Merge remote-tracking branch 'origin/main' into mobile-rearch

# Conflicts:
#	mobile/src/tasks/mobile-tasks-refactor-parity.test.ts
#	mobile/src/tasks/source-workspace-create.ts
#	mobile/src/tasks/worktree-create-retry.ts
#	src/main/daemon/headless-emulator.ts
#	src/main/index.ts
This commit is contained in:
Jinwoo-H
2026-08-31 22:10:08 -04:00
585 changed files with 33654 additions and 5440 deletions
+24
View File
@@ -1122,10 +1122,33 @@ jobs:
retention-days: 7
if-no-files-found: ignore
# Why: artifact jobs submit Windows binaries to SignPath. Keep every
# quota-consuming build behind all blocking release gates so a late test
# failure cannot create signing requests that can never be published.
release-preflight:
needs:
- cut
- terminal-rendering-golden
- skill-sharing-release-gate
- skill-sharing-linux-floor-release-gate
if: >-
always() &&
needs.cut.outputs.should_release == 'true' &&
needs.terminal-rendering-golden.result == 'success' &&
needs.skill-sharing-release-gate.result == 'success' &&
needs.skill-sharing-linux-floor-release-gate.result == 'success'
runs-on: ubuntu-latest
permissions:
contents: read
steps:
- name: Confirm blocking release gates passed
run: echo "All blocking release gates passed; artifact builds may start."
build:
needs:
- cut
- create-release
- release-preflight
if: needs.cut.outputs.should_release == 'true'
strategy:
fail-fast: false
@@ -2026,6 +2049,7 @@ jobs:
needs:
- cut
- create-release
- release-preflight
if: needs.cut.outputs.should_release == 'true'
# Why: SignPath requires every job in this signing workflow to be
# GitHub-hosted. The actual mac build runs in release-mac-build.yml so
+6
View File
@@ -76,6 +76,12 @@ When adding or changing a Git command:
- Keep the real-binary compatibility contract in PR CI current. When adopting a newer Git feature, add its version boundary so the preferred command and fallback both run against representative Git releases.
- Preserve commands that begin with global Git options such as `-c` before the subcommand, including auto-maintenance suppression used by worktree-create fetches.
## Git Scan Safety
- Never enumerate every ref and then run `git ls-tree -r` or `git show` once per ref. That ref × tree fan-out can retain gigabytes of output before a downstream `sort -u` or search can make progress.
- Prefer `rg` over the checked-out files for source searches. For history or refs, use a named ref, an explicit namespace/path, `--max-count`, and a bounded output; do not use an unqualified `--all` scan as a first diagnostic.
- Keep repository-wide commands targeted to the current repository and worktree. If an unbounded scan is genuinely required, measure the ref count first, explain the cost, and get confirmation before running it.
## Git Provider Compatibility
Source-control and review changes must consider GitLab and other supported git providers, not only GitHub. Keep provider-specific behavior behind explicit checks, and avoid GitHub-only naming for generic review concepts.
+2 -3
View File
@@ -36,7 +36,7 @@
Monitor and steer your agents from your phone — get notified when an agent finishes and send follow-ups from anywhere.
[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile)
[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile)
</td>
<td width="50%">
@@ -230,7 +230,7 @@ yay -S stably-orca-bin
Pair with your desktop app to monitor and steer your agents from your phone.
- **iOS:** [Download on the App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) or [join TestFlight](https://testflight.apple.com/join/YjeGMQBA)
- **Android:** [Download APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Install guide](https://www.onorca.dev/docs/android-apk)
- **Android:** [Download APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Install guide](https://www.onorca.dev/docs/android-apk)
---
@@ -262,7 +262,6 @@ Want to contribute or run locally? See our [CONTRIBUTING.md](.github/CONTRIBUTIN
</p>
## Signed Builds
Windows code signing sponored/provided by [SignPath.io](https://signpath.io), certificate by [SignPath Foundation](https://signpath.org).
## License
+295 -13
View File
@@ -1,6 +1,6 @@
{
"schemaVersion": 1,
"updatedAt": "2026-08-23",
"updatedAt": "2026-08-31",
"policy": {
"maturityLevels": ["experimental", "soak", "blocking", "accepted-gap", "deprecated"],
"blockingPromotion": {
@@ -5722,6 +5722,284 @@
],
"demotionRule": "Keep experimental or demote if ownership cardinality flakes, duplicate replay reaches a second renderer, active metadata or authority moves to the wrong leaf, deep normalization regresses, or a supported provider bypasses normalization."
},
{
"id": "terminal-session.split-activation-ordering",
"title": "Terminal splits activate before inherited-CWD lookup settles",
"maturity": "experimental",
"protection": "partial",
"owner": "terminal-renderer-lifecycle",
"layer": "renderer-unit-and-local-transport",
"surfaces": [
"terminal pane split",
"inherited working directory",
"pre-connect terminal input",
"split close cleanup",
"pre-bind pane detach"
],
"platforms": ["macos", "linux", "windows"],
"providers": ["local", "local-daemon", "ssh", "wsl", "remote-runtime"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "remote-runtime"],
"coverageNotes": "Deterministic renderer contracts hold CWD resolution behind an explicit promise, require the new pane to be created synchronously, and prove that close cancels the pending connection. A module-level stable-pane-key handoff preserves the exact CWD promise and bounded pre-connect input across whole-tab remounts, including a tab rehome between worktree buckets; stale owners are fenced, concrete PTY bind and definitive spawn failure clear the record, and explicit pane close discards it. Detach contracts reject cwd-pending and cwd-resolved deferred splits before PTY bind without mutation, carry resolved cwd for other unbound panes, and preserve persisted or live PTY handoff. Local IPC transport contracts exercise the real bounded pre-connect buffer and one live input FIFO across seeded and newly typed ordinary, acknowledged, and immediate writes, concurrent flushes, in-flight teardown, late spawn success or failure, attach failure, failed spawn, same-id reuse, stale-spawn retirement ownership, destroy, and mutable recovery metadata. The handoff registry is capped at 64 records for 15 seconds and shares the existing 1,024-entry/conservative UTF-16 input ceilings. A mocked direct-SSH authority-rotation contract proves a rejected stale spawn releases its deferred-CWD fence. The schema-v2 headful Electron benchmark records exact revision identity and attributes CWD request/settlement, PTY spawn request/result, bind, fixture unlock request/IPC write, fixture readiness, input, and first echo across 3 warmups and 20 measured cold-CWD cycles, requiring a distinct child PTY and observed child pty:exit before the next cycle. Remote-runtime coverage proves delegation remains host-owned; its host-delegated split path does not consume the local pre-connect input options, so remote-runtime input-remount replay and physical local-daemon, SSH, WSL, Linux, Windows, and folder-workspace latency journeys remain gaps.",
"motivatingLinks": ["https://github.com/stablyai/orca/commit/572ed1a8882"],
"invariant": "A terminal split creates and activates its renderer pane before an inherited-CWD lookup settles, starts its PTY only after the resolved directory is available, and cannot be externally detached while that deferred spawn remains unbound. A whole-tab remount or worktree rehome preserves the same stable pane's CWD promise and admitted local pre-connect bytes in order until a successor binds or the intent is definitively abandoned; stale owners cannot append or clear the successor's record. Other unbound panes preserve resolved cwd as startupCwd when detached. Bounded pre-connect input and later live local input share byte order, and teardown settles acknowledged writes without creating or rebinding a stale PTY. A disconnected or detached pending connect cannot bind its late fresh spawn, report its late failure through current callbacks, or ID-retire a newer same-ID owner; rejecting a stale direct-SSH spawn also releases the matching deferred-CWD fence. Natural exit cannot deliver queued work into a reused PTY id. Bound and remote-runtime splits remain owned by their execution host.",
"oracle": "Hold CWD resolution behind a controllable promise, invoke the production split path, and require manager.splitPane plus split telemetry before resolving it. Before PTY bind, require both cwd-pending and cwd-resolved deferred detach attempts to return null without layout, pane, tab, ownership, or focus mutation; separately require resolved cwd on an allowed unbound detach and unchanged persisted/live PTY adoption. Exercise the stable-pane handoff registry through repeated remounts and a worktree rehome, requiring the identical CWD promise, ordered ordinary/acknowledged/immediate seed replay, stale-owner fencing, 64-record/15-second bounds, and discard on bind, failure, or explicit close. Rotate a mocked direct-SSH authority while its delayed spawn is in flight, reject and disconnect the stale PTY claim, then require exactly one deferred-CWD cleanup when the delayed connect settles. At the local IPC PTY boundary, require zero connect calls while pending, the resolved CWD in spawn and local recovery metadata, one shared FIFO across seeded/new pre-connect and live ordinary/acknowledged/immediate input, prompt predecessor acknowledged-promise settlement on teardown, retirement of an unowned late fresh spawn, preservation of a newer same-ID owner, and zero stale delivery after failure, close, destroy, detach, natural exit, or same-id reuse. Capture callback exceptions must not change admission results. In visible Electron, press the real split shortcut for 3 warmup cycles, then after a cold inherited-CWD interval for each of 20 measured cycles, require an exact clean revision identity, complete focus/CWD/spawn/bind/fixture/input/echo attribution, distinct child PTYs, pane count, and child exits for every cycle; a timed-out, missing-event, or cleanup-aborted run must publish no headline latency and fail.",
"commands": [
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts src/renderer/src/lib/pane-manager/pane-split-close.test.ts src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/renderer/src/components/terminal-pane/deferred-split-pane-handoff.test.ts tests/e2e/terminal-split-activation-latency-artifact.unit.test.ts tests/e2e/terminal-split-activation-latency-main-probe.ts tests/e2e/terminal-split-activation-latency-phases.ts tests/e2e/terminal-split-activation-latency-report.unit.test.ts --reporter=dot",
// Historical evidence record retained so its seven-file command remains auditable.
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts src/renderer/src/lib/pane-manager/pane-split-close.test.ts src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts --reporter=dot",
// Historical evidence record retained so its nine-file command remains auditable.
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts src/renderer/src/lib/pane-manager/pane-split-close.test.ts src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/renderer/src/components/terminal-pane/deferred-split-pane-handoff.test.ts tests/e2e/terminal-split-activation-latency-artifact.unit.test.ts --reporter=dot",
// Historical schema-v1 evidence records only; their labels do not verify the checkout or harness.
"ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=baseline-df14d1a2983d8339e788d0e521f1c4affd9c6d5f-headful-run1 ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-baseline-df14d1a2983d8339e788d0e521f1c4affd9c6d5f-headful-run1.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-d453ffcdb704764daced1b2917fddee7224389f0-headful-run2 ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-candidate-d453ffcdb704764daced1b2917fddee7224389f0-headful-run2.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
// Exact-HEAD schema-v1 evidence records use the committed harness but predate revision metadata.
"ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-962faacec8c-headful-current ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-962faacec8c-headful.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-962faacec8c-headful-run2 ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-962faacec8c-headful-run2.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
// Schema-v2 evidence embeds the exact checkout identity and clean/dirty state.
"ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-073e6c7b0eb-headful-clean ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-073e6c7b0eb-headful-clean.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1"
],
"testFiles": [
"src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts",
"src/renderer/src/lib/pane-manager/pane-split-close.test.ts",
"src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts",
"src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts",
"src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts",
"src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts",
"src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts",
"src/renderer/src/components/terminal-pane/deferred-split-pane-handoff.test.ts",
"tests/e2e/terminal-split-activation-latency-artifact.unit.test.ts",
"tests/e2e/terminal-split-activation-latency-main-probe.ts",
"tests/e2e/terminal-split-activation-latency-phases.ts",
"tests/e2e/terminal-split-activation-latency-report.unit.test.ts",
"tests/e2e/terminal-split-activation-latency.spec.ts"
],
"assertionRefs": [
{
"file": "src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts",
"assertions": [
"creates and records the split before pending CWD resolution",
"passes the resolved CWD promise without reviving a stale manager",
"rapid nested splits reuse one pending CWD lookup",
"keeps remote-runtime split ownership on its execution host"
]
},
{
"file": "src/renderer/src/lib/pane-manager/pane-split-close.test.ts",
"assertions": ["focuses the new pane before publishing an unresolved CWD spawn hint"]
},
{
"file": "src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts",
"assertions": [
"preserves a deferred split fence when OSC 7 updates cwd",
"clears a settled deferred entry only for matching promise identity",
"keeps a newer deferred lookup when an older cleanup callback arrives"
]
},
{
"file": "src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts",
"assertions": [
"starts no PTY connection before inherited CWD resolves",
"applies the resolved directory to transport options",
"disposing the split before resolution cancels the pending spawn",
"invokes deferred-CWD cleanup exactly once after an authority-rotated direct-SSH spawn is disconnected and its delayed connect settles"
]
},
{
"file": "src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts",
"assertions": [
"rejects a deferred split while inherited CWD is pending without mutation",
"continues rejecting after CWD resolves until PTY bind",
"carries resolved CWD as startupCwd for an allowed unbound detach",
"preserves persisted and live remote PTY detach handoff"
]
},
{
"file": "src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts",
"assertions": [
"concurrent flush calls share one worker and preserve mixed input order",
"clear settles an in-flight acknowledged write before its late resolve or reject",
"in-flight acknowledged input remains charged to entry and code-unit caps"
]
},
{
"file": "src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts",
"assertions": [
"ordinary, acknowledged, and immediate pre-connect input flushes in byte order",
"pending acknowledged input settles on connect, destroy, and spawn failure",
"disconnect, destroy, and natural exit cancel an in-flight acknowledged write without blocking connect",
"live acknowledged input blocks later ordinary and immediate writes at its invocation position",
"the preconnect-to-live transition preserves the same input FIFO",
"disconnect and detach retire a late fresh spawn and suppress late failures before they reach current callbacks",
"natural exit fences queued ordinary and acknowledged chunks across same-id reuse",
"buffered exit and attach failure clear retained input",
"ordinary and acknowledged write failures drop later input without leaving promises pending",
"entry and code-unit ceilings bound pre-connect input retention",
"local recovery metadata observes the resolved split CWD",
"a stale fresh-spawn completion cannot retire a newer same-ID owner"
]
},
{
"file": "src/renderer/src/components/terminal-pane/deferred-split-pane-handoff.test.ts",
"assertions": [
"hands the same cwd promise and buffered input to a remounted leaf in order",
"keeps input across repeated remounts and fences stale owners",
"releases an unmounted owner without dropping its pending handoff",
"retains input within the shared preconnect entry and code-unit caps",
"evicts the oldest handoff when the record cap is reached",
"expires an abandoned handoff after the bounded remount window"
]
},
{
"file": "tests/e2e/terminal-split-activation-latency-artifact.unit.test.ts",
"assertions": [
"writes a passing benchmark report to the requested artifact path",
"fails when the benchmark artifact path cannot be written"
]
},
{
"file": "tests/e2e/terminal-split-activation-latency-main-probe.ts",
"assertions": [
"attributes CWD and PTY spawn request/settlement events to the source and child PTYs",
"captures the fixture unlock carriage return on both ordinary and acknowledged IPC channels",
"restores the intercepted IPC handlers and listener when the probe is disposed"
]
},
{
"file": "tests/e2e/terminal-split-activation-latency-phases.ts",
"assertions": [
"merges main-process events by operation and PTY identity without cross-cycle attribution",
"requires every activation, fixture, input, echo, pane, PTY, and cleanup observation for success",
"reports each attributed phase distribution with non-negative cross-clock durations"
]
},
{
"file": "tests/e2e/terminal-split-activation-latency-report.unit.test.ts",
"assertions": [
"attributes main-process phases to the matching source and child PTYs",
"embeds schema-v2 revision identity and summarizes the attributed phases",
"invalidates a sample when the actual fixture-unlock IPC write is missing"
]
},
{
"file": "tests/e2e/terminal-split-activation-latency.spec.ts",
"assertions": [
"requires a visible BrowserWindow and visible document before sampling",
"records schema-v2 revision identity plus attributed CWD, spawn, bind, fixture-ready, input, and echo phases",
"records 3 warmups, then 20 measured real-shortcut cycles after cold inherited-CWD intervals",
"requires every split to focus, bind a PTY distinct from its source, and echo immediate input",
"observes each closed child PTY exit before starting the next cycle",
"publishes headline latency only for a fully successful 3-warmup/20-measured run"
]
}
],
"evidenceRuns": [
{
"date": "2026-08-30",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts src/renderer/src/lib/pane-manager/pane-split-close.test.ts src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts --reporter=dot",
"durationSeconds": 56.59,
"summary": "Seven focused files and 78 tests passed. Vitest reported 49.92 seconds and the measured wall time was 56.59 seconds; coverage includes split creation and focus ordering, nested CWD lineage, promise-identity and SSH authority-rotation cleanup fencing, full pre-bind detach fencing, resolved-CWD detach handoff, close-cancellation, single-FIFO ordering, late-spawn retirement and error suppression, preservation of a newer same-ID owner, generation fencing, in-flight settlement across explicit teardown and natural exit, attach cleanup, bounded retention, and existing input-write contracts."
},
{
"date": "2026-08-31",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts src/renderer/src/lib/pane-manager/pane-split-close.test.ts src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/renderer/src/components/terminal-pane/deferred-split-pane-handoff.test.ts tests/e2e/terminal-split-activation-latency-artifact.unit.test.ts --reporter=dot",
"durationSeconds": 44.43,
"summary": "Nine focused files and 96 tests passed. Vitest reported 37.78 seconds and measured wall time was 44.43 seconds; the run adds stable-pane CWD/input handoff, repeated-remount stale-owner fencing, bounded 64-record/15-second retention, seeded-input caps, ordered ordinary/acknowledged/immediate replay, predecessor acknowledged-promise settlement, capture-callback failure containment, and benchmark-artifact write-failure coverage to the existing split, detach, CWD, and local transport contracts."
},
{
"date": "2026-08-31",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/terminal-pane-split-with-inherited-cwd.test.ts src/renderer/src/lib/pane-manager/pane-split-close.test.ts src/renderer/src/components/terminal-pane/resolve-split-cwd.test.ts src/renderer/src/components/terminal-pane/pty-connection-split-cwd-resolution.test.ts src/renderer/src/components/terminal-pane/terminal-pane-tab-detach.test.ts src/renderer/src/components/terminal-pane/pty-preconnect-input-buffer.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/renderer/src/components/terminal-pane/deferred-split-pane-handoff.test.ts tests/e2e/terminal-split-activation-latency-artifact.unit.test.ts tests/e2e/terminal-split-activation-latency-main-probe.ts tests/e2e/terminal-split-activation-latency-phases.ts tests/e2e/terminal-split-activation-latency-report.unit.test.ts --reporter=dot",
"durationSeconds": 4.14,
"summary": "The updated twelve-path focused command passed 99 tests (10 runnable test files plus 2 benchmark support modules), including schema-v2 main-process phase attribution, report revision identity, fixture IPC-write validation, stable-pane handoff, ordered pre-connect/live input, cleanup, detach, failure, and same-ID ownership contracts."
},
{
"date": "2026-08-30",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=baseline-df14d1a2983d8339e788d0e521f1c4affd9c6d5f-headful-run1 ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-baseline-df14d1a2983d8339e788d0e521f1c4affd9c6d5f-headful-run1.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"durationSeconds": 72,
"summary": "At baseline df14d1a2983d8339e788d0e521f1c4affd9c6d5f, the visible BrowserWindow and document completed 3/3 warmups followed by 20/20 measured cold-CWD cycles with every event present, distinct child PTYs, and observed child exits. Shortcut-to-focus p50/p95/max was 65.7/88.7/102.6 ms, PTY bind was 122.8/154.0/159.7 ms, and first echo was 203.8/264.2/332.7 ms. Artifact SHA-256: 6d860cd0cd210f55f2349a197318248af488042117b20c08b6831160950c3277."
},
{
"date": "2026-08-30",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-d453ffcdb704764daced1b2917fddee7224389f0-headful-run2 ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-candidate-d453ffcdb704764daced1b2917fddee7224389f0-headful-run2.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"durationSeconds": 72,
"summary": "At candidate d453ffcdb704764daced1b2917fddee7224389f0, the visible BrowserWindow and document completed 3/3 warmups followed by 20/20 measured cold-CWD cycles with every event present, distinct child PTYs, and observed child exits. Shortcut-to-focus p50/p95/max was 12.8/14.4/16.5 ms, PTY bind was 177.0/316.1/333.3 ms, and first echo was 268.6/627.4/710.7 ms. Artifact SHA-256: 9aef7fa842c732eb74f0066a77b2a0336e8f956a06ca61387f9999d6e76213cc."
},
{
"date": "2026-08-31",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-962faacec8c-headful-current ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-962faacec8c-headful.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"durationSeconds": 72,
"summary": "At exact HEAD 962faacec8c, the visible BrowserWindow and document completed 3/3 warmups followed by 20/20 measured cold-CWD cycles with every event present, distinct child PTYs, and observed child exits. Shortcut-to-focus p50/p95/max was 12.6/13.7/13.7 ms, PTY bind was 140.5/399.1/464.1 ms, and first echo was 192.0/519.3/2343.1 ms. Artifact SHA-256: 875e9d37dc711472a81438e4bbdbc8cbc7aada8c961d14195c024d8da350b9e2."
},
{
"date": "2026-08-31",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-962faacec8c-headful-run2 ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-962faacec8c-headful-run2.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"durationSeconds": 72,
"summary": "At exact HEAD 962faacec8c, the visible BrowserWindow and document completed 3/3 warmups followed by 20/20 measured cold-CWD cycles with every event present, distinct child PTYs, and observed child exits. Shortcut-to-focus p50/p95/max was 12.8/14.0/14.3 ms, PTY bind was 236.5/654.0/687.3 ms, and first echo was 566.8/1191.5/1219.7 ms. Artifact SHA-256: 4f66b93e5c936ade05f880010f4ec027385d5c89893430169af91c7fdfbe1d06."
},
{
"date": "2026-08-31",
"runner": "local",
"platform": "macos",
"result": "passed",
"command": "ORCA_TERMINAL_SPLIT_LATENCY_BENCH=1 ORCA_TERMINAL_SPLIT_LATENCY_LABEL=candidate-073e6c7b0eb-headful-clean ORCA_TERMINAL_SPLIT_LATENCY_OUTPUT=/private/tmp/orca-terminal-split-activation-073e6c7b0eb-headful-clean.json pnpm exec playwright test tests/e2e/terminal-split-activation-latency.spec.ts --config tests/playwright.config.ts --project electron-headful --workers=1",
"durationSeconds": 87.6,
"summary": "At exact clean HEAD 073e6c7b0eb1c0ccbb115db1528b641901997c73, the schema-v2 artifact records dirty=false, a visible BrowserWindow/document, and 3/3 warmups plus 20/20 measured cycles with every missing-event counter at zero, distinct child PTYs, and observed child exits. Measured focus p50/p95/max was 12.0/13.6/14.7 ms; attributed CWD lookup was 40/97/103 ms, CWD-settle to spawn request 1/4/4 ms, spawn request to result 48/120/122 ms, and spawn result to bind 2.1/2.5/2.5 ms. Fixture unlock request to IPC write was 0.5/0.8/0.9 ms, IPC write to fixture-ready parse was 35.1/87.4/141.8 ms, and input to first echo was 1.3/3.5/3.9 ms. Total shortcut-to-bind was 113.5/161.3/162.7 ms and shortcut-to-first-echo was 158.5/240.7/291.2 ms. Artifact SHA-256: 1e7ff9e658b717056273d64ecdf662cc6b5776bb6dae1827ed213b7647e4a5fb."
}
],
"evidenceProcedure": "Run the benchmark spec in a visible macOS Electron project from a clean primary worktree, complete 3 warmups followed by 20 measured cold-CWD cycles, require zero missing events and successful cleanup, save the schema-v2 JSON report, and record its SHA-256 plus embedded revision identity. The four older records are schema-v1 historical audit records whose labels do not verify the checkout or include phase attribution; the clean schema-v2 record is the current candidate evidence. A paired baseline/final rerun with one revision-verifying harness remains required before promotion or a readiness comparison claim.",
"runtimeBudget": {
"p95Seconds": 240,
"scope": "the full listed gate command set: one current twelve-path focused invocation (ten runnable test files plus two benchmark support modules), one historical nine-file unit invocation, one historical seven-file unit invocation, and five opt-in 3-warmup/20-measured visible Electron benchmark invocations (four schema-v1 historical records plus one clean schema-v2 record)"
},
"flakeHistory": {
"status": "not-started",
"evidence": "The focused promise-barrier, remount-handoff, benchmark-artifact, and schema-v2 attribution suite passes locally in 99 tests; the clean visible benchmark passes 3/3 warmups and 20/20 measured cycles with zero missing events. Its first attempt hit a transient warmup cleanup-dialog click timeout, then the exact command passed on retry without a launch, profile, or port workaround. Routed CI and soak history have not started. Historical benchmark labels do not verify the product checkout or embed revision identity."
},
"redGreenEvidence": {
"status": "partial",
"evidence": "Before the production seam landed, the split assertion failed with zero manager calls while CWD was pending, and the transport assertion rejected the first pre-connect input. Before the detach fence, a deferred split could be removed after CWD resolved but before PTY bind, dropping its pre-connect input. Before the single-flight hardening, the concurrent-flush oracle delivered ordinary input before the earlier acknowledged write and clear left the in-flight promise pending. Before stale-spawn ownership fencing, the combined deferred-connect and newer same-ID attach fixture called kill on the current PTY. An intentional one-line revert of deferred-CWD cleanup in the stale direct-SSH claim branch failed its focused callback assertion with zero calls instead of one; restoring it passed the prior focused tests. The remount-handoff, transport, artifact-write, and schema-v2 attribution regressions are green in the 99-test twelve-path run, but isolated intentional-revert evidence for each cleanup branch remains outstanding."
},
"performanceBudget": {
"required": true,
"evidence": "Pane creation and focus add no timer, polling, provider inventory, or subprocess work. CWD resolution remains one existing bounded request off the visible activation path, and detach admission adds only bounded map and record lookups. The remount handoff adds one module-level map lookup per pane lifecycle, a 64-record cap, and a 15-second expiry; it retains no unbounded payload. Pre-connect input, including an in-flight acknowledged write and remount seed replay, is capped at 1,024 entries and a conservative UTF-16 ceiling derived from the existing terminal-input byte limit, drains through one worker in order, and clears on teardown or failed connect. The historical schema-v1 same-mode pair recorded shortcut-to-focus p50/p95/max changing from 65.7/88.7/102.6 ms to 12.8/14.4/16.5 ms; those labels do not embed revision identity, so the comparison is directional evidence only. The clean schema-v2 candidate attributes focus at 12.0/13.6/14.7 ms while CWD lookup takes 40/97/103 ms and spawn request-to-result takes 48/120/122 ms, demonstrating that provider/process startup follows activation rather than blocking it. In that clean run, shortcut-to-bind is 113.5/161.3/162.7 ms, fixture IPC-write-to-ready is 35.1/87.4/141.8 ms, and input-to-echo is 1.3/3.5/3.9 ms; these readiness phases are diagnostic, one-host descriptive measurements, and no clean schema-v2 baseline exists to support a readiness improvement or regression claim. The n=20 empirical p95 values are descriptive, are not a distribution guarantee, and are not CI-enforced."
},
"promotionCriteria": [
"Record complete red/green evidence for close, remount/rehome handoff, mixed-input ordering, metadata, and failure cleanup.",
"Collect 100 consecutive focused CI passes or 14 days without an unexplained flake.",
"Run the committed real-shortcut Electron benchmark in routed CI or soak before enforcing a latency budget.",
"Collect physical local-daemon, SSH or WSL plus Linux, Windows, and folder-workspace evidence before claiming provider-complete coverage."
],
"knownGaps": [
"The clean schema-v2 candidate run and the historical schema-v1 comparison records ran on one Apple-silicon macOS host with a synthetic POSIX echo shell and a git-backed workspace; their n=20 empirical p95 values are descriptive and not CI-enforced.",
"No physical local-daemon, SSH, WSL, Linux, Windows, or folder-workspace latency journey has run; the synthetic fixture is currently skipped on Windows because it requires a POSIX shell.",
"Remote-runtime split creation remains host-delegated and its transport does not consume the local pre-connect seed/capture options; CWD handoff is covered, but remote-runtime pre-connect input replay has no implementation or evidence.",
"The four stored schema-v1 artifacts predate the final harness attribution/reporting and do not embed revision identity; the clean schema-v2 candidate artifact is revision-verified, but both product revisions still need a paired schema-v2 rerun with one committed harness before promotion.",
"No forced-failure visible benchmark artifact has been recorded; the focused artifact-write and missing-event report contracts verify local failure handling, while failure-report serialization remains unverified by a full visible run.",
"No clean schema-v2 baseline phase artifact exists, so the attributed CWD, spawn, fixture-ready, bind, and echo timings diagnose where time is spent but do not establish a shell-readiness improvement or regression."
],
"demotionRule": "Keep experimental or demote if pane activation waits on CWD, a deferred split can detach before PTY bind, a remount or rehome loses its stable CWD/input handoff, stale owners mutate a successor record, detached cwd is lost, input reorders or remains pending after cleanup, a closed pane can spawn, stale retirement kills a newer same-ID owner, remote-runtime delegation creates a competing local pane, or the focused suite flakes without an identified product or harness cause."
},
{
"id": "terminal-session.kill-all-surface-cleanup",
"title": "Kill all sessions removes only the confirmed terminal surfaces and current bindings",
@@ -8194,9 +8472,7 @@
"assertionRefs": [
{
"file": "tests/e2e/persisted-session-production-upgrade.spec.ts",
"assertions": [
"upgrades a legacy daemon session and keeps it stable after relaunch"
]
"assertions": ["upgrades a legacy daemon session and keeps it stable after relaunch"]
}
],
"evidenceRuns": [
@@ -13970,14 +14246,14 @@
"providers": ["local", "daemon", "ssh"],
"coveredPlatforms": ["macos"],
"coveredProviders": ["local", "ssh"],
"coverageNotes": "Deterministic renderer and IPC-transport tests prove count and text ceilings, oldest-reply shedding, explicit query-reply source routing, ordinary-input preservation, one-reply-per-write delivery for OSC, DA1, and CPR replies, real xterm OSC reply generation, drain-failure containment, and clear/reuse generation fencing. Remote-runtime tests preserve separate query-reply writes across pending input, async validation, and viewport-claim buffering. Host-contract tests prove a later DA1/CPR reply cannot overtake a deferred OSC reply, including a coalesced legacy-client payload. Live macOS Electron tests cover local PTY OSC replies and interactive typing; a macOS-hosted Docker OpenSSH test proves an upstream-node-pty Linux relay keeps OSC/DA1 replies out of the next fish child's stdin. No live daemon, paired-runtime, WSL, physical Linux/Windows client, or binary mixed-version run is registered.",
"coverageNotes": "Deterministic renderer and IPC-transport tests prove count and text ceilings, oldest-reply shedding, explicit query-reply source routing, ordinary-input preservation, acknowledged-write FIFO barriers, single-worker drain reentrancy, one-reply-per-write delivery for OSC, DA1, and CPR replies, real xterm OSC reply generation, drain-failure containment, teardown settlement, and clear/reuse generation fencing. Remote-runtime tests preserve separate query-reply writes across pending input, async validation, and viewport-claim buffering. Host-contract tests prove a later DA1/CPR reply cannot overtake a deferred OSC reply, including a coalesced legacy-client payload. Live macOS Electron tests cover local PTY OSC replies and interactive typing; a macOS-hosted Docker OpenSSH test proves an upstream-node-pty Linux relay keeps OSC/DA1 replies out of the next fish child's stdin. No live daemon, paired-runtime, WSL, physical Linux/Windows client, or binary mixed-version run is registered.",
"motivatingLinks": [
"https://github.com/stablyai/orca/issues/13137",
"https://github.com/stablyai/orca/issues/7329",
"https://github.com/stablyai/orca/issues/13892"
],
"invariant": "The desktop PTY input queue retains at most 64 explicitly sourced pending terminal query replies and 4096 UTF-16 code units. Every retained reply reaches the provider as one atomic write, and the host writes each reply the moment it accepts it, so replies reach the PTY in the order they were produced with no queue that could reorder them. A reply's own echo is contained on the output side by projecting its known echo shapes; the ESC-initial verbatim shape is matched only when complete, never held as a partial, so a query torn at its own ESC is still answered. Overflow removes only the oldest query replies, never ordinary input except the documented modified-F3/CPR byte collision, and drain failures cannot clear a newer queue generation.",
"oracle": "Synchronously enqueue separate 10,000-entry OSC and DA1 reply floods before the scheduled drain and assert that only the initial immediate reply and newest 64 pending replies are written, each as one provider write, before a trailing keystroke. At the host boundary, defer an OSC reply and assert that separate or legacy-coalesced DA1/CPR replies flush after it in observed query order. At the remote-runtime boundary, preserve separate writes around pending ordinary input, async validation, and viewport-claim buffering. Repeat behind 10,000 ordinary inputs and exercise the text ceiling, real xterm generation, provider-write failure, rejected yield, and clear/reuse generation fencing.",
"invariant": "The desktop PTY input queue retains at most 64 explicitly sourced pending terminal query replies and 4096 UTF-16 code units. Every retained reply reaches the provider as one atomic write, and ordinary, acknowledged, and reply input share one invocation-ordered FIFO so no later write overtakes an acknowledged write. Reentrant write callbacks cannot start a second drain worker or strand input admitted after clear/reuse. A reply's own echo is contained on the output side by projecting its known echo shapes; the ESC-initial verbatim shape is matched only when complete, never held as a partial, so a query torn at its own ESC is still answered. Overflow removes only the oldest query replies, never ordinary input except the documented modified-F3/CPR byte collision, and failure or teardown cannot clear a newer queue generation or strand an acknowledged promise.",
"oracle": "Synchronously enqueue separate 10,000-entry OSC and DA1 reply floods before the scheduled drain and assert that only the initial immediate reply and newest 64 pending replies are written, each as one provider write, before a trailing keystroke. Stall an acknowledged write between earlier and later ordinary/reply input, then require teardown to settle it and same-id reuse to receive no stale tail. Reenter the queue synchronously from an acknowledged write with both enqueue and clear/reuse, requiring one drain and fresh input delivery only after the stale acknowledged write settles false. At the host boundary, defer an OSC reply and assert that separate or legacy-coalesced DA1/CPR replies flush after it in observed query order. At the remote-runtime boundary, preserve separate writes around pending ordinary input, async validation, and viewport-claim buffering. Repeat behind 10,000 ordinary inputs and exercise the text ceiling, real xterm generation, provider-write failure, rejected yield, and clear/reuse generation fencing.",
"commands": [
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/shared/terminal-query-reply.test.ts src/shared/pty-startup-ingress-live-query-reply.test.ts src/shared/pty-startup-reply-echo-shapes.test.ts",
"pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-batching.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-query-reply-immediate.test.ts src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-input-coalescing.test.ts",
@@ -14009,14 +14285,20 @@
"real xterm OSC 10/11 query handlers remain subject to the same retention ceiling",
"retained OSC, DA1, and CPR replies stay one provider write each and both OSC and DA1 floods remain bounded",
"provider-write and yield failures settle without unhandled rejection, repeated same-generation admission, or stale-generation clearing",
"clear releases saturated reply accounting and fences in-flight validation before later input"
"clear releases saturated reply accounting and fences in-flight validation before later input",
"acknowledged input is serialized between earlier and later ordinary/reply writes",
"clear settles active and pending acknowledged input before same-id queue reuse",
"reentrant enqueue cannot start a second drain worker past an unacknowledged write",
"reentrant clear captures the stale cancellation and continues draining fresh input"
]
},
{
"file": "src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts",
"assertions": [
"sendInputImmediate applies the reply ceiling while sendInput preserves a reply-shaped ordinary payload in exact IPC write order",
"a thrown renderer write triggers one owning-transport recovery callback and rejects later input in that queue generation"
"a thrown renderer write triggers one owning-transport recovery callback and rejects later input in that queue generation",
"live and preconnect acknowledged writes remain FIFO barriers for later ordinary and immediate input",
"disconnect, detach, and natural exit settle acknowledged writes and fence same-id stale chunks"
]
},
{
@@ -14072,13 +14354,13 @@
],
"evidenceRuns": [
{
"date": "2026-08-25",
"date": "2026-08-30",
"runner": "local",
"platform": "macos",
"command": "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/components/terminal-pane/pty-input-write-queue.test.ts src/renderer/src/components/terminal-pane/pty-transport-input-write.test.ts src/shared/terminal-query-reply.test.ts src/shared/pty-startup-ingress-live-query-reply.test.ts src/shared/pty-startup-reply-echo-shapes.test.ts",
"result": "passed",
"durationSeconds": 1.05,
"summary": "Five files and 92 tests passed, including the live-pty echo-shape transcript, including explicit IPC reply-source routing, the 10,000-reply count ceiling, text ceiling, 10,000-entry ordinary backlog preservation, real xterm OSC query flood, single-shot drain-failure recovery, one-reply-per-write echo containment, and clear/reuse generation fencing."
"durationSeconds": 15,
"summary": "Five files and 149 tests passed in 15.16 seconds, including explicit IPC reply-source routing, the 10,000-reply count and text ceilings, 10,000-entry ordinary backlog preservation, acknowledged-write FIFO barriers, synchronous reentrancy fencing, prompt teardown settlement, same-id generation fencing, real xterm OSC query floods, single-shot drain-failure recovery, and one-reply-per-write echo containment."
},
{
"date": "2026-08-09",
@@ -14127,7 +14409,7 @@
},
"redGreenEvidence": {
"status": "partial",
"evidence": "Without the branch's admission cap, the 10,000-reply fixture writes all replies before the trailing keystroke. Before the source-routing repair, the queue had no API capable of distinguishing reply-shaped ordinary input; before failure containment, a thrown provider write rejected waitForDrain and Vitest recorded an unhandled rejection; the first containment pass retried a failed generation and invoked recovery twice; before generation fencing, a rejected stale yield cleared fresh input. No saved intentional-break artifact is attached yet."
"evidence": "Without the branch's admission cap, the 10,000-reply fixture writes all replies before the trailing keystroke. Before the source-routing repair, the queue had no API capable of distinguishing reply-shaped ordinary input; before failure containment, a thrown provider write rejected waitForDrain and Vitest recorded an unhandled rejection; the first containment pass retried a failed generation and invoked recovery twice; before generation fencing, a rejected stale yield cleared fresh input. Before reentrancy fencing, a synchronous accepted-write callback started a second drain that falsely accepted the pending write; clear/reuse also captured the replacement generation's cancellation and stranded fresh input. No saved intentional-break artifact is attached yet."
},
"performanceBudget": {
"required": true,
@@ -95,6 +95,66 @@ describe('benchmark artifact comparison', () => {
})
})
it('compares terminal split headline metrics in milliseconds', () => {
const dir = makeTempDir()
const baselinePath = writeArtifact(dir, 'split-baseline.json', {
label: 'split baseline',
headlineMs: {
shortcutToFocusP50: 284.2,
shortcutToFocusP95: 676.3
}
})
const candidatePath = writeArtifact(dir, 'split-candidate.json', {
label: 'split candidate',
headlineMs: {
shortcutToFocusP50: 12.7,
shortcutToFocusP95: 13.7
}
})
const comparison = comparePaths(baselinePath, candidatePath)
expect(comparison.baseline.kind).toBe('terminal-split-activation')
expect(comparison.metrics).toEqual(
expect.arrayContaining([
expect.objectContaining({
key: 'shortcutToFocusP50',
unit: 'ms',
baseline: 284.2,
candidate: 12.7,
status: 'improved'
}),
expect.objectContaining({
key: 'shortcutToFocusP95',
unit: 'ms',
baseline: 676.3,
candidate: 13.7,
status: 'improved'
})
])
)
})
it('rejects invalid benchmark artifacts before comparing partial metrics', () => {
const dir = makeTempDir()
const baselinePath = writeArtifact(dir, 'split-invalid.json', {
label: 'invalid split',
status: 'failed',
valid: false,
headlineMs: { shortcutToFocusP50: 0 }
})
const candidatePath = writeArtifact(dir, 'split-valid.json', {
label: 'valid split',
status: 'passed',
valid: true,
headlineMs: { shortcutToFocusP50: 10 }
})
expect(() => comparePaths(baselinePath, candidatePath)).toThrow(
'split-invalid.json: benchmark artifact is marked invalid'
)
})
it('compares numeric Playwright annotation metrics and omits metadata fields', () => {
const dir = makeTempDir()
const baselinePath = writeArtifact(dir, 'baseline-playwright.json', {
+13 -1
View File
@@ -74,6 +74,9 @@ export function readBenchmarkArtifact(path) {
}
export function normalizeBenchmarkArtifact(path, artifact = readBenchmarkArtifact(path)) {
if (artifact?.valid === false || artifact?.status === 'failed') {
throw new Error(`${path}: benchmark artifact is marked invalid`)
}
if (artifact?.summaryMedianMs != null) {
return normalizeNumericObject(path, artifact, 'startup', artifact.summaryMedianMs, () => 'ms')
}
@@ -82,6 +85,15 @@ export function normalizeBenchmarkArtifact(path, artifact = readBenchmarkArtifac
key.endsWith('Count') || key.endsWith('After') ? 'count' : 'ms'
)
}
if (artifact?.headlineMs != null) {
return normalizeNumericObject(
path,
artifact,
'terminal-split-activation',
artifact.headlineMs,
() => 'ms'
)
}
if (artifact?.suites != null) {
return normalizePlaywrightArtifact(path, artifact)
}
@@ -89,7 +101,7 @@ export function normalizeBenchmarkArtifact(path, artifact = readBenchmarkArtifac
return normalizeSummaryArtifact(path, artifact)
}
throw new Error(
`${path}: unsupported benchmark artifact; expected summaryMedianMs, summaryMedian, Playwright suites, or top-level summary`
`${path}: unsupported benchmark artifact; expected summaryMedianMs, summaryMedian, headlineMs, Playwright suites, or top-level summary`
)
}
@@ -12,11 +12,33 @@ const stubPath = join(projectDir, 'skills', 'computer-use', 'SKILL.md')
const bundledGuide = BUNDLED_SKILL_GUIDES.find((guide) => guide.name === 'computer-use')?.markdown
describe('computer-use skill guidance', () => {
it('keeps discovery scoped to desktop control and out of the embedded browser', () => {
const frontmatter = /^---\n([\s\S]*?)\n---\n/u.exec(readFileSync(guidePath, 'utf8'))?.[1] ?? ''
const description = frontmatter.replace(/\s+/gu, ' ')
expect(description).toContain('OS/window-level inspection and input')
expect(description).toContain('external browser window')
expect(description).toContain("Do not use for Orca's embedded browser")
expect(description).toContain('page-only browser automation')
expect(description).toContain("`orca-cli` for Orca's embedded pages")
expect(description).toContain(
'page-automation tool such as Playwright or CDP for external pages'
)
expect(description).not.toContain('read Slack')
expect(description).not.toContain('get app state')
const orcaCli = readFileSync(join(projectDir, 'skill-guides', 'orca-cli.md'), 'utf8').replace(
/\s+/gu,
' '
)
expect(orcaCli).toContain('browser embedded inside the Orca app')
})
it('keeps web-app targeting on the computer-use surface', () => {
const skill = readFileSync(guidePath, 'utf8')
expect(skill).toContain('Use this skill for desktop UI through `orca computer`')
expect(skill).toContain('operate the desktop browser app/window that contains the page')
expect(skill).toContain('external desktop browser window that needs desktop-level control')
expect(skill).not.toContain('orca goto')
expect(skill).not.toContain('orca snapshot')
expect(skill).not.toContain('orca click')
@@ -18,6 +18,24 @@ function readSkill(path = guidePath) {
}
describe('orca CLI skill guidance', () => {
it('keeps external browser routing at the OS/page boundary', () => {
const skill = readSkill(guidePath)
const description = skill.replace(/\s+/gu, ' ')
expect(description).toContain(
'Use Computer Use for external browser windows, webviews, or desktop UI only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots.'
)
expect(description).toContain(
"`orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages."
)
expect(skill).toContain(
'For external Chrome/Safari/webviews or Orca app chrome/settings, use the Computer Use skill/tool only when the task requires OS/window-level control'
)
expect(skill).toContain(
"Use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages"
)
})
it('keeps independent worktree lineage separate from Git base selection', () => {
const skill = readSkill()
@@ -25,6 +25,17 @@ function getSection(markdown, heading) {
}
describe('orchestration skill guidance', () => {
it('keeps external browser routing at the OS/page boundary', () => {
const description = readFileSync(guidePath, 'utf8').replace(/\s+/gu, ' ')
expect(description).toContain(
"Use Computer Use for external browser windows, webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when the task requires OS/window-level control such as focus, menus, dialogs, coordinates, or screenshots."
)
expect(description).toContain(
"`orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages."
)
})
it('requires Orca runtime state before claiming a worker was orchestrated', () => {
const skill = readSkill()
const toolBoundary = getSection(skill, 'Tool Boundary')
+3 -1
View File
@@ -316,7 +316,9 @@ describe('PR E2E gate contract', () => {
selectPrE2eSpecs(['src/renderer/src/hooks/remote-workspace-session-merge.test.ts'])
).toEqual([])
expect(
selectPrE2eSpecs(['src/renderer/src/hooks/remote-workspace-target-sync-test-harness.ts'])
selectPrE2eSpecs([
'src/renderer/src/hooks/__tests__/remote-workspace-target-sync-test-harness.ts'
])
).toEqual([])
})
@@ -1,4 +1,5 @@
import { existsSync, readFileSync, rmSync } from 'node:fs'
import { existsSync, readFileSync } from 'node:fs'
import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
@@ -44,7 +45,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
)
expect(existsSync(rebuildLogPath)).toBe(false)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
}
)
@@ -80,7 +81,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
)
).toBe('// napi.h\n')
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -104,7 +105,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
expect(readFileSync(join(runtimeDir, 'conpty.dll'), 'utf8')).toBe('conpty.dll x64')
expect(readFileSync(join(runtimeDir, 'OpenConsole.exe'), 'utf8')).toBe('OpenConsole.exe x64')
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -132,7 +133,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
const rebuildCall = JSON.parse(readFileSync(rebuildLogPath, 'utf8').trim())
expect(rebuildCall.onlyModules).toEqual(['windows-native-registry'])
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
}
)
@@ -162,7 +163,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
const rebuildCall = JSON.parse(readFileSync(rebuildLogPath, 'utf8').trim())
expect(rebuildCall.onlyModules).toEqual(['node-pty'])
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
}
)
@@ -193,7 +194,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
expect(rebuildCall.ignoreModules).toEqual(['cpu-features'])
expect(rebuildCall.force).toBe(true)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
}
)
@@ -221,7 +222,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
)
expect(existsSync(rebuildLogPath)).toBe(false)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
}
)
@@ -251,7 +252,7 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
expect(rebuildCall.onlyModules).toEqual(['node-pty'])
expect(rebuildCall.force).toBe(true)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
}
)
+10 -9
View File
@@ -1,6 +1,7 @@
import { existsSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts'
import {
mkTempProject,
@@ -36,7 +37,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'download attempted\n'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -60,7 +61,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'Continuing postinstall because Electron binary installation failed'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -81,7 +82,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'Continuing postinstall because Electron binary installation failed'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -117,7 +118,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'stale-path'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -141,7 +142,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'platform=linux arch=arm64\ndownload attempted\n'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -162,7 +163,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
expect(result.status, result.stderr).toBe(0)
expect(existsSync(join(projectDir, 'electron-get.log'))).toBe(false)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -188,7 +189,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'electron.exe'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -209,7 +210,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
'platform=linux arch=x64'
)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
@@ -230,7 +231,7 @@ describe('rebuild-native-deps Electron install fallback', () => {
expect(result.stdout).toContain('Repaired Electron path.txt -> electron')
expect(existsSync(join(projectDir, 'electron-get.log'))).toBe(false)
} finally {
rmSync(projectDir, { recursive: true, force: true })
removeTreeSync(projectDir)
}
})
})
@@ -31,6 +31,7 @@ const EXPECTED_MATRIX = {
},
[`${RELEASE_WORKFLOW}#post-release-e2e`]: { actions: 'write' },
[`${RELEASE_WORKFLOW}#publish-release`]: { contents: 'write' },
[`${RELEASE_WORKFLOW}#release-preflight`]: { contents: 'read' },
[`${RELEASE_WORKFLOW}#skill-sharing-linux-floor-release-gate`]: { contents: 'read' },
[`${RELEASE_WORKFLOW}#skill-sharing-release-gate`]: { contents: 'read' },
[`${RELEASE_WORKFLOW}#terminal-rendering-golden`]: { contents: 'read' },
@@ -10,6 +10,27 @@ function stepNamed(job, name) {
}
describe('skill-sharing release workflow', () => {
it('keeps artifact builds behind every blocking release gate', () => {
const preflight = workflow.jobs['release-preflight']
const build = workflow.jobs.build
const macBuild = workflow.jobs['build-mac']
expect(preflight.needs).toEqual([
'cut',
'terminal-rendering-golden',
'skill-sharing-release-gate',
'skill-sharing-linux-floor-release-gate'
])
expect(preflight.if).toContain('always()')
expect(preflight.if).toContain("needs.terminal-rendering-golden.result == 'success'")
expect(preflight.if).toContain("needs.skill-sharing-release-gate.result == 'success'")
expect(preflight.if).toContain(
"needs.skill-sharing-linux-floor-release-gate.result == 'success'"
)
expect(build.needs).toContain('release-preflight')
expect(macBuild.needs).toContain('release-preflight')
})
it('blocks publication on native Windows, macOS, and the Linux floor', () => {
const platform = workflow.jobs['skill-sharing-release-gate']
const linux = workflow.jobs['skill-sharing-linux-floor-release-gate']
+1
View File
@@ -117,6 +117,7 @@
"../src/main/hermes/hermes-home-filesystem.ts",
"../src/main/hermes/hermes-managed-plugin-source.ts",
"../src/main/hermes/hook-service.ts",
"../src/main/in-flight-run-dedupe.ts",
"../src/main/kimi/hook-service.ts",
"../src/main/kimi/kimi-hook-config-toml.ts",
"../src/main/openclaude/hook-service.ts",
+1
View File
@@ -10,6 +10,7 @@
"../src/main/gitlab/mappers.ts",
"../src/main/ipc/worktree-branch-name.ts",
"../src/main/ipc/worktree-logic.ts",
"../src/main/ipc/worktree-display-name.ts",
"../src/main/ipc/worktree-linked-work-item-metadata.ts",
"../src/main/ipc/worktree-metadata-merge.ts",
"../src/main/ipc/worktree-path-comparison.ts",
+7 -3
View File
@@ -3,18 +3,22 @@ title: How to use GLM-5.2 in Orca ADE
description: Configure Claude Code and other CLI agent harnesses to run GLM-5.2 inside Orca worktrees.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
GLM-5.2 works in Orca through the agent harness you already use. Configure GLM-5.2 in Claude Code, OpenCode, Cline, Kilo Code, Roo Code, Droid, OpenClaw, or another CLI agent, then launch that agent from Orca's picker.
Orca supplies the isolated worktree, terminal panes, browser tab, review flow, and session management. Your [Z.ai CodePlan subscription](https://z.ai/subscribe) and agent config supply the model access.
<Callout title="Prerequisite">
You need an active [Z.ai CodePlan subscription](https://z.ai/subscribe) with GLM Coding Plan access before configuring GLM-5.2 in an agent harness. OpenAI-compatible harnesses also need a Z.ai API key. Orca does not include or resell GLM access.
You need an active [Z.ai CodePlan subscription](https://z.ai/subscribe) with GLM Coding Plan
access before configuring GLM-5.2 in an agent harness. OpenAI-compatible harnesses also need a
Z.ai API key. Orca does not include or resell GLM access.
</Callout>
<Callout title="Source">
This page documents the GLM-5.2 configuration tested with Orca. Z.ai's [model guide](https://docs.z.ai/devpack/latest-model) may list newer models; verify model names, context limits, and harness compatibility there before substituting one.
This page documents the GLM-5.2 configuration tested with Orca. Z.ai's [model
guide](https://docs.z.ai/devpack/latest-model) may list newer models; verify model names, context
limits, and harness compatibility there before substituting one.
</Callout>
## Claude Code
@@ -3,12 +3,13 @@ title: Agent hibernation
description: Let Orca pause idle background agent terminals and auto-resume them when you reopen the worktree.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
When you keep dozens of worktrees open, idle agents add up — each one is a live PTY holding a model session in memory. Agent hibernation lets Orca quietly stop those terminals once they've been done and untouched long enough, then resume the same session the next time you open the worktree.
<Callout title="Experimental">
Agent hibernation is off by default. Turn it on under **Settings → Experimental → Agent hibernation** while we keep tuning the safety model.
Agent hibernation is off by default. Turn it on under **Settings → Experimental → Agent
hibernation** while we keep tuning the safety model.
</Callout>
## What gets hibernated
@@ -2,7 +2,7 @@
title: Agent hooks & memory
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca plays nicely with the agent hook and memory conventions Claude Code and Codex already use — it reads them, respects them, and gives you a UI for the ones that make sense in an IDE context.
@@ -27,5 +27,6 @@ Claude's `CLAUDE.md` and Codex's `AGENTS.md` (at repo root or nested) are left a
Hook endpoints are written to disk (`{userData}/agent-hooks/endpoint.env` on POSIX, `endpoint.cmd` on Windows) and re-sourced on every hook invocation, so long-lived agent sessions keep reaching the live Orca server even after an app restart — no more dead-port POSTs from a PTY that outlived the previous session.
<Callout>
The Orca CLI exposes a commented worktree status field agents can update themselves. See [Worktree checkpoints](/docs/cli/worktree-checkpoints).
The Orca CLI exposes a commented worktree status field agents can update themselves. See [Worktree
checkpoints](/docs/cli/worktree-checkpoints).
</Callout>
@@ -3,7 +3,7 @@ title: Chat UI (native chat)
description: Optional chat surface over supported agent terminals — skills, model pickers, and transcript view.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Chat UI is an experimental view layered on supported agent terminal sessions. The terminal remains the source of truth; Chat UI is a structured transcript + composer for the same PTY. Transcript decoding covers **Claude**, **Codex**, **Grok**, and **OMP** — OMP sessions open in Chat UI like the others instead of staying raw-terminal-only.
@@ -30,5 +30,6 @@ When Claude shows an **AskUserQuestion** (or similar structured permission/quest
Chat UI ships on desktop for supported local and remote (paired server) agent sessions. The [mobile companion](/docs/mobile) reuses chat-style transcript patterns for the same paired sessions.
<Callout title="Experimental">
Transcript fidelity, streaming, and terminal parity are still under active tuning. Prefer the raw TUI when you need every OSC/status detail.
Transcript fidelity, streaming, and terminal parity are still under active tuning. Prefer the raw
TUI when you need every OSC/status detail.
</Callout>
@@ -3,7 +3,7 @@ title: Agent session history
description: Browse and resume past Claude, Codex, Cursor, Gemini, and other agent sessions from Orca's right sidebar.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca scans the on-disk session transcripts that supported agent CLIs leave behind and lists them in a right-sidebar panel called **Agent Session History**. Pick a past session, click **Resume**, and Orca runs the agent's resume command in a fresh terminal — same `cwd`, same session ID, no manual `--resume` flag wrangling.
@@ -39,13 +39,16 @@ Click a session row to open its details: working directory, branch, model, messa
- **Resume** — opens a new terminal in the session's `cwd` and runs the agent's resume command (e.g. `claude --resume <id>`, `codex resume <id>`, `pi --session <session_file>`, `prime-agent --resume <path>`, `cursor-agent --resume <id>`, `acli rovodev run --restore <id>`). Codex sessions also re-export `CODEX_HOME` when the original session set one.
Pi resumes from the on-disk session file reported by its hooks (`--session <path>`), not from a bare session id. If that file is missing, Resume is unavailable for that row even when a session id exists.
- **Copy resume command** — copies the same shell command to the clipboard for use in an external terminal.
- **Copy session ID** / **Copy log path** — for scripting or attaching transcripts to bug reports.
- **Open log** / **Reveal log** — open the raw transcript file in Orca, or jump to it in your OS file manager.
- **Open cwd** — open the session's working directory as a workspace.
<Callout title="Resume needs a local workspace">
Resume runs the agent CLI on the machine where Orca is rendering. If you're connected to a remote workspace, switch back to a local one (or use **Copy resume command** and run it on the remote yourself) before clicking **Resume**.
Resume runs the agent CLI on the machine where Orca is rendering. If you're connected to a remote
workspace, switch back to a local one (or use **Copy resume command** and run it on the remote
yourself) before clicking **Resume**.
</Callout>
## Where the transcripts come from
+42 -39
View File
@@ -3,12 +3,15 @@ title: Supported agents
description: Every agent Orca ships with out of the box.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca works with **any CLI agent** — the agent combobox just launches a process in a terminal. The following ship preconfigured in the built-in agent picker with one-click launch/setup; deeper hooks, status, usage tracking, and account switching are noted where supported.
<Callout title="Permission safety">
The defaults below pass each agent's permission-bypass flag for new launches. A worktree is an isolated checkout, not a security sandbox: the agent can still access files and network resources available to its process. Choose **Manual** in **Settings → Agents → Agent Permissions** unless you intentionally trust the agent and the task.
The defaults below pass each agent's permission-bypass flag for new launches. A worktree is an
isolated checkout, not a security sandbox: the agent can still access files and network resources
available to its process. Choose **Manual** in **Settings → Agents → Agent Permissions** unless
you intentionally trust the agent and the task.
</Callout>
## Permissions default
@@ -19,40 +22,40 @@ Use **Settings → Agents → Agent Permissions** when you want to switch all un
To restore prompts for one agent only, edit that agent's default arguments or environment in Settings. Orca treats a non-empty custom value as an explicit override and opts that agent out of future permission-mode migrations.
| Agent | Notes | Docs |
| --- | --- | --- |
| Claude Code | Deep integration: usage, hot-swap, hooks | [Anthropic](https://docs.anthropic.com/claude/docs/claude-code) |
| Claude Agent Teams | Disabled by default — enable under Settings → Agents to launch via `orca claude-teams` with native panes for each teammate | [Anthropic](https://code.claude.com/docs/agent-teams) |
| Codex | Deep integration: usage, hot-swap | [OpenAI](https://github.com/openai/codex) |
| Grok | Auto-setup | [xAI](https://x.ai/cli) |
| GitHub Copilot CLI | Auto-setup | [GitHub](https://docs.github.com/en/copilot/how-tos/set-up/install-copilot-cli) |
| OpenCode | Auto-setup, status | [OpenCode](https://opencode.ai/docs/cli/) |
| Pi | Auto-setup, hooks, status | [Pi](https://pi.dev) |
| OMP | Auto-setup, hooks, status | [OMP](https://omp.sh) |
| Prime Agent | Auto-setup, hooks, status, session history | [Prime Intellect](https://github.com/PrimeIntellect-ai/prime-agent) |
| Gemini | Auto-setup | [Google](https://github.com/google-gemini/gemini-cli) |
| Antigravity | Auto-setup, hooks, status | [Google](https://antigravity.google/docs/cli-overview) |
| Ante | Auto-setup, status | [Ante](https://github.com/AntigmaLabs/ante-preview) |
| Aider | Auto-setup | [Aider](https://aider.chat/docs/) |
| Goose | Auto-setup | [Block](https://block.github.io/goose/docs/quickstart/) |
| Amp | Auto-setup | [Amp](https://ampcode.com/manual#install) |
| Kilocode | Auto-setup | [Kilo](https://kilo.ai/docs/cli) |
| Kiro | Auto-setup | [Kiro](https://kiro.dev/docs/cli/) |
| Charm Crush | Auto-setup | [Charm](https://github.com/charmbracelet/crush) |
| Auggie | Auto-setup | [Augment](https://docs.augmentcode.com/cli/overview) |
| Autohand | Auto-setup | [Autohand](https://github.com/autohandai/code-cli) |
| Cline | Auto-setup | [Cline](https://docs.cline.bot/cline-cli/overview) |
| Codebuff | Auto-setup | [Codebuff](https://www.codebuff.com/docs/help/quick-start) |
| Command Code | Auto-setup, status | [Command Code](https://commandcode.ai/docs/quickstart) |
| Continue | Auto-setup | [Continue](https://docs.continue.dev/guides/cli) |
| Cursor CLI | Deep integration | [Cursor](https://cursor.com/cli) |
| Devin | Auto-setup | [Devin](https://devin.ai/cli) |
| Droid (Factory) | Auto-setup, hooks, status | [Factory](https://docs.factory.ai/cli/getting-started/quickstart) |
| Kimi | Auto-setup | [Moonshot](https://www.kimi.com/code/docs/en/kimi-code-cli/getting-started.html) |
| Mistral Vibe | Auto-setup | [Mistral](https://github.com/mistralai/mistral-vibe) |
| MiniMax | Auto-setup, usage tracking, rate-limit tracking | [MiniMax](https://www.minimax.chat) |
| Qwen Code | Auto-setup via the installed `qwen` executable | [Qwen](https://github.com/QwenLM/qwen-code) |
| Rovo Dev | Auto-setup | [Atlassian](https://support.atlassian.com/rovo/docs/install-and-run-rovo-dev-cli-on-your-device/) |
| Hermes | Auto-setup | [Nous](https://hermes-agent.nousresearch.com/docs/) |
| OpenClaw | Auto-setup | [OpenClaw](https://github.com/openclaw/openclaw) |
| Trae | Auto-setup via `traecli` (TRAE CN CLI) | [Trae](https://www.trae.ai/) |
| Agent | Notes | Docs |
| ------------------ | -------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- |
| Claude Code | Deep integration: usage, hot-swap, hooks | [Anthropic](https://docs.anthropic.com/claude/docs/claude-code) |
| Claude Agent Teams | Disabled by default — enable under Settings → Agents to launch via `orca claude-teams` with native panes for each teammate | [Anthropic](https://code.claude.com/docs/agent-teams) |
| Codex | Deep integration: usage, hot-swap | [OpenAI](https://github.com/openai/codex) |
| Grok | Auto-setup | [xAI](https://x.ai/cli) |
| GitHub Copilot CLI | Auto-setup | [GitHub](https://docs.github.com/en/copilot/how-tos/set-up/install-copilot-cli) |
| OpenCode | Auto-setup, status | [OpenCode](https://opencode.ai/docs/cli/) |
| Pi | Auto-setup, hooks, status | [Pi](https://pi.dev) |
| OMP | Auto-setup, hooks, status | [OMP](https://omp.sh) |
| Prime Agent | Auto-setup, hooks, status, session history | [Prime Intellect](https://github.com/PrimeIntellect-ai/prime-agent) |
| Gemini | Auto-setup | [Google](https://github.com/google-gemini/gemini-cli) |
| Antigravity | Auto-setup, hooks, status | [Google](https://antigravity.google/docs/cli-overview) |
| Ante | Auto-setup, status | [Ante](https://github.com/AntigmaLabs/ante-preview) |
| Aider | Auto-setup | [Aider](https://aider.chat/docs/) |
| Goose | Auto-setup | [Block](https://block.github.io/goose/docs/quickstart/) |
| Amp | Auto-setup | [Amp](https://ampcode.com/manual#install) |
| Kilocode | Auto-setup | [Kilo](https://kilo.ai/docs/cli) |
| Kiro | Auto-setup | [Kiro](https://kiro.dev/docs/cli/) |
| Charm Crush | Auto-setup | [Charm](https://github.com/charmbracelet/crush) |
| Auggie | Auto-setup | [Augment](https://docs.augmentcode.com/cli/overview) |
| Autohand | Auto-setup | [Autohand](https://github.com/autohandai/code-cli) |
| Cline | Auto-setup | [Cline](https://docs.cline.bot/cline-cli/overview) |
| Codebuff | Auto-setup | [Codebuff](https://www.codebuff.com/docs/help/quick-start) |
| Command Code | Auto-setup, status | [Command Code](https://commandcode.ai/docs/quickstart) |
| Continue | Auto-setup | [Continue](https://docs.continue.dev/guides/cli) |
| Cursor CLI | Deep integration | [Cursor](https://cursor.com/cli) |
| Devin | Auto-setup | [Devin](https://devin.ai/cli) |
| Droid (Factory) | Auto-setup, hooks, status | [Factory](https://docs.factory.ai/cli/getting-started/quickstart) |
| Kimi | Auto-setup | [Moonshot](https://www.kimi.com/code/docs/en/kimi-code-cli/getting-started.html) |
| Mistral Vibe | Auto-setup | [Mistral](https://github.com/mistralai/mistral-vibe) |
| MiniMax | Auto-setup, usage tracking, rate-limit tracking | [MiniMax](https://www.minimax.chat) |
| Qwen Code | Auto-setup via the installed `qwen` executable | [Qwen](https://github.com/QwenLM/qwen-code) |
| Rovo Dev | Auto-setup | [Atlassian](https://support.atlassian.com/rovo/docs/install-and-run-rovo-dev-cli-on-your-device/) |
| Hermes | Auto-setup | [Nous](https://hermes-agent.nousresearch.com/docs/) |
| OpenClaw | Auto-setup | [OpenClaw](https://github.com/openclaw/openclaw) |
| Trae | Auto-setup via `traecli` (TRAE CN CLI) | [Trae](https://www.trae.ai/) |
@@ -16,7 +16,7 @@ Orca reads the local usage state each agent maintains on disk (under `~/.claude`
## Multi-account accounting
The status bar always reflects the *active* account. Other configured accounts are visible in the account switcher with their own usage.
The status bar always reflects the _active_ account. Other configured accounts are visible in the account switcher with their own usage.
## Usage roster
@@ -2,11 +2,14 @@
title: Design Mode
---
import { ImagePlaceholder } from '@/components/docs/prose';
import { ImagePlaceholder } from '@/components/docs/prose'
Design Mode turns the Orca browser into a pointer-to-code tool. Toggle it on, click any UI element on the rendered page, and the element drops into the agent chat as rich context — with its DOM, computed styles, and a screenshot.
<ImagePlaceholder src="/docs/orca-design-mode.gif" caption="Design Mode: click a button, it lands in the agent chat" />
<ImagePlaceholder
src="/docs/orca-design-mode.gif"
caption="Design Mode: click a button, it lands in the agent chat"
/>
## Turn it on
+5 -2
View File
@@ -2,11 +2,14 @@
title: Per-worktree browser
---
import { ImagePlaceholder } from '@/components/docs/prose';
import { ImagePlaceholder } from '@/components/docs/prose'
Every Orca worktree has its own browser. It's a real Chromium window — address bar, history, devtools — embedded in a pane. Tabs are scoped to the worktree, so the app you're building against stays out of the way of your other work.
<ImagePlaceholder src="/docs/orca-design-mode.gif" caption="Per-worktree browser pane with address bar and tab strip" />
<ImagePlaceholder
src="/docs/orca-design-mode.gif"
caption="Per-worktree browser pane with address bar and tab strip"
/>
## Controls
+4 -2
View File
@@ -3,12 +3,14 @@ title: Computer use
description: Drive local desktop apps from an agent via accessibility trees, screenshots, and safe UI actions.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
The `orca computer` CLI lets an agent inspect and control native desktop apps — list running apps, read accessibility trees, click controls, set values, type text, scroll, and take screenshots. Use it when a task needs to operate the OS or a third-party app rather than a terminal or the built-in browser.
<Callout title="Beta">
Computer use ships native helpers per platform and requires Accessibility (and on macOS, Screen Recording) permission. The command surface is stable enough for skills to build against, but flag names may still shift.
Computer use ships native helpers per platform and requires Accessibility (and on macOS, Screen
Recording) permission. The command surface is stable enough for skills to build against, but flag
names may still shift.
</Callout>
## First-time setup
+6 -3
View File
@@ -3,18 +3,21 @@ title: Orchestration
description: Coordinate agents with Runs, tasks, supervised workers, messages, and decision gates.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orchestration is Orca's structured multi-agent layer: a **Run** (namespace + coordinator inbox), **Tasks**, **Dispatches**, supervised **workers**, messages, and decision gates.
Use it when you need ownership, completion tracking, or a DAG. For one-off prompts, use `orca terminal send`. For full ownership handoffs without supervision, use worktree/terminal commands from the `orca-cli` skill.
<Callout title="Experimental">
Enable orchestration under Settings → Experimental before using these commands. The CLI talks to the running Orca runtime, so `orca status --json` should succeed first.
Enable orchestration under Settings → Experimental before using these commands. The CLI talks to
the running Orca runtime, so `orca status --json` should succeed first.
</Callout>
<Callout title="Legacy commands retired">
`orca orchestration run` and `run-stop` (and `coordinator-start` / `coordinator-stop`) perform **no effects**. They return recovery text pointing at `orca skills get orchestration --full`. Use the Run + worker-start flow below.
`orca orchestration run` and `run-stop` (and `coordinator-start` / `coordinator-stop`) perform
**no effects**. They return recovery text pointing at `orca skills get orchestration --full`. Use
the Run + worker-start flow below.
</Callout>
## Core model
+4 -2
View File
@@ -10,7 +10,7 @@ keywords:
- agent CLI
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
The Orca CLI is the `orca` command-line interface for scripting a running Orca editor from any shell. Use it to create and inspect worktrees, drive agent terminals, open files and diffs, automate the built-in browser, run scheduled automations, share HTML/Markdown artifacts, and control Orca-native tools from scripts or AI agents.
@@ -118,5 +118,7 @@ orca emulator kill --json
Use `--worktree <selector>`, `--device <udid-or-name>`, or `--emulator <id>` when a script needs an explicit target.
<Callout>
For the full command surface including tabs, waits, cookies, and frames, see [Orca CLI reference](/docs/cli/reference), then install the Orca CLI skill (see [Skills registry](/docs/cli/skills)) and point your agent at it.
For the full command surface including tabs, waits, cookies, and frames, see [Orca CLI
reference](/docs/cli/reference), then install the Orca CLI skill (see [Skills
registry](/docs/cli/skills)) and point your agent at it.
</Callout>
+3 -2
View File
@@ -3,7 +3,7 @@ title: Orca CLI reference
description: Commands, selectors, and agent-friendly patterns for driving Orca from a shell.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
The `orca` CLI talks to a running Orca runtime. Use it when a shell script or agent needs to inspect worktrees, launch terminals, open files, automate the built-in browser, or report progress back into Orca.
@@ -116,7 +116,8 @@ orca terminal close --terminal <handle> --json
Omit `--terminal` to target the active terminal in the current worktree. Read before sending when you are not sure what the terminal is waiting for.
<Callout title="Terminal handles">
Terminal handles are runtime-scoped. If Orca restarts or a command reports a stale terminal handle, run `orca terminal list --json` and reacquire the handle.
Terminal handles are runtime-scoped. If Orca restarts or a command reports a stale terminal
handle, run `orca terminal list --json` and reacquire the handle.
</Callout>
`terminal list` reports each terminal's `executionHostId` when Orca can verify it, plus a result-level `hostScope` with covered and omitted host IDs. Treat a missing host identity or scope as **unverifiable**, not local. A missing terminal is evidence that it exited only when its execution host is listed in `hostScope.hostIds`.
+10 -10
View File
@@ -12,21 +12,21 @@ keywords:
- orca-emulator-android skill
---
Orca ships **skills** that agents install into their skill directories. Public install packages are **hybrid discovery stubs**: short `SKILL.md` files that tell the agent *when* to engage Orca and how to load the full guide from the running CLI. Command flags live in the binary so they cannot drift from the app version.
Orca ships **skills** that agents install into their skill directories. Public install packages are **hybrid discovery stubs**: short `SKILL.md` files that tell the agent _when_ to engage Orca and how to load the full guide from the running CLI. Command flags live in the binary so they cannot drift from the app version.
## Installable Orca skills
Use `npx skills add` with the public Orca repo and the skill name. Default agent setup usually installs `orca-cli`, `computer-use`, and `orchestration`.
| Skill | Install | Use it for |
| --- | --- | --- |
| [`orca-cli`](#orca-cli) | `npx skills add https://github.com/stablyai/orca --skill orca-cli --global` | Worktrees, terminals, files, automations, embedded browser. |
| [`orchestration`](#orchestration) | `npx skills add https://github.com/stablyai/orca --skill orchestration --global` | Multi-agent Runs, tasks, supervised workers, messages, gates. |
| [`computer-use`](#computer-use) | `npx skills add https://github.com/stablyai/orca --skill computer-use --global` | Desktop apps via accessibility trees and safe UI actions. |
| [`orca-linear`](#orca-linear) | `npx skills add https://github.com/stablyai/orca --skill orca-linear --global` | Linear ticket read/write through `orca linear`. |
| [`orca-emulator`](#orca-emulator) | `npx skills add https://github.com/stablyai/orca --skill orca-emulator --global` | iOS Simulator control. |
| [`orca-emulator-android`](#orca-emulator-android) | `npx skills add https://github.com/stablyai/orca --skill orca-emulator-android --global` | Android emulator/device via adb. |
| [`orca-per-workspace-env`](#orca-per-workspace-env) | `npx skills add https://github.com/stablyai/orca --skill orca-per-workspace-env --global` | Per-workspace environment recipes (`orca.yaml`). |
| Skill | Install | Use it for |
| --------------------------------------------------- | ----------------------------------------------------------------------------------------- | ------------------------------------------------------------- |
| [`orca-cli`](#orca-cli) | `npx skills add https://github.com/stablyai/orca --skill orca-cli --global` | Worktrees, terminals, files, automations, embedded browser. |
| [`orchestration`](#orchestration) | `npx skills add https://github.com/stablyai/orca --skill orchestration --global` | Multi-agent Runs, tasks, supervised workers, messages, gates. |
| [`computer-use`](#computer-use) | `npx skills add https://github.com/stablyai/orca --skill computer-use --global` | Desktop apps via accessibility trees and safe UI actions. |
| [`orca-linear`](#orca-linear) | `npx skills add https://github.com/stablyai/orca --skill orca-linear --global` | Linear ticket read/write through `orca linear`. |
| [`orca-emulator`](#orca-emulator) | `npx skills add https://github.com/stablyai/orca --skill orca-emulator --global` | iOS Simulator control. |
| [`orca-emulator-android`](#orca-emulator-android) | `npx skills add https://github.com/stablyai/orca --skill orca-emulator-android --global` | Android emulator/device via adb. |
| [`orca-per-workspace-env`](#orca-per-workspace-env) | `npx skills add https://github.com/stablyai/orca --skill orca-per-workspace-env --global` | Per-workspace environment recipes (`orca.yaml`). |
## Hybrid stubs vs the live guide
+6 -6
View File
@@ -34,12 +34,12 @@ YAML and TOML front matter is shown in the rich editor and rendered preview by d
In rich markdown tables:
| Key | Behavior |
| --- | --- |
| **Tab** / **Shift-Tab** | Next / previous cell; Tab past the last cell inserts a row |
| **Enter** | Move to the cell below; on the last row, add a row |
| **Backspace** on a fully empty row | Delete the row (or the whole table if it is the last row) |
| **Backspace** in an empty cell when the row still has content | Step to the previous cell |
| Key | Behavior |
| ------------------------------------------------------------- | ---------------------------------------------------------- |
| **Tab** / **Shift-Tab** | Next / previous cell; Tab past the last cell inserts a row |
| **Enter** | Move to the cell below; on the last row, add a row |
| **Backspace** on a fully empty row | Delete the row (or the whole table if it is the last row) |
| **Backspace** in an empty cell when the row still has content | Step to the previous cell |
Use **Shift-Tab** to unindent a list item or the selected lines of a code block.
+3 -2
View File
@@ -2,7 +2,7 @@
title: HTML, Mermaid, PDF & image viewers
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca includes built-in viewers for the formats that show up in most repos.
@@ -35,5 +35,6 @@ Scroll, zoom, and text selection. Useful for design docs checked into the repo.
`.ipynb` files open in a notebook viewer with rendered markdown, syntax-highlighted code cells, and saved outputs. Editing cells writes back to the on-disk `.ipynb` while preserving nbformat, so diffs stay clean.
<Callout title="Beta">
The notebook editor is marked beta. Cell execution and richer output rendering are still settling — file an issue if a notebook in your repo doesn't load cleanly.
The notebook editor is marked beta. Cell execution and richer output rendering are still settling
— file an issue if a notebook in your repo doesn't load cleanly.
</Callout>
+3 -2
View File
@@ -3,7 +3,7 @@ title: Your first 3-agent session
description: From empty app to three agents running in parallel in under five minutes.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
This is the single most important page in the docs. By the end you'll have three agents running in parallel on three different approaches to the same task, with one PR shipped.
@@ -48,5 +48,6 @@ Once agents settle, open each worktree's diff view. Use [Annotate AI Diff](/docs
Commit and push directly from Orca — see [Commit & push from Orca](/docs/review/commit-push). The other two worktrees can be deleted with one click; their branches go with them.
<Callout title="That's it">
This flow — add → worktree → agent → split → diff → ship — is the whole of Orca. Every other page in these docs is a deeper look at one of those steps.
This flow — add → worktree → agent → split → diff → ship — is the whole of Orca. Every other page
in these docs is a deeper look at one of those steps.
</Callout>
+18 -18
View File
@@ -9,14 +9,14 @@ This page covers the errors you’ll see most often and how to fix them.
## Quick triage
| What you see | Likely cause | First thing to try |
| --- | --- | --- |
| “GitHub is rate-limiting requests” / “rate limit exceeded (core)” | GitHub REST (core) quota exhausted for your user | Wait for reset; stop extra `gh` / agent / Orca usage; check [Settings → Git → GitHub API Budget](/docs/settings) |
| “GitHub authentication is unavailable” / `gh auth` prompts | `gh` not logged in, expired token, or bad `GITHUB_TOKEN` | `gh auth status`, then `gh auth login` |
| “GitHub did not allow access” / HTTP 403 (not rate limit) | Missing scopes or no access to the repo | Re-auth with `repo` (and needed org SSO); confirm you can open the PR in the browser |
| “repository is unavailable” / HTTP 404 | Wrong remote, private repo without access, or renamed repo | Check `git remote -v` and browser access |
| “GitHub is unreachable” / timeouts | Network, proxy, VPN, or GitHub outage | Check [githubstatus.com](https://www.githubstatus.com/); retry off VPN |
| “GitHub CLI is unavailable” | `gh` missing from PATH Orca uses | Install `gh` and restart Orca |
| What you see | Likely cause | First thing to try |
| ----------------------------------------------------------------- | ---------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------- |
| “GitHub is rate-limiting requests” / “rate limit exceeded (core)” | GitHub REST (core) quota exhausted for your user | Wait for reset; stop extra `gh` / agent / Orca usage; check [Settings → Git → GitHub API Budget](/docs/settings) |
| “GitHub authentication is unavailable” / `gh auth` prompts | `gh` not logged in, expired token, or bad `GITHUB_TOKEN` | `gh auth status`, then `gh auth login` |
| “GitHub did not allow access” / HTTP 403 (not rate limit) | Missing scopes or no access to the repo | Re-auth with `repo` (and needed org SSO); confirm you can open the PR in the browser |
| “repository is unavailable” / HTTP 404 | Wrong remote, private repo without access, or renamed repo | Check `git remote -v` and browser access |
| “GitHub is unreachable” / timeouts | Network, proxy, VPN, or GitHub outage | Check [githubstatus.com](https://www.githubstatus.com/); retry off VPN |
| “GitHub CLI is unavailable” | `gh` missing from PATH Orca uses | Install `gh` and restart Orca |
## Rate limits (most common)
@@ -24,11 +24,11 @@ GitHub gives each **authenticated user** a shared hourly budget. **Every tool on
### Buckets Orca cares about
| Bucket | What it covers | Typical limit (authenticated) |
| --- | --- | --- |
| **REST (core)** | Most PR/issue/API calls (`gh pr view`, checks metadata, many REST endpoints) | 5,000 / hour |
| **GraphQL** | Project/Tasks and some richer PR queries | 5,000 points / hour |
| **Search** | Search-driven lists | 30 / minute |
| Bucket | What it covers | Typical limit (authenticated) |
| --------------- | ---------------------------------------------------------------------------- | ----------------------------- |
| **REST (core)** | Most PR/issue/API calls (`gh pr view`, checks metadata, many REST endpoints) | 5,000 / hour |
| **GraphQL** | Project/Tasks and some richer PR queries | 5,000 points / hour |
| **Search** | Search-driven lists | 30 / minute |
When a primary bucket is exhausted, GitHub returns HTTP **403** with a message like `API rate limit exceeded`. Orca classifies that as rate-limited, keeps the last known PR status when it can, and **stops spawning more `gh` calls** for a short window so a single limit doesn’t turn into a storm of failures.
@@ -106,11 +106,11 @@ GitHub auth is **per host**. Logging in on your laptop does not log in `gh` on a
## Permission and repository errors
| Symptom | Meaning |
| --- | --- |
| HTTP 403 without “rate limit” | Token lacks scope or you’re not allowed to see the resource |
| HTTP 404 / “could not resolve to a Repository” | Repo missing, renamed, or invisible to this token |
| “resource not accessible by integration” | App/token type can’t perform that action |
| Symptom | Meaning |
| ---------------------------------------------- | ----------------------------------------------------------- |
| HTTP 403 without “rate limit” | Token lacks scope or you’re not allowed to see the resource |
| HTTP 404 / “could not resolve to a Repository” | Repo missing, renamed, or invisible to this token |
| “resource not accessible by integration” | App/token type can’t perform that action |
Fixes:
+5 -3
View File
@@ -1,9 +1,9 @@
---
title: What is Orca?
description: "A 60-second pitch: who Orca is for and when to reach for it."
description: 'A 60-second pitch: who Orca is for and when to reach for it.'
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca is a desktop IDE for running multiple AI coding agents side by side. Every task gets its own git worktree, its own agent terminal, and its own browser tab — so you can fan out work across Claude Code, Codex, Cursor CLI, and friends without stashing, branch-juggling, or losing flow.
@@ -25,5 +25,7 @@ Orca is designed for people who already write code for a living and want to use
- **Not a hosted VPS product.** Orca runs on your desktop by default. Remote compute uses machines and cloud accounts you control — [SSH targets](/docs/ssh), [self-hosted Orca servers](/docs/remote-servers), or [Cloud VMs / per-workspace environments](/docs/ways-to-run#4-cloud-vms-per-workspace-environments).
<Callout title="Next steps">
Head to [Install](/docs/install), then walk through [Your first 3-agent session](/docs/first-session) — the single most important page in these docs. When you're ready to move agents off the laptop, start with [Ways to run Orca](/docs/ways-to-run).
Head to [Install](/docs/install), then walk through [Your first 3-agent
session](/docs/first-session) — the single most important page in these docs. When you're ready to
move agents off the laptop, start with [Ways to run Orca](/docs/ways-to-run).
</Callout>
+18 -13
View File
@@ -3,7 +3,7 @@ title: Install
description: Download Orca for macOS, Windows, or Linux, and opt into RC builds.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
## Download
@@ -21,17 +21,20 @@ import { Callout } from '@/components/docs/prose';
<ul className="hidden md:block">
<li>
**macOS:** [Apple Silicon](https://github.com/stablyai/orca/releases/latest/download/orca-macos-arm64.dmg) · [Intel](https://github.com/stablyai/orca/releases/latest/download/orca-macos-x64.dmg)
**macOS:** [Apple
Silicon](https://github.com/stablyai/orca/releases/latest/download/orca-macos-arm64.dmg) ·
[Intel](https://github.com/stablyai/orca/releases/latest/download/orca-macos-x64.dmg)
</li>
<li>
**Windows:** [installer](https://github.com/stablyai/orca/releases/latest/download/orca-windows-setup.exe)
**Windows:**
[installer](https://github.com/stablyai/orca/releases/latest/download/orca-windows-setup.exe)
</li>
<li>
**Linux:** [AppImage](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) · [.deb](https://github.com/stablyai/orca/releases)
</li>
<li>
Older versions: [GitHub Releases](https://github.com/stablyai/orca/releases).
**Linux:**
[AppImage](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) ·
[.deb](https://github.com/stablyai/orca/releases)
</li>
<li>Older versions: [GitHub Releases](https://github.com/stablyai/orca/releases).</li>
</ul>
### Homebrew (macOS)
@@ -58,16 +61,18 @@ Orca auto-updates by default, tracking the **stable** channel. Stable releases a
There is no permanent in-app opt-in for the RC channel. Modifier clicks on **Check for Updates** ([Settings → General → Updates](/docs/settings), or the app / Help menu):
| Modifier | Effect |
| --- | --- |
| **Shift+click** | Include the latest **RC** prerelease |
| **Cmd+click** (macOS) / **Ctrl+click** (Windows/Linux) | Latest **perf**-tagged prerelease |
| **Option+click** (macOS only) | Pick a **validated local macOS build** that passes Orca’s compatibility checks |
| Modifier | Effect |
| ------------------------------------------------------ | ------------------------------------------------------------------------------ |
| **Shift+click** | Include the latest **RC** prerelease |
| **Cmd+click** (macOS) / **Ctrl+click** (Windows/Linux) | Latest **perf**-tagged prerelease |
| **Option+click** (macOS only) | Pick a **validated local macOS build** that passes Orca’s compatibility checks |
You can still download any build directly from the [GitHub Releases page](https://github.com/stablyai/orca/releases).
<Callout title="Don't like the current update">
Older versions are always available on the [GitHub Releases page](https://github.com/stablyai/orca/releases). Orca will not force-downgrade your worktree data if you go back.
Older versions are always available on the [GitHub Releases
page](https://github.com/stablyai/orca/releases). Orca will not force-downgrade your worktree data
if you go back.
</Callout>
## Platform notes
+9 -3
View File
@@ -3,12 +3,15 @@ title: Mobile companion
description: Pair the Orca mobile app to your desktop to monitor agents and unstick worktrees from your phone.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
The Orca mobile companion is an iOS/Android app that pairs with your desktop Orca and gives you a read-mostly view of running agents — agent status, recent terminal scrollback, and the controls you actually want from a phone (replying to a prompt, sleeping a worktree, reviewing source control, switching agent accounts). Pairing is one-time and the desktop is always the source of truth.
<Callout title="Beta">
The mobile companion is in beta. Install iOS from the [App Store](https://apps.apple.com/us/app/orca-ide/id6766130217), join the [TestFlight preview channel](https://testflight.apple.com/join/YjeGMQBA), or install Android from the [current APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk).
The mobile companion is in beta. Install iOS from the [App
Store](https://apps.apple.com/us/app/orca-ide/id6766130217), join the [TestFlight preview
channel](https://testflight.apple.com/join/YjeGMQBA), or install Android from the [current APK
0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk).
</Callout>
## What you can do from mobile
@@ -110,5 +113,8 @@ The mobile terminal has a dedicated **Terminal settings** screen (Settings → T
- **Can't reach desktop** — phone and desktop must share a network path (LAN, Tailscale, or the pairing path you used). Closing the desktop app drops the connection; reopen desktop and the phone reconnects automatically.
<Callout title="Next steps">
Pair mobile notifications with [desktop notifications](/docs/notifications) so agent-finished alerts reach the right device. For headless machines, use [Remote Orca Server](/docs/remote-servers) mobile pairing. Track provider usage on desktop and mobile under [Usage & rate-limit tracking](/docs/agents/usage-tracking).
Pair mobile notifications with [desktop notifications](/docs/notifications) so agent-finished
alerts reach the right device. For headless machines, use [Remote Orca
Server](/docs/remote-servers) mobile pairing. Track provider usage on desktop and mobile under
[Usage & rate-limit tracking](/docs/agents/usage-tracking).
</Callout>
@@ -3,7 +3,7 @@ title: Agents & sessions
description: State dots, restart chips, and the lifecycle of an agent session.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
An **agent session** is one CLI agent running in one terminal in one worktree. Orca tracks its lifecycle so you always know which sessions are working and which are idle — without you having to click into each tab to check.
@@ -76,5 +76,6 @@ When an agent exits (clean or crash), the tab shows a **Restart** chip. One clic
1. **Exit** — the process ends; the Restart chip appears.
<Callout>
For the exact detection rules, see the `terminal wait --for tui-idle` command in the [Orca CLI](/docs/cli/overview).
For the exact detection rules, see the `terminal wait --for tui-idle` command in the [Orca
CLI](/docs/cli/overview).
</Callout>
+5 -7
View File
@@ -3,7 +3,7 @@ title: Quick Open & Jump Palette
description: Cmd-J scoped jump across worktrees, recents, and tabs.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Once you have more than a handful of worktrees, navigation becomes the bottleneck. Orca ships two keyboard-first navigation tools.
@@ -13,13 +13,13 @@ File search scoped to the current worktree. Type a fragment; Orca ranks by recen
## New-tab omnibox
The tab strip **+** omnibox searches **open tabs**, files, URLs, and agents in one field (placeholder: *Search open tabs, files, URLs, agents…*). File rows use the same filename-first layout as Quick Open. Matching an already-open editor tab prefers that tab over a duplicate file result, so you jump to the open buffer instead of opening a second copy.
The tab strip **+** omnibox searches **open tabs**, files, URLs, and agents in one field (placeholder: _Search open tabs, files, URLs, agents…_). File rows use the same filename-first layout as Quick Open. Matching an already-open editor tab prefers that tab over a duplicate file result, so you jump to the open buffer instead of opening a second copy.
Type a web search instead of a path or URL to open it in the worktree browser with your [Default Search Engine](/docs/settings). A single token still ranks file matches first; a multi-word phrase promotes the search row. Prefix the query with `?` to skip file and tab matching and search immediately.
## Worktree Jump Palette (Cmd-J)
Jump across every worktree and every tab in one search. The placeholder in the empty input reads *repo/worktree* — type either half and Orca filters accordingly. Once you start typing, search includes non-archived worktrees even if they are hidden by the sidebar's current filters. Slack-style emoji shortcodes (`:rocket:`) use the same suggestion popover as workspace naming.
Jump across every worktree and every tab in one search. The placeholder in the empty input reads _repo/worktree_ — type either half and Orca filters accordingly. Once you start typing, search includes non-archived worktrees even if they are hidden by the sidebar's current filters. Slack-style emoji shortcodes (`:rocket:`) use the same suggestion popover as workspace naming.
Press **Tab** in the palette for a host and project filter menu. Selected hosts and projects narrow the result set and show as chips you can remove one at a time; closing the palette clears the filter so the next open is unscoped.
@@ -32,12 +32,10 @@ Results include:
- Worktrees matched by cached GitHub PR title or number (`#123`) and cached GitLab merge request title or number (`!123`) when that review metadata is already available.
- Every open tab, scoped first by current worktree, then globally. A match in the tab's title or content ranks ahead of a match only in its worktree, branch, or repo; equally strong matches prefer the tab you used more recently. Rows show the last-active age when Orca has one. Type aliases such as `terminal` or `simulator` still match those tab types without cluttering the row label.
When a typed query hits **both** open tabs and worktrees, the palette interleaves a short preview of each section so neither primary list is buried. If a section has more matches, click **See more** in its *N more* row to reveal 20 additional entries at a time. Changing the query resets the expanded sections. Single-section results keep a full hard-capped list.
When a typed query hits **both** open tabs and worktrees, the palette interleaves a short preview of each section so neither primary list is buried. If a section has more matches, click **See more** in its _N more_ row to reveal 20 additional entries at a time. Changing the query resets the expanded sections. Single-section results keep a full hard-capped list.
Shift-Enter on a worktree opens it in a new split instead of swapping the current pane.
When the query does not match an existing worktree, the palette offers a **Create worktree** row using the typed text as the name. Existing matches stay selected first, so pressing Enter still jumps when a real result is available.
<Callout>
Shortcut bindings are remappable under [Settings → Shortcuts](/docs/settings).
</Callout>
<Callout>Shortcut bindings are remappable under [Settings → Shortcuts](/docs/settings).</Callout>
@@ -3,7 +3,7 @@ title: Session restore
description: Quit Orca, reopen it, and pick up exactly where you left off — worktrees, splits, scrollback, focused tab.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
When you close Orca, the next launch rehydrates the whole workspace: every open worktree, every terminal split, the scrollback in each pane, and the tab you had focused. You shouldn't have to remember which agents you had running on which branches — that's Orca's job.
@@ -26,15 +26,18 @@ A daemon crash while Orca is closed has the same effect for any sessions it was
Session restore runs on every launch. The cases divide by whether the **daemon** survived the gap or not:
**Daemon survives → agents keep running:**
- **Cmd-Q** — the normal way to quit. Agents keep working in the background.
- **Auto-updater relaunch** — Orca restarts to install an update; agents are unaffected.
- **App crash** — if Orca itself crashes, the daemon keeps your sessions alive for warm reattach on next launch.
**Daemon dies → agents are gone, layout still restores:**
- **Host reboot** — laptop restart, OS update, kernel panic, hard power-off. Worktrees, tabs, splits, and the last-persisted scrollback still come back on next launch.
<Callout title="Starting fresh">
If you want a clean slate, close worktrees explicitly before quitting. There is no "open in a new session" mode — Orca always restores. Closed worktrees stay closed.
If you want a clean slate, close worktrees explicitly before quitting. There is no "open in a new
session" mode — Orca always restores. Closed worktrees stay closed.
</Callout>
## Next steps
@@ -3,11 +3,14 @@ title: Tabs, panes & split layouts
description: Drag-to-split panes, tab groups, and pinned boundaries.
---
import { ImagePlaceholder } from '@/components/docs/prose';
import { ImagePlaceholder } from '@/components/docs/prose'
Orca's pane system is designed for watching multiple agents work without losing context. Tabs group into panes; panes split into layouts.
<ImagePlaceholder src="/docs/tab-split.gif" caption="Drag a tab to a pane edge to split — terminals, diffs, and browser tabs side by side" />
<ImagePlaceholder
src="/docs/tab-split.gif"
caption="Drag a tab to a pane edge to split — terminals, diffs, and browser tabs side by side"
/>
## Tabs
@@ -22,11 +25,11 @@ Each tab holds one thing: a terminal, an editor buffer, a browser, a diff, a PR.
Default chords on **new installs**:
| Action | macOS | Linux / Windows |
| --- | --- | --- |
| Next / previous tab (all types) | `Cmd+Shift+]` / `Cmd+Shift+[` | `Ctrl+Shift+]` / `Ctrl+Shift+[` |
| Next / previous tab (same type) | `Cmd+Option+]` / `Cmd+Option+[` | `Ctrl+Alt+]` / `Ctrl+Alt+[` |
| Previous recent tab | `Ctrl+Tab` | `Ctrl+Tab` |
| Action | macOS | Linux / Windows |
| ------------------------------- | ------------------------------- | ------------------------------- |
| Next / previous tab (all types) | `Cmd+Shift+]` / `Cmd+Shift+[` | `Ctrl+Shift+]` / `Ctrl+Shift+[` |
| Next / previous tab (same type) | `Cmd+Option+]` / `Cmd+Option+[` | `Ctrl+Alt+]` / `Ctrl+Alt+[` |
| Previous recent tab | `Ctrl+Tab` | `Ctrl+Tab` |
Remap under [Settings → Shortcuts](/docs/settings). Existing installs keep customized overrides in `~/.orca/keybindings.json`.
+4 -3
View File
@@ -3,7 +3,7 @@ title: Worktrees
description: How Orca turns every feature or bug into its own git worktree.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca is worktree-native. Instead of branching and stashing on one checkout, every task gets its own on-disk copy of the repo via `git worktree`. This is what makes parallel agents safe — they never step on each other's files.
@@ -126,7 +126,7 @@ Bulk-deleting workspaces (sidebar multi-select or Resource Manager cleanup) stil
When you import a parent folder that contains several Git repos, Orca can group those repos under a single **project group** in the sidebar. Each project group exposes a **folder workspace** flow — a worktree-like entry that lives at the parent-folder level and binds its task source to one of the repos underneath, so one feature's GitHub/GitLab/Linear/Jira task surface stays attached to the right repo even though the workspace itself is grouped with its siblings in the sidebar.
To create one, hover the project group's header row in the sidebar and click the **+** action (tooltip: "Create workspace for *group*"). The composer dialog ("Create Folder Workspace") asks you to pick the source project for the workspace's task source, name the workspace, and optionally attach a linked issue or PR. Submit, and the folder workspace appears under the project group alongside the regular repo-scoped worktrees.
To create one, hover the project group's header row in the sidebar and click the **+** action (tooltip: "Create workspace for _group_"). The composer dialog ("Create Folder Workspace") asks you to pick the source project for the workspace's task source, name the workspace, and optionally attach a linked issue or PR. Submit, and the folder workspace appears under the project group alongside the regular repo-scoped worktrees.
When deleting a project group, Orca also offers a checkbox to remove the group's contained projects (the underlying repo registrations) in the same action — so cleanup of a no-longer-used cluster is one confirmation, not several.
@@ -139,5 +139,6 @@ Worktrees you create yourself with `git worktree add` stay external until you sh
In **Settings → General → Workspace**, set global defaults for external-worktree sources: Claude Code worktrees, GSD worktrees, other locations, and any custom absolute-path locations you add. Those defaults apply to current and future worktrees on that host; new global custom roots start hidden. In a project's **Non-Orca worktrees** dialog, you can override a source for that project or return it to the global setting, then search and recover individual hidden worktrees.
<Callout>
If you `git worktree remove` from the CLI, Orca will notice and clean up its own state the next time it refreshes that repo.
If you `git worktree remove` from the CLI, Orca will notice and clean up its own state the next
time it refreshes that repo.
</Callout>
@@ -2,7 +2,7 @@
title: Work on a remote machine over SSH
---
Point Orca at any SSH target — a beefier dev box, a GPU host, a cloud sandbox — and it feels like a local worktree. Same editor, same diff view, same agents, different compute. You can open remote repos *or* just arbitrary folders. For the full menu of local / SSH / server / ephemeral-VM modes, see [Ways to run Orca](/docs/ways-to-run).
Point Orca at any SSH target — a beefier dev box, a GPU host, a cloud sandbox — and it feels like a local worktree. Same editor, same diff view, same agents, different compute. You can open remote repos _or_ just arbitrary folders. For the full menu of local / SSH / server / ephemeral-VM modes, see [Ways to run Orca](/docs/ways-to-run).
## Setup
+12 -10
View File
@@ -3,14 +3,15 @@ title: Remote Orca Servers
description: Keep Orca running on another computer and connect from your laptop.
---
import { Callout, ImagePlaceholder } from '@/components/docs/prose';
import { Callout, ImagePlaceholder } from '@/components/docs/prose'
A Remote Orca Server lets one computer do the work while another computer provides the UI. The server keeps the projects, worktrees, terminals, tabs, provider accounts, and agent sessions. Your laptop connects to that running Orca instance.
The easiest setup is the Orca desktop app on both computers, connected through [Tailscale](https://tailscale.com/). You do not need to run `orca serve` for this path.
<Callout title="Beta">
Remote Orca Servers are beta. Keep the server and client on a private network path you control, such as the same Tailscale tailnet or LAN.
Remote Orca Servers are beta. Keep the server and client on a private network path you control,
such as the same Tailscale tailnet or LAN.
</Callout>
## What runs where
@@ -68,7 +69,8 @@ On the computer that should keep the sessions running:
If the Tailscale address is missing, confirm Tailscale is connected and click the refresh button beside **Connection address**.
<Callout title="Keep the access link private">
The pairing URL grants access to this Orca runtime. Treat it like a password and send it only to the client you intend to pair.
The pairing URL grants access to this Orca runtime. Treat it like a password and send it only to
the client you intend to pair.
</Callout>
### 2. Add the server on your laptop
@@ -158,13 +160,13 @@ Keep the phone on the same tailnet, open Orca Mobile, choose **Pair**, and scan
## Desktop app or `orca serve`?
| | Desktop app on the server | `orca serve` |
| --- | --- | --- |
| Best for | An old laptop, Mac mini, or desktop | A headless Linux box, VM, or managed service |
| Setup | Settings and buttons | Terminal command and service configuration |
| Server window | Open | None |
| Access link | **New Link → Generate Access Link** | Printed in the terminal |
| Lifetime | While the desktop app is running | While the foreground process or service is running |
| | Desktop app on the server | `orca serve` |
| ------------- | ----------------------------------- | -------------------------------------------------- |
| Best for | An old laptop, Mac mini, or desktop | A headless Linux box, VM, or managed service |
| Setup | Settings and buttons | Terminal command and service configuration |
| Server window | Open | None |
| Access link | **New Link → Generate Access Link** | Printed in the terminal |
| Lifetime | While the desktop app is running | While the foreground process or service is running |
## Remote Orca Server or SSH?
+7 -3
View File
@@ -3,7 +3,7 @@ title: Jira items drawer
description: Browse, edit, and link Jira Cloud or self-hosted Server/Data Center issues to worktrees the same way you link Linear or GitHub items.
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Jira sits next to GitHub and Linear in the task drawer. Browse Jira issues, update them, and create a worktree from any issue without leaving Orca.
@@ -27,7 +27,9 @@ Jira sits next to GitHub and Linear in the task drawer. Browse Jira issues, upda
- **Username and password** — Basic auth for older instances without PATs.
<Callout title="Use HTTPS">
Use an `https://` URL for Cloud and self-hosted Jira whenever possible. Orca sends credentials in the `Authorization` header; plain HTTP can expose them on the network. If a server only offers HTTP, put it behind a trusted HTTPS tunnel or VPN before connecting.
Use an `https://` URL for Cloud and self-hosted Jira whenever possible. Orca sends credentials in
the `Authorization` header; plain HTTP can expose them on the network. If a server only offers
HTTP, put it behind a trusted HTTPS tunnel or VPN before connecting.
</Callout>
4. Click **Connect**. Orca verifies the credentials and loads your sites.
@@ -47,7 +49,9 @@ If you don't use Jira at all, hide it from the source picker via [Settings → T
- Orca remembers your last-used task source per repo, so a Jira-driven repo defaults to Jira on next open.
<Callout title="Where credentials live">
Your Atlassian API token or self-hosted credentials are encrypted via the OS keychain and stored locally — they're only used to call your configured Jira site. Revoke tokens from Atlassian account settings if you stop using Orca.
Your Atlassian API token or self-hosted credentials are encrypted via the OS keychain and stored
locally — they're only used to call your configured Jira site. Revoke tokens from Atlassian
account settings if you stop using Orca.
</Callout>
## Next steps
+3 -2
View File
@@ -2,7 +2,7 @@
title: Linear items drawer
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Linear sits next to hosted review providers in the task drawer. Browse, create, update, and link Linear issues to worktrees the same way you link GitHub issues.
@@ -26,7 +26,8 @@ Linear sits next to hosted review providers in the task drawer. Browse, create,
- Orca remembers your last-used task source (GitHub, Linear, or Jira) per repo.
<Callout>
Linear status sync (moving an issue to "In Progress" when a worktree is created) is opt-in per team.
Linear status sync (moving an issue to "In Progress" when a worktree is created) is opt-in per
team.
</Callout>
## Agents and CLI
+5 -5
View File
@@ -9,11 +9,11 @@ Settings are grouped into panes. Everything here is searchable with `Cmd-,` then
- **Orca CLI** — register the bundled command-line tool for shells and agents.
- **Updates** — check for and install updates. Modifier clicks on **Check for Updates**:
| Modifier | Effect |
| --- | --- |
| **Shift+click** | Include the latest **RC** prerelease |
| **Cmd+click** (macOS) / **Ctrl+click** (Windows/Linux) | Latest **perf**-tagged prerelease |
| **Option+click** (macOS only) | Pick a **validated local macOS build** (compatibility-checked). Failures surface as “Could Not Use Local Build” with **Choose Another Build**. |
| Modifier | Effect |
| ------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------- |
| **Shift+click** | Include the latest **RC** prerelease |
| **Cmd+click** (macOS) / **Ctrl+click** (Windows/Linux) | Latest **perf**-tagged prerelease |
| **Option+click** (macOS only) | Pick a **validated local macOS build** (compatibility-checked). Failures surface as “Could Not Use Local Build” with **Choose Another Build**. |
- **Open in menu** — choose apps on the worktree **Open in** menu. VS Code / Insiders enable **Remote SSH** open for SSH worktrees; other editors remain local-path only.
- **UI zoom** — per-install UI scale.
+3 -2
View File
@@ -2,12 +2,13 @@
title: SSH worktrees
---
import { Callout } from '@/components/docs/prose';
import { Callout } from '@/components/docs/prose'
Orca can drive agents on remote machines over SSH — useful for long-running builds, GPU boxes, or any environment where your laptop isn't the right place to run the work.
<Callout title="One of four run modes">
SSH is one way to put agents on remote compute. For local, self-hosted servers, and ephemeral VMs, see [Ways to run Orca](/docs/ways-to-run).
SSH is one way to put agents on remote compute. For local, self-hosted servers, and ephemeral VMs,
see [Ways to run Orca](/docs/ways-to-run).
</Callout>
## Add an SSH target
+10 -5
View File
@@ -3,7 +3,7 @@ title: Privacy & Telemetry
description: What anonymous usage data Orca collects, what it never collects, and how to opt out.
---
import { ImagePlaceholder } from '@/components/docs/prose';
import { ImagePlaceholder } from '@/components/docs/prose'
This page describes the anonymous product-usage telemetry we collect in packaged Orca builds, what we never collect, and how to opt out.
@@ -20,7 +20,7 @@ Alongside each event we include basic build and platform information: the Orca v
The categories of behavior we observe:
- **Lifecycle** — when the app opens. Used to estimate daily, weekly, and monthly active users.
- **Repos and workspaces** — when you add a repo or create a workspace. We record *how* you did it (e.g. folder picker vs. clone URL; command palette vs. drag-and-drop), never the repo name, URL, path, branch name, or any free-form text.
- **Repos and workspaces** — when you add a repo or create a workspace. We record _how_ you did it (e.g. folder picker vs. clone URL; command palette vs. drag-and-drop), never the repo name, URL, path, branch name, or any free-form text.
- **Agents** — when you start an agent, which agent kind it was (from a fixed list like `claude-code`, `codex`, `gemini`, etc.), and where you launched it from. Never the prompt, model details, or agent output.
- **Agent errors** — a coarse error category and which agent kind was involved. We never see raw error messages or stack traces; per-incident detail stays in a local diagnostic trace file on your machine and only reaches Orca if you explicitly share a diagnostic bundle.
- **Settings** — when you toggle one of a small whitelisted set of feature-flag or UX preferences. We record which preference changed and whether it's a boolean or an enum, never the raw value of any free-form setting.
@@ -43,9 +43,14 @@ You can disable telemetry in three ways. Any one of them is sufficient; they com
1. **In the app.** Settings → Privacy → toggle "Share anonymous usage data" off. The change is immediate and persistent.
<ImagePlaceholder src="/docs/privacy-toggle.png" caption="Settings → Privacy — toggle 'Share anonymous usage data' off to disable telemetry." />
2. **`DO_NOT_TRACK=1`** — community-standard environment variable. Disables transmission for that launch. Unsetting it restores your stored preference on the next launch.
3. **`ORCA_TELEMETRY_DISABLED=1`** — Orca-specific kill switch with the same semantics as `DO_NOT_TRACK`.
<ImagePlaceholder
src="/docs/privacy-toggle.png"
caption="Settings → Privacy — toggle 'Share anonymous usage data' off to disable telemetry."
/>
2. **`DO_NOT_TRACK=1`** — community-standard environment variable. Disables transmission for that
launch. Unsetting it restores your stored preference on the next launch. 3.
**`ORCA_TELEMETRY_DISABLED=1`** — Orca-specific kill switch with the same semantics as
`DO_NOT_TRACK`.
## Where the data goes
+1 -1
View File
@@ -81,4 +81,4 @@ Quick Commands save terminal commands you run often, such as `npm run dev`, `pnp
Each command has a label, command text, and scope. Use **Global** for commands that apply everywhere, or **Project** to show the command only in worktrees for a specific repo. The tab-bar button opens a fresh terminal tab and runs the command; the terminal context menu can insert commands into the current terminal. Use the copy control on a command row (Settings list, tab-bar menu, or [mobile Quick Commands](/docs/mobile#quick-commands)) to put the command body on the clipboard.
When you work with a paired [Remote Orca Server](/docs/remote-servers) (or another execution host), the picker can show **local and remote** collections side by side, labeled by host (for example *Local Mac* and *Orca Server*). **Saved on** is where the command is stored; running a command still executes in the terminal or workspace where you invoke it — so a client-owned command can run inside a remote worktree. Older servers that do not advertise multi-host Quick Commands fall back to the local-only list.
When you work with a paired [Remote Orca Server](/docs/remote-servers) (or another execution host), the picker can show **local and remote** collections side by side, labeled by host (for example _Local Mac_ and _Orca Server_). **Saved on** is where the command is stored; running a command still executes in the terminal or workspace where you invoke it — so a client-owned command can run inside a remote worktree. Older servers that do not advertise multi-host Quick Commands fall back to the local-only list.
+16 -14
View File
@@ -3,7 +3,7 @@ title: Ways to run Orca
description: Local desktop, SSH hosts, self-hosted Orca servers, and on-demand per-workspace VMs — pick the right compute for each task.
---
import { Callout, ImagePlaceholder } from '@/components/docs/prose';
import { Callout, ImagePlaceholder } from '@/components/docs/prose'
Orca is not locked to your laptop. Every worktree runs somewhere — on the machine in front of you, on a box you already own, on a shared always-on server, or on a fresh cloud VM spun up for that one workspace.
@@ -11,12 +11,12 @@ This page is the map. Deep dives live on the linked pages.
## At a glance
| Mode | Where files and agents live | Who owns the machine | Best for |
| --- | --- | --- | --- |
| **Local** | Your desktop | You | Day-to-day coding, fast iteration |
| **SSH target** | A remote host you connect to over SSH | You (or your team) | Dev boxes, GPU hosts, always-on VPS |
| **Remote Orca Server** | A machine running Orca desktop or `orca serve` | You (or your team) | Persistent shared runtime, mobile, automation |
| **Cloud VM / per-workspace environment** | A disposable VM/sandbox per workspace | Your cloud account (BYO provider) | Isolated, ephemeral agent compute |
| Mode | Where files and agents live | Who owns the machine | Best for |
| ---------------------------------------- | ---------------------------------------------- | --------------------------------- | --------------------------------------------- |
| **Local** | Your desktop | You | Day-to-day coding, fast iteration |
| **SSH target** | A remote host you connect to over SSH | You (or your team) | Dev boxes, GPU hosts, always-on VPS |
| **Remote Orca Server** | A machine running Orca desktop or `orca serve` | You (or your team) | Persistent shared runtime, mobile, automation |
| **Cloud VM / per-workspace environment** | A disposable VM/sandbox per workspace | Your cloud account (BYO provider) | Isolated, ephemeral agent compute |
Orca does **not** sell managed VPS hosting. Remote modes always use machines and cloud accounts you control.
@@ -62,12 +62,12 @@ Full detail: [Remote Orca Servers](/docs/remote-servers).
### SSH vs Remote Orca Server
| | SSH worktrees | Remote Orca Server |
| --- | --- | --- |
| Runtime owner | Laptop Orca | Remote machine (Orca desktop or `orca serve`) |
| Disconnect | Agents keep running on the host; laptop reattaches | Full session state lives on the server |
| Multi-client | One laptop drives the host | Laptop, web, mobile, and automation can share the same runtime |
| Typical setup | Import SSH config, pick **Run on** | Share the server app or run `orca serve`, then pair with a URL |
| | SSH worktrees | Remote Orca Server |
| ------------- | -------------------------------------------------- | -------------------------------------------------------------- |
| Runtime owner | Laptop Orca | Remote machine (Orca desktop or `orca serve`) |
| Disconnect | Agents keep running on the host; laptop reattaches | Full session state lives on the server |
| Multi-client | One laptop drives the host | Laptop, web, mobile, and automation can share the same runtime |
| Typical setup | Import SSH config, pick **Run on** | Share the server app or run `orca serve`, then pair with a URL |
## 4. Cloud VMs (per-workspace environments)
@@ -100,7 +100,9 @@ Providers people wire today include Vercel Sandbox, Fly, Modal, plain SSH hosts,
Recipes only show up for workspace create once the `environmentRecipes` entry is on the project's **primary** checkout of `orca.yaml` (not only a feature branch). Doctor and live provision can still run from any branch while you iterate on scripts.
<Callout title="BYO cloud — not an Orca VPS">
Cloud VMs do not give you an Orca-hosted VPS. You bring the provider (and pay that provider). Orca runs your create/suspend/resume/destroy scripts and connects over the pairing URL or SSH details they print.
Cloud VMs do not give you an Orca-hosted VPS. You bring the provider (and pay that provider). Orca
runs your create/suspend/resume/destroy scripts and connects over the pairing URL or SSH details
they print.
</Callout>
## How to choose
@@ -28,7 +28,7 @@ function fakeClient(script: (method: string, call: number) => unknown, calls: Ca
}
describe('createBlankWorkspace', () => {
it('sends no agent-launch fields for a blank workspace', async () => {
it('pins a manually entered blank-workspace name and sends no agent-launch fields', async () => {
const calls: Call[] = []
const client = fakeClient(() => ({ worktree: { id: 'wt-1' } }), calls)
@@ -51,6 +51,8 @@ describe('createBlankWorkspace', () => {
repo: 'id:repo-1',
setupDecision: 'inherit',
name: 'octopus',
displayName: 'octopus',
displayNameKind: 'user',
// Idempotency key so a create interrupted by a connection migration can be
// safely retried without the host spawning a duplicate worktree.
clientMutationId: expect.any(String)
@@ -32,6 +32,9 @@ export async function createBlankWorkspace(args: {
repo: `id:${args.repoId}`,
setupDecision: args.setupDecision,
name,
...(args.nameWasGenerated
? { displayNameKind: 'generated' as const }
: { displayName: args.baseName, displayNameKind: 'user' as const }),
...(args.nameWasGenerated ? { nameWasGenerated: true } : {}),
...agentLaunchCreateFields(args.createdWithAgentId)
}
@@ -174,7 +174,62 @@ describe('createWorkspaceFromComposerSource', () => {
})
})
it('suppresses displayName when the name is user-edited (not auto-managed)', async () => {
it('does not pin an automatically managed branch selection without a custom label', async () => {
const calls: Call[] = []
const client = fakeClient(() => ({ worktree: { id: 'wt-auto-branch' } }), calls)
const selection: MobileComposerCreateSelection = {
kind: 'branch',
baseBranch: 'main',
refName: 'main',
localBranchName: 'topic',
reuse: false,
branchNameOverride: 'topic'
}
await createWorkspaceFromComposerSource({ client, selection, ...baseArgs })
expect(calls[0]!.params).not.toHaveProperty('displayName')
expect(calls[0]!.params).not.toHaveProperty('displayNameKind')
})
it('does not pin an auto-derived branch label even when the draft is populated', async () => {
const calls: Call[] = []
const client = fakeClient(() => ({ worktree: { id: 'wt-auto-branch-draft' } }), calls)
const selection: MobileComposerCreateSelection = {
kind: 'new-branch',
branchName: 'topic'
}
await createWorkspaceFromComposerSource({
client,
selection,
...baseArgs,
workspaceName: 'topic',
nameIsAutoManaged: true
})
expect(calls[0]!.params).not.toHaveProperty('displayName')
expect(calls[0]!.params).not.toHaveProperty('displayNameKind')
})
it('pins a custom label for a new branch selection', async () => {
const calls: Call[] = []
const client = fakeClient(() => ({ worktree: { id: 'wt-labeled-branch' } }), calls)
const selection: MobileComposerCreateSelection = {
kind: 'new-branch',
branchName: 'feature/login'
}
await createWorkspaceFromComposerSource({
client,
selection,
...baseArgs,
workspaceName: ' Login work '
})
expect(calls[0]!.params).toMatchObject({
displayName: 'Login work',
displayNameKind: 'user'
})
})
it('pins displayName when the name is user-edited (not auto-managed)', async () => {
const calls: Call[] = []
const client = fakeClient(() => ({ worktree: { id: 'wt-dn' } }), calls)
const selection: MobileComposerCreateSelection = {
@@ -195,8 +250,12 @@ describe('createWorkspaceFromComposerSource', () => {
workspaceName: 'my-name',
nameIsAutoManaged: false
})
expect(calls[0]!.params.displayName).toBeUndefined()
expect(calls[0]!.params).toMatchObject({ name: 'my-name', linkedIssue: 7 })
expect(calls[0]!.params).toMatchObject({
name: 'my-name',
displayName: 'my-name',
displayNameKind: 'user',
linkedIssue: 7
})
})
it('creates a new branch off a ref, bumping the branch on collision', async () => {
+36 -3
View File
@@ -31,6 +31,7 @@ export type CreateWorkspaceFromComposerArgs = {
note: string | undefined
sparseCheckout?: { directories: string[]; presetId?: string }
nameIsAutoManaged?: boolean
note: string | undefined
worktreeCreateIdempotency: WorktreeCreateIdempotencyProbe
}
@@ -95,6 +96,7 @@ async function createWorkItemWorkspace(args: {
note: string | undefined
sparseCheckout?: { directories: string[]; presetId?: string }
nameIsAutoManaged?: boolean
note: string | undefined
worktreeCreateIdempotency: WorktreeCreateIdempotencyProbe
}): Promise<WorktreeCreateResult> {
const { client, selection, targetRepoId, setupDecision, agent, workspaceName, note } = args
@@ -153,12 +155,23 @@ async function createBranchWorkspace(args: {
setupDecision: WorkspaceCreateSetupDecision
agent: WorkspaceCreateAgentBundle
workspaceName: string | undefined
nameIsAutoManaged?: boolean
note: string | undefined
worktreeCreateIdempotency: WorktreeCreateIdempotencyProbe
}): Promise<WorktreeCreateResult> {
const { client, selection, targetRepoId, setupDecision, agent, workspaceName, note } = args
const {
client,
selection,
targetRepoId,
setupDecision,
agent,
workspaceName,
nameIsAutoManaged,
note
} = args
const createdWithAgentId = agent.choice === 'blank' ? undefined : agent.choice
const comment = note?.trim()
const manualDisplayName = nameIsAutoManaged === true ? undefined : workspaceName?.trim()
const applyCommon = (params: Record<string, unknown>): Record<string, unknown> => {
Object.assign(params, agentLaunchCreateFields(createdWithAgentId))
if (comment) {
@@ -184,6 +197,9 @@ async function createBranchWorkspace(args: {
applyCommon({
repo: `id:${targetRepoId}`,
name,
...(manualDisplayName
? { displayName: manualDisplayName, displayNameKind: 'user' as const }
: {}),
setupDecision,
baseBranch: selection.refName,
branchNameOverride: selection.localBranchName
@@ -206,7 +222,10 @@ async function createBranchWorkspace(args: {
repo: `id:${targetRepoId}`,
name: candidate,
setupDecision,
baseBranch: selection.baseBranch
baseBranch: selection.baseBranch,
...(manualDisplayName
? { displayName: manualDisplayName, displayNameKind: 'user' as const }
: {})
}
if (selection.branchNameOverride) {
params.branchNameOverride = candidate
@@ -223,11 +242,22 @@ async function createNewBranchWorkspace(args: {
setupDecision: WorkspaceCreateSetupDecision
agent: WorkspaceCreateAgentBundle
workspaceName: string | undefined
nameIsAutoManaged?: boolean
note: string | undefined
worktreeCreateIdempotency: WorktreeCreateIdempotencyProbe
}): Promise<WorktreeCreateResult> {
const { client, selection, targetRepoId, setupDecision, agent, note } = args
const {
client,
selection,
targetRepoId,
setupDecision,
agent,
workspaceName,
nameIsAutoManaged,
note
} = args
const createdWithAgentId = agent.choice === 'blank' ? undefined : agent.choice
const manualDisplayName = nameIsAutoManaged === true ? undefined : workspaceName?.trim()
const comment = note?.trim()
// A brand-new branch off the repo's default base. The typed name is kept as the
// git branch (via branchNameOverride) so a slash like `feature/login` survives;
@@ -243,6 +273,9 @@ async function createNewBranchWorkspace(args: {
name: candidate,
setupDecision,
branchNameOverride: candidate,
...(manualDisplayName
? { displayName: manualDisplayName, displayNameKind: 'user' as const }
: {}),
...agentLaunchCreateFields(createdWithAgentId)
}
if (comment) {
@@ -39,7 +39,8 @@ export function useMobileTasksWorkspaceCreateActions(model: WorkspaceSshStateMod
taskStateHydrated,
tasksSupported,
trustedOrcaHooks,
workspaceDetectedAgentIds
workspaceDetectedAgentIds,
workspaceLastAutoName
} = model
const createWorkspace = useCallback(
async (
@@ -149,6 +150,9 @@ export function useMobileTasksWorkspaceCreateActions(model: WorkspaceSshStateMod
})
return
}
const trimmedWorkspaceName = workspaceNameOverride?.trim() ?? ''
const nameIsAutoManaged =
!trimmedWorkspaceName || trimmedWorkspaceName === workspaceLastAutoName
let params: Record<string, unknown>
if (item.provider === 'github') {
const source = item.source
@@ -192,7 +196,8 @@ export function useMobileTasksWorkspaceCreateActions(model: WorkspaceSshStateMod
baseBranch: baseBranchOverride,
branchNameOverride,
sparseCheckout: sparseCheckoutOverride,
hostedStartPoint: prStartPoint
hostedStartPoint: prStartPoint,
nameIsAutoManaged
})
} else if (item.provider === 'gitlab') {
const source = item.source
@@ -236,7 +241,8 @@ export function useMobileTasksWorkspaceCreateActions(model: WorkspaceSshStateMod
baseBranch: baseBranchOverride,
branchNameOverride,
sparseCheckout: sparseCheckoutOverride,
hostedStartPoint: mrStartPoint
hostedStartPoint: mrStartPoint,
nameIsAutoManaged
})
} else {
params = buildTaskWorkspaceCreateParams({
@@ -248,7 +254,8 @@ export function useMobileTasksWorkspaceCreateActions(model: WorkspaceSshStateMod
note: comment,
baseBranch: baseBranchOverride,
branchNameOverride,
sparseCheckout: sparseCheckoutOverride
sparseCheckout: sparseCheckoutOverride,
nameIsAutoManaged
})
}
const response = await client.sendRequest('worktree.create', params, {
@@ -289,7 +296,8 @@ export function useMobileTasksWorkspaceCreateActions(model: WorkspaceSshStateMod
taskStateHydrated,
tasksSupported,
trustedOrcaHooks,
workspaceDetectedAgentIds
workspaceDetectedAgentIds,
workspaceLastAutoName
]
)
return Object.assign(model, { createWorkspace })
@@ -41,6 +41,7 @@ describe('task workspace create params', () => {
repo: 'id:repo-1',
name: 'mobile-tasks',
displayName: 'Fix mobile tasks',
displayNameKind: 'generated',
setupDecision: 'run',
activate: true,
startupDraft: 'https://github.com/acme/app/pull/123',
@@ -72,6 +73,7 @@ describe('task workspace create params', () => {
repo: 'id:repo-1',
name: 'issue-88',
displayName: 'Investigate login',
displayNameKind: 'generated',
setupDecision: 'skip',
activate: true,
linkedIssue: 88
@@ -80,6 +82,31 @@ describe('task workspace create params', () => {
expect(params).not.toHaveProperty('createdWithAgent')
})
it('marks an edited task label as user-owned', () => {
const params = buildTaskWorkspaceCreateParams({
item: {
provider: 'github',
source: {
type: 'issue',
repoId: 'repo-1',
number: 88,
title: 'Investigate login',
url: 'https://github.com/acme/app/issues/88'
}
},
targetRepoId: 'ignored-for-github',
setupDecision: 'skip',
workspaceName: 'My workspace',
nameIsAutoManaged: false
})
expect(params).toMatchObject({
name: 'My workspace',
displayName: 'My workspace',
displayNameKind: 'user'
})
})
it('keeps the startup draft when no agent was provided so the host can auto-pick', () => {
const params = buildTaskWorkspaceCreateParams({
item: {
+6 -3
View File
@@ -108,8 +108,7 @@ export function buildTaskWorkspaceCreateParams(args: {
const comment = note?.trim()
const selectedBaseBranch = baseBranch || hostedStartPoint?.baseBranch
const selectedPushTarget = pushTarget ?? hostedStartPoint?.pushTarget
// Why: desktop only sends displayName while the name is still auto-derived; a
// user-edited name suppresses it so the runtime keeps the user's chosen name.
// Preserve provenance so the host can distinguish an intentional label from a generated title.
const sourceName =
item.provider === 'linear'
? getWorkspaceSourceName({
@@ -121,7 +120,11 @@ export function buildTaskWorkspaceCreateParams(args: {
linearIdentifier: item.source.identifier
})
: getWorkspaceSourceName({ provider: item.provider, ...item.source })
const displayName = nameIsAutoManaged ? { displayName: sourceName.displayName } : {}
const displayName = nameIsAutoManaged
? { displayName: sourceName.displayName, displayNameKind: 'generated' as const }
: workspaceName?.trim()
? { displayName: workspaceName, displayNameKind: 'user' as const }
: {}
const common = {
setupDecision,
activate: true,
+41 -2
View File
@@ -55,7 +55,7 @@ async function flush(): Promise<void> {
// cutover). Records every call so tests can assert on the clientMutationId.
function scriptedClient(
outcomes: Array<
| { id: string }
| { id: string; displayName?: string }
| { errorMessage: string }
// takesMs models how long the ambiguity took to SURFACE — a clean close is
// instant, a half-open socket waits out the liveness watchdog or the timeout.
@@ -103,7 +103,12 @@ function scriptedClient(
return {
id: '1',
ok: true,
result: { worktree: { id: outcome.id } },
result: {
worktree: {
id: outcome.id,
...(outcome.displayName !== undefined ? { displayName: outcome.displayName } : {})
}
},
_meta: { runtimeId: 'r' }
}
}
@@ -227,6 +232,40 @@ describe('createWorktreeWithNameRetry', () => {
expect(attempts[1]!.params.name).toBe('topic-2')
})
it('uses the host-selected display name after a collision retry', async () => {
const attempts: Attempt[] = []
const client = scriptedClient(
[{ errorMessage: 'already exists locally' }, { id: 'wt-host-name', displayName: 'topic-3' }],
attempts
)
await expect(
createWorktreeWithNameRetry({
client,
baseName: 'topic',
buildParams: (name) => ({ repo: 'id:r', name }),
worktreeCreateIdempotency: false
})
).resolves.toEqual({ worktreeId: 'wt-host-name', name: 'topic-3' })
})
it('falls back to the client candidate when an older host omits displayName', async () => {
const attempts: Attempt[] = []
const client = scriptedClient(
[{ errorMessage: 'already exists locally' }, { id: 'wt-legacy' }],
attempts
)
await expect(
createWorktreeWithNameRetry({
client,
baseName: 'topic',
buildParams: (name) => ({ repo: 'id:r', name }),
worktreeCreateIdempotency: false
})
).resolves.toEqual({ worktreeId: 'wt-legacy', name: 'topic-2' })
})
it('advances generated retries without nesting suffixes', async () => {
const attempts: Attempt[] = []
const client = scriptedClient(
+6 -2
View File
@@ -85,12 +85,16 @@ export async function createWorktreeWithNameRetry(
const response = await sendWorktreeCreateResilient(client, params, worktreeCreateIdempotency)
if (response.ok) {
const result = (response as RpcSuccess).result as {
worktree: { id: string }
worktree: { id: string; displayName?: string }
warning?: string
}
const authoritativeName = result.worktree.displayName
return {
worktreeId: result.worktree.id,
name: candidateName,
name:
typeof authoritativeName === 'string' && authoritativeName.trim()
? authoritativeName
: candidateName,
...(result.warning ? { warning: result.warning } : {})
}
}
+20 -20
View File
@@ -4,18 +4,18 @@
{
"name": "computer-use",
"sourcePath": "skills/computer-use",
"releaseRevision": 8,
"packageDigest": "d1b4850c9a9ee9a32b855176c31cd357608bfedc845319c97e89960296303430",
"gitTreeSha": "2072384f53670cb61d93f4f6264ad2d8f6b5239c",
"releaseRevision": 9,
"packageDigest": "ddc9f910985ae67ab693263026d99c68dc34b6f0c12b444b620cd0e19c3df36a",
"gitTreeSha": "f0561c41d1f709a953684aef5d5368f8c58d269f",
"files": [
{
"path": "SKILL.md",
"size": 3667,
"size": 3465,
"executable": false,
"classification": "text",
"exactSha256": "c4a11596b7c0338f4c991b24ba7ba453d93fb8dc045c642c517e7ae6d3c88467",
"textNormalizedSha256": "c4a11596b7c0338f4c991b24ba7ba453d93fb8dc045c642c517e7ae6d3c88467",
"identitySha256": "c4a11596b7c0338f4c991b24ba7ba453d93fb8dc045c642c517e7ae6d3c88467"
"exactSha256": "a3fcca06875e354a470e7daef5619172a3615fbbf82900ff3e0a65636fc694c6",
"textNormalizedSha256": "a3fcca06875e354a470e7daef5619172a3615fbbf82900ff3e0a65636fc694c6",
"identitySha256": "a3fcca06875e354a470e7daef5619172a3615fbbf82900ff3e0a65636fc694c6"
}
]
},
@@ -41,17 +41,17 @@
"name": "orca-cli",
"sourcePath": "skills/orca-cli",
"releaseRevision": 37,
"packageDigest": "d5648df9c29b479bbbe0dea68f0506b2b0bad0827ce551451fb8f300c113b305",
"gitTreeSha": "ee3a35c76f875ea6ebaf68947af13ba6346b13f2",
"packageDigest": "d1b830256e3fda11408320631722e07bb4bbadf99d19c94f1e00f73b7bc8462d",
"gitTreeSha": "cdf89459f89dddf347ee2759ff884c369051f06a",
"files": [
{
"path": "SKILL.md",
"size": 3944,
"size": 4150,
"executable": false,
"classification": "text",
"exactSha256": "cdd5d9c8a95837a6cc24d68b150a117684afe9bbb2d763de2cc7fba1da1ed163",
"textNormalizedSha256": "cdd5d9c8a95837a6cc24d68b150a117684afe9bbb2d763de2cc7fba1da1ed163",
"identitySha256": "cdd5d9c8a95837a6cc24d68b150a117684afe9bbb2d763de2cc7fba1da1ed163"
"exactSha256": "6dcbd69045c74787be3385198750c67223709c5fa63a2ba3cfbbe0403e1219f5",
"textNormalizedSha256": "6dcbd69045c74787be3385198750c67223709c5fa63a2ba3cfbbe0403e1219f5",
"identitySha256": "6dcbd69045c74787be3385198750c67223709c5fa63a2ba3cfbbe0403e1219f5"
}
]
},
@@ -130,18 +130,18 @@
{
"name": "orchestration",
"sourcePath": "skills/orchestration",
"releaseRevision": 28,
"packageDigest": "ef5d5a744cdc700c51b4870cd2536b65b0b33d19413dfe238d43efdd01b5d14c",
"gitTreeSha": "9aa26fde93c0592e5983cdca1ccd33b402802255",
"releaseRevision": 29,
"packageDigest": "7a386ce558ba54abe02b4a0de5d71fe3d63c944ef0888ddde130c729b37f7cc8",
"gitTreeSha": "4199ec6988801dd491631706cba62631b4bed8fb",
"files": [
{
"path": "SKILL.md",
"size": 4220,
"size": 4451,
"executable": false,
"classification": "text",
"exactSha256": "9ca228137b9a442b98c761aa07adecc2265708132ab175ad7e22b163fdc0bd7f",
"textNormalizedSha256": "9ca228137b9a442b98c761aa07adecc2265708132ab175ad7e22b163fdc0bd7f",
"identitySha256": "9ca228137b9a442b98c761aa07adecc2265708132ab175ad7e22b163fdc0bd7f"
"exactSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f",
"textNormalizedSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f",
"identitySha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f"
}
]
}
+38 -6
View File
@@ -580,17 +580,17 @@
},
{
"releaseRevision": 37,
"packageDigest": "d5648df9c29b479bbbe0dea68f0506b2b0bad0827ce551451fb8f300c113b305",
"gitTreeSha": "ee3a35c76f875ea6ebaf68947af13ba6346b13f2",
"packageDigest": "d1b830256e3fda11408320631722e07bb4bbadf99d19c94f1e00f73b7bc8462d",
"gitTreeSha": "cdf89459f89dddf347ee2759ff884c369051f06a",
"files": [
{
"path": "SKILL.md",
"size": 3944,
"size": 4150,
"executable": false,
"classification": "text",
"exactSha256": "cdd5d9c8a95837a6cc24d68b150a117684afe9bbb2d763de2cc7fba1da1ed163",
"textNormalizedSha256": "cdd5d9c8a95837a6cc24d68b150a117684afe9bbb2d763de2cc7fba1da1ed163",
"identitySha256": "cdd5d9c8a95837a6cc24d68b150a117684afe9bbb2d763de2cc7fba1da1ed163"
"exactSha256": "6dcbd69045c74787be3385198750c67223709c5fa63a2ba3cfbbe0403e1219f5",
"textNormalizedSha256": "6dcbd69045c74787be3385198750c67223709c5fa63a2ba3cfbbe0403e1219f5",
"identitySha256": "6dcbd69045c74787be3385198750c67223709c5fa63a2ba3cfbbe0403e1219f5"
}
]
}
@@ -1043,6 +1043,22 @@
"identitySha256": "9ca228137b9a442b98c761aa07adecc2265708132ab175ad7e22b163fdc0bd7f"
}
]
},
{
"releaseRevision": 29,
"packageDigest": "7a386ce558ba54abe02b4a0de5d71fe3d63c944ef0888ddde130c729b37f7cc8",
"gitTreeSha": "4199ec6988801dd491631706cba62631b4bed8fb",
"files": [
{
"path": "SKILL.md",
"size": 4451,
"executable": false,
"classification": "text",
"exactSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f",
"textNormalizedSha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f",
"identitySha256": "a7e3350f037698ebbce36b2818d8383ec6e96c2a4825537caab39a83b4fb4b7f"
}
]
}
],
"mobile-fit-debug": [
@@ -1191,6 +1207,22 @@
"identitySha256": "c4a11596b7c0338f4c991b24ba7ba453d93fb8dc045c642c517e7ae6d3c88467"
}
]
},
{
"releaseRevision": 9,
"packageDigest": "ddc9f910985ae67ab693263026d99c68dc34b6f0c12b444b620cd0e19c3df36a",
"gitTreeSha": "f0561c41d1f709a953684aef5d5368f8c58d269f",
"files": [
{
"path": "SKILL.md",
"size": 3465,
"executable": false,
"classification": "text",
"exactSha256": "a3fcca06875e354a470e7daef5619172a3615fbbf82900ff3e0a65636fc694c6",
"textNormalizedSha256": "a3fcca06875e354a470e7daef5619172a3615fbbf82900ff3e0a65636fc694c6",
"identitySha256": "a3fcca06875e354a470e7daef5619172a3615fbbf82900ff3e0a65636fc694c6"
}
]
}
],
"orca-emulator": [
+8 -10
View File
@@ -1,19 +1,17 @@
---
name: computer-use
description: >-
Use Orca's computer-use CLI to inspect and operate local desktop app windows
through accessibility trees, screenshots, and safe UI actions. Use for
desktop app interaction: list apps/windows, get app state, read visible UI,
click controls, type, press keys, scroll, drag, set values, or perform
accessibility actions. Also use for browser windows, webviews, Orca app UI,
or other desktop UI. Triggers include "computer use", "orca computer", "read
Spotify", "read Slack", "control/click/read in a desktop app", and "get app
state".
Use Orca's computer-use CLI for OS/window-level inspection and input in visible
local app windows. Use when a task must read or operate a native app or an
external browser window (for example, Chrome, Edge, or Safari) or an app
webview. Do not use for Orca's embedded browser or page-only browser
automation. Use `orca-cli` for Orca's embedded pages and a page-automation
tool such as Playwright or CDP for external pages.
---
# Computer Use
Use this skill for desktop UI through `orca computer`. When the requested target is a website or web app, operate the desktop browser app/window that contains the page.
Use this skill for desktop UI through `orca computer`. For a website or web app, use it only when the page is in an external desktop browser window that needs desktop-level control. Do not use it for page-only automation: use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages.
## Preconditions
@@ -164,4 +162,4 @@ Slack: the accessibility tree may be shallow while the screenshot contains usefu
## Next Action
Confirm Orca status unless already checked, then run `ORCA computer capabilities --json`. For website or web-app targets such as Gmail, identify the desktop browser app/window that contains the page, then get that target app state with `ORCA computer get-app-state --app <app> --json`.
Confirm Orca status unless already checked, then run `ORCA computer capabilities --json`. For external browser targets such as Gmail, identify the desktop browser app/window that contains the page, then get that target app state with `ORCA computer get-app-state --app <app> --json`.
+5 -3
View File
@@ -10,8 +10,10 @@ description: >-
"share HTML/Markdown", "public artifact link", "share skills", or "control the browser inside
Orca". Prefer this over raw `git worktree`, ad hoc
PTYs, Playwright, or Computer Use when the task touches Orca-managed state.
Use Computer Use for browser windows, webviews, or desktop UI outside Orca's
embedded browser.
Use Computer Use for external browser windows, webviews, or desktop UI only
when the task requires OS/window-level control such as focus, menus, dialogs,
coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a
page-automation tool such as Playwright or CDP for external pages.
---
# Orca CLI
@@ -309,7 +311,7 @@ ORCA skills share --skill <selector> [--skill <selector> ...] --bundle-name <nam
The built-in browser is Orca's embedded browser tab surface, scoped to Orca worktrees; it is not Chrome/Safari or desktop app UI.
These commands control only Orca's embedded browser tabs. For external Chrome/Safari/webviews or Orca app chrome/settings, use the Computer Use skill/tool. If the user explicitly asks for Orca CLI desktop control, use `orca computer ...`; do not use browser commands for desktop UI.
These commands control only Orca's embedded browser tabs. For external Chrome/Safari/webviews or Orca app chrome/settings, use the Computer Use skill/tool only when the task requires OS/window-level control. Use `orca-cli` for Orca's embedded pages and a page-automation tool such as Playwright or CDP for external pages. If the user explicitly asks for Orca CLI desktop control, use `orca computer ...`; do not use browser commands for desktop UI.
Use a snapshot-interact-re-snapshot loop:
+6 -3
View File
@@ -8,10 +8,13 @@ description: >-
requests phrased as "hand off", "handoff", "handover", "give this to another
agent", or "another worktree" when the user did not explicitly ask to
supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for
ordinary terminal control, lightweight terminal prompts, shell commands, Orca
terminal control, lightweight terminal prompts, shell commands, Orca
worktree management, reading or waiting on terminals, and automation of the
browser embedded inside Orca. Use Computer Use for browser windows, webviews,
Orca app UI, or desktop UI outside Orca's embedded browser.
browser embedded inside Orca. Use Computer Use for external browser windows,
webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when
the task requires OS/window-level control such as focus, menus, dialogs,
coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a
page-automation tool such as Playwright or CDP for external pages.
---
# Orca Inter-Agent Orchestration
+4 -5
View File
@@ -4,11 +4,10 @@ This file is a discovery stub, not the usage guide. The full, version-matched co
reference is served by the `orca` binary itself — kept out of this file on purpose so it can
never drift from the binary that will actually run your commands.
Engage Orca's computer-use surface whenever you must inspect or operate a local desktop app
window — reading its accessibility tree, taking screenshots, or performing safe UI actions
(click controls, type, press keys, scroll, drag, set values). It also covers browser
windows, webviews, and Orca's own UI. Triggers include "computer use", "orca computer",
"read Spotify", "read Slack", "control/click/read in a desktop app", and "get app state".
Engage Orca's computer-use surface when a task requires desktop-level access to a visible local
app or window, including a native app or an external browser window/webview. Do not use for
Orca's embedded browser or page-only browser automation. Use `orca-cli` for Orca's embedded
pages and a page-automation tool such as Playwright or CDP for external pages.
## Resolve the CLI for this session
+10 -13
View File
@@ -1,14 +1,12 @@
---
name: computer-use
description: >-
Use Orca's computer-use CLI to inspect and operate local desktop app windows
through accessibility trees, screenshots, and safe UI actions. Use for
desktop app interaction: list apps/windows, get app state, read visible UI,
click controls, type, press keys, scroll, drag, set values, or perform
accessibility actions. Also use for browser windows, webviews, Orca app UI,
or other desktop UI. Triggers include "computer use", "orca computer", "read
Spotify", "read Slack", "control/click/read in a desktop app", and "get app
state".
Use Orca's computer-use CLI for OS/window-level inspection and input in visible
local app windows. Use when a task must read or operate a native app or an
external browser window (for example, Chrome, Edge, or Safari) or an app
webview. Do not use for Orca's embedded browser or page-only browser
automation. Use `orca-cli` for Orca's embedded pages and a page-automation
tool such as Playwright or CDP for external pages.
---
# Computer Use
@@ -17,11 +15,10 @@ This file is a discovery stub, not the usage guide. The full, version-matched co
reference is served by the `orca` binary itself — kept out of this file on purpose so it can
never drift from the binary that will actually run your commands.
Engage Orca's computer-use surface whenever you must inspect or operate a local desktop app
window — reading its accessibility tree, taking screenshots, or performing safe UI actions
(click controls, type, press keys, scroll, drag, set values). It also covers browser
windows, webviews, and Orca's own UI. Triggers include "computer use", "orca computer",
"read Spotify", "read Slack", "control/click/read in a desktop app", and "get app state".
Engage Orca's computer-use surface when a task requires desktop-level access to a visible local
app or window, including a native app or an external browser window/webview. Do not use for
Orca's embedded browser or page-only browser automation. Use `orca-cli` for Orca's embedded
pages and a page-automation tool such as Playwright or CDP for external pages.
## Resolve the CLI for this session
+4 -2
View File
@@ -10,8 +10,10 @@ description: >-
"share HTML/Markdown", "public artifact link", "share skills", or "control the browser inside
Orca". Prefer this over raw `git worktree`, ad hoc
PTYs, Playwright, or Computer Use when the task touches Orca-managed state.
Use Computer Use for browser windows, webviews, or desktop UI outside Orca's
embedded browser.
Use Computer Use for external browser windows, webviews, or desktop UI only
when the task requires OS/window-level control such as focus, menus, dialogs,
coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a
page-automation tool such as Playwright or CDP for external pages.
---
# Orca CLI
+6 -3
View File
@@ -8,10 +8,13 @@ description: >-
requests phrased as "hand off", "handoff", "handover", "give this to another
agent", or "another worktree" when the user did not explicitly ask to
supervise, monitor, wait for results, or coordinate a DAG. Use `orca-cli` for
ordinary terminal control, lightweight terminal prompts, shell commands, Orca
terminal control, lightweight terminal prompts, shell commands, Orca
worktree management, reading or waiting on terminals, and automation of the
browser embedded inside Orca. Use Computer Use for browser windows, webviews,
Orca app UI, or desktop UI outside Orca's embedded browser.
browser embedded inside Orca. Use Computer Use for external browser windows,
webviews, Orca app UI, or desktop UI outside Orca's embedded browser only when
the task requires OS/window-level control such as focus, menus, dialogs,
coordinates, or screenshots. Use `orca-cli` for Orca's embedded pages and a
page-automation tool such as Playwright or CDP for external pages.
---
# Orca Orchestration
File diff suppressed because one or more lines are too long
+4 -1
View File
@@ -230,9 +230,12 @@ export const WORKTREE_HANDLERS: Record<string, CommandHandler> = {
}
const linearIssueLink = getOptionalLinearIssueLinkFlag(flags, 'linear-issue')
const activate = flags.get('activate') === true || flags.get('run-hooks') === true
const name = getRequiredStringFlag(flags, 'name')
const result = await client.call<RuntimeWorktreeCreateResult>('worktree.create', {
repo: await getCreateRepoSelector(flags, cwdParentWorktree, client),
name: getRequiredStringFlag(flags, 'name'),
name,
displayName: name,
displayNameKind: 'user',
baseBranch: getOptionalStringFlag(flags, 'base-branch'),
linkedIssue: getOptionalNumberFlag(flags, 'issue'),
...linearIssueLink,
@@ -72,6 +72,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'feature',
displayName: 'feature',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -122,6 +124,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'agent-task',
displayName: 'agent-task',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -169,6 +173,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'agent-task',
displayName: 'agent-task',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -87,6 +87,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'feature',
displayName: 'feature',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
linkedLinearIssue: 'STA-335',
@@ -137,6 +139,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenCalledWith('worktree.create', {
repo: 'id:repo-1',
name: 'feature',
displayName: 'feature',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
linkedLinearIssue: 'STA-335',
@@ -106,6 +106,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenCalledWith('worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -152,6 +154,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenCalledWith('worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -219,6 +223,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenCalledWith('worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -263,6 +269,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -308,6 +316,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -369,6 +379,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -517,6 +529,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenCalledWith('worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -559,6 +573,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenCalledWith('worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -72,6 +72,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'feature',
displayName: 'feature',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -147,6 +149,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-gpu',
name: 'feature',
displayName: 'feature',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -259,6 +263,8 @@ describe('orca cli worktree awareness', () => {
expect(callMock).toHaveBeenNthCalledWith(2, 'worktree.create', {
repo: 'id:repo-1',
name: 'child',
displayName: 'child',
displayNameKind: 'user',
baseBranch: undefined,
linkedIssue: undefined,
comment: undefined,
@@ -7,7 +7,8 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import { spawn } from 'node:child_process'
import { createServer, type Server } from 'node:http'
import { mkdtempSync, readFileSync, rmSync } from 'node:fs'
import { mkdtempSync, readFileSync } from 'node:fs'
import { removeTreeSync } from '../../shared/windows-transient-lock-removal'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import type * as osModule from 'node:os'
@@ -149,7 +150,7 @@ describe.skipIf(process.platform !== 'win32')('Windows managed hook payload deli
server = null
homedirMock.mockImplementation(() => process.env.HOME ?? tmpdir())
if (home) {
rmSync(home, { recursive: true, force: true })
removeTreeSync(home)
home = ''
}
})
@@ -3,7 +3,11 @@ import { mkdtemp, readFile, readdir, rm, stat, truncate, writeFile } from 'node:
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { ARTIFACT_CLI_MAX_RPC_BYTES, artifactWriteRequestByteLength } from '../../shared/artifacts'
import {
ARTIFACT_MAX_CONTENT_BYTES,
ARTIFACT_MAX_REQUEST_BYTES,
artifactWriteRequestByteLength
} from '../../shared/artifacts'
import {
MAX_ARTIFACT_CREATE_INTENT_BYTES,
MAX_PENDING_ARTIFACT_CREATES,
@@ -270,12 +274,12 @@ describe('artifact create intent store', () => {
).toThrow(/unsupported format/)
})
it('persists a valid artifact request near the RPC limit', async () => {
it('persists a 5 MiB escaped artifact within the recovery limit', async () => {
const userDataPath = await createUserDataPath()
const nearLimitBody = { ...body, content: 'x'.repeat(ARTIFACT_CLI_MAX_RPC_BYTES - 200) }
const nearLimitBody = { ...body, content: '"'.repeat(ARTIFACT_MAX_CONTENT_BYTES) }
expect(
artifactWriteRequestByteLength({ sourceKey: '/repo/report.html', ...nearLimitBody })
).toBeLessThanOrEqual(ARTIFACT_CLI_MAX_RPC_BYTES)
).toBeLessThanOrEqual(ARTIFACT_MAX_REQUEST_BYTES)
expect(() =>
getOrCreateArtifactCreateIntent(
@@ -289,7 +293,41 @@ describe('artifact create intent store', () => {
).not.toThrow()
const directory = join(userDataPath, 'profiles', 'local-profile', 'artifact-create-intents')
const [fileName] = await readdir(directory)
expect((await stat(join(directory, fileName))).size).toBeGreaterThan(ARTIFACT_CLI_MAX_RPC_BYTES)
expect((await stat(join(directory, fileName))).size).toBeGreaterThan(ARTIFACT_MAX_CONTENT_BYTES)
})
it('rejects oversized artifact content before creating a recovery record', async () => {
const userDataPath = await createUserDataPath()
expect(() =>
getOrCreateArtifactCreateIntent(
'local-profile',
userDataPath,
'/repo/report.html',
scope,
'key-a',
{ ...body, content: 'x'.repeat(ARTIFACT_MAX_CONTENT_BYTES + 1) }
)
).toThrow(/5 MiB limit/)
})
it('rejects a recovery body whose escaped request exceeds the transport budget', async () => {
const userDataPath = await createUserDataPath()
const content = '\u0000'.repeat(Math.ceil(ARTIFACT_MAX_REQUEST_BYTES / 6))
expect(content.length).toBeLessThan(ARTIFACT_MAX_CONTENT_BYTES)
expect(
artifactWriteRequestByteLength({ sourceKey: '/repo/report.html', ...body, content })
).toBeGreaterThan(ARTIFACT_MAX_REQUEST_BYTES)
expect(() =>
getOrCreateArtifactCreateIntent(
'local-profile',
userDataPath,
'/repo/report.html',
scope,
'key-a',
{ ...body, content }
)
).toThrow(/supported size/)
})
it('rejects an oversized recovery record before reading it', async () => {
@@ -10,7 +10,12 @@ import {
writeFileSync
} from 'node:fs'
import { join } from 'node:path'
import { ARTIFACT_CLI_MAX_RPC_BYTES } from '../../shared/artifacts'
import {
ARTIFACT_MAX_CONTENT_BYTES,
ARTIFACT_MAX_REQUEST_BYTES,
artifactContentByteLength,
artifactWriteRequestByteLength
} from '../../shared/artifacts'
import {
bestEffortFsyncDirectorySync,
fsyncFileSync,
@@ -21,7 +26,7 @@ import type { ArtifactWriteBody } from './artifact-cloud-request'
import type { ArtifactShareScope } from './artifact-share-record-store'
export const MAX_PENDING_ARTIFACT_CREATES = 32
export const MAX_ARTIFACT_CREATE_INTENT_BYTES = ARTIFACT_CLI_MAX_RPC_BYTES + 128 * 1024
export const MAX_ARTIFACT_CREATE_INTENT_BYTES = ARTIFACT_MAX_REQUEST_BYTES + 128 * 1024
const MAX_HARDENED_INTENT_DIRECTORIES = 64
const hardenedIntentDirectories = new Set<string>()
@@ -116,12 +121,17 @@ function isWriteBody(value: unknown): value is ArtifactWriteBody {
const body = value as Partial<ArtifactWriteBody>
return (
typeof body.content === 'string' &&
artifactContentByteLength(body.content) <= ARTIFACT_MAX_CONTENT_BYTES &&
(body.contentType === 'text/html' || body.contentType === 'text/markdown') &&
typeof body.fileName === 'string' &&
(body.title === undefined || typeof body.title === 'string')
)
}
function artifactIntentRequestByteLength(sourceKey: string, body: ArtifactWriteBody): number {
return artifactWriteRequestByteLength({ sourceKey, ...body })
}
function isScope(value: unknown): value is ArtifactShareScope {
if (!value || typeof value !== 'object') {
return false
@@ -165,6 +175,9 @@ function readIntent(path: string): ArtifactCreateIntent {
) {
throw new Error('Artifact create recovery record has an unsupported format.')
}
if (artifactIntentRequestByteLength(intent.sourceKey, intent.body) > ARTIFACT_MAX_REQUEST_BYTES) {
throw new Error('Artifact create recovery record exceeds the supported size.')
}
return intent as ArtifactCreateIntent
}
@@ -193,6 +206,12 @@ export function getOrCreateArtifactCreateIntent(
idempotencyKey: string,
body: ArtifactWriteBody
): ArtifactCreateIntent {
if (artifactContentByteLength(body.content) > ARTIFACT_MAX_CONTENT_BYTES) {
throw new Error('Artifact content exceeds the 5 MiB limit.')
}
if (artifactIntentRequestByteLength(sourceKey, body) > ARTIFACT_MAX_REQUEST_BYTES) {
throw new Error('Artifact create recovery record exceeds the supported size.')
}
const existing = getArtifactCreateIntent(profileId, userDataPath, sourceKey, scope)
if (existing) {
return existing
@@ -0,0 +1,92 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { setupGuestMouseWheelZoomForwarding } from './browser-guest-wheel-zoom'
describe('guest viewport wheel forwarding', () => {
const rendererSend = vi.fn()
const guestOn = vi.fn()
beforeEach(() => {
rendererSend.mockReset()
guestOn.mockReset()
})
function trigger(
active: boolean,
mouse: Partial<Electron.MouseWheelInputEvent>
): ReturnType<typeof vi.fn> {
setupGuestMouseWheelZoomForwarding({
browserTabId: 'tab-1',
guest: { on: guestOn } as unknown as Electron.WebContents,
resolveRenderer: () => ({ send: rendererSend }) as unknown as Electron.WebContents,
isViewportPresetActive: () => active,
canViewportScroll: () => active
})
const handler = guestOn.mock.calls.at(-1)![1] as (
event: Electron.Event,
input: Electron.MouseInputEvent
) => void
const preventDefault = vi.fn()
handler({ preventDefault } as unknown as Electron.Event, {
type: 'mouseWheel',
x: 0,
y: 0,
modifiers: [],
deltaX: 0,
deltaY: 0,
...mouse
})
return preventDefault
}
it('forwards plain wheel deltas only for an active preset', () => {
const preventDefault = trigger(true, { deltaX: 24, deltaY: 120 })
expect(preventDefault).toHaveBeenCalledTimes(1)
expect(rendererSend).toHaveBeenCalledWith('ui:scrollBrowserPage', {
browserPageId: 'tab-1',
deltaX: 24,
deltaY: 120
})
trigger(false, { deltaY: 120 })
expect(rendererSend).toHaveBeenCalledTimes(1)
})
it('ignores zero deltas and preserves ctrl-wheel zoom routing', () => {
const zeroPreventDefault = trigger(true, {})
expect(zeroPreventDefault).not.toHaveBeenCalled()
const zoomPreventDefault = trigger(true, { modifiers: ['ctrl'], deltaY: -120 })
expect(zoomPreventDefault).toHaveBeenCalledTimes(1)
expect(rendererSend).toHaveBeenLastCalledWith('ui:zoomBrowserPage', 'in')
})
it('leaves fitting presets and host-edge wheels to the guest page', () => {
setupGuestMouseWheelZoomForwarding({
browserTabId: 'tab-1',
guest: { on: guestOn } as unknown as Electron.WebContents,
resolveRenderer: () => ({ send: rendererSend }) as unknown as Electron.WebContents,
isViewportPresetActive: () => true,
canViewportScroll: () => false
})
const handler = guestOn.mock.calls.at(-1)![1] as (
event: Electron.Event,
input: Electron.MouseWheelInputEvent
) => void
const preventDefault = vi.fn()
handler(
{ preventDefault } as unknown as Electron.Event,
{
type: 'mouseWheel',
x: 0,
y: 0,
modifiers: [],
deltaX: 0,
deltaY: 120
} as Electron.MouseWheelInputEvent
)
expect(preventDefault).not.toHaveBeenCalled()
expect(rendererSend).not.toHaveBeenCalled()
})
})
+36 -5
View File
@@ -68,17 +68,48 @@ export function setupGuestMouseWheelZoomForwarding(args: {
browserTabId: string
guest: Electron.WebContents
resolveRenderer: ResolveRenderer
isViewportPresetActive?: () => boolean
canViewportScroll?: (mouse: Electron.MouseWheelInputEvent) => boolean
onViewportWheelConsumed?: (deltaX: number, deltaY: number) => void
}): () => void {
const { browserTabId, guest, resolveRenderer } = args
const {
browserTabId,
guest,
resolveRenderer,
isViewportPresetActive,
canViewportScroll,
onViewportWheelConsumed
} = args
const handler = (event: Electron.Event, mouse: Electron.MouseInputEvent): void => {
const direction = resolveGuestMouseWheelZoomDirection(mouse)
if (!direction) {
if (direction) {
// Why: wheel input over a focused webview never reaches renderer DOM handlers, so consume and forward here.
event.preventDefault()
markGuestWheelZoom(guest, direction)
resolveRenderer(browserTabId)?.send('ui:zoomBrowserPage', direction)
return
}
// Why: wheel input over a focused webview never reaches renderer DOM handlers, so consume and forward here.
if (
!isViewportPresetActive?.() ||
mouse.type !== 'mouseWheel' ||
!canViewportScroll?.(mouse as Electron.MouseWheelInputEvent)
) {
return
}
const { deltaX, deltaY } = mouse as Electron.MouseWheelInputEvent
const safeDeltaX = typeof deltaX === 'number' && Number.isFinite(deltaX) ? deltaX : 0
const safeDeltaY = typeof deltaY === 'number' && Number.isFinite(deltaY) ? deltaY : 0
if (safeDeltaX === 0 && safeDeltaY === 0) {
return
}
// Why: the host owns panning once emulation makes the guest viewport larger than the pane.
event.preventDefault()
markGuestWheelZoom(guest, direction)
resolveRenderer(browserTabId)?.send('ui:zoomBrowserPage', direction)
onViewportWheelConsumed?.(safeDeltaX, safeDeltaY)
resolveRenderer(browserTabId)?.send('ui:scrollBrowserPage', {
browserPageId: browserTabId,
deltaX: safeDeltaX,
deltaY: safeDeltaY
})
}
guest.on('before-mouse-event', handler)
@@ -0,0 +1,153 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
const mocks = vi.hoisted(() => ({
appGetPathMock: vi.fn(() => '/downloads'),
shellOpenExternalMock: vi.fn(),
browserWindowFromWebContentsMock: vi.fn(),
menuBuildFromTemplateMock: vi.fn(),
guestOffMock: vi.fn(),
guestOnMock: vi.fn(),
guestSetBackgroundThrottlingMock: vi.fn(),
guestSetWindowOpenHandlerMock: vi.fn(),
guestOpenDevToolsMock: vi.fn(),
webContentsFromIdMock: vi.fn(),
screenGetCursorScreenPointMock: vi.fn(() => ({ x: 0, y: 0 })),
openPopupWithOriginBarMock: vi.fn()
}))
vi.mock('electron', () => ({
app: { getPath: mocks.appGetPathMock },
BrowserWindow: { fromWebContents: mocks.browserWindowFromWebContentsMock },
clipboard: { writeText: vi.fn() },
shell: { openExternal: mocks.shellOpenExternalMock },
Menu: { buildFromTemplate: mocks.menuBuildFromTemplateMock },
screen: { getCursorScreenPoint: mocks.screenGetCursorScreenPointMock },
webContents: { fromId: mocks.webContentsFromIdMock }
}))
vi.mock('./popup-origin-bar-window', () => ({
openPopupWithOriginBar: mocks.openPopupWithOriginBarMock
}))
import { browserManager } from './browser-manager'
import {
rendererWebContentsId,
resetBrowserManagerMocks,
resetBrowserManagerState
} from './browser-manager-test-harness'
import { createViewportGuestFactory } from './browser-manager-viewport-test-fixtures'
const makeGuest = createViewportGuestFactory(mocks)
const OVERRIDE = { width: 375, height: 667, deviceScaleFactor: 2, mobile: true } as const
describe('browserManager viewport partial failure', () => {
beforeEach(() => {
resetBrowserManagerMocks(mocks)
resetBrowserManagerState()
})
it('keeps wheel routing active when follow-up setup fails after metrics apply', async () => {
const { guest, debuggerSendCommand } = makeGuest(42421)
debuggerSendCommand.mockImplementation((method: string) =>
method === 'Emulation.setTouchEmulationEnabled'
? Promise.reject(new Error('touch setup failed'))
: Promise.resolve(undefined)
)
mocks.webContentsFromIdMock.mockReturnValue(guest)
browserManager.attachGuestPolicies(guest as never)
browserManager.registerGuest({
browserPageId: 'tab-partial-apply',
webContentsId: guest.id as number,
rendererWebContentsId
})
const renderer = { isDestroyed: vi.fn(() => false), send: vi.fn() }
mocks.webContentsFromIdMock.mockImplementation((id: number) =>
id === rendererWebContentsId ? renderer : guest
)
await expect(browserManager.setViewportOverride('tab-partial-apply', OVERRIDE)).resolves.toBe(
false
)
browserManager.setViewportScrollState('tab-partial-apply', rendererWebContentsId, {
scrollLeft: 0,
scrollTop: 0,
maxScrollLeft: 400,
maxScrollTop: 300
})
const beforeMouseEvent = mocks.guestOnMock.mock.calls.findLast(
([event]) => event === 'before-mouse-event'
)?.[1] as ((event: Electron.Event, mouse: Electron.MouseInputEvent) => void) | undefined
expect(beforeMouseEvent).toBeDefined()
const preventDefault = vi.fn()
beforeMouseEvent?.(
{ preventDefault } as unknown as Electron.Event,
{
type: 'mouseWheel',
x: 0,
y: 0,
modifiers: [],
deltaX: 0,
deltaY: 120
} as Electron.MouseWheelInputEvent
)
expect(preventDefault).toHaveBeenCalledTimes(1)
expect(renderer.send).toHaveBeenCalledWith('ui:scrollBrowserPage', {
browserPageId: 'tab-partial-apply',
deltaX: 0,
deltaY: 120
})
})
it('keeps host panning available when metrics setup fails', async () => {
const { guest, debuggerSendCommand } = makeGuest(42422)
debuggerSendCommand.mockImplementation((method: string) =>
method === 'Emulation.setDeviceMetricsOverride'
? Promise.reject(new Error('metrics setup failed'))
: Promise.resolve(undefined)
)
mocks.webContentsFromIdMock.mockReturnValue(guest)
browserManager.attachGuestPolicies(guest as never)
browserManager.registerGuest({
browserPageId: 'tab-metrics-failed',
webContentsId: guest.id as number,
rendererWebContentsId
})
const renderer = { isDestroyed: vi.fn(() => false), send: vi.fn() }
mocks.webContentsFromIdMock.mockImplementation((id: number) =>
id === rendererWebContentsId ? renderer : guest
)
await expect(browserManager.setViewportOverride('tab-metrics-failed', OVERRIDE)).resolves.toBe(
false
)
browserManager.setViewportScrollState('tab-metrics-failed', rendererWebContentsId, {
scrollLeft: 0,
scrollTop: 0,
maxScrollLeft: 400,
maxScrollTop: 300
})
const beforeMouseEvent = mocks.guestOnMock.mock.calls.findLast(
([event]) => event === 'before-mouse-event'
)?.[1] as ((event: Electron.Event, mouse: Electron.MouseInputEvent) => void) | undefined
expect(beforeMouseEvent).toBeDefined()
const preventDefault = vi.fn()
beforeMouseEvent?.(
{ preventDefault } as unknown as Electron.Event,
{
type: 'mouseWheel',
x: 0,
y: 0,
modifiers: [],
deltaX: 0,
deltaY: 120
} as Electron.MouseWheelInputEvent
)
expect(preventDefault).toHaveBeenCalledTimes(1)
expect(renderer.send).toHaveBeenCalledWith('ui:scrollBrowserPage', {
browserPageId: 'tab-metrics-failed',
deltaX: 0,
deltaY: 120
})
})
})
+108 -5
View File
@@ -56,7 +56,8 @@ import type {
BrowserCertificateFailure,
BrowserLoadError,
BrowserSessionUserAgentMode,
BrowserViewportOverride
BrowserViewportOverride,
BrowserViewportScrollState
} from '../../shared/browser-workspace-types'
import {
type BrowserAnnotationViewportBridgeOptions,
@@ -264,6 +265,13 @@ export class BrowserManager {
// Why: presence means the preset requires a CDP UA override (installed or in flight), so navigation
// can re-issue it against the target URL's identity.
private readonly viewportUaOverrideMobileByTabId = new Map<string, boolean>()
// Why: host-side wheel panning follows the requested local viewport on the owning guest;
// replacement guests must not inherit a retired guest's state.
private readonly viewportPresetActiveByTabId = new Map<
string,
{ guestWebContentsId: number; active: boolean }
>()
private readonly viewportScrollStateByTabId = new Map<string, BrowserViewportScrollState>()
// Why: the confirmed CDP identity outranks getUserAgent; pending intent keeps rapid navigations
// ordered without claiming a failed write was installed.
private readonly authUserAgentOverrideStateByGuestId = new Map<
@@ -303,6 +311,36 @@ export class BrowserManager {
this.shouldForwardDictationShortcut = predicate
}
setViewportScrollState(
browserTabId: string,
rendererWebContentsId: number,
state: BrowserViewportScrollState
): void {
if (this.rendererWebContentsIdByTabId.get(browserTabId) !== rendererWebContentsId) {
return
}
if (
![state.scrollLeft, state.scrollTop, state.maxScrollLeft, state.maxScrollTop].every(
(value) => typeof value === 'number' && Number.isFinite(value) && value >= 0
)
) {
return
}
this.viewportScrollStateByTabId.set(browserTabId, state)
}
recordViewportScrollDelta(browserTabId: string, deltaX: number, deltaY: number): void {
const state = this.viewportScrollStateByTabId.get(browserTabId)
if (!state) {
return
}
this.viewportScrollStateByTabId.set(browserTabId, {
...state,
scrollLeft: Math.min(state.maxScrollLeft, Math.max(0, state.scrollLeft + deltaX)),
scrollTop: Math.min(state.maxScrollTop, Math.max(0, state.scrollTop + deltaY))
})
}
setBrowserGuestStateChangedListener(listener: ((worktreeId: string) => void) | null): void {
this.browserGuestStateChangedListener = listener
}
@@ -1395,6 +1433,8 @@ export class BrowserManager {
const previousWebContentsId = this.webContentsIdByTabId.get(browserTabId)
if (previousWebContentsId !== undefined && previousWebContentsId !== webContentsId) {
this.retireStaleGuestWebContents(previousWebContentsId)
this.viewportPresetActiveByTabId.delete(browserTabId)
this.viewportScrollStateByTabId.delete(browserTabId)
}
this.webContentsIdByTabId.set(browserTabId, webContentsId)
this.tabIdByWebContentsId.set(webContentsId, browserTabId)
@@ -1479,6 +1519,8 @@ export class BrowserManager {
// Why: drop the viewport-op chain so the Map doesn't retain a promise keyed to a destroyed guest.
this.viewportOpsByTabId.delete(browserTabId)
this.viewportUaOverrideMobileByTabId.delete(browserTabId)
this.viewportPresetActiveByTabId.delete(browserTabId)
this.viewportScrollStateByTabId.delete(browserTabId)
if (wcId !== undefined) {
this.pendingNavigationByGuestId.delete(wcId)
}
@@ -1514,6 +1556,8 @@ export class BrowserManager {
const previousWebContentsId = this.webContentsIdByTabId.get(browserPageId)
if (previousWebContentsId !== undefined && previousWebContentsId !== webContentsId) {
this.retireStaleGuestWebContents(previousWebContentsId)
this.viewportPresetActiveByTabId.delete(browserPageId)
this.viewportScrollStateByTabId.delete(browserPageId)
}
this.webContentsIdByTabId.set(browserPageId, webContentsId)
this.tabIdByWebContentsId.set(webContentsId, browserPageId)
@@ -1555,6 +1599,8 @@ export class BrowserManager {
this.sessionProfileIdByPageId.clear()
this.userAgentModeByPageId.clear()
this.viewportUaOverrideMobileByTabId.clear()
this.viewportPresetActiveByTabId.clear()
this.viewportScrollStateByTabId.clear()
this.authUserAgentOverrideStateByGuestId.clear()
this.pendingNavigationByGuestId.clear()
this.pendingLoadFailuresByGuestId.clear()
@@ -1890,10 +1936,22 @@ export class BrowserManager {
override: BrowserViewportOverride | null
): Promise<boolean> {
// Why: chain per-tab so rapid toggles don't interleave CDP commands and the last-requested override wins.
const expectedWebContentsId = this.webContentsIdByTabId.get(browserTabId)
if (expectedWebContentsId !== undefined) {
// Keep host panning available while CDP applies the requested dimensions. The guest id fence
// prevents this intent from leaking to a replacement guest; clearing the preset removes it.
this.viewportPresetActiveByTabId.set(browserTabId, {
guestWebContentsId: expectedWebContentsId,
active: override !== null
})
}
// The renderer resizes the host before CDP completes; discard the old geometry until it
// reports the new pane bounds so a pending preset cannot route wheel input using stale limits.
this.viewportScrollStateByTabId.delete(browserTabId)
const prev = this.viewportOpsByTabId.get(browserTabId) ?? Promise.resolve()
const next = prev
.catch(() => {})
.then(() => this.doSetViewportOverrideImpl(browserTabId, override))
.then(() => this.doSetViewportOverrideImpl(browserTabId, override, expectedWebContentsId))
this.viewportOpsByTabId.set(browserTabId, next)
try {
return await next
@@ -1958,10 +2016,11 @@ export class BrowserManager {
private async doSetViewportOverrideImpl(
browserTabId: string,
override: BrowserViewportOverride | null
override: BrowserViewportOverride | null,
expectedWebContentsId: number | undefined
): Promise<boolean> {
const webContentsId = this.webContentsIdByTabId.get(browserTabId)
if (!webContentsId) {
if (!webContentsId || webContentsId !== expectedWebContentsId) {
return false
}
const guest = webContents.fromId(webContentsId)
@@ -1994,6 +2053,12 @@ export class BrowserManager {
deviceScaleFactor: override.deviceScaleFactor,
mobile: override.mobile
})
if (this.webContentsIdByTabId.get(browserTabId) === webContentsId) {
this.viewportPresetActiveByTabId.set(browserTabId, {
guestWebContentsId: webContentsId,
active: true
})
}
await dbg.sendCommand('Emulation.setTouchEmulationEnabled', {
enabled: override.mobile,
maxTouchPoints: override.mobile ? 5 : 0
@@ -2007,6 +2072,12 @@ export class BrowserManager {
}
} else {
await dbg.sendCommand('Emulation.clearDeviceMetricsOverride', {})
if (this.webContentsIdByTabId.get(browserTabId) === webContentsId) {
this.viewportPresetActiveByTabId.set(browserTabId, {
guestWebContentsId: webContentsId,
active: false
})
}
await dbg.sendCommand('Emulation.setTouchEmulationEnabled', {
enabled: false,
maxTouchPoints: 0
@@ -2037,6 +2108,9 @@ export class BrowserManager {
throw error
}
}
if (this.webContentsIdByTabId.get(browserTabId) !== webContentsId) {
return false
}
return true
} catch {
return false
@@ -2223,11 +2297,40 @@ export class BrowserManager {
browserTabId,
guest,
resolveRenderer: (tabId) =>
resolveRendererWebContents(this.rendererWebContentsIdByTabId, tabId)
resolveRendererWebContents(this.rendererWebContentsIdByTabId, tabId),
isViewportPresetActive: () => {
const state = this.viewportPresetActiveByTabId.get(browserTabId)
return state?.guestWebContentsId === guest.id && state.active
},
canViewportScroll: (mouse) => this.canViewportScroll(browserTabId, mouse),
onViewportWheelConsumed: (deltaX, deltaY) =>
this.recordViewportScrollDelta(browserTabId, deltaX, deltaY)
})
)
}
private canViewportScroll(browserTabId: string, mouse: Electron.MouseWheelInputEvent): boolean {
const state = this.viewportScrollStateByTabId.get(browserTabId)
if (!state) {
return false
}
const deltaX = typeof mouse.deltaX === 'number' ? mouse.deltaX : 0
const deltaY = typeof mouse.deltaY === 'number' ? mouse.deltaY : 0
const canScrollAxis = (delta: number, position: number, maximum: number): boolean => {
if (delta < 0) {
return position > 0
}
if (delta > 0) {
return position < maximum
}
return false
}
return (
canScrollAxis(deltaX, state.scrollLeft, state.maxScrollLeft) ||
canScrollAxis(deltaY, state.scrollTop, state.maxScrollTop)
)
}
private forwardOrQueueGuestLoadFailure(
guestWebContentsId: number,
loadError: { code: number; description: string; validatedUrl: string }
@@ -1,5 +1,6 @@
import { spawnSync } from 'node:child_process'
import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises'
import { mkdir, mkdtemp, writeFile } from 'node:fs/promises'
import { removeTree } from '../../shared/windows-transient-lock-removal'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
@@ -113,7 +114,7 @@ describe('WSL CLI PowerShell boundary', () => {
expect(exitResult.error).toBeUndefined()
expect(exitResult.status).toBe(23)
} finally {
await rm(root, { recursive: true, force: true })
await removeTree(root)
}
}
)
@@ -1,6 +1,10 @@
import { mkdirSync, readFileSync, realpathSync, rmSync, writeFileSync } from 'node:fs'
import { dirname, join, resolve, sep } from 'node:path'
import { isDefinitiveAbsence } from '../../shared/definitive-filesystem-absence'
import {
WINDOWS_RM_MAX_RETRIES,
WINDOWS_RM_RETRY_DELAY_MS
} from '../../shared/windows-transient-lock-removal'
import { quotePosixShell } from '../../shared/wsl-login-shell-command'
import { parseWslUncPath } from '../../shared/wsl-paths'
import { toWindowsWslPath } from '../wsl'
@@ -10,10 +14,6 @@ import { writeFileAtomically } from './fs-utils'
import { ManagedCodexHomeTemporarilyUnavailableError } from './host-codex-managed-home-ownership'
import type { CodexManagedHomePath } from './codex-managed-home-path'
// Why: mirrors the Windows rm retry policy in local-worktree-filesystem — a
// just-terminated codex login can briefly keep handles inside a managed home.
const WINDOWS_RM_MAX_RETRIES = 8
const WINDOWS_RM_RETRY_DELAY_MS = 150
const WSL_MANAGED_HOME_TIMEOUT_MS = 5_000
function removeManagedHomeTreeSync(targetPath: string): void {
@@ -22,6 +22,10 @@ export type CodexAppServerConnection = {
notify: (method: string, params?: Record<string, unknown>) => void
respond: (id: number | string, result: unknown) => void
respondWithError: (id: number | string, code: number, message: string) => void
/** Stops provider stdout at a record boundary while a durable sink drains. */
pauseReading?: () => void
/** Continues with any records retained from the chunk that triggered the pause. */
resumeReading?: () => void
/** Resolves true only after the child emitted `exit` or `close`; false is unproven. */
close: () => Promise<boolean>
}
@@ -5,6 +5,8 @@ import { PassThrough } from 'node:stream'
import { afterEach, describe, expect, it, vi } from 'vitest'
import type { spawnProcess } from '../../shared/child-process/run-process'
import {
CODEX_APP_SERVER_MAX_RECORD_BYTES,
CodexAppServerFrameSizeError,
isCodexAppServerRequestError,
openCodexAppServerConnection,
type CodexAppServerConnection,
@@ -128,6 +130,67 @@ function rejection(promise: Promise<unknown>): Promise<Error> {
)
}
function commandCompletionFixture(
targetBytes: number,
itemId = 'item-large'
): { line: string; output: string } {
const frame = {
method: 'item/completed',
params: {
turnId: 'turn-large',
item: { id: itemId, type: 'commandExecution', aggregated_output: '' }
}
}
const emptyBytes = Buffer.byteLength(JSON.stringify(frame), 'utf8')
const remaining = targetBytes - emptyBytes
if (remaining < 0) {
throw new Error(`target ${targetBytes} is smaller than fixture envelope ${emptyBytes}`)
}
const output = `${'\n'.repeat(Math.floor(remaining / 2))}${remaining % 2 ? 'x' : ''}`
frame.params.item.aggregated_output = output
const line = JSON.stringify(frame)
expect(Buffer.byteLength(line, 'utf8')).toBe(targetBytes)
return { line: `${line}\n`, output }
}
function commandCompletionLine(targetBytes: number): string {
return commandCompletionFixture(targetBytes).line
}
function responseLine(targetBytes: number, id: number): string {
const frame = { id, result: { data: '' } }
const emptyBytes = Buffer.byteLength(JSON.stringify(frame), 'utf8')
frame.result.data = 'x'.repeat(targetBytes - emptyBytes)
const line = JSON.stringify(frame)
expect(Buffer.byteLength(line, 'utf8')).toBe(targetBytes)
return `${line}\n`
}
function resultFirstResponseLine(targetBytes: number, id: number, resultKey: 'result' | 'error') {
const response =
resultKey === 'result'
? `{"result":{"turn":{"id":"turn-large"}},"id":${id},"padding":"`
: `{"error":{"code":-32000,"message":"too large"},"id":${id},"padding":"`
const suffix = '"}'
const padding = targetBytes - Buffer.byteLength(response + suffix, 'utf8')
if (padding < 0) {
throw new Error(`target ${targetBytes} is smaller than fixture envelope`)
}
const line = `${response}${'x'.repeat(padding)}${suffix}`
expect(Buffer.byteLength(line, 'utf8')).toBe(targetBytes)
return `${line}\n`
}
function giantContainerBeforeIdResponseLine(targetBytes: number, id: number): string {
const giantResult = `{"result":{"payload":"${'x'.repeat(62_000)}"},"id":${id},"padding":"`
const suffix = '"}'
const padding = targetBytes - Buffer.byteLength(giantResult + suffix, 'utf8')
if (padding < 0) {
throw new Error(`target ${targetBytes} is smaller than giant response envelope`)
}
return `${giantResult}${'x'.repeat(padding)}${suffix}\n`
}
describe('openCodexAppServerConnection', () => {
it('advertises the experimental API required for rollout-path resume', async () => {
const { child, spawnImpl, written } = stubChild()
@@ -413,25 +476,255 @@ describe('openCodexAppServerConnection', () => {
await expect(connection.close()).resolves.toBe(true)
})
it('ends the connection rather than buffering an oversized line', async () => {
it.each([1_090_188, 2_900_090])(
'accepts a realistic %i-byte escaped command completion and keeps processing',
async (frameBytes) => {
const { child, spawnImpl } = stubChild()
answerInitialize(child)
const completed: unknown[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{
onNotification: (method, params) => {
if (method === 'item/completed') {
completed.push(params)
}
}
},
spawnImpl
)
const line = Buffer.from(commandCompletionLine(frameBytes), 'utf8')
const split = Math.floor(line.length / 3)
child.stdout.write(line.subarray(0, split))
child.stdout.write(line.subarray(split, split * 2))
child.stdout.write(line.subarray(split * 2))
child.stdout.write('{"method":"turn/completed","params":{"turn":{"id":"turn-large"}}}\n')
await vi.waitFor(() => expect(completed).toHaveLength(1))
expect(
(completed[0] as { item: { aggregated_output: string } }).item.aggregated_output.length
).toBeGreaterThan(500_000)
expect(connection.closed).toBe(false)
await connection.close()
}
)
it('accepts two realistic large command completions without losing either payload', async () => {
const { child, spawnImpl } = stubChild()
answerInitialize(child)
const completed: { item: { id: string; aggregated_output: string } }[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{
onNotification: (method, params) => {
if (method === 'item/completed') {
completed.push(params as { item: { id: string; aggregated_output: string } })
}
}
},
spawnImpl
)
const fixtures = [
commandCompletionFixture(1_090_188, 'item-large-a'),
commandCompletionFixture(2_900_090, 'item-large-b')
]
child.stdout.write(fixtures[0]!.line)
child.stdout.write(fixtures[1]!.line)
await vi.waitFor(() => expect(completed).toHaveLength(2))
expect(completed.map((entry) => entry.item.id)).toEqual(['item-large-a', 'item-large-b'])
expect(
completed.map((entry) => Buffer.byteLength(entry.item.aggregated_output, 'utf8'))
).toEqual(fixtures.map((fixture) => Buffer.byteLength(fixture.output, 'utf8')))
expect(connection.closed).toBe(false)
await connection.close()
})
it('accepts the 16 MiB boundary and settles one byte above without killing the provider', async () => {
const { child, spawnImpl } = stubChild({ exitOnStdinEnd: false })
answerInitialize(child)
const exits: string[] = []
const frames: { kind: string; payload: unknown }[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{ onExit: (error) => exits.push(error.message) },
{
onExit: (error) => exits.push(error.message),
onUnhandledFrame: (kind, payload) => frames.push({ kind, payload })
},
spawnImpl
)
child.kill.mockImplementation(() => {
child.emit('exit', null, 'SIGKILL')
return true
})
const below = connection.request('thread/resume')
child.stdout.write(responseLine(CODEX_APP_SERVER_MAX_RECORD_BYTES - 1, 2))
expect(((await below) as { data: string }).data.length).toBeGreaterThan(
CODEX_APP_SERVER_MAX_RECORD_BYTES - 40
)
const at = connection.request('thread/resume')
child.stdout.write(responseLine(CODEX_APP_SERVER_MAX_RECORD_BYTES, 3))
expect(((await at) as { data: string }).data.length).toBeGreaterThan(
CODEX_APP_SERVER_MAX_RECORD_BYTES - 40
)
const inFlight = rejection(connection.request('turn/start'))
child.stdout.write('x'.repeat(1024 * 1024 + 1))
child.stdout.write(responseLine(CODEX_APP_SERVER_MAX_RECORD_BYTES + 1, 4))
expect(await inFlight).toBeInstanceOf(CodexAppServerFrameSizeError)
expect(frames).toEqual([
{
kind: 'frame:oversized-response',
payload: expect.objectContaining({ classification: 'response', id: 4 })
}
])
expect(exits).toEqual([])
expect(connection.closed).toBe(false)
const later = connection.request('turn/start')
child.stdout.write('{"id":5,"result":{"turn":{"id":"turn-next"}}}\n')
await expect(later).resolves.toEqual({ turn: { id: 'turn-next' } })
child.emit('exit', 0, null)
await connection.close()
})
it.each(['result', 'error'] as const)(
'classifies oversized responses with %s before id',
async (resultKey) => {
const { child, spawnImpl } = stubChild({ exitOnStdinEnd: false })
answerInitialize(child)
const frames: { kind: string; payload: unknown }[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{ onUnhandledFrame: (kind, payload) => frames.push({ kind, payload }) },
spawnImpl
)
const inFlight = rejection(connection.request('thread/resume'))
child.stdout.write(
resultFirstResponseLine(CODEX_APP_SERVER_MAX_RECORD_BYTES + 1, 2, resultKey)
)
expect(await inFlight).toBeInstanceOf(CodexAppServerFrameSizeError)
expect(frames).toEqual([
{
kind: 'frame:oversized-response',
payload: expect.objectContaining({ classification: 'response', id: 2 })
}
])
expect(connection.closed).toBe(false)
child.emit('exit', 0, null)
await connection.close()
}
)
it('classifies an oversized response when a giant result container precedes id', async () => {
const { child, spawnImpl } = stubChild({ exitOnStdinEnd: false })
answerInitialize(child)
const frames: { kind: string; payload: unknown }[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{ onUnhandledFrame: (kind, payload) => frames.push({ kind, payload }) },
spawnImpl
)
const inFlight = rejection(connection.request('thread/resume'))
child.stdout.write(giantContainerBeforeIdResponseLine(CODEX_APP_SERVER_MAX_RECORD_BYTES + 1, 2))
await expect(inFlight).resolves.toBeInstanceOf(CodexAppServerFrameSizeError)
expect(frames).toEqual([
{
kind: 'frame:oversized-response',
payload: expect.objectContaining({ classification: 'response', id: 2 })
}
])
child.emit('exit', 0, null)
await connection.close()
})
it('answers an oversized provider request once and resumes after its newline', async () => {
const { child, spawnImpl, written } = stubChild()
answerInitialize(child)
const frames: string[] = []
const notifications: string[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{
onUnhandledFrame: (kind) => frames.push(kind),
onNotification: (method) => notifications.push(method)
},
spawnImpl
)
child.stdout.write(
`{"id":"approval-1","method":"item/requestApproval","params":{"data":"${'x'.repeat(
CODEX_APP_SERVER_MAX_RECORD_BYTES
)}"}}\n{"method":"turn/completed","params":{}}\n`
)
await vi.waitFor(() => expect(notifications).toEqual(['turn/completed']))
expect(frames).toEqual(['frame:oversized-request'])
expect(written.at(-1)).toEqual({
id: 'approval-1',
error: {
code: -32001,
message: `request exceeds ${CODEX_APP_SERVER_MAX_RECORD_BYTES} byte limit`
}
})
expect(connection.closed).toBe(false)
await connection.close()
})
it('keeps malformed and non-object JSON non-fatal and processes the next record', async () => {
const { child, spawnImpl } = stubChild()
answerInitialize(child)
const frames: { kind: string; payload: unknown }[] = []
const notifications: string[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{
onUnhandledFrame: (kind, payload) => frames.push({ kind, payload }),
onNotification: (method) => notifications.push(method)
},
spawnImpl
)
child.stdout.write('not json\n[]\n{"method":"turn/completed","params":{}}\n')
await vi.waitFor(() => expect(notifications).toEqual(['turn/completed']))
expect(frames).toEqual([
{ kind: 'frame:invalid-json', payload: 'not json' },
{ kind: 'frame:invalid-json', payload: '[]' }
])
expect(connection.closed).toBe(false)
await connection.close()
})
it('pauses between coalesced records and resumes the retained remainder', async () => {
const { child, spawnImpl } = stubChild()
answerInitialize(child)
const notifications: string[] = []
let connection: CodexAppServerConnection
connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{
onNotification: (method) => {
notifications.push(method)
if (notifications.length === 1) {
connection.pauseReading?.()
}
}
},
spawnImpl
)
child.stdout.write(
'{"method":"item/started","params":{}}\n{"method":"item/completed","params":{}}\n'
)
await vi.waitFor(() => expect(notifications).toEqual(['item/started']))
connection.resumeReading?.()
await vi.waitFor(() => expect(notifications).toEqual(['item/started', 'item/completed']))
expect((await inFlight).message).toContain('oversized')
expect(exits[0]).toContain('oversized')
await connection.close()
})
@@ -485,8 +778,8 @@ describe('openCodexAppServerConnection', () => {
spawnImpl
)
// The oversized line kills the child, so its own `close` lands afterwards.
child.stdout.write('x'.repeat(1024 * 1024 + 1))
// An unclassifiable oversized line initiates recovery, then child exit lands afterwards.
child.stdout.write('x'.repeat(CODEX_APP_SERVER_MAX_RECORD_BYTES + 1))
child.stderr.write('killed\n')
await flushStreams()
child.emit('exit', null, 'SIGKILL')
@@ -498,6 +791,28 @@ describe('openCodexAppServerConnection', () => {
await connection.close()
})
it('does not report recovery for a protocol failure until child exit is observed', async () => {
const { child, spawnImpl } = stubChild({ exitOnStdinEnd: false })
answerInitialize(child)
const exits: string[] = []
const connection = await openCodexAppServerConnection(
{ command: 'codex', args: ['app-server'] },
{ onExit: (error) => exits.push(error.message) },
spawnImpl
)
const inFlight = rejection(connection.request('turn/start'))
child.stdout.write('x'.repeat(CODEX_APP_SERVER_MAX_RECORD_BYTES + 1))
await flushStreams()
expect(exits).toHaveLength(0)
expect((await inFlight).message).toContain('oversized')
child.emit('exit', null, 'SIGKILL')
expect(exits).toHaveLength(1)
await connection.close()
})
it('treats a broken stdin pipe as the end of the transport', async () => {
const { child, spawnImpl } = stubChild({ exitOnStdinEnd: false })
answerInitialize(child)
+52 -103
View File
@@ -4,16 +4,16 @@ import { createProviderSpawnSpec } from './codex-app-server-posix-supervisor'
import { buildCodexAppServerExitError } from './codex-app-server-exit-error'
import { initializeCodexAppServerConnection } from './codex-app-server-handshake'
import { CodexAppServerHandshakeExitUnprovenError } from './codex-app-server-handshake-exit-proof'
import { isAppServerRecord, parseCodexAppServerJsonLine } from './codex-app-server-jsonl'
import { terminateCodexAppServerProcessTree } from './codex-app-server-process-teardown'
import { CodexAppServerRequestError } from './codex-app-server-request-error'
import { CODEX_SPAWN_TOKEN_ENV } from './codex-structured-owner-identity'
import { waitForProcessExitUntil } from './codex-process-exit-deadline'
import { NDJSON_MAX_LINE_BYTES } from '../../shared/main-process-ndjson-framer'
import {
CodexAppServerTimeoutError,
CodexAppServerUnsupportedError,
isCodexMethodNotFoundError
CodexAppServerUnsupportedError
} from './codex-app-server-session'
import { createCodexAppServerRecordDispatcher } from './codex-app-server-record-dispatch'
import { createCodexAppServerRecordReader } from './codex-app-server-record-reader'
import type {
CodexAppServerConnection,
CodexAppServerConnectionHandlers
@@ -28,6 +28,7 @@ export {
CodexAppServerRequestError,
isCodexAppServerRequestError
} from './codex-app-server-request-error'
export { CodexAppServerFrameSizeError } from './codex-app-server-frame-size-error'
// Structured chat needs a persistent bidirectional child and per-request deadlines;
// the request-scoped app-server runner cannot carry approvals or streamed turns.
@@ -47,14 +48,7 @@ const DEFAULT_REQUEST_TIMEOUT_MS = 30_000
const GRACEFUL_EXIT_MS = 1_500
const FORCED_EXIT_MS = 1_000
const STDERR_TAIL_MAX_BYTES = 8192
const STDOUT_LINE_MAX_BYTES = 1024 * 1024
type PendingRequest = {
method: string
resolve: (result: unknown) => void
reject: (error: Error) => void
timer: ReturnType<typeof setTimeout>
}
export const CODEX_APP_SERVER_MAX_RECORD_BYTES = NDJSON_MAX_LINE_BYTES
/**
* Spawns `codex app-server`, completes the initialize handshake, and returns a
@@ -80,12 +74,12 @@ export async function openCodexAppServerConnection(
return terminateCodexAppServerProcessTree(child, spawnToken)
}
const pending = new Map<number, PendingRequest>()
let stderrTail = ''
let nextRequestId = 1
let exited = false
let exitObserved = false
let closing = false
let exitReported = false
const exitProof = new RetryableProcessExitProof()
/** First terminal cause, or null while the transport is still usable. Set once:
* a child that dies reaches us through several listeners, and the specific
@@ -103,31 +97,38 @@ export async function openCodexAppServerConnection(
resolveExit()
}
child.on('exit', observeExit)
child.on('exit', () => {
observeExit()
handleUnexpectedEnd()
})
function buildExitError(cause?: Error): Error {
return buildCodexAppServerExitError(stderrTail, cause)
}
function failPending(error: Error): void {
for (const waiter of pending.values()) {
clearTimeout(waiter.timer)
waiter.reject(error)
const dispatcher = createCodexAppServerRecordDispatcher({
handlers,
writeResponse,
onProtocolFailure: (error) => {
handleUnexpectedEnd(error)
void terminateProcessTree()
}
pending.clear()
}
})
/** A death nobody asked for kills every in-flight call AND tells the owner,
* which is the only signal the session has that its lease is now worthless.
* Once only: an oversized line kills the child and its `close` arrives after,
* and a spawn failure arrives as both `error` and `close`. */
function handleUnexpectedEnd(cause?: Error): void {
if (terminalError) {
return
if (!terminalError) {
terminalError = buildExitError(cause)
dispatcher.failPending(terminalError)
}
terminalError = buildExitError(cause)
failPending(terminalError)
if (!closing) {
// Transport/protocol failures make the connection unusable immediately so
// callers do not hang, but recovery must not treat that as a child exit
// until the execution host has observed `exit`/`close`.
if (exitObserved && !closing && !exitReported) {
exitReported = true
handlers.onExit?.(terminalError)
}
}
@@ -149,87 +150,33 @@ export async function openCodexAppServerConnection(
// close the reap is already under way and `exited` must stay honest, or
// `close` would skip the kill it still owes.
if (closing) {
failPending(error)
dispatcher.failPending(error)
return
}
void terminateProcessTree()
handleUnexpectedEnd(error)
void terminateProcessTree()
})
function dispatchMessage(message: Record<string, unknown>): void {
const hasMethod = typeof message.method === 'string'
const hasId = typeof message.id === 'number' || typeof message.id === 'string'
if (hasMethod && hasId) {
handlers.onServerRequest?.({
id: message.id as number | string,
method: message.method as string,
params: message.params
})
return
}
if (hasMethod) {
handlers.onNotification?.(message.method as string, message.params)
return
}
if (typeof message.id !== 'number') {
handlers.onUnhandledFrame?.('frame:unclassified', message)
return
}
const waiter = pending.get(message.id)
if (!waiter) {
handlers.onUnhandledFrame?.('response:unmatched', message)
return
}
pending.delete(message.id)
clearTimeout(waiter.timer)
const error = message.error
if (isAppServerRecord(error)) {
const detail = typeof error.message === 'string' ? error.message : 'unknown error'
waiter.reject(
isCodexMethodNotFoundError(error)
? new CodexAppServerUnsupportedError(
`codex app-server does not support ${waiter.method}: ${detail}`
)
: new CodexAppServerRequestError(
waiter.method,
typeof error.code === 'number' ? error.code : null,
`codex app-server ${waiter.method} failed: ${detail}`
)
)
return
}
waiter.resolve(message.result)
}
let stdoutBuffer = ''
child.stdout.setEncoding('utf8').on('data', (chunk: string) => {
stdoutBuffer += chunk
if (Buffer.byteLength(stdoutBuffer) > STDOUT_LINE_MAX_BYTES) {
child.stdout.destroy()
void terminateProcessTree()
handleUnexpectedEnd(new Error('codex app-server emitted an oversized JSONL line'))
return
}
let newlineIndex: number
while ((newlineIndex = stdoutBuffer.indexOf('\n')) !== -1) {
const line = stdoutBuffer.slice(0, newlineIndex).trim()
stdoutBuffer = stdoutBuffer.slice(newlineIndex + 1)
if (!line) {
continue
}
const parsed = parseCodexAppServerJsonLine(line)
if (!parsed) {
const recordReader = createCodexAppServerRecordReader({
stdout: child.stdout,
maxRecordBytes: CODEX_APP_SERVER_MAX_RECORD_BYTES,
onRecord: (parsed, line) => {
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
handlers.onUnhandledFrame?.('frame:invalid-json', line)
continue
}
try {
dispatchMessage(parsed)
} catch (error) {
child.stdout.destroy()
void terminateProcessTree()
handleUnexpectedEnd(error instanceof Error ? error : new Error(String(error)))
return
}
dispatcher.dispatch(parsed as Record<string, unknown>)
},
onRejected: (rejected) => {
if (rejected.kind === 'invalid-json') {
handlers.onUnhandledFrame?.('frame:invalid-json', rejected.line)
} else {
dispatcher.rejectOversized(rejected)
}
},
onFatal: (error) => {
handleUnexpectedEnd(error)
void terminateProcessTree()
}
})
@@ -268,14 +215,14 @@ export async function openCodexAppServerConnection(
// Why: per request, not per session — a chat session outlives every call,
// so only the individual call can carry a deadline.
const timer = setTimeout(() => {
pending.delete(id)
dispatcher.deletePending(id)
reject(new CodexAppServerTimeoutError(`codex app-server ${method} exceeded ${timeoutMs}ms`))
}, timeoutMs)
pending.set(id, { method, resolve, reject, timer })
dispatcher.addPending(id, { method, resolve, reject, timer })
try {
sendLine(params === undefined ? { method, id } : { method, id, params })
} catch (error) {
pending.delete(id)
dispatcher.deletePending(id)
clearTimeout(timer)
reject(error instanceof Error ? error : new Error(String(error)))
}
@@ -309,13 +256,13 @@ export async function openCodexAppServerConnection(
if (!exited) {
const treeExited = await terminateProcessTree()
if (!treeExited) {
failPending(new Error('codex app-server process-tree exit was not proven'))
dispatcher.failPending(new Error('codex app-server process-tree exit was not proven'))
return false
}
await waitForProcessExitUntil(exitPromise, FORCED_EXIT_MS)
}
}
failPending(new Error('codex app-server connection closed'))
dispatcher.failPending(new Error('codex app-server connection closed'))
return exitObserved
})
}
@@ -331,6 +278,8 @@ export async function openCodexAppServerConnection(
notify,
respond: (id, result) => writeResponse({ id, result }),
respondWithError: (id, code, message) => writeResponse({ id, error: { code, message } }),
pauseReading: recordReader.pause,
resumeReading: recordReader.resume,
close
}
@@ -0,0 +1,12 @@
export class CodexAppServerFrameSizeError extends Error {
constructor(
readonly method: string | null,
readonly observedBytes: number,
readonly maxBytes: number
) {
super(
`codex app-server${method ? ` ${method}` : ''} response exceeds ${maxBytes} byte limit (${observedBytes} bytes received)`
)
this.name = 'CodexAppServerFrameSizeError'
}
}
@@ -0,0 +1,166 @@
import type { NdjsonRejectedRecord } from '../../shared/main-process-ndjson-framer'
import type { CodexAppServerConnectionHandlers } from './codex-app-server-connection-types'
import { CodexAppServerFrameSizeError } from './codex-app-server-frame-size-error'
import { isAppServerRecord } from './codex-app-server-jsonl'
import { CodexAppServerRequestError } from './codex-app-server-request-error'
import {
CodexAppServerUnsupportedError,
isCodexMethodNotFoundError
} from './codex-app-server-session'
import { classifyJsonRpcPrefix } from './codex-app-server-record-prefix'
const OVERSIZED_REQUEST_ERROR_CODE = -32001
export type CodexPendingRequest = {
method: string
resolve: (result: unknown) => void
reject: (error: Error) => void
timer: ReturnType<typeof setTimeout>
}
export function createCodexAppServerRecordDispatcher(input: {
handlers: CodexAppServerConnectionHandlers
writeResponse: (payload: Record<string, unknown>) => void
onProtocolFailure: (error: Error) => void
}): {
addPending: (id: number, waiter: CodexPendingRequest) => void
deletePending: (id: number) => void
failPending: (error: Error) => void
dispatch: (message: Record<string, unknown>) => void
rejectOversized: (rejected: NdjsonRejectedRecord & { kind: 'line-too-long' }) => void
} {
const pending = new Map<number, CodexPendingRequest>()
const failPending = (error: Error): void => {
for (const waiter of pending.values()) {
clearTimeout(waiter.timer)
waiter.reject(error)
}
pending.clear()
}
const failPendingForOversizedUnknown = (
record: NdjsonRejectedRecord & { kind: 'line-too-long' }
): void => {
for (const waiter of pending.values()) {
clearTimeout(waiter.timer)
waiter.reject(
new CodexAppServerFrameSizeError(waiter.method, record.observedBytes, record.maxLineBytes)
)
}
pending.clear()
}
const dispatch = (message: Record<string, unknown>): void => {
const hasMethod = typeof message.method === 'string'
const hasId = typeof message.id === 'number' || typeof message.id === 'string'
if (hasMethod && hasId) {
input.handlers.onServerRequest?.({
id: message.id as number | string,
method: message.method as string,
params: message.params
})
return
}
if (hasMethod) {
input.handlers.onNotification?.(message.method as string, message.params)
return
}
if (typeof message.id !== 'number') {
input.handlers.onUnhandledFrame?.('frame:unclassified', message)
return
}
const waiter = pending.get(message.id)
if (!waiter) {
input.handlers.onUnhandledFrame?.('response:unmatched', message)
return
}
pending.delete(message.id)
clearTimeout(waiter.timer)
const error = message.error
if (isAppServerRecord(error)) {
const detail = typeof error.message === 'string' ? error.message : 'unknown error'
waiter.reject(
isCodexMethodNotFoundError(error)
? new CodexAppServerUnsupportedError(
`codex app-server does not support ${waiter.method}: ${detail}`
)
: new CodexAppServerRequestError(
waiter.method,
typeof error.code === 'number' ? error.code : null,
`codex app-server ${waiter.method} failed: ${detail}`
)
)
return
}
waiter.resolve(message.result)
}
const rejectOversized = (rejected: NdjsonRejectedRecord & { kind: 'line-too-long' }): void => {
const classification = classifyJsonRpcPrefix(rejected.prefix)
const payload = {
reason: 'record-too-large',
observedBytes: rejected.observedBytes,
maxBytes: rejected.maxLineBytes,
classification: classification.kind,
...('id' in classification ? { id: classification.id } : {}),
...('method' in classification ? { method: classification.method } : {})
}
if (classification.kind === 'response') {
const waiter = pending.get(classification.id)
if (waiter) {
pending.delete(classification.id)
clearTimeout(waiter.timer)
waiter.reject(
new CodexAppServerFrameSizeError(
waiter.method,
rejected.observedBytes,
rejected.maxLineBytes
)
)
} else {
input.handlers.onUnhandledFrame?.('frame:oversized-response', payload)
failPendingForOversizedUnknown(rejected)
input.onProtocolFailure(
new Error(
`codex app-server oversized response ${classification.id} had no pending request`
)
)
return
}
input.handlers.onUnhandledFrame?.('frame:oversized-response', payload)
return
}
if (classification.kind === 'server-request') {
input.writeResponse({
id: classification.id,
error: {
code: OVERSIZED_REQUEST_ERROR_CODE,
message: `request exceeds ${rejected.maxLineBytes} byte limit`
}
})
input.handlers.onUnhandledFrame?.('frame:oversized-request', payload)
return
}
if (classification.kind === 'notification') {
input.handlers.onUnhandledFrame?.('frame:oversized-notification', payload)
return
}
input.handlers.onUnhandledFrame?.('frame:oversized-unclassified', payload)
input.onProtocolFailure(
new Error(
classification.kind === 'response-unknown'
? 'codex app-server emitted an oversized response with an unknown shape'
: 'codex app-server emitted an oversized unclassifiable JSONL record'
)
)
}
return {
addPending: (id, waiter) => pending.set(id, waiter),
deletePending: (id) => pending.delete(id),
failPending,
dispatch,
rejectOversized
}
}
@@ -0,0 +1,186 @@
export type JsonRpcPrefix =
| { kind: 'response'; id: number }
| { kind: 'server-request'; id: number | string; method: string }
| { kind: 'notification'; method: string }
| { kind: 'response-unknown' }
| { kind: 'unknown' }
type LeadingProperty = { key: string; value?: string | number }
function readJsonStringEnd(value: string, start: number): number | null {
if (value[start] !== '"') {
return null
}
let escaped = false
for (let index = start + 1; index < value.length; index += 1) {
const character = value[index]
if (escaped) {
escaped = false
} else if (character === '\\') {
escaped = true
} else if (character === '"') {
return index + 1
}
}
return null
}
function skipJsonContainer(value: string, start: number): number | null {
const opening = value[start]
const closing = opening === '{' ? '}' : opening === '[' ? ']' : null
if (!closing) {
return null
}
const stack = [closing]
let escaped = false
let inString = false
for (let index = start + 1; index < value.length; index += 1) {
const character = value[index]
if (inString) {
if (escaped) {
escaped = false
} else if (character === '\\') {
escaped = true
} else if (character === '"') {
inString = false
}
continue
}
if (character === '"') {
inString = true
continue
}
if (character === '{') {
stack.push('}')
} else if (character === '[') {
stack.push(']')
} else if (character === stack.at(-1)) {
stack.pop()
if (stack.length === 0) {
return index + 1
}
}
}
return null
}
function skipJsonLiteral(value: string, start: number): number | null {
for (const literal of ['true', 'false', 'null']) {
if (value.startsWith(literal, start)) {
return start + literal.length
}
}
return null
}
function leadingJsonRpcProperties(prefix: string): LeadingProperty[] {
const properties: LeadingProperty[] = []
let cursor = 0
const skipWhitespace = (): void => {
while (/\s/.test(prefix[cursor] ?? '')) {
cursor += 1
}
}
skipWhitespace()
if (prefix[cursor] !== '{') {
return properties
}
cursor += 1
while (properties.length < 8) {
skipWhitespace()
const keyEnd = readJsonStringEnd(prefix, cursor)
if (keyEnd === null) {
break
}
let key: unknown
try {
key = JSON.parse(prefix.slice(cursor, keyEnd))
} catch {
break
}
cursor = keyEnd
skipWhitespace()
if (prefix[cursor] !== ':') {
break
}
cursor += 1
skipWhitespace()
if (prefix[cursor] === '{' || prefix[cursor] === '[') {
properties.push({ key: String(key) })
const valueEnd = skipJsonContainer(prefix, cursor)
if (valueEnd === null) {
break
}
cursor = valueEnd
skipWhitespace()
if (prefix[cursor] !== ',') {
break
}
cursor += 1
continue
}
const literalEnd = skipJsonLiteral(prefix, cursor)
if (literalEnd !== null) {
properties.push({ key: String(key) })
cursor = literalEnd
skipWhitespace()
if (prefix[cursor] !== ',') {
break
}
cursor += 1
continue
}
const stringEnd = readJsonStringEnd(prefix, cursor)
if (stringEnd !== null) {
try {
properties.push({ key: String(key), value: JSON.parse(prefix.slice(cursor, stringEnd)) })
} catch {
break
}
cursor = stringEnd
skipWhitespace()
if (prefix[cursor] !== ',') {
break
}
cursor += 1
continue
}
const match = /^-?(?:0|[1-9]\d*)/.exec(prefix.slice(cursor))
if (!match) {
break
}
properties.push({ key: String(key), value: Number(match[0]) })
cursor += match[0].length
skipWhitespace()
if (prefix[cursor] !== ',') {
break
}
cursor += 1
}
return properties
}
export function classifyJsonRpcPrefix(prefix: string): JsonRpcPrefix {
// Only inspect complete top-level properties. Searching arbitrary quoted
// keys would let a nested result/params object impersonate JSON-RPC fields.
const properties = leadingJsonRpcProperties(prefix)
const method = properties.find((property) => property.key === 'method')?.value
const id = properties.find((property) => property.key === 'id')?.value
if (typeof method === 'string' && (typeof id === 'number' || typeof id === 'string')) {
return { kind: 'server-request', id, method }
}
if (
typeof id === 'number' &&
properties.some((property) => property.key === 'id') &&
properties.some((property) => property.key === 'result' || property.key === 'error')
) {
return { kind: 'response', id }
}
if (typeof id === 'number') {
return { kind: 'response-unknown' }
}
if (typeof method === 'string' && properties.some((property) => property.key === 'params')) {
return { kind: 'notification', method }
}
return { kind: 'unknown' }
}
@@ -0,0 +1,51 @@
import type { Readable } from 'node:stream'
import {
createIncrementalNdjsonFramer,
type NdjsonRejectedRecord
} from '../../shared/main-process-ndjson-framer'
type RecordReaderStream = Pick<Readable, 'on' | 'pause' | 'resume' | 'setEncoding'>
export type CodexAppServerRecordReader = {
pause: () => void
resume: () => void
}
export function createCodexAppServerRecordReader(input: {
stdout: RecordReaderStream
maxRecordBytes: number
onRecord: (record: unknown, line: string) => void
onRejected: (rejected: NdjsonRejectedRecord) => void
onFatal: (error: Error) => void
}): CodexAppServerRecordReader {
let paused = false
const framer = createIncrementalNdjsonFramer(input.onRecord, input.onRejected, {
maxLineBytes: input.maxRecordBytes,
shouldPause: () => paused
})
input.stdout.setEncoding('utf8').on('data', (chunk: string) => {
try {
framer.feed(chunk)
} catch (error) {
input.onFatal(error instanceof Error ? error : new Error(String(error)))
}
})
return {
pause: () => {
paused = true
input.stdout.pause()
},
resume: () => {
if (!paused) {
return
}
paused = false
framer.resume()
if (!paused) {
input.stdout.resume()
}
}
}
}
@@ -1,6 +1,8 @@
import { win32 as pathWin32 } from 'node:path'
import type { SFTPWrapper } from 'ssh2'
import type { AgentHookInstallStatus } from '../../shared/agent-hook-types'
import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path'
import { dedupeInFlightRun } from '../in-flight-run-dedupe'
import { refreshManagedScriptIfPresent } from '../agent-hooks/managed-hook-script-refresh'
import { getOrcaManagedCodexHomePath } from './codex-home-paths'
import { getManagedScriptPath } from './codex-hook-definition'
@@ -27,6 +29,11 @@ import {
} from './codex-wsl-hook-install-plan'
import type { CodexTrustEntry } from './config-toml-trust'
/** Lane-scoped so the hooks-on install never joins the hooks-off refresh. */
function launchPrepKey(lane: 'install' | 'refresh', runtimeHomePath: string): string {
return `${lane}\0${normalizeRuntimePathForComparison(runtimeHomePath)}`
}
export class CodexHookService {
async refreshManagedScripts(): Promise<void> {
await refreshManagedScriptIfPresent(getManagedScriptPath(), getManagedScript())
@@ -34,6 +41,7 @@ export class CodexHookService {
private readonly wslReconciliationGeneration = new Map<string, number>()
private readonly wslInstallsInFlight = new Map<string, Promise<AgentHookInstallStatus | null>>()
private readonly launchPrepInFlight = new Map<string, Promise<AgentHookInstallStatus>>()
private supersedeWslReconciliation(runtimeHomePath: string | null | undefined): number {
if (!runtimeHomePath) {
@@ -138,20 +146,11 @@ export class CodexHookService {
return Promise.resolve(null)
}
const targetKey = target?.runtime === 'wsl' ? target.wslDistro?.trim().toLowerCase() : ''
const key = `${getWslReconciliationKey(runtimeHomePath)}\0${targetKey ?? ''}`
const active = this.wslInstallsInFlight.get(key)
if (active) {
return active
}
const install = this.installForRuntimeHome(runtimeHomePath, target)
this.wslInstallsInFlight.set(key, install)
const clear = (): void => {
if (this.wslInstallsInFlight.get(key) === install) {
this.wslInstallsInFlight.delete(key)
}
}
void install.then(clear, clear)
return install
return dedupeInFlightRun(
this.wslInstallsInFlight,
`${getWslReconciliationKey(runtimeHomePath)}\0${targetKey ?? ''}`,
() => this.installForRuntimeHome(runtimeHomePath, target)
)
}
async prepareRuntimeHomeForLaunch(
@@ -164,12 +163,12 @@ export class CodexHookService {
// so hooks/trust must install there rather than the shared mirror.
return (
(await this.installForRuntimeHomeSerialized(runtimeHomePath, target)) ??
(await this.install(runtimeHomePath ?? undefined))
(await this.installForLaunchPrep(runtimeHomePath ?? undefined))
)
}
return (
this.refreshRuntimeUserHooksForRuntimeHome(runtimeHomePath, target) ??
(await this.refreshRuntimeUserHooks(runtimeHomePath ?? undefined))
(await this.refreshRuntimeUserHooksForLaunchPrep(runtimeHomePath ?? undefined))
)
}
@@ -205,6 +204,33 @@ export class CodexHookService {
)
}
/**
* Launch prep runs on every local PTY spawn, and both lanes below serialize
* globally per Codex home, so activating a multi-pane worktree used to pay one
* full hook install per pane back to back (measured ~790ms for 7 panes, and a
* resumed Codex pane prepares twice). Spawns racing for the same home all want
* the same on-disk outcome, so they share one run — the same reason the WSL
* lane above shares `installForRuntimeHome`.
*
* Invalidation: `dedupeInFlightRun` drops the run the moment it settles, so the
* next launch re-reads hooks.json and the user's trust state. Never widen this
* into a time-based cache — the hooks setting, ~/.codex approvals and the
* managed script can all change between spawns, and only a fresh run sees them.
*/
installForLaunchPrep(runtimeHomePath?: string): Promise<AgentHookInstallStatus> {
const homePath = runtimeHomePath ?? getOrcaManagedCodexHomePath()
return dedupeInFlightRun(this.launchPrepInFlight, launchPrepKey('install', homePath), () =>
this.install(homePath)
)
}
refreshRuntimeUserHooksForLaunchPrep(runtimeHomePath?: string): Promise<AgentHookInstallStatus> {
const homePath = runtimeHomePath ?? getOrcaManagedCodexHomePath()
return dedupeInFlightRun(this.launchPrepInFlight, launchPrepKey('refresh', homePath), () =>
this.refreshRuntimeUserHooks(homePath)
)
}
private installExclusively(runtimeHomePath: string): Promise<AgentHookInstallStatus> {
return installCodexHooksExclusively(runtimeHomePath, (recentGrantEntries, homePath) =>
this.getStatusAfterInstall(recentGrantEntries, homePath)
@@ -0,0 +1,108 @@
import {
boundPayload,
digestPayload
} from '../native-chat/agent-session-journal/journal-payload-bounds'
export const CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES = 256
export const CODEX_JOURNAL_PROMPT_OPTION_ID_MAX_BYTES = 1024
export const CODEX_PROMPT_MAX_QUESTIONS = 64
export const CODEX_PROMPT_MAX_QUESTION_BYTES = 32 * 1024
export const CODEX_PROMPT_MAX_OPTIONS = 256
export const CODEX_PROMPT_MAX_OPTION_BYTES = 64 * 1024
export const CODEX_PROMPT_MAX_ANSWER_BYTES = 64 * 1024
export const MAX_CODEX_PROMPT_REGISTRY_ENTRIES = 128
export const MAX_CODEX_PROMPT_JOURNAL_BINDINGS = 256
export const MAX_CODEX_PROMPT_REGISTRY_BYTES = 4 * 1024 * 1024
export function codexJournalPromptIdPart(value: string): string {
if (Buffer.byteLength(value, 'utf8') <= CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES) {
return value
}
const suffix = `#${digestPayload(value).slice(0, 32)}`
const bounded = boundPayload(value, {
inlineHeadBytes: CODEX_JOURNAL_PROMPT_ID_COMPONENT_MAX_BYTES - suffix.length,
maxSessionBytes: Number.MAX_SAFE_INTEGER,
maxAppendsPerWindow: Number.MAX_SAFE_INTEGER,
appendWindowMs: Number.MAX_SAFE_INTEGER
})
return `${bounded.head}${suffix}`
}
export function encodeCodexJournalQuestionOptionId(questionId: string, answer: string): string {
const exact = `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}`
if (Buffer.byteLength(exact, 'utf8') <= CODEX_JOURNAL_PROMPT_OPTION_ID_MAX_BYTES) {
return exact
}
const bounded = `${encodeURIComponent(codexJournalPromptIdPart(questionId))}:${encodeURIComponent(codexJournalPromptIdPart(answer))}`
if (Buffer.byteLength(bounded, 'utf8') <= CODEX_JOURNAL_PROMPT_OPTION_ID_MAX_BYTES) {
return bounded
}
return `#${digestPayload(questionId).slice(0, 32)}:#${digestPayload(answer).slice(0, 32)}`
}
export function readQuestionIds(params: unknown): string[] | null {
const questions = (params as { questions?: unknown } | null)?.questions
if (!Array.isArray(questions)) {
return []
}
const ids: string[] = []
let bytes = 0
for (const question of questions) {
const id = (question as { id?: unknown })?.id
if (typeof id !== 'string' || id.length === 0) {
continue
}
if (ids.length >= CODEX_PROMPT_MAX_QUESTIONS) {
return null
}
bytes += Buffer.byteLength(id, 'utf8')
if (bytes > CODEX_PROMPT_MAX_QUESTION_BYTES) {
return null
}
ids.push(id)
}
return ids
}
export function readQuestionOptionAnswers(
params: unknown
): Map<string, { questionId: string; answer: string }> | null {
const questions = (params as { questions?: unknown } | null)?.questions
const answers = new Map<string, { questionId: string; answer: string }>()
if (!Array.isArray(questions)) {
return answers
}
let optionCount = 0
let optionBytes = 0
for (const entry of questions) {
const question = typeof entry === 'object' && entry !== null ? entry : {}
const questionId = (question as { id?: unknown }).id
const options = (question as { options?: unknown }).options
if (typeof questionId !== 'string' || !Array.isArray(options)) {
continue
}
for (const option of options) {
const record = typeof option === 'object' && option !== null ? option : {}
const label = (record as { label?: unknown }).label
if (
typeof label !== 'string' ||
label.length === 0 ||
(record as { isOther?: unknown }).isOther === true
) {
continue
}
if (++optionCount > CODEX_PROMPT_MAX_OPTIONS) {
return null
}
optionBytes += Buffer.byteLength(questionId, 'utf8') + Buffer.byteLength(label, 'utf8')
if (optionBytes > CODEX_PROMPT_MAX_OPTION_BYTES) {
return null
}
answers.set(encodeCodexJournalQuestionOptionId(questionId, label), {
questionId,
answer: label
})
}
}
return answers
}
@@ -78,7 +78,7 @@ describe('Codex blocking server request dispositions', () => {
)
})
it('cancels a malformed interactive request instead of using method-not-found', () => {
it('refuses a malformed interactive request instead of inventing an answer', () => {
const { registry, connection } = harness()
disposeCodexServerRequest(registry, connection, {
@@ -87,7 +87,12 @@ describe('Codex blocking server request dispositions', () => {
params: {}
})
expect(connection.respond).toHaveBeenCalledWith(4, { decision: 'cancel' })
expect(connection.respond).not.toHaveBeenCalled()
expect(connection.respondWithError).toHaveBeenCalledWith(
4,
-32001,
'Orca could not model item/commandExecution/requestApproval as a durable prompt'
)
})
it('enumerates every server request in the negotiated stable schema', () => {
@@ -51,10 +51,12 @@ export function disposeCodexServerRequest(
switch (request.method) {
case CODEX_COMMAND_APPROVAL_METHOD:
case CODEX_FILE_CHANGE_APPROVAL_METHOD:
connection.respond(request.id, { decision: 'cancel' })
break
case CODEX_USER_INPUT_METHOD:
connection.respond(request.id, { answers: {} })
connection.respondWithError(
request.id,
-32001,
`Orca could not model ${request.method} as a durable prompt`
)
break
case CODEX_MCP_ELICITATION_METHOD:
connection.respond(request.id, { action: 'decline', content: null, _meta: null })
@@ -8,26 +8,50 @@
import type { CodexAppServerConnection } from './codex-app-server-connection'
import { CodexPromptRegistry } from './codex-structured-prompt-replies'
/** Pre-publication buffering is bounded so a provider cannot pin closures. */
export const MAX_CODEX_ACQUISITION_BUFFER_OPERATIONS = 1024
export const MAX_CODEX_ACQUISITION_BUFFER_BYTES = 4 * 1024 * 1024
export class CodexAcquisitionWindow {
readonly prompts = new CodexPromptRegistry()
/** Null until the spawn resolves; the handshake can already emit events. */
connection: CodexAppServerConnection | null = null
private readonly buffered: (() => void)[] = []
private retainedBytes = 0
private open = true
private overflowed = false
get isOverflowed(): boolean {
return this.overflowed
}
/** Returns false once the session is published, which is the caller's cue to
* deliver live rather than buffer. */
buffer(event: () => void): boolean {
buffer(event: () => void, retainedBytes = 256): boolean {
if (!this.open) {
return false
}
const bytes = Number.isFinite(retainedBytes) && retainedBytes > 0 ? Math.ceil(retainedBytes) : 1
if (
this.buffered.length >= MAX_CODEX_ACQUISITION_BUFFER_OPERATIONS ||
this.retainedBytes + bytes > MAX_CODEX_ACQUISITION_BUFFER_BYTES
) {
// Refuse the acquisition rather than dropping an event and continuing.
this.overflowed = true
this.open = false
this.buffered.length = 0
this.retainedBytes = 0
return false
}
this.buffered.push(event)
this.retainedBytes += bytes
return true
}
/** Closes the window and hands back what arrived while it was open, in order. */
drain(): (() => void)[] {
this.open = false
this.retainedBytes = 0
return this.buffered.splice(0)
}
}
@@ -0,0 +1,42 @@
export const MAX_CODEX_ITEM_STREAM_STATES = 256
export const MAX_CODEX_ITEM_STREAM_PENDING_PATCHES = 128
export const MAX_CODEX_ITEM_STREAM_RETAINED_BYTES = 32 * 1024 * 1024
export const MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES = 8 * 1024 * 1024
export const MAX_CODEX_ITEM_STREAM_ITEM_BYTES = 64 * 1024
export function codexStructuredItemKey(threadId: string, itemId: string): string {
const key = `${encodeURIComponent(threadId)}:${encodeURIComponent(itemId)}`
if (Buffer.byteLength(key, 'utf8') <= 1024) {
return key
}
let hash = 2166136261
for (const byte of Buffer.from(key, 'utf8')) {
hash ^= byte
hash = Math.imul(hash, 16777619)
}
return `${key.slice(0, 960)}:${(hash >>> 0).toString(16)}`
}
export function pendingPatchBytes(pending: {
body: unknown
blobs: readonly { payload: string }[]
}): number {
return (
Buffer.byteLength(JSON.stringify(pending.body), 'utf8') +
pending.blobs.reduce((total, blob) => total + Buffer.byteLength(blob.payload, 'utf8'), 0)
)
}
export function boundStreamItem(item: Record<string, unknown>): Record<string, unknown> {
if (Buffer.byteLength(JSON.stringify(item), 'utf8') <= MAX_CODEX_ITEM_STREAM_ITEM_BYTES) {
return item
}
return {
type: item.type,
id: item.id,
...(typeof item.command === 'string' ? { command: item.command.slice(0, 4096) } : {}),
...(typeof item.cwd === 'string' ? { cwd: item.cwd.slice(0, 4096) } : {}),
...(typeof item.status === 'string' ? { status: item.status } : {}),
...(typeof item.exitCode === 'number' ? { exitCode: item.exitCode } : {})
}
}
@@ -0,0 +1,53 @@
import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types'
import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer'
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
import type { codexJournalItem, CodexThreadItem } from './codex-structured-item-translation'
export type CodexItemStreamDeps = {
sink: StructuredAgentSessionEventSink
identityFor: (
threadId: string,
params: unknown,
item: CodexThreadItem
) => AgentJournalItemIdentity
coalesceMs?: number
maxRetainedBytes?: number
maxTotalRetainedBytes?: number
schedule?: AgentSessionDeltaCoalescerDeps['schedule']
}
export type CodexItemStreamState = {
identity: AgentJournalItemIdentity
item: CodexThreadItem
}
export type CodexPendingItemPatch = {
identity: AgentJournalItemIdentity
body: NonNullable<ReturnType<typeof codexJournalItem>['body']>
blobs: ReturnType<typeof codexJournalItem>['blobs']
}
export type CodexStructuredItemStreamAdmission =
| { accepted: true }
| { accepted: false; reason: 'backpressure' | 'failed' | 'closed' }
export type CodexStructuredItemStreamHandleResult = {
handled: boolean
admission: CodexStructuredItemStreamAdmission
}
export type CodexStructuredItemStreams = {
track: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => void
handle: (
threadId: string,
method: string,
params: unknown
) => CodexStructuredItemStreamHandleResult
forget: (threadId: string, itemId: string) => void
flush: () => boolean
dispose: () => void
snapshot: (
threadId: string,
itemId: string
) => { text: string; observedBytes: number; truncated: boolean } | null
}
@@ -0,0 +1,42 @@
import { MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES } from './codex-structured-item-stream-bounds'
export const CODEX_ITEM_STREAM_TYPES = {
'item/agentMessage/delta': 'agentMessage',
'item/plan/delta': 'plan',
'item/commandExecution/outputDelta': 'commandExecution',
'item/fileChange/outputDelta': 'fileChange',
'item/reasoning/summaryTextDelta': 'reasoning',
'item/reasoning/textDelta': 'reasoning'
} as const
export const PATCH_UPDATED_METHOD = 'item/fileChange/patchUpdated'
export const REASONING_PART_METHOD = 'item/reasoning/summaryPartAdded'
export const TERMINAL_INTERACTION_METHOD = 'item/commandExecution/terminalInteraction'
export function readCodexItemStreamRecord(value: unknown): Record<string, unknown> {
return typeof value === 'object' && value !== null ? (value as Record<string, unknown>) : {}
}
export function readCodexItemStreamString(
source: Record<string, unknown>,
key: string
): string | null {
const value = source[key]
return typeof value === 'string' && value.length > 0 ? value : null
}
export function codexPatchChangeBytes(changes: readonly unknown[]): number {
let total = 0
for (const change of changes) {
const record = readCodexItemStreamRecord(change)
const path = readCodexItemStreamString(record, 'path')
const diff = readCodexItemStreamString(record, 'diff')
if (path && diff) {
total += Buffer.byteLength(path, 'utf8') + Buffer.byteLength(diff, 'utf8') + 1
if (total > MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES) {
return total
}
}
}
return total
}
+217 -92
View File
@@ -1,96 +1,142 @@
import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types'
import {
createAgentSessionDeltaCoalescer,
type AgentSessionDeltaCoalescerDeps
} from '../native-chat/agent-session-wire/agent-session-delta-coalescer'
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key'
import { createAgentSessionDeltaCoalescer } from '../native-chat/agent-session-wire/agent-session-delta-coalescer'
import {
codexJournalItem,
codexStreamingJournalItem,
type CodexThreadItem
} from './codex-structured-item-translation'
const CODEX_ITEM_STREAM_TYPES = {
'item/agentMessage/delta': 'agentMessage',
'item/plan/delta': 'plan',
'item/commandExecution/outputDelta': 'commandExecution',
'item/fileChange/outputDelta': 'fileChange',
'item/reasoning/summaryTextDelta': 'reasoning',
'item/reasoning/textDelta': 'reasoning'
} as const
const PATCH_UPDATED_METHOD = 'item/fileChange/patchUpdated'
const REASONING_PART_METHOD = 'item/reasoning/summaryPartAdded'
const TERMINAL_INTERACTION_METHOD = 'item/commandExecution/terminalInteraction'
type CodexItemStreamDeps = {
sink: StructuredAgentSessionEventSink
identityFor: (
threadId: string,
params: unknown,
item: CodexThreadItem
) => AgentJournalItemIdentity
coalesceMs?: number
schedule?: AgentSessionDeltaCoalescerDeps['schedule']
}
type StreamState = { identity: AgentJournalItemIdentity; item: CodexThreadItem }
export type CodexStructuredItemStreams = {
track: (threadId: string, item: CodexThreadItem, identity: AgentJournalItemIdentity) => void
handle: (threadId: string, method: string, params: unknown) => boolean
forget: (threadId: string, itemId: string) => void
flush: () => void
dispose: () => void
}
function readRecord(value: unknown): Record<string, unknown> {
return typeof value === 'object' && value !== null ? (value as Record<string, unknown>) : {}
}
function readString(source: Record<string, unknown>, key: string): string | null {
const value = source[key]
return typeof value === 'string' && value.length > 0 ? value : null
}
export function codexStructuredItemKey(threadId: string, itemId: string): string {
return `${encodeURIComponent(threadId)}:${encodeURIComponent(itemId)}`
}
import {
codexStructuredItemKey,
MAX_CODEX_ITEM_STREAM_PENDING_PATCHES,
MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES,
MAX_CODEX_ITEM_STREAM_RETAINED_BYTES,
MAX_CODEX_ITEM_STREAM_STATES,
boundStreamItem,
pendingPatchBytes
} from './codex-structured-item-stream-bounds'
import {
CODEX_ITEM_STREAM_TYPES,
codexPatchChangeBytes,
PATCH_UPDATED_METHOD,
readCodexItemStreamRecord,
readCodexItemStreamString,
REASONING_PART_METHOD,
TERMINAL_INTERACTION_METHOD
} from './codex-structured-item-stream-events'
import type {
CodexItemStreamDeps,
CodexItemStreamState,
CodexPendingItemPatch,
CodexStructuredItemStreamAdmission,
CodexStructuredItemStreams
} from './codex-structured-item-stream-contracts'
export type {
CodexStructuredItemStreamAdmission,
CodexStructuredItemStreamHandleResult,
CodexStructuredItemStreams
} from './codex-structured-item-stream-contracts'
export { codexStructuredItemKey } from './codex-structured-item-stream-bounds'
export {
MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES,
MAX_CODEX_ITEM_STREAM_RETAINED_BYTES,
MAX_CODEX_ITEM_STREAM_PENDING_PATCHES,
MAX_CODEX_ITEM_STREAM_STATES
} from './codex-structured-item-stream-bounds'
/** Delta-only item ids are provider input; retain only a deterministic recent window. */
export function createCodexStructuredItemStreams(
deps: CodexItemStreamDeps
): CodexStructuredItemStreams {
const states = new Map<string, StreamState>()
const latestText = new Map<string, string>()
const states = new Map<string, CodexItemStreamState>()
const checkpointLengths = new Map<string, number>()
// Patch updates are authoritative item snapshots. Keep the latest rejected
// snapshot until the journal admits it; unlike streamed deltas, there is no
// coalescer timer to retry these events for us.
const pendingPatches = new Map<string, CodexPendingItemPatch>()
let retainedPatchBytes = 0
const append = (state: StreamState, text: string): void => {
const translated = codexStreamingJournalItem(state.item, text)
if (!translated.body) {
return
const forgetState = (key: string): void => {
coalescer.forget(key)
states.delete(key)
checkpointLengths.delete(key)
const pending = pendingPatches.get(key)
if (pending) {
retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending))
pendingPatches.delete(key)
}
deps.sink.appendItem(state.identity, translated.body, translated.blobs)
deps.sink.publish()
}
const persist = (key: string, text: string, force: boolean): void => {
latestText.set(key, text)
const trimStates = (): void => {
while (states.size > MAX_CODEX_ITEM_STREAM_STATES) {
const oldest = states.keys().next().value
if (typeof oldest !== 'string') {
break
}
const pending = coalescer.snapshot(oldest)
if (pending && pending.text.length > 0 && !persist(oldest, pending.text, true)) {
// Keep the state (and its buffered text) until the sink recovers. A
// bounded map is preferable to silently losing streamed output.
break
}
forgetState(oldest)
}
}
const trimPendingPatches = (): void => {
while (pendingPatches.size > MAX_CODEX_ITEM_STREAM_PENDING_PATCHES) {
const oldest = pendingPatches.keys().next().value
if (typeof oldest !== 'string') {
break
}
const pending = pendingPatches.get(oldest)
if (pending) {
retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending))
}
pendingPatches.delete(oldest)
}
}
const append = (state: CodexItemStreamState, text: string): boolean => {
const translated = codexStreamingJournalItem(state.item, text)
if (!translated.body) {
return true
}
const options = { coalescingKey: `checkpoint:${agentJournalItemKey(state.identity)}` }
const admission = deps.sink.tryAppendItem
? deps.sink.tryAppendItem(state.identity, translated.body, translated.blobs, options)
: (deps.sink.appendItem(state.identity, translated.body, translated.blobs, options),
{ accepted: true as const })
if (!admission.accepted) {
return false
}
const published = deps.sink.tryPublish
? deps.sink.tryPublish()
: (deps.sink.publish(), { accepted: true as const })
return published.accepted
}
const persist = (key: string, text: string, force: boolean): boolean => {
const checkpointLength = checkpointLengths.get(key) ?? 0
const nextLength = Math.max(checkpointLength + 32, Math.ceil(checkpointLength * 1.125))
if (!force && checkpointLength > 0 && text.length < nextLength) {
return
return true
}
checkpointLengths.set(key, text.length)
const state = states.get(key)
if (state) {
append(state, text)
if (state && append(state, text)) {
checkpointLengths.set(key, text.length)
return true
}
return false
}
const coalescer = createAgentSessionDeltaCoalescer({
windowMs: deps.coalesceMs,
maxRetainedBytes: deps.maxRetainedBytes,
maxTotalRetainedBytes: deps.maxTotalRetainedBytes,
schedule: deps.schedule,
emit: (key, text) => persist(key, text, false)
emit: (key, text) => {
return persist(key, text, false)
}
})
const ensureState = (
@@ -98,7 +144,7 @@ export function createCodexStructuredItemStreams(
itemId: string,
type: string,
params: unknown
): StreamState => {
): CodexItemStreamState => {
const key = codexStructuredItemKey(threadId, itemId)
const existing = states.get(key)
if (existing) {
@@ -107,70 +153,149 @@ export function createCodexStructuredItemStreams(
const item = { type, id: itemId }
const state = { item, identity: deps.identityFor(threadId, params, item) }
states.set(key, state)
trimStates()
return state
}
const flush = (): void => {
coalescer.flushAll()
for (const [key, text] of latestText) {
if (checkpointLengths.get(key) !== text.length) {
persist(key, text, true)
const flush = (): boolean => {
let flushed = coalescer.flushAll()
for (const key of states.keys()) {
const snapshot = coalescer.snapshot(key)
if (snapshot && checkpointLengths.get(key) !== snapshot.text.length) {
flushed = persist(key, snapshot.text, true) && flushed
}
}
for (const [key, pending] of pendingPatches) {
const admission = deps.sink.tryAppendItem
? deps.sink.tryAppendItem(pending.identity, pending.body, pending.blobs)
: (deps.sink.appendItem(pending.identity, pending.body, pending.blobs),
{ accepted: true as const })
if (!admission.accepted) {
flushed = false
continue
}
const published = deps.sink.tryPublish
? deps.sink.tryPublish()
: (deps.sink.publish(), { accepted: true as const })
if (!published.accepted) {
flushed = false
continue
}
retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending))
pendingPatches.delete(key)
}
return flushed
}
const flushPatch = (key: string): CodexStructuredItemStreamAdmission => {
const pending = pendingPatches.get(key)
if (!pending) {
return { accepted: true }
}
const admission = deps.sink.tryAppendItem
? deps.sink.tryAppendItem(pending.identity, pending.body, pending.blobs)
: (deps.sink.appendItem(pending.identity, pending.body, pending.blobs),
{ accepted: true as const })
if (!admission.accepted) {
return admission
}
const published = deps.sink.tryPublish
? deps.sink.tryPublish()
: (deps.sink.publish(), { accepted: true as const })
if (!published.accepted) {
return published
}
retainedPatchBytes = Math.max(0, retainedPatchBytes - pendingPatchBytes(pending))
pendingPatches.delete(key)
return { accepted: true }
}
return {
track: (threadId, item, identity) => {
states.set(codexStructuredItemKey(threadId, item.id), { item, identity })
const key = codexStructuredItemKey(threadId, item.id)
states.delete(key)
states.set(key, { item: boundStreamItem(item) as CodexThreadItem, identity })
trimStates()
},
handle: (threadId, method, params) => {
const paramsRecord = readRecord(params)
const itemId = readString(paramsRecord, 'itemId')
const paramsRecord = readCodexItemStreamRecord(params)
const itemId = readCodexItemStreamString(paramsRecord, 'itemId')
if (method === PATCH_UPDATED_METHOD) {
if (!itemId || !Array.isArray(paramsRecord.changes)) {
return true
return { handled: true, admission: { accepted: true } }
}
if (
codexPatchChangeBytes(paramsRecord.changes) > MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES
) {
return { handled: true, admission: { accepted: false, reason: 'backpressure' } }
}
const key = codexStructuredItemKey(threadId, itemId)
coalescer.flush(key)
const streamFlushed = coalescer.flush(key)
const state = ensureState(threadId, itemId, 'fileChange', params)
state.item = { ...state.item, changes: paramsRecord.changes }
const translated = codexJournalItem(state.item)
if (translated.body) {
deps.sink.appendItem(state.identity, translated.body, translated.blobs)
deps.sink.publish()
const nextPending: CodexPendingItemPatch = {
identity: state.identity,
body: translated.body,
blobs: translated.blobs
}
const previous = pendingPatches.get(key)
const previousBytes = previous ? pendingPatchBytes(previous) : 0
const nextBytes = pendingPatchBytes(nextPending)
if (
nextBytes > MAX_CODEX_ITEM_STREAM_PENDING_PATCH_BYTES ||
retainedPatchBytes - previousBytes + nextBytes > MAX_CODEX_ITEM_STREAM_RETAINED_BYTES
) {
return { handled: true, admission: { accepted: false, reason: 'backpressure' } }
}
retainedPatchBytes = Math.max(0, retainedPatchBytes - previousBytes) + nextBytes
pendingPatches.set(key, nextPending)
trimPendingPatches()
if (streamFlushed) {
const admission = flushPatch(key)
if (!admission.accepted) {
return { handled: true, admission }
}
}
}
return true
return { handled: true, admission: { accepted: true } }
}
if (method === TERMINAL_INTERACTION_METHOD) {
return true
return { handled: true, admission: { accepted: true } }
}
const type = CODEX_ITEM_STREAM_TYPES[method as keyof typeof CODEX_ITEM_STREAM_TYPES]
if (!type && method !== REASONING_PART_METHOD) {
return false
return { handled: false, admission: { accepted: true } }
}
if (!itemId) {
return true
return { handled: true, admission: { accepted: true } }
}
const state = ensureState(threadId, itemId, type ?? 'reasoning', params)
const delta = method === REASONING_PART_METHOD ? '\n' : paramsRecord.delta
if (typeof delta === 'string') {
coalescer.append(codexStructuredItemKey(threadId, state.item.id), delta)
const accepted = coalescer.append(codexStructuredItemKey(threadId, state.item.id), delta)
if (!accepted) {
return { handled: true, admission: { accepted: false, reason: 'backpressure' } }
}
}
return true
return { handled: true, admission: { accepted: true } }
},
forget: (threadId, itemId) => {
const key = codexStructuredItemKey(threadId, itemId)
coalescer.forget(key)
states.delete(key)
latestText.delete(key)
checkpointLengths.delete(key)
forgetState(key)
},
flush,
dispose: () => {
coalescer.dispose()
states.clear()
latestText.clear()
checkpointLengths.clear()
pendingPatches.clear()
retainedPatchBytes = 0
},
snapshot: (threadId, itemId) => {
const key = codexStructuredItemKey(threadId, itemId)
return coalescer.snapshot(key)
}
}
}

Some files were not shown because too many files have changed in this diff Show More