Merge origin/main into OrcaWin/win-edr-process-table-flags

This commit is contained in:
Orca Worker
2026-09-05 21:19:57 -07:00
222 changed files with 13755 additions and 1061 deletions
+8 -1
View File
@@ -8,7 +8,14 @@
/src/cli/bundled-skill-guides.ts text eol=lf
# Bundled plugin trees are byte-hashed; CRLF checkout would break the pinned hash.
/resources/plugins/** text eol=lf
# pnpm hashes every patch byte-for-byte, so a CRLF checkout breaks the install.
# Pin the bytes so a patch reads and diffs identically on every host. It is NOT
# what makes the hash right: pnpm hashes a patch LF-normalized, so a CRLF checkout
# cannot change it. Believing otherwise put a hand-computed raw digest in the
# lockfile twice and broke every install (#17886).
# These files are stored LF, which is not always the encoding they were written
# against -- @vscode/windows-process-tree ships CRLF sources -- so any code that
# runs `git apply` on one must force `-c core.autocrlf=input` rather than trust
# the host's setting. See config/scripts/windows-process-tree-gyp-rebuild.mjs.
/config/patches/*.patch -text
# The xterm bundle hunks also make a diff nobody can read; review the hand-written
# source patch under xterm-src/ instead. The sibling patches stay diffable.
@@ -0,0 +1,26 @@
#!/usr/bin/env bash
set -euo pipefail
openbox --sm-disable > /tmp/orca-e2e-window-manager.log 2>&1 &
wm_pid=$!
cleanup() {
kill "$wm_pid" 2>/dev/null || true
wait "$wm_pid" 2>/dev/null || true
}
trap cleanup EXIT
ready=false
for attempt in {1..100}; do
if xprop -root _NET_SUPPORTING_WM_CHECK 2>/dev/null | rg -q 'window id # 0x[1-9a-fA-F]'; then
ready=true
break
fi
if ! kill -0 "$wm_pid" 2>/dev/null; then
cat /tmp/orca-e2e-window-manager.log
exit 1
fi
sleep 0.1
done
if [ "$ready" != true ]; then
echo 'Window manager did not acquire the Xvfb root window' >&2
exit 1
fi
"$@"
+3
View File
@@ -160,6 +160,9 @@ jobs:
- name: Checkout the requested ref
uses: actions/checkout@v6
env:
# Full-history checkout must also preserve case-twin branch and tag names.
GIT_DEFAULT_REF_FORMAT: reftable
with:
# Why an input at all rather than just github.ref: the whole point is to
# build code that has not landed, and the workflow definition itself
+5 -4
View File
@@ -25,9 +25,10 @@ defaults:
working-directory: cloud
jobs:
# Public-repository hosted runners preserve Blacksmith allowance for macOS.
security:
name: Secret scan
runs-on: blacksmith-2vcpu-ubuntu-2204
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
with:
@@ -53,7 +54,7 @@ jobs:
# Compiles the workspace. No Postgres service: nothing here reaches a
# database, and the service container costs ~13s of startup.
build:
runs-on: blacksmith-4vcpu-ubuntu-2204
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
@@ -73,7 +74,7 @@ jobs:
# package it needs through the relay pretest hook, so it does not depend on
# `pnpm build` having run.
test:
runs-on: blacksmith-4vcpu-ubuntu-2204
runs-on: ubuntu-22.04
services:
postgres:
image: postgres:16-alpine
@@ -107,7 +108,7 @@ jobs:
# Fork pull requests reach this job, so it never configures a backend, never plans, and never
# holds a credential. Only the relay root ships here; foundation and apps stay private.
terraform:
runs-on: blacksmith-2vcpu-ubuntu-2204
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
+12 -8
View File
@@ -27,6 +27,10 @@ on:
description: Ref to check out (defaults to the workflow ref)
required: false
type: string
test_files:
description: JSON array of specs to run; empty runs the full suite
required: false
type: string
schedule:
# Why: GitHub cron uses UTC; these slots map to 10am and 3pm
# America/Phoenix for the default-branch E2E run.
@@ -146,7 +150,7 @@ jobs:
# Native cache misses need the compiler, Electron needs Xvfb, and paired
# Quick Open needs ripgrep. Install them in one apt transaction per shard.
- name: Install native build and headless UI tools
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk python3 ripgrep xvfb zsh openbox x11-utils
- uses: ./.github/actions/install-node-dependencies
with:
@@ -167,7 +171,7 @@ jobs:
# ORCA_E2E_FORWARD_APP_LOGS keeps startup failures visible when Electron
# launches but never creates a BrowserWindow.
- name: Run E2E tests (${{ matrix.shard_name }})
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }}
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --shard=${{ matrix.shard }}
# Why: Playwright retains traces/screenshots only on failure. Uploading
# them as an artifact makes post-mortem debugging on CI possible without
@@ -201,7 +205,7 @@ jobs:
# unbounded inventory fallback; the paired fixture exercises that real boundary.
# Why openssh-client: the Docker-SSH fixture shells out to ssh/ssh-keygen, and this
# lane now receives those specs from pr.yml's SSH source mapping.
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils
- uses: ./.github/actions/install-node-dependencies
with:
@@ -241,7 +245,7 @@ jobs:
if grep -l '@headful' "${TEST_FILES[@]}" >/dev/null; then
E2E_PROJECT_ARGS+=(--project=electron-headful)
fi
xvfb-run --auto-servernum env "${E2E_ENV[@]}" \
xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env "${E2E_ENV[@]}" \
pnpm run test:e2e "${TEST_FILES[@]}" --workers=1 "${E2E_PROJECT_ARGS[@]}"
- name: Upload Playwright traces
@@ -278,7 +282,7 @@ jobs:
ref: ${{ inputs.ref || github.ref }}
- name: Install native build and headless UI tools
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 xvfb zsh
run: sudo apt-get update && sudo apt-get install -y build-essential fonts-noto-cjk openssh-client python3 ripgrep xvfb zsh openbox x11-utils
- uses: ./.github/actions/install-node-dependencies
with:
@@ -293,7 +297,7 @@ jobs:
# Why: this is the release-path proof that the deployed Linux relay keeps
# its PTY and explorer live across a real watcher SIGSEGV.
- name: Run Docker SSH watcher isolation E2E
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-watcher-isolation
# Why: Playwright empties test-results/ when it starts, so each step here used to
# destroy the previous step's traces. Only the last lane's failure was ever
@@ -310,7 +314,7 @@ jobs:
# readiness across live SSH, headed paired, and headless serve topologies.
- name: Run Docker SSH terminal parking + startup readiness E2E
if: always()
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking
- name: Keep terminal-parking traces
if: always()
@@ -326,7 +330,7 @@ jobs:
# legible as an SSH-named failure.
- name: Run remaining Docker SSH E2E
if: always()
run: xvfb-run --auto-servernum env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker
- name: Keep remaining-ssh-docker traces
if: always()
+11 -2
View File
@@ -93,12 +93,13 @@ jobs:
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
echo "native_ime_source_changed=$NATIVE_IME_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "Native IME source changed: $NATIVE_IME_SOURCE_CHANGED"
if [ "$TEST_FILES_JSON" != '[]' ]; then
SHOULD_RUN="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --reusable-workflow)"
if [ "$SHOULD_RUN" = true ]; then
echo "should_run=true" >> "$GITHUB_OUTPUT"
echo "Changed E2E specs: $TEST_FILES_JSON"
else
echo "should_run=false" >> "$GITHUB_OUTPUT"
echo "No changed E2E specs"
echo "No specs requiring the reusable E2E workflow"
fi
static_analysis:
@@ -831,10 +832,13 @@ jobs:
node_modules/.pnpm/@vscode+windows-process-tree@*/node_modules/@vscode/windows-process-tree/build
key: native-modules-${{ runner.os }}-${{ steps.deps.outputs.native-cache-scope }}-${{ runner.arch }}-node-node${{ steps.deps.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch') }}
# vitest runs here directly rather than through `pnpm test`, so the addon
# assertions only hold once install-node-dependencies has rebuilt natives.
- name: Test Windows-specific boundaries
run: >-
pnpm exec vitest run --config config/vitest.config.ts
config/scripts/rebuild-native-deps.test.mjs
config/scripts/rebuild-native-deps-windows-process-tree.test.mjs
src/main/browser/browser-client-page-renderer-lifecycle.electron.test.ts
src/main/browser/browser-route-tcp-egress.electron.test.ts
src/main/browser/browser-route-webrtc-egress.electron.test.ts
@@ -847,6 +851,7 @@ jobs:
src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts
src/main/windows/windows-pty-job.win32.test.ts
src/main/windows/windows-host-job.win32.test.ts
src/main/windows/windows-process-tree-command-line-patch.test.ts
src/main/windows-live-tree-kill.win32.test.ts
src/main/wsl/wsl-runner.test.ts
src/main/wsl/wsl-guest-environment.test.ts
@@ -855,14 +860,18 @@ jobs:
src/main/wsl/wsl-w1-w3-contract.test.ts
src/shared/source-scan/source-tree-scan.test.ts
src/main/cli/wsl-cli-powershell-boundary.test.ts
src/main/computer/desktop-script-runtime-host.win32.test.ts
src/main/cursor/hook-service.test.ts
src/main/orca-profiles/profile-index-store.test.ts
src/main/startup/windows-install-dir-acl-repair.win32.test.ts
src/main/runtime/repo-worktree-admin-fingerprint.test.ts
src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts
src/shared/secure-file-fsync-flags.test.ts
src/shared/secure-path-windows-acl.win32.test.ts
src/main/runtime/unreadable-secret-store-preservation.win32.test.ts
src/main/ipc/pty-codex-account-attribution.test.ts
src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts
src/relay/windows-port-scan.win32.test.ts
# Why the :parallel variant: identical to build:release except the three
# electron-vite targets overlap instead of running back to back. The Linux package
+13 -10
View File
@@ -858,16 +858,17 @@ jobs:
if: runner.os == 'Linux'
run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: Linux terminal golden E2E uses the same native install path as
# release CI, which needs pnpm to bypass its non-executable gyp_main.py.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
@@ -1074,16 +1075,17 @@ jobs:
if: runner.os == 'Linux'
run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: keep the non-blocking evidence lane on the same Linux native
# install path as the blocking golden and release build jobs.
- name: Use external node-gyp to avoid pnpm's bundled copy (Linux only)
@@ -1716,6 +1718,7 @@ jobs:
with:
name: orca-windows-unsigned-${{ needs.cut.outputs.tag }}
path: dist/orca-windows-setup.exe
compression-level: 0
if-no-files-found: error
# Why: SignPath Foundation production certificates require manual review,
@@ -0,0 +1,38 @@
name: Release ref validation
on:
pull_request:
paths:
- '.github/workflows/adhoc-mac-build.yml'
- '.github/workflows/dev-channel-win-build.yml'
- '.github/workflows/release-ref-validation.yml'
- 'config/scripts/workflow-ref-reachability.test.mjs'
- 'config/scripts/workflow-ref-mirror-case-safety.test.mjs'
workflow_dispatch:
permissions:
contents: read
concurrency:
group: release-ref-validation-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
jobs:
validate:
strategy:
fail-fast: false
matrix:
os: [macos-15, windows-2022]
runs-on: ${{ matrix.os }}
timeout-minutes: 10
steps:
- uses: actions/checkout@v6
with:
persist-credentials: false
- uses: ./.github/actions/install-node-dependencies
- name: Verify case-twin refs and release trust boundary
run: >-
pnpm exec vitest run --config config/vitest.config.ts
config/scripts/workflow-ref-reachability.test.mjs
config/scripts/workflow-ref-mirror-case-safety.test.mjs
config/scripts/dev-channel-windows-workflow-contract.test.mjs
@@ -45,7 +45,9 @@ jobs:
steps:
- uses: actions/checkout@v6
with:
# Historical skill snapshots need tags, but only their blobs are read.
fetch-depth: 0
filter: blob:none
persist-credentials: false
- uses: actions/setup-node@v6
with:
+2 -16
View File
@@ -38,23 +38,9 @@ jobs:
xfwm4
xvfb
- name: Setup Node.js
uses: actions/setup-node@v6
- uses: ./.github/actions/install-node-dependencies
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Use external node-gyp to avoid pnpm bundled copy
run: |
npm install -g node-gyp@11.5.0
echo "npm_config_node_gyp=$(npm root -g)/node-gyp/bin/node-gyp.js" >> "$GITHUB_ENV"
- name: Install dependencies
run: pnpm install --frozen-lockfile
native-runtime: electron
- name: Build Electron app for E2E
run: pnpm exec electron-vite build --mode e2e
+6 -5
View File
@@ -67,16 +67,17 @@ jobs:
- name: Install native build tools and xvfb
run: sudo apt-get update && sudo apt-get install -y build-essential python3 xvfb zsh
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
- name: Setup pnpm
uses: pnpm/setup@v2
with:
install: false
- name: Setup Node.js
uses: actions/setup-node@v6
with:
node-version-file: package.json
cache: pnpm
# Why: this scheduled/manual workflow uses the same native install path as
# PR and E2E CI, which needs pnpm to bypass its bundled gyp_main.py.
- name: Use external node-gyp to avoid pnpm's bundled copy
@@ -215,6 +215,7 @@ jobs:
with:
name: orca-windows-installer-unsigned-${{ github.run_id }}
path: dist/orca-windows-setup.exe
compression-level: 0
if-no-files-found: error
- name: Submit Windows installer signing request
+1
View File
@@ -110,6 +110,7 @@ docs/**
!docs/reference/macos-press-and-hold.md
!docs/reference/orcad-operations.md
!docs/reference/relay-grace-time-reconfiguration.md
!docs/reference/windows-daemon-host-relocation.md
!docs/reference/windows-edr-posture.md
!docs/reference/windows-process-enumeration.md
!docs/reference/wsl-runner-verification.md
+7
View File
@@ -4,6 +4,12 @@ All UI work — layout, color, typography, spacing, component selection, UX beha
## Electron UI Validation
Always run tests and agent-launched apps in the background with `ORCA_BACKGROUND_LAUNCH=1`.
Never steal monitor focus or reveal test windows: no `show()`, `showInactive()`, `bringToFront()`,
`app.focus()`, or OS activation. Use CDP screenshots of hidden renderers. Keep native-focus and
visible-window tests paused on the user's desktop; run them on an isolated display or CI.
Rebuild modified launch-policy code before running an app; stale build wrappers are not safe.
Use the `$electron` skill and Playwright CDP for rendered Orca UI checks. Do not use computer-use for Orca UI validation.
# Style
@@ -49,6 +55,7 @@ Orca targets macOS, Linux, and Windows. Keep all platform-dependent behavior beh
- **Windows setup scripts**: the setup/issue-command runner is a `.cmd` batch file unless the script starts with a `#!` line — never derive that from the user's terminal-shell preference, and never launch a `.cmd` runner with a bare `cmd.exe /c` from a Git Bash pane (MSYS rewrites the `/c`). See [`docs/reference/windows-setup-shell.md`](./docs/reference/windows-setup-shell.md).
- **Windows child processes**: start them through `runProcess`/`spawnProcess` in `src/shared/child-process/` — never `child_process` directly. It pins `windowsHide`, refuses `shell: true`, and encodes `.cmd`/`.bat` arguments so neither `CommandLineToArgvW` nor `cmd.exe` mangles them. A ratchet test fails on any new direct import.
- **Windows process enumeration**: read the table through `src/main/windows/windows-process-table.ts`, never by forking `powershell.exe`. See [`docs/reference/windows-process-enumeration.md`](./docs/reference/windows-process-enumeration.md).
- **Windows daemon-host relocation**: the terminal daemon runs from a copy of the app runtime under `%LOCALAPPDATA%`, which is what survives an auto-update. Before touching that copy, its exe name, or the NSIS uninstall macro, read [`docs/reference/windows-daemon-host-relocation.md`](./docs/reference/windows-daemon-host-relocation.md).
- **Windows EDR signal**: don't add `-ExecutionPolicy Bypass`, `-EncodedCommand`, `cmd.exe /c` with escaped free text, per-operation interpreter spawning, or runtime `Add-Type` compilation without reading [`docs/reference/windows-edr-posture.md`](./docs/reference/windows-edr-posture.md) first — behavioural EDR scores each of those, and being signed does not clear them.
- **WSL commands**: build argv with `buildWslExecArgs` (always `--exec` — under `--`, `wsl.exe` expands `$name` in every argument and silently rewrites the script), and fence anything whose stdout you parse with `buildWslCapturedLoginShellCommand`, because the interactive login shell prints the distro banner to stdout. See [`docs/reference/wsl-command-execution.md`](./docs/reference/wsl-command-execution.md).
- **Linux native modules**: keep the glibc floor at Ubuntu 20.04 / glibc 2.31. A module compiled from source on a newer runner can reference symbol versions absent on the floor and crash the app on startup. See [`docs/reference/linux-glibc-compatibility.md`](./docs/reference/linux-glibc-compatibility.md); packaging fails if a bundled native binary needs newer glibc.
+35 -9
View File
@@ -49,22 +49,48 @@
; ---------------------------------------------------------------------------
; Clean up the relocated terminal daemon on a REAL uninstall.
;
; Why: the daemon host is deliberately copied to a distinct image name
; (orca-terminal-daemon.exe) under %LOCALAPPDATA%\Orca\daemon-host so that app
; UPDATES cannot kill it — that relocation is what keeps terminals alive across
; updates. The same design means a normal uninstall's process sweep and file
; removal both miss it, leaving an orphaned daemon plus its runtime copy behind.
; Why: the daemon host is deliberately copied OUT of the install dir into
; %LOCALAPPDATA%\Orca\daemon-host so that app UPDATES cannot kill it —
; electron-builder's kill sweep selects processes whose image path is under
; $INSTDIR, and that relocation is what keeps terminals alive across updates.
; The same design means a normal uninstall's process sweep and file removal both
; miss it, leaving an orphaned daemon plus its runtime copy behind.
;
; The ${isUpdated} guard is essential: electron-builder runs this uninstaller as
; part of uninstallOldVersion on EVERY update, and killing the daemon there would
; defeat the whole feature. Only clean up on a genuine uninstall.
;
; The image name and the LOCALAPPDATA folder name must stay in sync with
; DAEMON_HOST_EXE_NAME and LOCAL_HOST_ROOT_NAME in
; src/main/daemon/daemon-host-relocation.ts.
; The LOCALAPPDATA folder name must stay in sync with LOCAL_HOST_ROOT_NAME in
; src/main/daemon/daemon-host-relocation.ts. See
; docs/reference/windows-daemon-host-relocation.md.
!macro customUnInstall
${ifNot} ${isUpdated}
nsExec::Exec 'taskkill /F /IM orca-terminal-daemon.exe'
Push $0
Push $1
Push $2
; The host exe is a verbatim copy of the app exe, so the app's own image name
; reaches it; the second name covers hosts left by builds that renamed the copy.
; Filtered to the current user like upstream's per-user KILL_PROCESS, so an
; elevated machine-wide uninstall cannot reach another logged-on user's session.
; NSIS expands USERNAME itself: routing through cmd.exe only to get %USERNAME%
; would add two interpreter spawns to the uninstall path for nothing.
ReadEnvStr $1 USERNAME
${if} $1 == ""
; Measured: taskkill rejects an empty filter value outright ("The search filter
; cannot be recognized") and kills nothing, so with no USERNAME to scope by,
; kill unfiltered rather than not at all. USERNAME is set in every session an
; uninstaller runs in, so this is a backstop, not the expected path.
StrCpy $2 ""
${else}
StrCpy $2 '/FI "USERNAME eq $1"'
${endIf}
nsExec::Exec 'taskkill /F /IM "${APP_EXECUTABLE_FILENAME}" $2'
Pop $0
nsExec::Exec 'taskkill /F /IM "orca-terminal-daemon.exe" $2'
Pop $0
Pop $2
Pop $1
Pop $0
; Give the OS a moment to release the image lock before removing the tree.
Sleep 500
RMDir /r "$LOCALAPPDATA\Orca\daemon-host"
@@ -27,15 +27,424 @@ index 855bd4b86f0a3c18c7594212c0e42b6e35bc4001..33774e7ae296f0de39dd94156673c9e7
"/guard:cf",
"/sdl",
diff --git a/src/process.cc b/src/process.cc
index 3eea92077c4d1d433119361d5c432881859131e9..1998f4addd4d7e9aba946ea6f7f7a4a5d13291bc 100644
index 3eea92077c4d1d433119361d5c432881859131e9..738775f6fcdfb676054386fe34c0380327ed1863 100644
--- a/src/process.cc
+++ b/src/process.cc
@@ -37,7 +37,7 @@ uint32_t GetRawProcessList(std::vector<ProcessInfo>& process_info,
process_info.push_back(std::move(pinfo));
process_count++;
}
@@ -1,108 +1,112 @@
-/*---------------------------------------------------------------------------------------------
- * Copyright (c) Microsoft Corporation. All rights reserved.
- * Licensed under the MIT License. See License.txt in the project root for license information.
- *--------------------------------------------------------------------------------------------*/
-
-#include "process.h"
-#include "process_commandline.h"
-
-#include <tlhelp32.h>
-#include <psapi.h>
-#include <limits>
-
-uint32_t GetRawProcessList(std::vector<ProcessInfo>& process_info,
- DWORD process_data_flags) {
- // Fetch the PID and PPIDs
- PROCESSENTRY32 process_entry = { 0 };
- DWORD parent_pid = 0;
- uint32_t process_count = 0;
- HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0);
- process_entry.dwSize = sizeof(PROCESSENTRY32);
- if (Process32First(snapshot_handle, &process_entry)) {
- do {
- if (process_entry.th32ProcessID != 0) {
- ProcessInfo pinfo;
- pinfo.pid = process_entry.th32ProcessID;
- pinfo.ppid = process_entry.th32ParentProcessID;
-
- if (MEMORY & process_data_flags) {
- GetProcessMemoryUsage(pinfo);
- }
-
- if (COMMANDLINE & process_data_flags) {
- GetProcessCommandLine(pinfo);
- }
-
- strcpy(pinfo.name, process_entry.szExeFile);
- process_info.push_back(std::move(pinfo));
- process_count++;
- }
- } while (process_count < 1024 && Process32Next(snapshot_handle, &process_entry));
- }
-
- CloseHandle(snapshot_handle);
- return process_count;
-}
-
-void GetProcessMemoryUsage(ProcessInfo& process_info) {
- DWORD pid = process_info.pid;
- HANDLE hProcess;
- PROCESS_MEMORY_COUNTERS pmc;
-
- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid);
-
- if (hProcess == NULL) {
- return;
- }
-
- if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) {
- process_info.memory = (DWORD)pmc.WorkingSetSize;
- }
-
- CloseHandle(hProcess);
-}
-
-// Per documentation, it is not recommended to add or subtract values from the FILETIME
-// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows.
-// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead.
-// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx
-ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) {
- ULARGE_INTEGER kt, ut;
- kt.LowPart = (*kernelTime).dwLowDateTime;
- kt.HighPart = (*kernelTime).dwHighDateTime;
-
- ut.LowPart = (*userTime).dwLowDateTime;
- ut.HighPart = (*userTime).dwHighDateTime;
-
- return kt.QuadPart + ut.QuadPart;
-}
-
-void GetCpuUsage(Cpu& cpu_info, bool first_pass) {
- DWORD pid = cpu_info.pid;
- HANDLE hProcess;
-
- hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid);
-
- if (hProcess == NULL) {
- return;
- }
-
- FILETIME creationTime, exitTime, kernelTime, userTime;
- FILETIME sysIdleTime, sysKernelTime, sysUserTime;
- if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime)
- && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) {
- if (first_pass) {
- cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime);
- cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime);
- } else {
- ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime);
- ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime);
-
- cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime);
- }
- } else {
- cpu_info.cpu = std::numeric_limits<double>::quiet_NaN();
- }
-
- CloseHandle(hProcess);
+/*---------------------------------------------------------------------------------------------
+ * Copyright (c) Microsoft Corporation. All rights reserved.
+ * Licensed under the MIT License. See License.txt in the project root for license information.
+ *--------------------------------------------------------------------------------------------*/
+
+#include "process.h"
+#include "process_commandline.h"
+
+#include <tlhelp32.h>
+#include <psapi.h>
+#include <limits>
+
+uint32_t GetRawProcessList(std::vector<ProcessInfo>& process_info,
+ DWORD process_data_flags) {
+ // Fetch the PID and PPIDs
+ PROCESSENTRY32 process_entry = { 0 };
+ DWORD parent_pid = 0;
+ uint32_t process_count = 0;
+ HANDLE snapshot_handle = CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0);
+ process_entry.dwSize = sizeof(PROCESSENTRY32);
+ if (Process32First(snapshot_handle, &process_entry)) {
+ do {
+ if (process_entry.th32ProcessID != 0) {
+ // Value-initialize: `memory` is otherwise stack garbage when the flag is unset.
+ ProcessInfo pinfo{};
+ pinfo.pid = process_entry.th32ProcessID;
+ pinfo.ppid = process_entry.th32ParentProcessID;
+
+ if (MEMORY & process_data_flags) {
+ GetProcessMemoryUsage(pinfo);
+ }
+
+ if (COMMANDLINE & process_data_flags) {
+ GetProcessCommandLine(pinfo);
+ }
+
+ strcpy(pinfo.name, process_entry.szExeFile);
+ process_info.push_back(std::move(pinfo));
+ process_count++;
+ }
+ } while (Process32Next(snapshot_handle, &process_entry));
}
CloseHandle(snapshot_handle);
+ }
+
+ CloseHandle(snapshot_handle);
+ return process_count;
+}
+
+void GetProcessMemoryUsage(ProcessInfo& process_info) {
+ DWORD pid = process_info.pid;
+ HANDLE hProcess;
+ PROCESS_MEMORY_COUNTERS pmc;
+
+ // PROCESS_VM_READ is never used here -- GetProcessMemoryInfo reads counters the
+ // kernel keeps, not the address space -- and acquiring it is what EDR scores.
+ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid);
+
+ if (hProcess == NULL) {
+ return;
+ }
+
+ if (GetProcessMemoryInfo(hProcess, &pmc, sizeof(pmc))) {
+ process_info.memory = (DWORD)pmc.WorkingSetSize;
+ }
+
+ CloseHandle(hProcess);
+}
+
+// Per documentation, it is not recommended to add or subtract values from the FILETIME
+// structure, or to cast it to ULARGE_INTEGER as this can cause alignment faults on 64-bit Windows.
+// Copy the high and low part to a ULARGE_INTEGER and peform arithmetic on that instead.
+// See https://msdn.microsoft.com/en-us/library/windows/desktop/ms724284(v=vs.85).aspx
+ULONGLONG GetTotalTime(const FILETIME* kernelTime, const FILETIME* userTime) {
+ ULARGE_INTEGER kt, ut;
+ kt.LowPart = (*kernelTime).dwLowDateTime;
+ kt.HighPart = (*kernelTime).dwHighDateTime;
+
+ ut.LowPart = (*userTime).dwLowDateTime;
+ ut.HighPart = (*userTime).dwHighDateTime;
+
+ return kt.QuadPart + ut.QuadPart;
+}
+
+void GetCpuUsage(Cpu& cpu_info, bool first_pass) {
+ DWORD pid = cpu_info.pid;
+ HANDLE hProcess;
+
+ // GetProcessTimes needs no more than PROCESS_QUERY_LIMITED_INFORMATION.
+ hProcess = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid);
+
+ if (hProcess == NULL) {
+ return;
+ }
+
+ FILETIME creationTime, exitTime, kernelTime, userTime;
+ FILETIME sysIdleTime, sysKernelTime, sysUserTime;
+ if (GetProcessTimes(hProcess, &creationTime, &exitTime, &kernelTime, &userTime)
+ && GetSystemTimes(&sysIdleTime, &sysKernelTime, &sysUserTime)) {
+ if (first_pass) {
+ cpu_info.initialProcRunTime = GetTotalTime(&kernelTime, &userTime);
+ cpu_info.initialSystemTime = GetTotalTime(&sysKernelTime, &sysUserTime);
+ } else {
+ ULONGLONG endProcTime = GetTotalTime(&kernelTime, &userTime);
+ ULONGLONG endSysTime = GetTotalTime(&sysKernelTime, &sysUserTime);
+
+ cpu_info.cpu = 100.0 * (endProcTime - cpu_info.initialProcRunTime) / (endSysTime - cpu_info.initialSystemTime);
+ }
+ } else {
+ cpu_info.cpu = std::numeric_limits<double>::quiet_NaN();
+ }
+
+ CloseHandle(hProcess);
}
\ No newline at end of file
diff --git a/src/process_commandline.cc b/src/process_commandline.cc
index ea822b120e8038a4803e34647042f08f4aaf5ca1..25907c0bf542bed6c72b1b462b19bcf3210c3cfd 100644
--- a/src/process_commandline.cc
+++ b/src/process_commandline.cc
@@ -1,67 +1,125 @@
-/*---------------------------------------------------------------------------------------------
- * Copyright (c) Microsoft Corporation. All rights reserved.
- * Licensed under the MIT License. See License.txt in the project root for license information.
- *--------------------------------------------------------------------------------------------*/
-
-#include "process.h"
-#include "process_commandline.h"
-#include <windows.h>
-#include <winternl.h>
-#include <iostream>
-
-bool GetProcessCommandLine(ProcessInfo& process_info) {
- HINSTANCE ntdll = GetModuleHandleW(L"ntdll.dll");
- if (!ntdll) {
- return false;
- }
-
- decltype(NtQueryInformationProcess)* nt_query_information_process =
- reinterpret_cast<decltype(NtQueryInformationProcess)*>(
- GetProcAddress(ntdll, "NtQueryInformationProcess"));
-
- if (!nt_query_information_process) {
- return false;
- }
-
- PROCESS_BASIC_INFORMATION pbi{};
- PEB peb = {NULL};
- RTL_USER_PROCESS_PARAMETERS process_parameters = {NULL};
-
- // Get process handle
- DWORD pid = process_info.pid;
- HANDLE hProcess = OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, FALSE, pid);
- if (hProcess == INVALID_HANDLE_VALUE) {
- return false;
- }
-
- // Get Process Environment Block (PEB)
- NTSTATUS status = nt_query_information_process(hProcess, ProcessBasicInformation, &pbi, sizeof(pbi), nullptr);
- if (NT_SUCCESS(status) && pbi.PebBaseAddress) {
- // Read PEB
- if (ReadProcessMemory(hProcess, pbi.PebBaseAddress, &peb, sizeof(peb), nullptr)) {
- // Read the processs parameters
- if (ReadProcessMemory(hProcess, peb.ProcessParameters, &process_parameters, sizeof(RTL_USER_PROCESS_PARAMETERS), nullptr)) {
- if (process_parameters.CommandLine.Length > 0) {
- std::wstring buffer;
- buffer.resize(process_parameters.CommandLine.Length / sizeof(wchar_t));
- if (ReadProcessMemory(hProcess, process_parameters.CommandLine.Buffer, &buffer[0], process_parameters.CommandLine.Length, nullptr)) {
- int wide_length = static_cast<int>(buffer.length());
- int charcount = WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length,
- NULL, 0, NULL, NULL);
- if (charcount) {
- process_info.commandLine.resize(static_cast<size_t>(charcount));
- WideCharToMultiByte(CP_UTF8, 0, buffer.data(), wide_length,
- &process_info.commandLine[0], charcount,
- NULL, NULL);
- }
- CloseHandle(hProcess);
- return true;
- }
- }
- }
- }
- }
-
- CloseHandle(hProcess);
- return false;
-}
+/*---------------------------------------------------------------------------------------------
+ * Copyright (c) Microsoft Corporation. All rights reserved.
+ * Licensed under the MIT License. See License.txt in the project root for license information.
+ *--------------------------------------------------------------------------------------------*/
+
+#include "process.h"
+#include "process_commandline.h"
+#include <windows.h>
+#include <winternl.h>
+#include <vector>
+
+namespace {
+
+// Windows 8.1 and later hand back a process's command line as a UNICODE_STRING
+// the kernel builds, needing only PROCESS_QUERY_LIMITED_INFORMATION.
+//
+// There is deliberately no PEB fallback. Reading the command line out of the
+// target's address space -- opening it for VM reads and then chaining
+// memory reads across every pid on a timer -- is the credential-dumping
+// primitive this reader exists to not perform, so it is absent from the binary
+// rather than one anomalous NTSTATUS away. Electron's floor is Windows 10, so
+// every OS Orca supports has this class; if a hooked ntdll refuses it anyway,
+// the command line comes back empty, which callers already handle, instead of
+// silently reinstating the primitive on exactly the instrumented machines this
+// reader was written for.
+const ULONG kProcessCommandLineInformation = 60;
+
+const NTSTATUS kStatusInfoLengthMismatch = static_cast<NTSTATUS>(0xC0000004L);
+const NTSTATUS kStatusBufferTooSmall = static_cast<NTSTATUS>(0xC0000023L);
+
+// A command line is a UNICODE_STRING, whose Length is a USHORT, so the kernel
+// can never need more than the header plus 64 KiB. Refusing anything larger
+// keeps a bogus size from throwing bad_alloc out of a scan that has already
+// walked most of the table.
+const ULONG kMaxCommandLineBytes = sizeof(UNICODE_STRING) + 0xFFFF + sizeof(wchar_t);
+
+// winternl.h's PROCESSINFOCLASS does not name class 60 and its enumerator range
+// stops far short of it, so the class travels as a ULONG rather than a cast enum.
+typedef NTSTATUS(NTAPI* NtQueryInformationProcessFn)(HANDLE, ULONG, PVOID, ULONG, PULONG);
+
+// ntdll ships no import library for this entry point; it has to be resolved.
+NtQueryInformationProcessFn ResolveNtQueryInformationProcess() {
+ HMODULE ntdll = GetModuleHandleW(L"ntdll.dll");
+ if (!ntdll) {
+ return nullptr;
+ }
+ return reinterpret_cast<NtQueryInformationProcessFn>(
+ GetProcAddress(ntdll, "NtQueryInformationProcess"));
+}
+
+NtQueryInformationProcessFn NtQueryInformationProcessEntry() {
+ static NtQueryInformationProcessFn entry = ResolveNtQueryInformationProcess();
+ return entry;
+}
+
+bool StoreCommandLineUtf8(ProcessInfo& process_info, const wchar_t* data, size_t wide_length) {
+ if (wide_length == 0) {
+ return false;
+ }
+ int length = static_cast<int>(wide_length);
+ int charcount = WideCharToMultiByte(CP_UTF8, 0, data, length, NULL, 0, NULL, NULL);
+ if (!charcount) {
+ return false;
+ }
+ process_info.commandLine.resize(static_cast<size_t>(charcount));
+ WideCharToMultiByte(CP_UTF8, 0, data, length, &process_info.commandLine[0], charcount, NULL,
+ NULL);
+ return true;
+}
+
+} // namespace
+
+bool GetProcessCommandLine(ProcessInfo& process_info) {
+ NtQueryInformationProcessFn query = NtQueryInformationProcessEntry();
+ if (!query) {
+ return false;
+ }
+
+ HANDLE process = OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, FALSE, process_info.pid);
+ if (process == NULL) {
+ return false;
+ }
+
+ ULONG size = 0;
+ NTSTATUS status = query(process, kProcessCommandLineInformation, nullptr, 0, &size);
+ if (NT_SUCCESS(status)) {
+ // Nothing was written, so there is no command line to read.
+ CloseHandle(process);
+ return false;
+ }
+ if (status != kStatusInfoLengthMismatch && status != kStatusBufferTooSmall) {
+ CloseHandle(process);
+ return false;
+ }
+ if (size < sizeof(UNICODE_STRING) || size > kMaxCommandLineBytes) {
+ CloseHandle(process);
+ return false;
+ }
+
+ std::vector<unsigned char> buffer(size);
+ status = query(process, kProcessCommandLineInformation, &buffer[0], size, &size);
+ CloseHandle(process);
+ if (!NT_SUCCESS(status)) {
+ return false;
+ }
+
+ // Header and characters arrive in one allocation, but treat the header as
+ // untrusted: a hooked ntdll is the case this reader is written for, and an
+ // unchecked Buffer/Length here would be an over-read encoded straight into JS.
+ // Bound against buffer.size(), never `size` -- the second query overwrote it.
+ const UNICODE_STRING* command_line = reinterpret_cast<const UNICODE_STRING*>(&buffer[0]);
+ const unsigned char* begin = &buffer[0];
+ const unsigned char* end = begin + buffer.size();
+ const unsigned char* chars = reinterpret_cast<const unsigned char*>(command_line->Buffer);
+ if (chars == nullptr || chars < begin + sizeof(UNICODE_STRING) || chars > end ||
+ command_line->Length > static_cast<ULONG>(end - chars)) {
+ return false;
+ }
+
+ // True only when a command line was actually stored, so "empty" and "not
+ // recovered" stay the same answer they were before this reader replaced the
+ // PEB read. `src/process.cc` discards the result either way.
+ return StoreCommandLineUtf8(process_info, command_line->Buffer,
+ command_line->Length / sizeof(wchar_t));
+}
@@ -0,0 +1,110 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
import { stripTypeScriptTypes } from 'node:module'
import { performance } from 'node:perf_hooks'
// Run from the worktree root: node config/scripts/benchmark-browser-tunnel-framing.mjs [base-ref]
const path = 'src/shared/browser-network-tunnel-stream-framing.ts'
const baselineRef = process.argv[2] ?? 'HEAD'
const beforeSource = execFileSync('git', ['show', `${baselineRef}:${path}`], {
encoding: 'utf8'
})
const afterSource = readFileSync(path, 'utf8')
const load = (source) =>
import(
`data:text/javascript;base64,${Buffer.from(
stripTypeScriptTypes(source, { mode: 'transform' })
).toString('base64')}`
)
const before = await load(beforeSource)
const after = await load(afterSource)
function measure(module, chunks, payload, repetitions) {
let frameCount = 0
let lastFrame
const onFrame = (frame) => {
frameCount++
lastFrame = frame
}
const onError = (error) => {
throw error
}
const run = () => {
const decoder = new module.BrowserNetworkTunnelStreamFrameDecoder(onFrame, onError)
for (const chunk of chunks) {
decoder.feed(chunk)
}
}
run()
assert.deepEqual(lastFrame, payload)
const samples = []
for (let sample = 0; sample < 5; sample++) {
const start = performance.now()
for (let iteration = 0; iteration < repetitions; iteration++) {
run()
}
samples.push((performance.now() - start) / repetitions)
}
assert.equal(frameCount, 1 + 5 * repetitions)
return samples.sort((a, b) => a - b)[2]
}
function countCopies(module, chunks) {
const originalSet = Uint8Array.prototype.set
const originalSlice = Uint8Array.prototype.slice
let copied = 0
Uint8Array.prototype.set = function (source, offset) {
copied += source.length
return originalSet.call(this, source, offset)
}
Uint8Array.prototype.slice = function (...args) {
const result = originalSlice.apply(this, args)
copied += result.length
return result
}
try {
const decoder = new module.BrowserNetworkTunnelStreamFrameDecoder(
() => {},
(error) => {
throw error
}
)
for (const chunk of chunks) {
decoder.feed(chunk)
}
} finally {
Uint8Array.prototype.set = originalSet
Uint8Array.prototype.slice = originalSlice
}
return copied
}
const rows = []
for (const [payloadBytes, chunkBytes, repetitions] of [
[1, 5, 10000],
[64 * 1024, 65540, 1000],
[64 * 1024, 4096, 100],
[64 * 1024, 256, 25],
[64 * 1024, 16, 5],
[64 * 1024, 1, 1]
]) {
const payload = Uint8Array.from({ length: payloadBytes }, (_, index) => index % 251)
const encoded = before.encodeBrowserNetworkTunnelStreamFrame(payload)
const chunks = []
for (let offset = 0; offset < encoded.length; offset += chunkBytes) {
chunks.push(encoded.subarray(offset, offset + chunkBytes))
}
const beforeMs = measure(before, chunks, payload, repetitions)
const afterMs = measure(after, chunks, payload, repetitions)
rows.push({
payloadBytes,
chunkBytes,
beforeMs: +beforeMs.toFixed(6),
afterMs: +afterMs.toFixed(6),
speedup: +(beforeMs / afterMs).toFixed(2),
beforeCopiedBytes: countCopies(before, chunks),
afterCopiedBytes: countCopies(after, chunks)
})
}
console.log(JSON.stringify({ node: process.version, baselineRef, rows }, null, 2))
@@ -0,0 +1,121 @@
import assert from 'node:assert/strict'
import { createRequire } from 'node:module'
import { existsSync, realpathSync } from 'node:fs'
import { delimiter, join, resolve } from 'node:path'
// Emit each revision with tsc -p config/tsconfig.cli.json --outDir <dir> --composite false --incremental false.
// Run: node config/scripts/benchmark-cli-error-imports.mjs <before-dir> <after-dir>
const [beforeDir, afterDir] = process.argv.slice(2)
assert.ok(beforeDir && afterDir, 'Pass distinct before and after TypeScript output directories.')
assert.notEqual(
realpathSync(beforeDir),
realpathSync(afterDir),
'Do not compare a build to itself.'
)
const entries = {
before: join(resolve(beforeDir), 'cli', 'index.js'),
after: join(resolve(afterDir), 'cli', 'index.js')
}
for (const entry of Object.values(entries)) {
assert.ok(existsSync(entry), `Missing emitted CLI: ${entry}`)
}
const { runProcessSync } = createRequire(import.meta.url)(
join(resolve(afterDir), 'shared', 'child-process', 'run-process.js')
)
const child = String.raw`
const { performance } = require('node:perf_hooks')
const { writeSync } = require('node:fs')
const { createHash } = require('node:crypto')
const { basename } = require('node:path')
let stdout = '', stderr = ''
process.stdout.write = (text) => { stdout += text; return true }
process.stderr.write = (text) => { stderr += text; return true }
const started = performance.now()
const cli = require(process.argv[1])
const importMs = performance.now() - started
cli.main(JSON.parse(process.argv[2])).then(() => {
const totalMs = performance.now() - started
const modules = Object.keys(require.cache)
writeSync(1, JSON.stringify({
importMs, totalMs, modules: modules.length,
featureFormatters: modules.filter((file) => ['browser', 'terminal', 'project', 'automation', 'workspace', 'computer'].some((name) => basename(file) === name + '-format.js')),
stdout: createHash('sha256').update(stdout).digest('hex'),
stderr: createHash('sha256').update(stderr).digest('hex'),
exitCode: process.exitCode || 0
}))
process.exitCode = 0
}).catch((error) => { writeSync(2, String(error)); process.exitCode = 1 })
`
const cases = [
['--help'],
['help', 'terminal', 'read'],
['does-not-exist'],
['computer', 'click', '--does-not-exist'],
['does-not-exist', '--json']
]
const median = (values) => [...values].sort((a, b) => a - b)[Math.floor(values.length / 2)]
const summarize = (samples) => ({
importMs: median(samples.map((sample) => sample.importMs)),
totalMs: median(samples.map((sample) => sample.totalMs)),
modules: samples[0].modules
})
const rows = []
for (const args of cases) {
const samples = { before: [], after: [] }
let expected
for (let run = 0; run < 22; run++) {
for (const variant of run % 2 ? ['after', 'before'] : ['before', 'after']) {
const result = runProcessSync({
program: process.execPath,
args: ['-e', child, entries[variant], JSON.stringify(args)],
timeoutMs: 30_000,
env: {
...process.env,
NODE_PATH: [resolve('node_modules'), process.env.NODE_PATH]
.filter(Boolean)
.join(delimiter)
}
})
assert.equal(result.timedOut, false, 'CLI child timed out.')
assert.equal(result.code, 0, result.stderr)
const sample = JSON.parse(result.stdout)
const output = { stdout: sample.stdout, stderr: sample.stderr, exitCode: sample.exitCode }
expected ??= output
assert.deepEqual(output, expected, `${variant} output changed for ${args.join(' ')}`)
if (variant === 'after') {
assert.deepEqual(
sample.featureFormatters,
[],
'Help and syntax errors must skip feature formatters.'
)
}
if (run >= 2) {
samples[variant].push(sample)
}
}
}
assert.ok(samples.after[0].modules < samples.before[0].modules, 'Expected fewer loaded modules.')
rows.push({
args,
before: summarize(samples.before),
after: summarize(samples.after),
output: expected,
samples
})
}
console.log(
JSON.stringify(
{
node: process.version,
platform: process.platform,
measurement:
'Fresh-process import + main; excludes process creation; warmed filesystem; 2 warmups and 20 samples per variant, alternating order.',
entries,
rows
},
null,
2
)
)
@@ -0,0 +1,128 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { EventEmitter } from 'node:events'
import { readFileSync } from 'node:fs'
import Module from 'node:module'
import { dirname, resolve } from 'node:path'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
// Run from the worktree root: node config/scripts/benchmark-cli-response-framing.mjs <base-ref>
const sourcePath = 'src/cli/runtime/transport.ts'
const baselineRef = process.argv[2]
assert.ok(baselineRef, 'Pass the pre-change transport revision as base-ref.')
const beforeSource = execFileSync('git', ['show', `${baselineRef}:${sourcePath}`], {
encoding: 'utf8'
})
let chunks = []
async function loadTransport(source) {
const built = await build({
stdin: { contents: source, loader: 'ts', resolveDir: dirname(resolve(sourcePath)) },
bundle: true,
platform: 'node',
format: 'cjs',
write: false,
logLevel: 'silent'
})
const module = new Module(resolve(sourcePath))
const originalRequire = module.require.bind(module)
module.require = (name) => {
if (name === 'node:crypto') {
return { randomUUID: () => 'benchmark-request' }
}
if (name !== 'node:net') {
return originalRequire(name)
}
return {
createConnection() {
const socket = new EventEmitter()
socket.setEncoding = () => {}
socket.end = () => {}
socket.destroy = () => {}
socket.write = () => {
for (const chunk of chunks) {
socket.emit('data', chunk)
}
}
queueMicrotask(() => socket.emit('connect'))
return socket
}
}
}
module._compile(built.outputFiles[0].text, resolve(sourcePath))
return module.exports.sendRequest
}
const before = await loadTransport(beforeSource)
const after = await loadTransport(readFileSync(sourcePath, 'utf8'))
const metadata = {
runtimeId: 'benchmark-runtime',
authToken: 'benchmark-token',
transports: [{ kind: 'unix', endpoint: 'injected-socket' }]
}
const run = (sendRequest) => sendRequest(metadata, 'terminal.read', {}, 30000)
async function measure(sendRequest, payloadBytes, repetitions) {
const warmup = await run(sendRequest)
assert.equal(warmup.result.data.length, payloadBytes)
const samples = []
for (let sample = 0; sample < 5; sample++) {
const start = performance.now()
for (let iteration = 0; iteration < repetitions; iteration++) {
await run(sendRequest)
}
samples.push((performance.now() - start) / repetitions)
}
return samples.sort((a, b) => a - b)[2]
}
async function searchedCharacters(sendRequest) {
const original = String.prototype.indexOf
let searched = 0
String.prototype.indexOf = function (needle, position) {
if (needle === '\n') {
searched += this.length - (position ?? 0)
}
return original.call(this, needle, position)
}
try {
await run(sendRequest)
} finally {
String.prototype.indexOf = original
}
return searched
}
const rows = []
for (const [payloadBytes, chunkChars, repetitions] of [
[32, 65536, 1000],
[1024 * 1024, 2 * 1024 * 1024, 20],
[1024 * 1024, 65536, 10],
[1024 * 1024, 4096, 5],
[4 * 1024 * 1024, 4096, 2],
[4 * 1024 * 1024, 256, 1]
]) {
const line = `${JSON.stringify({
id: 'benchmark-request',
ok: true,
result: { data: 'x'.repeat(payloadBytes) },
_meta: { runtimeId: 'benchmark-runtime' }
})}\n`
chunks = []
for (let offset = 0; offset < line.length; offset += chunkChars) {
chunks.push(line.slice(offset, offset + chunkChars))
}
const beforeMs = await measure(before, payloadBytes, repetitions)
const afterMs = await measure(after, payloadBytes, repetitions)
rows.push({
payloadBytes,
chunkChars,
beforeMs: +beforeMs.toFixed(6),
afterMs: +afterMs.toFixed(6),
speedup: +(beforeMs / afterMs).toFixed(2),
beforeSearchedCharacters: await searchedCharacters(before),
afterSearchedCharacters: await searchedCharacters(after)
})
}
console.log(JSON.stringify({ node: process.version, baselineRef, rows }, null, 2))
@@ -0,0 +1,165 @@
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import Module from 'node:module'
import { resolve } from 'node:path'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
// Pass the pre-change file-explorer-entries.ts snapshot as the only argument.
const baselinePath = process.argv[2]
assert.ok(baselinePath, 'Pass a pre-change file-explorer-entries.ts snapshot.')
const entry = 'src/renderer/src/components/right-sidebar/file-explorer-entries.ts'
const baseline = readFileSync(baselinePath, 'utf8')
assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.')
async function load(useBaseline) {
const result = await build({
stdin: {
contents: `export { isDotfileRelativePath } from './${entry}';
export { createNameFilteredFileExplorerProjection } from './src/renderer/src/components/right-sidebar/file-explorer-name-filter-projection.ts';`,
resolveDir: process.cwd(),
loader: 'ts'
},
bundle: true,
platform: 'node',
format: 'cjs',
write: false,
logLevel: 'silent',
alias: { '@': resolve('src/renderer/src') },
plugins: useBaseline
? [
{
name: 'baseline-dotfile-predicate',
setup(builder) {
builder.onLoad({ filter: /file-explorer-entries\.ts$/ }, () => ({
contents: baseline,
loader: 'ts'
}))
}
}
]
: []
})
const module = new Module(resolve('dotfile-benchmark.cjs'))
module.paths = Module._nodeModulePaths(process.cwd())
module._compile(result.outputFiles[0].text, module.id)
return module.exports
}
const versions = [await load(true), await load(false)]
let parityCases = 0
function check(path, depth) {
assert.equal(
versions[0].isDotfileRelativePath(path),
versions[1].isDotfileRelativePath(path),
path
)
parityCases++
if (depth > 0) {
for (const character of ['.', '/', '\\', 'a', '\n']) {
check(path + character, depth - 1)
}
}
}
check('', 8)
function measure(functions, iterations = 1) {
let sink = 0
const run = (fn) => {
for (let i = 0; i < iterations; i++) {
sink += Number(fn())
}
}
for (const fn of functions) {
for (let warmup = 0; warmup < 3; warmup++) {
run(fn)
}
}
const samples = [[], []]
for (let round = 0; round < 11; round++) {
for (const variant of round % 2 ? [1, 0] : [0, 1]) {
const start = performance.now()
run(functions[variant])
samples[variant].push(performance.now() - start)
}
}
return {
beforeMs: samples[0].sort((a, b) => a - b)[5],
afterMs: samples[1].sort((a, b) => a - b)[5],
iterations,
sink
}
}
const predicates = []
for (const path of [
'a',
'.env',
'packages/pkg/src/file.tsx',
`a${'.'.repeat(254)}`,
`${'/'.repeat(4096)}.`,
`${'../'.repeat(1000)}file.ts`,
'😀/.你好',
'\n/.\n'
]) {
check(path, 0)
predicates.push({
pathLength: path.length,
prefix: path.slice(0, 40),
...measure(
versions.map((version) => () => version.isDotfileRelativePath(path)),
10_000
)
})
}
const projections = []
for (const count of [1000, 10_000, 100_000]) {
for (const query of ['nonmatching-needle', 'file-42']) {
const args = {
ignoredSet: new Set(['unrelated']),
nameFilter: {
query,
relativePaths: Array.from(
{ length: count },
(_, i) => `packages/package-${i % 50}/src/components/section-${i % 10}/file-${i}.tsx`
)
},
showDotfiles: false,
showGitIgnoredFiles: false,
worktreePath: '/workspace'
}
const functions = versions.map(
(version) => () => version.createNameFilteredFileExplorerProjection(args)
)
const rows = functions.map((fn) => {
const projection = fn()
return Array.from({ length: projection.getVisibleCount() }, (_, i) =>
projection.getRowAtIndex(i)
)
})
assert.deepEqual(rows[0], rows[1])
projections.push({
count,
query,
visibleRows: rows[0].length,
...measure(functions.map((fn) => () => fn().getVisibleCount()))
})
}
}
console.log(
JSON.stringify(
{
node: process.version,
platform: process.platform,
baselinePath: resolve(baselinePath),
parityCases,
samples: 11,
warmups: 3,
predicates,
projections
},
null,
2
)
)
@@ -0,0 +1,72 @@
import { strict as assert } from 'node:assert'
import { EventEmitter } from 'node:events'
import { mkdtemp, rm } from 'node:fs/promises'
import { createRequire } from 'node:module'
import { tmpdir } from 'node:os'
import { join, resolve } from 'node:path'
import { build } from 'esbuild'
if (!global.gc) {
throw new Error('Run with node --expose-gc')
}
const root = resolve(import.meta.dirname, '../..')
const directory = await mkdtemp(join(tmpdir(), 'orca-sentinel-retention-'))
const output = join(directory, 'sentinel.cjs')
try {
await build({
stdin: {
contents: `export {waitForSentinel} from './src/main/ssh/ssh-relay-deploy-helpers';
export {RELAY_SENTINEL} from './src/main/ssh/relay-protocol';`,
resolveDir: root,
loader: 'ts'
},
bundle: true,
platform: 'node',
format: 'cjs',
packages: 'external',
banner: {
js: `var require = require('node:module').createRequire(${JSON.stringify(join(root, 'package.json'))});`
},
outfile: output
})
const { waitForSentinel, RELAY_SENTINEL } = createRequire(import.meta.url)(output)
const held = []
const banners = []
for (let i = 0; i < 100; i++) {
const channel = Object.assign(new EventEmitter(), {
stderr: new EventEmitter(),
stdin: { write: () => true },
close: () => {}
})
const pending = waitForSentinel(channel)
banners.push(feedBanner(channel))
channel.emit('data', Buffer.from(RELAY_SENTINEL))
const transport = await pending
const received = []
transport.onData((bytes) => received.push(bytes.toString()))
channel.emit('data', Buffer.from('frame'))
assert.deepEqual(received, ['frame'])
held.push({ channel, transport })
}
await new Promise((resolve) => setImmediate(resolve))
for (let i = 0; i < 5; i++) {
global.gc()
}
const retained = banners.filter((reference) => reference.deref() !== undefined).length
console.log(
JSON.stringify({
connections: held.length,
bannerBytes: 65536,
retainedBannerBuffers: retained,
retainedBannerBytes: retained * 65536
})
)
} finally {
await rm(directory, { recursive: true, force: true })
}
function feedBanner(channel) {
const banner = Buffer.alloc(65536, 120)
channel.emit('data', banner)
return new WeakRef(banner.buffer)
}
+122
View File
@@ -0,0 +1,122 @@
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import * as fs from 'node:fs/promises'
import Module from 'node:module'
import { tmpdir } from 'node:os'
import { join, resolve } from 'node:path'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
// Pass a pre-change skill-root-file-walk.ts snapshot as the only argument.
const baselinePath = process.argv[2]
const brokenLinks = process.argv.includes('--broken')
assert.ok(baselinePath, 'Pass a pre-change skill-root-file-walk.ts snapshot.')
const entry = 'src/main/skills/skill-root-file-walk.ts'
const baseline = readFileSync(baselinePath, 'utf8')
assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.')
let statCalls = 0
async function load(useBaseline) {
const result = await build({
entryPoints: [entry],
bundle: true,
platform: 'node',
format: 'cjs',
write: false,
logLevel: 'silent',
plugins: useBaseline
? [
{
name: 'baseline-skill-depth',
setup(builder) {
builder.onLoad({ filter: /skill-root-file-walk\.ts$/ }, () => ({
contents: baseline,
loader: 'ts'
}))
}
}
]
: []
})
const module = new Module(resolve('skill-depth-benchmark.cjs'))
module.paths = Module._nodeModulePaths(process.cwd())
const originalRequire = module.require.bind(module)
module.require = (name) =>
name === 'node:fs/promises'
? {
...fs,
stat: (...args) => {
statCalls++
return fs.stat(...args)
}
}
: originalRequire(name)
module._compile(result.outputFiles[0].text, module.id)
return module.exports.findSkillFiles
}
const before = await load(true)
const after = await load(false)
const median = (values) => values.sort((a, b) => a - b)[Math.floor(values.length / 2)]
const temporaryRoot = await fs.mkdtemp(join(tmpdir(), 'orca-skill-depth-benchmark-'))
try {
for (const links of [0, 8, 100, 1000]) {
const root = join(temporaryRoot, String(links))
const edge = join(root, 'a', 'b', 'c', 'd')
const target = join(temporaryRoot, 'target')
await fs.mkdir(edge, { recursive: true })
await fs.mkdir(target, { recursive: true })
await fs.writeFile(join(target, 'SKILL.md'), 'skill')
await fs.writeFile(join(edge, 'SKILL.md'), 'edge')
for (let index = 0; index < links; index++) {
await fs.symlink(
brokenLinks ? join(target, 'missing') : target,
join(edge, `link${index}`),
process.platform === 'win32' ? 'junction' : 'dir'
)
}
for (const depth of [4, 5]) {
const timings = { before: [], after: [] }
const counts = {}
let rows
for (let sample = 0; sample < 13; sample++) {
const versions =
sample % 2
? [
['after', after],
['before', before]
]
: [
['before', before],
['after', after]
]
for (const [name, walk] of versions) {
statCalls = 0
const start = performance.now()
const result = await walk(root, depth)
const elapsed = performance.now() - start
if (rows) {
assert.deepEqual(result, rows)
}
rows = result
counts[name] = statCalls
if (sample >= 2) {
timings[name].push(elapsed)
}
}
}
console.log(
JSON.stringify({
links,
brokenLinks,
depth,
statCalls: counts,
rows: rows.length,
medianMs: { before: median(timings.before), after: median(timings.after) }
})
)
}
}
} finally {
await fs.rm(temporaryRoot, { recursive: true, force: true })
}
@@ -0,0 +1,80 @@
import { strict as assert } from 'node:assert'
import { mkdtemp, readFile, rm } from 'node:fs/promises'
import { createRequire } from 'node:module'
import { tmpdir } from 'node:os'
import { join, resolve } from 'node:path'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
const root = resolve(import.meta.dirname, '../..')
const source = join(root, 'src/renderer/src/store/slices/tab-group-reference-repair.ts')
const directory = await mkdtemp(join(tmpdir(), 'orca-tab-repair-'))
const current = await readFile(source, 'utf8')
const indexed = `const orderedTabIds = new Set(group.tabOrder)
const missingTabIds = ownedTabIds.filter((tabId) => !orderedTabIds.has(tabId))`
assert(current.includes(indexed), 'Expected indexed implementation')
try {
const implementations = []
for (const baseline of [true, false]) {
const outfile = join(directory, baseline ? 'before.cjs' : 'after.cjs')
await build({
stdin: {
contents: baseline
? current.replace(
indexed,
'const missingTabIds = ownedTabIds.filter((tabId) => !group.tabOrder.includes(tabId))'
)
: current,
resolveDir: resolve(source, '..'),
loader: 'ts'
},
bundle: true,
platform: 'node',
format: 'cjs',
outfile,
alias: { '@': join(root, 'src/renderer/src') }
})
implementations.push(createRequire(import.meta.url)(outfile).appendOwnedTabIdsToGroups)
}
const rows = []
for (const count of [1, 10, 100, 1_000, 10_000]) {
for (const missing of [false, true]) {
const ids = Array.from({ length: count }, (_, i) => `tab-${i}`)
const groups = [
{ id: 'group', worktreeId: 'workspace', activeTabId: null, tabOrder: ids, recentTabIds: [] }
]
const owners = new Map(ids.map((id) => [missing ? `missing-${id}` : id, 'group']))
assert.deepEqual(implementations[0](groups, owners), implementations[1](groups, owners))
const iterations = Math.max(1, Math.floor(10_000 / count))
const samples = [[], []]
for (let sample = -3; sample < 11; sample++) {
for (const index of sample % 2 === 0 ? [0, 1] : [1, 0]) {
const start = performance.now()
for (let i = 0; i < iterations; i++) {
implementations[index](groups, owners)
}
const elapsed = (performance.now() - start) / iterations
if (sample >= 0) {
samples[index].push(elapsed)
}
}
}
rows.push({
count,
missing,
iterations,
beforeMs: samples[0].sort((a, b) => a - b)[5],
afterMs: samples[1].sort((a, b) => a - b)[5]
})
}
}
console.log(
JSON.stringify(
{ node: process.version, platform: process.platform, samples: 11, warmups: 3, rows },
null,
2
)
)
} finally {
await rm(directory, { recursive: true, force: true })
}
@@ -0,0 +1,124 @@
import assert from 'node:assert/strict'
import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import Module from 'node:module'
import { tmpdir } from 'node:os'
import { join, resolve } from 'node:path'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
const entry = 'src/shared/agent-hook-listener/transcript-reader.ts'
assert.ok(process.argv[2], 'Pass a pre-change transcript-reader.ts snapshot.')
const baseline = readFileSync(process.argv[2], 'utf8')
assert.notEqual(baseline, readFileSync(entry, 'utf8'), 'Do not compare the source to itself.')
async function load(useBaseline) {
const result = await build({
stdin: {
contents: `export * from './${entry}';
export { extractAssistantTextFromLine } from './src/shared/agent-hook-listener/transcript-entry-text.ts';`,
resolveDir: process.cwd(),
loader: 'ts'
},
bundle: true,
platform: 'node',
format: 'cjs',
write: false,
logLevel: 'silent',
plugins: useBaseline
? [
{
name: 'baseline-transcript-reader',
setup(builder) {
builder.onLoad({ filter: /transcript-reader\.ts$/ }, () => ({
contents: baseline,
loader: 'ts'
}))
}
}
]
: []
})
const module = new Module(resolve('transcript-benchmark.cjs'))
module.paths = Module._nodeModulePaths(process.cwd())
module._compile(result.outputFiles[0].text, module.id)
return module.exports
}
const versions = [await load(true), await load(false)]
function measure(functions, iterations) {
let sink = 0
const run = (fn) => {
for (let i = 0; i < iterations; i++) {
sink += fn()?.length ?? 0
}
}
for (const fn of functions) {
for (let i = 0; i < 3; i++) {
run(fn)
}
}
const samples = [[], []]
for (let round = 0; round < 11; round++) {
for (const index of round % 2 ? [1, 0] : [0, 1]) {
const start = performance.now()
run(functions[index])
samples[index].push((performance.now() - start) / iterations)
}
}
return {
beforeMs: samples[0].sort((a, b) => a - b)[5],
afterMs: samples[1].sort((a, b) => a - b)[5],
iterations,
sink
}
}
const cases = [
['tiny', `${JSON.stringify({ role: 'assistant', content: 'hello' })}\n`, 10000],
['64KiB line', `${JSON.stringify({ role: 'assistant', content: 'x'.repeat(65500) })}\n`, 100],
[
'4MiB line',
`${JSON.stringify({ role: 'assistant', content: 'x'.repeat(4 * 1024 * 1024 - 40) })}\n`,
10
],
[
'1000 short tool lines',
Array.from({ length: 1000 }, () =>
JSON.stringify({ role: 'tool', content: 'x'.repeat(100) })
).join('\n'),
50
],
[
'Unicode line',
`${JSON.stringify({ role: 'assistant', content: '😀漢字'.repeat(16000) })}\n`,
100
],
[
'leading and trailing blank lines',
`\n\r\n${JSON.stringify({ role: 'assistant', content: 'hello' })}\n\n`,
10000
]
]
const directory = mkdtempSync(join(tmpdir(), 'orca-transcript-benchmark-'))
try {
for (const [name, text, iterations] of cases) {
const file = join(directory, 'transcript.jsonl')
writeFileSync(file, text)
const scanners = versions.map(
(v) => () => v.findLastExtractedTranscriptLineText(text, v.extractAssistantTextFromLine)
)
const readers = versions.map((v) => () => v.readLastAssistantFromTranscriptOnce(file))
assert.equal(scanners[0](), scanners[1](), name)
assert.equal(readers[0](), readers[1](), name)
console.log(
JSON.stringify({
name,
bytes: Buffer.byteLength(text),
scanner: measure(scanners, iterations),
warmFileReader: measure(readers, Math.min(iterations, 100))
})
)
}
} finally {
rmSync(directory, { recursive: true, force: true })
}
@@ -32,6 +32,8 @@ import {
import { join, resolve } from 'node:path'
import { RELAY_WINDOWS_PROCESS_TREE_FILENAME } from '../../src/shared/relay-artifacts.ts'
import {
ensureWindowsProcessTreeCommandLinePatch,
inspectWindowsProcessTreeAddon,
nodeGypRebuildInvocation,
stageWindowsProcessTreeNodeAddonApiHeaders,
WINDOWS_PROCESS_TREE_PACKAGE_DIR as PACKAGE_DIR
@@ -89,6 +91,13 @@ function assertPatchApplied() {
'config/patches/@vscode__windows-process-tree@0.8.0.patch; run pnpm install.'
)
}
if (processCc.includes('OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ')) {
throw new Error(
'src/process.cc still takes PROCESS_VM_READ for memory or CPU counters it never reads ' +
'from the address space. pnpm did not apply ' +
'config/patches/@vscode__windows-process-tree@0.8.0.patch; run pnpm install.'
)
}
}
// pnpm can materialize this CRLF package without applying its patch. Repair the
@@ -123,6 +132,13 @@ function applyWindowsProcessTreeBuildFixes() {
''
)
processCc = processCc.replace(/process_count < 1024 && /, '')
// The memory and CPU readers only ever call GetProcessMemoryInfo/GetProcessTimes,
// which need no more than PROCESS_QUERY_LIMITED_INFORMATION; taking VM_READ is
// what EDR scores.
processCc = processCc.replaceAll(
'OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ, false, pid)',
'OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, false, pid)'
)
if (bindingGyp !== originalBinding) {
writeFileSync(bindingPath, bindingGyp)
@@ -131,7 +147,8 @@ function applyWindowsProcessTreeBuildFixes() {
writeFileSync(processPath, processCc)
}
stageWindowsProcessTreeNodeAddonApiHeaders(PACKAGE_DIR)
if (bindingGyp !== originalBinding || processCc !== originalProcess) {
const repairedCommandLine = ensureWindowsProcessTreeCommandLinePatch(PACKAGE_DIR)
if (bindingGyp !== originalBinding || processCc !== originalProcess || repairedCommandLine) {
console.warn('[windows-process-tree] Repaired un-applied pnpm patch hunks before build.')
}
}
@@ -173,6 +190,14 @@ function main() {
if (!existsSync(built)) {
throw new Error(`node-gyp reported success but ${built} is missing.`)
}
// Why check the artifact and not only the source: the source checks above run
// before node-gyp, and a stale build directory can outlive them.
if (inspectWindowsProcessTreeAddon(built) === 'unpatched') {
throw new Error(
'The built addon still calls ReadProcessMemory, so it did not come from the patched ' +
'command-line reader. A relay would get the primitive MDE scores as credential dumping.'
)
}
const machine = readPeMachine(built)
if (machine !== PE_MACHINE[arch]) {
throw new Error(
@@ -2,7 +2,7 @@
// Equivalence check for deferring the RuntimeClient module graph in the CLI.
//
// Builds the CLI twice with the REAL tsc emit — once from the working tree and
// once with the seven touched files restored from git HEAD~ (the pre-deferral
// once with the touched files restored from git HEAD~ (the pre-deferral
// implementation) — then compares stdout, stderr and exit code BYTE FOR BYTE
// across a matrix of invocations.
//
@@ -13,7 +13,7 @@
//
// Usage: node config/scripts/cli-runtime-client-deferral-equivalence.mjs [--baseline <rev>]
import { execFileSync, spawnSync } from 'node:child_process'
import { mkdirSync, mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs'
import { existsSync, mkdirSync, mkdtempSync, rmSync, writeFileSync, readFileSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { fileURLToPath } from 'node:url'
@@ -21,8 +21,11 @@ const REPO = fileURLToPath(new URL('../..', import.meta.url))
// The files this change touches. Restoring exactly these from the baseline rev
// reconstructs the old implementation without disturbing anything else.
// Files absent at the baseline (e.g. cli-error.ts, split out of format.ts
// later) are removed for the baseline build and put back afterwards.
const TOUCHED = [
'src/cli/args.ts',
'src/cli/cli-error.ts',
'src/cli/dispatch.ts',
'src/cli/flags.ts',
'src/cli/format.ts',
@@ -72,12 +75,16 @@ function buildTree(label, baselineRev) {
if (baselineRev) {
for (const file of TOUCHED) {
const path = join(REPO, file)
restored.push([path, readFileSync(path)])
const old = execFileSync('git', ['show', `${baselineRev}:${file}`], {
restored.push([path, existsSync(path) ? readFileSync(path) : null])
const old = spawnSync('git', ['show', `${baselineRev}:${file}`], {
cwd: REPO,
maxBuffer: 64 * 1024 * 1024
})
writeFileSync(path, old)
if (old.status === 0) {
writeFileSync(path, old.stdout)
} else {
rmSync(path, { force: true })
}
}
}
execFileSync(
@@ -97,7 +104,11 @@ function buildTree(label, baselineRev) {
)
} finally {
for (const [path, contents] of restored) {
writeFileSync(path, contents)
if (contents === null) {
rmSync(path, { force: true })
} else {
writeFileSync(path, contents)
}
}
}
return join(outDir, 'cli/index.js')
@@ -103,14 +103,24 @@ describe('electron-builder markdown file associations', () => {
// Why: this include was renamed from daemon-host-uninstall.nsh to carry the markdown
// hooks too. electron-builder allows only one include, so a merge that drops the daemon
// sweep would silently orphan a running orca-terminal-daemon.exe on every uninstall.
// sweep would silently orphan a running daemon host on every uninstall.
//
// Asserted against comment-stripped script, and on the app exe name first: the relocated
// host is a verbatim copy of the app exe (daemonHostExeName, daemon-host-relocation.ts),
// so a macro that kills only orca-terminal-daemon.exe matches no running process. The
// prose above the macro names both, so a toContain over the raw file proves nothing.
it('keeps the daemon-host uninstall sweep across the include rename', async () => {
const hooks = await readInstallerHooks()
const script = stripNsisCommentLines(await readInstallerHooks())
expect(hooks).toContain('orca-terminal-daemon.exe')
expect(hooks).toContain('$LOCALAPPDATA\\Orca\\daemon-host')
expect(script).toMatch(/taskkill[^\n]*\/IM\s+"?\$\{APP_EXECUTABLE_FILENAME\}"?/)
// Legacy name, so hosts left by builds that renamed the copy still get reaped.
expect(script).toMatch(/taskkill[^\n]*\/IM\s+"?orca-terminal-daemon\.exe"?/)
// Scopes both kills to the uninstalling user: an elevated machine-wide uninstall must
// not reach another logged-on user's session.
expect(script).toMatch(/\/FI\s+"USERNAME eq /)
expect(script).toContain('$LOCALAPPDATA\\Orca\\daemon-host')
// Without this guard, uninstallOldVersion would kill the daemon on every update —
// defeating the relocation that keeps terminals alive across updates.
expect(hooks).toMatch(/\$\{ifNot\}\s+\$\{isUpdated\}/)
expect(script).toMatch(/\$\{ifNot\}\s+\$\{isUpdated\}/)
})
})
+25 -4
View File
@@ -5,6 +5,12 @@ import { createRequire } from 'node:module'
import { existsSync, readFileSync } from 'node:fs'
import { release } from 'node:os'
import { basename, dirname, resolve } from 'node:path'
import {
ensureWindowsProcessTreeCommandLinePatch,
inspectWindowsProcessTreeAddon,
stageWindowsProcessTreeNodeAddonApiHeaders,
windowsProcessTreeAddonPath
} from './windows-process-tree-gyp-rebuild.mjs'
const require = createRequire(import.meta.url)
const { assertNodePtyJobOwnership } = require('./node-pty-job-ownership.cjs')
@@ -253,11 +259,18 @@ function collectNativeModuleFailures() {
function loadNativeModule(moduleName) {
if (moduleName === '@vscode/windows-process-tree') {
// A bare require already loads the .node addon on win32, so it catches an
// ABI mismatch on its own. What it cannot catch is a snapshot that comes
// back empty -- the shape a blocked CreateToolhelp32Snapshot produces --
// so check the addon actually enumerates before calling the runtime healthy.
// A bare require loads the .node addon on win32, so it catches an ABI
// mismatch on its own. What it cannot catch is *which* addon loaded: the
// published tarball ships a prebuilt built from unpatched source that is
// node-addon-api, so it requires cleanly and then reads every process's
// command line out of its address space. Check the binary, not the load.
require(moduleName)
if (inspectWindowsProcessTreeAddon(windowsProcessTreeAddonPath()) === 'unpatched') {
throw new Error(
'the loaded addon still calls ReadProcessMemory, so it was not built from the patched ' +
'source. Rebuild it (pnpm run rebuild:electron) rather than using the published prebuild.'
)
}
return
}
if (moduleName === 'windows-native-registry') {
@@ -368,6 +381,14 @@ function getWindowsBuildNumber() {
function rebuildNodeRuntimeModules(moduleNames) {
for (const moduleName of moduleNames) {
const moduleDir = dirname(require.resolve(`${moduleName}/package.json`))
if (moduleName === '@vscode/windows-process-tree') {
// Why before node-gyp: this module is rebuilt precisely because the
// binary was the unpatched one, and pnpm materializes it unpatched often
// enough that compiling the source as-is would just rebuild the same
// reader and fail the verify pass.
ensureWindowsProcessTreeCommandLinePatch(moduleDir)
stageWindowsProcessTreeNodeAddonApiHeaders(moduleDir)
}
console.warn(`[native-runtime] Rebuilding ${moduleName} with node-gyp.`)
runPnpm(['exec', 'node-gyp', 'rebuild'], { cwd: moduleDir })
if (moduleName === 'node-pty' && process.platform === 'win32') {
@@ -12,6 +12,7 @@ import { tmpdir } from 'node:os'
import { delimiter, join } from 'node:path'
import { fileURLToPath } from 'node:url'
import { describe, expect, it } from 'vitest'
import { copyScriptWithLocalModules } from './script-module-dependencies.mjs'
const sourceScriptPath = fileURLToPath(new URL('./ensure-native-runtime.mjs', import.meta.url))
const sourceNodePtyJobOwnershipPath = fileURLToPath(
@@ -27,7 +28,6 @@ describe('ensure-native-runtime', () => {
const logPath = join(projectDir, 'native-runtime.log')
const markerPath = join(projectDir, 'rebuilt.marker')
const binDir = join(projectDir, 'bin')
copyFileSync(sourceScriptPath, scriptPath)
writeFakeNativeModules(projectDir)
writeNodePtyPatchFile(projectDir)
writeFakePnpm(binDir)
@@ -67,7 +67,6 @@ describe('ensure-native-runtime', () => {
const logPath = join(projectDir, 'native-runtime.log')
const markerPath = join(projectDir, 'rebuilt.marker')
const binDir = join(projectDir, 'bin')
copyFileSync(sourceScriptPath, scriptPath)
writeFakeNativeModules(projectDir, { windowsRegistryRequiresMarker: true })
writeNodePtyPatchFile(projectDir)
writeFakePnpm(binDir)
@@ -102,7 +101,6 @@ describe('ensure-native-runtime', () => {
const logPath = join(projectDir, 'native-runtime.log')
const markerPath = join(projectDir, 'rebuilt.marker')
const binDir = join(projectDir, 'bin')
copyFileSync(sourceScriptPath, scriptPath)
writeLoadableNativeModules(projectDir)
writeNodePtyPatchFile(projectDir)
writeFakePnpm(binDir)
@@ -137,7 +135,6 @@ describe('ensure-native-runtime', () => {
const logPath = join(projectDir, 'native-runtime.log')
const markerPath = join(projectDir, 'rebuilt.marker')
const binDir = join(projectDir, 'bin')
copyFileSync(sourceScriptPath, scriptPath)
writeLoadableNativeModules(projectDir)
writeNodePtyPatchFile(projectDir)
writePatchedNodePtyBuildArtifacts(projectDir)
@@ -171,7 +168,6 @@ describe('ensure-native-runtime', () => {
const logPath = join(projectDir, 'native-runtime.log')
const markerPath = join(projectDir, 'rebuilt.marker')
const binDir = join(projectDir, 'bin')
copyFileSync(sourceScriptPath, scriptPath)
writeLoadableNativeModules(projectDir, { nativeDir: '../build/Release/' })
writeNodePtyPatchFile(projectDir)
writePatchedNodePtyBuildArtifacts(projectDir)
@@ -198,7 +194,9 @@ describe('ensure-native-runtime', () => {
function mkTempProject() {
const projectDir = mkdtempSync(join(tmpdir(), 'orca-native-runtime-'))
mkdirSync(join(projectDir, 'config', 'scripts'), { recursive: true })
// Walked, not listed: the script imports windows-process-tree-gyp-rebuild.mjs, and a fixture
// missing it fails every case with a module-resolution error instead of the defect under test.
copyScriptWithLocalModules(sourceScriptPath, join(projectDir, 'config', 'scripts'))
copyFileSync(
sourceNodePtyJobOwnershipPath,
join(projectDir, 'config', 'scripts', 'node-pty-job-ownership.cjs')
@@ -0,0 +1,75 @@
import assert from 'node:assert/strict'
import { join } from 'node:path'
import { performance } from 'node:perf_hooks'
import { fileURLToPath } from 'node:url'
import { build } from 'esbuild'
const root = fileURLToPath(new URL('../..', import.meta.url))
const bundled = await build({
stdin: {
contents: `export { selectDeletionRoots } from './file-explorer-batch-deletion';
export { isPathEqualOrDescendant } from './file-explorer-paths';`,
resolveDir: join(root, 'src/renderer/src/components/right-sidebar'),
loader: 'ts'
},
alias: { '@': join(root, 'src/renderer/src') },
bundle: true,
platform: 'node',
format: 'esm',
write: false,
logLevel: 'silent'
})
const { selectDeletionRoots, isPathEqualOrDescendant } = await import(
`data:text/javascript;base64,${Buffer.from(bundled.outputFiles[0].text).toString('base64')}`
)
// Original production selector; both paths use the same path-comparison implementation.
function original(nodes) {
return nodes.filter(
(n) =>
!nodes.some(
(other) => other !== n && other.isDirectory && isPathEqualOrDescendant(n.path, other.path)
)
)
}
function measure(run, nodes) {
for (let index = 0; index < 3; index++) {
run(nodes)
}
const samples = []
for (let index = 0; index < 11; index++) {
const start = performance.now()
run(nodes)
samples.push(performance.now() - start)
}
return samples.sort((a, b) => a - b)[5]
}
const results = []
for (const [fileCount, directoryCount] of [
[100, 0],
[1000, 0],
[5000, 0],
[5000, 5],
[0, 100]
]) {
const nodes = Array.from({ length: fileCount + directoryCount }, (_, index) => ({
name: `item-${index}`,
path: `/repo/item-${index}`,
relativePath: `item-${index}`,
isDirectory: index >= fileCount,
depth: 0
}))
const expected = original(nodes)
const actual = selectDeletionRoots(nodes)
assert.equal(actual.length, expected.length)
actual.forEach((node, index) => assert.equal(node, expected[index]))
results.push({
fileCount,
directoryCount,
beforeMs: measure(original, nodes),
afterMs: measure(selectDeletionRoots, nodes)
})
}
console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2))
@@ -0,0 +1,53 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
import { stripTypeScriptTypes } from 'node:module'
import { performance } from 'node:perf_hooks'
const baseline = process.argv[2]
if (!baseline) {
throw new Error('Usage: node config/scripts/mobile-file-ranking-benchmark.mjs <baseline-ref>')
}
async function load(source) {
const js = stripTypeScriptTypes(source, { mode: 'transform' })
return await import(`data:text/javascript;base64,${Buffer.from(js).toString('base64')}`)
}
function measure(fn, paths, query) {
for (let warmup = 0; warmup < 10; warmup++) {
fn(paths, query, 16)
}
const samples = []
for (let i = 0; i < 9; i++) {
const start = performance.now()
fn(paths, query, 16)
samples.push(performance.now() - start)
}
return samples.sort((a, b) => a - b)[4]
}
const results = []
for (const [file, name] of [
['src/main/runtime/runtime-mobile-file-path-search.ts', 'rankRuntimeMobileFilePaths'],
['mobile/src/session/mobile-native-chat-autocomplete.ts', 'rankSuggestions']
]) {
const before = (
await load(execFileSync('git', ['show', `${baseline}:${file}`], { encoding: 'utf8' }))
)[name]
const after = (await load(readFileSync(file, 'utf8')))[name]
for (const count of [100, 100000]) {
const paths = Array.from(
{ length: count },
(_, i) => `src/components/workspace/group-${i % 100}/file-${i}.tsx`
)
for (const query of ['file-9', 'missing', 'workspace']) {
assert.deepEqual(after(paths, query, 16), before(paths, query, 16))
results.push({
function: name,
paths: count,
query,
beforeMs: measure(before, paths, query),
afterMs: measure(after, paths, query)
})
}
}
}
console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2))
@@ -0,0 +1,58 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { readFileSync } from 'node:fs'
import { dirname, resolve } from 'node:path'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
const sourcePath = 'mobile/src/components/mobile-markdown-preview-html.ts'
const baselineRef = process.argv[2]
if (!baselineRef) {
throw new Error(
'Usage: node config/scripts/mobile-markdown-placeholder-benchmark.mjs <baseline-ref>'
)
}
async function load(source) {
const result = await build({
stdin: { contents: source, resolveDir: dirname(resolve(sourcePath)), loader: 'ts' },
bundle: true,
write: false,
platform: 'node',
format: 'esm'
})
return (
await import(
`data:text/javascript;base64,${Buffer.from(result.outputFiles[0].text).toString('base64')}`
)
).normalizeMobileMarkdownPreviewHtml
}
const before = await load(
execFileSync('git', ['show', `${baselineRef}:${sourcePath}`], { encoding: 'utf8' })
)
const after = await load(readFileSync(sourcePath, 'utf8'))
function measure(fn, input, repeats) {
const samples = []
for (let run = 0; run < repeats; run++) {
const start = performance.now()
fn(input)
samples.push(performance.now() - start)
}
return samples.sort((a, b) => a - b)[Math.floor(samples.length / 2)]
}
const results = []
for (const [shape, input] of [
['ordinary Markdown', '# Hello\n\n<p>Use `Array<string>` and <b>bold</b>.</p>'],
...[2048, 8192, 16384].map((length) => [
`${length} underscore collision`,
`\uE000ORCA_MD_CODE_${'_'.repeat(length)}0\uE000 and \`Array<string>\``
])
]) {
assert.equal(after(input), before(input))
results.push({
shape,
bytes: Buffer.byteLength(input),
beforeMs: measure(before, input, 5),
afterMs: measure(after, input, 15)
})
}
console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2))
@@ -0,0 +1,155 @@
import {
cpSync,
copyFileSync,
existsSync,
mkdirSync,
mkdtempSync,
readFileSync,
writeFileSync
} from 'node:fs'
import { tmpdir } from 'node:os'
import { isAbsolute, join, parse, resolve } from 'node:path'
import { describe, expect, it } from 'vitest'
import { runProcessSync } from '../../src/shared/child-process/run-process.ts'
import { resolveCliCommand } from '../../src/shared/node-cli-command-resolution.ts'
import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts'
import { resolvePnpmCliInvocation } from './pnpm-cli-invocation.mjs'
/**
* Run the command that actually consumes the patch hashes.
*
* A hash comparison is not this check. `@vscode/windows-process-tree@0.8.0` shipped
* twice with a hand-computed `sha256(patchBytes)` in the lockfile, and two separate
* reviews "verified" it by recomputing the same number the same wrong way. pnpm
* hashes the **LF-normalized** content, so a CRLF patch makes the raw digest a value
* pnpm will never produce, and `--frozen-lockfile` dies with
* ERR_PNPM_LOCKFILE_CONFIG_MISMATCH on every runner. An independent check that
* repeats the original assumption is not independent; only the installer is.
*
* `--lockfile-only --ignore-scripts` keeps it to the resolution pnpm rejects on,
* with no node_modules and no native builds.
*/
const PROJECT_DIR = resolve(import.meta.dirname, '../..')
const WINDOWS_PROCESS_TREE_PATCH = '@vscode__windows-process-tree@0.8.0.patch'
/**
* Which pnpm to run belongs to pnpm-cli-invocation.mjs, not to this file: naming
* the Windows shim here is what windows-cmd-shim-spawn-boundary.test.mjs rejects.
* Its `shell` is dropped on purpose -- runProcessSync refuses that flag and
* already drives a shim through the interpreter itself.
*/
function resolvePnpmInvocation() {
const { command, prefixArgs } = resolvePnpmCliInvocation()
if (isAbsolute(command)) {
return existsSync(command) ? { program: command, prefixArgs } : null
}
// Bare name only when npm_execpath is unset (bare `vitest`, not `pnpm test`).
// Drop the extension so the shared resolver tries every executable form of it.
const resolved = resolveCliCommand(parse(command).name)
return isAbsolute(resolved) ? { program: resolved, prefixArgs } : null
}
describe('patched dependencies', () => {
it('installs with --frozen-lockfile, which is what validates every patch hash', () => {
const pnpm = resolvePnpmInvocation()
expect(pnpm, 'pnpm must be installed; it is the only thing that can check this').not.toBeNull()
// A copy, because a --frozen-lockfile run still rewrites parts of the
// lockfile this repo does not track, and the real one must not move.
const scratch = mkdtempSync(join(tmpdir(), 'orca-frozen-install-'))
try {
for (const file of ['package.json', 'pnpm-lock.yaml', 'pnpm-workspace.yaml']) {
copyFileSync(join(PROJECT_DIR, file), join(scratch, file))
}
mkdirSync(join(scratch, 'config'), { recursive: true })
cpSync(join(PROJECT_DIR, 'config', 'patches'), join(scratch, 'config', 'patches'), {
recursive: true
})
const result = runProcessSync({
program: pnpm.program,
args: [
...pnpm.prefixArgs,
'install',
'--frozen-lockfile',
'--lockfile-only',
'--ignore-scripts'
],
cwd: scratch,
timeoutMs: 300_000
})
expect(result.code, `${result.stdout}\n${result.stderr}`).toBe(0)
} finally {
removeTreeSync(scratch)
}
// The 300s spawn budget is only reachable if the case is allowed to take it;
// config/vitest.config.ts caps every case at 30s by default.
}, 300_000)
/**
* `--lockfile-only` resolves; it never applies a patch. So the case above is
* bounded to hash consistency, and the actual question -- can pnpm still put
* the patched reader on disk? -- had nothing covering it.
*
* One package, patch applied for real, assert the marker landed. Scoped to the
* single dependency so it stays a ~2s check rather than a full install.
*/
it('materializes the patched command-line reader on a real install', () => {
const pnpm = resolvePnpmInvocation()
expect(pnpm, 'pnpm must be installed; it is the only thing that can check this').not.toBeNull()
const scratch = mkdtempSync(join(tmpdir(), 'orca-patch-apply-'))
try {
mkdirSync(join(scratch, 'config', 'patches'), { recursive: true })
copyFileSync(
join(PROJECT_DIR, 'config', 'patches', WINDOWS_PROCESS_TREE_PATCH),
join(scratch, 'config', 'patches', WINDOWS_PROCESS_TREE_PATCH)
)
writeFileSync(
join(scratch, 'package.json'),
`${JSON.stringify(
{
name: 'orca-patch-apply-probe',
version: '1.0.0',
dependencies: { '@vscode/windows-process-tree': '0.8.0' }
},
null,
2
)}\n`
)
writeFileSync(
join(scratch, 'pnpm-workspace.yaml'),
'packages: []\n' +
'patchedDependencies:\n' +
` '@vscode/windows-process-tree@0.8.0': config/patches/${WINDOWS_PROCESS_TREE_PATCH}\n`
)
const result = runProcessSync({
program: pnpm.program,
args: [...pnpm.prefixArgs, 'install', '--no-frozen-lockfile', '--ignore-scripts'],
cwd: scratch,
timeoutMs: 300_000
})
expect(result.code, `${result.stdout}\n${result.stderr}`).toBe(0)
const materialized = readFileSync(
join(
scratch,
'node_modules',
'@vscode',
'windows-process-tree',
'src',
'process_commandline.cc'
),
'utf8'
)
expect(materialized).toContain('kProcessCommandLineInformation')
// The whole point of the patch: the upstream reader is gone, not merely
// supplemented.
expect(materialized).not.toContain('ReadProcessMemory')
} finally {
removeTreeSync(scratch)
}
}, 300_000)
})
+7 -1
View File
@@ -213,6 +213,7 @@ const LINUX_PACKAGE_TESTS = [
const WINDOWS_PACKAGE_TESTS = [
...LINUX_PACKAGE_TESTS,
'config/scripts/rebuild-native-deps.test.mjs',
'config/scripts/rebuild-native-deps-windows-process-tree.test.mjs',
'src/main/providers/windows-conpty-wide-char-duplication.node-pty.test.ts',
'src/main/providers/pty-repaint-wide-char-buffer.node-pty.test.ts',
'src/shared/child-process/windows-command-line.win32.test.ts',
@@ -220,6 +221,7 @@ const WINDOWS_PACKAGE_TESTS = [
'src/main/agent-hooks/windows-direct-cmd-hook-command.test.ts',
'src/main/windows/windows-pty-job.win32.test.ts',
'src/main/windows/windows-host-job.win32.test.ts',
'src/main/windows/windows-process-tree-command-line-patch.test.ts',
'src/main/windows-live-tree-kill.win32.test.ts',
'src/main/wsl/wsl-runner.test.ts',
'src/main/wsl/wsl-guest-environment.test.ts',
@@ -228,14 +230,18 @@ const WINDOWS_PACKAGE_TESTS = [
'src/main/wsl/wsl-w1-w3-contract.test.ts',
'src/shared/source-scan/source-tree-scan.test.ts',
'src/main/cli/wsl-cli-powershell-boundary.test.ts',
'src/main/computer/desktop-script-runtime-host.win32.test.ts',
'src/main/cursor/hook-service.test.ts',
'src/main/orca-profiles/profile-index-store.test.ts',
'src/main/startup/windows-install-dir-acl-repair.win32.test.ts',
'src/main/runtime/repo-worktree-admin-fingerprint.test.ts',
'src/main/runtime/worktree-scan-admin-fingerprint-gate.test.ts',
'src/shared/secure-file-fsync-flags.test.ts',
'src/shared/secure-path-windows-acl.win32.test.ts',
'src/main/runtime/unreadable-secret-store-preservation.win32.test.ts',
'src/main/ipc/pty-codex-account-attribution.test.ts',
'src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts'
'src/main/ipc/pty-spawn-env-codex-resume-provenance.test.ts',
'src/relay/windows-port-scan.win32.test.ts'
]
const DESKTOP_IRRELEVANT_PREFIXES = [
@@ -0,0 +1,33 @@
import { readFileSync } from 'node:fs'
import { describe, expect, it } from 'vitest'
import { parse } from 'yaml'
import { hasNativeImeSourceChange, shouldRunReusablePrE2e } from './pr-e2e-source-routing.mjs'
const workflow = parse(readFileSync('.github/workflows/pr.yml', 'utf8'))
const filterStep = workflow.jobs.code_paths.steps.find((step) => step.id === 'e2e_filter')
describe('native-only PR E2E routing', () => {
it('avoids generic E2E allocation for native-only changes while preserving its IME lane', () => {
for (const file of [
'tests/e2e/terminal-ibus-hangul-native.spec.ts',
'config/scripts/run-terminal-ibus-hangul-e2e.mjs'
]) {
expect(hasNativeImeSourceChange([file])).toBe(true)
expect(shouldRunReusablePrE2e([file])).toBe(false)
}
expect(shouldRunReusablePrE2e([])).toBe(false)
for (const spec of [
'tests/e2e/ssh-startup-exec-readiness.spec.ts',
'tests/e2e/paired-startup-exec-readiness.spec.ts',
'tests/e2e/terminal-ime-exact-byte.spec.ts',
'tests/e2e/future.spec.ts'
]) {
expect(shouldRunReusablePrE2e([spec])).toBe(true)
expect(shouldRunReusablePrE2e(['tests/e2e/terminal-ibus-hangul-native.spec.ts', spec])).toBe(
true
)
}
expect(filterStep.run).toContain('pr-e2e-source-routing.mjs --reusable-workflow')
expect(filterStep.run).toContain('if [ "$SHOULD_RUN" = true ]; then')
})
})
+12
View File
@@ -217,6 +217,16 @@ export function hasNativeImeSourceChange(changedPaths) {
).some((route) => changedPaths.some(route.matches))
}
export function shouldRunReusablePrE2e(changedPaths) {
// Native IME has its own workflow; SSH still runs inside the reusable workflow.
return (
hasSshSourceChange(changedPaths) ||
selectPrE2eSpecs(changedPaths).some(
(spec) => spec !== 'tests/e2e/terminal-ibus-hangul-native.spec.ts'
)
)
}
if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
let input = ''
process.stdin.setEncoding('utf8')
@@ -226,6 +236,8 @@ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href)
const changedPaths = input.split(/\r?\n/).filter(Boolean)
if (process.argv.includes('--ssh-source')) {
process.stdout.write(`${hasSshSourceChange(changedPaths)}\n`)
} else if (process.argv.includes('--reusable-workflow')) {
process.stdout.write(`${shouldRunReusablePrE2e(changedPaths)}\n`)
} else if (process.argv.includes('--native-ime-source')) {
process.stdout.write(`${hasNativeImeSourceChange(changedPaths)}\n`)
} else {
@@ -0,0 +1,60 @@
import assert from 'node:assert/strict'
import { performance } from 'node:perf_hooks'
import { build } from 'esbuild'
const bundled = await build({
entryPoints: ['src/shared/quick-open-filter.ts'],
bundle: true,
platform: 'node',
format: 'esm',
write: false,
logLevel: 'silent'
})
const { shouldExcludeQuickOpenRelPath: after } = await import(
`data:text/javascript;base64,${Buffer.from(bundled.outputFiles[0].text).toString('base64')}`
)
// Original production predicate, including its exact boundary check.
function before(relPath, prefixes) {
for (const prefix of prefixes) {
if (relPath === prefix) {
return true
}
if (relPath.length > prefix.length && relPath.startsWith(`${prefix}/`)) {
return true
}
}
return false
}
const files = Array.from(
{ length: 100000 },
(_, index) => `src/components/group-${index % 100}/file-${index}.tsx`
)
function run(fn, prefixes) {
let excluded = 0
for (const file of files) {
excluded += Number(fn(file, prefixes))
}
return excluded
}
function measure(fn, prefixes) {
run(fn, prefixes)
const samples = []
for (let index = 0; index < 5; index++) {
const start = performance.now()
run(fn, prefixes)
samples.push(performance.now() - start)
}
return samples.sort((a, b) => a - b)[2]
}
const results = []
for (const count of [0, 10, 100, 500]) {
const prefixes = Array.from({ length: count }, (_, index) => `nested-worktrees/worktree-${index}`)
assert.equal(run(after, prefixes), run(before, prefixes))
results.push({
files: files.length,
exclusions: count,
beforeMs: measure(before, prefixes),
afterMs: measure(after, prefixes)
})
}
console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2))
@@ -4,6 +4,8 @@ import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
import {
gitLineEndingEnv,
initGitWorkTree,
mkTempProject,
runRebuildScript,
writeFakeElectronRebuild,
@@ -14,7 +16,8 @@ import {
writeFakeWindowsProcessTreeWithNodeAddonApi,
writeFakeWindowsRegistry,
writeNodePtyPatchFile,
writePatchedNodePtyBuildArtifacts
writePatchedNodePtyBuildArtifacts,
writeWindowsProcessTreePatchFile
} from './rebuild-native-deps-test-fixtures.mjs'
describe('rebuild-native-deps patched node-pty rebuild', () => {
@@ -85,6 +88,91 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
}
})
const commandLineSourcePath = (projectDir) =>
join(
projectDir,
'node_modules',
'@vscode',
'windows-process-tree',
'src',
'process_commandline.cc'
)
// Why inside a git work tree: `git apply` run under one prefixes patch paths
// with the cwd-relative prefix, silently skips what does not match, and still
// exits 0. The package dir is always under the project root in production, so
// a fixture in %TEMP% alone would pass while the real repair did nothing.
//
// Why both line-ending modes: the patch is stored LF while upstream ships this
// source CRLF, so whether the pre-image matches depends on `core.autocrlf` --
// and under `false`, Git's own built-in default, it did not. The repair blinds
// git to the repo, so that value comes from global config, i.e. from whichever
// option the developer's installer wrote. Pinning both makes the case cover the
// host that breaks rather than the host that happens to run it.
for (const autocrlf of ['false', 'true']) {
it(`repairs an un-applied command-line patch in a work tree (autocrlf=${autocrlf})`, () => {
const projectDir = mkTempProject()
try {
initGitWorkTree(projectDir)
writeFakeUsableElectronPackage(projectDir, { platform: 'win32' })
writeFakeElectronRebuild(projectDir)
writeFakeNodePtyConptyPayload(projectDir, 'x64')
writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir, {
commandLinePatchApplied: false
})
writeWindowsProcessTreePatchFile(projectDir)
const result = runRebuildScript(
projectDir,
{
npm_config_platform: 'win32',
npm_config_arch: 'x64',
...gitLineEndingEnv(autocrlf)
},
['--platform=win32', '--arch=x64', '--force']
)
expect(result.status, result.stderr).toBe(0)
expect(readFileSync(commandLineSourcePath(projectDir), 'utf8')).toContain(
'kProcessCommandLineInformation'
)
} finally {
removeTreeSync(projectDir)
}
})
}
// Why fail rather than build: an unpatched command-line reader compiles fine
// and then opens every process with PROCESS_VM_READ to walk its PEB, which is
// the primitive the patch exists to remove.
it('refuses a Windows rebuild when the command-line patch cannot be applied', () => {
const projectDir = mkTempProject()
try {
initGitWorkTree(projectDir)
writeFakeUsableElectronPackage(projectDir, { platform: 'win32' })
writeFakeElectronRebuild(projectDir)
writeFakeNodePtyConptyPayload(projectDir, 'x64')
writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir, { commandLinePatchApplied: false })
// No patch file, so the repair has nothing to apply.
const result = runRebuildScript(
projectDir,
{ npm_config_platform: 'win32', npm_config_arch: 'x64' },
['--platform=win32', '--arch=x64', '--force']
)
expect(result.status).not.toBe(0)
expect(result.stderr).toContain('process_commandline.cc')
expect(readFileSync(commandLineSourcePath(projectDir), 'utf8')).not.toContain(
'kProcessCommandLineInformation'
)
} finally {
removeTreeSync(projectDir)
}
})
it('restores the ConPTY runtime payload after a Windows Electron rebuild', () => {
const projectDir = mkTempProject()
@@ -256,4 +344,37 @@ describe('rebuild-native-deps patched node-pty rebuild', () => {
}
}
)
// The binary this step produces is the one copied into the packaged app. The
// relay build checks its own artifact and ensure-native-runtime checks what it
// loads; nothing checked this one, so a rebuild that quietly emitted the
// upstream reader shipped. Both non-clean states have to fail, which is the
// caller the tri-state was missing: after a rebuild that reported success, an
// absent binary is a broken build, not an absence to shrug at.
for (const [addon, expected] of [
['unpatched', 'still imports ReadProcessMemory'],
['none', 'is not there']
]) {
it(`fails a Windows rebuild that leaves ${addon} windows-process-tree bytes`, () => {
const projectDir = mkTempProject()
try {
writeFakeUsableElectronPackage(projectDir, { platform: 'win32' })
writeFakeElectronRebuild(projectDir, { addon })
writeFakeNodePtyConptyPayload(projectDir, 'x64')
writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir)
const result = runRebuildScript(
projectDir,
{ npm_config_platform: 'win32', npm_config_arch: 'x64' },
['--platform=win32', '--arch=x64', '--force']
)
expect(result.status).not.toBe(0)
expect(result.stderr).toContain(expected)
} finally {
removeTreeSync(projectDir)
}
})
}
})
@@ -1,5 +1,12 @@
import { spawnSync } from 'node:child_process'
import { chmodSync, copyFileSync, mkdirSync, mkdtempSync, writeFileSync } from 'node:fs'
import {
chmodSync,
copyFileSync,
mkdirSync,
mkdtempSync,
readFileSync,
writeFileSync
} from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { fileURLToPath } from 'node:url'
@@ -15,6 +22,68 @@ const sourceNodePtyJobOwnershipPath = fileURLToPath(
const sourceWindowsProcessTreeGypRebuildPath = fileURLToPath(
new URL('./windows-process-tree-gyp-rebuild.mjs', import.meta.url)
)
const sourceWindowsProcessTreePatchPath = fileURLToPath(
new URL('../patches/@vscode__windows-process-tree@0.8.0.patch', import.meta.url)
)
/**
* The command-line reader as it is *before* the patch, taken from the patch's
* own pre-image so no upstream copy has to be vendored.
*
* Written back as **CRLF**, which is what `@vscode/windows-process-tree@0.8.0`
* actually ships: all 67 pre-image lines of this file carried a CR before the
* patch was normalized to LF. Rebuilding it with the patch's current newline
* instead would make fixture and patch agree by construction, on any encoding —
* which is exactly how a repair that cannot apply to the real package passed
* this suite.
*/
function unpatchedWindowsProcessTreeCommandLineSource() {
const lines = readFileSync(sourceWindowsProcessTreePatchPath, 'utf8').split('\n')
const start = lines.findIndex((line) =>
line.startsWith('diff --git a/src/process_commandline.cc ')
)
const rest = lines.slice(start + 1)
const end = rest.findIndex((line) => line.startsWith('diff --git '))
const preImage = (end === -1 ? rest : rest.slice(0, end))
.filter((line) => line.startsWith(' ') || line.startsWith('-'))
.filter((line) => !line.startsWith('---'))
.map((line) => line.slice(1).replace(/\r$/, ''))
.join('\r\n')
// Splitting drops the file's own trailing newline as an empty element, and
// `git apply` needs the bytes exact.
return `${preImage}\r\n`
}
/**
* Pin `core.autocrlf` for a spawned repair, whatever the host is set to.
*
* The repair blinds git to the surrounding repo with `GIT_DIR`, so the value it
* sees comes from global/system config — on a Git for Windows box that is
* whichever line-ending option the installer wrote, and `false` (Git's built-in
* default, "checkout as-is") is the one the repair used to fail under. A global
* config in a temp HOME outranks the system file, so this is deterministic
* rather than whatever the developer happens to have.
*/
export function gitLineEndingEnv(autocrlf) {
const home = mkdtempSync(join(tmpdir(), `orca-git-home-${autocrlf}-`))
writeFileSync(join(home, '.gitconfig'), `[core]\n\tautocrlf = ${autocrlf}\n`)
return { HOME: home, USERPROFILE: home }
}
/** Production always runs the repair from inside a work tree; `git apply` behaves differently there. */
export function initGitWorkTree(projectDir) {
for (const args of [['init'], ['config', 'user.email', 'a@b.c'], ['config', 'user.name', 't']]) {
spawnSync('git', args, { cwd: projectDir, encoding: 'utf8' })
}
}
export function writeWindowsProcessTreePatchFile(projectDir) {
mkdirSync(join(projectDir, 'config', 'patches'), { recursive: true })
copyFileSync(
sourceWindowsProcessTreePatchPath,
join(projectDir, 'config', 'patches', '@vscode__windows-process-tree@0.8.0.patch')
)
}
export function mkTempProject() {
const projectDir = mkdtempSync(join(tmpdir(), 'orca-rebuild-native-deps-'))
@@ -143,17 +212,46 @@ if (${JSON.stringify(createExecutable)}) {
)
}
export function writeFakeElectronRebuild(projectDir, { logPathEnv = null } = {}) {
/** Bytes that stand in for a compiled addon's import table. */
const FAKE_ADDON_BYTES = {
clean: 'MZ\0ntdll.dll\0NtQueryInformationProcess\0',
unpatched: 'MZ\0KERNEL32.dll\0ReadProcessMemory\0'
}
/**
* A rebuild that produces nothing leaves no addon to inspect, and the script now
* asserts the binary it just built is a patched one. Emit a stand-in so the
* fixture models a rebuild that actually succeeded. `addon` picks which kind,
* because "produced the upstream reader" and "produced nothing" are both real
* outcomes that assertion has to tell apart.
*/
export function writeFakeElectronRebuild(projectDir, { logPathEnv = null, addon = 'clean' } = {}) {
const rebuildDir = join(projectDir, 'node_modules', '@electron', 'rebuild')
mkdirSync(rebuildDir, { recursive: true })
writeFileSync(join(rebuildDir, 'package.json'), JSON.stringify({ type: 'module' }))
const emitAddon =
addon === 'none'
? ''
: `
const packageDir = join('node_modules', '@vscode', 'windows-process-tree')
if (existsSync(join(packageDir, 'package.json'))) {
mkdirSync(join(packageDir, 'build', 'Release'), { recursive: true })
writeFileSync(
join(packageDir, 'build', 'Release', 'windows_process_tree.node'),
${JSON.stringify(FAKE_ADDON_BYTES[addon])}
)
}`
const emitImports =
addon === 'none'
? ''
: "import { existsSync, mkdirSync, writeFileSync } from 'node:fs'\nimport { join } from 'node:path'\n"
writeFileSync(
join(rebuildDir, 'index.js'),
logPathEnv
? `
import { appendFileSync } from 'node:fs'
export async function rebuild(options) {
${emitImports}
export async function rebuild(options) {${emitAddon}
const logPath = process.env[${JSON.stringify(logPathEnv)}]
if (!logPath) {
return
@@ -171,7 +269,10 @@ export async function rebuild(options) {
)
}
`
: 'export async function rebuild() {}\n'
: `${emitImports}
export async function rebuild() {${emitAddon}
}
`
)
}
@@ -271,12 +372,22 @@ export function writeFakeWindowsProcessTree(projectDir) {
writeFileSync(join(processTreeDir, 'index.js'), 'module.exports = {}\n')
}
export function writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir) {
export function writeFakeWindowsProcessTreeWithNodeAddonApi(
projectDir,
{ commandLinePatchApplied = true } = {}
) {
const processTreeDir = join(projectDir, 'node_modules', '@vscode', 'windows-process-tree')
const nodeAddonApiDir = join(processTreeDir, 'node_modules', 'node-addon-api')
mkdirSync(nodeAddonApiDir, { recursive: true })
writeFileSync(join(processTreeDir, 'package.json'), '{"dependencies":{"node-addon-api":"*"}}\n')
writeFileSync(join(processTreeDir, 'index.js'), 'module.exports = {}\n')
mkdirSync(join(processTreeDir, 'src'), { recursive: true })
writeFileSync(
join(processTreeDir, 'src', 'process_commandline.cc'),
commandLinePatchApplied
? '// kProcessCommandLineInformation = 60\n'
: unpatchedWindowsProcessTreeCommandLineSource()
)
writeFileSync(join(nodeAddonApiDir, 'package.json'), '{"name":"node-addon-api"}\n')
writeFileSync(join(nodeAddonApiDir, 'napi.h'), '// napi.h\n')
writeFileSync(join(nodeAddonApiDir, 'napi-inl.h'), '// napi-inl.h\n')
@@ -0,0 +1,103 @@
import { spawn } from 'node:child_process'
import { appendFileSync, copyFileSync, existsSync, mkdirSync } from 'node:fs'
import { createRequire } from 'node:module'
import { join } from 'node:path'
import { describe, expect, it } from 'vitest'
import { removeTreeSync } from '../../src/shared/windows-transient-lock-removal.ts'
import {
mkTempProject,
runRebuildScript,
writeFakeElectronRebuild,
writeFakeNodePtyConptyPayload,
writeFakeUsableElectronPackage,
writeFakeWindowsProcessTreeWithNodeAddonApi
} from './rebuild-native-deps-test-fixtures.mjs'
const require = createRequire(import.meta.url)
/** A real loadable addon, so the OS holds the same lock a running Orca holds. */
function repoAddonPath() {
try {
const entry = require.resolve('@vscode/windows-process-tree')
const built = join(entry, '..', '..', 'build', 'Release', 'windows_process_tree.node')
return existsSync(built) ? built : null
} catch {
return null
}
}
/**
* Stage a stale addon and keep it loaded, exactly as a running Orca does.
*
* The bytes are the repo's own patched build with the flagged import appended,
* because the guard keys on that symbol and the patched binary does not carry
* it. Trailing bytes are PE overlay, so the file still loads.
*/
async function stageLoadedStaleAddon(projectDir) {
const source = repoAddonPath()
const releaseDir = join(
projectDir,
'node_modules',
'@vscode',
'windows-process-tree',
'build',
'Release'
)
mkdirSync(releaseDir, { recursive: true })
const stale = join(releaseDir, 'windows_process_tree.node')
copyFileSync(source, stale)
appendFileSync(stale, 'ReadProcessMemory')
const holder = spawn(
process.execPath,
['-e', 'require(process.argv[1]); process.send("held"); setInterval(() => {}, 1000)', stale],
{ stdio: ['ignore', 'ignore', 'ignore', 'ipc'] }
)
await new Promise((resolve, reject) => {
holder.once('message', resolve)
holder.once('exit', () => reject(new Error('the addon holder exited before loading')))
})
return holder
}
// Why an end-to-end run: the defect was purely one of placement. The guard threw
// a real EPERM, and the classifier that turns that into "close running Orca"
// already existed -- the throw simply happened before the try that reaches it.
// Only the whole script exercises that.
describe.runIf(process.platform === 'win32')('rebuild-native-deps stale addon under lock', () => {
it.skipIf(!repoAddonPath())(
'reports a locked stale addon as a Windows file lock instead of an EPERM stack',
async () => {
const projectDir = mkTempProject()
let holder
try {
writeFakeUsableElectronPackage(projectDir, { platform: 'win32' })
writeFakeElectronRebuild(projectDir)
writeFakeNodePtyConptyPayload(projectDir, process.arch)
writeFakeWindowsProcessTreeWithNodeAddonApi(projectDir)
holder = await stageLoadedStaleAddon(projectDir)
const result = runRebuildScript(
projectDir,
{
npm_lifecycle_event: 'postinstall',
npm_config_platform: 'win32',
npm_config_arch: process.arch
},
['--platform=win32', `--arch=${process.arch}`, '--force']
)
expect(result.stderr).toContain(
'Close running Orca/Electron/dev processes for this worktree'
)
// Non-strict postinstall soft-exits on a lock; the next dev/start re-checks.
expect(result.status, result.stderr).toBe(0)
} finally {
holder?.kill()
removeTreeSync(projectDir)
}
}
)
})
+55 -9
View File
@@ -20,7 +20,12 @@
import { rebuild } from '@electron/rebuild'
import { execFileSync, spawnSync } from 'node:child_process'
import { stageWindowsProcessTreeNodeAddonApiHeaders } from './windows-process-tree-gyp-rebuild.mjs'
import {
ensureWindowsProcessTreeCommandLinePatch,
inspectWindowsProcessTreeAddon,
stageWindowsProcessTreeNodeAddonApiHeaders,
windowsProcessTreeAddonPath
} from './windows-process-tree-gyp-rebuild.mjs'
import {
copyFileSync,
existsSync,
@@ -141,15 +146,21 @@ if (!ignoreModules.includes('cpu-features')) {
}
}
if (
rebuildPlatform === 'win32' &&
modulesToRebuild.includes('@vscode/windows-process-tree') &&
existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json'))
) {
stageWindowsProcessTreeNodeAddonApiHeaders()
}
try {
// Why inside the try: the patch guard deletes a stale addon binary, and that
// delete fails EPERM when the addon is loaded -- exactly the running-Orca case
// the catch below is written for. Outside, it aborted `pnpm install` with a
// raw stack instead of the "close running Orca/Electron processes" message.
if (
rebuildPlatform === 'win32' &&
modulesToRebuild.includes('@vscode/windows-process-tree') &&
existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json'))
) {
stageWindowsProcessTreeNodeAddonApiHeaders()
if (ensureWindowsProcessTreeCommandLinePatch()) {
console.warn('[rebuild] Repaired the un-applied windows-process-tree command-line patch.')
}
}
await rebuild({
buildPath: projectDir,
electronVersion,
@@ -165,6 +176,7 @@ try {
force: true
})
restoreNodePtyWindowsConptyRuntime()
assertWindowsProcessTreeAddonIsPatched()
} catch (/** @type {any} */ err) {
console.error('[rebuild] Native module rebuild failed:', err?.message ?? err)
if (isWindowsNativeLockError(err)) {
@@ -184,6 +196,40 @@ try {
process.exit(1)
}
/**
* The binary this rebuild just produced is the one the packaged app ships.
*
* The relay build asserts its own artifact and `ensure-native-runtime.mjs`
* asserts what it loads, but nothing checked the addon that gets copied into the
* packaged `node_modules` -- so a rebuild that silently produced the upstream
* reader would reach users. Anything but `clean` fails: after a rebuild that
* reported success the binary must exist, so `missing` is a broken build, not an
* absence to shrug at. This is the caller that needs the state to be a state and
* not a boolean.
*/
function assertWindowsProcessTreeAddonIsPatched() {
if (
rebuildPlatform !== 'win32' ||
!modulesToRebuild.includes('@vscode/windows-process-tree') ||
!existsSync(join(projectDir, 'node_modules', '@vscode', 'windows-process-tree', 'package.json'))
) {
return
}
const addonPath = windowsProcessTreeAddonPath()
const state = inspectWindowsProcessTreeAddon(addonPath)
if (state === 'clean') {
return
}
throw new Error(
state === 'missing'
? `the rebuild reported success but ${addonPath} is not there, so the packaged app would ` +
'ship no windows-process-tree addon at all.'
: `${addonPath} still imports ReadProcessMemory, so it was not built from the patched ` +
'command-line reader. The packaged app would carry the primitive MDE scores as ' +
'credential dumping.'
)
}
function restoreNodePtyWindowsConptyRuntime() {
if (rebuildPlatform !== 'win32' || !onlyModules.includes('node-pty')) {
return
@@ -0,0 +1,62 @@
#!/usr/bin/env node
import assert from 'node:assert/strict'
import { readFileSync } from 'node:fs'
import { stripTypeScriptTypes } from 'node:module'
import { performance } from 'node:perf_hooks'
// Pass the pre-change source saved with git show <base>:src/shared/relay-frame-buffer.ts.
const baselinePath = process.argv[2]
if (!baselinePath) {
throw new Error('Usage: node config/scripts/relay-frame-buffer-benchmark.mjs <baseline.ts>')
}
async function load(source) {
return (
await import(
`data:text/javascript;base64,${Buffer.from(stripTypeScriptTypes(source)).toString('base64')}`
)
).RelayFrameBuffer
}
const Before = await load(readFileSync(baselinePath, 'utf8'))
const After = await load(
readFileSync(new URL('../../src/shared/relay-frame-buffer.ts', import.meta.url), 'utf8')
)
function median(values) {
return values.sort((a, b) => a - b)[Math.floor(values.length / 2)]
}
for (const count of [1, 256, 16384, 65536]) {
const chunks = Array.from({ length: count }, (_, index) => Buffer.alloc(64, index % 256))
const expected = Buffer.concat(chunks)
for (const mode of ['take', 'discard']) {
const times = [[], []]
for (let round = 0; round < 9; round += 1) {
for (const arm of round % 2 === 0 ? [0, 1] : [1, 0]) {
const FrameBuffer = arm === 0 ? Before : After
const buffer = new FrameBuffer()
for (const chunk of chunks) {
buffer.append(chunk)
}
const start = performance.now()
const output = buffer[mode](expected.length)
times[arm].push(performance.now() - start)
if (mode === 'take') {
assert.deepEqual(output, expected)
}
assert.equal(buffer.length, 0)
buffer.append(Buffer.from('tail'))
assert.equal(buffer.drain().toString(), 'tail')
}
}
const beforeMs = median(times[0]),
afterMs = median(times[1])
console.log(
JSON.stringify({
mode,
chunks: count,
bytes: expected.length,
beforeMs,
afterMs,
speedup: beforeMs / afterMs
})
)
}
}
@@ -0,0 +1,55 @@
import assert from 'node:assert/strict'
import { performance } from 'node:perf_hooks'
import { extractIconHref } from '../../src/main/repo-icon-source-href.ts'
// Original production expressions, preserved for the before/after measurement.
const html =
/<link\b(?=[^>]*\brel=["'](?:icon|shortcut icon)["'])(?=[^>]*\bhref=["']([^"'?]+))[^>]*>/i
const object =
/(?=[^}]*\brel\s*:\s*["'](?:icon|shortcut icon)["'])(?=[^}]*\bhref\s*:\s*["']([^"'?]+))[^}]*/i
const original = (source) => source.match(html)?.[1] ?? source.match(object)?.[1] ?? null
function measurePair(source) {
original(source)
extractIconHref(source)
const beforeSamples = []
const afterSamples = []
for (let run = 0; run < 5; run++) {
const measurements = [
[original, beforeSamples],
[extractIconHref, afterSamples]
]
if (run % 2 === 1) {
measurements.reverse()
}
for (const [fn, samples] of measurements) {
const started = performance.now()
fn(source)
samples.push(performance.now() - started)
}
}
return {
beforeMs: beforeSamples.sort((a, b) => a - b)[2],
afterMs: afterSamples.sort((a, b) => a - b)[2]
}
}
const results = []
for (const size of [8192, 16384, 32768]) {
for (const shape of ['no icon', 'rel without href', 'unterminated link starts']) {
const source =
shape === 'unterminated link starts'
? '<link '.repeat(Math.floor(size / 6))
: 'a'.repeat(size) + (shape === 'rel without href' ? ' rel:"icon"' : '')
assert.equal(extractIconHref(source), original(source))
const { beforeMs, afterMs } = measurePair(source)
results.push({
shape,
bytes: Buffer.byteLength(source),
beforeMs,
afterMs,
speedup: beforeMs / afterMs
})
}
}
console.log(JSON.stringify({ node: process.version, platform: process.platform, results }, null, 2))
@@ -0,0 +1,79 @@
import assert from 'node:assert/strict'
import { execFileSync } from 'node:child_process'
import { stripTypeScriptTypes } from 'node:module'
import { performance } from 'node:perf_hooks'
import { blankStringContents as after } from '../../src/shared/source-scan/source-tree-scan.ts'
const ref = process.argv[2]
if (!ref) {
throw new Error('Usage: node config/scripts/source-string-blanking-benchmark.mjs <baseline-ref>')
}
const source = execFileSync('git', ['show', `${ref}:src/shared/source-scan/source-tree-scan.ts`], {
encoding: 'utf8'
})
const { blankStringContents: before } = await import(
`data:text/javascript;base64,${Buffer.from(stripTypeScriptTypes(source)).toString('base64')}`
)
const tokens = [
'a',
'/',
'*',
' ',
'\n',
'\r',
'\t',
'\u00a0',
'\u2028',
'"',
"'",
'`',
'${',
'}',
'{',
'\\',
'(',
')',
'[',
']',
'=',
'+',
'-',
';'
]
let seed = 173
for (let sample = 0; sample < 3000; sample++) {
let input = ''
for (let token = 0; token < 40; token++) {
seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0
input += tokens[seed % tokens.length]
}
assert.equal(after(input), before(input), JSON.stringify(input))
assert.equal(after(input, true), before(input, true), JSON.stringify(input))
}
function measure(fn, input) {
const samples = []
for (let run = 0; run < 3; run++) {
const start = performance.now()
fn(input)
samples.push(performance.now() - start)
}
return samples.sort((a, b) => a - b)[1]
}
const results = []
for (const lines of [100, 1000, 5000, 10000]) {
const input = 'const x = value / 2;\n'.repeat(lines)
assert.equal(after(input), before(input))
results.push({
lines,
bytes: Buffer.byteLength(input),
beforeMs: measure(before, input),
afterMs: measure(after, input)
})
}
console.log(
JSON.stringify(
{ node: process.version, platform: process.platform, differentialCases: 3000, results },
null,
2
)
)
@@ -9,7 +9,8 @@
* hop escapes the store and configure fails with "node_addon_api.gyp not
* found" (run 32999886072).
*/
import { copyFileSync, mkdirSync, realpathSync } from 'node:fs'
import { execFileSync } from 'node:child_process'
import { copyFileSync, existsSync, mkdirSync, readFileSync, realpathSync, rmSync } from 'node:fs'
import { createRequire } from 'node:module'
import { dirname, join, resolve } from 'node:path'
@@ -22,6 +23,16 @@ export const WINDOWS_PROCESS_TREE_PACKAGE_DIR = join(
'windows-process-tree'
)
export const WINDOWS_PROCESS_TREE_PATCH_PATH = join(
ROOT,
'config',
'patches',
'@vscode__windows-process-tree@0.8.0.patch'
)
/** Only the patched reader defines this; the upstream one walks the PEB. */
const COMMAND_LINE_PATCH_MARKER = 'kProcessCommandLineInformation'
export const WINDOWS_PROCESS_TREE_NODE_ADDON_API_HEADERS = [
'napi.h',
'napi-inl.h',
@@ -39,6 +50,119 @@ export function nodeGypRebuildInvocation(arch, packageDir = WINDOWS_PROCESS_TREE
}
}
/** The binary the addon actually loads. */
export function windowsProcessTreeAddonPath(packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR) {
return join(packageDir, 'build', 'Release', 'windows_process_tree.node')
}
/** The import whose absence tells the patched binary from the published prebuilt. */
const FLAGGED_IMPORT = 'ReadProcessMemory'
/**
* Does this compiled addon still carry the flagged primitive?
*
* The patched reader never calls `ReadProcessMemory`, so the symbol is absent
* from its import table; the upstream build imports it. That makes this a
* property of the binary rather than of the source next to it, which matters
* because the published tarball ships a *loadable* prebuilt built from
* unpatched source: it is node-addon-api, so it satisfies a bare `require()`
* under both Node and Electron, and a skipped rebuild would use it.
*
* Tri-state, not a predicate: a binary that is not there has not been cleared,
* and a boolean makes "absent" indistinguishable from "verified clean" at every
* call site. Takes the binary path so the relay's staged addon -- which sits
* beside the bundle, with no package around it -- gets the same check.
*
* @param {string} addonPath
* @returns {'clean' | 'unpatched' | 'missing'}
*/
export function inspectWindowsProcessTreeAddon(addonPath) {
if (!existsSync(addonPath)) {
return 'missing'
}
return readFileSync(addonPath).includes(FLAGGED_IMPORT) ? 'unpatched' : 'clean'
}
/**
* Refuse to compile or load the upstream command-line reader.
*
* Unpatched, it opens every process with `PROCESS_VM_READ` and walks the PEB to
* recover the command line -- the primitive MDE scores as credential dumping,
* and the reason this package is patched at all. pnpm has been seen
* materializing this CRLF package with its patch missing, so repair the source
* from the patch file, and drop any binary that predates the repair.
*/
export function ensureWindowsProcessTreeCommandLinePatch(
packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR
) {
const source = join(packageDir, 'src', 'process_commandline.cc')
if (!existsSync(source)) {
throw new Error(
`${source} is missing, so the command-line patch cannot be verified. Run pnpm install.`
)
}
let repaired = false
if (!readFileSync(source, 'utf8').includes(COMMAND_LINE_PATCH_MARKER)) {
try {
execFileSync(
'git',
[
// Why force the line-ending mode: the patch is stored LF (a contract
// test forbids CR bytes in it), but upstream ships this source CRLF,
// so its pre-image lines and the file's differ by a CR. Under
// `core.autocrlf=false` -- Git's own built-in default, and what
// "checkout as-is" selects in the Git for Windows installer -- git
// compares them literally, the hunk does not match, and the repair
// throws. `input` normalizes line endings for that comparison and
// nothing else, so a hunk whose real content drifted is still
// rejected. Measured: without it, apply exits 1 at autocrlf=false and
// 0 at true/input; with it, 0 for CRLF and LF sources under all three.
'-c',
'core.autocrlf=input',
'apply',
'--include=src/process_commandline.cc',
WINDOWS_PROCESS_TREE_PATCH_PATH
],
{
cwd: realpathSync(packageDir),
stdio: 'pipe',
// Why blind git to the repo: run inside a work tree, `git apply`
// prefixes patch paths with the cwd-relative prefix, silently skips
// everything that does not match -- and still exits 0. The package
// dir is always under the project root, so without this the repair
// reports success and changes nothing.
env: { ...process.env, GIT_DIR: join(packageDir, '.orca-no-such-git-dir') }
}
)
} catch (error) {
throw new Error(
'src/process_commandline.cc still reads the PEB, and repairing it from ' +
`${WINDOWS_PROCESS_TREE_PATCH_PATH} failed: ${error?.message ?? error}. Run pnpm install.`
)
}
if (!readFileSync(source, 'utf8').includes(COMMAND_LINE_PATCH_MARKER)) {
throw new Error(
'src/process_commandline.cc still reads the PEB after repair, so the patch did not ' +
'apply. Run pnpm install.'
)
}
repaired = true
}
// A binary from before the repair -- or the tarball's own prebuilt -- would
// otherwise survive a skipped rebuild and load the flagged reader anyway.
// Deleting it can fail EPERM against a loaded (memory-mapped) addon, which
// `force: true` does not cover -- it only swallows ENOENT. That throw is the
// caller's to classify as a Windows file lock, so it must not be swallowed.
if (inspectWindowsProcessTreeAddon(windowsProcessTreeAddonPath(packageDir)) === 'unpatched') {
rmSync(windowsProcessTreeAddonPath(packageDir), { force: true })
repaired = true
}
return repaired
}
// Patched binding.gyp includes deps/node-addon-api; the tarball does not ship those headers.
export function stageWindowsProcessTreeNodeAddonApiHeaders(
packageDir = WINDOWS_PROCESS_TREE_PACKAGE_DIR
@@ -10,8 +10,9 @@ import {
} from 'node:fs'
import { tmpdir } from 'node:os'
import { join, resolve } from 'node:path'
import { describe, expect, it } from 'vitest'
import { afterEach, beforeEach, describe, expect, it } from 'vitest'
import {
inspectWindowsProcessTreeAddon,
nodeGypRebuildInvocation,
stageWindowsProcessTreeNodeAddonApiHeaders,
WINDOWS_PROCESS_TREE_NODE_ADDON_API_HEADERS,
@@ -59,3 +60,40 @@ describe('windows-process-tree node-gyp rebuild', () => {
}
})
})
describe('inspecting a compiled windows-process-tree addon', () => {
let dir
beforeEach(() => {
dir = mkdtempSync(join(tmpdir(), 'orca-windows-process-tree-addon-'))
})
afterEach(() => {
rmSync(dir, { recursive: true, force: true })
})
it('reports a binary that still imports ReadProcessMemory as unpatched', () => {
const addonPath = join(dir, 'windows_process_tree.node')
writeFileSync(addonPath, Buffer.from('MZ\0\0KERNEL32.dll\0ReadProcessMemory\0', 'binary'))
expect(inspectWindowsProcessTreeAddon(addonPath)).toBe('unpatched')
})
it('reports a binary without the import as clean', () => {
const addonPath = join(dir, 'windows_process_tree.node')
writeFileSync(addonPath, Buffer.from('MZ\0\0ntdll.dll\0NtQueryInformationProcess\0', 'binary'))
expect(inspectWindowsProcessTreeAddon(addonPath)).toBe('clean')
})
// The whole point of the tri-state: absence is not evidence of safety, and a
// boolean made "there is no binary" indistinguishable from "checked, clean".
it('reports an absent binary as missing rather than clean', () => {
expect(inspectWindowsProcessTreeAddon(join(dir, 'windows_process_tree.node'))).toBe('missing')
})
it('inspects whatever path it is handed, including a relay-staged addon', () => {
// The relay loads `./windows-process-tree.node` beside its bundle, which is
// nowhere near a node_modules package directory.
const staged = join(dir, 'windows-process-tree.node')
writeFileSync(staged, Buffer.from('MZ\0\0ReadProcessMemory\0', 'binary'))
expect(inspectWindowsProcessTreeAddon(staged)).toBe('unpatched')
})
})
@@ -15,6 +15,16 @@ const REF_MIRRORS = [
]
describe('ref-mirroring vet steps', () => {
it('keeps the full-history adhoc checkout on the same case-safe backend', () => {
const steps = readWorkflow('.github/workflows/adhoc-mac-build.yml').jobs['build-adhoc-mac']
.steps
const checkout = steps.find((step) => step.name === 'Checkout the requested ref')
expect(checkout.env.GIT_DEFAULT_REF_FORMAT).toBe('reftable')
expect(checkout.with.ref).toBe('${{ steps.vetted.outputs.sha }}')
expect(checkout.with['fetch-depth']).toBe(0)
expect(checkout.with['persist-credentials']).toBe(false)
})
// Why: macOS and Windows runner disks are case-insensitive, and this repo has
// branches that differ only in casing. The files backend cannot store both, and
// it fails the whole fetch rather than the one ref — so the vet step dies before
@@ -0,0 +1,125 @@
import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import { pathToFileURL } from 'node:url'
import { afterAll, beforeAll, describe, expect, it } from 'vitest'
import { parse } from 'yaml'
import { runProcess } from '../../src/shared/child-process/run-process'
const readWorkflow = (name) => parse(readFileSync(`.github/workflows/${name}.yml`, 'utf8'))
const windowsVet = readWorkflow('dev-channel-win-build').jobs['build-win'].steps.find(
(step) => step.id === 'vetted'
)
const macSteps = readWorkflow('adhoc-mac-build').jobs['build-adhoc-mac'].steps
const macVet = macSteps.find((step) => step.id === 'vetted')
const macCheckout = macSteps.find((step) => step.name === 'Checkout the requested ref')
const directory = mkdtempSync(join(tmpdir(), 'workflow-ref-reachability-'))
const repository = join(directory, 'remote.git')
const identity = {
...process.env,
GIT_AUTHOR_NAME: 'Ref test',
GIT_AUTHOR_EMAIL: 'ref-test@example.com',
GIT_COMMITTER_NAME: 'Ref test',
GIT_COMMITTER_EMAIL: 'ref-test@example.com'
}
let ancestor, upper, lower, untrusted
async function git(args, env = identity) {
const result = await runProcess({ program: 'git', args, env })
expect(result.code, result.stderr).toBe(0)
return result.stdout.trim()
}
beforeAll(async () => {
await git(['init', '--bare', '--ref-format=reftable', repository])
const tree = await git(['-C', repository, 'mktree'])
ancestor = await git(['-C', repository, 'commit-tree', tree, '-m', 'ancestor'])
upper = await git(['-C', repository, 'commit-tree', tree, '-p', ancestor, '-m', 'upper'])
lower = await git(['-C', repository, 'commit-tree', tree, '-p', ancestor, '-m', 'lower'])
untrusted = await git(['-C', repository, 'commit-tree', tree, '-m', 'PR only'])
for (const [ref, sha] of [
['refs/heads/Fix', upper],
['refs/heads/fix', lower],
['refs/pull/1/head', untrusted]
]) {
await git(['-C', repository, 'update-ref', ref, sha])
}
await git(['-C', repository, 'tag', '-a', 'Release', upper, '-m', 'upper tag'])
await git(['-C', repository, 'tag', '-a', 'release', lower, '-m', 'lower tag'])
await git(['-C', repository, 'config', 'uploadpack.allowFilter', 'true'])
})
afterAll(() => rmSync(directory, { recursive: true, force: true }))
async function vet(step, ref) {
const scratch = mkdtempSync(join(directory, 'attempt-'))
const script = join(scratch, 'vet.sh')
writeFileSync(script, step.run)
return runProcess({
program: 'bash',
args: [script],
env: {
...identity,
REPO_URL: pathToFileURL(repository).href,
RUNNER_TEMP: scratch,
GITHUB_OUTPUT: join(scratch, 'output'),
REQUESTED_REF: ref,
REQUESTED_SHA: ref,
CHANNEL: 'hourly',
TAG: 'v1.0.0-hourly.test',
VERSION: '1.0.0-hourly.test'
}
})
}
describe('release ref trust with case-twin names', () => {
it('accepts both branch tips, annotated tags, and their common ancestor', async () => {
for (const sha of [upper, lower, ancestor]) {
const result = await vet(windowsVet, sha)
expect(result.code, result.stderr).toBe(0)
}
for (const ref of ['Fix', 'fix', 'Release', 'release', ancestor]) {
const result = await vet(macVet, ref)
expect(result.code, result.stderr).toBe(0)
}
})
it('rejects PR-only commits even when the server has their objects', async () => {
for (const step of [windowsVet, macVet]) {
const result = await vet(step, untrusted)
expect(result.code).not.toBe(0)
expect(result.stdout).toContain('not reachable from any branch or tag')
}
const result = await vet(macVet, 'refs/pull/1/head')
expect(result.code).not.toBe(0)
expect(result.stdout).toContain('Refusing to build PR ref')
})
it('preserves both case variants in the subsequent full-history checkout', async () => {
const checkout = join(directory, 'checkout')
const env = { ...identity, ...macCheckout.env }
await git(['init', checkout], env)
await git(
[
'-C',
checkout,
'fetch',
'--no-tags',
repository,
'+refs/heads/*:refs/remotes/origin/*',
'+refs/tags/*:refs/tags/*'
],
env
)
await git(['-C', checkout, 'checkout', '--detach', upper], env)
for (const [ref, sha] of [
['refs/remotes/origin/Fix', upper],
['refs/remotes/origin/fix', lower],
['refs/tags/Release', upper],
['refs/tags/release', lower]
]) {
expect(await git(['-C', checkout, 'rev-parse', `${ref}^{commit}`], env)).toBe(sha)
}
expect(await git(['-C', checkout, 'rev-parse', 'HEAD'], env)).toBe(upper)
})
})
+66
View File
@@ -131,3 +131,69 @@ environments currently exist. An environment-gated design adds a GitHub
approval after each SignPath approval and changes the current automatic inner
signing timeout fallback; those are explicit release-policy decisions, so this
PR leaves production signing behavior unchanged.
## Second audit and hosted trials
- Cloud Verify ran 100 times in a sampled 39-hour window (84 PR and 16 push
runs). Move its four Ubuntu 22.04 jobs from Blacksmith to standard hosted
Ubuntu 22.04, preserving Postgres, secret scanning, build, tests, and Terraform
validation. Baseline [34001538145](https://github.com/stablyai/orca/actions/runs/34001538145)
used 64/72/26/19 seconds for security/test/build/Terraform respectively.
This conserves the shared provider allowance; hosted latency must be checked.
- Keep full tag history for the 13-job skill round-trip matrix, but fetch blobs
lazily. Only two historical SKILL.md files are materialized. Baseline
[33999994876](https://github.com/stablyai/orca/actions/runs/33999994876)
spent 42–84 seconds per checkout, about 14 aggregate runner minutes. A hosted
trial must verify historical blob fetches on all three operating systems.
- Use the existing Electron/native dependency cache for native IME CI. Keep
both deterministic boundary and real IBus tests. Add pnpm store caching to
terminal perf and release golden/evidence lanes; retain their raw installs
because manually selected older refs may not contain the shared action.
- Disable ZIP recompression only for already-compressed NSIS installers sent
to SignPath. Installer contents, release compression, and signing stay intact.
- Advance existing placement and startup deadlines with scoped fake timers in
three renderer test files. All 34 tests pass in 62 ms of local test execution,
versus 65.182 seconds in the sampled hosted baseline. Imports and transforms
still dominate invocation time; this is not a claim of equal PR wall savings.
Eight unit shards already have balanced 260–296-second sample durations.
Reducing shards or removing test isolation lacks evidence of a net gain. Real
subprocess tests intentionally cover lifecycle behavior and retain real clocks.
The 14-way E2E split retains headroom after earlier 12-way timeouts. Lowering
coverage or schedule frequency is outside this efficiency pass. Cache complexity
for a seven-second docs install is unlikely to pay back. Release build reuse
across modes risks differing telemetry identities and native platform artifacts.
Terminal Perf's baseline [33955846492](https://github.com/stablyai/orca/actions/runs/33955846492)
failed waiting 30 seconds for workspaceSessionReady in its shared-page fixture,
before measuring terminal performance. Compare hosted trials against that known
failure rather than attributing it to dependency cache changes.
Hosted trials for the second audit:
- [Cloud Verify 34002295216](https://github.com/stablyai/orca/actions/runs/34002295216)
passed all four jobs on standard hosted Ubuntu: security 57s, test 102s, build
35s, Terraform 19s. The test lane is 30s slower than the Blacksmith sample;
retain this modest latency tradeoff to conserve shared allowance.
- [Skill matrix 34002295221](https://github.com/stablyai/orca/actions/runs/34002295221)
passed all 13 legs, including historical blob materialization. Checkout took
18–20s on Linux, 39–45s on macOS, and 49–58s on Windows, versus the earlier
42–84s range across platforms. These are observational samples.
- [Native IME 34002299594](https://github.com/stablyai/orca/actions/runs/34002299594)
passed both deterministic and real IBus checks. Shared dependency setup took
29s, versus 35s for the old install/toolchain steps in the sampled baseline.
- Native-IME-only source/spec changes no longer allocate the reusable E2E
build, cache, and consumer jobs just to filter out the native spec. The
separate native workflow still runs; SSH-only and mixed spec lists still
allocate the reusable workflow. Routing contracts exercise these cases.
- [Hourly 34001816449](https://github.com/stablyai/orca/actions/runs/34001816449)
exercised the new five-second preflight and successfully published macOS.
The Windows follow-up failed in its unchanged input-vetting fetch because
remote refs differ only by case on its case-insensitive filesystem. The
requested SHA was correct; this does not validate an unchanged-main skip yet.
Moving the daily Mac freshness check has lower expected value than hourly:
only one potential idle allocation per day, and active development usually
requires that build. Defer another release-graph change until skip frequency
justifies it. The substantive remaining release occupancy opportunity is the
separately documented asynchronous signing policy decision.
@@ -0,0 +1,118 @@
# Windows daemon-host relocation
On Windows the terminal daemon does not run from the install directory. Before it forks the
daemon, Orca materializes a trimmed copy of its own runtime under
`%LOCALAPPDATA%\Orca\daemon-host\<app version>\` and forks the daemon from there
(`src/main/daemon/daemon-host-relocation.ts`). This is what keeps live terminals alive across an
auto-update and across a crash of the main process.
Read this before changing the copy plan, the host exe name, the LOCALAPPDATA layout, or
`config/nsis/orca-installer-hooks.nsh`.
## What the relocation actually escapes
The killer is **electron-builder's process sweep, matched on image path** — not file deletion.
Windows will not delete a running image, so `RMDir /r "$INSTDIR"` cannot end the daemon on its own.
In app-builder-lib's `allowOnlyOneInstallerInstance.nsh`, `FIND_PROCESS` / `KILL_PROCESS` have two
branches:
| Branch | Condition | Selector |
| -------- | --------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| Primary | `powershell.exe` runs, `Get-CimInstance` resolves, and `Get-ExecutionPolicy -Scope Process` is not `Restricted` | `Win32_Process` where `$_.Path.StartsWith('$INSTDIR', 'CurrentCultureIgnoreCase')` — **path-scoped** |
| Fallback | otherwise | per-user: `taskkill /F /IM "<AppName>.exe" /FI "PID ne $pid" /FI "USERNAME eq %USERNAME%"`; per-machine: the same without the username filter — **image-name-scoped** |
The probe reads the **process** scope, not the effective policy, and Group Policy writes
`MachinePolicy`/`UserPolicy` — so a GPO-managed host whose effective policy is `Restricted` still
exits 0 and takes the primary branch. The fallback is reached only when `powershell.exe` is absent,
`Get-CimInstance` does not resolve, PowerShell is blocked outright (WDAC/AppLocker, Server Core), or
an inherited `PSExecutionPolicyPreference=Restricted` is in the environment.
So on essentially every machine the sweep is path-scoped, and a daemon whose image lives under
`%LOCALAPPDATA%` is out of range regardless of what the file is called. **Survival is a property of
the path.** The name only matters on the fallback branch.
## Why the exe is copied verbatim (and not renamed)
The host exe keeps the app exe's own file name (`daemonHostExeName()` returns
`basename(process.execPath)`), so the relocated image is a byte-for-byte copy of the app binary
under its original name.
An earlier revision copied it as `orca-terminal-daemon.exe` specifically so the fallback
`taskkill /IM Orca.exe` could not match. That bought survival on the rare no-PowerShell host and
cost a textbook defence-evasion signature: _a process copies its own image into a user-writable
directory under a different name so a kill-by-image-name cannot match it, then runs detached and
survives the installer._ Microsoft Defender for Endpoint flagged it as MITRE **T1036
(Masquerading)**, and — because it is the process every other flagged action is attributed to — it
acted as a reputation multiplier on unrelated findings. No VS Code fork does this.
Trading the fallback branch for the name is the right trade:
- On the primary branch nothing changes: the daemon still survives the update.
- On the fallback branch the daemon is killed with the app and terminals **cold-restore** on
relaunch. That is the documented pre-relocation behaviour, a first-class outcome the update
harness already asserts (`--expect cold-restore`), not a failure.
- Relocation is fail-open end to end anyway: any materialization failure returns `null` and the
caller forks the install-dir host.
One new failure mode comes with it, on the fallback branch only. The daemon now matches
`FIND_PROCESS` under the app's image name, so it enters electron-builder's retry loop
(`allowOnlyOneInstallerInstance.nsh:136-141`). If the `taskkill` there fails to end it — an elevated
or otherwise unkillable host — the loop reaches `MessageBox ... /SD IDCANCEL` and `Quit`s, aborting a
silent update rather than completing it. Under the old distinct name the daemon was invisible to
that loop. Low probability (fallback branch _and_ an unkillable daemon), but it is a real new path.
What this does **not** buy. Two things bound the win honestly:
- The strongest T1036 indicator is a PE-resource-vs-disk-name mismatch, and it was **never firing**:
the shipped binary's `OriginalFilename` is empty (only `InternalName = Orca` is set), so there was
no embedded name for the old disk name to contradict.
- The remaining behaviour — a signed app copying its own ~225 MB image into user-writable
`%LOCALAPPDATA%` and running it detached under `ELECTRON_RUN_AS_NODE=1` — is still execution from
a non-standard user-writable location, which maps to **T1036.005** and is a standard heuristic on
its own.
So this removes a real but partial signal. Expect the score to drop; do not expect the process to
stop being scored.
## Options that were rejected
| Option | Why not |
| ----------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| Materialize the tree from the NSIS installer | The daemon host is ~246 MB. Writing it at install time doubles install footprint and lengthens the window in which the app is down during a silent update. Worse, on a per-machine install (`INSTALL_MODE_PER_ALL_USERS`) the installer runs as the installing admin, so `$LOCALAPPDATA` is the wrong user's — every other user still needs the runtime path, which means the runtime self-copy stays in the product and the signal is only made rarer. |
| Ship a second signed `orca-terminal-daemon.exe` in the installer | `Orca.exe` is 235,555,328 bytes (224.6 MiB). electron-builder's NSIS uses solid LZMA with a 64 MB dictionary, so a second copy 224 MB downstream does not dedupe; the compressed installer grows by roughly a whole compressed Electron binary, paid by every user on every update download. It also does not remove the runtime copy — the helper still has to reach `%LOCALAPPDATA%` to escape the sweep — so it buys the same signal reduction as the verbatim copy at a large download cost. |
| Override `customCheckAppRunning` to force a path-scoped kill on both branches | Cheap to write (~6 lines: `!include "getProcessInfo.nsh"`, `Var pid`, and a macro that pins `IsPowerShellAvailable`, reusing upstream's dialog, retry loop and elevated handling) — but wrong at any size. Forcing the PowerShell branch on a host where PowerShell is genuinely absent makes `FIND_PROCESS` and `KILL_PROCESS` silently no-op, so the installer proceeds with the **real app** still running and its files in use. That is a worse outcome than the cold restore it would prevent, so this is not worth doing ever, not merely not now. |
| Hardlink instead of copy | Avoids the 246 MB entirely and is not a "copy" at all, but is NTFS-and-same-volume-only and introduces fresh failure modes (link counts, AV interception, cross-volume installs). Worth revisiting deliberately, not as part of a signal fix. |
## Invariants to preserve
- The host exe name is **derived from `process.execPath`**, never a literal. A future
`executableName` or dev-channel rename must follow automatically; pinning a name of our own is
how the mismatch creeps back.
- The daemon is identified by **PID and command line**, never by image name — in the product
(`daemon-pid-file-parse`, `daemon-process-inspection`) and in the harness
(`tests/tools/win-update-e2e/daemon-processes.mjs`). Nothing may start matching on the exe name.
- `config/nsis/orca-installer-hooks.nsh` kills the daemon by image name. That now also matches the
app's own exe, which is correct on a genuine uninstall — the product is being removed — but its
`${isUpdated}` guard must stay: electron-builder runs the uninstaller during every update's
`uninstallOldVersion`, and killing the daemon there defeats the whole feature. The legacy
`orca-terminal-daemon.exe` name stays in the macro to reap hosts left by older builds.
- `LOCAL_HOST_ROOT_NAME` in `daemon-host-relocation.ts` and the path in the uninstall macro are the
same directory. Change both together.
## Verifying a change
Unit coverage lives in `src/main/daemon/daemon-host-relocation.test.ts` (copy plan, verbatim
naming, marker/atomic publish, fail-open, prune veto). Nothing in unit tests can prove survival, so
any change to this file or to the NSIS macro needs the packaged harnesses:
- `.github/workflows/win-update-survival-e2e.yml` — builds an installer from the branch and updates
it over itself with `--expect survival`. The primary proof.
- `.github/workflows/win-crash-survival-e2e.yml` — proves the daemon survives a main-process crash.
- `.github/workflows/windows-terminal-restart-e2e.yml` — terminal restart behaviour.
- `.github/workflows/win-update-e2e.yml` — release-tag-to-release-tag update, both `survival` and
`cold-restore` profiles.
All four are `workflow_dispatch`-only (the two update workflows also carry a push trigger pinned to
one historical feature branch), so they must be dispatched by hand against this branch before
merging a change here — which requires the workflow files to already exist on `main`.
+25 -12
View File
@@ -50,25 +50,37 @@ and `orca-terminal-daemon.exe` report `Valid CN=SignPath Foundation`.
## The behaviours, and why each one exists
### The daemon runs from a renamed copy of our own image
### The daemon runs from a copy of our own image
`src/main/daemon/daemon-host-relocation.ts` copies the Electron runtime into
`%LOCALAPPDATA%\Orca\daemon-host\<version>\` and renames `Orca.exe` to
`orca-terminal-daemon.exe`. The comment on `DAEMON_HOST_EXE_NAME` states the
reason without varnish: _"so the NSIS updater's `taskkill /IM Orca.exe` can't
match it."_
`%LOCALAPPDATA%\Orca\daemon-host\<version>\` and forks the terminal daemon from
there.
It exists because the NSIS installer deletes the old install directory and force-
kills every process imaged under it. Without relocation, an auto-update kills the
terminal daemon and every live terminal with it. The copy is a run-as-node
`Orca.exe` rather than `node.exe` so there is no console flash and asar still
resolves; `config/nsis/daemon-host-uninstall.nsh` reaps it on a real uninstall
resolves; `config/nsis/orca-installer-hooks.nsh` reaps it on a real uninstall
(guarded by `${isUpdated}` so an update's `uninstallOldVersion` never fires it).
**How an EDR reads it: MITRE T1036, masquerading.** A signed executable copied
out of the install directory into `%LOCALAPPDATA%` under a different name, which
then spawns shells, matches the textbook description closely enough that no
behavioural engine can be expected to score it low.
**At the time of these incidents the copy was also renamed** to
`orca-terminal-daemon.exe`, the image name every incident here reports, and
`DAEMON_HOST_EXE_NAME`'s comment stated the reason without varnish: _"so the NSIS
updater's `taskkill /IM Orca.exe` can't match it."_ The rename has since been
removed; the copy now keeps the app exe's own file name, because the updater's
kill sweep is path-scoped on every host that has PowerShell and the rename only
ever bought the no-PowerShell fallback. See
[`windows-daemon-host-relocation.md`](./windows-daemon-host-relocation.md).
**How an EDR reads it: MITRE T1036, masquerading** — and, for what remains,
**T1036.005**. A signed executable copied out of the install directory into
`%LOCALAPPDATA%` under a different name, which then spawns shells, matches the
textbook description closely enough that no behavioural engine can be expected to
score it low. Dropping the rename removes that literal indicator but not the
underlying shape: execution from a non-standard user-writable location is scored
on its own. Note also that the strongest form of the T1036 signal was never
present here — the shipped binary's `OriginalFilename` is empty, so there was no
embedded name for the old disk name to contradict.
### Every process gets a handle, on a timer
@@ -245,7 +257,8 @@ obfuscated-command-line detector is tuned on.
### The spawn tree itself
`Orca.exe` → `orca-terminal-daemon.exe` → a shell → an agent CLI is what a
`Orca.exe` → the relocated daemon host (`orca-terminal-daemon.exe` in the builds
these incidents cover, `Orca.exe` since) → a shell → an agent CLI is what a
terminal multiplexer for coding agents *is*. `reg.exe` appears from
`src/main/win32-utils.ts`,
`src/main/agent-hooks/managed-hook-owner-identity.ts` and
@@ -376,7 +389,7 @@ The checklist. On Windows, do not reach for:
| Forking `powershell.exe` to read system state | The native reader — [`windows-process-enumeration.md`](./windows-process-enumeration.md) is the standing rule for the process table |
| A process per operation in a loop | One long-lived helper with a request channel. A burst of short-lived interpreters under one parent is itself the signal |
| `Add-Type -TypeDefinition` at runtime | A precompiled, signed assembly, or a native helper |
| Copying our own image under a different name | An installer or updater that does not need the rename. Where the rename is load-bearing, document it as such |
| Copying our own image under a different name | Copy it verbatim — [`windows-daemon-host-relocation.md`](./windows-daemon-host-relocation.md) (done for the daemon host) |
| Deriving a script runner from a UI preference | [`windows-setup-shell.md`](./windows-setup-shell.md) — the script declares its own interpreter |
Two framing rules that outlast the table:
+117 -1
View File
@@ -344,7 +344,7 @@ on any other OS keeps using the scan.
## Why the package is patched
`config/patches/@vscode__windows-process-tree@0.8.0.patch` carries three hunks.
`config/patches/@vscode__windows-process-tree@0.8.0.patch` carries four hunks.
1. **Spectre mitigation.** The upstream `binding.gyp` requires Spectre-mitigated
libraries, which Orca's Windows build agents do not install. `node-pty` is
@@ -359,10 +359,122 @@ on any other OS keeps using the scan.
realpath, then loads the relative path from the `node_modules` symlink, so
`node_addon_api.gyp` resolves outside the repo and hourly Windows builds
die at configure. `node-pty` is patched the same way for the same reason.
4. **No PEB reads, no `PROCESS_VM_READ`.** See below.
The typings claim `commandLine` is truncated at 512 characters. Measured, it is
not: the longest observed on a real host was 26,059.
### The command line comes from the kernel, not the target's memory
Upstream, `GetProcessCommandLine` opens every process with
`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and issues three chained
`ReadProcessMemory` calls — PEB, `RTL_USER_PROCESS_PARAMETERS`, then the string
— to recover the command line. Walking another process's address space for
credentials-adjacent data on a repeating timer is what a credential dumper does,
so Defender for Endpoint scores it as such regardless of intent. Nothing about
the flag sets above changes that; only removing the read does.
Windows 8.1 added `NtQueryInformationProcess`'s `ProcessCommandLineInformation`
class (60), which returns the same string as a `UNICODE_STRING` the kernel
builds, needing only `PROCESS_QUERY_LIMITED_INFORMATION`. Electron's floor is
Windows 10, so every OS Orca supports has it. The entry point is resolved with
`GetProcAddress` on `ntdll.dll` — it has no import library — and the size is
probed with a null-buffer call that answers `STATUS_INFO_LENGTH_MISMATCH`.
The same hunk drops `PROCESS_VM_READ` from `GetProcessMemoryUsage` and
`GetCpuUsage`, which acquired it and never read an address space:
`GetProcessMemoryInfo` and `GetProcessTimes` are satisfied by
`PROCESS_QUERY_LIMITED_INFORMATION`. Measured, both return identical values
under the weaker right on every process that opens at all.
Measured on Windows 11, ~540 processes, counted in-process by replacing the
addon's import table entries with counting stubs:
| per `CommandLine` scan | before | after |
| ---------------------- | ----------------------------------------- | -------------------------------------- |
| `OpenProcess` calls | 543 | 543 |
| desired access | `0x0410` (`VM_READ \| QUERY_INFORMATION`) | `0x1000` (`QUERY_LIMITED_INFORMATION`) |
| `ReadProcessMemory` | 1128 | **0** |
| p50 / p95 | 13.5 / 14.5 ms | 12.3 / 13.5 ms |
Command lines were byte-identical on every process both readers recovered
(405/405, and 399/399 and 376/376 on other runs), including a 24,087-character
argv with embedded quotes, non-ASCII characters and trailing whitespace, and a
WOW64 target. The weaker right is also a strict superset in reach: three
processes that refused `PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` granted
`PROCESS_QUERY_LIMITED_INFORMATION`, and none went the other way.
### There is no PEB fallback, deliberately
An earlier revision kept the PEB reader for a kernel without class 60, behind a
latch. That was wrong, and the reason is worth recording: `ClassifyQueryFailure`
mapped `STATUS_INVALID_INFO_CLASS` / `NOT_SUPPORTED` / `NOT_IMPLEMENTED` from
**any single target** onto a process-wide, one-way switch back to
`PROCESS_VM_READ` plus three `ReadProcessMemory` per pid per scan, for the life
of the process, with nothing observable from JS.
The environment this reader exists for is one where an EDR hooks `ntdll`. A hook
that returns `STATUS_INVALID_INFO_CLASS` for a class it does not recognise would
have silently reinstated the exact primitive the patch removes, on precisely the
machines it was written for — and one stray status from one process was enough.
The same applies under Wine or any instrumented `ntdll`.
So the fallback is gone rather than guarded. `GetProcessCommandLine` returns
false and leaves the command line empty, which is already a normal outcome
(`WindowsProcessRow.command` is documented as empty when a process denies a
query handle, and callers fall back to the image name). Degrading to no command
line is recoverable; silently resuming address-space reads is not.
This also makes the property checkable on the artifact rather than the source:
the patched reader never calls `ReadProcessMemory`, so the symbol is absent from
the compiled addon's import table. `inspectWindowsProcessTreeAddon()` in
`config/scripts/windows-process-tree-gyp-rebuild.mjs` is that check, and it is
the only way to tell the two binaries apart — see below. It answers
`clean` / `unpatched` / `missing` rather than a boolean, because a binary that is
not there has not been cleared, and a caller reading `false` as “verified” would
pass exactly the thing the check exists to catch.
Because the returned `UNICODE_STRING` comes from that same hookable boundary,
its `Buffer` and `Length` are bounds-checked against the allocation before the
characters are encoded, and the probed size is capped at the header plus 64 KiB
(`Length` is a `USHORT`) so a bogus size cannot turn into a `bad_alloc` that
fails an entire scan instead of one process.
### The published tarball ships a loadable unpatched prebuilt
`@vscode/windows-process-tree@0.8.0` publishes
`build/Release/windows_process_tree.node` in the tarball. It is node-addon-api,
so it is ABI-stable and loads cleanly under both Node and Electron — and it was
built from unpatched source, so it performs 1179 `ReadProcessMemory` calls and
opens every process at `0x0410` per scan.
That matters because `allowBuilds` is `false` for this package and CI installs
with `--ignore-scripts`, so nothing compiles it at install time. A `require()`
health check cannot tell the two binaries apart, and a rebuild that is skipped —
`rebuild-native-deps.mjs` soft-exits 0 on a Windows file lock during postinstall
— leaves the upstream prebuilt in place and cached.
Four checks close that, all keyed on the absent `ReadProcessMemory` import:
- `ensureWindowsProcessTreeCommandLinePatch()` deletes a binary that still has
it, so a skipped rebuild fails loudly instead of using the prebuilt;
- `ensure-native-runtime.mjs` treats such a binary as a load failure, which is
what triggers the rebuild;
- the relay build asserts it on the artifact it just produced;
- `loadWindowsProcessTree()` asserts it again on the addon staged beside a relay
bundle and refuses to bind one that still imports the symbol, falling back to
the CIM scan. The build-time assertion is not enough on its own: a bundle and
the addon beside it redeploy independently, so a host that has not taken a new
bundle keeps whatever `.node` is already there.
What none of this does is narrow _which_ processes are asked. A detailed scan
still queries every pid, including `lsass.exe`; it now asks with the same right
Task Manager uses instead of `PROCESS_VM_READ`. Restricting the command-line
pass to Orca's own subtree is the complementary change, and it belongs with the
identity/detailed reader split rather than here — a ppid-derived allowlist would
miss exactly the detached, reparented descendants the trackers exist to find
(#9045, #10475), so it needs the job-object membership as its source of truth.
## Packaging
The addon is Windows-only, so it follows the same contract as
@@ -376,6 +488,10 @@ The addon is Windows-only, so it follows the same contract as
`ensure-native-runtime.mjs`;
- copied into the packaged `node_modules` for win32 only.
The relay's copy is a separate artifact staged beside the bundle, so a relay host
only picks up a rebuilt addon on redeploy. Until then it keeps whatever binary it
already has, which is why the addon is checked again at load.
## What the snapshot does not provide
`CreationDate` (process start time) has no equivalent. Anything using a start
@@ -204,11 +204,18 @@ function escapeRegExp(value: string): string {
}
function codePlaceholderPrefix(content: string): string {
let prefix = CODE_PLACEHOLDER_PREFIX_BASE
while (content.includes(prefix)) {
prefix = `${prefix}_`
let suffixLength = 0
let cursor = 0
while ((cursor = content.indexOf(CODE_PLACEHOLDER_PREFIX_BASE, cursor)) !== -1) {
cursor += CODE_PLACEHOLDER_PREFIX_BASE.length
const suffixStart = cursor
while (content[cursor] === '_') {
cursor += 1
}
// One extra underscore keeps the prefix longer than every authored run.
suffixLength = Math.max(suffixLength, cursor - suffixStart + 1)
}
return prefix
return CODE_PLACEHOLDER_PREFIX_BASE + '_'.repeat(suffixLength)
}
function protectMarkdownCode(content: string): {
@@ -0,0 +1,33 @@
import { describe, expect, it } from 'vitest'
import { normalizeMobileMarkdownPreviewHtml } from './mobile-markdown-preview-html'
const marker = '\uE000ORCA_MD_CODE_'
const suffix = '\uE000'
describe('mobile Markdown code placeholder collisions', () => {
it.each([0, 1, 2, 15, 128, 16384])('preserves a literal marker with %i underscores', (length) => {
const literal = `${marker}${'_'.repeat(length)}0${suffix}`
const input = `${literal} and \`Array<string>\`\n\n\`\`\`html\n<p>literal</p>\n\`\`\``
expect(normalizeMobileMarkdownPreviewHtml(input)).toBe(input)
})
it('handles adjacent markers and repeated maximum suffixes', () => {
const literal = `${marker}${marker}__0${suffix}${marker}__1${suffix}${marker}_2${suffix}`
expect(normalizeMobileMarkdownPreviewHtml(`<p>${literal} and \`<div>\`</p>`)).toBe(
`${literal} and \`<div>\``
)
})
it('preserves authored markers across generated suffix orders and HTML islands', () => {
let seed = 173
for (let sample = 0; sample < 500; sample++) {
const literals: string[] = []
for (let index = 0; index < 8; index++) {
seed = (Math.imul(seed, 1664525) + 1013904223) >>> 0
literals.push(`${marker}${'_'.repeat(seed % 32)}${index}${suffix}`)
}
const text = literals.join(' ') + ' and `Array<string>`'
expect(normalizeMobileMarkdownPreviewHtml(`<p>${text}</p>`)).toBe(text)
}
})
})
@@ -81,7 +81,7 @@ export function rankSuggestions(candidates: readonly string[], query: string, li
const substring: string[] = []
for (const candidate of candidates) {
const lower = candidate.toLowerCase()
const base = lower.split('/').pop() ?? lower
const base = lower.slice(lower.lastIndexOf('/') + 1)
if (lower.startsWith(q) || base.startsWith(q)) {
prefix.push(candidate)
} else if (lower.includes(q)) {
@@ -31,7 +31,14 @@ export function directPathForEndpoint(
// instead of holding the supervisor's operation mutex for the full outer bound.
const RECONNECT_GRACE_MS = 2_000
function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Promise<void> {
function waitForAuthenticatedSession(
session: RpcClient,
timeoutMs: number,
signal?: AbortSignal
): Promise<void> {
if (signal?.aborted) {
return Promise.reject(new Error('probe cancelled'))
}
if (session.getState() === 'connected') {
return Promise.resolve()
}
@@ -72,7 +79,13 @@ function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Pro
finish()
reject(new Error('probe session authentication timed out'))
}, timeoutMs)
const onAbort = (): void => {
finish()
reject(new Error('probe cancelled'))
}
signal?.addEventListener('abort', onAbort, { once: true })
function finish(): void {
signal?.removeEventListener('abort', onAbort)
if (timer) {
clearTimeout(timer)
}
@@ -87,8 +100,12 @@ function waitForAuthenticatedSession(session: RpcClient, timeoutMs: number): Pro
export async function openAuthenticatedDirectEndpoint(
host: HostProfile,
openDirect: (endpoint: string) => RpcClient,
timeoutMs: number
timeoutMs: number,
signal?: AbortSignal
): Promise<{ client: RpcClient; path: Exclude<MobileConnectionPath, 'relay'> } | null> {
if (signal?.aborted) {
return null
}
const endpoints = directEndpointUrls(host)
return await new Promise((resolve) => {
const clients = new Set<RpcClient>()
@@ -110,8 +127,13 @@ export async function openAuthenticatedDirectEndpoint(
continue
}
clients.add(client)
void waitForAuthenticatedSession(client, timeoutMs).then(
void waitForAuthenticatedSession(client, timeoutMs, signal).then(
() => {
if (signal?.aborted) {
client.close()
rejectCandidate()
return
}
if (settled) {
client.close()
return
@@ -0,0 +1,175 @@
import { expect, it, vi } from 'vitest'
import {
dependencies,
FakeLogicalClient,
FakeSession,
host
} from './mobile-endpoint-supervisor-test-fakes'
import { MobileEndpointHysteresis } from './mobile-endpoint-hysteresis'
import { createStableLogicalRpcClient } from './stable-logical-rpc-client'
import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor'
vi.mock('react-native', () => ({ Platform: { OS: 'ios' } }))
vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' }))
vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) }))
it('closes in-flight candidates and clears their timeout when the owner stops', async () => {
vi.useFakeTimers()
try {
const candidate = new FakeSession('connecting')
const logical = new FakeLogicalClient('connected', 'relay')
const deps = dependencies({ openDirect: vi.fn(() => candidate) })
const supervisor = new MobileEndpointSupervisor(logical, host, deps)
await supervisor.start()
await vi.advanceTimersByTimeAsync(15_000)
expect(deps.openDirect).toHaveBeenCalledOnce()
supervisor.stop()
await vi.advanceTimersByTimeAsync(0)
expect(candidate.close).toHaveBeenCalledOnce()
expect(vi.getTimerCount()).toBe(0)
await vi.advanceTimersByTimeAsync(12_000)
expect(vi.getTimerCount()).toBe(0)
expect(candidate.close).toHaveBeenCalledOnce()
expect(logical.migrateTo).not.toHaveBeenCalled()
expect(deps.openDirect).toHaveBeenCalledOnce()
} finally {
vi.restoreAllMocks()
vi.useRealTimers()
}
})
it('closes an authenticated candidate when stop races its completion', async () => {
vi.useFakeTimers()
try {
const candidate = new FakeSession('connecting')
const logical = new FakeLogicalClient('connected', 'relay')
const deps = dependencies({ openDirect: vi.fn(() => candidate) })
const supervisor = new MobileEndpointSupervisor(logical, host, deps)
await supervisor.start()
await vi.advanceTimersByTimeAsync(15_000)
candidate.publishState('connected')
supervisor.stop()
await vi.advanceTimersByTimeAsync(0)
expect(candidate.close).toHaveBeenCalledOnce()
expect(logical.migrateTo).not.toHaveBeenCalled()
expect(vi.getTimerCount()).toBe(0)
} finally {
vi.restoreAllMocks()
vi.useRealTimers()
}
})
it('preserves an in-flight probe across a transient background pause', async () => {
vi.useFakeTimers()
try {
const candidate = new FakeSession('connecting')
const logical = new FakeLogicalClient('connected', 'relay')
const deps = dependencies({ openDirect: vi.fn(() => candidate) })
const supervisor = new MobileEndpointSupervisor(logical, host, deps)
await supervisor.start()
await vi.advanceTimersByTimeAsync(15_000)
supervisor.setForeground(false)
await vi.advanceTimersByTimeAsync(0)
expect(candidate.close).not.toHaveBeenCalled()
supervisor.stop()
await vi.advanceTimersByTimeAsync(0)
expect(candidate.close).toHaveBeenCalledOnce()
expect(vi.getTimerCount()).toBe(0)
} finally {
vi.restoreAllMocks()
vi.useRealTimers()
}
})
it('releases every candidate when multiple endpoint probes are pending', async () => {
vi.useFakeTimers()
try {
const candidates: FakeSession[] = []
const logical = new FakeLogicalClient('connected', 'relay')
const deps = dependencies({
openDirect: vi.fn(() => {
const candidate = new FakeSession('connecting')
candidates.push(candidate)
return candidate
})
})
const supervisor = new MobileEndpointSupervisor(
logical,
{
...host,
endpoints: [{ id: 'alternate', kind: 'tailscale', url: 'ws://100.64.0.2:6768' }]
},
deps
)
await supervisor.start()
await vi.advanceTimersByTimeAsync(15_000)
expect(candidates).toHaveLength(2)
supervisor.stop()
await vi.advanceTimersByTimeAsync(0)
expect(vi.getTimerCount()).toBe(0)
for (const candidate of candidates) {
expect(candidate.close).toHaveBeenCalledOnce()
candidate.publishState('connected')
}
await vi.advanceTimersByTimeAsync(60_000)
expect(logical.migrateTo).not.toHaveBeenCalled()
expect(deps.openDirect).toHaveBeenCalledTimes(2)
expect(vi.getTimerCount()).toBe(0)
} finally {
vi.restoreAllMocks()
vi.useRealTimers()
}
})
it.each([false, true])(
'fences migration finishing after stop (already swapped: %s)',
async (alreadySwapped) => {
vi.useFakeTimers()
try {
const recordedMigration = vi.spyOn(MobileEndpointHysteresis.prototype, 'recordMigration')
const relay = new FakeSession('connected')
const logical = createStableLogicalRpcClient(relay, 'relay')
const candidates: FakeSession[] = []
const deps = dependencies({
openDirect: vi.fn(() => {
const candidate = new FakeSession('connected')
candidates.push(candidate)
return candidate
})
})
const supervisor = new MobileEndpointSupervisor(logical, host, deps)
const migrate = logical.migrateTo.bind(logical)
let release!: () => void
const pending = new Promise<void>((resolve) => {
release = resolve
})
const migration = vi.spyOn(logical, 'migrateTo').mockImplementation(async (...args) => {
if (alreadySwapped) {
await migrate(...args)
}
await pending
if (!alreadySwapped) {
await migrate(...args)
}
})
await supervisor.start()
await vi.advanceTimersByTimeAsync(60_000)
expect(migration).toHaveBeenCalledOnce()
const requestsBeforeStop = relay.sendRequest.mock.calls.length
const candidateRequestsBeforeStop = candidates[3].sendRequest.mock.calls.length
const migrationsBeforeStop = recordedMigration.mock.calls.length
supervisor.stop()
release()
await vi.advanceTimersByTimeAsync(0)
expect(logical.getActivePath()).toBe(alreadySwapped ? 'lan' : 'relay')
expect(logical.getGeneration()).toBe(alreadySwapped ? 2 : 1)
expect(relay.sendRequest).toHaveBeenCalledTimes(requestsBeforeStop)
expect(candidates[3].sendRequest).toHaveBeenCalledTimes(candidateRequestsBeforeStop)
expect(recordedMigration).toHaveBeenCalledTimes(migrationsBeforeStop)
expect(candidates[3].close).toHaveBeenCalledTimes(alreadySwapped ? 0 : 1)
expect(vi.getTimerCount()).toBe(0)
logical.close()
} finally {
vi.restoreAllMocks()
vi.useRealTimers()
}
}
)
@@ -11,6 +11,9 @@ const DIRECT_PROBE_INTERVAL_MS = 15_000
export class DirectReturnProbe {
private timer: ReturnType<typeof setTimeout> | null = null
private stopped = false
private activeProbe: AbortController | null = null
constructor(
private readonly deps: {
now: () => number
@@ -24,14 +27,18 @@ export class DirectReturnProbe {
canSchedule: () => boolean
canAttempt: () => boolean
beginOperation: () => void
migrate: (client: RpcClient, path: MobileConnectionPath) => Promise<void>
migrate: (
client: RpcClient,
path: MobileConnectionPath,
shouldAbort: () => boolean
) => Promise<void>
onDirectMigrated: () => Promise<void>
afterProbe: () => void
}
) {}
schedule(delayMs = DIRECT_PROBE_INTERVAL_MS): void {
if (!this.hooks.canSchedule() || this.timer) {
if (this.stopped || !this.hooks.canSchedule() || this.timer) {
return
}
this.timer = this.deps.setTimer(() => {
@@ -47,19 +54,34 @@ export class DirectReturnProbe {
}
}
stop(): void {
this.stopped = true
this.clear()
this.activeProbe?.abort()
}
private async probe(): Promise<void> {
if (this.stopped) {
return
}
if (!this.hooks.canAttempt() || !this.hooks.hysteresis.canProbe(this.deps.now())) {
this.schedule()
return
}
const controller = new AbortController()
this.activeProbe = controller
this.hooks.beginOperation()
let successful: Awaited<ReturnType<typeof openAuthenticatedDirectEndpoint>> = null
try {
successful = await openAuthenticatedDirectEndpoint(
this.hooks.host(),
this.deps.openDirect,
12_000
12_000,
controller.signal
)
if (this.stopped) {
return
}
if (!successful) {
this.hooks.hysteresis.recordDirectFailure(this.deps.now())
return
@@ -68,11 +90,24 @@ export class DirectReturnProbe {
successful.client.close()
return
}
await this.hooks.migrate(successful.client, successful.path)
const candidate = successful
// Migration owns the candidate, including closing it if cutover is canceled.
successful = null
try {
await this.hooks.migrate(candidate.client, candidate.path, () => this.stopped)
} catch (error) {
if (this.stopped) {
return
}
throw error
}
if (this.stopped) {
return
}
this.hooks.hysteresis.recordMigration(this.deps.now())
await this.hooks.onDirectMigrated()
} finally {
this.activeProbe = null
successful?.client.close()
// Why: a relay drop or backoff timer can arrive while the probe owns the
// operation mutex; afterProbe releases it and replays deferred recovery.
@@ -120,7 +120,7 @@ export class MobileEndpointSupervisor {
canSchedule: () => this.isActive() && this.logical.getActivePath() === 'relay',
canAttempt: () => this.isActive() && !this.operationInFlight,
beginOperation: () => (this.operationInFlight = true),
migrate: (client, path) => this.logical.migrateTo(client, path),
migrate: (client, path, abort) => this.logical.migrateTo(client, path, undefined, abort),
onDirectMigrated: async () => {
this.leaseRotation.clear()
this.relayRotationPending = false
@@ -195,6 +195,7 @@ export class MobileEndpointSupervisor {
stop(): void {
this.stopped = true
this.directProbe.stop()
this.unsubscribeState?.()
this.unsubscribeState = null
this.backgroundGrace.stop()
@@ -0,0 +1,109 @@
import { describe, expect, it } from 'vitest'
import { RuntimeBrowserScreencastController } from '../../../src/main/runtime/runtime-browser-screencast-controller'
import type { RuntimeBrowserCommands } from '../../../src/main/runtime/orca-runtime-browser'
import type { BrowserScreencastResult } from '../../../src/shared/runtime-types'
import { MobileRelayRpcStreams } from './mobile-relay-rpc-streams'
import type { RpcResponse } from './types'
describe('relay browser cancellation resource budget', () => {
it.each([false, true])('stops host frames when cancellation precedes ready=%s', async (early) => {
const subscriptions = new Map<string, () => void | Promise<void>>()
const done = Promise.withResolvers<void>()
const ready = Promise.withResolvers<RpcResponse>()
let sequence = 0
let stopped = false
let frameSends = 0
let frameBytes = 0
let sendBinary: (bytes: Uint8Array) => boolean | void = () => false
let hostRun: Promise<void> | undefined
const methods: string[] = []
const cleanup = (id: string): void => {
const release = subscriptions.get(id)
subscriptions.delete(id)
void release?.()
}
const host = new RuntimeBrowserScreencastController({
getCommands: () =>
({
browserScreencast: async (_params, stream) => {
sendBinary = stream.sendBinary
return {
subscriptionId: 'server-stream',
ready: { type: 'ready', subscriptionId: 'server-stream', browserPageId: 'page' },
session: {
done: done.promise,
stop: () => {
stopped = true
done.resolve()
}
},
flushPendingFrame: () => {}
}
}
}) as RuntimeBrowserCommands,
registerSubscriptionCleanup: (id, release) => subscriptions.set(id, release),
cleanupSubscription: cleanup,
getDriver: () => ({ kind: 'idle' }),
setDriver: () => {},
notifyRemoteViewersChanged: () => {}
})
const streams = new MobileRelayRpcStreams({
nextId: () => `request-${++sequence}`,
waitForConnected: async () => {},
sendFrame: (request) => {
methods.push(request.method)
if (request.method === 'browser.screencast' && (request.params as { page?: string }).page) {
hostRun = host.start(request.params as Parameters<typeof host.start>[0], {
connectionId: 'relay-connection',
sendBinary: (bytes) => {
frameSends++
frameBytes += bytes.byteLength
return true
},
emit: (result: BrowserScreencastResult) => {
if (result.type === 'ready') {
ready.resolve({
id: request.id,
ok: true,
streaming: true,
result,
_meta: { runtimeId: 'host' }
})
}
}
})
} else if (request.method === 'browser.screencast.unsubscribe') {
cleanup((request.params as { subscriptionId: string }).subscriptionId)
}
return true
}
})
const cancel = streams.subscribe('browser.screencast', { page: 'page' }, () => {})
try {
const response = await ready.promise
if (early) {
cancel()
}
streams.handleResponse(response)
if (!early) {
cancel()
}
for (let frame = 0; frame < 100; frame++) {
if (!stopped) {
sendBinary(new Uint8Array(65_536))
}
}
expect({ stopped, subscriptions: subscriptions.size, frameSends, frameBytes }).toEqual({
stopped: true,
subscriptions: 0,
frameSends: 0,
frameBytes: 0
})
expect(methods).toEqual(['browser.screencast', 'browser.screencast.unsubscribe'])
} finally {
cleanup('server-stream')
await hostRun
streams.clear()
}
})
})
@@ -136,6 +136,36 @@ describe('mobile relay RPC session', () => {
})
afterEach(() => vi.useRealTimers())
it('releases stream listeners on failure even when close follows it', async () => {
const { session } = await authenticateSession()
const listener = vi.fn()
session.subscribe('runtime.clientEvents.subscribe', {}, listener)
await Promise.resolve()
const request = JSON.parse(fakes.sendText.mock.calls[0]![0] as string) as { id: string }
fakes.linkOptions!.onText(
JSON.stringify({
id: request.id,
ok: true,
streaming: true,
result: { type: 'ready', subscriptionId: 'server-events' },
_meta: { runtimeId: 'runtime-1' }
})
)
expect(listener).toHaveBeenCalledTimes(1)
fakes.linkOptions!.onError(new Error('relay lost'))
session.close()
fakes.linkOptions!.onText(
JSON.stringify({
id: request.id,
ok: true,
streaming: true,
result: { type: 'event' },
_meta: { runtimeId: 'runtime-1' }
})
)
expect(listener).toHaveBeenCalledTimes(1)
})
it('requires exact resume observations and confirms by request ID before becoming connected', async () => {
const { session, confirmationRequest, capabilityRequest } = await authenticateSession()
@@ -294,6 +294,7 @@ export function connectMobileRelayRpcSession(args: {
closed = true
failure = error
livenessWatchdog.stop(livenessIdentity)
streams.clear()
link.close()
pending.rejectAll(error)
publishState(error instanceof MobileE2EEAuthenticationError ? 'auth-failed' : 'disconnected')
@@ -0,0 +1,259 @@
import { describe, expect, it, vi } from 'vitest'
import { MobileRelayRpcStreams } from './mobile-relay-rpc-streams'
import type { RpcResponse } from './types'
function createStreams(waitForConnected = async () => {}) {
let sequence = 0
const sendFrame = vi.fn((_request: { id: string; method: string; params?: unknown }) => true)
const streams = new MobileRelayRpcStreams({
nextId: () => `request-${++sequence}`,
sendFrame,
waitForConnected
})
return { streams, sendFrame }
}
function response(id: string, result: unknown): RpcResponse {
return { id, ok: true, streaming: true, result, _meta: { runtimeId: 'test' } }
}
const serverSubscriptions = [
['browser.screencast', 'browser.screencast.unsubscribe'],
['runtime.clientEvents.subscribe', 'runtime.clientEvents.unsubscribe']
] as const
describe('mobile relay subscription cancellation', () => {
it.each(serverSubscriptions)('cleans up ready %s exactly once', async (method, unsubscribe) => {
const { streams, sendFrame } = createStreams()
const listener = vi.fn()
const cancel = streams.subscribe(method, {}, listener)
await Promise.resolve()
streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' }))
cancel()
cancel()
expect(sendFrame.mock.calls).toEqual([
[{ id: 'request-1', method, params: {} }],
[{ id: 'request-2', method: unsubscribe, params: { subscriptionId: 'server-1' } }]
])
expect(streams.handleResponse(response('request-1', { type: 'end' }))).toBe(false)
expect(listener).toHaveBeenCalledTimes(1)
})
it.each(serverSubscriptions)(
'cleans up late-ready %s without calling disposed listeners',
async (method, unsubscribe) => {
const { streams, sendFrame } = createStreams()
const listener = vi.fn()
const cancel = streams.subscribe(method, {}, listener)
await Promise.resolve()
cancel()
cancel()
expect(sendFrame).toHaveBeenCalledTimes(1)
expect(streams.handleResponse(response('request-1', { type: 'starting' }))).toBe(true)
streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' }))
expect(sendFrame).toHaveBeenLastCalledWith({
id: 'request-2',
method: unsubscribe,
params: { subscriptionId: 'server-1' }
})
expect(
streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' }))
).toBe(false)
expect(listener).not.toHaveBeenCalled()
}
)
it.each(['error', 'end', 'disconnect', 'completed'])(
'forgets cancelled cleanup routes on %s',
async (ending) => {
const { streams, sendFrame } = createStreams()
const cancel = streams.subscribe('browser.screencast', {}, vi.fn())
await Promise.resolve()
cancel()
if (ending === 'disconnect') {
streams.clear()
} else if (ending === 'completed') {
streams.handleResponse({
id: 'request-1',
ok: true,
result: null,
_meta: { runtimeId: 'test' }
})
} else if (ending === 'error') {
streams.handleResponse({
id: 'request-1',
ok: false,
error: { code: 'unsupported', message: 'failed' },
_meta: { runtimeId: 'test' }
})
} else {
streams.handleResponse(response('request-1', { type: 'end', subscriptionId: 'server-1' }))
}
expect(
streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' }))
).toBe(false)
expect(sendFrame).toHaveBeenCalledTimes(1)
}
)
it.each([
[
'terminal.subscribe',
{ terminal: 'term', client: { id: 'phone' } },
'terminal.unsubscribe',
{ subscriptionId: 'term:phone', client: { id: 'phone' } }
],
[
'session.tabs.subscribe',
{ worktree: 'id:workspace' },
'session.tabs.unsubscribe',
{ worktree: 'id:workspace', subscriptionId: 'request-1' }
],
[
'nativeChat.subscribe',
{ subscriptionId: 'chat' },
'nativeChat.unsubscribe',
{ subscriptionId: 'chat' }
]
])(
'cancels %s using its request cleanup identity',
async (method, params, unsubscribe, unsubscribeParams) => {
const { streams, sendFrame } = createStreams()
const cancel = streams.subscribe(method as string, params, vi.fn())
await Promise.resolve()
if (method === 'session.tabs.subscribe') {
streams.handleResponse(response('request-1', { type: 'snapshot' }))
}
cancel()
expect(sendFrame).toHaveBeenLastCalledWith({
id: 'request-2',
method: unsubscribe,
params: unsubscribeParams
})
}
)
it.each([
'terminal.subscribe',
'browser.screencast',
'runtime.clientEvents.subscribe',
'session.tabs.subscribe',
'nativeChat.subscribe'
])('does not unsubscribe an unsent %s', async (method) => {
const wait = Promise.withResolvers<void>()
const { streams, sendFrame } = createStreams(() => wait.promise)
const cancel = streams.subscribe(
method,
{ terminal: 'term', worktree: 'id:workspace', subscriptionId: 'chat' },
vi.fn()
)
cancel()
wait.resolve()
await Promise.resolve()
expect(sendFrame).not.toHaveBeenCalled()
expect(
streams.handleResponse(response('request-1', { type: 'ready', subscriptionId: 'server-1' }))
).toBe(false)
})
it.each([false, true])(
'preserves a same-worktree sibling when cancellation precedes snapshot=%s',
async (early) => {
const { streams, sendFrame } = createStreams()
const first = vi.fn()
const second = vi.fn()
const cancel = streams.subscribe(
'session.tabs.subscribe',
{ worktree: 'id:workspace' },
first
)
streams.subscribe('session.tabs.subscribe', { worktree: 'id:workspace' }, second)
await Promise.resolve()
if (early) {
cancel()
}
expect(sendFrame).toHaveBeenCalledTimes(2)
streams.handleResponse(response('request-1', { type: 'snapshot' }))
if (!early) {
cancel()
}
expect(sendFrame).toHaveBeenLastCalledWith({
id: 'request-3',
method: 'session.tabs.unsubscribe',
params: { worktree: 'id:workspace', subscriptionId: 'request-1' }
})
streams.handleResponse(response('request-2', { type: 'snapshot' }))
streams.handleResponse(response('request-2', { type: 'updated' }))
expect(second).toHaveBeenCalledTimes(2)
expect(first).toHaveBeenCalledTimes(early ? 0 : 1)
expect(streams.handleResponse(response('request-1', { type: 'updated' }))).toBe(false)
}
)
it.each([
['nativeChat.subscribe', { agent: 'claude', sessionId: 's1', subscriptionId: 'claude:s1' }],
['terminal.subscribe', { terminal: 'term', client: { id: 'phone' } }]
])(
'keeps the newer %s live when an older same-token subscription unmounts',
async (method, params) => {
const { streams, sendFrame } = createStreams()
const older = vi.fn()
const newer = vi.fn()
const cancelOlder = streams.subscribe(method, params, older)
const cancelNewer = streams.subscribe(method, { ...params }, newer)
await Promise.resolve()
expect(sendFrame).toHaveBeenCalledTimes(2)
cancelOlder()
// The host keys cleanup by the deterministic token, so unsubscribing would evict the newer.
expect(sendFrame).toHaveBeenCalledTimes(2)
streams.handleResponse(response('request-2', { type: 'snapshot' }))
expect(newer).toHaveBeenCalledTimes(1)
expect(streams.handleResponse(response('request-1', { type: 'snapshot' }))).toBe(false)
expect(older).not.toHaveBeenCalled()
cancelNewer()
expect(sendFrame).toHaveBeenCalledTimes(3)
expect(sendFrame).toHaveBeenLastCalledWith(
expect.objectContaining({ method: method.replace(/\.subscribe$/, '.unsubscribe') })
)
}
)
it('still unsubscribes a shared-token nativeChat stream when the sibling is unsent', async () => {
const wait = Promise.withResolvers<void>()
let connected = false
const { streams, sendFrame } = createStreams(() =>
connected ? Promise.resolve() : wait.promise
)
const params = { agent: 'claude', sessionId: 's1', subscriptionId: 'claude:s1' }
connected = true
const cancelOlder = streams.subscribe('nativeChat.subscribe', params, vi.fn())
await Promise.resolve()
connected = false
streams.subscribe('nativeChat.subscribe', params, vi.fn())
cancelOlder()
expect(sendFrame).toHaveBeenCalledTimes(2)
expect(sendFrame).toHaveBeenLastCalledWith({
id: 'request-3',
method: 'nativeChat.unsubscribe',
params: { subscriptionId: 'claude:s1' }
})
})
it('cleans up every cancelled server subscription across repeated late-ready cycles', async () => {
const { streams, sendFrame } = createStreams()
const listener = vi.fn()
for (let i = 0; i < 100; i++) {
const cancel = streams.subscribe('runtime.clientEvents.subscribe', {}, listener)
await Promise.resolve()
const requestId = `request-${2 * i + 1}`
cancel()
streams.handleResponse(response(requestId, { type: 'ready', subscriptionId: `server-${i}` }))
}
expect(
sendFrame.mock.calls.filter(
([request]) => (request as { method: string }).method === 'runtime.clientEvents.unsubscribe'
)
).toHaveLength(100)
expect(listener).not.toHaveBeenCalled()
})
})
+104 -15
View File
@@ -4,9 +4,11 @@ import {
type TerminalSnapshotState
} from './rpc-client-terminal-binary-frame'
import {
buildStreamUnsubscribe,
buildTerminalUnsubscribeParams,
updateTerminalSubscriptionViewport
} from './rpc-client-terminal-subscription'
import { buildReadyStreamUnsubscribe } from './rpc-client-server-subscription'
import type { RpcClient } from './rpc-client'
import type { RpcResponse, RpcSuccess } from './types'
@@ -22,6 +24,23 @@ type StreamRecord = {
streamIds: Set<number>
subscriptionId?: string
cancelled: boolean
sent: boolean
receivedSnapshot?: boolean
}
type StreamUnsubscribe = { method: string; params: unknown }
/** Unsubscribe derived from the subscribe params alone (no server-assigned id). */
function buildParamsUnsubscribe(
method: string,
params: unknown,
requestId: string
): StreamUnsubscribe | null {
if (method === 'terminal.subscribe') {
const unsubscribeParams = buildTerminalUnsubscribeParams(params)
return unsubscribeParams ? { method: 'terminal.unsubscribe', params: unsubscribeParams } : null
}
return buildStreamUnsubscribe(method, params, requestId)
}
type StreamManagerOptions = {
@@ -32,6 +51,10 @@ type StreamManagerOptions = {
export class MobileRelayRpcStreams {
private readonly streams = new Map<string, StreamRecord>()
private readonly cancelledSubscriptions = new Map<
string,
{ method: string; unsubscribe?: StreamUnsubscribe }
>()
private readonly terminalListeners = new Map<number, (result: unknown) => void>()
private readonly terminalSnapshots = new Map<number, TerminalSnapshotState>()
private activeBrowserStream: StreamRecord | null = null
@@ -51,13 +74,15 @@ export class MobileRelayRpcStreams {
listener,
onBinaryFrame: subscribeOptions?.onBinaryFrame,
streamIds: new Set(),
cancelled: false
cancelled: false,
sent: false
}
this.streams.set(id, stream)
void this.options
.waitForConnected()
.then(() => {
if (!stream.cancelled) {
stream.sent = true
if (!this.options.sendFrame({ id, method, params: stream.params })) {
this.fail(id, stream, 'Connection interrupted')
}
@@ -75,6 +100,30 @@ export class MobileRelayRpcStreams {
}
handleResponse(response: RpcResponse): boolean {
const cancelled = this.cancelledSubscriptions.get(response.id)
if (cancelled) {
if (!response.ok) {
this.cancelledSubscriptions.delete(response.id)
} else if (response.result && typeof response.result === 'object') {
const result = response.result as { subscriptionId?: unknown; type?: unknown }
if (result.type === 'end') {
this.cancelledSubscriptions.delete(response.id)
} else if (result.type === 'snapshot' && cancelled.unsubscribe) {
this.cancelledSubscriptions.delete(response.id)
this.options.sendFrame({ id: this.options.nextId(), ...cancelled.unsubscribe })
} else if (typeof result.subscriptionId === 'string') {
this.cancelledSubscriptions.delete(response.id)
const unsubscribe = buildReadyStreamUnsubscribe(cancelled.method, result.subscriptionId)
if (unsubscribe) {
this.options.sendFrame({ id: this.options.nextId(), ...unsubscribe })
}
}
}
if (response.ok && response.streaming !== true) {
this.cancelledSubscriptions.delete(response.id)
}
return true
}
const stream = this.streams.get(response.id)
if (!stream) {
return false
@@ -86,6 +135,9 @@ export class MobileRelayRpcStreams {
const result = (response as RpcSuccess).result
if (result && typeof result === 'object') {
const metadata = result as { subscriptionId?: unknown; streamId?: unknown; type?: unknown }
if (stream.method === 'session.tabs.subscribe' && metadata.type === 'snapshot') {
stream.receivedSnapshot = true
}
if (typeof metadata.subscriptionId === 'string') {
stream.subscriptionId = metadata.subscriptionId
}
@@ -125,6 +177,7 @@ export class MobileRelayRpcStreams {
stream.cancelled = true
}
this.streams.clear()
this.cancelledSubscriptions.clear()
this.terminalListeners.clear()
this.terminalSnapshots.clear()
this.activeBrowserStream = null
@@ -136,25 +189,61 @@ export class MobileRelayRpcStreams {
return
}
stream.cancelled = true
if (stream.method === 'terminal.subscribe') {
const params = buildTerminalUnsubscribeParams(stream.params)
if (params) {
this.options.sendFrame({
id: this.options.nextId(),
method: 'terminal.unsubscribe',
params
})
if (stream.sent) {
const byParams = buildParamsUnsubscribe(stream.method, stream.params, id)
if (stream.method === 'terminal.subscribe') {
if (byParams) {
this.sendUnsubscribe(byParams)
}
} else {
const unsubscribe = stream.subscriptionId
? buildReadyStreamUnsubscribe(stream.method, stream.subscriptionId)
: null
if (byParams && stream.method === 'session.tabs.subscribe' && !stream.receivedSnapshot) {
// The host registers cleanup only after resolving the initial snapshot.
this.cancelledSubscriptions.set(id, { method: stream.method, unsubscribe: byParams })
} else if (unsubscribe || byParams) {
this.sendUnsubscribe((unsubscribe ?? byParams)!)
} else if (
stream.method === 'browser.screencast' ||
stream.method === 'runtime.clientEvents.subscribe'
) {
// Keep only the cleanup route while the server assigns its subscription ID.
this.cancelledSubscriptions.set(id, { method: stream.method })
} else if (stream.subscriptionId) {
this.sendUnsubscribe({
method: stream.method.replace(/\.subscribe$/, '.unsubscribe'),
params: { subscriptionId: stream.subscriptionId }
})
}
}
} else if (stream.subscriptionId) {
this.options.sendFrame({
id: this.options.nextId(),
method: stream.method.replace(/\.subscribe$/, '.unsubscribe'),
params: { subscriptionId: stream.subscriptionId }
})
}
this.remove(id)
}
/** Skip the unsubscribe when a live sibling shares the host cleanup token (e.g. nativeChat's
* deterministic `agent:sessionId`), since the host would evict the sibling's registration. */
private sendUnsubscribe(unsubscribe: StreamUnsubscribe): void {
if (this.hasLiveOwner(unsubscribe)) {
return
}
this.options.sendFrame({ id: this.options.nextId(), ...unsubscribe })
}
private hasLiveOwner(unsubscribe: StreamUnsubscribe): boolean {
const token = JSON.stringify(unsubscribe)
for (const [siblingId, sibling] of this.streams) {
if (sibling.cancelled || !sibling.sent) {
continue
}
const siblingUnsubscribe = buildParamsUnsubscribe(sibling.method, sibling.params, siblingId)
if (siblingUnsubscribe && JSON.stringify(siblingUnsubscribe) === token) {
return true
}
}
return false
}
private remove(id: string): void {
const stream = this.streams.get(id)
if (!stream) {
@@ -38,7 +38,8 @@ export function updateTerminalSubscriptionViewport(
* the per-method echo logic out of the rpc-client teardown closure. */
export function buildStreamUnsubscribe(
method: string | undefined,
params: unknown
params: unknown,
requestId?: string
): { method: string; params: Record<string, unknown> } | null {
if (!params || typeof params !== 'object') {
return null
@@ -46,7 +47,10 @@ export function buildStreamUnsubscribe(
if (method === 'session.tabs.subscribe') {
const worktree = (params as { worktree?: unknown }).worktree
return typeof worktree === 'string'
? { method: 'session.tabs.unsubscribe', params: { worktree } }
? {
method: 'session.tabs.unsubscribe',
params: { worktree, ...(requestId ? { subscriptionId: requestId } : {}) }
}
: null
}
if (method === 'nativeChat.subscribe') {
+60 -7
View File
@@ -1,9 +1,15 @@
param(
[Parameter(Mandatory = $true)]
[string]$OperationPath
[Parameter(Position = 0)]
[string]$OperationPath,
# Serve mode keeps one process alive so the Add-Type P/Invoke assembly below
# is emitted once per session instead of once per operation.
[switch]$Serve
)
$ErrorActionPreference = "Stop"
# Progress records render to the host, which in serve mode is a pipe carrying
# one JSON response per line; a stray record would desynchronise the stream.
$ProgressPreference = "SilentlyContinue"
$utf8NoBom = New-Object System.Text.UTF8Encoding $false
[Console]::InputEncoding = $utf8NoBom
[Console]::OutputEncoding = $utf8NoBom
@@ -1313,9 +1319,56 @@ function Invoke-OrcaOperation($Operation) {
[pscustomobject]@{ ok = $true; action = $action; snapshot = $snapshot }
}
try {
$operation = Read-OrcaOperation $OperationPath
Write-OrcaJson (Invoke-OrcaOperation $operation)
} catch {
Write-OrcaJson ([pscustomobject]@{ ok = $false; error = [string]$_.Exception.Message })
function Invoke-OrcaServeLoop {
# Announced before the first read, and after every Add-Type above: a caller
# that never sees this line knows the helper cannot have read a request, let
# alone synthesized a click, so replaying it is provably safe. Inferring that
# from a missing response instead would replay operations that did run.
[Console]::Out.WriteLine('{"ready":true}')
[Console]::Out.Flush()
# One NDJSON request per line in, one response per line out, until stdin closes.
# Responses carry base64 screenshots and routinely exceed a megabyte; ReadLine
# and the console writer are both length-bounded only by memory.
while ($true) {
$line = [Console]::In.ReadLine()
if ($null -eq $line) { break }
if ([string]::IsNullOrWhiteSpace($line)) { continue }
$requestId = $null
try {
$operation = $line | ConvertFrom-Json
$requestId = $operation.requestId
$response = Invoke-OrcaOperation $operation
} catch {
$response = [pscustomobject]@{ ok = $false; error = [string]$_.Exception.Message }
# ConvertFrom-Json throws before the id is read, so recover it from the
# raw line. An error the caller can match is delivered to the request
# that caused it; an unmatched one only trips the caller's desync
# guard, which kills this helper, charges a failure toward its cooldown
# and discards the message below - so a malformed request would be
# reported as a broken stream and its real cause never surface.
if ($null -eq $requestId -and $line -match '"requestId"\s*:\s*(\d+)') {
$requestId = [long]$Matches[1]
}
}
# Echoed so the caller can prove which request a line answers; a reply it
# cannot match is a desynchronised stream, not a usable response.
if ($null -ne $requestId) {
$response | Add-Member -NotePropertyName requestId -NotePropertyValue $requestId -Force
}
[Console]::Out.WriteLine((ConvertTo-Json $response -Depth 100 -Compress))
[Console]::Out.Flush()
}
}
if ($Serve) {
Invoke-OrcaServeLoop
} elseif ([string]::IsNullOrWhiteSpace($OperationPath)) {
Write-OrcaJson ([pscustomobject]@{ ok = $false; error = "runtime.ps1 requires an operation path or -Serve" })
} else {
try {
$operation = Read-OrcaOperation $OperationPath
Write-OrcaJson (Invoke-OrcaOperation $operation)
} catch {
Write-OrcaJson ([pscustomobject]@{ ok = $false; error = [string]$_.Exception.Message })
}
}
+3 -3
View File
@@ -109,7 +109,7 @@ overrides:
monaco-editor>dompurify: 3.4.13
patchedDependencies:
'@vscode/windows-process-tree@0.8.0': 9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585
'@vscode/windows-process-tree@0.8.0': f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e
'@xterm/addon-ligatures@0.11.0-beta.300': 47405b9994b5acf1b4e90b49250358c1ca03649854d59560e7732b72fe336920
'@xterm/addon-search@0.17.0-beta.300': eee5338dd2621ece46e79c61ec06766cd7fadaf79ffdb24e2a8ab68e97ef31f0
'@xterm/addon-serialize@0.15.0-beta.300': 851eac3d75e6d8c013b9f4c053e61d824b23965cb19ecc28e335e05059f3a294
@@ -510,7 +510,7 @@ importers:
optionalDependencies:
'@vscode/windows-process-tree':
specifier: 0.8.0
version: 0.8.0(patch_hash=9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585)
version: 0.8.0(patch_hash=f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e)
sherpa-onnx-darwin-arm64:
specifier: 1.12.37
version: 1.12.37
@@ -9821,7 +9821,7 @@ snapshots:
convert-source-map: 2.0.0
tinyrainbow: 3.1.0
'@vscode/windows-process-tree@0.8.0(patch_hash=9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585)':
'@vscode/windows-process-tree@0.8.0(patch_hash=f8ea245391c94da5770045aeea01fa6de466c2199c6ef46b5b769b398aa9823e)':
dependencies:
node-addon-api: 7.1.0
optional: true
+144
View File
@@ -0,0 +1,144 @@
import { computerUseErrorRecoveryData } from '../shared/computer-use-error-recovery'
import {
matchAutomationOwnerConflict,
stripAutomationOwnerConflictCode
} from '../shared/automation-owner-conflict'
import { automationOwnerConflictRecovery } from './automation-owner-conflict-recovery'
import type { RuntimeRpcFailure } from './runtime-client'
import { RuntimeClientError, RuntimeRpcFailureError } from './runtime/types'
type CliErrorContext = {
commandPath?: readonly string[]
}
export function formatCliError(error: unknown, context: CliErrorContext = {}): string {
const message = error instanceof Error ? error.message : String(error)
if (error instanceof RuntimeClientError && error.code === 'runtime_unavailable') {
if (hasOrchestrationRequestId(error.data)) {
return message
}
return `${message}\nOrca is not running. Run 'orca open' first.`
}
// Why: error-specific recovery must win over the generic computer fallback.
// Classified from the whole error, not just `.code`: a hop that flattens the class leaves only the token.
const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error))
if (conflict) {
return formatMessageWithNextSteps(stripAutomationOwnerConflictCode(message), conflict.nextSteps)
}
if (error instanceof RuntimeClientError) {
const nextSteps = nextStepsFromData(error.data)
if (nextSteps.length > 0) {
return formatMessageWithNextSteps(message, nextSteps)
}
if (error.code === 'invalid_argument' && context.commandPath?.[0] === 'computer') {
return formatMessageWithNextSteps(
message,
computerUseErrorRecoveryData('invalid_argument')?.nextSteps ?? []
)
}
}
if (
error instanceof RuntimeRpcFailureError &&
error.response.error.code === 'runtime_unavailable'
) {
return `${message}\nOrca is not running. Run 'orca open' first.`
}
if (error instanceof RuntimeRpcFailureError) {
return formatMessageWithNextSteps(message, nextStepsFromData(error.response.error.data))
}
return message
}
function hasOrchestrationRequestId(data: unknown): boolean {
return (
data !== null &&
typeof data === 'object' &&
typeof (data as { orchestrationRequestId?: unknown }).orchestrationRequestId === 'string'
)
}
export function reportCliError(error: unknown, json: boolean, context: CliErrorContext = {}): void {
if (json) {
if (error instanceof RuntimeRpcFailureError) {
console.log(JSON.stringify(withAutomationOwnerConflictRecovery(error.response), null, 2))
} else {
const response: RuntimeRpcFailure = {
id: 'local',
ok: false,
error: {
code:
matchAutomationOwnerConflict(error) ??
(error instanceof RuntimeClientError ? error.code : 'runtime_error'),
message: stripAutomationOwnerConflictCode(
error instanceof Error ? error.message : String(error)
),
data: localCliErrorData(error, context)
},
_meta: {
runtimeId: null
}
}
console.log(JSON.stringify(response, null, 2))
}
} else {
console.error(formatCliError(error, context))
}
}
/** Machine-readable half of the same recovery the human message carries. */
function withAutomationOwnerConflictRecovery(response: RuntimeRpcFailure): RuntimeRpcFailure {
const code = matchAutomationOwnerConflict(response)
const conflict = automationOwnerConflictRecovery(code)
if (!conflict || !code) {
return response
}
return {
...response,
error: {
...response.error,
// Restores the classification a flattening hop dropped, so --json consumers read the conflict, not the transport.
code,
message: stripAutomationOwnerConflictCode(response.error.message),
data: response.error.data ?? conflict
}
}
}
function formatMessageWithNextSteps(message: string, nextSteps: readonly string[]): string {
if (nextSteps.length === 0) {
return message
}
return `${message}\n${nextSteps.map((step) => `Next step: ${step}`).join('\n')}`
}
function nextStepsFromData(data: unknown): string[] {
if (
data &&
typeof data === 'object' &&
Array.isArray((data as { nextSteps?: unknown }).nextSteps)
) {
return (data as { nextSteps: unknown[] }).nextSteps.filter(
(step): step is string => typeof step === 'string'
)
}
return []
}
function localCliErrorData(error: unknown, context: CliErrorContext): unknown {
// Why: error-specific recovery must win over the generic computer fallback.
if (error instanceof RuntimeClientError && error.data !== undefined) {
return error.data
}
const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error))
if (conflict) {
return conflict
}
if (
error instanceof RuntimeClientError &&
error.code === 'invalid_argument' &&
context.commandPath?.[0] === 'computer'
) {
return computerUseErrorRecoveryData('invalid_argument')
}
return undefined
}
+44
View File
@@ -0,0 +1,44 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import * as distance from '../shared/edit-distance'
import { suggestCommands, unknownFlagData } from './command-suggestion'
import type { CommandSpec } from './command-spec'
const specs: CommandSpec[] = [
{ path: ['list'], summary: '', usage: '', allowedFlags: [] },
{ path: ['remove'], summary: '', usage: '', allowedFlags: [], destructive: true }
]
afterEach(() => vi.restoreAllMocks())
describe('suggestion distance work', () => {
it('does no distance calculations for a long command, including destructive intent', () => {
const spy = vi.spyOn(distance, 'levenshtein')
expect(suggestCommands(specs, ['x'.repeat(32_768)])).toEqual([])
expect(spy).not.toHaveBeenCalled()
})
it('does no distance calculations for a long flag but still lists valid flags', () => {
const spy = vi.spyOn(distance, 'levenshtein')
expect(unknownFlagData('x'.repeat(32_768), ['worktree', 'json'])).toEqual({
validFlags: ['json', 'worktree'],
suggestions: [],
nextSteps: ['Valid flags: --json, --worktree']
})
expect(spy).not.toHaveBeenCalled()
})
it('keeps the inclusive three-edit suggestion boundary', () => {
expect(suggestCommands(specs, ['listxxx'])).toEqual(['list'])
expect(unknownFlagData('jsonxxx', ['json']).suggestions).toEqual(['json'])
})
it('keeps the inclusive one-edit destructive intent boundary', () => {
expect(suggestCommands(specs, ['remov'])).toEqual(['remove'])
expect(suggestCommands(specs, ['remo'])).toEqual([])
})
it('retains UTF-16 distance semantics at the length boundary', () => {
expect(unknownFlagData('json😀x', ['json']).suggestions).toEqual(['json'])
expect(unknownFlagData('json😀😀', ['json']).suggestions).toEqual([])
})
})
+14 -5
View File
@@ -37,7 +37,10 @@ function destructiveVerbs(specs: CommandSpec[]): Set<string> {
// input token is itself a near-miss of a destructive verb. #6303
function intendsDestruction(inputToken: string, verbs: Set<string>): boolean {
for (const verb of verbs) {
if (levenshtein(inputToken, verb) <= DESTRUCTIVE_INTENT_THRESHOLD) {
if (
Math.abs(inputToken.length - verb.length) <= DESTRUCTIVE_INTENT_THRESHOLD &&
levenshtein(inputToken, verb) <= DESTRUCTIVE_INTENT_THRESHOLD
) {
return true
}
}
@@ -85,7 +88,9 @@ export function suggestCommands(specs: CommandSpec[], commandPath: string[]): st
continue
}
seen.add(joined)
scored.push({ label: joined, distance: levenshtein(input, joined) })
if (Math.abs(input.length - joined.length) <= SUGGESTION_THRESHOLD) {
scored.push({ label: joined, distance: levenshtein(input, joined) })
}
}
}
return rankByDistance(scored)
@@ -106,9 +111,13 @@ export type FlagErrorData = {
}
function suggestFlags(flag: string, validFlags: string[]): string[] {
return rankByDistance(
validFlags.map((candidate) => ({ label: candidate, distance: levenshtein(flag, candidate) }))
)
const scored: { label: string; distance: number }[] = []
for (const candidate of validFlags) {
if (Math.abs(flag.length - candidate.length) <= SUGGESTION_THRESHOLD) {
scored.push({ label: candidate, distance: levenshtein(flag, candidate) })
}
}
return rankByDistance(scored)
}
// Why: include the accepted set so agents can recover without another help call.
+3 -144
View File
@@ -1,13 +1,8 @@
import type { CliStatusResult } from '../shared/runtime-types'
import { computerUseErrorRecoveryData } from '../shared/computer-use-error-recovery'
import {
matchAutomationOwnerConflict,
stripAutomationOwnerConflictCode
} from '../shared/automation-owner-conflict'
import { automationOwnerConflictRecovery } from './automation-owner-conflict-recovery'
import { prepareComputerCliJsonResult } from './computer-format'
import type { RuntimeRpcFailure, RuntimeRpcSuccess } from './runtime-client'
import { RuntimeClientError, RuntimeRpcFailureError } from './runtime/types'
import type { RuntimeRpcSuccess } from './runtime-client'
export { formatCliError, reportCliError } from './cli-error'
export {
formatBrowserProfileList,
@@ -67,10 +62,6 @@ export {
formatWorktreeShow
} from './workspace-format'
type CliErrorContext = {
commandPath?: readonly string[]
}
export function printResult<TResult>(
response: RuntimeRpcSuccess<TResult>,
json: boolean,
@@ -83,138 +74,6 @@ export function printResult<TResult>(
console.log(formatter(response.result))
}
export function formatCliError(error: unknown, context: CliErrorContext = {}): string {
const message = error instanceof Error ? error.message : String(error)
if (error instanceof RuntimeClientError && error.code === 'runtime_unavailable') {
if (hasOrchestrationRequestId(error.data)) {
return message
}
return `${message}\nOrca is not running. Run 'orca open' first.`
}
// Why: error-specific recovery must win over the generic computer fallback.
// Classified from the whole error, not just `.code`: a hop that flattens the class leaves only the token.
const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error))
if (conflict) {
return formatMessageWithNextSteps(stripAutomationOwnerConflictCode(message), conflict.nextSteps)
}
if (error instanceof RuntimeClientError) {
const nextSteps = nextStepsFromData(error.data)
if (nextSteps.length > 0) {
return formatMessageWithNextSteps(message, nextSteps)
}
if (error.code === 'invalid_argument' && context.commandPath?.[0] === 'computer') {
return formatMessageWithNextSteps(
message,
computerUseErrorRecoveryData('invalid_argument')?.nextSteps ?? []
)
}
}
if (
error instanceof RuntimeRpcFailureError &&
error.response.error.code === 'runtime_unavailable'
) {
return `${message}\nOrca is not running. Run 'orca open' first.`
}
if (error instanceof RuntimeRpcFailureError) {
return formatMessageWithNextSteps(message, nextStepsFromData(error.response.error.data))
}
return message
}
function hasOrchestrationRequestId(data: unknown): boolean {
return (
data !== null &&
typeof data === 'object' &&
typeof (data as { orchestrationRequestId?: unknown }).orchestrationRequestId === 'string'
)
}
export function reportCliError(error: unknown, json: boolean, context: CliErrorContext = {}): void {
if (json) {
if (error instanceof RuntimeRpcFailureError) {
console.log(JSON.stringify(withAutomationOwnerConflictRecovery(error.response), null, 2))
} else {
const response: RuntimeRpcFailure = {
id: 'local',
ok: false,
error: {
code:
matchAutomationOwnerConflict(error) ??
(error instanceof RuntimeClientError ? error.code : 'runtime_error'),
message: stripAutomationOwnerConflictCode(
error instanceof Error ? error.message : String(error)
),
data: localCliErrorData(error, context)
},
_meta: {
runtimeId: null
}
}
console.log(JSON.stringify(response, null, 2))
}
} else {
console.error(formatCliError(error, context))
}
}
/** Machine-readable half of the same recovery the human message carries. */
function withAutomationOwnerConflictRecovery(response: RuntimeRpcFailure): RuntimeRpcFailure {
const code = matchAutomationOwnerConflict(response)
const conflict = automationOwnerConflictRecovery(code)
if (!conflict || !code) {
return response
}
return {
...response,
error: {
...response.error,
// Restores the classification a flattening hop dropped, so --json consumers read the conflict, not the transport.
code,
message: stripAutomationOwnerConflictCode(response.error.message),
data: response.error.data ?? conflict
}
}
}
function formatMessageWithNextSteps(message: string, nextSteps: readonly string[]): string {
if (nextSteps.length === 0) {
return message
}
return `${message}\n${nextSteps.map((step) => `Next step: ${step}`).join('\n')}`
}
function nextStepsFromData(data: unknown): string[] {
if (
data &&
typeof data === 'object' &&
Array.isArray((data as { nextSteps?: unknown }).nextSteps)
) {
return (data as { nextSteps: unknown[] }).nextSteps.filter(
(step): step is string => typeof step === 'string'
)
}
return []
}
function localCliErrorData(error: unknown, context: CliErrorContext): unknown {
// Why: error-specific recovery must win over the generic computer fallback.
if (error instanceof RuntimeClientError && error.data !== undefined) {
return error.data
}
const conflict = automationOwnerConflictRecovery(matchAutomationOwnerConflict(error))
if (conflict) {
return conflict
}
if (
error instanceof RuntimeClientError &&
error.code === 'invalid_argument' &&
context.commandPath?.[0] === 'computer'
) {
return computerUseErrorRecoveryData('invalid_argument')
}
return undefined
}
export type HostListEntry = {
kind: 'local' | 'ssh' | 'environment'
name: string
+1 -1
View File
@@ -15,7 +15,7 @@ import {
resolveHostFlagEnvironmentId
} from './execution-host-flag'
import { listSshTargets } from './host-selector-alternatives'
import { reportCliError } from './format'
import { reportCliError } from './cli-error'
import { printHelp } from './help'
import type { RuntimeClient } from './runtime-client'
import { COMMAND_SPECS } from './specs'
+3 -4
View File
@@ -84,14 +84,12 @@ describe('RuntimeClient module-graph deferral', () => {
process.exitCode = 0
})
// Why: the whole point of the change. These six modules load on EVERY
// invocation, so a value-import of the barrel from any of them drags the
// RuntimeClient graph (zod, ws, tweetnacl) back onto the --help path.
// These eager modules must not pull the RuntimeClient dependency graph into help.
it.each([
'args.ts',
'flags.ts',
'dispatch.ts',
'format.ts',
'cli-error.ts',
'selectors.ts',
'execution-host-flag.ts'
])('%s imports error classes from ./runtime/types, not the barrel', (file) => {
@@ -110,6 +108,7 @@ describe('RuntimeClient module-graph deferral', () => {
expect(source).toContain("import type { RuntimeClient } from './runtime-client'")
expect(source).not.toMatch(/^import \{[^}]*RuntimeClient[^}]*\} from '\.\/runtime-client'/m)
expect(source).toContain("await import('./runtime-client.js')")
expect(source).toContain("import { reportCliError } from './cli-error'")
})
it('constructs no client for --help', async () => {
+150
View File
@@ -0,0 +1,150 @@
import { EventEmitter } from 'node:events'
import { StringDecoder } from 'node:string_decoder'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import type { RuntimeMetadata } from '../../shared/runtime-bootstrap'
import { sendRequest } from './transport'
const { createConnection } = vi.hoisted(() => ({ createConnection: vi.fn() }))
vi.mock('node:net', () => ({ createConnection }))
vi.mock('node:crypto', () => ({ randomUUID: () => 'request-1' }))
const metadata: RuntimeMetadata = {
runtimeId: 'runtime-1',
pid: 123,
transports: [{ kind: 'unix', endpoint: 'test-only' }],
authToken: 'token',
startedAt: 1
}
const reply = (result: unknown) =>
`${JSON.stringify({ id: 'request-1', ok: true, result, _meta: { runtimeId: 'runtime-1' } })}\n`
class TestSocket extends EventEmitter {
setEncoding = vi.fn()
write = vi.fn()
end = vi.fn()
destroy = vi.fn()
}
let socket: TestSocket
beforeEach(() => {
socket = new TestSocket()
createConnection.mockReturnValue(socket)
})
afterEach(() => {
vi.restoreAllMocks()
vi.useRealTimers()
})
describe('CLI runtime response framing', () => {
it.each([1, 7, 256, 4096])(
'reads a fragmented response with %i-character chunks',
async (size) => {
const result = { data: '界😀'.repeat(10000) }
const encoded = reply(result)
const pending = sendRequest(metadata, 'terminal.read', {}, 30000)
for (let offset = 0; offset < encoded.length; offset += size) {
socket.emit('data', encoded.slice(offset, offset + size))
}
await expect(pending).resolves.toMatchObject({ result })
expect(socket.setEncoding).toHaveBeenCalledExactlyOnceWith('utf8')
expect(socket.end).toHaveBeenCalledOnce()
}
)
it('accepts Unicode split across socket bytes using the existing UTF-8 decoder', async () => {
const pending = sendRequest(metadata, 'terminal.read', {}, 30000)
const decoder = new StringDecoder('utf8')
for (const byte of Buffer.from(reply({ data: '界😀é' }))) {
socket.emit('data', decoder.write(Buffer.from([byte])))
}
socket.emit('data', decoder.end())
await expect(pending).resolves.toMatchObject({ result: { data: '界😀é' } })
})
it('searches each fragment once without rescanning the accumulated reply', async () => {
const encoded = reply({ data: 'x'.repeat(1024 * 1024) })
const pending = sendRequest(metadata, 'terminal.read', {}, 30000)
const originalIndexOf = String.prototype.indexOf
let searchedCharacters = 0
const search = vi
.spyOn(String.prototype, 'indexOf')
.mockImplementation(function (this: string, value, position) {
if (value === '\n') {
searchedCharacters += this.length - (position ?? 0)
}
return originalIndexOf.call(this, value, position)
})
try {
for (let offset = 0; offset < encoded.length; offset += 256) {
socket.emit('data', encoded.slice(offset, offset + 256))
}
} finally {
search.mockRestore()
}
await expect(pending).resolves.toMatchObject({ ok: true })
expect(searchedCharacters).toBe(encoded.length)
})
it('refreshes keepalives across chunks and ignores blanks and data after the final frame', async () => {
vi.useFakeTimers()
const pending = sendRequest(metadata, 'terminal.read', {}, 100)
await vi.advanceTimersByTimeAsync(90)
socket.emit('data', ' \r\n{"_keep')
socket.emit('data', 'alive":true}\n\t\n')
await vi.advanceTimersByTimeAsync(90)
socket.emit('data', `${reply({ data: 'done' })}invalid JSON\n`)
socket.emit('data', 'more ignored data')
await expect(pending).resolves.toMatchObject({ result: { data: 'done' } })
expect(socket.end).toHaveBeenCalledOnce()
expect(vi.getTimerCount()).toBe(0)
})
it.each([
['broken JSON\n', 'invalid_runtime_response'],
['{}\n', 'invalid_runtime_response'],
['{"id":"other","ok":true,"result":{}}\n', 'invalid_runtime_response'],
[
'{"id":"request-1","ok":true,"result":{},"_meta":{"runtimeId":"other"}}\n',
'runtime_unavailable'
]
])(
'rejects a fragmented invalid first frame before subsequent valid frames',
async (line, code) => {
const pending = sendRequest(metadata, 'terminal.read', {}, 30000)
socket.emit('data', line.slice(0, 2))
socket.emit('data', line.slice(2) + reply({ data: 'ignored' }))
await expect(pending).rejects.toMatchObject({ code })
expect(socket.end).toHaveBeenCalledOnce()
}
)
it('preserves terminal failure envelopes', async () => {
const pending = sendRequest(metadata, 'terminal.read', {}, 30000)
socket.emit('data', '{"id":"request-1","ok":false,"error":{"code":"bad","message":"no"}}\n')
await expect(pending).resolves.toMatchObject({
ok: false,
error: { code: 'bad', message: 'no' }
})
})
it('rejects close with an incomplete frame and does not parse later data', async () => {
const pending = sendRequest(metadata, 'terminal.read', {}, 30000)
socket.emit('data', '{"id":')
socket.emit('close')
socket.emit('data', reply({ data: 'ignored' }))
await expect(pending).rejects.toMatchObject({ code: 'runtime_unavailable' })
expect(socket.end).toHaveBeenCalledOnce()
})
it('destroys a timed out socket holding an incomplete frame', async () => {
vi.useFakeTimers()
const pending = sendRequest(metadata, 'terminal.read', {}, 100)
const rejected = expect(pending).rejects.toMatchObject({ code: 'runtime_timeout' })
socket.emit('data', '{"id":')
await vi.advanceTimersByTimeAsync(100)
socket.emit('data', reply({ data: 'ignored' }))
await rejected
expect(socket.destroy).toHaveBeenCalledOnce()
expect(vi.getTimerCount()).toBe(0)
})
})
+18 -9
View File
@@ -31,7 +31,7 @@ export async function sendRequest<TResult>(
return
}
const socket = createConnection(transport.endpoint)
let buffer = ''
let lineSegments: string[] = []
let settled = false
const requestId = randomUUID()
@@ -40,6 +40,7 @@ export async function sendRequest<TResult>(
return
}
settled = true
lineSegments = []
socket.destroy()
reject(
new RuntimeClientError(
@@ -56,6 +57,7 @@ export async function sendRequest<TResult>(
return
}
settled = true
lineSegments = []
clearTimeout(timeout)
socket.end()
if (result.ok === false) {
@@ -89,18 +91,27 @@ export async function sendRequest<TResult>(
})
})
socket.on('data', (chunk: string) => {
buffer += chunk
// Why: the server may interleave `{"_keepalive":true}\n` frames with the
// final success/failure frame to keep both idle timers alive during a
// long-poll (see design doc §3.1). Read frames in a loop until we see a
// terminal frame. Each keepalive refreshes the client-side timer so a
// 10 min wait doesn't trip the 60 s default ceiling.
let newlineIndex = buffer.indexOf('\n')
while (newlineIndex !== -1 && !settled) {
const line = buffer.slice(0, newlineIndex)
buffer = buffer.slice(newlineIndex + 1)
let cursor = 0
while (cursor < chunk.length && !settled) {
const newlineIndex = chunk.indexOf('\n', cursor)
if (newlineIndex === -1) {
lineSegments.push(chunk.slice(cursor))
return
}
const segment = chunk.slice(cursor, newlineIndex)
let line = segment
if (lineSegments.length > 0) {
lineSegments.push(segment)
line = lineSegments.join('')
lineSegments = []
}
cursor = newlineIndex + 1
if (line.trim().length === 0) {
newlineIndex = buffer.indexOf('\n')
continue
}
@@ -124,7 +135,6 @@ export async function sendRequest<TResult>(
// major). See §7 risk #9.
if (isKeepaliveFrame(raw)) {
timeout.refresh()
newlineIndex = buffer.indexOf('\n')
continue
}
@@ -150,7 +160,6 @@ export async function sendRequest<TResult>(
const frame = parsed.data
if ('_keepalive' in frame) {
timeout.refresh()
newlineIndex = buffer.indexOf('\n')
continue
}
@@ -1,4 +1,3 @@
import { execFileSync } from 'node:child_process'
import { mkdtemp, readFile, readdir, rm, stat, truncate, writeFile } from 'node:fs/promises'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
@@ -16,9 +15,14 @@ import {
getOrCreateArtifactCreateIntent,
removeArtifactCreateIntent
} from './artifact-create-intent-store'
import { runProcessSync } from '../../shared/child-process/run-process'
import { __resetSecureFileWindowsUserSidForTests } from '../../shared/secure-file'
import type { ArtifactShareScope } from './artifact-share-record-store'
vi.mock('node:child_process', () => ({ execFile: vi.fn(), execFileSync: vi.fn() }))
vi.mock('../../shared/child-process/run-process', () => ({
runProcess: vi.fn(),
runProcessSync: vi.fn()
}))
const createdPaths: string[] = []
const scope: ArtifactShareScope = {
@@ -168,12 +172,27 @@ describe('artifact create intent store', () => {
expect((await readdir(directory)).some((name) => name.endsWith('.tmp'))).toBe(false)
})
it('hardens one Windows journal directory without per-file PowerShell launches', async () => {
it('hardens one Windows journal directory without per-file ACL launches', async () => {
const originalPlatform = Object.getOwnPropertyDescriptor(process, 'platform')
Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' })
vi.mocked(execFileSync).mockImplementation((file) =>
String(file).endsWith('whoami.exe') ? '"USER","S-1-5-21-1000"' : ''
)
const ok = { code: 0, signal: null, stdout: '', stderr: '', timedOut: false }
// Earlier cases in this file already resolved (and cached) the SID against an unstubbed mock.
__resetSecureFileWindowsUserSidForTests()
vi.mocked(runProcessSync).mockImplementation((spec) => {
if (spec.program.endsWith('whoami.exe')) {
return { ...ok, stdout: '"USER","S-1-5-21-1000"' }
}
const args = spec.args ?? []
if (args.length > 1) {
return ok // /reset and the /grant:r pass
}
// The verify pass re-reads the DACL; answer with the three protected inheritable rules.
const rules = ['host\\me', 'NT AUTHORITY\\SYSTEM', 'BUILTIN\\Administrators'].map(
(name, index) =>
index === 0 ? `${args[0]} ${name}:(OI)(CI)(F)` : ` ${name}:(OI)(CI)(F)`
)
return { ...ok, stdout: `${rules.join('\r\n')}\r\n\r\nSuccessfully processed 1 files\r\n` }
})
try {
const userDataPath = await createUserDataPath()
getOrCreateArtifactCreateIntent(
@@ -193,16 +212,20 @@ describe('artifact create intent store', () => {
body
)
const powershellCalls = vi
.mocked(execFileSync)
.mock.calls.filter(([file]) => String(file).endsWith('powershell.exe'))
expect(powershellCalls).toHaveLength(1)
expect((powershellCalls[0]![1] as string[]).at(-1)).toBe('1')
// One harden across both intents: counted by its /reset pass, which opens each harden.
const aclCalls = vi
.mocked(runProcessSync)
.mock.calls.map(([spec]) => spec)
.filter((spec) => spec.program.endsWith('icacls.exe'))
expect(aclCalls.filter((spec) => spec.args?.includes('/reset'))).toHaveLength(1)
// The child intent files rely on inheritance, so the directory rules must carry (OI)(CI).
const grant = aclCalls.find((spec) => spec.args?.includes('/grant:r'))
expect(grant?.args?.filter((arg) => arg.endsWith(':(OI)(CI)(F)'))).toHaveLength(3)
} finally {
if (originalPlatform) {
Object.defineProperty(process, 'platform', originalPlatform)
}
vi.mocked(execFileSync).mockReset()
vi.mocked(runProcessSync).mockReset()
}
})
@@ -1,4 +1,4 @@
import { describe, expect, it } from 'vitest'
import { describe, expect, it, vi } from 'vitest'
import type { BrowserClientHostCommandEvent } from '../../shared/browser-client-host-protocol'
import {
@@ -102,3 +102,38 @@ describe('readBrowserClientUploadPaths', () => {
)
})
})
it.each([0, 1, 128 * 1024])(
'avoids recopying 16 single-chunk uploads of %i bytes',
async (size) => {
const source = Buffer.alloc(size, 171)
const response = {
contentBase64: source.toString('base64'),
bytesRead: size,
totalBytes: size,
eof: true
}
const remotePaths = Array.from({ length: 16 }, (_, i) => `file-${i}.bin`)
const request = vi.fn(async () => response)
const concat = vi.spyOn(Buffer, 'concat')
let copies = 0
let files: Awaited<ReturnType<typeof fetchBrowserClientUploadFiles>>
try {
files = await fetchBrowserClientUploadFiles({ request, event, remotePaths })
copies = concat.mock.calls.length
} finally {
concat.mockRestore()
}
expect(copies).toBe(0)
expect(request).toHaveBeenCalledTimes(16)
expect(files.map((file) => file.remotePath)).toEqual(remotePaths)
for (const file of files) {
expect(file.contents).toEqual(source)
}
if (size > 0) {
files[0].contents[0] = 0
expect(files[1].contents[0]).toBe(171)
expect(source[0]).toBe(171)
}
}
)
@@ -74,7 +74,7 @@ export async function fetchBrowserClientUploadFiles(options: {
throw new Error('browser_client_upload_transfer_stalled')
}
}
files.push({ remotePath, contents: Buffer.concat(chunks) })
files.push({ remotePath, contents: chunks.length === 1 ? chunks[0] : Buffer.concat(chunks) })
}
return files
}
@@ -0,0 +1,45 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import {
isComputerSidecarDiagnostic,
reportComputerDiagnostic
} from './computer-sidecar-diagnostics'
describe('computer sidecar diagnostics', () => {
const originalSend = process.send
afterEach(() => {
process.send = originalSend
vi.restoreAllMocks()
})
it('sends over IPC when running inside the sidecar', () => {
const send = vi.fn((_message: unknown) => true)
process.send = send as unknown as typeof process.send
const console_ = vi.spyOn(console, 'warn').mockImplementation(() => {})
reportComputerDiagnostic('fell back to Bypass')
// The sidecar's stdout is piped and never read, so this must not go there.
expect(console_).not.toHaveBeenCalled()
expect(send).toHaveBeenCalledWith({
kind: 'computer-sidecar-diagnostic',
message: 'fell back to Bypass'
})
expect(isComputerSidecarDiagnostic(send.mock.calls[0][0])).toBe(true)
})
it('logs directly when there is no IPC channel', () => {
process.send = undefined
const console_ = vi.spyOn(console, 'warn').mockImplementation(() => {})
reportComputerDiagnostic('fell back to Bypass')
expect(console_).toHaveBeenCalledWith('[computer-use] fell back to Bypass')
})
it('does not mistake a sidecar response for a diagnostic', () => {
expect(isComputerSidecarDiagnostic({ id: 1, ok: true, result: {} })).toBe(false)
expect(isComputerSidecarDiagnostic({ kind: 'computer-sidecar-diagnostic' })).toBe(false)
expect(isComputerSidecarDiagnostic(null)).toBe(false)
})
})
@@ -0,0 +1,38 @@
/**
* Warnings from the computer-use provider, routed to somewhere a human sees.
*
* Why not `console.warn`: the provider runs inside the forked sidecar, which
* `sidecar-client.ts` starts with piped stdio that nothing ever reads. Anything
* written there is discarded — including the only signal that a machine has
* fallen back to `-ExecutionPolicy Bypass`, a state that persists for the
* session. The sidecar has an IPC channel already, so the warning takes it.
*/
export type ComputerSidecarDiagnostic = {
kind: 'computer-sidecar-diagnostic'
message: string
}
const DIAGNOSTIC_KIND = 'computer-sidecar-diagnostic'
export function isComputerSidecarDiagnostic(
message: unknown
): message is ComputerSidecarDiagnostic {
if (!message || typeof message !== 'object') {
return false
}
const record = message as Record<string, unknown>
return record.kind === DIAGNOSTIC_KIND && typeof record.message === 'string'
}
export function reportComputerDiagnostic(message: string): void {
if (process.send) {
process.send({ kind: DIAGNOSTIC_KIND, message } satisfies ComputerSidecarDiagnostic)
return
}
logComputerDiagnostic(message)
}
/** The main-process end: how a sidecar's forwarded diagnostic is printed. */
export function logComputerDiagnostic(message: string): void {
console.warn(`[computer-use] ${message}`)
}
@@ -228,3 +228,17 @@ export function elementParam(
}
return element
}
/**
* Tools that only observe, and so may be safely re-sent to a fresh helper.
*
* Why an allowlist: a helper can die after running an operation but before
* writing its reply, so a replayed mutation is a second click, keystroke or
* paste. Only the observation tools are provably safe to repeat, and a tool
* added later has to opt in rather than inherit a replay by default.
*/
const OBSERVATION_TOOLS = new Set(['handshake', 'list_apps', 'list_windows', 'get_app_state'])
export function isReplayableTool(tool: string): boolean {
return OBSERVATION_TOOLS.has(tool)
}
@@ -1,28 +1,98 @@
import { execFile } from 'node:child_process'
import { windowsPowerShellPath } from '../../shared/child-process/windows-system-binary'
import { reportComputerDiagnostic } from './computer-sidecar-diagnostics'
import { RuntimeClientError } from './runtime-client-error'
import type { DesktopScriptPlatform } from './desktop-script-provider-paths'
import {
FALLBACK_WINDOWS_EXECUTION_POLICY,
PREFERRED_WINDOWS_EXECUTION_POLICY,
isExecutionPolicyBlocked,
windowsPowerShellRuntimeArgs
} from './windows-powershell-execution-policy'
const REQUEST_TIMEOUT_MS = 30_000
const FORCE_KILL_GRACE_MS = 1_000
export function execBridge(
export async function execBridge(
platform: DesktopScriptPlatform,
scriptPath: string,
operationPath: string
): Promise<{ stdout: string; stderr: string }> {
const command = platform === 'windows' ? 'powershell.exe' : 'python3'
const args =
platform === 'windows'
? [
'-NoProfile',
'-NonInteractive',
'-ExecutionPolicy',
'Bypass',
'-File',
scriptPath,
operationPath
]
: [scriptPath, operationPath]
if (platform !== 'windows') {
return await mapped(runBridgeProcess('python3', [scriptPath, operationPath]))
}
const command = windowsPowerShellPath()
try {
return await runBridgeProcess(
command,
windowsPowerShellRuntimeArgs(scriptPath, PREFERRED_WINDOWS_EXECUTION_POLICY, [operationPath])
)
} catch (error) {
if (!isPolicyBlockedStart(error)) {
throw error instanceof BridgeProcessFailure ? error.mapped : error
}
reportComputerDiagnostic(
`bridge start blocked at ${PREFERRED_WINDOWS_EXECUTION_POLICY}; retrying once with ${FALLBACK_WINDOWS_EXECUTION_POLICY}`
)
return await mapped(
runBridgeProcess(
command,
windowsPowerShellRuntimeArgs(scriptPath, FALLBACK_WINDOWS_EXECUTION_POLICY, [operationPath])
)
)
}
}
/** Unwrap the raw-stream carrier back into the error callers expect. */
async function mapped(
run: Promise<{ stdout: string; stderr: string }>
): Promise<{ stdout: string; stderr: string }> {
try {
return await run
} catch (error) {
throw error instanceof BridgeProcessFailure ? error.mapped : error
}
}
/**
* Only a run that produced no stdout at all may be replayed.
*
* What the stdout guard covers: operations are not idempotent, and the response
* embeds window titles and element names, so a snapshot that merely contains
* the word "SecurityError" must not be read as a policy block and replayed as a
* second click, keystroke or paste. It closes that injection route only.
*
* What it does not cover: one-shot mode runs the operation to completion and
* writes stdout only afterwards, so stdout is empty for the whole action, not
* just before it starts. A crash after the click but before the write looks
* identical to a helper that never started. Nothing here can tell those apart —
* only a policy pattern that cannot match a non-policy failure keeps the replay
* off, which is why its `\b` is load-bearing rather than cosmetic.
*/
function isPolicyBlockedStart(error: unknown): error is BridgeProcessFailure {
return (
error instanceof BridgeProcessFailure &&
!error.stdout.trim() &&
isExecutionPolicyBlocked(error.stderr)
)
}
/** Carries the raw streams so the retry decision does not read a mapped message. */
class BridgeProcessFailure extends Error {
constructor(
readonly stdout: string,
readonly stderr: string,
readonly mapped: RuntimeClientError
) {
super(mapped.message)
this.name = 'BridgeProcessFailure'
}
}
function runBridgeProcess(
command: string,
args: readonly string[]
): Promise<{ stdout: string; stderr: string }> {
return new Promise((resolve, reject) => {
let child: ReturnType<typeof execFile> | null = null
let settled = false
@@ -76,7 +146,7 @@ export function execBridge(
try {
child = execFile(
command,
args,
[...args],
{
env: process.env,
maxBuffer: 20 * 1024 * 1024,
@@ -86,11 +156,10 @@ export function execBridge(
(error, stdout, stderr) => {
if (error) {
const message = stderr.trim() || stdout.trim() || error.message
finish(
error.killed
? new RuntimeClientError('action_timeout', message)
: mapBridgeError(message)
)
const mapped = error.killed
? new RuntimeClientError('action_timeout', message)
: mapBridgeError(message)
finish(new BridgeProcessFailure(stdout, stderr, mapped))
return
}
finish(null, { stdout, stderr })
@@ -35,6 +35,7 @@ import type {
BridgeResponse,
NativeActionMethod
} from './desktop-script-provider-types'
import { DesktopScriptRuntimeHost, isRuntimeHostUnavailable } from './desktop-script-runtime-host'
import { DesktopScriptSnapshotStore } from './desktop-script-snapshot-store'
import { normalizeBridgeApp, renderSnapshot } from './desktop-script-snapshot-rendering'
import { normalizeComputerActionResult } from './computer-action-verification-normalization'
@@ -51,12 +52,17 @@ export class DesktopScriptProviderClient {
constructor(
private readonly platform: DesktopScriptPlatform = requiredPlatform(),
private readonly scriptPath: string = requiredScriptPath()
private readonly scriptPath: string = requiredScriptPath(),
private readonly runtimeHost: DesktopScriptRuntimeHost | null = defaultRuntimeHost(
platform,
scriptPath
)
) {}
shutdown(): void {
this.snapshotStore.clear()
this.providerCapabilities = null
this.runtimeHost?.dispose()
}
async listApps(): Promise<ComputerListAppsResult> {
@@ -203,6 +209,23 @@ export class DesktopScriptProviderClient {
}
private async callBridge(request: BridgeRequest): Promise<BridgeResponse> {
const host = this.runtimeHost
if (host) {
try {
return checkedBridgeResponse(await host.request(request), '')
} catch (error) {
// Only a helper that cannot start falls back; operation errors surface.
// The host is kept: it re-probes after its cooldown, so a transient bad
// spawn cannot strand the session on one powershell.exe per operation.
if (!isRuntimeHostUnavailable(error)) {
throw error
}
}
}
return await this.callOneShotBridge(request)
}
private async callOneShotBridge(request: BridgeRequest): Promise<BridgeResponse> {
const operationDirectory = await mkdtemp(join(tmpdir(), 'orca-computer-use-'))
const operationPath = join(operationDirectory, 'operation.json')
try {
@@ -217,10 +240,7 @@ export class DesktopScriptProviderClient {
`desktop provider returned invalid JSON: ${error instanceof Error ? error.message : String(error)}`
)
}
if (!response.ok) {
throw mapBridgeError(response.error ?? stderr)
}
return response
return checkedBridgeResponse(response, stderr)
} finally {
await rm(operationDirectory, { force: true, recursive: true })
}
@@ -255,6 +275,21 @@ export class DesktopScriptProviderClient {
}
}
function checkedBridgeResponse(response: BridgeResponse, stderr: string): BridgeResponse {
if (!response.ok) {
throw mapBridgeError(response.error ?? stderr)
}
return response
}
// Why Windows only: the Linux provider is a python3 one-shot with no serve mode.
function defaultRuntimeHost(
platform: DesktopScriptPlatform,
scriptPath: string
): DesktopScriptRuntimeHost | null {
return platform === 'windows' ? new DesktopScriptRuntimeHost(scriptPath) : null
}
function requiredPlatform(): DesktopScriptPlatform {
const platform = desktopScriptPlatform()
if (!platform) {
@@ -0,0 +1,139 @@
import { afterEach, describe, expect, it, vi } from 'vitest'
import {
bridgeProcessArgs,
createDesktopScriptProviderClient,
expectDesktopProviderSubprocessStartCount,
mockBridgeProcessFailure,
mockBridgeResponse,
resetDesktopScriptProviderTestHarness,
sampleCapabilities
} from './desktop-script-provider-test-harness'
import type { BridgeResponse } from './desktop-script-provider-types'
import type { DesktopScriptRuntimeHost } from './desktop-script-runtime-host'
import { RuntimeClientError } from './runtime-client-error'
const POLICY_STDERR =
'File runtime.ps1 cannot be loaded because running scripts is disabled on this system. + CategoryInfo : SecurityError'
function fakeRuntimeHost(request: DesktopScriptRuntimeHost['request']) {
const dispose = vi.fn()
return { host: { request, dispose } as unknown as DesktopScriptRuntimeHost, dispose }
}
describe('desktop script provider runtime host routing', () => {
afterEach(resetDesktopScriptProviderTestHarness)
it('serves Windows operations from the runtime host without spawning a one-shot bridge', async () => {
const request = vi.fn(
async () => ({ ok: true, capabilities: sampleCapabilities() }) as BridgeResponse
)
const { host } = fakeRuntimeHost(request)
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host)
await expect(client.capabilities()).resolves.toMatchObject({ platform: 'linux' })
expect(request).toHaveBeenCalledWith({ tool: 'handshake' })
expectDesktopProviderSubprocessStartCount(0)
})
it('maps runtime host operation failures without falling back to the one-shot bridge', async () => {
const { host } = fakeRuntimeHost(
vi.fn(async () => ({ ok: false, error: 'appBlocked("1Password")' }) as BridgeResponse)
)
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host)
await expect(client.listApps()).rejects.toMatchObject({ code: 'app_blocked' })
expectDesktopProviderSubprocessStartCount(0)
})
it('degrades to the one-shot bridge for the operations a host cannot serve', async () => {
const request = vi.fn(async () => {
throw new RuntimeClientError('runtime_host_unavailable', 'could not start')
})
const { host, dispose } = fakeRuntimeHost(request as never)
mockBridgeResponse({ ok: true, apps: [{ name: 'Notepad', pid: 42 }] })
mockBridgeResponse({ ok: true, apps: [{ name: 'Notepad', pid: 42 }] })
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host)
await expect(client.listApps()).resolves.toMatchObject({ apps: [{ pid: 42 }] })
await client.listApps()
// The host is kept and asked again: it owns its own cooldown, so one bad
// spawn must not stand the session down to a powershell.exe per click.
expect(request).toHaveBeenCalledTimes(2)
expect(dispose).not.toHaveBeenCalled()
expectDesktopProviderSubprocessStartCount(2)
})
it('returns to the runtime host once it recovers', async () => {
let healthy = false
const request = vi.fn(async () => {
if (!healthy) {
throw new RuntimeClientError('runtime_host_unavailable', 'could not start')
}
return { ok: true, apps: [] } as BridgeResponse
})
const { host } = fakeRuntimeHost(request)
mockBridgeResponse({ ok: true, apps: [] })
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1', host)
await client.listApps()
expectDesktopProviderSubprocessStartCount(1)
healthy = true
await expect(client.listApps()).resolves.toEqual({ apps: [] })
expectDesktopProviderSubprocessStartCount(1)
})
it('runs the one-shot bridge under RemoteSigned and falls back to Bypass once', async () => {
mockBridgeProcessFailure(POLICY_STDERR)
mockBridgeResponse({ ok: true, apps: [] })
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1')
await expect(client.listApps()).resolves.toEqual({ apps: [] })
expectDesktopProviderSubprocessStartCount(2)
expect(bridgeProcessArgs(0)).toContain('-NoLogo')
expect(bridgeProcessArgs(0)).toContain('RemoteSigned')
expect(bridgeProcessArgs(0)).not.toContain('Bypass')
expect(bridgeProcessArgs(1)).toContain('Bypass')
})
it('does not retry the one-shot bridge for a non-policy failure', async () => {
mockBridgeProcessFailure('No top-level UI Automation window is available for Notepad')
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1')
await expect(client.listApps()).rejects.toMatchObject({ code: 'window_not_found' })
expectDesktopProviderSubprocessStartCount(1)
})
it('never replays an operation whose own output merely mentions a policy error', async () => {
// Window titles and element names are user-controlled text that lands in
// stdout; matching them would double a click, a keystroke or a paste.
mockBridgeProcessFailure({
stdout: JSON.stringify({
ok: true,
snapshot: { windowTitle: 'SecurityError - UnauthorizedAccess.log - Notepad' }
}),
stderr: ''
})
const client = await createDesktopScriptProviderClient('windows', 'C:\\runtime.ps1')
await expect(client.listApps()).rejects.toBeInstanceOf(Error)
expectDesktopProviderSubprocessStartCount(1)
})
it('keeps Linux on the one-shot python bridge with no execution policy flags', async () => {
mockBridgeResponse({ ok: true, apps: [] })
const client = await createDesktopScriptProviderClient('linux', '/tmp/runtime.py')
await expect(client.listApps()).resolves.toEqual({ apps: [] })
expect(bridgeProcessArgs(0)).toEqual(['/tmp/runtime.py', expect.any(String)])
})
})
@@ -1,4 +1,5 @@
import { expect, vi } from 'vitest'
import type { DesktopScriptRuntimeHost } from './desktop-script-runtime-host'
const { execFileMock, operationFiles, mkdtempMock, rmMock, writeFileMock } = vi.hoisted(() => {
const files = new Map<string, string>()
@@ -23,12 +24,14 @@ vi.mock('fs/promises', () => ({
writeFile: writeFileMock
}))
/** Builds a client on the one-shot bridge; pass a host to exercise serve mode. */
export async function createDesktopScriptProviderClient(
platform: 'linux' | 'windows',
executablePath: string
executablePath: string,
runtimeHost: DesktopScriptRuntimeHost | null = null
) {
const { DesktopScriptProviderClient } = await import('./desktop-script-provider-client')
return new DesktopScriptProviderClient(platform, executablePath)
return new DesktopScriptProviderClient(platform, executablePath, runtimeHost)
}
export function resetDesktopScriptProviderTestHarness(): void {
@@ -77,6 +80,19 @@ export function mockBridgeResponse(
})
}
export function mockBridgeProcessFailure(streams: string | { stdout?: string; stderr?: string }) {
const { stdout = '', stderr = '' } = typeof streams === 'string' ? { stderr: streams } : streams
execFileMock.mockImplementationOnce((_command, _args, _options, callback) => {
const done = callback as (error: Error | null, stdout: string, stderr: string) => void
done(new Error('Command failed'), stdout, stderr)
return null as never
})
}
export function bridgeProcessArgs(call: number): string[] {
return (execFileMock.mock.calls[call]?.[1] ?? []) as string[]
}
export function sampleBridgeSnapshot(name: string, value: string) {
return {
app: { name, bundleIdentifier: name, pid: 100 },
@@ -101,6 +101,8 @@ export type BridgeWindow = {
export type BridgeResponse = {
ok: boolean
/** Echo of BridgeRequest.requestId; set only on the persistent serve path. */
requestId?: number
error?: string
capabilities?: ComputerProviderCapabilities
apps?: {
@@ -122,6 +124,8 @@ export type BridgeResponse = {
export type BridgeRequest = {
tool: string
/** Correlates a serve-mode reply with its request; the one-shot path omits it. */
requestId?: number
app?: string
element?: BridgeElement
fromElement?: BridgeElement
@@ -0,0 +1,73 @@
import { RuntimeClientError } from './runtime-client-error'
/**
* Serializes operations onto one helper and bounds how long one may wait its
* turn.
*
* Why the wait needs its own deadline: the in-flight timeout is armed only once
* a request reaches a helper, so a request behind N timing-out ones waited N
* times that timeout with no deadline of its own — bounded, but the caller sees
* an `await` that looks hung for minutes and gets no error to act on.
*
* Why only the wait: a request that reaches a helper still gets its full
* execution budget. A single deadline covering both would fail operations that
* queued briefly and would otherwise have succeeded.
*/
export class DesktopScriptRequestQueue {
/**
* Never rejects: downstream turns chain onto it, and a rejection here would
* be delivered to whichever request happened to queue behind the failure.
*/
private tail: Promise<void> | null = null
constructor(
private readonly waitTimeoutMs: number,
/** Called when the queue empties, so the host can arm its idle shutdown. */
private readonly onDrained: () => void
) {}
enqueue<T>(run: () => Promise<T>): Promise<T> {
const queued = this.tail
if (!queued) {
return this.track(run())
}
let expiry: RuntimeClientError | null = null
let waitTimer: NodeJS.Timeout | undefined
const waited = new Promise<never>((_resolve, reject) => {
waitTimer = setTimeout(() => {
expiry = new RuntimeClientError(
'action_timeout',
`desktop provider timed out after ${this.waitTimeoutMs}ms waiting for earlier operations`
)
reject(expiry)
}, this.waitTimeoutMs)
waitTimer.unref?.()
})
// An abandoned request is never handed to a helper. The caller has already
// been told it failed, and a click delivered after that is worse than none.
const turn = (): Promise<T> => {
clearTimeout(waitTimer)
return expiry ? Promise.reject(expiry) : run()
}
// The tail chains on the turn, not on the race: a caller giving up early
// must not release the next request while this one's predecessor is still
// in flight.
return Promise.race([waited, this.track(queued.then(turn, turn))])
}
private track<T>(result: Promise<T>): Promise<T> {
const tail = result.then(
() => undefined,
() => undefined
)
this.tail = tail
void tail.finally(() => {
if (this.tail !== tail) {
return
}
this.tail = null
this.onDrained()
})
return result
}
}
@@ -0,0 +1,176 @@
import {
FALLBACK_WINDOWS_EXECUTION_POLICY,
PREFERRED_WINDOWS_EXECUTION_POLICY,
type WindowsExecutionPolicy
} from './windows-powershell-execution-policy'
/**
* Consecutive child failures before the helper is believed dead, and how long
* the one-shot bridge covers for it afterwards.
*
* Why not a latch: every plausible cause is transient — a Defender scan touching
* the script mid-launch, a locked CSC temp directory failing one `Add-Type`,
* momentary memory pressure. Giving up permanently silently restores the
* per-click process burst the host exists to remove, and computer use keeps
* working throughout, so nothing looks wrong while the MDE signature returns.
*/
export const MAX_START_ATTEMPTS = 3
export const START_FAILURE_COOLDOWN_MS = 60_000
/**
* Why not `Date.now`: an NTP correction, a VM snapshot restore or a user changing
* the clock steps the wall clock backwards, which extended the cooldown by the
* size of the step. Nothing shortens it from there — only `recordSuccess` clears
* it, and no request can reach a helper to succeed while it holds — so a one-hour
* step disabled the persistent helper for the life of the sidecar, silently
* restoring the per-click process burst. Elapsed monotonic time cannot go
* backwards.
*/
const monotonicNowMs = (): number => performance.now()
/**
* Whether the persistent helper is currently believed usable, and the execution
* policy it should be started under.
*
* Split from the host so the recovery rules are readable on their own: they are
* what stands between a transient bad spawn and a session that silently spends
* the rest of its life on one powershell.exe per click.
*/
export class RuntimeHostAvailability {
private policy: WindowsExecutionPolicy = PREFERRED_WINDOWS_EXECUTION_POLICY
private retryUnderFallbackPolicy = false
private consecutiveFailures = 0
private consecutiveSuccesses = 0
/** Null, not 0, for "no cooldown": `performance.now()` legitimately returns 0. */
private cooldownStartedAtMs: number | null = null
/**
* Set while the escalated policy has yet to start a helper, so a wrong
* diagnosis can be taken back.
*
* Why it can be wrong: AppLocker and WDAC constrained language mode raise
* PSSecurityException under the same SecurityError category a policy block
* uses, but they refuse the script at parse time, which `Bypass` cannot lift.
* Latching there would spend the session putting the most heavily weighted
* MDE token on every command line, on exactly the hardened hosts watching
* for it.
*/
private fallbackPolicyUnproven = false
constructor(
private readonly cooldownMs: number,
/** Public so the host can report its own start attempts to the same sink. */
readonly warn: (message: string) => void,
/** Overridden only by tests; the default must stay monotonic. */
private readonly now: () => number = monotonicNowMs
) {}
get executionPolicy(): WindowsExecutionPolicy {
return this.policy
}
get policyRetryPending(): boolean {
return this.retryUnderFallbackPolicy
}
get atPreferredPolicy(): boolean {
return this.policy === PREFERRED_WINDOWS_EXECUTION_POLICY
}
/** Milliseconds left before the host may try a helper again; 0 when it may. */
remainingCooldown(): number {
if (this.cooldownStartedAtMs === null) {
return 0
}
// Elapsed since the cooldown began, never a stored deadline: a deadline is
// only as trustworthy as the clock it was computed against.
return Math.max(0, Math.ceil(this.cooldownMs - (this.now() - this.cooldownStartedAtMs)))
}
requestPolicyRetry(): void {
this.retryUnderFallbackPolicy = true
}
escalateExecutionPolicy(): void {
this.retryUnderFallbackPolicy = false
this.policy = FALLBACK_WINDOWS_EXECUTION_POLICY
this.fallbackPolicyUnproven = true
// Sticky once proven: a genuinely Restricted machine would otherwise pay a
// guaranteed failed spawn per operation. Only a helper that produced no
// output at all can reach here, so a snapshot cannot talk the host into it.
this.warn(
`runtime host start blocked at ${PREFERRED_WINDOWS_EXECUTION_POLICY}; trying ${FALLBACK_WINDOWS_EXECUTION_POLICY}`
)
}
/** A helper started under the current policy, so the policy is the right one. */
confirmExecutionPolicy(): void {
this.fallbackPolicyUnproven = false
}
/**
* Undo an escalation the fallback never justified.
*
* The escalation is a diagnosis, and a fallback that cannot start a helper
* either disproves it: the policy was not what stopped the first attempt. Go
* back rather than latch, so a re-probe can escalate again later if the real
* cause clears. Re-probing costs one spawn per outage, which the failure
* count and its cooldown already bound, and never latching is the whole point
* of this class.
*/
abandonUnprovenFallback(): void {
if (!this.fallbackPolicyUnproven) {
return
}
this.fallbackPolicyUnproven = false
this.policy = PREFERRED_WINDOWS_EXECUTION_POLICY
this.warn(
`${FALLBACK_WINDOWS_EXECUTION_POLICY} did not start a helper either, so the execution policy was not the cause; returning to ${PREFERRED_WINDOWS_EXECUTION_POLICY}`
)
}
recordFailure(): void {
this.consecutiveSuccesses = 0
this.consecutiveFailures++
}
/** True once a helper has died often enough that respawning is just thrash. */
get exhausted(): boolean {
return this.consecutiveFailures >= MAX_START_ATTEMPTS
}
recordSuccess(): void {
this.consecutiveSuccesses++
this.fallbackPolicyUnproven = false
// Why a clean run and not a single reply: a helper that answers one
// operation and dies on the next would otherwise reset the count forever,
// and respawn once per operation — the exact burst the host removes.
if (this.consecutiveSuccesses >= MAX_START_ATTEMPTS) {
this.consecutiveFailures = 0
}
if (this.cooldownStartedAtMs === null) {
return
}
this.cooldownStartedAtMs = null
this.warn('runtime host recovered; operations are served by the persistent helper again')
}
enterCooldown(): void {
// An escalation that never started a helper must not outlive the outage it
// was guessed from; the next one re-diagnoses from the preferred policy.
this.abandonUnprovenFallback()
const failures = this.consecutiveFailures
this.cooldownStartedAtMs = this.now()
// The wait is the penalty; leaving the count at the limit would charge twice
// and let the first death after recovery re-enter a full cooldown, so an
// interleaved workload would spend its life on the one-shot bridge.
this.consecutiveFailures = 0
this.consecutiveSuccesses = 0
this.warn(
`runtime host unavailable after ${failures} consecutive failures; falling back to one powershell.exe per operation for ${this.cooldownMs}ms`
)
}
clearCooldown(): void {
this.cooldownStartedAtMs = null
}
}
@@ -0,0 +1,857 @@
import { EventEmitter } from 'node:events'
import { afterEach, describe, expect, it, vi } from 'vitest'
import type { ProcessSpec } from '../../shared/child-process/process-spec'
import type { RuntimeChildProcess } from './desktop-script-serve-channel'
import { DesktopScriptRuntimeHost, isRuntimeHostUnavailable } from './desktop-script-runtime-host'
const POLICY_ERROR =
'File runtime.ps1 cannot be loaded because running scripts\nis disabled on this system.\n + CategoryInfo : SecurityError'
class FakeRuntimeChild extends EventEmitter {
readonly stdout = new EventEmitter()
readonly stderr = new EventEmitter()
readonly writes: string[] = []
killed = false
stdinEnded = false
/** Holds write callbacks so a late stdin failure can be fired deliberately. */
deferWrites = false
private readonly pendingWrites: ((error?: Error | null) => void)[] = []
readonly stdin = {
write: (chunk: string, callback?: (error?: Error | null) => void): boolean => {
this.writes.push(chunk)
if (this.deferWrites) {
if (callback) {
this.pendingWrites.push(callback)
}
return true
}
callback?.(null)
return true
},
end: (): void => {
this.stdinEnded = true
},
on: (): void => {}
}
kill(): boolean {
this.killed = true
return true
}
/** What a destroyed stdin does to writes still queued at teardown. */
failQueuedWrites(): void {
for (const callback of this.pendingWrites.splice(0)) {
callback(new Error('ERR_STREAM_DESTROYED'))
}
}
/** Fail one queued write, leaving later ones outstanding. */
failQueuedWrite(index: number): void {
this.pendingWrites.splice(index, 1)[0](new Error('EPIPE'))
}
/** Requests written to this child, decoded. */
requests(): Record<string, unknown>[] {
return this.writes.map((line) => JSON.parse(line) as Record<string, unknown>)
}
/** The id the host is currently waiting on, so replies can echo it. */
pendingId(): number {
return this.requests().at(-1)?.requestId as number
}
/** The announcement the real serve loop writes before its first read. */
ready(): void {
this.write('{"ready":true}\n')
}
respond(response: Record<string, unknown>, requestId = this.pendingId()): void {
this.write(`${JSON.stringify({ ...response, requestId })}\n`)
}
write(raw: string): void {
this.stdout.emit('data', Buffer.from(raw, 'utf8'))
}
exit(code: number | null, stderr = ''): void {
if (stderr) {
this.stderr.emit('data', Buffer.from(stderr, 'utf8'))
}
this.emit('close', code, null)
}
}
function createHost(
options: {
idleShutdownMs?: number
requestTimeoutMs?: number
cooldownMs?: number
now?: () => number
deferWrites?: boolean
} = {}
) {
const children: FakeRuntimeChild[] = []
const specs: ProcessSpec[] = []
const warnings: string[] = []
const host = new DesktopScriptRuntimeHost('C:\\orca\\runtime.ps1', {
...options,
powerShellPath: () => 'C:\\Windows\\System32\\powershell.exe',
warn: (message) => warnings.push(message),
spawn: (spec) => {
specs.push(spec)
const child = new FakeRuntimeChild()
child.deferWrites = options.deferWrites === true
children.push(child)
return child as unknown as RuntimeChildProcess
}
})
return { host, children, specs, warnings }
}
/** Let the host's queue microtasks drain so the next request reaches its child. */
async function settle(): Promise<void> {
for (let index = 0; index < 6; index++) {
await Promise.resolve()
}
}
/** The wait the host reported, read back out of its refusal message. */
function remainingCooldownMs(error: Error | null): number {
const match = /retrying the runtime host in (\d+)ms/.exec(error?.message ?? '')
return match ? Number(match[1]) : Number.NaN
}
/** Kill each helper the host starts, until it stops starting them. */
async function failEveryStart(children: FakeRuntimeChild[], stderr: string): Promise<void> {
for (let index = 0; index < 8; index++) {
if (index >= children.length) {
return
}
children[index].exit(1, stderr)
await settle()
}
}
describe('DesktopScriptRuntimeHost', () => {
afterEach(() => {
vi.useRealTimers()
vi.restoreAllMocks()
})
it('starts one helper for many operations and never writes an operation file', async () => {
const { host, children, specs } = createHost()
const first = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await expect(first).resolves.toMatchObject({ ok: true })
for (let index = 0; index < 5; index++) {
const next = host.request({ tool: 'click', app: 'Notepad' })
await settle()
children[0].respond({ ok: true, action: { path: 'synthetic' } })
await expect(next).resolves.toMatchObject({ ok: true })
}
expect(children).toHaveLength(1)
expect(children[0].requests()).toHaveLength(6)
expect(specs[0].args).toEqual([
'-NoLogo',
'-NoProfile',
'-NonInteractive',
'-ExecutionPolicy',
'RemoteSigned',
'-File',
'C:\\orca\\runtime.ps1',
'-Serve'
])
host.dispose()
})
it('serializes requests so only one operation is ever in flight', async () => {
const { host, children } = createHost()
const first = host.request({ tool: 'click', app: 'A' })
const second = host.request({ tool: 'click', app: 'B' })
await settle()
expect(children[0].requests()).toEqual([{ tool: 'click', app: 'A', requestId: 1 }])
children[0].respond({ ok: true, action: { path: 'synthetic' } })
await expect(first).resolves.toMatchObject({ ok: true })
await settle()
expect(children[0].requests()).toHaveLength(2)
children[0].respond({ ok: true, action: { path: 'accessibility' } })
await expect(second).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('strips the echoed id from the response it hands back', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await expect(promise).resolves.toEqual({ ok: true, capabilities: {} })
host.dispose()
})
it('reassembles a response split across chunks, including a split code point', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'get_app_state', app: 'Editor' })
await settle()
const payload = Buffer.from(
`${JSON.stringify({ ok: true, snapshot: { app: 'né' }, requestId: 1 })}\r\n`,
'utf8'
)
const split = payload.indexOf(Buffer.from('é', 'utf8')) + 1
children[0].stdout.emit('data', payload.subarray(0, split))
children[0].stdout.emit('data', payload.subarray(split))
await expect(promise).resolves.toEqual({ ok: true, snapshot: { app: 'né' } })
host.dispose()
})
it('kills the helper rather than answering a request with another reply', async () => {
const { host, children } = createHost()
const first = host.request({ tool: 'handshake' })
await settle()
// A stray line would otherwise shift every later response by one.
children[0].respond({ ok: true, capabilities: {} }, 999)
await expect(first).rejects.toThrow(/did not match the pending request/)
expect(children[0].killed).toBe(true)
host.dispose()
})
it('kills the helper when an unsolicited line arrives with nothing pending', async () => {
const { host, children } = createHost()
const first = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await first
children[0].write(`${JSON.stringify({ ok: true, requestId: 77 })}\n`)
expect(children[0].killed).toBe(true)
host.dispose()
})
it('times out a wedged operation and starts a fresh helper for the next one', async () => {
vi.useFakeTimers()
const { host, children } = createHost({ requestTimeoutMs: 30_000 })
const promise = host.request({ tool: 'click', app: 'Frozen' })
await settle()
await vi.advanceTimersByTimeAsync(30_001)
await expect(promise).rejects.toMatchObject({ code: 'action_timeout' })
expect(children[0].killed).toBe(true)
const next = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(2)
children[1].respond({ ok: true, capabilities: {} })
await expect(next).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('rejects the in-flight request when a working helper crashes, then restarts', async () => {
const { host, children } = createHost()
const first = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await first
const second = host.request({ tool: 'click', app: 'Notepad' })
await settle()
children[0].exit(1, 'boom')
await expect(second).rejects.toMatchObject({ code: 'accessibility_error' })
await expect(second).rejects.toThrow(/runtime host exited/)
const third = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(2)
children[1].respond({ ok: true, capabilities: {} })
await expect(third).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('stops respawning a helper that dies on every second operation', async () => {
let clock = 1_000
const { host, children } = createHost({ cooldownMs: 60_000, now: () => clock })
// One good answer per helper is exactly the pattern that used to respawn
// forever: the success reset the failure count before it could ever trip.
for (let round = 0; round < 3; round++) {
const good = host.request({ tool: 'handshake' })
await settle()
children.at(-1)?.respond({ ok: true, capabilities: {} })
await expect(good).resolves.toMatchObject({ ok: true })
await settle()
const crash = host.request({ tool: 'click', app: 'Crashy' })
await settle()
children.at(-1)?.exit(1, 'boom')
await expect(crash).rejects.toThrow(/runtime host exited/)
await settle()
}
const spawned = children.length
await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable)
expect(children).toHaveLength(spawned)
host.dispose()
})
it('keeps serving a healthy helper after an isolated crash', async () => {
const { host, children } = createHost({ cooldownMs: 60_000 })
const crashed = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await crashed
const second = host.request({ tool: 'click', app: 'Notepad' })
await settle()
children[0].exit(1, 'boom')
await expect(second).rejects.toThrow(/runtime host exited/)
for (let index = 0; index < 4; index++) {
const next = host.request({ tool: 'handshake' })
await settle()
children.at(-1)?.respond({ ok: true, capabilities: {} })
await expect(next).resolves.toMatchObject({ ok: true })
}
// A clean run clears the count, so one bad helper cannot degrade a good one.
expect(children).toHaveLength(2)
host.dispose()
})
it('stops respawning a helper that keeps answering the wrong request', async () => {
let clock = 1_000
const { host, children } = createHost({ cooldownMs: 60_000, now: () => clock })
// Desync is host-detected, so it bypassed the exit handler entirely: without
// its own accounting this respawned once per operation, forever.
for (let round = 0; round < 3; round++) {
const promise = host.request({ tool: 'handshake' })
await settle()
const child = children.at(-1)
child?.respond({ ok: true, capabilities: {} }, child.pendingId() + 500)
await expect(promise).rejects.toThrow(/did not match the pending request/)
await settle()
}
const spawned = children.length
await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable)
expect(children).toHaveLength(spawned)
host.dispose()
})
it('stops respawning a helper that times out on every operation', async () => {
vi.useFakeTimers()
let clock = 1_000
const { host, children } = createHost({
requestTimeoutMs: 1_000,
cooldownMs: 60_000,
now: () => clock
})
for (let round = 0; round < 3; round++) {
const promise = host.request({ tool: 'get_app_state', app: 'Frozen' })
await settle()
await vi.advanceTimersByTimeAsync(1_001)
await expect(promise).rejects.toMatchObject({ code: 'action_timeout' })
await settle()
}
const spawned = children.length
await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable)
expect(children).toHaveLength(spawned)
host.dispose()
})
it('never re-sends a mutation to a fresh helper after a pre-answer death', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'click', app: 'Notepad', x: 10, y: 10 })
await settle()
children[0].exit(1, 'Add-Type : Cannot access the temporary directory')
// The click may already have landed inside the helper that died; replaying
// it would click twice. An observation in the same position is retried.
await expect(promise).rejects.toSatisfy(isRuntimeHostUnavailable)
expect(children).toHaveLength(1)
host.dispose()
})
it('never replays a mutation once the helper announced it was reading', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'click', app: 'Notepad', x: 10, y: 10 })
await settle()
children[0].ready()
// Past the announcement the click may already have been synthesized: the
// snapshot that follows it is the fault-prone part, so a missing reply
// proves nothing about whether the input landed.
children[0].exit(1, 'faulting module gdiplus.dll')
await expect(promise).rejects.toThrow(/runtime host exited/)
expect(children).toHaveLength(1)
host.dispose()
})
it('replays a mutation only for a helper that died before announcing readiness', async () => {
const { host, children } = createHost()
const first = host.request({ tool: 'handshake' })
await settle()
children[0].ready()
children[0].respond({ ok: true, capabilities: {} })
await first
const crashed = host.request({ tool: 'click', app: 'Notepad', x: 1, y: 1 })
await settle()
children[0].exit(1, 'boom')
await expect(crashed).rejects.toThrow(/runtime host exited/)
const retried = host.request({ tool: 'click', app: 'Notepad', x: 1, y: 1 })
await settle()
// This helper never announced, so it cannot have read the click: replaying
// is a fact rather than a guess, and the caller never sees the stumble.
children[1].exit(1, 'Add-Type : Cannot access the temporary directory')
await settle()
expect(children).toHaveLength(3)
children[2].ready()
children[2].respond({ ok: true, action: { path: 'synthetic' } })
await expect(retried).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('does not treat the readiness announcement as an unmatched reply', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'handshake' })
await settle()
children[0].ready()
expect(children[0].killed).toBe(false)
children[0].respond({ ok: true, capabilities: {} })
await expect(promise).resolves.toEqual({ ok: true, capabilities: {} })
host.dispose()
})
it('charges one cooldown per outage, not one per later death', async () => {
let clock = 1_000
const { host, children } = createHost({ cooldownMs: 60_000, now: () => clock })
const failed = host.request({ tool: 'handshake' })
await settle()
await failEveryStart(children, 'The term is not recognized')
await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable)
clock += 61_000
const recovered = host.request({ tool: 'handshake' })
await settle()
children.at(-1)?.respond({ ok: true, capabilities: {} })
await recovered
// One death after recovery must not re-enter a full cooldown; the previous
// outage was already paid for.
const crashed = host.request({ tool: 'handshake' })
await settle()
children.at(-1)?.exit(1, 'boom')
await expect(crashed).rejects.toBeInstanceOf(Error)
const next = host.request({ tool: 'handshake' })
await settle()
children.at(-1)?.respond({ ok: true, capabilities: {} })
await expect(next).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('charges one failure when a write fails after the helper was torn down', async () => {
const { host, children, warnings } = createHost({ deferWrites: true })
const promise = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} }, 999)
await expect(promise).rejects.toThrow(/did not match the pending request/)
// stop() destroys stdin, so the queued write calls back with an error. That
// is the same operation failing, not a second one, and counting it twice
// would drive a 3-strike cooldown at half the intended rate.
children[0].failQueuedWrites()
expect(warnings.filter((line) => /helper stopped/.test(line))).toHaveLength(1)
host.dispose()
})
it('never lets a stale write error stop a replacement helper', async () => {
const { host, children } = createHost({ deferWrites: true })
const first = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} }, 999)
await expect(first).rejects.toBeInstanceOf(Error)
const second = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(2)
// The late callback belongs to a channel and a request that are both gone.
children[0].failQueuedWrites()
expect(children[1].killed).toBe(false)
children[1].respond({ ok: true, capabilities: {} })
await expect(second).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('ignores a write error for a request that already finished', async () => {
const { host, children } = createHost({ deferWrites: true })
const first = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await first
const second = host.request({ tool: 'handshake' })
await settle()
// Backpressure can hold a write callback past its own response. The channel
// is alive and was never stopped, so only the request id can tell that this
// report is stale — this is what pins the host-side guard on its own.
children[0].failQueuedWrite(0)
expect(children[0].killed).toBe(false)
children[0].respond({ ok: true, capabilities: {} })
await expect(second).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('shuts the helper down when idle and starts a new one on the next operation', async () => {
vi.useFakeTimers()
const { host, children } = createHost({ idleShutdownMs: 60_000 })
const first = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await first
await settle()
expect(children[0].killed).toBe(false)
await vi.advanceTimersByTimeAsync(60_001)
expect(children[0].stdinEnded).toBe(true)
expect(children[0].killed).toBe(true)
const next = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(2)
children[1].respond({ ok: true, capabilities: {} })
await expect(next).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('disposes the helper and rejects the in-flight request', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'click', app: 'Notepad' })
await settle()
host.dispose()
expect(children[0].stdinEnded).toBe(true)
expect(children[0].killed).toBe(true)
await expect(promise).rejects.toThrow(/shut down/)
})
it('never respawns for a request queued behind dispose', async () => {
const { host, children } = createHost()
const first = host.request({ tool: 'handshake' })
const queued = host.request({ tool: 'handshake' })
await settle()
host.dispose()
await expect(first).rejects.toBeInstanceOf(Error)
await expect(queued).rejects.toSatisfy(isRuntimeHostUnavailable)
await settle()
expect(children).toHaveLength(1)
})
it('falls back to Bypass once when the execution policy blocks the start', async () => {
const { host, children, specs, warnings } = createHost()
const promise = host.request({ tool: 'handshake' })
await settle()
children[0].exit(1, POLICY_ERROR)
await settle()
expect(children).toHaveLength(2)
expect(specs[1].args).toContain('Bypass')
children[1].respond({ ok: true, capabilities: {} })
await expect(promise).resolves.toMatchObject({ ok: true })
expect(warnings.some((line) => /trying Bypass/.test(line))).toBe(true)
// A helper started under Bypass, so the diagnosis is proven and the fallback
// is remembered for the session rather than re-probed per call.
const next = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(2)
children[1].respond({ ok: true, capabilities: {} })
await next
expect(warnings.some((line) => /returning to RemoteSigned/.test(line))).toBe(false)
host.dispose()
})
it('returns to RemoteSigned when Bypass does not start a helper either', async () => {
let clock = 1_000
const { host, children, specs, warnings } = createHost({ cooldownMs: 60_000, now: () => clock })
// What AppLocker and WDAC constrained language mode look like: the same
// SecurityError category, but the block is at script load, so Bypass cannot
// lift it and the escalation was a misdiagnosis.
const promise = host.request({ tool: 'handshake' })
await settle()
await failEveryStart(children, POLICY_ERROR)
await expect(promise).rejects.toSatisfy(isRuntimeHostUnavailable)
expect(specs[1].args).toContain('Bypass')
expect(warnings.some((line) => /returning to RemoteSigned/.test(line))).toBe(true)
// The revert lands inside the outage, not just at its end: every attempt
// after the fallback is disproved is back on the preferred policy, so the
// misdiagnosis costs one Bypass command line rather than one per attempt.
expect(specs).toHaveLength(3)
expect(specs[2].args).not.toContain('Bypass')
// Latching here would put the most heavily weighted MDE token on every
// later command line, on exactly the hardened host that is watching.
clock += 61_000
const recovered = host.request({ tool: 'handshake' })
await settle()
expect(specs.at(-1)?.args).not.toContain('Bypass')
children.at(-1)?.respond({ ok: true, capabilities: {} })
await expect(recovered).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('reports itself unavailable when Bypass is also refused', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'handshake' })
await settle()
await failEveryStart(children, POLICY_ERROR)
await expect(promise).rejects.toSatisfy(isRuntimeHostUnavailable)
host.dispose()
})
it('reports itself unavailable when the helper cannot be spawned at all', async () => {
const host = new DesktopScriptRuntimeHost('C:\\orca\\runtime.ps1', {
powerShellPath: () => 'C:\\Windows\\System32\\powershell.exe',
warn: () => {},
spawn: () => {
throw new Error('spawn ENOENT')
}
})
await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable)
host.dispose()
})
it('retries a transient pre-answer death without the caller ever seeing it', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'handshake' })
await settle()
children[0].exit(1, 'Add-Type : Cannot access the temporary directory')
await settle()
expect(children).toHaveLength(2)
children[1].respond({ ok: true, capabilities: {} })
await expect(promise).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('gives up only after repeated start failures, then serves from the host again after the cooldown', async () => {
let clock = 1_000
const { host, children, warnings } = createHost({ cooldownMs: 60_000, now: () => clock })
const failed = host.request({ tool: 'handshake' })
await settle()
await failEveryStart(children, 'The term is not recognized')
await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable)
const attempts = children.length
expect(attempts).toBe(3)
// Inside the cooldown the host stays out of the way without respawning.
clock += 30_000
await expect(host.request({ tool: 'handshake' })).rejects.toSatisfy(isRuntimeHostUnavailable)
expect(children).toHaveLength(attempts)
// Past it, the next operation re-probes rather than staying degraded forever.
clock += 31_000
const recovered = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(attempts + 1)
children[attempts].respond({ ok: true, capabilities: {} })
await expect(recovered).resolves.toMatchObject({ ok: true })
expect(warnings.at(-1)).toMatch(/recovered/)
host.dispose()
})
it('keeps the helper account of a reply it could not tag', async () => {
const { host, children } = createHost()
const promise = host.request({ tool: 'handshake' })
await settle()
const child = children[0]
// What an old runtime.ps1 sends when a request will not parse: a real error,
// with no id to route it by. The desync is honest, but replacing its message
// reports a broken stream and loses the only account of the cause.
child.respond({ ok: false, error: 'Invalid object passed in' }, child.pendingId() + 500)
await expect(promise).rejects.toThrow(
/did not match the pending request: Invalid object passed in/
)
host.dispose()
})
it('does not charge a cooldown for requests the helper rejects as malformed', async () => {
const { host, children } = createHost({ cooldownMs: 60_000 })
// A tagged error is the helper working, not failing. Three of them used to
// arrive untagged, and three desync aborts is exactly the cooldown.
for (let round = 0; round < 3; round++) {
const promise = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: false, error: 'Invalid object passed in' })
await expect(promise).resolves.toMatchObject({ ok: false })
await settle()
}
expect(children).toHaveLength(1)
const next = host.request({ tool: 'handshake' })
await settle()
children[0].respond({ ok: true, capabilities: {} })
await expect(next).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('fails a request that spends its whole timeout queued behind others', async () => {
vi.useFakeTimers()
const { host, children } = createHost({ requestTimeoutMs: 1_000 })
// Two ahead of it, because one puts the turn exactly on the deadline.
const first = host.request({ tool: 'get_app_state', app: 'Frozen' })
const second = host.request({ tool: 'get_app_state', app: 'Frozen' })
const queued = host.request({ tool: 'click', app: 'Notepad' })
// Asserted before the clock moves: both reject while the test is still
// inside advanceTimersByTimeAsync.
const firstFailed = expect(first).rejects.toMatchObject({ code: 'action_timeout' })
// Its own deadline, not the one it would inherit by reaching the head.
const queuedFailed = expect(queued).rejects.toMatchObject({
code: 'action_timeout',
message: /waiting for earlier operations/
})
await settle()
expect(children[0].requests()).toHaveLength(1)
await vi.advanceTimersByTimeAsync(1_001)
await firstFailed
await queuedFailed
// Drain past the abandoned request: it is never handed to a helper, because
// a click the caller has been told failed must not still land.
children[1].respond({ ok: true, state: {} })
await expect(second).resolves.toMatchObject({ ok: true })
await settle()
expect(children.flatMap((child) => child.requests())).not.toContainEqual(
expect.objectContaining({ tool: 'click' })
)
// The request that gave up does not poison the queue behind it.
const next = host.request({ tool: 'handshake' })
await settle()
children[1].respond({ ok: true, capabilities: {} })
await expect(next).resolves.toMatchObject({ ok: true })
host.dispose()
})
it('gives a queued request its full timeout once it reaches the helper', async () => {
vi.useFakeTimers()
const { host, children } = createHost({ requestTimeoutMs: 1_000 })
const head = host.request({ tool: 'handshake' })
const queued = host.request({ tool: 'get_app_state', app: 'Slow' })
await settle()
await vi.advanceTimersByTimeAsync(900)
children[0].respond({ ok: true, capabilities: {} })
await expect(head).resolves.toMatchObject({ ok: true })
await settle()
// Past the point the enqueue deadline would have fired: waiting its turn
// must not eat the budget the operation itself is entitled to.
await vi.advanceTimersByTimeAsync(900)
children[0].respond({ ok: true, state: {} })
await expect(queued).resolves.toMatchObject({ ok: true })
host.dispose()
})
// Both of these deliberately leave `now` unset: the bug was in the default the
// host picks, so a test that injects a clock cannot see it.
it('does not stretch the cooldown when the wall clock steps backwards', async () => {
const wallClock = vi.spyOn(Date, 'now').mockReturnValue(2_000_000_000_000)
const { host, children } = createHost({ cooldownMs: 60_000 })
const failed = host.request({ tool: 'handshake' })
await settle()
await failEveryStart(children, 'The term is not recognized')
await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable)
// An NTP correction, a VM snapshot restore, a user changing the clock.
wallClock.mockReturnValue(2_000_000_000_000 - 3_600_000)
const refused = await host.request({ tool: 'handshake' }).then(
() => null,
(error: Error) => error
)
expect(refused?.message).toMatch(/retrying the runtime host in/)
expect(remainingCooldownMs(refused)).toBeLessThanOrEqual(60_000)
host.dispose()
})
it('serves from the persistent helper again after a backwards clock step', async () => {
vi.spyOn(Date, 'now').mockReturnValue(2_000_000_000_000)
const { host, children } = createHost({ cooldownMs: 25 })
const failed = host.request({ tool: 'handshake' })
await settle()
await failEveryStart(children, 'The term is not recognized')
await expect(failed).rejects.toSatisfy(isRuntimeHostUnavailable)
const attempts = children.length
vi.mocked(Date.now).mockReturnValue(2_000_000_000_000 - 3_600_000)
// Real elapsed time, because the clock under test is the real monotonic one.
await new Promise((resolve) => setTimeout(resolve, 60))
const recovered = host.request({ tool: 'handshake' })
await settle()
expect(children).toHaveLength(attempts + 1)
children[attempts].respond({ ok: true, capabilities: {} })
await expect(recovered).resolves.toMatchObject({ ok: true })
host.dispose()
})
})
@@ -0,0 +1,384 @@
import { spawnProcess } from '../../shared/child-process/run-process'
import { windowsPowerShellPath } from '../../shared/child-process/windows-system-binary'
import { reportComputerDiagnostic } from './computer-sidecar-diagnostics'
import { isReplayableTool } from './desktop-script-action'
import type { BridgeRequest, BridgeResponse } from './desktop-script-provider-types'
import { DesktopScriptRequestQueue } from './desktop-script-request-queue'
import {
startServeChannel,
type DesktopScriptServeChannel,
type RuntimeProcessSpawn
} from './desktop-script-serve-channel'
import {
MAX_START_ATTEMPTS,
RuntimeHostAvailability,
START_FAILURE_COOLDOWN_MS
} from './desktop-script-runtime-availability'
import { RuntimeClientError } from './runtime-client-error'
import {
isExecutionPolicyBlocked,
windowsPowerShellRuntimeArgs
} from './windows-powershell-execution-policy'
const REQUEST_TIMEOUT_MS = 30_000
const IDLE_SHUTDOWN_MS = 120_000
/** Code the client keys on to serve this one operation from the one-shot bridge. */
export const RUNTIME_HOST_UNAVAILABLE = 'runtime_host_unavailable'
export type DesktopScriptRuntimeHostOptions = {
spawn?: RuntimeProcessSpawn
powerShellPath?: () => string
requestTimeoutMs?: number
idleShutdownMs?: number
cooldownMs?: number
now?: () => number
warn?: (message: string) => void
}
type PendingRequest = {
id: number
resolve: (response: BridgeResponse) => void
reject: (error: Error) => void
timer: NodeJS.Timeout
}
export function isRuntimeHostUnavailable(error: unknown): boolean {
return error instanceof RuntimeClientError && error.code === RUNTIME_HOST_UNAVAILABLE
}
/**
* One long-lived `runtime.ps1 -Serve` process serving every computer-use
* operation over NDJSON on stdin/stdout.
*
* Why persistent: the one-shot bridge started a powershell.exe per click, and
* each one re-emitted the script's inline `Add-Type` P/Invoke assembly, which
* Defender for Endpoint reports as suspicious MSIL emission alongside the
* screen capture. Compiling once per session collapses a burst of short-lived
* PIDs into a single process.
*
* Requests are strictly serialized, and each carries an id the helper echoes.
* Serialization alone would leave a single stray line answering every later
* request with the previous response — silently acting on stale element
* indexes, with no error raised — so the id is checked and a mismatch is fatal
* to the child rather than merely logged.
*/
export class DesktopScriptRuntimeHost {
private channel: DesktopScriptServeChannel | null = null
private pending: PendingRequest | null = null
private idleTimer: NodeJS.Timeout | null = null
private childReady = false
private childAnswered = false
/**
* Set once any helper has announced itself, which proves the script on disk
* speaks the ready protocol. Until then a mutating request is not replayed
* even on a clean start failure, because ORCA_COMPUTER_DESKTOP_SCRIPT_PROVIDER_PATH
* can point at an older runtime.ps1 that simply never announces.
*/
private readyProtocolConfirmed = false
private disposed = false
private nextRequestId = 1
private readonly availability: RuntimeHostAvailability
private readonly queue: DesktopScriptRequestQueue
private readonly requestTimeoutMs: number
private readonly idleShutdownMs: number
constructor(
private readonly scriptPath: string,
private readonly options: DesktopScriptRuntimeHostOptions = {}
) {
this.requestTimeoutMs = options.requestTimeoutMs ?? REQUEST_TIMEOUT_MS
this.idleShutdownMs = options.idleShutdownMs ?? IDLE_SHUTDOWN_MS
this.queue = new DesktopScriptRequestQueue(this.requestTimeoutMs, () => this.armIdleTimer())
this.availability = new RuntimeHostAvailability(
options.cooldownMs ?? START_FAILURE_COOLDOWN_MS,
(message) => (options.warn ?? reportComputerDiagnostic)(message),
options.now
)
}
request(request: BridgeRequest): Promise<BridgeResponse> {
return this.queue.enqueue(() => this.send(request))
}
/** Permanently stop this host. Callers build a new one for a new session. */
dispose(): void {
this.disposed = true
this.clearIdleTimer()
this.availability.clearCooldown()
this.stopChannel()
this.rejectPending(
new RuntimeClientError('accessibility_error', 'desktop provider runtime host was shut down')
)
}
private async send(request: BridgeRequest): Promise<BridgeResponse> {
this.clearIdleTimer()
// Why checked here and not only on entry: requests queue, and dispose can
// land while one waits its turn. Without this a teardown respawns a helper.
if (this.disposed) {
throw this.unavailableError('runtime host was disposed')
}
const cooldown = this.availability.remainingCooldown()
if (cooldown > 0) {
throw this.unavailableError(`retrying the runtime host in ${cooldown}ms`)
}
let lastError: unknown
for (let attempt = 1; attempt <= MAX_START_ATTEMPTS; attempt++) {
try {
const response = await this.sendOnce(request)
this.availability.recordSuccess()
return response
} catch (error) {
lastError = error
if (this.availability.policyRetryPending) {
this.availability.escalateExecutionPolicy()
continue
}
// Only this error proves no helper started, which is what disproves the
// escalation; a helper that started and then died proves the opposite.
if (isRuntimeHostUnavailable(error)) {
this.availability.abandonUnprovenFallback()
}
// A helper that answered and then died is a crash, not a bad start: the
// caller sees it and the next operation gets a fresh process — unless it
// keeps happening, which is thrash the one-shot bridge should absorb.
if (!isRuntimeHostUnavailable(error) || !this.mayReplay(request)) {
if (this.availability.exhausted) {
this.availability.enterCooldown()
}
throw error
}
this.availability.warn(
`runtime host failed to start (attempt ${attempt}/${MAX_START_ATTEMPTS}): ${errorText(error)}`
)
}
}
this.availability.enterCooldown()
throw lastError
}
private sendOnce(request: BridgeRequest): Promise<BridgeResponse> {
let channel: DesktopScriptServeChannel
try {
channel = this.ensureChannel()
} catch (error) {
this.availability.recordFailure()
return Promise.reject(this.unavailableError(errorText(error)))
}
const id = this.nextRequestId++
return new Promise((resolve, reject) => {
// Why kill rather than wait: a hung UI Automation call cannot be
// cancelled, so the process itself is the only thing left to reclaim.
const timer = setTimeout(() => {
this.abortChannel(
new RuntimeClientError(
'action_timeout',
`desktop provider timed out after ${this.requestTimeoutMs}ms`
)
)
}, this.requestTimeoutMs)
timer.unref?.()
this.pending = { id, resolve, reject, timer }
channel.write(`${JSON.stringify({ ...request, requestId: id })}\n`, (error) => {
// Bind the report to what it was written for: a late callback must not
// charge a second failure for this operation, nor stop a replacement
// helper and reject a later request with this one's error. Deliberately
// redundant with the channel's own closed guard — keep both. This one
// also covers a live channel whose request has already been answered,
// which the channel cannot see; that case is what pins it.
//
// Redundant does not mean untested: removing either guard alone fails a
// test, so neither can be deleted as "the one the other covers".
if (this.channel !== channel || this.pending?.id !== id) {
return
}
this.abortChannel(new RuntimeClientError('accessibility_error', error.message))
})
})
}
private ensureChannel(): DesktopScriptServeChannel {
if (this.channel) {
return this.channel
}
this.childReady = false
this.childAnswered = false
const channel: DesktopScriptServeChannel = startServeChannel(
{
program: (this.options.powerShellPath ?? windowsPowerShellPath)(),
args: windowsPowerShellRuntimeArgs(this.scriptPath, this.availability.executionPolicy, [
'-Serve'
]),
env: process.env
},
this.options.spawn ?? spawnProcess,
{
onLine: (line) => this.deliver(line),
// A replaced channel can still report; that must not fail the live one.
onGone: (detail) => {
if (this.channel === channel) {
this.handleGone(detail)
}
},
onOverflow: () =>
this.abortChannel(
new RuntimeClientError(
'accessibility_error',
'desktop provider response exceeded the runtime host buffer'
)
)
}
)
this.channel = channel
return channel
}
/**
* Whether the helper that just died can be proved not to have run the request.
*
* Why proof and not inference: "no reply came back" is not "nothing happened".
* runtime.ps1 synthesizes the input and only then builds the snapshot, which
* allocates a full-window bitmap and walks the UIA tree — a native fault there
* is uncatchable and would leave a click already delivered. Retrying on that
* inference turns one requested click into four.
*/
private mayReplay(request: BridgeRequest): boolean {
if (this.childReady || this.childAnswered) {
return false
}
return this.readyProtocolConfirmed || isReplayableTool(request.tool)
}
private deliver(line: string): void {
let parsed: Record<string, unknown>
try {
parsed = JSON.parse(line) as Record<string, unknown>
} catch {
// Not a response at all — a PowerShell banner, a stray write. Dropping it
// is safe now that the id below is what decides which request is answered,
// and it keeps a chatty console from making the helper unusable.
return
}
// The readiness announcement carries no request id and answers nothing.
if (parsed.ready === true && parsed.requestId === undefined) {
this.childReady = true
this.readyProtocolConfirmed = true
this.availability.confirmExecutionPolicy()
return
}
const pending = this.pending
if (!pending || parsed.requestId !== pending.id) {
// One unmatched reply would otherwise shift every later response by one.
// Carry the helper's own message when it sent one: a line it could not tag
// with an id is usually the only account of what went wrong, and reporting
// a bare desync in its place loses the cause for good.
const reported = typeof parsed.error === 'string' ? `: ${parsed.error}` : ''
this.abortChannel(
new RuntimeClientError(
'accessibility_error',
`desktop provider response did not match the pending request${reported}`
)
)
return
}
// Only a reply this host can prove is its own counts as the helper working.
this.childAnswered = true
this.pending = null
clearTimeout(pending.timer)
const { requestId: _echoed, ...response } = parsed
pending.resolve(response as BridgeResponse)
}
private handleGone(detail: string): void {
const started = this.childReady || this.childAnswered
this.channel = null
this.availability.recordFailure()
if (!started && this.availability.atPreferredPolicy && isExecutionPolicyBlocked(detail)) {
this.availability.requestPolicyRetry()
// Unavailable rather than a generic error, because this can now be the
// final attempt: reverting an unproven escalation puts the host back on
// the preferred policy, so a later attempt can land here again. Only this
// code routes the operation to the one-shot bridge, which carries its own
// policy fallback; anything else fails the operation outright.
this.rejectPending(this.unavailableError(detail))
return
}
if (!started) {
this.rejectPending(this.unavailableError(detail))
return
}
this.rejectPending(
new RuntimeClientError(
'accessibility_error',
`desktop provider runtime host exited: ${detail}`
)
)
}
/**
* Stop a helper this host has judged unusable — a timeout, a desynchronised
* reply, an oversized line.
*
* Why it counts as a failure: stopping the channel suppresses the exit
* handler, so without this these paths bypassed the accounting entirely and a
* helper that failed this way on every operation was respawned once per
* operation forever — the burst this host exists to remove, restored through
* its own recovery path.
*/
private abortChannel(error: Error): void {
this.stopChannel()
this.availability.recordFailure()
this.availability.warn(`runtime host helper stopped: ${error.message}`)
this.rejectPending(error)
}
private stopChannel(): void {
const channel = this.channel
this.channel = null
channel?.stop()
}
private takePending(): PendingRequest | null {
const pending = this.pending
this.pending = null
if (pending) {
clearTimeout(pending.timer)
}
return pending
}
private rejectPending(error: Error): void {
this.takePending()?.reject(error)
}
private armIdleTimer(): void {
this.clearIdleTimer()
if (!this.channel) {
return
}
this.idleTimer = setTimeout(() => {
this.idleTimer = null
this.stopChannel()
}, this.idleShutdownMs)
this.idleTimer.unref?.()
}
private clearIdleTimer(): void {
if (this.idleTimer) {
clearTimeout(this.idleTimer)
this.idleTimer = null
}
}
private unavailableError(message: string): RuntimeClientError {
return new RuntimeClientError(
RUNTIME_HOST_UNAVAILABLE,
`desktop provider runtime host could not start: ${message}`
)
}
}
function errorText(error: unknown): string {
return error instanceof Error ? error.message : String(error)
}
@@ -0,0 +1,130 @@
import { resolve } from 'node:path'
import { afterEach, describe, expect, it } from 'vitest'
import { spawnProcess } from '../../shared/child-process/run-process'
import { windowsPowerShellPath } from '../../shared/child-process/windows-system-binary'
import { DesktopScriptRuntimeHost } from './desktop-script-runtime-host'
import { startServeChannel } from './desktop-script-serve-channel'
import {
PREFERRED_WINDOWS_EXECUTION_POLICY,
windowsPowerShellRuntimeArgs
} from './windows-powershell-execution-policy'
/**
* The other half of the serve-mode proof: the unit test drives a fake child,
* this one drives the real `runtime.ps1 -Serve` on a real Windows box.
*
* Both are needed. The framing that matters — one NDJSON line per response,
* megabyte-scale screenshot payloads, a console writer that actually flushes —
* only exists in PowerShell, and a fake child cannot disprove any of it.
*
* Runs only on win32; skipped elsewhere.
*/
const describeOnWindows = process.platform === 'win32' ? describe : describe.skip
const SCRIPT_PATH = resolve(__dirname, '../../../native/computer-use-windows/runtime.ps1')
describeOnWindows('runtime.ps1 serve mode', () => {
let host: DesktopScriptRuntimeHost | null = null
let spawns = 0
function startHost(): DesktopScriptRuntimeHost {
spawns = 0
host = new DesktopScriptRuntimeHost(SCRIPT_PATH, {
warn: () => {},
spawn: (spec) => {
spawns++
return spawnProcess(spec)
}
})
return host
}
afterEach(() => {
host?.dispose()
host = null
})
it('answers repeated operations from a single PowerShell process', async () => {
const runtime = startHost()
await expect(runtime.request({ tool: 'handshake' })).resolves.toMatchObject({
ok: true,
capabilities: { protocolVersion: 1, provider: 'orca-computer-use-windows' }
})
const apps = await runtime.request({ tool: 'list_apps' })
expect(apps.ok).toBe(true)
expect(Array.isArray(apps.apps)).toBe(true)
await expect(runtime.request({ tool: 'handshake' })).resolves.toMatchObject({ ok: true })
expect(spawns).toBe(1)
})
it('returns a structured error for a bad request without killing the helper', async () => {
const runtime = startHost()
await expect(runtime.request({ tool: 'not_a_tool' })).resolves.toMatchObject({ ok: false })
await expect(runtime.request({ tool: 'handshake' })).resolves.toMatchObject({ ok: true })
expect(spawns).toBe(1)
})
/**
* The host can only write well-formed JSON, so the parse-failure branch of the
* serve loop is unreachable through it. Driving the channel directly is the
* only way to prove what the real PowerShell answers.
*/
it('echoes the id it can recover when a request will not parse', async () => {
const answer = await answerRawLine('{"tool":"handshake","requestId":7')
// Tagged, so the host resolves the waiting request with a failed operation
// instead of reading an untagged line as a desynchronised stream.
expect(answer).toMatchObject({ ok: false, requestId: 7 })
expect(String(answer.error)).not.toBe('')
})
it('reports an error for a line with no recoverable id', async () => {
const answer = await answerRawLine('{"tool":"handshake"')
expect(answer).toMatchObject({ ok: false })
expect(answer.requestId).toBeUndefined()
expect(String(answer.error)).not.toBe('')
})
})
/** One raw line into a real `runtime.ps1 -Serve`, and the line it writes back. */
function answerRawLine(raw: string): Promise<Record<string, unknown>> {
return new Promise((settle, fail) => {
const channel = startServeChannel(
{
program: windowsPowerShellPath(),
args: windowsPowerShellRuntimeArgs(SCRIPT_PATH, PREFERRED_WINDOWS_EXECUTION_POLICY, [
'-Serve'
]),
env: process.env
},
spawnProcess,
{
onLine: (line) => {
let parsed: Record<string, unknown>
try {
parsed = JSON.parse(line) as Record<string, unknown>
} catch {
return
}
if (parsed.ready === true) {
channel.write(`${raw}\n`, fail)
return
}
channel.stop()
settle(parsed)
},
onGone: (detail) => fail(new Error(`helper exited before answering: ${detail}`)),
onOverflow: () => {
channel.stop()
fail(new Error('helper overflowed the response buffer'))
}
}
)
})
}
@@ -0,0 +1,99 @@
import { EventEmitter } from 'node:events'
import { describe, expect, it, vi } from 'vitest'
import { DesktopScriptServeChannel, type RuntimeChildProcess } from './desktop-script-serve-channel'
class FakeChild extends EventEmitter {
readonly stdout = new EventEmitter()
readonly stderr = new EventEmitter()
readonly writes: string[] = []
killed = false
private readonly pendingWrites: ((error?: Error | null) => void)[] = []
readonly stdin = {
write: (chunk: string, callback?: (error?: Error | null) => void): boolean => {
this.writes.push(chunk)
if (callback) {
this.pendingWrites.push(callback)
}
return true
},
end: (): void => {},
on: (): void => {}
}
kill(): boolean {
this.killed = true
return true
}
/** What a destroyed stdin does to writes still queued at teardown. */
failQueuedWrites(): void {
for (const callback of this.pendingWrites.splice(0)) {
callback(new Error('ERR_STREAM_DESTROYED'))
}
}
}
function createChannel() {
const child = new FakeChild()
const handlers = { onLine: vi.fn(), onGone: vi.fn(), onOverflow: vi.fn() }
const channel = new DesktopScriptServeChannel(child as unknown as RuntimeChildProcess, handlers)
return { channel, child, handlers }
}
describe('DesktopScriptServeChannel', () => {
it('splits responses into lines and tolerates a trailing carriage return', () => {
const { child, handlers } = createChannel()
child.stdout.emit('data', Buffer.from('{"a":1}\r\n{"b":2}\n', 'utf8'))
expect(handlers.onLine.mock.calls.map(([line]) => line)).toEqual(['{"a":1}', '{"b":2}'])
})
it('reports the exit reason with the stderr tail', () => {
const { child, handlers } = createChannel()
child.stderr.emit('data', Buffer.from('it broke', 'utf8'))
child.emit('close', 1, null)
expect(handlers.onGone).toHaveBeenCalledWith('code 1: it broke')
})
describe('once stopped', () => {
/**
* The channel's half of the stale-callback guard, pinned here rather than
* through the host: the host refuses a stale report too, so a host-level
* test passes with either guard alone and neither ends up covered.
*/
it('accepts no further writes', () => {
const { channel, child } = createChannel()
channel.stop()
channel.write('{"tool":"click"}\n', vi.fn())
expect(child.writes).toEqual([])
})
it('reports no error from a write that was already queued', () => {
const { channel, child } = createChannel()
const onError = vi.fn()
channel.write('{"tool":"click"}\n', onError)
channel.stop()
child.failQueuedWrites()
expect(onError).not.toHaveBeenCalled()
})
it('reports neither lines nor the exit it was asked to cause', () => {
const { channel, child, handlers } = createChannel()
channel.stop()
child.stdout.emit('data', Buffer.from('{"a":1}\n', 'utf8'))
child.emit('close', 0, null)
expect(handlers.onLine).not.toHaveBeenCalled()
expect(handlers.onGone).not.toHaveBeenCalled()
})
})
})
@@ -0,0 +1,145 @@
import { StringDecoder } from 'node:string_decoder'
import type { ProcessSpec } from '../../shared/child-process/process-spec'
import type { spawnProcess } from '../../shared/child-process/run-process'
/** The all-pipes child `spawnProcess` returns; avoids a node:child_process import. */
export type RuntimeChildProcess = ReturnType<typeof spawnProcess>
export type RuntimeProcessSpawn = (spec: ProcessSpec) => RuntimeChildProcess
/** UTF-16 units, not bytes — this bounds the buffer, it is not a payload contract. */
const MAX_RESPONSE_CHARS = 20 * 1024 * 1024
const MAX_STDERR_CHARS = 4096
export type ServeChannelHandlers = {
/** One complete line from the helper, without its terminator. */
onLine: (line: string) => void
/** The helper is gone; detail carries the exit reason and its stderr tail. */
onGone: (detail: string) => void
/** The helper produced more than one buffer's worth without a line break. */
onOverflow: () => void
}
/**
* One `runtime.ps1 -Serve` child, framed as NDJSON lines.
*
* Split from the host so the host reads as what it is — a queue, a retry policy
* and a correlation check — rather than that plus stream plumbing. Responses
* carry base64 screenshots and routinely exceed a megabyte, so lines are
* reassembled across chunks with a decoder that survives a code point split
* across a chunk boundary.
*/
export class DesktopScriptServeChannel {
private readonly decoder = new StringDecoder('utf8')
private buffer = ''
private stderrTail = ''
private detach: (() => void) | null = null
private closed = false
constructor(
private readonly child: RuntimeChildProcess,
private readonly handlers: ServeChannelHandlers
) {
const onStdout = (chunk: Buffer | string): void => this.readStdout(chunk)
const onStderr = (chunk: Buffer | string): void => {
this.stderrTail = `${this.stderrTail}${chunk.toString()}`.slice(-MAX_STDERR_CHARS)
}
// Why close and not exit: the caller classifies the failure from stderr, and
// only close guarantees the stdio streams were drained first.
const onClose = (code: number | null, signal: NodeJS.Signals | null): void =>
this.reportGone(signal ? `signal ${signal}` : `code ${code ?? 'unknown'}`)
const onError = (error: Error): void => this.reportGone(error.message)
child.stdout.on('data', onStdout)
child.stderr.on('data', onStderr)
child.once('close', onClose)
child.once('error', onError)
// An unhandled stream error is an uncaught exception in the main process.
child.stdin.on('error', () => {})
this.detach = (): void => {
child.stdout.off('data', onStdout)
child.stderr.off('data', onStderr)
child.off('close', onClose)
child.off('error', onError)
child.on('error', () => {})
}
}
write(payload: string, onError: (error: Error) => void): void {
if (this.closed) {
return
}
this.child.stdin.write(payload, (error) => {
// A destroyed stdin calls back after stop(); reporting then charges the
// caller a second failure for one operation. Deliberately redundant with
// the host's own staleness check — keep both, and note that each is
// pinned separately, this one by the "once stopped" tests here.
if (error && !this.closed) {
onError(error)
}
})
}
/** Stop the helper and go silent; handlers are not called afterwards. */
stop(): void {
if (this.closed) {
return
}
this.closed = true
this.detach?.()
this.detach = null
this.buffer = ''
// Closing stdin ends the serve loop; the kill covers a wedged helper.
try {
this.child.stdin.end()
} catch {
/* already closed */
}
this.child.kill()
}
private reportGone(detail: string): void {
if (this.closed) {
return
}
const text = [detail, this.stderrTail.trim()].filter(Boolean).join(': ')
this.closed = true
this.detach?.()
this.detach = null
this.handlers.onGone(text)
}
private readStdout(chunk: Buffer | string): void {
if (this.closed) {
return
}
this.buffer += typeof chunk === 'string' ? chunk : this.decoder.write(chunk)
if (this.buffer.length > MAX_RESPONSE_CHARS) {
this.buffer = ''
this.handlers.onOverflow()
return
}
for (let newline = this.buffer.indexOf('\n'); newline >= 0;) {
// Slice a trailing CR off by index; trimming copies the whole payload.
const end = newline > 0 && this.buffer.charCodeAt(newline - 1) === 13 ? newline - 1 : newline
const line = this.buffer.slice(0, end)
this.buffer = this.buffer.slice(newline + 1)
if (line.length > 0) {
this.handlers.onLine(line)
// A handler may have stopped this channel; stop reading its backlog.
if (this.closed) {
this.buffer = ''
return
}
}
newline = this.buffer.indexOf('\n')
}
}
}
export function startServeChannel(
spec: ProcessSpec,
spawn: RuntimeProcessSpawn,
handlers: ServeChannelHandlers
): DesktopScriptServeChannel {
return new DesktopScriptServeChannel(spawn(spec), handlers)
}
+6
View File
@@ -9,6 +9,7 @@ import type {
ComputerSnapshotResult
} from '../../shared/runtime-types'
import { normalizeComputerActionResult } from './computer-action-verification-normalization'
import { isComputerSidecarDiagnostic, logComputerDiagnostic } from './computer-sidecar-diagnostics'
import { validateComputerSidecarPasteText } from './computer-sidecar-paste-validation'
import { RuntimeClientError } from './runtime-client-error'
@@ -245,6 +246,11 @@ class ComputerSidecarProcess {
}
private handleMessage(message: unknown): void {
// The sidecar's stdio is piped and unread, so its warnings arrive here.
if (isComputerSidecarDiagnostic(message)) {
logComputerDiagnostic(message.message)
return
}
if (!isSidecarResponse(message)) {
return
}
+5
View File
@@ -8,6 +8,11 @@ type SidecarRequest = {
params?: Record<string, unknown>
}
// Why disconnect carries the weight on Windows: the parent stops the sidecar
// with kill('SIGTERM'), which is TerminateProcess there, so the SIGTERM handler
// below never runs and teardown rides on the IPC channel closing instead. A
// helper wedged inside a UI Automation call can still outlive that and deliver
// input after teardown; only a real signal would preempt it.
process.once('disconnect', shutdownProviders)
process.once('SIGTERM', () => {
shutdownProviders()
@@ -0,0 +1,92 @@
import { describe, expect, it } from 'vitest'
import {
FALLBACK_WINDOWS_EXECUTION_POLICY,
PREFERRED_WINDOWS_EXECUTION_POLICY,
isExecutionPolicyBlocked,
windowsPowerShellRuntimeArgs
} from './windows-powershell-execution-policy'
/**
* Captured from powershell.exe on Windows, verbatim including the hard wrapping.
*
* The discriminator has to be pinned in both directions: a policy block must
* escalate once, and a plain access denial must not, because escalation is
* sticky for the session and lands on `-ExecutionPolicy Bypass`.
*/
const POLICY_BLOCKED_RESTRICTED = [
'File C:\\Temp\\runtime.ps1 cannot be loaded because running scripts is disabled on this system. For more ',
'information, see about_Execution_Policies at https:/go.microsoft.com/fwlink/?LinkID=135170.',
' + CategoryInfo : SecurityError: (:) [], ParentContainsErrorRecordException',
' + FullyQualifiedErrorId : UnauthorizedAccess'
].join('\r\n')
const POLICY_BLOCKED_REMOTE_SIGNED = [
'File C:\\Temp\\runtime.ps1 cannot be loaded. The file ',
'C:\\Temp\\runtime.ps1 is not digitally signed. You cannot run this script on the current system. For more ',
'information about running scripts and setting execution policy, see about_Execution_Policies at https:/go.microsoft.com/fwlink/?LinkID=135170.',
' + CategoryInfo : SecurityError: (:) [], ParentContainsErrorRecordException',
' + FullyQualifiedErrorId : UnauthorizedAccess'
].join('\r\n')
/** No execution policy involved: .NET refusing a file the process may not read. */
const GENUINE_ACCESS_DENIED = [
'Exception calling "ReadAllText" with "1" argument(s): "Access to the path \'C:\\Windows\\System32\\config\\SAM\' is denied."',
'At C:\\Temp\\runtime.ps1:1 char:1',
'+ [System.IO.File]::ReadAllText("C:\\Windows\\System32\\config\\SAM")',
'+ ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~',
' + CategoryInfo : NotSpecified: (:) [], MethodInvocationException',
' + FullyQualifiedErrorId : UnauthorizedAccessException'
].join('\r\n')
describe('isExecutionPolicyBlocked', () => {
it('recognises a policy block under either policy', () => {
expect(isExecutionPolicyBlocked(POLICY_BLOCKED_RESTRICTED)).toBe(true)
expect(isExecutionPolicyBlocked(POLICY_BLOCKED_REMOTE_SIGNED)).toBe(true)
})
it('does not read a plain access denial as a policy block', () => {
// UnauthorizedAccessException merely starts with the policy error id. Without
// the word boundary this matched, and one locked file downgraded the whole
// session to Bypass with no path back.
expect(isExecutionPolicyBlocked(GENUINE_ACCESS_DENIED)).toBe(false)
})
it('keeps recognising a block when the record labels are localized', () => {
// The labels are translated on a non-English host; the ids and the help
// topic are not, so the match must not depend on the labels.
const localized = POLICY_BLOCKED_RESTRICTED.replace('CategoryInfo', 'Categoria')
.replace('FullyQualifiedErrorId', 'IdErroreCompleto')
.replace(
'cannot be loaded because running scripts is disabled on this system',
'non puo essere caricato'
)
expect(isExecutionPolicyBlocked(localized)).toBe(true)
})
it('ignores the failures the helper reports every day', () => {
expect(isExecutionPolicyBlocked('code 1: The term is not recognized')).toBe(false)
expect(isExecutionPolicyBlocked('Add-Type : Cannot access the temporary directory')).toBe(false)
expect(isExecutionPolicyBlocked('')).toBe(false)
})
})
describe('windowsPowerShellRuntimeArgs', () => {
it('never emits Bypass unless the caller escalated to it', () => {
const preferred = windowsPowerShellRuntimeArgs(
'C:\\orca\\runtime.ps1',
PREFERRED_WINDOWS_EXECUTION_POLICY,
['-Serve']
)
expect(preferred).not.toContain(FALLBACK_WINDOWS_EXECUTION_POLICY)
expect(preferred).toEqual([
'-NoLogo',
'-NoProfile',
'-NonInteractive',
'-ExecutionPolicy',
'RemoteSigned',
'-File',
'C:\\orca\\runtime.ps1',
'-Serve'
])
})
})
@@ -0,0 +1,59 @@
/**
* Execution-policy handling for the Windows computer-use runtime script.
*
* Why not `Bypass` outright: it is the highest-weighted token on a
* powershell.exe command line for Defender for Endpoint, and the shipped
* runtime.ps1 does not need it — NSIS extraction writes no Zone.Identifier, so
* an unsigned local script runs under `RemoteSigned`. `Restricted` is still the
* Windows client default though, so a policy-blocked start must fall back once
* rather than leaving computer use broken.
*/
export type WindowsExecutionPolicy = 'RemoteSigned' | 'Bypass'
export const PREFERRED_WINDOWS_EXECUTION_POLICY: WindowsExecutionPolicy = 'RemoteSigned'
export const FALLBACK_WINDOWS_EXECUTION_POLICY: WindowsExecutionPolicy = 'Bypass'
/**
* Matches the SecurityError PowerShell emits for `-File` under a blocking policy.
*
* Every alternative is a PowerShell or .NET identifier, never prose. The prose
* differs by policy ("running scripts is disabled" under Restricted, "is not
* digitally signed" under RemoteSigned), is localized, and PowerShell hard-wraps
* it mid-sentence at the console width, so it can anchor nothing.
*
* The `\b` after UnauthorizedAccess is the whole discriminator and must not be
* dropped. `UnauthorizedAccess` is the FullyQualifiedErrorId of a policy block,
* but it is also a strict prefix of `UnauthorizedAccessException`, which .NET
* raises for an ordinary locked or ACL-denied file: an AV scan holding
* runtime.ps1, a locked CSC temp directory, a roaming-profile hiccup. Matching
* that escalates to `Bypass` for the rest of the session — the exact command
* line token this stack exists to stop emitting — and on the one-shot path
* replays an operation that already ran.
*
* Anchoring on the `FullyQualifiedErrorId:`/`CategoryInfo:` labels would be more
* precise still, but the labels are localized where these values are not, so a
* non-English host would stop recognising a real block and lose the fallback.
*/
const EXECUTION_POLICY_BLOCKED = /\bUnauthorizedAccess\b|\bSecurityError\b|about_Execution_Policies/
export function isExecutionPolicyBlocked(text: string): boolean {
return EXECUTION_POLICY_BLOCKED.test(text)
}
export function windowsPowerShellRuntimeArgs(
scriptPath: string,
policy: WindowsExecutionPolicy,
scriptArgs: readonly string[] = []
): string[] {
return [
// -NoLogo: a banner on stdout would be read as a malformed response line.
'-NoLogo',
'-NoProfile',
'-NonInteractive',
'-ExecutionPolicy',
policy,
'-File',
scriptPath,
...scriptArgs
]
}
+30 -8
View File
@@ -5,12 +5,13 @@ import {
mkdtempSync,
readFileSync,
readdirSync,
renameSync,
rmSync,
utimesSync,
writeFileSync
} from 'node:fs'
import os from 'node:os'
import { dirname, join } from 'node:path'
import { basename, dirname, join } from 'node:path'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { setAppEnvironment, type AppEnvironment } from '../../shared/app-environment'
@@ -141,12 +142,11 @@ describe('buildDaemonHostManifest', () => {
entryRelPath: 'resources/app.asar.unpacked/out/main/daemon-entry.js'
})
const byDest = new Map(ops.map((op) => [op.destRel, op]))
// The host exe is renamed to a distinct image name (NOT the source basename)
// so the NSIS updater's name-based `taskkill /IM Orca.exe` can't kill it.
expect(byDest.get('orca-terminal-daemon.exe')?.kind).toBe('file')
expect(byDest.has('Orca.exe')).toBe(false)
// The host exe keeps the source basename: a verbatim, signature-preserving copy with no
// image-name mismatch. What escapes the updater's sweep is the path, not the name.
expect(byDest.get('Orca.exe')?.kind).toBe('file')
const exeOp = ops.find((op) => op.sourcePath === 'C:\\app\\Orca.exe')
expect(exeOp?.destRel).not.toBe('Orca.exe')
expect(exeOp?.destRel).toBe('Orca.exe')
// V8/ICU data blobs are read by the Electron bootstrap and kept.
expect(byDest.has('icudtl.dat')).toBe(true)
// GPU/graphics DLLs are never loaded by the windowless host, so not copied.
@@ -170,7 +170,7 @@ describe('materializeRelocatedDaemonHost', () => {
const result = materializeRelocatedDaemonHost()
expect(result).not.toBeNull()
const dest = join(localAppDataDir, 'Orca', 'daemon-host', '9.9.9')
expect(result?.execPath).toBe(join(dest, 'orca-terminal-daemon.exe'))
expect(result?.execPath).toBe(join(dest, 'Orca.exe'))
expect(result?.entryPath).toBe(
join(dest, 'resources', 'app.asar.unpacked', 'out', 'main', 'daemon-entry.js')
)
@@ -203,6 +203,28 @@ describe('materializeRelocatedDaemonHost', () => {
expect(marker.entryRelPath).toBe('resources/app.asar.unpacked/out/main/daemon-entry.js')
})
it('copies the exe verbatim: same file name and same bytes as the install-dir exe', () => {
const result = materializeRelocatedDaemonHost()
const sourceExe = join(installDir, 'Orca.exe')
// Byte-for-byte under the same name is what preserves the Authenticode signature and leaves
// no renamed-image signal for endpoint detection to read as masquerading.
expect(basename(result!.execPath)).toBe(basename(sourceExe))
expect(readFileSync(result!.execPath)).toEqual(readFileSync(sourceExe))
})
it('tracks a differently-named app exe rather than pinning an image name of its own', () => {
// A dev-channel or rebranded build ships a different executableName; the host copy must follow
// it, which is what keeps the copy verbatim instead of reintroducing a name mismatch.
renameSync(join(installDir, 'Orca.exe'), join(installDir, 'Orca Nightly.exe'))
setProcessProp('execPath', join(installDir, 'Orca Nightly.exe'))
const result = materializeRelocatedDaemonHost()
const dest = join(localAppDataDir, 'Orca', 'daemon-host', '9.9.9')
expect(result?.execPath).toBe(join(dest, 'Orca Nightly.exe'))
expect(existsSync(join(dest, 'orca-terminal-daemon.exe'))).toBe(false)
// Re-resolution must agree with materialization or the fork would target a missing exe.
expect(getRelocatedDaemonHost()?.execPath).toBe(join(dest, 'Orca Nightly.exe'))
})
it('is idempotent: a valid marker short-circuits without recopying', () => {
materializeRelocatedDaemonHost()
const dest = join(localAppDataDir, 'Orca', 'daemon-host', '9.9.9')
@@ -210,7 +232,7 @@ describe('materializeRelocatedDaemonHost', () => {
const sentinel = join(dest, 'sentinel.txt')
writeFileSync(sentinel, 'keep')
const result = materializeRelocatedDaemonHost()
expect(result?.execPath).toBe(join(dest, 'orca-terminal-daemon.exe'))
expect(result?.execPath).toBe(join(dest, 'Orca.exe'))
expect(existsSync(sentinel)).toBe(true)
})
+18 -6
View File
@@ -22,6 +22,10 @@ import { inspectProcessLiveness, mergeProcessLivenessVerdict } from './daemon-pr
* imaged under it, which would otherwise kill the daemon and its live terminals. The relocated exe is a
* run-as-node Orca.exe copy (not node.exe) so there's no console flash and asar still resolves. Fail-open:
* any failure returns null and the caller forks the install-dir host (pre-relocation behavior).
*
* What escapes the updater is the PATH, not the file name: electron-builder's kill sweep selects
* processes whose image path sits under $INSTDIR. See docs/reference/windows-daemon-host-relocation.md
* for the survival contract and why the exe is copied verbatim rather than renamed.
*/
export type RelocatedDaemonHost = {
@@ -37,8 +41,14 @@ const MARKER_NAME = '.materialized.json'
// LOCAL appData (not roaming) so OneDrive/roaming never syncs this ~260MB runtime. Shared with NSIS uninstall (config/nsis/orca-installer-hooks.nsh) — keep in sync.
const LOCAL_HOST_ROOT_NAME = 'Orca'
// Copy of Orca.exe renamed to a distinct image name so the NSIS updater's `taskkill /IM Orca.exe` can't match it.
const DAEMON_HOST_EXE_NAME = 'orca-terminal-daemon.exe'
/**
* The host exe keeps the app exe's own file name, so the relocated image is a byte-for-byte,
* name-included copy of a signed binary — nothing for EDR to read as a renamed image (MITRE T1036).
* Survival comes from the path (see the module header). The one name-sensitive updater path is the
* no-PowerShell `taskkill /IM` fallback, where the daemon is killed and terminals cold-restore —
* the documented pre-relocation outcome, not a failure.
*/
const daemonHostExeName = (execPath: string): string => winPath.basename(execPath)
// V8 snapshots + ICU data the Electron bootstrap reads even under ELECTRON_RUN_AS_NODE; siblings of Orca.exe.
const RUNTIME_DATA_FILES = ['icudtl.dat', 'snapshot_blob.bin', 'v8_context_snapshot.bin']
@@ -146,8 +156,8 @@ export function buildDaemonHostManifest(sources: DaemonHostSources): CopyOp[] {
const { appDir, execPath, resourcesPath, entrySourcePath, entryRelPath } = sources
const ops: CopyOp[] = []
// Host exe (renamed) + V8/ICU blobs at dest root. Top-level DLLs omitted: GPU/media libs a windowless run-as-node host never loads (~48MB saved).
ops.push({ sourcePath: execPath, destRel: DAEMON_HOST_EXE_NAME, kind: 'file' })
// Host exe (verbatim name) + V8/ICU blobs at dest root. Top-level DLLs omitted: GPU/media libs a windowless run-as-node host never loads (~48MB saved).
ops.push({ sourcePath: execPath, destRel: daemonHostExeName(execPath), kind: 'file' })
for (const name of RUNTIME_DATA_FILES) {
ops.push({ sourcePath: join(appDir, name), destRel: name, kind: 'file', optional: true })
}
@@ -245,7 +255,7 @@ export function getRelocatedDaemonHost(): RelocatedDaemonHost | null {
if (!marker || marker.version !== version) {
return null
}
const execPath = join(dest, DAEMON_HOST_EXE_NAME)
const execPath = join(dest, daemonHostExeName(sources.execPath))
const entryPath = destPath(dest, marker.entryRelPath)
if (!existsSync(execPath) || !existsSync(entryPath)) {
return null
@@ -281,7 +291,9 @@ export function materializeRelocatedDaemonHost(): RelocatedDaemonHost | null {
entryRelPath: sources.entryRelPath
}
writeFileSync(join(staging, MARKER_NAME), JSON.stringify(marker))
// Replace any stale/partial dest, then publish the staging dir atomically.
// Replace any stale/partial dest, then publish atomically. Windows refuses to delete a running
// image, so a live daemon already hosted in THIS version's dir (same-version reinstall, or a dev
// channel reusing a version) throws here and materialization fails open to the install-dir host.
rmSync(dest, { recursive: true, force: true })
renameSync(staging, dest)
} catch {
+15 -2
View File
@@ -14,7 +14,10 @@ import {
} from '../../../shared/git-status-line-stats-cache'
import { resolveWorktreeHostPath } from '../../../shared/git-metadata-path'
import { gitOptionalLocksDisabledEnv, gitStreamStdout } from '../runner'
import { findExistingWorktreeSymlinkPaths } from '../worktree-symlink-detection'
import {
findExistingWorktreeSymlinkPaths,
getSafeRelativePath
} from '../worktree-symlink-detection'
import type { GetStatusOptions } from './get-status-options'
import { statusReadLeaseOwner } from './git-read-cache-invalidation'
import { detectConflictOperation } from './git-conflict-operation'
@@ -89,8 +92,18 @@ async function dropSharedSymlinkUntrackedEntries(
if (sharedLinkPaths.length === 0 || !entries.some((entry) => entry.area === 'untracked')) {
return
}
const untrackedPaths = new Set(
entries.filter((entry) => entry.area === 'untracked').map((entry) => entry.path)
)
const candidatePaths = sharedLinkPaths.filter((rawPath) => {
const path = getSafeRelativePath(rawPath)
return path.safe && untrackedPaths.has(path.rel)
})
if (candidatePaths.length === 0) {
return
}
const sharedLinks = new Set(
await findExistingWorktreeSymlinkPaths(worktreePath, sharedLinkPaths, {
await findExistingWorktreeSymlinkPaths(worktreePath, candidatePaths, {
wslDistro: options.wslDistro
})
)

Some files were not shown because too many files have changed in this diff Show More