Free PR CI capacity by avoiding repeated setup and real-time test waits (#24355)

* Reduce repeated PR setup and transcript timing waits; add hosted comparisons

* Align parallelism contract with Node-only external rebuild toolchain

* Record hosted coverage and launch package, store, and cancellation comparisons

* Apply hosted Windows setup savings and remove measured test waits

* Keep measured PR package gains and remove completed comparison jobs

* Report measured test counts with precise units
This commit is contained in:
Neil
2026-10-01 11:51:43 -07:00
committed by GitHub
parent 99db2bfae4
commit 197ea3a3b3
33 changed files with 1471 additions and 198 deletions
@@ -80,7 +80,9 @@ runs:
# PR-local stores compete with reusable build caches for the repository quota.
- name: Resolve pnpm download store
id: pnpm-store
if: github.event_name == 'pull_request'
if: >-
github.event_name == 'pull_request' &&
(runner.os != 'Windows' || runner.arch != 'X64' || !contains(inputs.cache-dependency-path, 'mobile/pnpm-lock.yaml'))
shell: bash
env:
LOCKFILE_HASH: ${{ hashFiles(inputs.cache-dependency-path) }}
@@ -92,8 +94,11 @@ runs:
printf 'arch=%s\n' "$(node -p 'require("node:os").arch()')" >> "$GITHUB_OUTPUT"
# Match setup-node's key and path so existing default-branch stores remain reusable.
# Hosted Windows x64 mixed installs cost less than restoring their root/mobile store.
- name: Restore pnpm download store without saving
if: github.event_name == 'pull_request'
if: >-
github.event_name == 'pull_request' &&
(runner.os != 'Windows' || runner.arch != 'X64' || !contains(inputs.cache-dependency-path, 'mobile/pnpm-lock.yaml'))
uses: actions/cache/restore@v5
with:
path: ${{ steps.pnpm-store.outputs.path }}
@@ -237,9 +242,9 @@ runs:
node_modules/.pnpm/@vscode+windows-process-tre*/node_modules/@vscode/windows-process-tree/build
key: native-modules-${{ runner.os }}-${{ steps.native-cache-scope.outputs.scope }}-${{ runner.arch }}-${{ inputs.native-runtime }}-node${{ steps.requested-node.outputs.node-version || steps.default-node.outputs.node-version }}-${{ hashFiles('pnpm-lock.yaml', '.github/actions/install-node-dependencies/action.yml', 'config/scripts/ensure-native-runtime.mjs', 'config/scripts/rebuild-native-deps.mjs', 'config/patches/node-pty@1.1.0.patch', 'config/patches/@vscode__windows-process-tree@0.8.0.patch', 'native/windows-registry/src/addon.cc', 'native/windows-registry/binding.gyp', 'native/windows-registry/package.json') }}
# pnpm's bundled gyp_main.py is not executable on fresh Linux runners.
# pnpm's bundled gyp_main.py is not executable; Electron rebuild uses its own node-gyp API.
- name: Use external node-gyp
if: runner.os == 'Linux' && inputs.native-runtime != 'none'
if: runner.os == 'Linux' && inputs.native-runtime == 'node'
shell: bash
env:
NATIVE_RUNTIME: ${{ inputs.native-runtime }}
@@ -32,7 +32,7 @@ runs:
run: |
set -euo pipefail
case "$FIXTURE" in
headless-serve-shutdown|cli-launch-contract) ;;
headless-serve-shutdown|cli-launch-contract|daemon-shutdown-descendants) ;;
*) echo "Unsupported package fixture: $FIXTURE" >&2; exit 1 ;;
esac
archive="$RUNNER_TEMP/orca-package-fixtures/$FIXTURE.tar"
+6
View File
@@ -13,6 +13,7 @@ on:
- '.github/actions/prepare-linux-package-fixture/**'
- 'config/docker/headless-serve-shutdown/**'
- 'config/docker/cli-launch-contract/**'
- 'config/docker/daemon-shutdown-descendants/**'
- 'package.json'
- 'pnpm-lock.yaml'
- 'pnpm-workspace.yaml'
@@ -29,6 +30,7 @@ on:
- '.github/actions/prepare-linux-package-fixture/**'
- 'config/docker/headless-serve-shutdown/**'
- 'config/docker/cli-launch-contract/**'
- 'config/docker/daemon-shutdown-descendants/**'
- 'config/scripts/ci-cache-warmup-workflow.test.mjs'
permissions:
@@ -121,6 +123,10 @@ jobs:
- uses: actions/checkout@v6
with:
persist-credentials: false
- uses: ./.github/actions/prepare-linux-package-fixture
with:
fixture: daemon-shutdown-descendants
save-cache: ${{ github.event_name != 'pull_request' }}
- uses: ./.github/actions/prepare-linux-package-fixture
with:
fixture: headless-serve-shutdown
+57 -43
View File
@@ -17,6 +17,16 @@ on:
description: JSON array of changed specs; empty runs the full suite
required: false
type: string
run_changed_e2e:
description: PR detector found specs requiring the general changed-spec lane.
required: false
type: boolean
default: true
needs_build:
description: PR detector found an Electron consumer requiring build and native cache.
required: false
type: boolean
default: true
ssh_source_changed:
description: '"true" when the PR touches SSH execution source; gates the Docker-SSH lane'
required: false
@@ -38,6 +48,7 @@ on:
jobs:
build:
name: build e2e app
if: inputs.test_files == '' || github.event_name != 'pull_request' || inputs.needs_build
runs-on: ubuntu-latest
timeout-minutes: 10
@@ -97,6 +108,7 @@ jobs:
# this immutable cache instead of compiling the same ABI concurrently.
prepare-native-cache:
name: prepare Electron native cache
if: inputs.test_files == '' || github.event_name != 'pull_request' || inputs.needs_build
runs-on: ubuntu-latest
timeout-minutes: 15
@@ -240,7 +252,7 @@ jobs:
changed-e2e:
name: changed e2e specs
needs: [build, prepare-native-cache]
if: inputs.test_files != ''
if: inputs.test_files != '' && (github.event_name != 'pull_request' || inputs.run_changed_e2e)
runs-on: ubuntu-latest
timeout-minutes: 45
@@ -271,44 +283,47 @@ jobs:
env:
TEST_FILES_JSON: ${{ inputs.test_files }}
run: |
# Why the native IME spec is dropped: it test.skip()s itself without
# ORCA_E2E_NATIVE_IBUS_HANGUL, which this lane cannot set because it has no ibus
# session. Running it here reported a green skip as coverage.
mapfile -t TEST_FILES < <(jq -r '.[] | select(
. != "tests/e2e/local-ssh-browser-routing.spec.ts" and
. != "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" and
. != "tests/e2e/pty-input-write-queue-ssh.spec.ts" and
. != "tests/e2e/ssh-ai-vault-session-history.spec.ts" and
. != "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts" and
. != "tests/e2e/ssh-cold-activation-restore.spec.ts" and
. != "tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts" and
. != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts" and
. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and
. != "tests/e2e/ssh-docker-half-open-link.spec.ts" and
. != "tests/e2e/ssh-docker-quick-open-large-listing.spec.ts" and
. != "tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts" and
. != "tests/e2e/ssh-docker-relay-stall-credential.spec.ts" and
. != "tests/e2e/ssh-docker-resource-accumulation.spec.ts" and
. != "tests/e2e/ssh-docker-transport-drop-recovery.spec.ts" and
. != "tests/e2e/ssh-external-image-preview.spec.ts" and
. != "tests/e2e/ssh-lost-kill-tab-resurrection.spec.ts" and
. != "tests/e2e/ssh-pi-compatible-agent-title.spec.ts" and
. != "tests/e2e/ssh-port-forward-lifecycle.spec.ts" and
. != "tests/e2e/ssh-reconnect-tab-destruction.spec.ts" and
. != "tests/e2e/ssh-restart-tab-accumulation.spec.ts" and
. != "tests/e2e/ssh-skill-installation.spec.ts" and
. != "tests/e2e/ssh-stale-resume-execution-host-scope.spec.ts" and
. != "tests/e2e/ssh-terminal-window-wake-stale-grid-repro.spec.ts" and
. != "tests/e2e/terminal-inline-images-ssh.spec.ts" and
. != "tests/e2e/ssh-docker-watcher-isolation.spec.ts" and
. != "tests/e2e/ssh-terminal-parking.spec.ts" and
. != "tests/e2e/terminal-retention-budget.spec.ts" and
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
. != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and
. != "tests/e2e/ssh-localhost.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
)' <<<"$TEST_FILES_JSON")
if [ -f config/scripts/ci-e2e-job-selection.mjs ]; then
printf '%s\n' "$TEST_FILES_JSON" | node config/scripts/ci-e2e-job-selection.mjs > "$RUNNER_TEMP/general-e2e-specs"
else
# Dispatching an older ref retains its existing dedicated-owner exclusions.
jq -r '.[] | select(
. != "tests/e2e/local-ssh-browser-routing.spec.ts" and
. != "tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts" and
. != "tests/e2e/pty-input-write-queue-ssh.spec.ts" and
. != "tests/e2e/ssh-ai-vault-session-history.spec.ts" and
. != "tests/e2e/ssh-codex-display-artifacts-repro.spec.ts" and
. != "tests/e2e/ssh-cold-activation-restore.spec.ts" and
. != "tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts" and
. != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts" and
. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts" and
. != "tests/e2e/ssh-docker-half-open-link.spec.ts" and
. != "tests/e2e/ssh-docker-quick-open-large-listing.spec.ts" and
. != "tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts" and
. != "tests/e2e/ssh-docker-relay-stall-credential.spec.ts" and
. != "tests/e2e/ssh-docker-resource-accumulation.spec.ts" and
. != "tests/e2e/ssh-docker-transport-drop-recovery.spec.ts" and
. != "tests/e2e/ssh-external-image-preview.spec.ts" and
. != "tests/e2e/ssh-lost-kill-tab-resurrection.spec.ts" and
. != "tests/e2e/ssh-pi-compatible-agent-title.spec.ts" and
. != "tests/e2e/ssh-port-forward-lifecycle.spec.ts" and
. != "tests/e2e/ssh-reconnect-tab-destruction.spec.ts" and
. != "tests/e2e/ssh-restart-tab-accumulation.spec.ts" and
. != "tests/e2e/ssh-skill-installation.spec.ts" and
. != "tests/e2e/ssh-stale-resume-execution-host-scope.spec.ts" and
. != "tests/e2e/ssh-terminal-window-wake-stale-grid-repro.spec.ts" and
. != "tests/e2e/terminal-inline-images-ssh.spec.ts" and
. != "tests/e2e/ssh-docker-watcher-isolation.spec.ts" and
. != "tests/e2e/ssh-terminal-parking.spec.ts" and
. != "tests/e2e/terminal-retention-budget.spec.ts" and
. != "tests/e2e/ssh-startup-exec-readiness.spec.ts" and
. != "tests/e2e/paired-startup-exec-readiness.spec.ts" and
. != "tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts" and
. != "tests/e2e/ssh-localhost.spec.ts" and
. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"
)' <<<"$TEST_FILES_JSON" > "$RUNNER_TEMP/general-e2e-specs"
fi
mapfile -t TEST_FILES < "$RUNNER_TEMP/general-e2e-specs"
if [ "${#TEST_FILES[@]}" -eq 0 ]; then
echo "Changed specs are all owned by dedicated lanes."
exit 0
@@ -417,10 +432,9 @@ jobs:
mv test-results "e2e-traces/watcher-isolation"
fi
# Why always(): this lane gates SSH parking/retention plus startup-exec
# readiness across live SSH, headed paired, and headless serve topologies.
# Keep every SSH journey after a test failure; stop work on superseded commits.
- name: Run Docker SSH terminal parking + startup readiness E2E
if: always() && matrix.shard == 1
if: '!cancelled() && matrix.shard == 1'
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker-terminal-parking
- name: Keep terminal-parking traces
@@ -433,7 +447,7 @@ jobs:
# Separate VMs isolate destructive SSH fixtures while keeping one worker per shard.
- name: Run remaining Docker SSH E2E
if: always()
if: '!cancelled()'
run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 pnpm run test:e2e:ssh-docker --shard=${{ matrix.shard }}/4
- name: Keep remaining-ssh-docker traces
+15
View File
@@ -51,6 +51,8 @@ jobs:
package_windows: ${{ steps.readiness.outputs.reused != 'true' && steps.filter.outputs.package_windows }}
e2e_should_run: ${{ steps.e2e_filter.outputs.should_run }}
test_files: ${{ steps.e2e_filter.outputs.test_files }}
e2e_run_changed: ${{ steps.e2e_filter.outputs.e2e_run_changed }}
e2e_needs_build: ${{ steps.e2e_filter.outputs.e2e_needs_build }}
ssh_source_changed: ${{ steps.e2e_filter.outputs.ssh_source_changed }}
native_ime_source_changed: ${{ steps.e2e_filter.outputs.native_ime_source_changed }}
wsl_source_changed: ${{ steps.e2e_filter.outputs.wsl_source_changed }}
@@ -74,6 +76,7 @@ jobs:
/config/scripts/check-readme-local-links.mjs
/config/scripts/pr-code-change-scope.mjs
/config/scripts/pr-e2e-source-routing.mjs
/config/scripts/ci-e2e-job-selection.mjs
/config/scripts/pr-ready-check-reuse.mjs
sparse-checkout-cone-mode: false
persist-credentials: false
@@ -133,6 +136,7 @@ jobs:
SSH_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --ssh-source)"
echo "ssh_source_changed=$SSH_SOURCE_CHANGED" >> "$GITHUB_OUTPUT"
echo "SSH source changed: $SSH_SOURCE_CHANGED"
printf '%s\n' "$TEST_FILES_JSON" | E2E_SSH_SOURCE_CHANGED="$SSH_SOURCE_CHANGED" node config/scripts/ci-e2e-job-selection.mjs --job-outputs >> "$GITHUB_OUTPUT"
# Why its own signal: the real-IME lane is a whole ibus session, not a spec, so it must
# trigger on IME source rather than on a spec name in some route's list.
NATIVE_IME_SOURCE_CHANGED="$(printf '%s\n' "$CHANGED" | node config/scripts/pr-e2e-source-routing.mjs --native-ime-source)"
@@ -918,6 +922,12 @@ jobs:
restore-keys: |
electron-builder-linux-
- uses: ./.github/actions/prepare-linux-package-fixture
id: daemon-fixture-cache
background: true
with:
fixture: daemon-shutdown-descendants
- uses: ./.github/actions/install-node-dependencies
with:
native-runtime: electron
@@ -930,9 +940,12 @@ jobs:
- uses: ./.github/actions/install-mobile-dependencies
# The checkout must pass without depending on a historical baseline that leaks.
- wait: daemon-fixture-cache
- name: Verify Linux daemon shutdown descendant cleanup
env:
ORCA_BACKGROUND_LAUNCH: '1'
ORCA_DAEMON_SHUTDOWN_FIXTURE_CACHE_IMAGE: ${{ steps.daemon-fixture-cache.outputs.image }}
run: node config/scripts/run-daemon-shutdown-descendants-docker.mjs
# Why --no-file-parallelism: every file here launches a full Electron stack twice, and each
@@ -1228,6 +1241,8 @@ jobs:
ref: ${{ github.event.pull_request.head.sha }}
test_files: ${{ needs.code_paths.outputs.test_files }}
ssh_source_changed: ${{ needs.code_paths.outputs.ssh_source_changed }}
run_changed_e2e: ${{ needs.code_paths.outputs.e2e_run_changed != 'false' }}
needs_build: ${{ needs.code_paths.outputs.e2e_needs_build != 'false' }}
# Why this is not in verify's needs: it is the first PR-gate run of a harness whose reliability
# is only known from nightly main runs (20/20 green, 2026-08-09..2026-08-29, p50 3m25s). It
@@ -1,4 +1,5 @@
import { readFileSync } from 'node:fs'
import { runInNewContext } from 'node:vm'
import { describe, expect, it } from 'vitest'
import { parse } from 'yaml'
@@ -29,12 +30,14 @@ describe('CI dependency download caches', () => {
])
})
it('restores PR stores with setup-node keys without registering a post-job save', () => {
it('restores PR stores except measured Windows mixed installs, without a post-job save', () => {
const resolve = action.runs.steps.find((step) => step.id === 'pnpm-store')
const restore = action.runs.steps.find(
(step) => step.name === 'Restore pnpm download store without saving'
)
expect(resolve.if).toBe("github.event_name == 'pull_request'")
expect(resolve.if).toBe(
"github.event_name == 'pull_request' && (runner.os != 'Windows' || runner.arch != 'X64' || !contains(inputs.cache-dependency-path, 'mobile/pnpm-lock.yaml'))"
)
expect(restore.if).toBe(resolve.if)
expect(restore.uses).toBe('actions/cache/restore@v5')
expect(restore.with.path).toBe('${{ steps.pnpm-store.outputs.path }}')
@@ -53,6 +56,49 @@ describe('CI dependency download caches', () => {
expect(saves[0].if).toContain("github.ref == 'refs/heads/main'")
expect(saves[0].if).toContain("github.event_name != 'pull_request'")
expect(saves[0].with.path).toBe('${{ steps.verification-cache.outputs.path }}')
const windows = workflow('pr').jobs.package_windows.steps.find((step) =>
step.uses?.includes('install-node-dependencies')
)
expect(windows.with['cache-dependency-path'].trim().split('\n')).toEqual([
'pnpm-lock.yaml',
'mobile/pnpm-lock.yaml'
])
})
it.each([
['Windows x64 mixed PR', 'pull_request', 'Windows', 'X64', true, false, ''],
['Windows ARM64 mixed PR', 'pull_request', 'Windows', 'ARM64', true, true, ''],
['Windows x86 mixed PR', 'pull_request', 'Windows', 'X86', true, true, ''],
['Windows x64 root-only PR', 'pull_request', 'Windows', 'X64', false, true, ''],
['Linux x64 mixed PR', 'pull_request', 'Linux', 'X64', true, true, ''],
['Linux ARM64 mixed PR', 'pull_request', 'Linux', 'ARM64', true, true, ''],
['macOS ARM64 mixed PR', 'pull_request', 'macOS', 'ARM64', true, true, ''],
['Windows x64 mixed push', 'push', 'Windows', 'X64', true, false, 'pnpm'],
['Windows x64 mixed manual run', 'workflow_dispatch', 'Windows', 'X64', true, false, 'pnpm']
])('%s keeps its scoped store policy', (_name, event, os, arch, mixed, restore, cache) => {
const context = {
github: { event_name: event },
runner: { os, arch },
inputs: {
'cache-dependency-path': mixed ? 'pnpm-lock.yaml\nmobile/pnpm-lock.yaml' : 'pnpm-lock.yaml'
},
contains: (value, search) => value.toLowerCase().includes(search.toLowerCase())
}
const evaluate = (expression) =>
runInNewContext(
expression.replaceAll('inputs.cache-dependency-path', 'inputs["cache-dependency-path"]'),
context
)
for (const step of action.runs.steps.filter(
(step) =>
step.id === 'pnpm-store' || step.name === 'Restore pnpm download store without saving'
)) {
expect(evaluate(step.if)).toBe(restore)
}
for (const step of action.runs.steps.filter((step) => step.uses === 'actions/setup-node@v6')) {
expect(evaluate(step.with.cache.slice(3, -2))).toBe(cache)
expect(step.with['package-manager-cache']).toBe(false)
}
})
it('restores Windows packaging downloads from the release cache without a PR upload', () => {
+103
View File
@@ -0,0 +1,103 @@
import { pathToFileURL } from 'node:url'
export const DOCKER_SSH_E2E_SPECS = [
'tests/e2e/local-ssh-browser-routing.spec.ts',
'tests/e2e/ssh-client-hosted-browser-drop-reconnect.spec.ts',
'tests/e2e/pty-input-write-queue-ssh.spec.ts',
'tests/e2e/ssh-ai-vault-session-history.spec.ts',
'tests/e2e/ssh-codex-display-artifacts-repro.spec.ts',
'tests/e2e/ssh-cold-activation-restore.spec.ts',
'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts',
'tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts',
'tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts',
'tests/e2e/ssh-docker-half-open-link.spec.ts',
'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts',
'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts',
'tests/e2e/ssh-docker-relay-stall-credential.spec.ts',
'tests/e2e/ssh-docker-resource-accumulation.spec.ts',
'tests/e2e/ssh-docker-transport-drop-recovery.spec.ts',
'tests/e2e/ssh-external-image-preview.spec.ts',
'tests/e2e/ssh-lost-kill-tab-resurrection.spec.ts',
'tests/e2e/ssh-pi-compatible-agent-title.spec.ts',
'tests/e2e/ssh-port-forward-lifecycle.spec.ts',
'tests/e2e/ssh-reconnect-tab-destruction.spec.ts',
'tests/e2e/ssh-restart-tab-accumulation.spec.ts',
'tests/e2e/ssh-skill-installation.spec.ts',
'tests/e2e/ssh-stale-resume-execution-host-scope.spec.ts',
'tests/e2e/ssh-terminal-window-wake-stale-grid-repro.spec.ts',
'tests/e2e/terminal-inline-images-ssh.spec.ts',
'tests/e2e/ssh-docker-watcher-isolation.spec.ts',
'tests/e2e/ssh-terminal-parking.spec.ts',
'tests/e2e/terminal-retention-budget.spec.ts',
'tests/e2e/ssh-startup-exec-readiness.spec.ts',
'tests/e2e/paired-startup-exec-readiness.spec.ts'
]
export const NODE_NETWORK_E2E_SPEC =
'tests/e2e/ssh-browser-network-execution-route.docker.unit.test.ts'
export const LOCALHOST_SSH_E2E_SPEC = 'tests/e2e/ssh-localhost.spec.ts'
export const NATIVE_IME_E2E_SPEC = 'tests/e2e/terminal-ibus-hangul-native.spec.ts'
export const DEDICATED_E2E_SPECS = [
...DOCKER_SSH_E2E_SPECS,
NODE_NETWORK_E2E_SPEC,
LOCALHOST_SSH_E2E_SPEC,
NATIVE_IME_E2E_SPEC
]
const dedicatedSpecs = new Set(DEDICATED_E2E_SPECS)
const dockerSpecs = new Set(DOCKER_SSH_E2E_SPECS)
export function selectGeneralE2eSpecs(specs) {
return specs.filter((spec) => !dedicatedSpecs.has(spec))
}
function parseSpecs(input) {
const specs = JSON.parse(input)
if (!Array.isArray(specs) || specs.some((spec) => typeof spec !== 'string' || !spec)) {
throw new Error('Expected a JSON array of nonempty E2E spec paths')
}
return specs
}
export function classifyE2eJobs(input, sshSourceChanged = 'false') {
const conservative = { e2e_run_changed: true, e2e_needs_build: true }
let specs
try {
specs = parseSpecs(input)
} catch {
return conservative
}
// Empty evidence keeps allocations; the consumer still validates its input.
if (specs.length === 0) {
return conservative
}
const runChanged = selectGeneralE2eSpecs(specs).length > 0
return {
e2e_run_changed: runChanged,
e2e_needs_build:
runChanged ||
sshSourceChanged !== 'false' ||
specs.some((spec) => dockerSpecs.has(spec) || spec === LOCALHOST_SSH_E2E_SPEC)
}
}
if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
let input = ''
process.stdin.setEncoding('utf8')
for await (const chunk of process.stdin) {
input += chunk
}
if (process.argv.includes('--job-outputs')) {
for (const [name, value] of Object.entries(
classifyE2eJobs(input, process.env.E2E_SSH_SOURCE_CHANGED ?? 'false')
)) {
process.stdout.write(`${name}=${value}\n`)
}
} else {
for (const spec of selectGeneralE2eSpecs(parseSpecs(input))) {
if (/[\r\n]/.test(spec)) {
throw new Error('E2E spec paths cannot contain newlines')
}
process.stdout.write(`${spec}\n`)
}
}
}
@@ -0,0 +1,147 @@
import { readFileSync } from 'node:fs'
import { expect, it } from 'vitest'
import { parse } from 'yaml'
import { runProcess } from '../../src/shared/child-process/run-process'
import {
classifyE2eJobs,
DEDICATED_E2E_SPECS,
DOCKER_SSH_E2E_SPECS,
LOCALHOST_SSH_E2E_SPEC,
NATIVE_IME_E2E_SPEC,
NODE_NETWORK_E2E_SPEC,
selectGeneralE2eSpecs
} from './ci-e2e-job-selection.mjs'
import { selectPrE2eSpecs } from './pr-e2e-source-routing.mjs'
const workflow = parse(readFileSync('.github/workflows/e2e.yml', 'utf8'))
const prWorkflow = parse(readFileSync('.github/workflows/pr.yml', 'utf8'))
const classify = (specs, ssh = 'false') => classifyE2eJobs(JSON.stringify(specs), ssh)
it('skips the general consumer only when every requested spec has a dedicated owner', () => {
for (const spec of DEDICATED_E2E_SPECS) {
expect(classify([spec]).e2e_run_changed, spec).toBe(false)
}
expect(classify(DEDICATED_E2E_SPECS).e2e_run_changed).toBe(false)
const future = 'tests/e2e/future-unclassified.spec.ts'
expect(classify([future])).toEqual({ e2e_run_changed: true, e2e_needs_build: true })
expect(selectGeneralE2eSpecs([...DEDICATED_E2E_SPECS, future])).toEqual([future])
expect(classify([...DEDICATED_E2E_SPECS, future]).e2e_run_changed).toBe(true)
})
it('keeps Electron build and native prerequisites for every consumer requiring them', () => {
for (const spec of [...DOCKER_SSH_E2E_SPECS, LOCALHOST_SSH_E2E_SPEC]) {
expect(classify([spec]), spec).toEqual({
e2e_run_changed: false,
e2e_needs_build: true
})
}
for (const spec of [NODE_NETWORK_E2E_SPEC, NATIVE_IME_E2E_SPEC]) {
expect(classify([spec]), spec).toEqual({
e2e_run_changed: false,
e2e_needs_build: false
})
expect(classify([spec], 'true').e2e_needs_build).toBe(true)
}
expect(workflow.jobs['ssh-browser-network-route'].needs).toBeUndefined()
for (const name of ['e2e', 'changed-e2e', 'ssh-docker-watcher-isolation', 'ssh-localhost']) {
expect(workflow.jobs[name].needs, name).toEqual(['build', 'prepare-native-cache'])
}
})
it('retains allocations when selection or SSH evidence is incomplete', () => {
for (const input of ['', '[]', 'null', '{}', '[null]', '[""]', 'malformed']) {
expect(classifyE2eJobs(input), input).toEqual({
e2e_run_changed: true,
e2e_needs_build: true
})
}
expect(classify([NODE_NETWORK_E2E_SPEC], '').e2e_needs_build).toBe(true)
})
it('preserves the requested specs across source-routed and mixed selections', () => {
for (const files of [
['src/main/ssh/connection.ts'],
['src/main/browser/ssh-browser-network-execution-route.ts'],
['src/main/agent-hooks/server.ts'],
['src/shared/terminal-unicode-provider.ts'],
['tests/e2e/ssh-localhost.spec.ts', 'tests/e2e/future-unclassified.spec.ts']
]) {
const specs = selectPrE2eSpecs(files)
const general = selectGeneralE2eSpecs(specs)
const dedicated = specs.filter((spec) => DEDICATED_E2E_SPECS.includes(spec))
expect([...general, ...dedicated].sort(), files.join(', ')).toEqual(specs)
expect(new Set([...general, ...dedicated]).size).toBe(specs.length)
}
})
it('applies allocation hints only to PRs and retains other callers and full references', () => {
for (const name of ['run_changed_e2e', 'needs_build']) {
expect(workflow.on.workflow_call.inputs[name]).toMatchObject({
type: 'boolean',
required: false,
default: true
})
}
for (const name of ['build', 'prepare-native-cache']) {
expect(workflow.jobs[name].if).toBe(
"inputs.test_files == '' || github.event_name != 'pull_request' || inputs.needs_build"
)
}
expect(workflow.jobs.e2e.if).toBe("inputs.test_files == ''")
const changed = workflow.jobs['changed-e2e']
expect(changed.if).toContain("github.event_name != 'pull_request' || inputs.run_changed_e2e")
const command = changed.steps.find((step) => step.name === 'Run changed E2E specs').run
expect(command).toContain('node config/scripts/ci-e2e-job-selection.mjs >')
expect(command).not.toContain('mapfile -t TEST_FILES < <(')
const fallback = [...command.matchAll(/\. != "([^"]+)"/g)].map((match) => match[1])
expect(fallback).toEqual(DEDICATED_E2E_SPECS)
})
it('publishes conservative hints for malformed evidence and refuses a malformed consumer list', async () => {
const program = 'config/scripts/ci-e2e-job-selection.mjs'
const consumer = await runProcess({
program: process.execPath,
args: [program],
input: 'malformed',
timeoutMs: 10000
})
expect(consumer.code).not.toBe(0)
const hints = await runProcess({
program: process.execPath,
args: [program, '--job-outputs'],
input: 'malformed',
timeoutMs: 10000
})
expect(hints.code, hints.stderr).toBe(0)
expect(hints.stdout).toBe('e2e_run_changed=true\ne2e_needs_build=true\n')
})
it('classifies PR consumers in the existing detector and passes conservative allocation hints', () => {
const detector = prWorkflow.jobs.code_paths
expect(detector.steps[0].with['sparse-checkout']).toContain(
'/config/scripts/ci-e2e-job-selection.mjs'
)
for (const name of ['e2e_run_changed', 'e2e_needs_build']) {
expect(detector.outputs[name]).toBe(`\${{ steps.e2e_filter.outputs.${name} }}`)
}
const command = detector.steps.find((step) => step.id === 'e2e_filter').run
expect(command).toContain('E2E_SSH_SOURCE_CHANGED="$SSH_SOURCE_CHANGED"')
expect(command).toContain('ci-e2e-job-selection.mjs --job-outputs >> "$GITHUB_OUTPUT"')
expect(prWorkflow.jobs.e2e.with.run_changed_e2e).toBe(
"${{ needs.code_paths.outputs.e2e_run_changed != 'false' }}"
)
expect(prWorkflow.jobs.e2e.with.needs_build).toBe(
"${{ needs.code_paths.outputs.e2e_needs_build != 'false' }}"
)
})
it('runs remaining SSH tests after real failures and stops them when a run is cancelled', () => {
const steps = workflow.jobs['ssh-docker-watcher-isolation'].steps
expect(steps.find((step) => step.name === 'Run remaining Docker SSH E2E').if).toBe('!cancelled()')
expect(
steps.find((step) => step.name === 'Run Docker SSH terminal parking + startup readiness E2E').if
).toBe('!cancelled() && matrix.shard == 1')
for (const step of steps.filter((step) => step.name?.startsWith('Keep '))) {
expect(step.if).toContain('always()')
}
})
+11 -4
View File
@@ -17,7 +17,7 @@ describe('CI native toolchain preparation', () => {
expect(toolchain.env.NATIVE_CACHE_HIT).toContain(`steps.${id}.outputs.cache-hit`)
}
expect(index).toBeLessThan(steps.findIndex((step) => step.name === 'Prepare native runtime'))
expect(toolchain.if).toBe("runner.os == 'Linux' && inputs.native-runtime != 'none'")
expect(toolchain.if).toBe("runner.os == 'Linux' && inputs.native-runtime == 'node'")
})
// The action's toolchain workaround only runs in Linux Bash.
@@ -25,9 +25,7 @@ describe('CI native toolchain preparation', () => {
['node', 'true', '0', false],
['node', 'true', '1', true],
['node', 'false', '0', true],
['node', '', '0', true],
['electron', 'true', '0', true],
['electron', 'false', '0', true]
['node', '', '0', true]
])('runtime=%s cache=%s probe=%s installs=%s', (runtime, hit, probeStatus, installs) => {
const directory = mkdtempSync(join(tmpdir(), 'orca-ci-native-toolchain-'))
const log = join(directory, 'commands')
@@ -66,4 +64,13 @@ describe('CI native toolchain preparation', () => {
rmSync(directory, { recursive: true, force: true })
}
})
it('prepares Electron native modules through the existing rebuild path without the global toolchain', () => {
const preparation = steps.find((step) => step.name === 'Prepare native runtime')
expect(preparation.if).toBe("inputs.native-runtime != 'none'")
expect(preparation.env.NATIVE_RUNTIME).toBe('${{ inputs.native-runtime }}')
expect(preparation.run).toBe(
'node config/scripts/ensure-native-runtime.mjs --runtime="$NATIVE_RUNTIME"'
)
})
})
@@ -18,7 +18,13 @@ afterEach(() => {
}
})
function exercise({ hit = false, save = false, loadFails = false, inspectFails = false } = {}) {
function exercise({
hit = false,
save = false,
loadFails = false,
inspectFails = false,
fixture = 'cli-launch-contract'
} = {}) {
directory = mkdtempSync(join(tmpdir(), 'orca-package-cache-'))
const bin = join(directory, 'bin')
mkdirSync(bin)
@@ -48,7 +54,7 @@ function exercise({ hit = false, save = false, loadFails = false, inspectFails =
RUNNER_TEMP: directory,
GITHUB_OUTPUT: output,
COMMAND_LOG: commandLog,
FIXTURE: 'cli-launch-contract',
FIXTURE: fixture,
CACHE_HIT: String(hit),
SAVE_CACHE: String(save),
LOAD_FAILS: String(loadFails),
@@ -81,6 +87,12 @@ describe.runIf(process.platform !== 'win32')('fixture cache fallbacks', () => {
expect(result.commands).not.toMatch(/(?:build|save) /)
})
it('loads the daemon shutdown fixture without running package or source tests in the cache action', () => {
const result = exercise({ hit: true, fixture: 'daemon-shutdown-descendants' })
expect(result.output).toBe('image=orca-package-fixture-daemon-shutdown-descendants:cache\n')
expect(result.commands).not.toMatch(/(?:build|save|run) /)
})
it('builds and exports inline cache only when the warmer has no usable image', () => {
const result = exercise({ save: true })
expect(result.commands).toContain(
@@ -105,6 +117,7 @@ it('uses exact context/platform keys and the same restore-only action in package
step.uses?.endsWith('/prepare-linux-package-fixture')
)
expect(prepared.map((step) => step.with)).toEqual([
{ fixture: 'daemon-shutdown-descendants' },
{ fixture: 'headless-serve-shutdown' },
{ fixture: 'cli-launch-contract' }
])
@@ -0,0 +1,64 @@
import { mkdirSync, symlinkSync, writeFileSync } from 'node:fs'
import { createRequire } from 'node:module'
import { dirname, join, resolve } from 'node:path'
import { describeProcessFailure, runProcessSync } from './script-child-process.mjs'
const require = createRequire(import.meta.url)
async function getPinnedTools(version) {
const { getAppImageTools } = require('app-builder-lib/out/toolsets/linux.js')
return getAppImageTools(version, require('builder-util').Arch.x64)
}
export async function preparePrAppImageTools({
directory: requestedDirectory,
configuration = require('../electron-builder-pr-linux.config.cjs'),
getTools = getPinnedTools,
platform = process.platform,
architecture = process.arch
}) {
if (platform !== 'linux' || architecture !== 'x64') {
throw new Error('PR AppImage compression requires a Linux x64 host')
}
if (configuration.toolsets?.appimage !== '1.0.3') {
throw new Error('PR AppImage compression requires pinned AppImage toolset 1.0.3')
}
if (
configuration.compression === 'store' ||
(configuration.appImage?.compression && configuration.appImage.compression !== 'zstd')
) {
throw new Error('PR AppImage compression requires the existing zstd configuration')
}
// Resolve the original override before installing the private, child-only overlay.
const tools = await getTools(configuration.toolsets.appimage)
const version = runProcessSync({
program: tools.mksquashfs,
args: ['-version'],
timeoutMs: 10_000,
maxOutputBytes: 64 * 1024
})
if (
version.code !== 0 ||
version.timedOut ||
version.outputTruncated ||
!/^mksquashfs version 4\.6\.1(?:\s|$)/m.test(version.stdout)
) {
throw new Error(
`PR AppImage compression requires mksquashfs 4.6.1: ${describeProcessFailure(version)}`
)
}
const directory = resolve(requestedDirectory)
mkdirSync(directory)
symlinkSync(tools.desktopFileValidate, join(directory, 'desktop-file-validate'))
symlinkSync(dirname(tools.runtime), join(directory, 'runtimes'))
symlinkSync(dirname(tools.runtimeLibraries), join(directory, 'lib'))
writeFileSync(
join(directory, 'mksquashfs'),
'#!/usr/bin/env bash\nset -euo pipefail\nexec "$ORCA_PR_APPIMAGE_MKSQUASHFS" "$@" -Xcompression-level 3\n',
{ mode: 0o755 }
)
return {
APPIMAGE_TOOLS_PATH: directory,
ORCA_PR_APPIMAGE_MKSQUASHFS: tools.mksquashfs
}
}
@@ -0,0 +1,220 @@
import {
existsSync,
mkdirSync,
mkdtempSync,
readFileSync,
realpathSync,
rmSync,
statSync,
writeFileSync
} from 'node:fs'
import { createRequire } from 'node:module'
import { tmpdir } from 'node:os'
import { join, relative } from 'node:path'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { preparePrAppImageTools } from './package-linux-formats-appimage.mjs'
import { runProcessSync } from './script-child-process.mjs'
const require = createRequire(import.meta.url)
const configuration = { toolsets: { appimage: '1.0.3' } }
let root
beforeEach(() => {
root = mkdtempSync(join(tmpdir(), 'orca pr AppImage tools-'))
})
afterEach(() => {
vi.unstubAllEnvs()
rmSync(root, { recursive: true, force: true })
})
it.each([
['linux', 'arm64'],
['darwin', 'x64'],
['win32', 'x64']
])('refuses unmeasured %s/%s hosts before resolving tools', async (platform, architecture) => {
const getTools = vi.fn()
await expect(
preparePrAppImageTools({
directory: join(root, 'overlay'),
configuration,
getTools,
platform,
architecture
})
).rejects.toThrow('requires a Linux x64 host')
expect(getTools).not.toHaveBeenCalled()
})
it.each([
[{ toolsets: { appimage: '1.0.2' } }, 'pinned AppImage toolset 1.0.3'],
[{ ...configuration, appImage: { compression: 'gzip' } }, 'existing zstd configuration'],
[{ ...configuration, compression: 'store' }, 'existing zstd configuration']
])('rejects unsupported configuration %j before resolving tools', async (value, message) => {
const getTools = vi.fn()
await expect(
preparePrAppImageTools({
directory: join(root, 'overlay'),
configuration: value,
getTools,
platform: 'linux',
architecture: 'x64'
})
).rejects.toThrow(message)
expect(getTools).not.toHaveBeenCalled()
})
describe.skipIf(process.platform === 'win32')('real executable AppImage tool overlay', () => {
let original
let overlay
beforeEach(() => {
original = join(root, 'custom tools with spaces')
overlay = join(root, 'private tools')
mkdirSync(join(original, 'runtimes'), { recursive: true })
mkdirSync(join(original, 'lib', 'x64'), { recursive: true })
writeFileSync(join(original, 'runtimes', 'runtime-x64'), 'unchanged static runtime')
writeFileSync(join(original, 'lib', 'x64', 'lib.so'), 'unchanged runtime library')
writeFileSync(join(original, 'desktop-file-validate'), '#!/usr/bin/env bash\nexit 0\n', {
mode: 0o755
})
writeFileSync(
join(original, 'mksquashfs'),
[
'#!/usr/bin/env node',
'if (process.argv[2] === "-version") {',
' console.log(`mksquashfs version ${process.env.ORCA_TEST_SQUASHFS_VERSION ?? "4.6.1"}`)',
'} else { console.log(JSON.stringify(process.argv.slice(2))) }',
'process.exitCode = Number(process.env.ORCA_TEST_SQUASHFS_EXIT ?? 0)',
''
].join('\n'),
{ mode: 0o755 }
)
vi.stubEnv('APPIMAGE_TOOLS_PATH', original)
})
function prepare(options = {}) {
return preparePrAppImageTools({
directory: overlay,
configuration,
platform: 'linux',
architecture: 'x64',
...options
})
}
it('honors a custom toolset, preserving every original argument and runtime byte', async () => {
const originalProgram = join(original, 'mksquashfs')
const programBytes = readFileSync(originalProgram)
const environment = await prepare()
const args = [
'app spaces \' " $HOME $(touch expanded) `touch expanded`',
'output\nwith newline',
'-offset',
'944632',
'-comp',
'zstd'
]
const result = runProcessSync({
program: join(overlay, 'mksquashfs'),
args,
cwd: root,
env: { ...process.env, ...environment },
timeoutMs: 10_000
})
expect(result.code).toBe(0)
expect(JSON.parse(result.stdout)).toEqual([...args, '-Xcompression-level', '3'])
expect(existsSync(join(root, 'expanded'))).toBe(false)
expect(statSync(join(overlay, 'mksquashfs')).mode & 0o777).toBe(0o755)
expect(readFileSync(originalProgram)).toEqual(programBytes)
expect(process.env.APPIMAGE_TOOLS_PATH).toBe(original)
expect(environment.ORCA_PR_APPIMAGE_MKSQUASHFS).toBe(originalProgram)
vi.stubEnv('APPIMAGE_TOOLS_PATH', environment.APPIMAGE_TOOLS_PATH)
const tools = await require('app-builder-lib/out/toolsets/linux.js').getAppImageTools(
'1.0.3',
require('builder-util').Arch.x64
)
for (const [actual, expected] of [
[tools.desktopFileValidate, join(original, 'desktop-file-validate')],
[tools.runtime, join(original, 'runtimes', 'runtime-x64')],
[tools.runtimeLibraries, join(original, 'lib', 'x64')]
]) {
expect(realpathSync(actual)).toBe(realpathSync(expected))
}
expect(readFileSync(tools.runtime, 'utf8')).toBe('unchanged static runtime')
expect(readFileSync(join(tools.runtimeLibraries, 'lib.so'), 'utf8')).toBe(
'unchanged runtime library'
)
})
it('quotes an original executable path containing shell metacharacters', async () => {
const tools = await require('app-builder-lib/out/toolsets/linux.js').getAppImageTools(
'1.0.3',
require('builder-util').Arch.x64
)
const program = join(root, 'tool \' " $(touch expanded) `touch expanded`')
writeFileSync(program, readFileSync(tools.mksquashfs), { mode: 0o755 })
const environment = await prepare({ getTools: async () => ({ ...tools, mksquashfs: program }) })
const result = runProcessSync({
program: join(overlay, 'mksquashfs'),
args: ['app', 'output', '-comp', 'zstd'],
cwd: root,
env: { ...process.env, ...environment },
timeoutMs: 10_000
})
expect(result.code).toBe(0)
expect(JSON.parse(result.stdout)).toEqual([
'app',
'output',
'-comp',
'zstd',
'-Xcompression-level',
'3'
])
expect(existsSync(join(root, 'expanded'))).toBe(false)
})
it('normalizes a relative overlay into an absolute builder override', async () => {
const environment = await prepare({ directory: relative(process.cwd(), overlay) })
expect(environment.APPIMAGE_TOOLS_PATH).toBe(overlay)
expect(existsSync(join(overlay, 'mksquashfs'))).toBe(true)
})
it('propagates the original executable failure without a fallback', async () => {
const environment = await prepare()
vi.stubEnv('ORCA_TEST_SQUASHFS_EXIT', '17')
const result = runProcessSync({
program: join(overlay, 'mksquashfs'),
args: ['app', 'output', '-comp', 'zstd'],
env: { ...process.env, ...environment },
timeoutMs: 10_000
})
expect(result.code).toBe(17)
})
it.each([
['ORCA_TEST_SQUASHFS_VERSION', '4.7.0'],
['ORCA_TEST_SQUASHFS_EXIT', '2']
])(
'rejects an unsupported original tool (%s=%s) without creating an overlay',
async (key, value) => {
vi.stubEnv(key, value)
await expect(prepare()).rejects.toThrow('requires mksquashfs 4.6.1')
expect(existsSync(overlay)).toBe(false)
}
)
it('keeps invalid custom toolset errors instead of downloading another toolset', async () => {
vi.stubEnv('APPIMAGE_TOOLS_PATH', join(root, 'missing custom tools'))
await expect(prepare()).rejects.toThrow(/APPIMAGE_TOOLS_PATH|AppImage tool/)
expect(existsSync(overlay)).toBe(false)
})
it('preserves upstream rejection of unsafe custom override paths', async () => {
vi.stubEnv('APPIMAGE_TOOLS_PATH', join(root, 'tools $(touch expanded)'))
await expect(prepare()).rejects.toThrow('APPIMAGE_TOOLS_PATH contains shell-unsafe characters')
expect(existsSync(overlay)).toBe(false)
expect(existsSync(join(root, 'expanded'))).toBe(false)
})
})
+10 -3
View File
@@ -14,6 +14,7 @@ import {
import { join, resolve } from 'node:path'
import { copyPrivateTree } from './space-sharing-copy.mjs'
import { spawnProcess } from './script-child-process.mjs'
import { preparePrAppImageTools } from './package-linux-formats-appimage.mjs'
const require = createRequire(import.meta.url)
const formats = ['AppImage', 'deb', 'rpm']
@@ -39,12 +40,12 @@ export function linuxFormatArguments({ format, appDirectory, outputDirectory })
]
}
function runElectronBuilder(args) {
function runElectronBuilder(args, environment) {
return new Promise((resolveBuild, reject) => {
const child = spawnProcess({
program: process.execPath,
args: [require.resolve('electron-builder/cli.js'), ...args],
env: { ...process.env, ORCA_BACKGROUND_LAUNCH: '1' },
env: { ...process.env, ORCA_BACKGROUND_LAUNCH: '1', ...environment },
stdio: 'inherit'
})
child.once('error', reject)
@@ -61,6 +62,7 @@ function runElectronBuilder(args) {
export async function packageLinuxFormats({
preparedDirectory = resolve('dist/linux-unpacked'),
outputDirectory = resolve('dist'),
prepareAppImageTools = preparePrAppImageTools,
runBuilder = runElectronBuilder
} = {}) {
const marker = join(preparedDirectory, 'resources/package-type')
@@ -82,8 +84,13 @@ export async function packageLinuxFormats({
console.log(
`[linux-package] ${format} copied in ${Math.round(performance.now() - startedFormatAt)}ms`
)
const environment =
format === 'AppImage'
? await prepareAppImageTools({ directory: join(staging, format, 'tools') })
: {}
await runBuilder(
linuxFormatArguments({ format, appDirectory, outputDirectory: formatOutput })
linuxFormatArguments({ format, appDirectory, outputDirectory: formatOutput }),
environment
)
const artifacts = readdirSync(formatOutput).filter((name) => name.endsWith(`.${format}`))
const artifactStats =
+88 -33
View File
@@ -85,6 +85,7 @@ describe('independent Linux package formats', () => {
await packageLinuxFormats({
preparedDirectory,
outputDirectory,
prepareAppImageTools: async () => ({}),
runBuilder: async (args) => {
const app = valueAfter(args, '--prepackaged')
const format = valueAfter(args, '--linux')
@@ -127,42 +128,93 @@ describe('independent Linux package formats', () => {
).toBe(false)
})
it('settles every worker before cleanup and exposes no partial artifact on failure', async () => {
const finished = []
let release
const gate = new Promise((resolve) => {
release = resolve
})
await expect(
packageLinuxFormats({
preparedDirectory,
outputDirectory,
runBuilder: async (args) => {
const format = valueAfter(args, '--linux')
if (format === 'AppImage') {
throw new Error('compression failed')
}
if (format === 'rpm') {
release()
}
await gate
const { app } = emitPackage(args)
expect(existsSync(app)).toBe(true)
finished.push(format)
}
it.each(['builder', 'tool preparation'])(
'settles every worker before cleanup and exposes no partial artifact on %s failure',
async (failurePhase) => {
const finished = []
let overlay
let release
const gate = new Promise((resolve) => {
release = resolve
})
).rejects.toMatchObject({
message: 'Linux package formats failed',
errors: [
expect.objectContaining({
message: 'AppImage packaging failed',
cause: expect.objectContaining({ message: 'compression failed' })
await expect(
packageLinuxFormats({
preparedDirectory,
outputDirectory,
prepareAppImageTools: async ({ directory }) => {
overlay = directory
mkdirSync(directory)
writeFileSync(join(directory, 'mksquashfs'), 'private wrapper')
if (failurePhase === 'tool preparation') {
throw new Error('compression failed')
}
return { APPIMAGE_TOOLS_PATH: directory, ORCA_PR_APPIMAGE_MKSQUASHFS: 'original tool' }
},
runBuilder: async (args, environment) => {
const format = valueAfter(args, '--linux')
if (format === 'AppImage') {
expect(environment).toEqual({
APPIMAGE_TOOLS_PATH: overlay,
ORCA_PR_APPIMAGE_MKSQUASHFS: 'original tool'
})
throw new Error('compression failed')
}
expect(environment).toEqual({})
if (format === 'rpm') {
release()
}
await gate
const { app } = emitPackage(args)
expect(existsSync(app)).toBe(true)
expect(existsSync(overlay)).toBe(true)
finished.push(format)
}
})
]
).rejects.toMatchObject({
message: 'Linux package formats failed',
errors: [
expect.objectContaining({
message: 'AppImage packaging failed',
cause: expect.objectContaining({ message: 'compression failed' })
})
]
})
expect(finished.sort()).toEqual(['deb', 'rpm'])
expect(existsSync(overlay)).toBe(false)
expect(readdirSync(outputDirectory)).toEqual([])
expect(readFileSync(join(preparedDirectory, 'resources/package-type'), 'utf8')).toBe(
'AppImage'
)
}
)
it('passes its private tool environment only to AppImage and removes it after success', async () => {
const originalToolsPath = process.env.APPIMAGE_TOOLS_PATH
let overlay
const calls = []
await packageLinuxFormats({
preparedDirectory,
outputDirectory,
prepareAppImageTools: async ({ directory }) => {
overlay = directory
mkdirSync(directory)
return { APPIMAGE_TOOLS_PATH: directory, ORCA_PR_APPIMAGE_MKSQUASHFS: 'original tool' }
},
runBuilder: async (args, environment) => {
const format = valueAfter(args, '--linux')
calls.push(format)
expect(environment).toEqual(
format === 'AppImage'
? { APPIMAGE_TOOLS_PATH: overlay, ORCA_PR_APPIMAGE_MKSQUASHFS: 'original tool' }
: {}
)
expect(process.env.APPIMAGE_TOOLS_PATH).toBe(originalToolsPath)
emitPackage(args)
}
})
expect(finished.sort()).toEqual(['deb', 'rpm'])
expect(readdirSync(outputDirectory)).toEqual([])
expect(readFileSync(join(preparedDirectory, 'resources/package-type'), 'utf8')).toBe('AppImage')
expect(calls.sort()).toEqual(targets.slice().sort())
expect(existsSync(overlay)).toBe(false)
expect(process.env.APPIMAGE_TOOLS_PATH).toBe(originalToolsPath)
})
it('rejects missing artifacts even when the builder reports success', async () => {
@@ -170,6 +222,7 @@ describe('independent Linux package formats', () => {
packageLinuxFormats({
preparedDirectory,
outputDirectory,
prepareAppImageTools: async () => ({}),
runBuilder: async (args) => {
const result = emitPackage(args)
if (result.format === 'rpm') {
@@ -187,6 +240,7 @@ describe('independent Linux package formats', () => {
packageLinuxFormats({
preparedDirectory,
outputDirectory,
prepareAppImageTools: async () => ({}),
runBuilder: async (args) => {
emitPackage(args)
}
@@ -202,6 +256,7 @@ describe('independent Linux package formats', () => {
packageLinuxFormats({
preparedDirectory,
outputDirectory,
prepareAppImageTools: async () => ({}),
runBuilder: async () => {
throw new Error('must not run')
}
@@ -212,6 +212,7 @@ describe('per-job path classification', () => {
it('runs Linux packaging when an artifact contract changes', () => {
for (const file of [
'config/scripts/package-linux-formats.mjs',
'config/scripts/package-linux-formats-appimage.mjs',
'config/scripts/script-child-process.mjs',
'config/scripts/space-sharing-copy.mjs',
'.github/actions/prepare-linux-package-fixture/action.yml',
+10 -10
View File
@@ -1,3 +1,4 @@
import { DEDICATED_E2E_SPECS } from './ci-e2e-job-selection.mjs'
import { existsSync, readdirSync, readFileSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { parse as parseJsonc } from 'jsonc-parser'
@@ -153,7 +154,9 @@ describe('PR E2E gate contract', () => {
it('uses one runner for changed specs and keeps full runs sharded', () => {
expect(e2eWorkflow.jobs.e2e.if).toBe("inputs.test_files == ''")
expect(e2eWorkflow.jobs['changed-e2e'].if).toBe("inputs.test_files != ''")
expect(e2eWorkflow.jobs['changed-e2e'].if).toBe(
"inputs.test_files != '' && (github.event_name != 'pull_request' || inputs.run_changed_e2e)"
)
expect(e2eWorkflow.jobs['changed-e2e'].strategy).toBeUndefined()
expect(e2eWorkflow.jobs.e2e.strategy.matrix.include).toEqual(
Array.from({ length: 14 }, (_, index) => ({
@@ -165,12 +168,12 @@ describe('PR E2E gate contract', () => {
(step) => step.name === 'Run changed E2E specs'
)
expect(changedRun.env.TEST_FILES_JSON).toBe('${{ inputs.test_files }}')
expect(changedRun.run).toContain('. != "tests/e2e/ssh-startup-exec-readiness.spec.ts"')
expect(changedRun.run).toContain('. != "tests/e2e/paired-startup-exec-readiness.spec.ts"')
expect(changedRun.run).toContain(
'. != "tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts"'
expect(DEDICATED_E2E_SPECS).toContain('tests/e2e/ssh-startup-exec-readiness.spec.ts')
expect(DEDICATED_E2E_SPECS).toContain('tests/e2e/paired-startup-exec-readiness.spec.ts')
expect(DEDICATED_E2E_SPECS).toContain(
'tests/e2e/ssh-docker-five-pane-input-under-flood.spec.ts'
)
expect(changedRun.run).toContain('. != "tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts"')
expect(DEDICATED_E2E_SPECS).toContain('tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts')
expect(changedRun.run).toContain('if [ "${#TEST_FILES[@]}" -eq 0 ]')
expect(changedRun.run).toContain('grep -l \'@headful\' "${TEST_FILES[@]}"')
expect(changedRun.run).toContain('E2E_PROJECT_ARGS+=(--project=electron-headful)')
@@ -695,10 +698,7 @@ describe('PR E2E gate contract', () => {
})
it('keeps the native IME spec out of the lane that would silently skip it', () => {
const changedRun = e2eWorkflow.jobs['changed-e2e'].steps.find(
(step) => step.name === 'Run changed E2E specs'
)
expect(changedRun.run).toContain('. != "tests/e2e/terminal-ibus-hangul-native.spec.ts"')
expect(DEDICATED_E2E_SPECS).toContain('tests/e2e/terminal-ibus-hangul-native.spec.ts')
// Why it still has to be routed: the dedicated lane is selected by the same route, so the
// spec appearing in test_files is how a spec-only edit reaches the real-IME lane at all.
expect(selectPrE2eSpecs(['src/shared/terminal-unicode-provider.ts'])).toContain(
@@ -300,6 +300,9 @@ describe('PR workflow parallelism', () => {
steps.findIndex((step) => step.name === 'Install dependencies')
)
expect(steps[restoreIndex].uses).toBe('actions/cache/restore@v5')
expect(steps[restoreIndex].if).toBe(
"github.event_name == 'pull_request' && (runner.os != 'Windows' || runner.arch != 'X64' || !contains(inputs.cache-dependency-path, 'mobile/pnpm-lock.yaml'))"
)
})
it('uses the repository package-manager version for every direct pnpm setup', () => {
@@ -359,7 +362,7 @@ describe('PR workflow parallelism', () => {
expect(dependencyAction.inputs['persist-native-cache'].default).toBe('true')
expect(
dependencyAction.runs.steps.find((step) => step.name === 'Use external node-gyp').if
).toBe("runner.os == 'Linux' && inputs.native-runtime != 'none'")
).toBe("runner.os == 'Linux' && inputs.native-runtime == 'node'")
const dependencyInstall = dependencyAction.runs.steps.find(
(step) => step.name === 'Install dependencies'
)
+28 -16
View File
@@ -20,6 +20,7 @@ const manifest = JSON.parse(readFileSync(manifestPath, 'utf8'))
const selectedFiles = new Set(['web-index.html'])
const visitedEntries = new Set()
const PDFJS_VIEWER_ASSET_DIRS = ['cmaps', 'standard_fonts', 'wasm']
const TEXT_REFERENCE_OUTPUT = /\.(?:css|html|m?js|svg)$/
function assertEntryIsolation() {
const entryKeys = new Set(
@@ -96,26 +97,37 @@ function listOutputFiles(directory, prefix = '') {
function includeReferencedOutputs() {
const candidates = listOutputFiles(rendererOutput).filter(
(outputPath) => !outputPath.startsWith('.vite/') && !outputPath.endsWith('.html')
(outputPath) =>
!outputPath.startsWith('.vite/') &&
!outputPath.endsWith('.html') &&
!selectedFiles.has(outputPath) &&
// Viewer binaries are copied unconditionally and cannot extend the reference closure.
(TEXT_REFERENCE_OUTPUT.test(outputPath) ||
!PDFJS_VIEWER_ASSET_DIRS.some((directory) => outputPath.startsWith(`${directory}/`)))
)
let foundReference = true
const referencesByDirectory = new Map()
while (foundReference) {
foundReference = false
for (const selectedFile of selectedFiles) {
if (!/\.(?:css|html|m?js|svg)$/.test(selectedFile)) {
// Set iteration also visits newly discovered files, including transitive references.
for (const selectedFile of selectedFiles) {
if (!TEXT_REFERENCE_OUTPUT.test(selectedFile)) {
continue
}
const directory = posix.dirname(selectedFile)
let references = referencesByDirectory.get(directory)
if (!references) {
references = candidates.map((candidate) => ({
candidate,
localReference: posix.relative(directory, candidate)
}))
referencesByDirectory.set(directory, references)
}
const contents = readFileSync(join(rendererOutput, selectedFile), 'utf8')
for (const { candidate, localReference } of references) {
if (selectedFiles.has(candidate)) {
continue
}
const contents = readFileSync(join(rendererOutput, selectedFile), 'utf8')
for (const candidate of candidates) {
if (selectedFiles.has(candidate)) {
continue
}
const localReference = posix.relative(posix.dirname(selectedFile), candidate)
if (contents.includes(candidate) || contents.includes(localReference)) {
selectedFiles.add(candidate)
foundReference = true
}
if (contents.includes(candidate) || contents.includes(localReference)) {
selectedFiles.add(candidate)
}
}
}
@@ -1,12 +1,22 @@
import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join, resolve } from 'node:path'
import { spawnSync } from 'node:child_process'
import { runProcess } from '../../src/shared/child-process/run-process'
import { afterEach, describe, expect, it } from 'vitest'
const scriptPath = resolve('config/scripts/project-renderer-web-client.mjs')
const temporaryRoots = []
function projectFixture(root) {
return runProcess({
program: process.execPath,
args: [scriptPath],
cwd: root,
env: { ...process.env, ORCA_BACKGROUND_LAUNCH: '1' },
timeoutMs: 30000
})
}
function writeFixtureFile(root, relativePath, contents) {
const targetPath = join(root, relativePath)
mkdirSync(dirname(targetPath), { recursive: true })
@@ -70,14 +80,11 @@ describe('renderer web client projection', () => {
expect(builderConfig).toContain("'!out/renderer/.vite{,/**/*}'")
})
it('copies and minifies only the web dependency closure', () => {
it('copies and minifies only the web dependency closure', async () => {
const root = createRendererFixture()
const result = spawnSync(process.execPath, [scriptPath], {
cwd: root,
encoding: 'utf8'
})
const result = await projectFixture(root)
expect(result.status, result.stderr).toBe(0)
expect(result.code, result.stderr).toBe(0)
expect(result.stdout).toContain('Projected web client: 7 files')
expect(existsSync(join(root, 'out/web/web-index.html'))).toBe(true)
expect(existsSync(join(root, 'out/web/assets/editor.worker-fixture.js'))).toBe(true)
@@ -87,31 +94,114 @@ describe('renderer web client projection', () => {
expect(readFileSync(join(root, 'out/web/assets/web.css'), 'utf8')).toBe('.root{color:red}\n')
})
it('fails when the renderer manifest omits the web entry', () => {
it('follows transitive relative references and cycles with the existing substring behavior', async () => {
const root = createRendererFixture()
writeFixtureFile(
root,
'out/renderer/assets/editor.worker-fixture.js',
'const referenced = "../workers/one.mjs"; const overlapping = "assets/token.png.backup";'
)
writeFixtureFile(root, 'out/renderer/workers/one.mjs', 'const icon = "../icons/link.svg";')
writeFixtureFile(root, 'out/renderer/icons/link.svg', '<image href="../workers/one.mjs"/>')
writeFixtureFile(root, 'out/renderer/assets/token.png', 'substring-token')
writeFixtureFile(root, 'out/renderer/assets/token.png.backup', 'longer-token')
writeFixtureFile(root, 'out/renderer/workers/unreferenced.mjs', 'export const absent = true;')
const result = await projectFixture(root)
expect(result.code, result.stderr).toBe(0)
for (const file of [
'workers/one.mjs',
'icons/link.svg',
'assets/token.png',
'assets/token.png.backup'
]) {
expect(existsSync(join(root, 'out/web', file)), file).toBe(true)
}
expect(readFileSync(join(root, 'out/web/assets/token.png'), 'utf8')).toBe('substring-token')
expect(existsSync(join(root, 'out/web/workers/unreferenced.mjs'))).toBe(false)
})
it('retains all PDFJS viewer asset directories even without textual references', async () => {
const root = createRendererFixture()
const assets = ['cmaps/fixture.bcmap', 'standard_fonts/fixture.pfb', 'wasm/fixture.wasm']
for (const file of assets) {
writeFixtureFile(root, `out/renderer/${file}`, Buffer.from([0, 255, 42]))
}
const result = await projectFixture(root)
expect(result.code, result.stderr).toBe(0)
for (const file of assets) {
expect(readFileSync(join(root, 'out/web', file))).toEqual(Buffer.from([0, 255, 42]))
}
})
it('follows referenced PDFJS text assets without expanding unreferenced viewer modules', async () => {
const root = createRendererFixture()
writeFixtureFile(
root,
'out/renderer/assets/web-entry.js',
'const viewer = "../wasm/viewer.mjs";'
)
writeFixtureFile(
root,
'out/renderer/wasm/viewer.mjs',
'const image = "../assets/viewer-image.png";'
)
writeFixtureFile(
root,
'out/renderer/wasm/unused.mjs',
'const unused = "../assets/unreferenced-image.png";'
)
writeFixtureFile(root, 'out/renderer/assets/viewer-image.png', 'viewer-image')
writeFixtureFile(root, 'out/renderer/assets/unreferenced-image.png', 'unreferenced-image')
const result = await projectFixture(root)
expect(result.code, result.stderr).toBe(0)
expect(existsSync(join(root, 'out/web/wasm/viewer.mjs'))).toBe(true)
expect(existsSync(join(root, 'out/web/wasm/unused.mjs'))).toBe(true)
expect(readFileSync(join(root, 'out/web/assets/viewer-image.png'), 'utf8')).toBe('viewer-image')
expect(existsSync(join(root, 'out/web/assets/unreferenced-image.png'))).toBe(false)
})
it.each(['../outside.js', '/absolute.js', 'C:/drive.js', 'assets\\backslash.js'])(
'rejects invalid manifest output paths: %s',
async (outputPath) => {
const root = createRendererFixture()
const manifestPath = join(root, 'out/renderer/.vite/manifest.json')
const manifest = JSON.parse(readFileSync(manifestPath, 'utf8'))
manifest['web-index.html'].file = outputPath
writeFileSync(manifestPath, JSON.stringify(manifest))
const result = await projectFixture(root)
expect(result.code).toBe(1)
expect(result.stderr).toContain('Invalid renderer output path:')
expect(readFileSync(join(root, 'out/web/stale.js'), 'utf8')).toBe('stale')
}
)
it('fails when the renderer manifest omits the web entry', async () => {
const root = createRendererFixture()
writeFixtureFile(root, 'out/renderer/.vite/manifest.json', '{}')
const result = spawnSync(process.execPath, [scriptPath], {
cwd: root,
encoding: 'utf8'
})
const result = await projectFixture(root)
expect(result.status).toBe(1)
expect(result.code).toBe(1)
expect(result.stderr).toContain('Renderer manifest is missing entry: web-index.html')
})
it('rejects renderer entries that execute another entry root', () => {
it('rejects renderer entries that execute another entry root', async () => {
const root = createRendererFixture()
const manifestPath = join(root, 'out/renderer/.vite/manifest.json')
const manifest = JSON.parse(readFileSync(manifestPath, 'utf8'))
manifest['web-index.html'].dynamicImports.push('index.html')
writeFileSync(manifestPath, JSON.stringify(manifest))
const result = spawnSync(process.execPath, [scriptPath], {
cwd: root,
encoding: 'utf8'
})
const result = await projectFixture(root)
expect(result.status).toBe(1)
expect(result.code).toBe(1)
expect(result.stderr).toContain('Renderer entry web-index.html executes entry index.html')
})
})
@@ -115,7 +115,17 @@ try {
bundles.unshift(['baseline', baseline])
}
runDocker(['build', '--platform', platform, '-t', image, dockerDir])
runDocker([
'build',
'--platform',
platform,
...(process.env.ORCA_DAEMON_SHUTDOWN_FIXTURE_CACHE_IMAGE
? ['--cache-from', process.env.ORCA_DAEMON_SHUTDOWN_FIXTURE_CACHE_IMAGE]
: []),
'-t',
image,
dockerDir
])
for (const [mode, bundlePath] of bundles) {
const result = runDocker(
[
@@ -1,3 +1,4 @@
import { DEDICATED_E2E_SPECS } from './ci-e2e-job-selection.mjs'
import { readFileSync } from 'node:fs'
import { join, resolve } from 'node:path'
import { parse } from 'yaml'
@@ -22,7 +23,8 @@ it('routes SSH browser specs to a lane that enables their opt-ins', () => {
expect(runner).toContain(`'${spec}'`)
expect(runner).toContain(`${flag}: '1'`)
expect(workflow.jobs['ssh-docker-watcher-isolation'].if).toContain(spec)
expect(changedRun.run).toContain(`. != "${spec}"`)
expect(DEDICATED_E2E_SPECS).toContain(spec)
expect(changedRun.run).toContain('node config/scripts/ci-e2e-job-selection.mjs')
}
})
@@ -42,9 +44,7 @@ it('executes both Docker network routes in a Node job with their opt-in enabled'
expect(run.env.ORCA_RUN_DOCKER_SSH_BROWSER_E2E).toBe('1')
expect(run.run).toContain(`vitest run --config config/vitest.config.ts ${spec}`)
expect(run['continue-on-error']).toBeUndefined()
expect(
workflow.jobs['changed-e2e'].steps.find((step) => step.name === 'Run changed E2E specs').run
).toContain(`. != "${spec}"`)
expect(DEDICATED_E2E_SPECS).toContain(spec)
for (const changed of [
spec,
'src/main/browser/ssh-browser-network-execution-route.ts',
@@ -1,3 +1,4 @@
import { DEDICATED_E2E_SPECS } from './ci-e2e-job-selection.mjs'
import { readFileSync } from 'node:fs'
import { createRequire } from 'node:module'
import { dirname, join } from 'node:path'
@@ -22,7 +23,8 @@ it('gives every removed SSH spec a dedicated owner even for test-only edits', ()
const changedRun = workflow.jobs['changed-e2e'].steps.find(
(step) => step.name === 'Run changed E2E specs'
).run
const excluded = [...changedRun.matchAll(/\. != "([^"]+)"/g)].map((match) => match[1])
expect(changedRun).toContain('node config/scripts/ci-e2e-job-selection.mjs')
const excluded = DEDICATED_E2E_SPECS
const owned = runners.flatMap(runnerSpecs)
expect(owned.length).toBeGreaterThan(25)
expect(new Set(owned).size).toBe(owned.length)
@@ -1,3 +1,4 @@
import { DEDICATED_E2E_SPECS } from './ci-e2e-job-selection.mjs'
import { existsSync, readFileSync } from 'node:fs'
import { resolve } from 'node:path'
import { parse } from 'yaml'
@@ -30,9 +31,7 @@ it('gives the localhost SSH journey its same-filesystem server and agent prerequ
expect(run.run).toContain('--project=electron-headless')
expect(run.run).not.toContain('--retries')
expect(run['continue-on-error']).toBeUndefined()
expect(
workflow.jobs['changed-e2e'].steps.find((step) => step.name === 'Run changed E2E specs').run
).toContain(`. != "${spec}"`)
expect(DEDICATED_E2E_SPECS).toContain(spec)
})
it('selects the localhost journey for its remote hook authorities', () => {
+379
View File
@@ -5,6 +5,208 @@ unit-selection evidence, headless runtime qualification, review cancellation and
## October 1 PR concurrency follow-up
### Where the next gains are
The [September 30 demand report](https://github.com/stablyai/orca/actions/runs/36816009362)
samples 282 of 4,194 runs across workflow/conclusion strata. It estimates full job
duration for runs created in the reporting window, rather than occupancy clipped
to that window. PR CI accounts for about 537 runner-hours, including about 321
hours of unit shards, 73 hours of E2E, 41 hours of Windows packaging, 35 hours of
Linux packaging, 18 hours of static analysis, and 11 hours of typechecking. These
are weighted estimates, not exact billing totals. Unassigned and incomplete jobs
are excluded. The report predates the merged planning/setup change below.
The [full reference run](https://github.com/stablyai/orca/actions/runs/36841821670)
ran 10,310 files exactly once. Across its five shards Vitest reports about 5,063
worker-seconds importing modules, 2,838 running tests, 685 transforming source,
267 setting up tests, and 390 preparing environments. Workers overlap, so these
figures cannot be added to predict job elapsed time. They identify repeated
imports and real-time test waits as larger targets than line-count reporting,
which uses about 1.1 runner-hours in the same demand sample.
Parallel steps share the job's CPU and memory. Their
[background/wait support](https://github.blog/changelog/2026-06-25-actions-steps-can-now-be-run-in-parallel/)
saves repeated runner setup when independent checks fit together, but does not
increase machine resources or the account's concurrent-job allowance. Cheap
preflight checks still gate expensive unit and package jobs.
Organization metadata reported the Team plan on October 1. GitHub documents
[60 standard concurrent jobs by default](https://docs.github.com/en/actions/reference/limits#job-concurrency-limits-for-github-hosted-runners)
and allows support requests for increases. The effective configured allowance
was not exposed by the API. A seven-second, repository-only sample at 10:48 UTC
found 32 queued jobs and 70 assigned job records marked in progress, including
one Blacksmith label. Those non-atomic records, runner turnover, other repositories
and provider labels cannot establish the actual allowance or simultaneous usage.
These changes reduce demand; they do not change account settings. If queues
remain, ask GitHub Support to confirm the effective organization limit before
choosing a larger allowance or paid runners.
The [follow-up PR](https://github.com/stablyai/orca/pull/24355) measures virtual
readiness deadlines in captured-transcript tests, E2E allocations with no general
consumer, renderer projection, native setup, Docker fixtures, and Windows store
restoration. Alternating hosted comparisons check output parity. Completed
benchmark workflows and drivers are removed; their trial commits retain the
exact reproduction code.
A [three-pair hosted transcript comparison](https://github.com/stablyai/orca/actions/runs/36849458799)
on one four-worker ARM runner measured baseline invocations at 127.001 / 114.307 /
114.265 seconds, versus 28.903 / 28.663 / 28.455 seconds with virtual readiness
deadlines. Median elapsed time for these five files fell about 75%. All 259
original named tests passed in every baseline and candidate, and candidates
also passed three repaint checks. Summed test-body time fell from a median 325.85
to 31.80 worker-seconds. This comparison includes Vitest startup/import work but
excludes checkout, dependency setup, and queues; it is not a measured percentage
improvement in the full unit matrix. Real emulator setup and drains remain real,
and the same captured bytes, readiness/refusal deadlines, and assertions run.
The same hosted trial passed three forced Electron native rebuilds while the
external node-gyp path pointed to a nonexistent file. A separate fresh consumer
then restored the native cache, required a real hit, and passed the existing
Electron binary probe. The Linux Node-runtime workaround remains intact.
For a network-only E2E selection in that trial, both dedicated network jobs
passed. The previous allocations consumed 87 seconds for the Electron build,
34 for the native primer, and 67 for a general job whose log confirmed that
every selected spec belonged to a dedicated lane. The candidate skipped those
three jobs before runner allocation, avoiding 188 runner-seconds in this case.
This single-case measurement excludes queues and does not predict savings for
mixed selections; the existing dedicated SSH, IME, and ordinary E2E routes remain.
A [controlled cancellation trial](https://github.com/stablyai/orca/actions/runs/36853246785)
verified both condition outcomes. After intentional cancellation, `always()`
started another 60-second follow-up, while `!cancelled()` skipped it. Cleanup
and artifact uploads succeeded in both treatments. A separate deliberately
failed test still ran its later `!cancelled()` test and cleanup. The 60 seconds
are synthetic condition evidence, not a measurement of a real SSH test's cost.
Completed comparison workflows are removed after recording their evidence;
the exact drivers and workflow remain reproducible at the trial's source commit.
The [first full PR validation](https://github.com/stablyai/orca/actions/runs/36849458648)
passed all five unit shards and both Linux/Windows package checks. Its reports
contain 10,326 unique files, each once, with zero unhandled errors. Full shard
job durations ranged from 544 to 581 seconds. They ran a different merged source
on different allocations from the earlier reference, so comparing their totals
does not establish an end-to-end speedup. The alternating transcript comparison
above is the controlled timing evidence.
The [updated full PR validation](https://github.com/stablyai/orca/actions/runs/36855833565)
passed all required gates and both package checks. All 10,335 discovered files
appear once across five passing timing reports, with zero unhandled errors;
122 modules have the existing expected skipped status. Named merge-tree discovery
and saved assignment replay match both successful validation runs. The latest
unit jobs took 520–558 seconds. Refreshing weights with that same measurement set
would reduce the largest projected load from 1,756.706 to 1,708.187 worker-seconds
(2.76%), while retaining the same 8,540.831 total. Applying the earlier successful
run's proposed weights to the latest measurements improves the maximum only
1.54% and the median 0.68%. These small, variable projections do not establish
an elapsed-time gain, so the existing weights remain.
A [three-pair Windows store comparison](https://github.com/stablyai/orca/actions/runs/36853246494)
used fresh dependency trees, stores and pnpm metadata before each treatment. The
middle pair reversed order. Cached totals include archive restoration and both
unchanged frozen installs; the mobile install ran every existing postinstall
generator. Setup/reset time and runner queues are excluded.
| Pair | First treatment | Cached total | Registry total | Registry saving |
| ---- | --------------- | ------------ | -------------- | --------------- |
| 1 | Cached | 71.092s | 37.500s | 33.592s |
| 2 | Registry | 73.556s | 39.935s | 33.621s |
| 3 | Cached | 75.660s | 37.001s | 38.659s |
Treatment medians were 73.556s cached and 37.500s registry; median paired saving
was 33.621s. Cache restore and step overhead alone cost a median 27.955s. Policy
and all six generated-output digests matched across all six treatments. Windows
x64 PR jobs using the mixed root/mobile key now skip its download-store restore;
PRs already skip store saves. Root-only Windows stores, Windows ARM64/x86, other
operating systems, non-PR writers, native/Electron caches and frozen-install policy retain their existing
behavior. The trial covers this mixed install on Windows 2022, not every Windows
dependency key or a whole PR's elapsed time.
A [three-pair daemon fixture comparison](https://github.com/stablyai/orca/actions/runs/36853246776)
reset Docker build caches and the fixture/base image before every treatment.
Warm totals include the existing action's archive restore/load and the unchanged
daemon descendant oracle; reset time and runner queues are excluded. The middle
pair again reversed order.
| Pair | First treatment | Cold total | Warm total | Warm saving |
| ---- | --------------- | ---------- | ---------- | ----------- |
| 1 | Cold | 28.984s | 20.673s | 8.311s |
| 2 | Warm | 21.798s | 28.953s | -7.155s |
| 3 | Cold | 28.964s | 17.614s | 11.350s |
Treatment medians were 28.964s cold and 20.673s warm; median paired saving was
8.311s. Oracle medians fell from 28.886s to 4.907s, while archive restore/load
cost 13.025–24.024s (15.766s median) for a 714,643,456-byte archive. BuildKit
confirmed warm provisioning was cached and cold provisioning was not; every
treatment used the same immutable base image, reaped the descendant and kept
the canary alive. The PR keeps restoration in the background during root/mobile
installation. One serial pair was slower, so the roughly 24s oracle reduction
is not a guaranteed total runner saving. The drivers and exact workflows remain
available at source commit `9231d1be6c76ccc1d2fef741a4e68ae29735a5c8`.
A [three-pair AppImage compression comparison](https://github.com/stablyai/orca/actions/runs/36855833100)
packaged the same complete Linux x64 app with the pinned 1.0.3 toolset and
mksquashfs 4.6.1. Each timing includes the private app copy, electron-builder
and blockmap generation. Tool download, app compilation, extraction, parity
checks and queues are excluded; the middle pair reversed order.
| Pair | First treatment | Default zstd 15 | PR zstd 3 | Saving |
| ---- | --------------- | --------------- | --------- | ------- |
| 1 | Default | 22.580s | 11.760s | 10.820s |
| 2 | PR | 22.612s | 12.560s | 10.052s |
| 3 | Default | 22.512s | 11.815s | 10.697s |
Median packaging time fell from 22.580s to 11.815s (47.7%); median package size
grew from 213,873,601 to 237,570,611 bytes (11.1%). All six extracted manifests
matched every path, byte, mode and symlink target: 4,096 files and 601,194,650
file bytes. The runtime prefix matched the pinned runtime exactly, and stored
SquashFS options confirmed the actual level-3 override. All static checks passed;
the representative baseline and candidate each passed the unchanged headless
and CLI journeys, including all entrypoints and both shutdown signals. The
directory build passed the existing glibc floor checks on all 19 native binaries.
Only the PR Linux x64 AppImage child receives the private tool overlay. It resolves
the existing custom-tool override first, checks the pinned tool/version and zstd
configuration, and reuses the original runtime, validator and libraries. Cleanup
waits for every package worker even on failure. Release settings, Linux ARM,
Debian/RPM packaging and all native/package gates retain their existing behavior.
The isolated AppImage result does not establish the full three-format job's gain.
The same hosted comparison projected one already-built renderer three times per
treatment, again alternating order. Baseline times were 8.079 / 8.219 / 8.129s;
candidate times were 2.568 / 2.477 / 2.403s. Median projection fell from 8.129s to
2.477s (69.5%, 5.653s saved). All six web snapshots matched all 1,137 files and
51,957,014 bytes, and the renderer input remained unchanged after every run.
These timings include the projector process but exclude renderer compilation,
checkout, setup and queues. The drivers and workflow remain available at source
commit `8ba5c9bf9f734d585f5e89945519aef4f607face`.
Two local cache screens do not justify enabling Node's compile cache. A 96-file
screen with an explicit worker flush produced a small, noisy difference. A larger
256-file screen retained all 2,088 tests: baseline elapsed times were
54.630 / 55.105 / 55.171 seconds, fresh caches 53.719 / 54.577, and a warm cache
53.732. The roughly 1.7% median difference is too small to justify cache transfer
and another test hook without stronger hosted evidence.
Vitest 4.1.11's experimental filesystem module cache is more promising locally,
but raw reuse is unsafe. A 96-file screen fell from about 8.4 to 5.9 seconds with
a warm cache, while a cold cache cost about 3%. Negative controls then reproduced
false passes after adding a preferred import extension, retargeting a symlink,
changing package exports, or changing transform inputs. Cache-disabled controls
failed correctly. A cache key must cover resolution and transform inputs as well
as file contents before any production trial; source hashes alone do not suffice.
Affected-test selection remains in shadow mode. Its first merge was September
28, so October 1 cannot satisfy the documented week of evidence. Seven sampled
complete reference reports included one red run, but only two evaluated a smaller
candidate set; each omitted about 177–179 worker-seconds out of 8,073–8,392. Five
full fallbacks are not selection-validation evidence, and two other sampled red
runs had no review artifact. These samples support keeping the conservative
policy, rather than claiming that omitting roughly 12% of files would omit the
same fraction of work.
### Shared planning and typechecking
PR planning now shares checkout and dependency setup with typechecking. Planning
runs in the background, with an explicit failure-propagating join before its
artifact is published. Static analysis remains on a separate runner. Its existing
@@ -590,3 +792,180 @@ only one potential idle allocation per day, and active development usually
requires that build. Defer another release-graph change until skip frequency
justifies it. The substantive remaining release occupancy opportunity is the
separately documented asynchronous signing policy decision.
## Persistent Vitest transform cache: rejected for now
A local 96-file import-heavy sample with Vitest 4.1.11 took 8.31/8.53 seconds
without its filesystem module cache, 8.63/8.69 seconds cold, and 5.91/5.91
seconds warm: about 30% faster warm. The cache held 3,770 modules and 83 MiB.
These timings exclude hosted cache transfer and do not establish a PR saving.
Correctness probes found eight changes that incorrectly kept a test passing
against the old transformed import or compiler output:
| Change after warming | Raw cache | Startup fingerprint |
| ----------------------------------------------------------------- | ---------- | ------------------- |
| Add preferred `value.js` beside previously resolved `value.ts` | False pass | Correctly fails |
| Retarget a source symlink while its old target still exists | False pass | Correctly fails |
| Change an inlined package's `exports` to another existing file | False pass | Correctly fails |
| Add a preferred extension in a generated source directory | False pass | Correctly fails |
| Change TypeScript's JSX factory in `tsconfig.json` | False pass | Correctly fails |
| Create the preferred file from setup after startup fingerprinting | False pass | False pass |
| Add a preferred file inside an external symlinked directory | False pass | False pass |
| Change an external file read by a transform plugin | False pass | False pass |
The fingerprint included file names/types, symlink targets, package/config/
TypeScript metadata contents, and effective alias/define options. Following
external symlink inventories and hashing declared transform inputs repaired the
last two rows, but did not repair files created after fingerprinting. All ten
cache-disabled changed-input controls failed correctly; initial and repeated
warm controls passed. Effective alias and simple define changes also invalidated
correctly without the added fingerprint.
The [Vitest 4.1.11 documentation](https://github.com/vitest-dev/vitest/blob/v4.1.11/docs/config/experimental.md#known-issues)
documents incomplete plugin-input tracking. Its
[cache implementation](https://github.com/vitest-dev/vitest/blob/v4.1.11/packages/vitest/src/node/cache/fsModuleCache.ts)
hashes the module and selected configuration, but retains previously resolved
import URLs. [Upstream fix #11381](https://github.com/vitest-dev/vitest/pull/11381)
merged September 29 and revalidates those URLs. A disposable Vitest 5.0.3 probe,
which contains that fix, reproduced all eight false-pass categories: the old
target still exists, so checking its resolved URL misses a newly preferred file
or changed package export. Upgrading alone does not make reuse safe.
To reproduce the simplest negative control outside the worktree:
1. Create `value.ts` containing `export const value = 1`, and a test importing
`./value` and asserting `expect(value).toBe(1)`. Use an isolated config/cache,
one fork worker, `ORCA_BACKGROUND_LAUNCH=1`, and
`NODE_DISABLE_COMPILE_CACHE=1`.
2. Run the installed CLI with `--experimental.fsModuleCache=true` to warm it.
Add `value.js` containing `export const value = 2`; retain `value.ts` and the
unchanged test. The same cached invocation incorrectly passes.
3. Repeat with `--experimental.fsModuleCache=false`. The assertion correctly
fails. Vitest 5 uses `--fsModuleCache` for the equivalent controls.
4. For the startup-inventory control, use unchanged setup code that creates
`value.js` only when a runtime environment switch is enabled. Remove that
file before each invocation/fingerprint; warm with the switch off, then run
with it on. The cached importer still points at `value.ts`, while the fresh
module graph correctly fails. This is an additional persistent-cache error,
not a claim that normal in-process module reuse supports arbitrary mutation.
Orca currently resolves only pinned Vitest/Vite built-in transform plugins.
Its setup files install runtime guards/shims and temporary user data, rather
than custom transforms. A narrower policy could cache only proven immutable
source/dependency inputs and leave tests, setup, virtual modules, external
fixtures, and unknown plugins cold. That requires a validated transitive input
boundary and mutation policy; hashing every source tree on each lookup would
also spend the gain. Until that policy and hosted transfer cost are measured,
the local warm result does not justify adding a persistent cache to CI.
## Test fixture imports: reuse the existing narrow builder
The pointer-drag test imported only `makeWorktree` from `store-test-helpers`,
which also loads the real store slices. Its existing identical export in
`worktrees-slice-test-fixtures` supplies the same defaults without that graph.
Changing this single import preserves the five tests, fork workers,
isolation, and disabled filesystem/Node compile caches.
Three local interleaved before/after pairs took 2.066/2.047/2.031 seconds versus
0.351/0.388/0.353 seconds: the isolated median fell 82.7%, from 2.047 to 0.353
seconds. Transformed modules fell from 1,086 to 15. This is an isolated test
result, not a whole-shard estimate: other tests need the store modules anyway.
A broader 20-file screening sample saved only 0.295 seconds at its median,
which does not justify splitting the fixture module across those consumers.
Two further import-only reuses passed the same six-run controls. The kanban
lane test mocks its card component, so switching its builder import reduced
the isolated median from 2.202 to 0.495 seconds (77.5%) and transformed modules
from 1,082 to 13, with all six tests unchanged. The autosave fixture needs
the real editor slice, but not every store slice: the same import change across
its three consuming suites reduced the median from 3.027 to 1.301 seconds
(57.0%), with 1,107 to 320 modules and all 17 tests unchanged. These
results also measure isolated file groups; they are not additive shard savings.
The remaining inspected builder-only imports already load the full store as
their subject, or use builders whose defaults differ from existing exports.
## Vitest threads: retain forks after the scoped pilot
Three local interleaved comparisons kept four workers, `isolate: true`, both
persistent caches disabled, and the same test assertions/module graph. The
94-file happy-dom renderer cohort passed all 570 tests: forks took
17.480/17.382/17.657 seconds and threads 15.048/14.890/15.013 seconds, a 14.1%
median reduction. A 23-file shared JavaScript cohort passed all 248 tests,
with its median falling from 1.435 to 1.274 seconds (11.2%). No main-process
module or native addon loaded; guards reject native loading, `chdir`, and
process signals. These Mac/Node 24 timings motivated the hosted comparison.
The [pinned Vitest pool documentation](https://github.com/vitest-dev/vitest/blob/v4.1.11/docs/config/pool.md)
defaults to forks and documents thread limitations around process APIs and
native libraries. [Node's worker documentation](https://nodejs.org/docs/latest-v24.x/api/worker_threads.html#new-workerfilename-options)
also excludes V8 flags from worker `execArgv`. Orca's `--expose-gc` worker flag
fails with `ERR_WORKER_INVALID_EXEC_ARGV` under threads. The experiment starts
both parent processes with that flag, retains it on fork workers, and removes
it only from thread worker arguments; GC availability is checked in every
test environment. Process, native, lifecycle, and GC-retention tests stay out
of this comparison.
The [hosted pilot](https://github.com/stablyai/orca/actions/runs/36855833033)
passed on Linux ARM64, four CPUs, Node 24.21.0, and Ubuntu image
`20260927.135.1`. Three alternating pairs preserved source hashes and complete
module graphs, with no main-process module or native addon loaded:
| Audited cohort | Forks, seconds | Threads, seconds | Median saving |
| -------------------------------------- | ------------------------ | ------------------------ | --------------------- |
| 94 renderer files / 570 tests | 48.591 / 48.144 / 47.299 | 42.486 / 42.402 / 42.221 | 5.742 seconds (11.9%) |
| 23 shared JavaScript files / 248 tests | 3.386 / 3.330 / 3.424 | 3.203 / 3.237 / 3.278 | 0.149 seconds (4.4%) |
Each timed group also included two isolation sentinels: totals were 96 files /
572 tests and 25 files / 250 tests, respectively, with no skips.
Their graph hashes matched in every pair (4,015 and 251 modules). Separate
single-worker positive controls passed both sentinels in each pool. With
isolation disabled, the second sentinel correctly failed on leaked state.
Missing GC failed setup, and native loading, `chdir`, and process-signal probes
failed at the guard in both pools. Those expected failures did not pass silently.
The fixed 117-file sample accounts for about 1.5% of the baseline's aggregate
module time; its isolated gains are not a whole-suite estimate. Maintaining
that exact file list for this benefit is not justified. A broader route needs
a safe eligibility policy, Node 26/Windows evidence, and a mixed full-shard
comparison: separate Vitest projects can repeat shared transforms and erase
the pool-startup saving. A renderer path alone does not prove that future
imports avoid process or native behavior. Production retains forks, and the
temporary workflow, driver, and cohort list were removed after measurement.
## Oxlint scan consolidation: rejected
The [hosted comparison](https://github.com/stablyai/orca/actions/runs/36855833063)
used one Linux ARM64/four-CPU runner, Node 24.21.0, and three alternating
baseline/candidate pairs. The baseline kept root lint and anti-slop in parallel,
then native and type-aware audits in parallel. The candidate merged the first
three scans and ran the unchanged type-aware audit alongside them.
| Pair/order | Baseline stage | Candidate stage | Change |
| ------------------ | -------------- | --------------- | ------ |
| 1: baseline first | 49.409 s | 53.197 s | +7.7% |
| 2: candidate first | 50.249 s | 60.297 s | +20.0% |
| 3: baseline first | 50.040 s | 56.672 s | +13.3% |
| Median | 50.040 s | 56.672 s | +13.3% |
These complete stage timings include anti-slop synchronization in both variants
and candidate configuration generation. Candidate preparation took only
0.141–0.255 seconds. The unchanged type-aware scan took 16.386–17.279 seconds
in the baseline wave, versus 33.561–55.736 seconds beside the merged scan;
these timings are consistent with contention on the four-CPU runner.
The corrected local Mac/16-CPU comparison had reduced the median from 16.676
to 14.381 seconds (13.8%). Both comparisons limited each Oxlint invocation to
four threads and used identical source configuration hashes. The hosted result
shows why the local gain did not justify adoption on the actual CI runner.
Coverage controls passed: the merged scan matched the exact 28,621-file union
with no missing or extra files. Thirty-one fault fixture files produced the
exact 16-diagnostic union, including all seven active root JavaScript rules.
Eighteen focused controls preserved nested-mobile exemptions, type-aware
exclusions, and exit behavior. The native audit's warnings still failed its
original `--deny-warnings` gate and became errors in the merged scan; root
warnings remained non-fatal. Every full-repository scan passed cleanly.
Keep the existing production waves. The temporary workflow and 601-line
benchmark driver were removed after recording this rejected result.
@@ -91,3 +91,22 @@ export async function createTranscriptPane(
}
return { runtime, handle: terminal.handle }
}
/** Advance readiness deadlines after real pane creation and emulator drains. */
export async function waitForTranscriptIdle(
pane: Awaited<ReturnType<typeof createTranscriptPane>>,
timeoutMs: number
) {
vi.useFakeTimers({
toFake: ['Date', 'setTimeout', 'clearTimeout', 'setInterval', 'clearInterval']
})
try {
const waiting = pane.runtime.waitForTerminal(pane.handle, { condition: 'tui-idle', timeoutMs })
// Expected refusals must have a rejection handler before advancing their deadline.
void waiting.catch(() => {})
await vi.advanceTimersByTimeAsync(timeoutMs)
return await waiting
} finally {
vi.useRealTimers()
}
}
@@ -1,5 +1,9 @@
import { describe, expect, it, vi } from 'vitest'
import { createTranscriptPane, TRANSCRIPT_PANE_PTY_ID } from './agent-transcript-pane-test-harness'
import {
createTranscriptPane,
TRANSCRIPT_PANE_PTY_ID,
waitForTranscriptIdle
} from './agent-transcript-pane-test-harness'
import {
readRuntimeFixture,
replayTranscript,
@@ -200,13 +204,15 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
describe('through the runtime', () => {
async function codexPane(name: string, size?: { cols: number; rows: number }) {
return createTranscriptPane({
const created = await createTranscriptPane({
paneTitle: 'Terminal',
foregroundProcess: 'codex',
launchAgent: 'codex',
data: readRuntimeFixture(name),
size
})
await created.runtime.readTerminal(created.handle, { screen: true })
return created
}
it.each(ALL_FIXTURES)(
@@ -214,9 +220,10 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
async (name) => {
const { runtime, handle } = await codexPane(name, { cols: 120, rows: 40 })
// Why 8s: quiescence (3s) plus the 2s poll re-reading the grid.
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 8_000 })
).resolves.toMatchObject({ condition: 'tui-idle', satisfied: true })
await expect(waitForTranscriptIdle({ runtime, handle }, 8_000)).resolves.toMatchObject({
condition: 'tui-idle',
satisfied: true
})
},
15_000
)
@@ -235,10 +242,9 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
data: provisional,
size: { cols: 120, rows: 40 }
})
await runtime.readTerminal(handle, { screen: true })
// Why 6s: past the 3s quiescence, so the quiet lane's provisional veto is what holds.
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 6_000 })
).rejects.toThrow(/timeout/)
await expect(waitForTranscriptIdle({ runtime, handle }, 6_000)).rejects.toThrow(/timeout/)
}, 15_000)
// Why no bytes: worker-start waits on a pane that has printed nothing yet, which is when the
@@ -251,6 +257,7 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
data: '',
size: { cols: 120, rows: 40 }
})
await runtime.readTerminal(handle, { screen: true })
const readVisibleScreen = vi.spyOn(runtime, 'readTerminal').mockResolvedValue({
handle,
status: 'running',
@@ -267,9 +274,7 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
nextCursor: null,
source: 'screen'
})
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 2_500 })
).rejects.toThrow(/timeout/)
await expect(waitForTranscriptIdle({ runtime, handle }, 2_500)).rejects.toThrow(/timeout/)
// Presence precondition: the visible-screen probe actually ran.
expect(readVisibleScreen).toHaveBeenCalled()
}, 15_000)
@@ -282,6 +287,7 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
launchAgent: 'codex',
data: ''
})
await runtime.readTerminal(handle, { screen: true })
runtime.seedTerminalRestoreTail(TRANSCRIPT_PANE_PTY_ID, {
text: [
'╭──────────────────────────────────────────╮',
@@ -302,16 +308,15 @@ describe('Codex 0.157 header readiness from captured bytes', () => {
nextCursor: null,
source: 'screen-unavailable'
})
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 2_500 })
).resolves.toMatchObject({ condition: 'tui-idle', satisfied: true })
await expect(waitForTranscriptIdle({ runtime, handle }, 2_500)).resolves.toMatchObject({
condition: 'tui-idle',
satisfied: true
})
}, 15_000)
it('keeps timing out on the garbled 80x24 default grid, as before', async () => {
const { runtime, handle } = await codexPane(EFFORT_OVERRIDE)
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 6_000 })
).rejects.toThrow(/timeout/)
await expect(waitForTranscriptIdle({ runtime, handle }, 6_000)).rejects.toThrow(/timeout/)
}, 15_000)
})
})
@@ -2,7 +2,7 @@ import { readFileSync } from 'node:fs'
import { join } from 'node:path'
import { afterEach, describe, expect, it, vi } from 'vitest'
import type { TuiAgent } from '../../shared/tui-agent'
import { createTranscriptPane } from './agent-transcript-pane-test-harness'
import { createTranscriptPane, waitForTranscriptIdle } from './agent-transcript-pane-test-harness'
import {
readRuntimeFixture,
replayTranscript,
@@ -435,21 +435,24 @@ describe('through the runtime', () => {
// Why 0.158 alone: its header carries no `model:`, so only this lane settles it.
const CODEX_0158 = 'codex-0-158-0-timed-turn'
async function pane(name: string, launchAgent: TuiAgent, size?: { cols: number; rows: number }) {
return createTranscriptPane({
const created = await createTranscriptPane({
paneTitle: 'Terminal',
foregroundProcess: launchAgent,
launchAgent,
data: readRuntimeFixture(name),
size
})
await created.runtime.readTerminal(created.handle, { screen: true })
return created
}
// Why 8s: quiescence (3s) plus the 2s poll re-reading the grid.
it('codex 0.158: a tui-idle wait settles once the composer is quiet', async () => {
const { runtime, handle } = await pane(CODEX_0158, 'codex', { cols: 120, rows: 40 })
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 8_000 })
).resolves.toMatchObject({ condition: 'tui-idle', satisfied: true })
await expect(waitForTranscriptIdle({ runtime, handle }, 8_000)).resolves.toMatchObject({
condition: 'tui-idle',
satisfied: true
})
}, 15_000)
it('keeps a Claude pane showing the same screen pending', async () => {
@@ -457,8 +460,6 @@ describe('through the runtime', () => {
cols: 120,
rows: 40
})
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 6_000 })
).rejects.toThrow(/timeout/)
await expect(waitForTranscriptIdle({ runtime, handle }, 6_000)).rejects.toThrow(/timeout/)
}, 15_000)
})
@@ -1,6 +1,10 @@
// One replay suite for every agent whose readiness its live screen decides (terminal-wait-detection.ts).
import { describe, expect, it, vi } from 'vitest'
import { createTranscriptPane, TRANSCRIPT_PANE_PTY_ID } from './agent-transcript-pane-test-harness'
import {
createTranscriptPane,
TRANSCRIPT_PANE_PTY_ID,
waitForTranscriptIdle
} from './agent-transcript-pane-test-harness'
import {
finalReadProjection,
finalReplayFrame,
@@ -96,8 +100,11 @@ export function describeScreenRuledAgentTranscripts(suite: ScreenRuledAgentSuite
async (fixture) => {
const { runtime, handle } = await pane(fixture)
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: READY_TIMEOUT_MS })
).resolves.toMatchObject({ condition: 'tui-idle', satisfied: true })
waitForTranscriptIdle({ runtime, handle }, READY_TIMEOUT_MS)
).resolves.toMatchObject({
condition: 'tui-idle',
satisfied: true
})
},
READY_TIMEOUT_MS + PANE_SETUP_SLACK_MS
)
@@ -106,9 +113,9 @@ export function describeScreenRuledAgentTranscripts(suite: ScreenRuledAgentSuite
'$name: a tui-idle wait does not settle ready',
async (fixture) => {
const { runtime, handle } = await pane(fixture)
const result = await runtime
.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: REFUSAL_TIMEOUT_MS })
.catch((error: unknown) => ({ satisfied: false, error: String(error) }))
const result = await waitForTranscriptIdle({ runtime, handle }, REFUSAL_TIMEOUT_MS).catch(
(error: unknown) => ({ satisfied: false, error: String(error) })
)
expect(result.satisfied).toBe(false)
},
REFUSAL_TIMEOUT_MS + PANE_SETUP_SLACK_MS
@@ -128,9 +135,9 @@ export function describeScreenRuledAgentTranscripts(suite: ScreenRuledAgentSuite
const { runtime, handle } = await createTranscriptPane(options)
await runtime.readTerminal(handle, { screen: true })
options.size = { cols: fixture.cols, rows: fixture.rows }
const result = await runtime
.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: READY_TIMEOUT_MS })
.catch(() => ({ satisfied: false }))
const result = await waitForTranscriptIdle({ runtime, handle }, READY_TIMEOUT_MS).catch(
() => ({ satisfied: false })
)
expect(result.satisfied).toBe(suite.readyWithoutScreen.includes(fixture.name))
},
READY_TIMEOUT_MS + PANE_SETUP_SLACK_MS
@@ -157,9 +164,9 @@ export function describeScreenRuledAgentTranscripts(suite: ScreenRuledAgentSuite
// Why: an echo of the reflowed size sends no SIGWINCH, so nothing repaints.
runtime.onExternalPtyResize(TRANSCRIPT_PANE_PTY_ID, firstReady.cols, firstReady.rows)
await runtime.readTerminal(handle, { screen: true })
const result = await runtime
.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: READY_TIMEOUT_MS })
.catch(() => ({ satisfied: false }))
const result = await waitForTranscriptIdle({ runtime, handle }, READY_TIMEOUT_MS).catch(
() => ({ satisfied: false })
)
expect(result.satisfied).toBe(suite.readyWithoutScreen.includes(firstReady.name))
},
READY_TIMEOUT_MS + PANE_SETUP_SLACK_MS
@@ -184,9 +191,9 @@ export function describeScreenRuledAgentTranscripts(suite: ScreenRuledAgentSuite
nextCursor: null,
source: 'screen'
})
const result = await runtime
.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 2_500 })
.catch(() => ({ satisfied: false }))
const result = await waitForTranscriptIdle({ runtime, handle }, 2_500).catch(() => ({
satisfied: false
}))
// Presence precondition: the visible-screen probe actually ran.
expect(read).toHaveBeenCalled()
return result.satisfied === true
@@ -207,6 +214,33 @@ export function describeScreenRuledAgentTranscripts(suite: ScreenRuledAgentSuite
},
15_000
)
it('waits for quiet again after a ready pane repaints', async () => {
const { runtime, handle } = await pane(firstReady)
vi.useFakeTimers({
toFake: ['Date', 'setTimeout', 'clearTimeout', 'setInterval', 'clearInterval']
})
try {
const settled = vi.fn()
const waiting = runtime.waitForTerminal(handle, {
condition: 'tui-idle',
timeoutMs: READY_TIMEOUT_MS
})
void waiting.then(settled, () => {})
await vi.advanceTimersByTimeAsync(2_000)
expect(settled).not.toHaveBeenCalled()
runtime.onPtyData(TRANSCRIPT_PANE_PTY_ID, readRuntimeFixture(firstReady.name), Date.now())
await runtime.readTerminal(handle, { screen: true })
await vi.advanceTimersByTimeAsync(2_000)
expect(settled).not.toHaveBeenCalled()
await vi.advanceTimersByTimeAsync(2_000)
await expect(waiting).resolves.toMatchObject({ condition: 'tui-idle', satisfied: true })
} finally {
vi.useRealTimers()
}
})
})
it('never reads the screen for another agent', async () => {
@@ -1,5 +1,5 @@
import { describe, expect, it, vi } from 'vitest'
import { createTranscriptPane } from './agent-transcript-pane-test-harness'
import { createTranscriptPane, waitForTranscriptIdle } from './agent-transcript-pane-test-harness'
import { readRuntimeFixture } from './agent-transcript-replay-test-harness'
vi.mock('electron', () => ({
@@ -24,8 +24,9 @@ describe('screen-rule trust stays with screen-ruled agents', () => {
const { runtime, handle } = await createTranscriptPane(options)
await runtime.readTerminal(handle, { screen: true })
options.size = { cols: 100, rows: 30 }
await expect(
runtime.waitForTerminal(handle, { condition: 'tui-idle', timeoutMs: 8_000 })
).resolves.toMatchObject({ condition: 'tui-idle', satisfied: true })
await expect(waitForTranscriptIdle({ runtime, handle }, 8_000)).resolves.toMatchObject({
condition: 'tui-idle',
satisfied: true
})
}, 20_000)
})
@@ -5,7 +5,7 @@ import { vi } from 'vitest'
import { createStore, type StoreApi } from 'zustand/vanilla'
import { createEditorSlice } from '@/store/slices/editor'
import type { AppState } from '@/store'
import { makeWorktree } from '@/store/slices/store-test-helpers'
import { makeWorktree } from '@/store/slices/worktrees-slice-test-fixtures'
export type FakeEditorDisk = {
files: Map<string, string>
@@ -14,7 +14,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest'
import type { Repo } from '../../../../shared/repo-types'
import type { Worktree } from '../../../../shared/worktree/types'
import { getWorktreeHostIdentity } from '../../../../shared/worktree/host-qualified-identity'
import { makeWorktree } from '../../store/slices/store-test-helpers'
import { makeWorktree } from '../../store/slices/worktrees-slice-test-fixtures'
const WINDOW_START = 20
const WINDOW_END = 25
@@ -3,7 +3,7 @@ import {
resolveWorkspaceKanbanPointerDragSelection,
shouldStartWorkspaceKanbanCardPointerDrag
} from './use-workspace-kanban-card-pointer-drag'
import { makeWorktree } from '../../store/slices/store-test-helpers'
import { makeWorktree } from '../../store/slices/worktrees-slice-test-fixtures'
function pointerEvent(overrides: Partial<PointerEvent> = {}): PointerEvent {
return {
@@ -1,4 +1,4 @@
import { beforeEach, describe, expect, it, vi } from 'vitest'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { sendFollowupPromptWhenAgentReady } from './agent-followup-delivery'
import {
inspectRuntimeTerminalProcess,
@@ -21,11 +21,17 @@ const INTERPRETER_WRAPPED_AGENTS = [
describe('sendFollowupPromptWhenAgentReady — interpreter-wrapped agents', () => {
beforeEach(() => {
vi.clearAllMocks()
vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout'] })
vi.stubGlobal('globalThis', globalThis)
// Deliver the prompt write eagerly so the test does not depend on retries.
vi.mocked(sendRuntimePtyInputVerified).mockResolvedValue(true)
})
afterEach(() => {
vi.useRealTimers()
vi.unstubAllGlobals()
})
for (const { agent, expectedProcess } of INTERPRETER_WRAPPED_AGENTS) {
it(`types the prompt once ${agent} is up behind a python3 wrapper with a live child`, async () => {
// The console-script agent is running: foreground comm is python3 and the
@@ -35,13 +41,16 @@ describe('sendFollowupPromptWhenAgentReady — interpreter-wrapped agents', () =
hasChildProcesses: true
})
const delivered = await sendFollowupPromptWhenAgentReady({
const delivery = sendFollowupPromptWhenAgentReady({
ptyId: 'pty-1',
expectedProcess,
prompt: 'ship it',
settings: null
})
await vi.advanceTimersByTimeAsync(4 * 150)
const delivered = await delivery
expect(delivered).toBe(true)
expect(sendRuntimePtyInputVerified).toHaveBeenCalledWith(null, 'pty-1', 'ship it\r', 'launch')
})
@@ -54,13 +63,16 @@ describe('sendFollowupPromptWhenAgentReady — interpreter-wrapped agents', () =
hasChildProcesses: false
})
const delivered = await sendFollowupPromptWhenAgentReady({
const delivery = sendFollowupPromptWhenAgentReady({
ptyId: 'pty-1',
expectedProcess,
prompt: 'ship it',
settings: null
})
await vi.advanceTimersByTimeAsync(29 * 150)
const delivered = await delivery
expect(delivered).toBe(false)
expect(sendRuntimePtyInputVerified).not.toHaveBeenCalled()
})
@@ -71,13 +83,16 @@ describe('sendFollowupPromptWhenAgentReady — interpreter-wrapped agents', () =
hasChildProcesses: false
})
const delivered = await sendFollowupPromptWhenAgentReady({
const delivery = sendFollowupPromptWhenAgentReady({
ptyId: 'pty-1',
expectedProcess,
prompt: 'ship it',
settings: null
})
await vi.advanceTimersByTimeAsync(29 * 150)
const delivered = await delivery
expect(delivered).toBe(false)
expect(sendRuntimePtyInputVerified).not.toHaveBeenCalled()
})