diff --git a/.github/workflows/bun-profile-tests.yml b/.github/workflows/bun-profile-tests.yml
index e579782c651..5fae18f2f84 100644
--- a/.github/workflows/bun-profile-tests.yml
+++ b/.github/workflows/bun-profile-tests.yml
@@ -2,6 +2,7 @@ name: Bun profile persistence
on:
pull_request:
+ types: [opened, synchronize, reopened, ready_for_review]
paths:
- 'src/**'
- 'config/**'
@@ -17,6 +18,8 @@ on:
- '.github/actions/install-node-dependencies/**'
- '.github/workflows/bun-profile-tests.yml'
workflow_dispatch:
+ schedule:
+ - cron: '30 11 * * *'
permissions:
contents: read
@@ -32,6 +35,8 @@ jobs:
timeout-minutes: 5
outputs:
should_run: ${{ steps.scope.outputs.should_run }}
+ qualification: ${{ steps.scope.outputs.qualification }}
+ runners: ${{ steps.scope.outputs.runners }}
steps:
- uses: actions/checkout@v6
with:
@@ -56,7 +61,7 @@ jobs:
strategy:
fail-fast: false
matrix:
- os: [ubuntu-22.04, ubuntu-24.04-arm, macos-14, macos-15-intel, windows-2022, windows-11-arm]
+ os: ${{ fromJSON(needs.changes.outputs.runners || '["ubuntu-22.04","ubuntu-24.04-arm","macos-14","macos-15-intel","windows-2022","windows-11-arm"]') }}
runs-on: ${{ matrix.os }}
timeout-minutes: 20
env:
@@ -82,9 +87,12 @@ jobs:
node out/orcad/orcad.js --orcad-profile-state-preflight 00000000-0000-4000-8000-000000000018
linux_glibc_floor:
- needs: changes
- # Missing/failed detection runs the full matrix; manual runs remain unconditional.
- if: ${{ !cancelled() && needs.changes.outputs.should_run != 'false' }}
+ needs: [changes, persistence]
+ # A failed smoke already blocks qualification; missing scope still selects every platform.
+ if: >-
+ ${{ !cancelled() && needs.persistence.result == 'success' &&
+ needs.changes.outputs.should_run != 'false' &&
+ needs.changes.outputs.qualification != 'false' }}
strategy:
fail-fast: false
matrix:
@@ -107,9 +115,12 @@ jobs:
- run: pnpm test:bun:profile --artifact
linux_musl:
- needs: changes
- # Missing/failed detection runs the full matrix; manual runs remain unconditional.
- if: ${{ !cancelled() && needs.changes.outputs.should_run != 'false' }}
+ needs: [changes, persistence]
+ # A failed smoke already blocks qualification; missing scope still selects every platform.
+ if: >-
+ ${{ !cancelled() && needs.persistence.result == 'success' &&
+ needs.changes.outputs.should_run != 'false' &&
+ needs.changes.outputs.qualification != 'false' }}
strategy:
fail-fast: false
matrix:
diff --git a/.github/workflows/ci-runner-demand.yml b/.github/workflows/ci-runner-demand.yml
new file mode 100644
index 00000000000..966813a65e4
--- /dev/null
+++ b/.github/workflows/ci-runner-demand.yml
@@ -0,0 +1,33 @@
+name: CI runner demand
+
+on:
+ schedule:
+ - cron: '23 4 * * *'
+ workflow_dispatch:
+
+permissions:
+ contents: read
+ actions: read
+
+concurrency:
+ group: ci-runner-demand
+ cancel-in-progress: false
+
+jobs:
+ report:
+ runs-on: ubuntu-slim
+ timeout-minutes: 15
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ sparse-checkout: config/scripts
+ persist-credentials: false
+ - name: Measure the previous complete 24 hours
+ env:
+ GH_TOKEN: ${{ github.token }}
+ run: node config/scripts/ci-runner-demand.mjs
+ - uses: actions/upload-artifact@v7
+ with:
+ name: ci-runner-demand-${{ github.run_id }}
+ path: ci-demand/
+ retention-days: 30
diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml
index ad9c1ee5139..4177b141d8a 100644
--- a/.github/workflows/e2e.yml
+++ b/.github/workflows/e2e.yml
@@ -32,9 +32,8 @@ on:
required: false
type: string
schedule:
- # Why: GitHub cron uses UTC; these slots map to 10am and 3pm
- # America/Phoenix for the default-branch E2E run.
- - cron: '0 17,22 * * *'
+ # One complete daily reference run; targeted PR coverage remains unchanged.
+ - cron: '0 17 * * *'
jobs:
build:
@@ -201,7 +200,14 @@ jobs:
node config/scripts/ci-e2e-shard-plan.mjs --verify ci-shards/assignment.json ci-shards/selected-discovery.json
- name: Run E2E tests (${{ matrix.shard_name }})
- run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --test-list=ci-shards/selected.txt
+ env:
+ PLAYWRIGHT_JSON_OUTPUT_FILE: ci-shards/results.json
+ run: xvfb-run --auto-servernum bash .github/scripts/e2e-with-window-manager.sh env SKIP_BUILD=1 ORCA_E2E_FORWARD_APP_LOGS=1 ORCA_E2E_WEB_CLIENT=1 ORCA_RELAY_PATH="$GITHUB_WORKSPACE/out/relay" pnpm run test:e2e --test-list=ci-shards/selected.txt --reporter=list,json
+
+ - name: Summarize E2E failures
+ if: always()
+ continue-on-error: true
+ run: node config/scripts/ci-e2e-failure-summary.mjs ci-shards/results.json
- name: Upload E2E shard assignment
if: always()
diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml
index 4580a6e2385..5c9094ae26a 100644
--- a/.github/workflows/pr.yml
+++ b/.github/workflows/pr.yml
@@ -1,5 +1,5 @@
name: PR Checks
-run-name: 'PR ${{ github.event.pull_request.number }} | source ${{ github.sha }} | workflow ${{ github.workflow_sha }}'
+run-name: "PR ${{ github.event.pull_request.number }} | source ${{ github.sha }} | workflow ${{ github.workflow_sha }} | unit ${{ github.event.pull_request.draft && vars.ORCA_UNIT_SELECTION_MODE == 'selected' && 'selected' || 'full' }}"
on:
pull_request:
@@ -161,7 +161,17 @@ jobs:
filter: blob:none
persist-credentials: false
+ # Why two guarded installs: the mixed root+mobile store entry is 537 MB against
+ # 321 MB for root alone, and restoring it costs 8.6s against 4.6s. Most PRs skip the
+ # mobile install below, so they were paying 216 MB for packages they never link. The
+ # root-only key is also the one the hourly warmer reseeds. Only one of these runs.
- uses: ./.github/actions/install-node-dependencies
+ if: needs.code_paths.outputs.mobile_dependencies != 'true'
+ with:
+ native-runtime: node
+
+ - uses: ./.github/actions/install-node-dependencies
+ if: needs.code_paths.outputs.mobile_dependencies == 'true'
with:
native-runtime: node
cache-dependency-path: |
@@ -620,16 +630,19 @@ jobs:
node-version: '24'
test:
- needs: [code_paths, test_native_cache]
+ needs: [code_paths, test_native_cache, static_analysis, typecheck]
# Honor cancellation while allowing the optional native-cache primer to skip.
if: >-
!cancelled() &&
needs.code_paths.outputs.test == 'true' &&
+ needs.static_analysis.result == 'success' &&
+ needs.typecheck.result == 'success' &&
(needs.test_native_cache.result == 'success' || needs.test_native_cache.result == 'skipped')
uses: ./.github/workflows/unit-tests.yml
with:
node_versions: '["24"]'
runner: ubuntu-24.04-arm
+ selection_mode: ${{ vars.ORCA_UNIT_SELECTION_MODE || 'shadow' }}
# Why a separate job: the test needs a real Chrome, and the sharded `test` matrix
# would pay for it on every shard to run one file in whichever shard it landed in.
@@ -784,6 +797,7 @@ jobs:
tests/e2e/cross-version-wire/cross-version-worktree-identity-downgrade.unit.test.ts
tests/e2e/cross-version-wire/cross-version-session-tabs-retirement-proof.unit.test.ts
tests/e2e/cross-version-wire/agent-session-unproven-release-downgrade.unit.test.ts
+ tests/e2e/cross-version-wire/cross-version-worktree-ps-verdict.unit.test.ts
managed_hook_node18:
name: managed hooks on Node 18
@@ -812,7 +826,7 @@ jobs:
package:
name: package
- needs: [code_paths]
+ needs: [code_paths, static_analysis, typecheck]
if: needs.code_paths.outputs.package == 'true'
runs-on: ubuntu-latest
# Let the serial Docker gates reach their own deadlines and report cleanup failures.
@@ -960,7 +974,7 @@ jobs:
package_windows:
name: package (windows)
- needs: [code_paths]
+ needs: [code_paths, static_analysis, typecheck]
if: needs.code_paths.outputs.package_windows == 'true'
runs-on: windows-2022
timeout-minutes: 30
diff --git a/.github/workflows/pullfrog.yml b/.github/workflows/pullfrog.yml
index d5030252f88..2fc81c887bb 100644
--- a/.github/workflows/pullfrog.yml
+++ b/.github/workflows/pullfrog.yml
@@ -1,6 +1,6 @@
-# PULLFROG ACTION — DO NOT EDIT EXCEPT WHERE INDICATED
+# Explicit review identities share workflow concurrency; legacy names use ordered cancellation.
name: Pullfrog
-run-name: ${{ inputs.name || github.workflow }}
+run-name: ${{ inputs.name || github.workflow }}${{ inputs.pull_request_number && format(' | PR {0}', inputs.pull_request_number) || '' }}
on:
workflow_dispatch:
inputs:
@@ -11,21 +11,71 @@ on:
type: string
description: Run name
+ pull_request_number:
+ type: string
+ description: Optional PR identity for cancelling superseded reviews
+ head_sha:
+ type: string
+ description: Optional expected PR head; stale reviews are skipped
+
permissions:
contents: read
+ pull-requests: read
+
+concurrency:
+ group: ${{ inputs.pull_request_number && format('pullfrog-pr-{0}', inputs.pull_request_number) || format('pullfrog-run-{0}', github.run_id) }}
+ cancel-in-progress: true
jobs:
+ review_scope:
+ runs-on: ubuntu-slim
+ permissions:
+ contents: read
+ pull-requests: read
+ actions: write
+ outputs:
+ current: ${{ steps.scope.outputs.current }}
+ number: ${{ steps.scope.outputs.number }}
+ head: ${{ steps.scope.outputs.head }}
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ sparse-checkout: config/scripts/pullfrog-review-scope.cjs
+ sparse-checkout-cone-mode: false
+ persist-credentials: false
+ - uses: actions/github-script@v8
+ id: scope
+ with:
+ script: |
+ const { reviewScope } = require('./config/scripts/pullfrog-review-scope.cjs')
+ await reviewScope({ github, context, core })
+
pullfrog:
+ needs: review_scope
+ if: needs.review_scope.outputs.current == 'true'
runs-on: ubuntu-latest
permissions:
id-token: write
contents: read
+ pull-requests: read
steps:
- name: Checkout code
uses: actions/checkout@v6
with:
fetch-depth: 1
+ - name: Recheck review head before starting agent
+ id: freshness
+ if: needs.review_scope.outputs.number != ''
+ uses: actions/github-script@v8
+ env:
+ REVIEW_NUMBER: ${{ needs.review_scope.outputs.number }}
+ REVIEW_HEAD: ${{ needs.review_scope.outputs.head }}
+ with:
+ script: |
+ const { data: pr } = await github.rest.pulls.get({ ...context.repo, pull_number: Number(process.env.REVIEW_NUMBER) })
+ core.setOutput('current', pr.state === 'open' && pr.head.sha === process.env.REVIEW_HEAD)
- name: Run agent
+ if: needs.review_scope.outputs.number == '' || steps.freshness.outputs.current == 'true'
uses: pullfrog/pullfrog@v0
with:
prompt: ${{ inputs.prompt }}
diff --git a/.github/workflows/unit-tests.yml b/.github/workflows/unit-tests.yml
index ccb44c0faa6..2691f4907bc 100644
--- a/.github/workflows/unit-tests.yml
+++ b/.github/workflows/unit-tests.yml
@@ -14,19 +14,48 @@ on:
default: ubuntu-latest
type: string
+ selection_mode:
+ description: Shadow validates selection; selected applies it only to draft PRs.
+ required: false
+ default: shadow
+ type: string
+
permissions:
contents: read
jobs:
+ plan:
+ runs-on: ubuntu-latest
+ timeout-minutes: 5
+ outputs:
+ shards: ${{ steps.plan.outputs.shards }}
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ fetch-depth: 2
+ persist-credentials: false
+ - uses: ./.github/actions/install-node-dependencies
+ - name: Plan unit selection
+ id: plan
+ env:
+ ORCA_UNIT_SELECTION_MODE: ${{ inputs.selection_mode }}
+ run: node config/scripts/ci-unit-plan.mjs
+ - uses: actions/upload-artifact@v7
+ continue-on-error: true
+ with:
+ name: unit-selection-attempt-${{ github.run_attempt }}
+ path: ci-shards/unit-selection.json
+ retention-days: 14
+
test:
- name: tests node ${{ matrix.node }} ${{ matrix.shard }}/${{ matrix.shard_total }}
+ needs: plan
+ name: tests node ${{ matrix.node }} ${{ matrix.shard.index }}/${{ matrix.shard.count }}
runs-on: ${{ inputs.runner }}
strategy:
fail-fast: false
matrix:
node: ${{ fromJSON(inputs.node_versions) }}
- shard: [1, 2, 3, 4, 5, 6, 7, 8]
- shard_total: [8]
+ shard: ${{ fromJSON(needs.plan.outputs.shards) }}
steps:
- name: Checkout
@@ -43,6 +72,12 @@ jobs:
- name: Install Electron package binary for tests
run: node config/scripts/install-electron-package-binary.mjs
+ - uses: actions/download-artifact@v8
+ continue-on-error: true
+ with:
+ name: unit-selection-attempt-${{ github.run_attempt }}
+ path: ci-shards/
+
- name: Test shard
env:
ORCA_BALANCE_UNIT_SHARDS: '1'
@@ -50,27 +85,7 @@ jobs:
run: |
export ORCA_SHARD_SOURCE_SHA="$(git rev-parse HEAD)"
pnpm exec vitest run --config config/vitest.config.ts \
- --exclude=src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts \
- --exclude=src/main/daemon/shell-ready.test.ts \
- --exclude=src/main/daemon/node-pty-fd-leak.test.ts \
- --exclude=src/main/providers/local-pty-shell-ready-zsh-launch-environment.test.ts \
- --exclude=src/main/providers/__tests__/shell-ready-framework-example.test.ts \
- --exclude=src/main/pty/omp-shell-wrapper-alias-safety.test.ts \
- --exclude=src/main/pty/omp-shell-wrapper.node-pty.test.ts \
- --exclude=src/main/shell-startup-feature-channel.test.ts \
- --exclude=src/main/terminal-history-fish-session.node-pty.test.ts \
- --exclude=src/main/zsh-scoped-histfile.live-shell.test.ts \
- --exclude=src/main/zsh-startup-hook-user-config-equivalence.live-shell.test.ts \
- --exclude=src/main/zsh-wrapper-version-mismatch.live-shell.test.ts \
- --exclude=src/renderer/src/components/terminal-pane/fish-color-scheme-child-stdin.node-pty.test.ts \
- --exclude=src/shared/fish-query-reply-child-stdin.node-pty.test.ts \
- --exclude=src/shared/pty-reply-echo-shapes.node-pty.test.ts \
- --exclude=src/shared/startup-shell-portability.live-shell.test.ts \
- --exclude=src/shared/posix-command-path-lookup.test.ts \
- --exclude=tests/e2e/relay-region-compatibility.unit.test.ts \
- --exclude=tests/e2e/relay-region-correction.unit.test.ts \
- --exclude=tests/e2e/cross-version-wire/** \
- --shard=${{ matrix.shard }}/${{ matrix.shard_total }}
+ --shard=${{ matrix.shard.index }}/${{ matrix.shard.count }}
- name: Upload unit shard assignment
if: always()
@@ -78,11 +93,40 @@ jobs:
continue-on-error: true
uses: actions/upload-artifact@v7
with:
- name: unit-shard-node-${{ matrix.node }}-${{ matrix.shard }}-attempt-${{ github.run_attempt }}
+ name: unit-shard-node-${{ matrix.node }}-${{ matrix.shard.index }}-attempt-${{ github.run_attempt }}
path: ci-shards/
retention-days: 14
if-no-files-found: warn
+ selection_evidence:
+ needs: test
+ if: ${{ !cancelled() }}
+ continue-on-error: true
+ runs-on: ubuntu-slim
+ steps:
+ - uses: actions/checkout@v6
+ with:
+ sparse-checkout: config/scripts/ci-unit-selection-review.mjs
+ sparse-checkout-cone-mode: false
+ persist-credentials: false
+ - uses: actions/setup-node@v6
+ with:
+ node-version: '24'
+ - uses: actions/download-artifact@v8
+ with:
+ pattern: unit-shard-node-*-attempt-${{ github.run_attempt }}
+ path: unit-evidence/
+ - name: Compare selection with full results
+ continue-on-error: true
+ run: node config/scripts/ci-unit-selection-review.mjs unit-evidence
+ - uses: actions/upload-artifact@v7
+ if: always()
+ continue-on-error: true
+ with:
+ name: unit-selection-review-attempt-${{ github.run_attempt }}
+ path: unit-evidence/selection-review.json
+ retention-days: 30
+
relay_integration:
name: relay integration node ${{ matrix.node }}
strategy:
diff --git a/README.md b/README.md
index 98efc301c2f..708d3311bb2 100644
--- a/README.md
+++ b/README.md
@@ -179,6 +179,7 @@ Works with **any CLI agent** — if it runs in a terminal, it runs in Orca.
Cursor
GitHub Copilot
Muse
+
DeepSeek Harness
ZCode
OpenCode
MiMo Code
diff --git a/config/e2e-failure-tracking.json b/config/e2e-failure-tracking.json
new file mode 100644
index 00000000000..fe51488c706
--- /dev/null
+++ b/config/e2e-failure-tracking.json
@@ -0,0 +1 @@
+[]
diff --git a/config/scripts/anti-slop-shards.test.mjs b/config/scripts/anti-slop-shards.test.mjs
new file mode 100644
index 00000000000..9f2aa9db161
--- /dev/null
+++ b/config/scripts/anti-slop-shards.test.mjs
@@ -0,0 +1,71 @@
+import { describe, expect, it } from 'vitest'
+import { buildUnits, packShards, planShards } from './run-anti-slop-shards.mjs'
+
+// Why these properties: the sharded pass equals a single pass only if every file is linted by
+// exactly one shard. Coverage and disjointness are what carry that, so they are asserted
+// directly rather than by diffing two multi-minute lint runs.
+function filesUnder(unit, files) {
+ return files.filter((file) => file === unit || file.startsWith(`${unit}/`))
+}
+
+function coveredBy(units, files) {
+ return files.filter((file) => units.some((unit) => file === unit || file.startsWith(`${unit}/`)))
+}
+
+const SAMPLE = [
+ 'src/renderer/src/components/a.tsx',
+ 'src/renderer/src/components/b.tsx',
+ 'src/renderer/src/hooks/c.ts',
+ 'src/renderer/index.ts',
+ 'src/main/agent/d.ts',
+ 'src/main/agent/e.ts',
+ 'src/main/f.ts',
+ 'src/shared/g.ts',
+ 'config/scripts/h.mjs',
+ 'tests/e2e/i.spec.ts',
+ 'mobile/src/j.tsx',
+ 'mobile/src/k.tsx'
+]
+
+describe('anti-slop shard planning', () => {
+ it('covers every file exactly once across units', () => {
+ const units = buildUnits(SAMPLE, 3).map((entry) => entry.unit)
+ for (const file of SAMPLE) {
+ const owners = units.filter((unit) => file === unit || file.startsWith(`${unit}/`))
+ expect(owners, `${file} owned by ${JSON.stringify(owners)}`).toHaveLength(1)
+ }
+ })
+
+ it('reports a unit count that matches the files it owns', () => {
+ for (const { unit, count } of buildUnits(SAMPLE, 3)) {
+ expect(count).toBe(filesUnder(unit, SAMPLE).length)
+ }
+ })
+
+ it('splits a directory larger than the target instead of leaving it whole', () => {
+ // src/renderer holds 4 of 12 sample files through a single child directory, so the
+ // splitter has to descend more than one level to get under a small target.
+ const units = buildUnits(SAMPLE, 2).map((entry) => entry.unit)
+ expect(units).not.toContain('src')
+ expect(units.some((unit) => unit.startsWith('src/renderer/'))).toBe(true)
+ })
+
+ it('assigns every unit to exactly one shard and keeps shards disjoint', () => {
+ const { units, bins } = planShards(SAMPLE, 3)
+ const assigned = bins.flatMap((bin) => bin.units)
+ expect(assigned.slice().sort()).toEqual(units.map((entry) => entry.unit).sort())
+ expect(new Set(assigned).size).toBe(assigned.length)
+ expect(coveredBy(assigned, SAMPLE)).toHaveLength(SAMPLE.length)
+ })
+
+ it('keeps the heaviest shard near the mean so wall time is not bound by one shard', () => {
+ const files = Array.from({ length: 400 }, (_, index) => `src/pkg${index % 40}/file${index}.ts`)
+ const { bins } = planShards(files, 4)
+ const heaviest = Math.max(...bins.map((bin) => bin.count))
+ expect(heaviest).toBeLessThanOrEqual(Math.ceil(files.length / 4) * 1.35)
+ })
+
+ it('never emits more shards than there are units', () => {
+ expect(packShards(buildUnits(['src/a.ts'], 1), 4)).toHaveLength(1)
+ })
+})
diff --git a/config/scripts/bun-profile-change-scope.mjs b/config/scripts/bun-profile-change-scope.mjs
index c30c4a282ac..7b8b47b69aa 100644
--- a/config/scripts/bun-profile-change-scope.mjs
+++ b/config/scripts/bun-profile-change-scope.mjs
@@ -8,6 +8,7 @@ import {
ORCAD_ENTRY_POINT
} from './orcad-entry-build.mjs'
import { bunProfileTestPaths } from './bun-profile-test-paths.mjs'
+import { bunProfileQualification } from './bun-profile-qualification.mjs'
const ROOT = resolve(import.meta.dirname, '../..')
const BUILD_SCRIPTS = [
@@ -29,7 +30,9 @@ const ALWAYS_FILES = new Set([
'tsconfig.json',
'.github/workflows/bun-profile-tests.yml',
'config/scripts/bun-profile-change-scope.mjs',
- 'config/scripts/bun-profile-change-scope.test.mjs'
+ 'config/scripts/bun-profile-change-scope.test.mjs',
+ 'config/scripts/bun-profile-qualification.mjs',
+ 'config/scripts/bun-profile-qualification.test.mjs'
])
const ALWAYS_PREFIXES = [
'.github/actions/install-node-dependencies/',
@@ -110,7 +113,11 @@ export async function classifyBunProfileChanges(changedFiles, collect = collectB
reason: matched ? `Runtime or test dependency changed: ${matched}` : 'No Bun inputs changed'
}
} catch (error) {
- return { shouldRun: true, reason: `Dependency graph unavailable: ${String(error)}` }
+ return {
+ shouldRun: true,
+ graphUnavailable: true,
+ reason: `Dependency graph unavailable: ${String(error)}`
+ }
}
}
@@ -118,7 +125,14 @@ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href)
const changedFiles = readFileSync(process.argv[2], 'utf8').split('\0').filter(Boolean)
const result = await classifyBunProfileChanges(changedFiles)
console.log(result.reason)
- const output = `should_run=${String(result.shouldRun)}\n`
+ let event = {}
+ try {
+ event = JSON.parse(readFileSync(process.env.GITHUB_EVENT_PATH, 'utf8'))
+ } catch {
+ // Missing event evidence retains full qualification.
+ }
+ const policy = bunProfileQualification(changedFiles, result, event)
+ const output = `should_run=${result.shouldRun}\nqualification=${policy.qualification}\nrunners=${JSON.stringify(policy.runners)}\n`
if (process.env.GITHUB_OUTPUT) {
appendFileSync(process.env.GITHUB_OUTPUT, output)
} else {
diff --git a/config/scripts/bun-profile-change-scope.test.mjs b/config/scripts/bun-profile-change-scope.test.mjs
index 4b6d1647ffc..e813be93621 100644
--- a/config/scripts/bun-profile-change-scope.test.mjs
+++ b/config/scripts/bun-profile-change-scope.test.mjs
@@ -78,6 +78,7 @@ it.each([
'.npmrc',
'tsconfig.json',
'config/tsconfig.node.json',
+ 'config/scripts/bun-profile-qualification.mjs',
'config/patches/node-pty@1.1.0.patch',
'native/windows-registry/src/addon.cc',
'.github/actions/install-node-dependencies/action.yml',
@@ -140,12 +141,15 @@ it('keeps all ten platform jobs and runs them when detection is skipped or fails
expect(workflow.jobs.changes.steps[0].with['persist-credentials']).toBe(false)
const detect = workflow.jobs.changes.steps.find((step) => step.id === 'scope')
expect(detect.run).toContain('git diff --name-only --no-renames -z HEAD^1 HEAD')
- let count = 0
- for (const jobName of ['persistence', 'linux_glibc_floor', 'linux_musl']) {
+ expect(workflow.on.pull_request.types).toContain('ready_for_review')
+ expect(workflow.on.schedule).toHaveLength(1)
+ expect(workflow.jobs.persistence.strategy.matrix.os).toContain('needs.changes.outputs.runners')
+ for (const jobName of ['linux_glibc_floor', 'linux_musl']) {
const job = workflow.jobs[jobName]
- expect(job.needs).toBe('changes')
- expect(job.if).toBe("${{ !cancelled() && needs.changes.outputs.should_run != 'false' }}")
- count += job.strategy.matrix.os.length
+ expect(job.needs).toEqual(['changes', 'persistence'])
+ expect(job.if).toContain("needs.persistence.result == 'success'")
+ expect(job.if).toContain("needs.changes.outputs.qualification != 'false'")
+ expect(job.if).toContain("needs.changes.outputs.should_run != 'false'")
+ expect(job.strategy.matrix.os).toEqual(['ubuntu-22.04', 'ubuntu-24.04-arm'])
}
- expect(count).toBe(10)
})
diff --git a/config/scripts/bun-profile-qualification.mjs b/config/scripts/bun-profile-qualification.mjs
new file mode 100644
index 00000000000..799f459ff85
--- /dev/null
+++ b/config/scripts/bun-profile-qualification.mjs
@@ -0,0 +1,42 @@
+export const BUN_PERSISTENCE_RUNNERS = [
+ 'ubuntu-22.04',
+ 'ubuntu-24.04-arm',
+ 'macos-14',
+ 'macos-15-intel',
+ 'windows-2022',
+ 'windows-11-arm'
+]
+
+const QUALIFICATION_PREFIXES = [
+ 'config/',
+ 'native/',
+ 'resources/',
+ '.github/',
+ 'src/main/persistence/',
+ 'src/main/sqlite/',
+ 'src/main/orcad/',
+ 'src/main/providers/',
+ 'src/main/daemon/',
+ 'src/main/ssh/',
+ 'src/relay/',
+ 'src/shared/child-process/'
+]
+
+export function bunProfileQualification(changedFiles, scope, event = {}) {
+ const sensitive = changedFiles.some(
+ (file) =>
+ !file.includes('/') ||
+ QUALIFICATION_PREFIXES.some((prefix) => file.startsWith(prefix)) ||
+ /(?:^|[/.-])(?:windows|win32|wsl|macos|darwin|linux|posix|bun)(?:[/.-]|$)/i.test(file)
+ )
+ // Only a proven unrelated platform change in a draft may defer qualification.
+ const full =
+ event.pull_request?.draft !== true ||
+ changedFiles.length === 0 ||
+ scope.graphUnavailable === true ||
+ sensitive
+ return {
+ qualification: full,
+ runners: full ? BUN_PERSISTENCE_RUNNERS : ['ubuntu-22.04']
+ }
+}
diff --git a/config/scripts/bun-profile-qualification.test.mjs b/config/scripts/bun-profile-qualification.test.mjs
new file mode 100644
index 00000000000..9ebc5e437e1
--- /dev/null
+++ b/config/scripts/bun-profile-qualification.test.mjs
@@ -0,0 +1,47 @@
+import { expect, it } from 'vitest'
+import { BUN_PERSISTENCE_RUNNERS, bunProfileQualification } from './bun-profile-qualification.mjs'
+
+const draft = { pull_request: { draft: true } }
+const scope = { shouldRun: true }
+
+it('defers only platform qualification for ordinary draft runtime changes', () => {
+ expect(
+ bunProfileQualification(['src/main/runtime/rpc/methods/example.ts'], scope, draft)
+ ).toEqual({
+ qualification: false,
+ runners: ['ubuntu-22.04']
+ })
+})
+
+it.each([
+ 'package.json',
+ 'native/windows-registry/src/addon.cc',
+ 'config/vitest.config.ts',
+ 'src/main/ssh/ssh-provider.ts',
+ 'src/main/providers/local-pty-provider.ts',
+ 'src/shared/child-process/run-process.ts',
+ 'src/main/persistence/profile-state/store.ts',
+ 'src/main/sqlite/database.ts',
+ 'src/main/orcad/entry.ts',
+ 'src/main/runtime/windows-terminal.ts',
+ 'src/shared/linux-glibc.ts',
+ 'src/main/daemon/entry.ts',
+ 'src/relay/index.ts',
+ 'src/main/wsl/runner.ts'
+])('retains all platforms for sensitive input %s', (file) => {
+ expect(bunProfileQualification([file], scope, draft)).toEqual({
+ qualification: true,
+ runners: BUN_PERSISTENCE_RUNNERS
+ })
+})
+
+it('qualifies every ready commit, scheduled/manual runs and incomplete evidence', () => {
+ const paths = ['src/main/runtime/rpc/methods/example.ts']
+ for (const event of [{}, { pull_request: { draft: false } }]) {
+ expect(bunProfileQualification(paths, scope, event).qualification).toBe(true)
+ }
+ expect(bunProfileQualification([], scope, draft).qualification).toBe(true)
+ expect(
+ bunProfileQualification(paths, { ...scope, graphUnavailable: true }, draft).qualification
+ ).toBe(true)
+})
diff --git a/config/scripts/ci-cache-warmup-workflow.test.mjs b/config/scripts/ci-cache-warmup-workflow.test.mjs
index 803563f61c8..e913f780422 100644
--- a/config/scripts/ci-cache-warmup-workflow.test.mjs
+++ b/config/scripts/ci-cache-warmup-workflow.test.mjs
@@ -1,6 +1,7 @@
import { readFileSync } from 'node:fs'
import { expect, it } from 'vitest'
import { parse } from 'yaml'
+import { BUN_PERSISTENCE_RUNNERS } from './bun-profile-qualification.mjs'
const readWorkflow = (name) =>
parse(readFileSync(new URL(`../../.github/workflows/${name}.yml`, import.meta.url), 'utf8'))
@@ -50,9 +51,8 @@ it('bounds warming to the required platforms and validates changes without grant
it('warms and probes both Windows images with the persistence job runtime', () => {
const job = workflow.jobs['warm-windows']
- const persistence = readWorkflow('bun-profile-tests').jobs.persistence
expect(job.strategy.matrix.os).toEqual(
- persistence.strategy.matrix.os.filter((os) => os.startsWith('windows-'))
+ BUN_PERSISTENCE_RUNNERS.filter((os) => os.startsWith('windows-'))
)
expect(job['runs-on']).toBe('${{ matrix.os }}')
expect(job.strategy['fail-fast']).toBe(false)
diff --git a/config/scripts/ci-e2e-failure-summary.mjs b/config/scripts/ci-e2e-failure-summary.mjs
new file mode 100644
index 00000000000..f95fed20dac
--- /dev/null
+++ b/config/scripts/ci-e2e-failure-summary.mjs
@@ -0,0 +1,92 @@
+import { appendFileSync, readFileSync } from 'node:fs'
+import { pathToFileURL } from 'node:url'
+import { trackE2eFailures } from './ci-e2e-failure-tracking.mjs'
+
+export function e2eFailureSummary(report) {
+ const failures = []
+ let passed = 0
+ let skipped = 0
+ function visit(suite, parents = [], parentFile) {
+ const titles = [...parents, suite.title].filter(Boolean)
+ const file = suite.file ?? parentFile
+ for (const spec of suite.specs ?? []) {
+ for (const test of spec.tests ?? []) {
+ if (test.status === 'expected') {
+ passed++
+ continue
+ }
+ if (test.status === 'skipped') {
+ skipped++
+ continue
+ }
+ const errors = test.results.flatMap((result) => result.errors ?? [])
+ failures.push({
+ file: spec.file ?? file,
+ title: [...titles, spec.title].join(' › '),
+ project: test.projectName,
+ status: test.status,
+ message: errors
+ .map((error) => error.message ?? error.value ?? '')
+ .join('\n')
+ .slice(0, 2000)
+ })
+ }
+ }
+ for (const child of suite.suites ?? []) {
+ visit(child, titles, file)
+ }
+ }
+ for (const suite of report.suites ?? []) {
+ visit(suite)
+ }
+ return { passed, skipped, failures, errors: report.errors ?? [] }
+}
+
+export function renderE2eFailures(summary, records = [], now = new Date()) {
+ const escape = (value) =>
+ String(value ?? '')
+ .replaceAll('&', '&')
+ .replaceAll('<', '<')
+ .replaceAll('>', '>')
+ const tracked = trackE2eFailures(summary.failures, records, now)
+ const details = (failure) =>
+ `
${escape(failure.status)}: ${escape(failure.file)} — ${escape(failure.title)}
${escape(failure.message)}
${escape(error.message ?? error.value)}`),
+ '',
+ 'Failures retain their original verdict. Repeated failures need a tracked owner, reproduction and review date; do not treat a red baseline as passing.',
+ ''
+ ].join('\n')
+}
+
+if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
+ try {
+ const summary = e2eFailureSummary(JSON.parse(readFileSync(process.argv[2], 'utf8')))
+ const records = JSON.parse(
+ readFileSync(new URL('../e2e-failure-tracking.json', import.meta.url), 'utf8')
+ )
+ appendFileSync(process.env.GITHUB_STEP_SUMMARY, renderE2eFailures(summary, records))
+ } catch (error) {
+ appendFileSync(
+ process.env.GITHUB_STEP_SUMMARY,
+ `E2E report unavailable (${error.code ?? 'invalid report'}); inspect the failing step and traces.\n`
+ )
+ }
+}
diff --git a/config/scripts/ci-e2e-failure-summary.test.mjs b/config/scripts/ci-e2e-failure-summary.test.mjs
new file mode 100644
index 00000000000..5594e4faa56
--- /dev/null
+++ b/config/scripts/ci-e2e-failure-summary.test.mjs
@@ -0,0 +1,79 @@
+import { expect, it } from 'vitest'
+import { e2eFailureSummary, renderE2eFailures } from './ci-e2e-failure-summary.mjs'
+import { trackE2eFailures } from './ci-e2e-failure-tracking.mjs'
+
+it('requires exact failures with an owner, issue and unexpired review date', () => {
+ const failure = {
+ file: 'test.spec.ts',
+ title: 'case',
+ project: 'electron',
+ message: 'expected focus failed'
+ }
+ const record = {
+ ...failure,
+ message: 'expected focus',
+ owner: '@owner',
+ issue: 'https://github.com/stablyai/orca/issues/123',
+ expires: '2026-10-01'
+ }
+ const now = new Date('2026-09-28')
+ expect(trackE2eFailures([failure], [null, record], now)).toMatchObject({
+ known: [{ ...failure, tracking: record }],
+ invalid: [null]
+ })
+ expect(trackE2eFailures([failure], null, now).untracked).toEqual([failure])
+ for (const change of [
+ { expires: '2026-09-27' },
+ { expires: '2026-09-31' },
+ { owner: '' },
+ { issue: '' },
+ { title: 'different' },
+ { message: 'different' },
+ { project: 'other' }
+ ]) {
+ expect(trackE2eFailures([failure], [{ ...record, ...change }], now).untracked).toEqual([
+ failure
+ ])
+ }
+})
+
+it('keeps failure, flaky, skipped and startup-error evidence separate', () => {
+ const report = {
+ errors: [{ message: 'startup failed' }],
+ suites: [
+ {
+ title: 'file',
+ file: 'tests/e2e/test.spec.ts',
+ suites: [
+ {
+ title: 'feature',
+ specs: [
+ {
+ title: 'case',
+ tests: [
+ { status: 'expected' },
+ { status: 'skipped' },
+ {
+ status: 'unexpected',
+ projectName: 'electron-headless',
+ results: [{ errors: [{ message: ') {
+ const facts = new TerminalRunFactsRegister()
+ const stops = new TerminalIntentionalStops()
+ stops.mark(PTY_ID, 'reversible', null)(true)
+ const runtime = {
+ terminalRunFacts: facts,
+ intentionalPtyStops: stops,
+ // Why the real method: the case under test is what the runtime does with each commit.
+ noteTerminalSpawnCommit: OrcaRuntimeWithRuntimeId.prototype.noteTerminalSpawnCommit,
+ registerPreAllocatedHandleForPty: vi.fn(),
+ registerPty: vi.fn(),
+ cancelPendingPtyRegistration: vi.fn(),
+ reflowHeadlessTerminalToPtyGrid: vi.fn(),
+ seedHeadlessTerminal: vi.fn(),
+ noteTerminalSpawnCommand: vi.fn()
+ }
+ const ports = {
+ runtime,
+ store: { persistPtyBinding },
+ options: {},
+ sendPtySpawnedToRenderer: vi.fn()
+ }
+ // oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: the commit reads only the ports above, and the store only for its binding save.
+ const deps = ports as unknown as PtySpawnIpcDeps
+ const ctx = createPtyIpcSpawnState(deps, {
+ worktreeId: 'wt-1',
+ tabId: 'tab-1',
+ leafId: 'leaf-1',
+ cols: 120,
+ rows: 40
+ })
+ ctx.validatedLeafId = 'leaf-1'
+ ctx.provider = { ...ctx.provider, shutdown: vi.fn().mockResolvedValue(undefined) }
+ ctx.result = { id: PTY_ID, incarnationId: INCARNATION_ID }
+ const outcome = await commitPtyIpcSpawn(ctx).then(
+ () => 'committed',
+ () => 'discarded'
+ )
+ return { outcome, facts, stops }
+}
+
+describe('renderer spawn commit: run facts', () => {
+ afterEach(() => {
+ ptySizes.delete(PTY_ID)
+ ptyOwnership.delete(PTY_ID)
+ ptyIncarnationById.delete(PTY_ID)
+ })
+
+ it('records a committed spawn and lets it supersede a landed stop no exit pinned', async () => {
+ const { outcome, facts, stops } = await commit(vi.fn().mockResolvedValue(true))
+
+ expect(outcome).toBe('committed')
+ expect(facts.read(PTY_ID, INCARNATION_ID).freshSpawn).toBe(true)
+ expect(stops.claimExit(PTY_ID, INCARNATION_ID)).toEqual([])
+ })
+
+ it('records nothing for a spawn discarded because its binding save failed', async () => {
+ const persistPtyBinding = vi.fn().mockRejectedValue(new Error('disk full'))
+ const { outcome, facts, stops } = await commit(persistPtyBinding)
+
+ expect(outcome).toBe('discarded')
+ expect(persistPtyBinding).toHaveBeenCalledOnce()
+ expect(facts.read(PTY_ID, INCARNATION_ID).freshSpawn).toBe(false)
+ expect(stops.claimExit(PTY_ID, INCARNATION_ID)).toEqual(['reversible'])
+ })
+})
diff --git a/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts b/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts
index ceeb5127b7d..4afd2b008f1 100644
--- a/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts
+++ b/src/main/ipc/pty/ipc/write-input-chunk-yield.test.ts
@@ -68,7 +68,7 @@ describe('chunked pty write yield', () => {
}) as typeof setImmediate)
const outcome = await Promise.race([
- createWriteInput()({ id: PTY_ID, data: THREE_CHUNK_INPUT }),
+ createWriteInput()({ inputKind: 'driving', id: PTY_ID, data: THREE_CHUNK_INPUT }),
afterImmediateTurns(50)
])
@@ -88,6 +88,7 @@ describe('chunked pty write yield', () => {
const immediate = vi.spyOn(globalThis, 'setImmediate')
const outcome = createWriteInput()({
+ inputKind: 'driving',
id: PTY_ID,
data: 'x'.repeat(TERMINAL_INPUT_CHUNK_MAX_BYTES)
})
diff --git a/src/main/ipc/pty/ipc/write-input-user-input.test.ts b/src/main/ipc/pty/ipc/write-input-user-input.test.ts
new file mode 100644
index 00000000000..d4f20f00a68
--- /dev/null
+++ b/src/main/ipc/pty/ipc/write-input-user-input.test.ts
@@ -0,0 +1,63 @@
+import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
+import { TerminalRunFactsRegister } from '../../../runtime/terminal-run-facts'
+import { ptyOwnership } from '../provider/ownership-state'
+import { createPtyWriteInput } from './write-input'
+
+const PTY_ID = 'pty-user-input'
+
+const { provider } = vi.hoisted(() => ({
+ provider: { write: vi.fn(), hasPty: vi.fn(() => true) }
+}))
+
+vi.mock('../provider/registry', () => ({
+ tryGetProviderForPty: (id: string) => (id === PTY_ID ? provider : undefined)
+}))
+
+function createWriteInput(facts: TerminalRunFactsRegister) {
+ const runtime = { getDriver: () => ({ kind: 'desktop' }), terminalRunFacts: facts }
+ const mainWindow = { isDestroyed: () => false, webContents: { send: vi.fn() } }
+ // oxlint-disable-next-line typescript/consistent-type-assertions -- SAFETY: write input reads only getDriver and terminalRunFacts from the runtime, and isDestroyed/webContents from the window.
+ return createPtyWriteInput({ mainWindow: mainWindow as never, runtime: runtime as never })
+}
+
+beforeEach(() => {
+ ptyOwnership.set(PTY_ID, null)
+ provider.write.mockReset()
+})
+
+afterEach(() => {
+ ptyOwnership.delete(PTY_ID)
+})
+
+describe('renderer PTY writes: input kind', () => {
+ it.each(['writePtyInput', 'writePtyInputAccepted'] as const)(
+ '%s records driving input before the provider write',
+ async (writer) => {
+ const facts = new TerminalRunFactsRegister()
+ facts.recordSpawnCommit({ id: PTY_ID, incarnationId: 'inc-1' })
+ const recordedAtWrite: (number | null)[] = []
+ provider.write.mockImplementation(() => {
+ recordedAtWrite.push(facts.read(PTY_ID, 'inc-1').firstUserInputAt)
+ })
+
+ await createWriteInput(facts)[writer]({ id: PTY_ID, data: 'exit\r', inputKind: 'driving' })
+
+ expect(recordedAtWrite).toEqual([expect.any(Number)])
+ }
+ )
+
+ it.each([
+ ['a launch write', 'launch', 'echo startup\r'],
+ ['a query reply', 'query-reply', '\x1b[3;4R'],
+ ['a driving write that is only a reply', 'driving', '\x1b[3;4R'],
+ ['a driving write that is only focus reports', 'driving', '\x1b[I\x1b[O']
+ ] as const)('records nothing for %s', async (_label, inputKind, data) => {
+ const facts = new TerminalRunFactsRegister()
+ facts.recordSpawnCommit({ id: PTY_ID, incarnationId: 'inc-1' })
+
+ await createWriteInput(facts).writePtyInput({ id: PTY_ID, data, inputKind })
+
+ expect(provider.write).toHaveBeenCalledOnce()
+ expect(facts.read(PTY_ID, 'inc-1').firstUserInputAt).toBeNull()
+ })
+})
diff --git a/src/main/ipc/pty/ipc/write-input.ts b/src/main/ipc/pty/ipc/write-input.ts
index 77b27ef16f0..2f8ea72cd9b 100644
--- a/src/main/ipc/pty/ipc/write-input.ts
+++ b/src/main/ipc/pty/ipc/write-input.ts
@@ -9,6 +9,7 @@ import {
} from '../../../../shared/terminal-input'
import { ptyOwnership } from '../provider/ownership-state'
import { tryGetProviderForPty } from '../provider/registry'
+import type { TerminalInputKind } from '../../../../shared/terminal-input-kind'
import { interactiveOutputCharsByPty, lastInputAtByPty } from '../delivery/visibility-state'
export function isMainWindowPtyIpcEvent(
@@ -25,7 +26,7 @@ export function isMainWindowPtyIpcEvent(
)
}
-export type PtyWritePayload = { id: string; data: string }
+export type PtyWritePayload = { id: string; data: string; inputKind: TerminalInputKind }
export type PtyViewportClaimPayload = { id: string; cols: number; rows: number }
export function createPtyWriteInput(deps: {
@@ -148,6 +149,12 @@ export function createPtyWriteInput(deps: {
const isPtyWriteEventFromMainWindow = (event: IpcMainEvent | IpcMainInvokeEvent): boolean =>
isMainWindowPtyIpcEvent(event, mainWindow)
+ const noteRendererPtyInput = (args: PtyWritePayload): void => {
+ lastInputAtByPty.set(args.id, performance.now())
+ interactiveOutputCharsByPty.set(args.id, 0)
+ runtime?.terminalRunFacts?.recordInput(args.id, args.inputKind, args.data)
+ }
+
const writePtyInput = (args: PtyWritePayload): boolean | Promise