diff --git a/.gitattributes b/.gitattributes index 736d59473f6..1b447a9189e 100644 --- a/.gitattributes +++ b/.gitattributes @@ -31,6 +31,13 @@ # the reviewable change, and pin LF because they are compared byte-for-byte. # Not -diff: the shell diff is the review surface when a wrapper does change. /src/main/__fixtures__/shell-wrapper-snapshots/*.txt linguist-generated=true text eol=lf +# Captured agent PTY transcripts. -text, not `text eol=lf` like the wrapper snapshots above: +# these carry real CR and CRLF bytes as the terminal emitted them, and line-ending +# normalisation on a Windows checkout would rewrite the evidence the fixture exists to be. +/src/main/runtime/__fixtures__/*.txt -text # Generated runtime English subset: compared byte-for-byte by # verify:localization-runtime-catalog, so a CRLF checkout would fail the gate. /src/renderer/src/i18n/en-runtime-required.json linguist-generated=true text eol=lf +# Generated method->params catalog: compared byte-for-byte by +# verify:rpc-params-catalog, so a CRLF checkout would fail the gate. +/src/shared/rpc-contract/rpc-params-catalog.generated.ts linguist-generated=true text eol=lf diff --git a/.github/workflows/cloud-push-deploy.yml b/.github/workflows/cloud-push-deploy.yml index 6014681372d..1b909d42f63 100644 --- a/.github/workflows/cloud-push-deploy.yml +++ b/.github/workflows/cloud-push-deploy.yml @@ -106,7 +106,7 @@ jobs: set -euo pipefail docker run --rm --network none --entrypoint node "${IMAGE}" --input-type=module -e ' import { loadPushConfig } from "./apps/push/dist/config.js"; - const env = { ORCA_PUSH_PUBLIC_URL: "https://push.onorca.dev", ORCA_PUSH_MODE: "validation" }; + const env = { ORCA_PUSH_PUBLIC_URL: "https://push.onorca.dev", ORCA_PUSH_MODE: "validation", ORCA_PUSH_FCM_PROJECT_ID: "onorca-cloud" }; if (loadPushConfig(env).mode !== "validation") throw new Error("validation_mode_unsupported"); let rejected = false; try { loadPushConfig({ ...env, ORCA_PUSH_MODE: "invalid" }); } catch { rejected = true; } diff --git a/.github/workflows/cloud-verify.yml b/.github/workflows/cloud-verify.yml index e2ba9407ac4..5e24cae76cc 100644 --- a/.github/workflows/cloud-verify.yml +++ b/.github/workflows/cloud-verify.yml @@ -90,6 +90,7 @@ jobs: --health-timeout 5s --health-retries 10 env: + ORCA_PUSH_TEST_DATABASE_URL: postgres://relay_test:relay_test@127.0.0.1:5432/orca_relay_test ORCA_RELAY_TEST_POSTGRES_URL: postgres://relay_test:relay_test@127.0.0.1:5432/orca_relay_test steps: - uses: actions/checkout@v4 diff --git a/.github/workflows/pi-owner-runtime.yml b/.github/workflows/pi-owner-runtime.yml new file mode 100644 index 00000000000..373afb7a539 --- /dev/null +++ b/.github/workflows/pi-owner-runtime.yml @@ -0,0 +1,29 @@ +name: Pi owner runtime verification +on: + pull_request: + paths: + - 'src/main/pi/agent-status-handler-source.ts' + - 'tests/tools/pi-owner-runtime-smoke.mjs' + - '.github/workflows/pi-owner-runtime.yml' + workflow_dispatch: +permissions: + contents: read +jobs: + runtime: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + runs-on: ${{ matrix.os }} + timeout-minutes: 20 + env: + ORCA_BACKGROUND_LAUNCH: '1' + steps: + - uses: actions/checkout@v6 + with: + persist-credentials: false + - uses: ./.github/actions/install-node-dependencies + - name: Install pinned extension loader + run: npm install --prefix .cache/pi-owner --ignore-scripts --no-audit --no-fund @earendil-works/pi-coding-agent@0.83.0 + - name: Verify real owner exit and hook delivery + run: node tests/tools/pi-owner-runtime-smoke.mjs .cache/pi-owner/node_modules/@earendil-works/pi-coding-agent diff --git a/.github/workflows/pi-provider-runtime.yml b/.github/workflows/pi-provider-runtime.yml new file mode 100644 index 00000000000..837c38baf9a --- /dev/null +++ b/.github/workflows/pi-provider-runtime.yml @@ -0,0 +1,28 @@ +name: Pi extension provider verification +on: + pull_request: + paths: + - 'src/shared/commit-message-agent-specs-primary.ts' + - 'tests/tools/pi-provider-runtime-smoke.mjs' + - '.github/workflows/pi-provider-runtime.yml' +permissions: + contents: read +jobs: + runtime: + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, macos-latest, windows-latest] + runs-on: ${{ matrix.os }} + timeout-minutes: 20 + env: + ORCA_BACKGROUND_LAUNCH: '1' + steps: + - uses: actions/checkout@v6 + with: + persist-credentials: false + - uses: ./.github/actions/install-node-dependencies + - name: Install pinned Pi runtime + run: npm install --prefix .cache/pi-provider --ignore-scripts --no-audit --no-fund @earendil-works/pi-coding-agent@0.84.2 + - name: Verify extension model generation before and after + run: node tests/tools/pi-provider-runtime-smoke.mjs .cache/pi-provider/node_modules/@earendil-works/pi-coding-agent/dist/cli.js diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index 268b6ad66e3..c49091ab148 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -203,6 +203,9 @@ jobs: - name: Boot orcad and round-trip a terminal run: pnpm run smoke:orcad-terminal + - name: Verify the generated RPC params catalog + run: pnpm run verify:rpc-params-catalog + - name: Verify bundled skill guides run: pnpm run verify:bundled-skill-guides diff --git a/.gitignore b/.gitignore index 6722fc5ae54..e5207a25015 100644 --- a/.gitignore +++ b/.gitignore @@ -103,6 +103,9 @@ docs/** !docs/agent-skill-sharing-implementation-checklist.md !docs/mobile-terminal-shortcut-bar.md !docs/reference/ +!docs/reference/agent-pty-transcript-capture.md +!docs/reference/agent-status-store.md +!docs/reference/antigravity-readiness-evidence.md !docs/reference/git-compatibility.md !docs/reference/headless-linux-server.md !docs/reference/ime-regression-checklist.md diff --git a/.oxlintrc.json b/.oxlintrc.json index 03cc659f494..55758560478 100644 --- a/.oxlintrc.json +++ b/.oxlintrc.json @@ -180,6 +180,7 @@ } ], "ignorePatterns": [ + "src/shared/rpc-contract/rpc-params-catalog.generated.ts", "**/node_modules", "**/dist", "**/out", diff --git a/AGENTS.md b/AGENTS.md index b0947da0c2f..f1ce31e404b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -68,6 +68,14 @@ All changes must consider the SSH use case. Don't assume local-only execution. B All changes must consider folder workspaces as well as git worktrees. Don't assume every workspace is a git worktree. +## Agent Status + +The execution host owns agent status in one store, the hook server's, and every reader (sidebar, `worktree ps`, mobile, dashboard) subscribes to it. Before adding a producer, a cache, or a reader-side precedence rule, read [`docs/reference/agent-status-store.md`](./docs/reference/agent-status-store.md): new producers write into that store, and readers keep only presentation policy. + +## Agent Terminal Screens + +A rule that reads what an agent CLI paints on a terminal — readiness, blocked prompts, idle — must be written against a captured transcript, not a remembered screen. Record one with [`docs/reference/agent-pty-transcript-capture.md`](./docs/reference/agent-pty-transcript-capture.md), which keeps escapes and wrapping intact and scrubs account identifiers before they reach git. Antigravity readiness has no transcript yet and five failed attempts without one; before touching it, read [`docs/reference/antigravity-readiness-evidence.md`](./docs/reference/antigravity-readiness-evidence.md). + ## Remote Wire Compatibility Clients and remote Orca servers update independently, so mixed versions are the normal state. Before changing anything a paired client and host exchange — RPC params, stream frames, or the content either side publishes over them — follow [`docs/reference/remote-wire-compatibility.md`](./docs/reference/remote-wire-compatibility.md). A new optional field is safe; a new stream opcode must be capability-negotiated because decoders drop unknown opcodes silently; and changing what the host publishes reaches old clients even with no wire change. diff --git a/cloud/README.md b/cloud/README.md index 8ffcd9fa6b3..6ca96f4cc99 100644 --- a/cloud/README.md +++ b/cloud/README.md @@ -24,6 +24,40 @@ the repository's root [MIT license](../LICENSE). - `apps/relay-ops`: the relay operations console and the incident monitor behind `pnpm ops:relay`, `pnpm incident:relay`, and `pnpm incident:relay-preflight`. +- `apps/push` and `packages/push-contract`: the mobile push gateway that holds + the APNs key and sends to phones through APNs and FCM, and its wire contract. + It is deployed and operated from here but is not part of the relay data path; + see [docs/push-gateway.md](docs/push-gateway.md). + +## Mobile push gateway + +`apps/push` is a separate Cloud Run service from the relay. Phones never hold an +Orca credential for it: the desktop host authenticates with the same X25519 +key it uses for the relay, answering an encrypted challenge to mint a 24 hour +session, then registers each paired phone's native push token and asks the +gateway to push. The gateway queues each event as its own notification, +enforces per-host quotas and request limits, and retires a +registration as soon as Apple or Google reports the token unregistered. +Provider push is the only ordinary mobile OS-banner path. The notification +socket is retained only for live dismissal and reconnect tray reconciliation; +it never creates or recovers banners. Desktop notification categories remain +authoritative. +Each delivery is persisted as one notification event. Before deploying an +incompatible queue format, stop all older push gateway revisions and clear only +unpublished push delivery fixtures; no queue preservation or migration is required. +FCM notification messages are inherently collapsible while offline and have a +small concurrent collapse-key budget, so every pending alert is not guaranteed. + +Storage follows the relay pattern: PostgreSQL in production, SQLite for tests +and local development. Configure it with `ORCA_PUSH_PUBLIC_URL`, `ORCA_PUSH_FCM_PROJECT_ID`, +`ORCA_PUSH_DATABASE_URL`, the three APNs variables (`ORCA_PUSH_APNS_KEY`, +`ORCA_PUSH_APNS_KEY_ID`, `ORCA_PUSH_APPLE_TEAM_ID`, all three or none), and +optionally `ORCA_PUSH_APNS_TOPIC`. The FCM credential comes from +the runtime service account, so no key material is configured for Android. See +[push gateway operations](docs/push-gateway.md) for deployment and recovery. + +Logging is aggregate counters only. Tokens, notification titles, notification +bodies, and full host fingerprints never reach a log line. ## Infrastructure and operations @@ -38,16 +72,18 @@ the repository's root [MIT license](../LICENSE). - `dev/contracts` and `dev/fixtures`: the checked-in data those contract tests read, including the Terraform root partition. - `docs/`: the relay runbooks, capacity-testing guide, incident-monitor - reference, and the workflow variable reference in `docs/relay-workflows.md`. + reference, the workflow variable reference in `docs/relay-workflows.md`, and + the push gateway runbook in `docs/push-gateway.md`. ## Workflows -The 24 `.github/workflows/cloud-*.yml` workflows are the relay's deploy and -operate surface: publish and deploy the director, roll GCE cell capacity, -operate Asia admission and regional rehoming, prove staging capacity, monitor -production, and power staging up and down. `.github/actions/cloud-sql-rollout-lease` -is the compare-and-swap lease that serializes every rollout against the shared -Cloud SQL instance. +The 25 `.github/workflows/cloud-*.yml` workflows are the deploy and operate +surface: publish and deploy the director, roll GCE cell capacity, operate Asia +admission and regional rehoming, prove staging capacity, monitor production, +power staging up and down, and deploy the mobile push gateway. +`.github/actions/cloud-sql-rollout-lease` is the compare-and-swap lease that +serializes rollouts against the shared Cloud SQL instance. Push reuses that +action with its own lease object and deployment concurrency group. Every one of them is inert. Each top-level job is gated on `vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true'`, a repository variable that is diff --git a/cloud/apps/push/Dockerfile b/cloud/apps/push/Dockerfile new file mode 100644 index 00000000000..efdc85fc404 --- /dev/null +++ b/cloud/apps/push/Dockerfile @@ -0,0 +1,29 @@ +FROM node:24-alpine AS build +WORKDIR /app +RUN corepack enable +COPY package.json pnpm-lock.yaml pnpm-workspace.yaml tsconfig.base.json ./ +COPY packages/push-contract/package.json packages/push-contract/package.json +COPY packages/postgres-schema/package.json packages/postgres-schema/package.json +COPY apps/push/package.json apps/push/package.json +RUN pnpm install --frozen-lockfile +COPY packages/push-contract packages/push-contract +COPY apps/push apps/push +COPY packages/postgres-schema packages/postgres-schema +RUN pnpm --filter @orca-cloud/postgres-schema build && pnpm --filter @orca-cloud/push-contract build && pnpm --filter @orca-cloud/push build + +FROM node:24-alpine AS runtime +ENV NODE_ENV=production +ENV PORT=8080 +WORKDIR /app +RUN corepack enable +COPY package.json pnpm-lock.yaml pnpm-workspace.yaml ./ +COPY packages/push-contract/package.json packages/push-contract/package.json +COPY packages/postgres-schema/package.json packages/postgres-schema/package.json +COPY apps/push/package.json apps/push/package.json +COPY --from=build /app/packages/push-contract/dist packages/push-contract/dist +COPY --from=build /app/packages/postgres-schema/dist packages/postgres-schema/dist +COPY --from=build /app/apps/push/dist apps/push/dist +RUN pnpm install --prod --frozen-lockfile --filter @orca-cloud/push... +USER node +EXPOSE 8080 +CMD ["node", "apps/push/dist/index.js"] diff --git a/cloud/apps/push/package.json b/cloud/apps/push/package.json new file mode 100644 index 00000000000..1f27749024c --- /dev/null +++ b/cloud/apps/push/package.json @@ -0,0 +1,35 @@ +{ + "name": "@orca-cloud/push", + "private": true, + "version": "0.0.0", + "type": "module", + "main": "dist/index.js", + "scripts": { + "build": "pnpm clean && tsc -p tsconfig.build.json", + "clean": "node -e \"require('fs').rmSync('dist', { recursive: true, force: true })\"", + "dev": "tsx watch src/index.ts", + "lint": "tsc -p tsconfig.json --noEmit", + "pretest": "pnpm --filter @orca-cloud/postgres-schema build && pnpm --filter @orca-cloud/push-contract build", + "start": "node dist/index.js", + "test": "vitest run", + "typecheck": "tsc -p tsconfig.json --noEmit" + }, + "dependencies": { + "@hono/node-server": "^1.19.17", + "@orca-cloud/postgres-schema": "workspace:*", + "@orca-cloud/push-contract": "workspace:*", + "google-auth-library": "^10.5.0", + "hono": "^4.13.7", + "pg": "^8.22.0", + "pg-connection-string": "2.14.0", + "tweetnacl": "^1.0.3", + "zod": "^3.25.76" + }, + "devDependencies": { + "@types/node": "^24.10.0", + "@types/pg": "^8.20.0", + "tsx": "^4.21.0", + "typescript": "^5.9.3", + "vitest": "^4.0.8" + } +} diff --git a/cloud/apps/push/src/apns-authentication-token.ts b/cloud/apps/push/src/apns-authentication-token.ts new file mode 100644 index 00000000000..34def16e86e --- /dev/null +++ b/cloud/apps/push/src/apns-authentication-token.ts @@ -0,0 +1,42 @@ +import { createPrivateKey, type KeyObject, sign } from 'node:crypto' +import type { ApnsCredentials } from './config.js' + +// Apple rejects a provider token older than an hour and throttles reissue +// under about 20 minutes, so 50 minutes is the safe rotation point. +export const APNS_TOKEN_ROTATION_MS = 50 * 60 * 1000 + +function base64UrlJson(value: Record): string { + return Buffer.from(JSON.stringify(value), 'utf8').toString('base64url') +} + +export class ApnsAuthenticationToken { + private readonly privateKey: KeyObject + private cached: { token: string; issuedAtMs: number } | null = null + + constructor( + private readonly credentials: ApnsCredentials, + private readonly now: () => number = Date.now, + private readonly rotationMs: number = APNS_TOKEN_ROTATION_MS + ) { + this.privateKey = createPrivateKey(credentials.keyPem) + } + + value(): string { + const nowMs = this.now() + if (this.cached && nowMs - this.cached.issuedAtMs < this.rotationMs) return this.cached.token + const header = base64UrlJson({ alg: 'ES256', kid: this.credentials.keyId }) + const payload = base64UrlJson({ + iss: this.credentials.teamId, + iat: Math.floor(nowMs / 1000) + }) + const signingInput = `${header}.${payload}` + // ES256 requires the raw r||s pair; Node emits DER unless asked otherwise. + const signature = sign('sha256', Buffer.from(signingInput, 'utf8'), { + key: this.privateKey, + dsaEncoding: 'ieee-p1363' + }).toString('base64url') + const token = `${signingInput}.${signature}` + this.cached = { token, issuedAtMs: nowMs } + return token + } +} diff --git a/cloud/apps/push/src/apns-client.test.ts b/cloud/apps/push/src/apns-client.test.ts new file mode 100644 index 00000000000..e4af2bb1ecb --- /dev/null +++ b/cloud/apps/push/src/apns-client.test.ts @@ -0,0 +1,220 @@ +import { generateKeyPairSync } from 'node:crypto' +import { describe, expect, it } from 'vitest' +import { ApnsAuthenticationToken, APNS_TOKEN_ROTATION_MS } from './apns-authentication-token.js' +import { ApnsClient } from './apns-client.js' +import type { ApnsRequest, ApnsResponse } from './apns-http2-transport.js' +import type { ApnsCredentials } from './config.js' +import { buildPushDelivery } from './push-delivery-message.js' + +const HOST = 'abcdefghijklmnop' + +function credentials(): ApnsCredentials { + const { privateKey } = generateKeyPairSync('ec', { + namedCurve: 'P-256', + privateKeyEncoding: { type: 'pkcs8', format: 'pem' }, + publicKeyEncoding: { type: 'spki', format: 'pem' } + }) + return { keyPem: privateKey, keyId: 'ABCDE12345', teamId: 'TEAM123456' } +} + +function delivery(now = Date.now()) { + return buildPushDelivery({ + expiresAt: now + 300_000, + registrationId: 'reg-1', + hostFingerprint: HOST, + notification: { + notificationId: 'note-1', + notificationSeq: 7, + notificationEpoch: 'epoch-1', + source: 'agent-task-complete', + agentState: 'needs-input', + title: 'Agent needs input', + body: 'Waiting on your answer', + worktreeId: 'wt-1' + } + }) +} + +function fakeTransport(response: ApnsResponse) { + const requests: ApnsRequest[] = [] + return { + requests, + transport: async (request: ApnsRequest): Promise => { + requests.push(request) + return response + } + } +} + +describe('apns authentication token', () => { + it('signs an ES256 provider token and caches it until the rotation point', () => { + let clock = 1_700_000_000_000 + const authentication = new ApnsAuthenticationToken(credentials(), () => clock) + const first = authentication.value() + const [header, payload, signature] = first.split('.') + expect(JSON.parse(Buffer.from(header!, 'base64url').toString('utf8'))).toEqual({ + alg: 'ES256', + kid: 'ABCDE12345' + }) + expect(JSON.parse(Buffer.from(payload!, 'base64url').toString('utf8'))).toEqual({ + iss: 'TEAM123456', + iat: Math.floor(clock / 1000) + }) + expect(Buffer.from(signature!, 'base64url').byteLength).toBe(64) + + clock += APNS_TOKEN_ROTATION_MS - 1 + expect(authentication.value()).toBe(first) + clock += 1 + expect(authentication.value()).not.toBe(first) + }) +}) + +describe('apns client', () => { + it('sends the specified headers, path, and alert body', async () => { + const clock = 1_700_000_000_000 + const fake = fakeTransport({ status: 200, body: '' }) + const client = new ApnsClient({ + topic: 'com.stably.orca.mobile', + credentials: credentials(), + transport: fake.transport, + now: () => clock + }) + await expect( + client.send(delivery(clock), { token: 'a'.repeat(64), apnsEnvironment: 'production' }) + ).resolves.toEqual({ status: 'sent' }) + const request = fake.requests[0]! + expect(request.host).toBe('api.push.apple.com') + expect(request.path).toBe(`/3/device/${'a'.repeat(64)}`) + expect(request.headers).toMatchObject({ + 'apns-topic': 'com.stably.orca.mobile', + 'apns-push-type': 'alert', + 'apns-priority': '10', + 'apns-expiration': String(Math.floor(clock / 1000) + 5 * 60), + 'apns-collapse-id': expect.stringMatching(/^[a-f0-9]{64}$/) + }) + expect(request.headers.authorization).toMatch(/^bearer /) + expect(JSON.parse(request.body)).toEqual({ + aps: { + alert: { title: 'Agent needs input', body: 'Waiting on your answer' }, + sound: 'default', + 'thread-id': HOST + }, + orca: { + hostFingerprint: HOST, + worktreeId: 'wt-1', + notificationId: 'note-1', + notificationSeq: 7, + notificationEpoch: 'epoch-1', + source: 'agent-task-complete', + agentState: 'needs-input' + } + }) + }) + + it('targets the sandbox host and keeps the individual collapse id', async () => { + const fake = fakeTransport({ status: 200, body: '' }) + const client = new ApnsClient({ + topic: 'com.stably.orca.mobile', + credentials: credentials(), + transport: fake.transport + }) + await client.send(delivery(), { token: 'b'.repeat(64), apnsEnvironment: 'sandbox' }) + expect(fake.requests[0]?.host).toBe('api.sandbox.push.apple.com') + expect(fake.requests[0]?.headers['apns-collapse-id']).toMatch(/^[a-f0-9]{64}$/) + }) + + it.each([ + [410, 'Unregistered'], + [400, 'BadDeviceToken'], + [400, 'Unregistered'] + ])('classifies %i %s as a dead token', async (status, reason) => { + const fake = fakeTransport({ status, body: JSON.stringify({ reason }) }) + const client = new ApnsClient({ + topic: 'com.stably.orca.mobile', + credentials: credentials(), + transport: fake.transport + }) + await expect( + client.send(delivery(), { token: 'a'.repeat(64), apnsEnvironment: 'production' }) + ).resolves.toEqual({ status: 'dead', reason }) + }) + + it.each([ + [400, 'PayloadTooLarge'], + [400, 'DeviceTokenNotForTopic'], + [429, 'TooManyRequests'], + [500, 'InternalServerError'] + ])('treats %i %s with the appropriate retry policy', async (status, reason) => { + const fake = fakeTransport({ status, body: JSON.stringify({ reason }) }) + const client = new ApnsClient({ + topic: 'com.stably.orca.mobile', + credentials: credentials(), + transport: fake.transport + }) + await expect( + client.send(delivery(), { token: 'a'.repeat(64), apnsEnvironment: 'production' }) + ).resolves.toEqual({ status: 'error', reason, retryable: status === 429 || status >= 500 }) + }) + + it('reports a transport failure as an error rather than throwing', async () => { + const client = new ApnsClient({ + topic: 'com.stably.orca.mobile', + credentials: credentials(), + transport: async () => { + throw new Error('socket hang up') + } + }) + await expect( + client.send(delivery(), { token: 'a'.repeat(64), apnsEnvironment: 'production' }) + ).resolves.toEqual({ status: 'error', reason: 'Error', retryable: true }) + }) +}) + +it('does not collapse background dismissals with visible alerts', async () => { + const fake = fakeTransport({ status: 200, body: '' }) + const apns = new ApnsClient({ + topic: 'test', + credentials: credentials(), + transport: fake.transport + }) + const alert = delivery() + await apns.send( + { ...alert, orca: { ...alert.orca, kind: 'dismiss' } }, + { + token: 'test', + apnsEnvironment: 'sandbox' + } + ) + expect(fake.requests[0]?.headers).not.toHaveProperty('apns-collapse-id') + expect(fake.requests[0]?.headers).toMatchObject({ + 'apns-push-type': 'background', + 'apns-priority': '5' + }) + expect(JSON.parse(fake.requests[0]!.body).aps).toEqual({ 'content-available': 1 }) +}) + +it('keeps the absolute deadline across retries and refuses expired delivery', async () => { + let now = 1_700_000_000_000 + const fake = fakeTransport({ status: 503, body: '{}' }) + const client = new ApnsClient({ + topic: 'test', + credentials: credentials(), + now: () => now, + transport: fake.transport + }) + const pending = delivery(now) + const device = { token: 'test', apnsEnvironment: 'sandbox' as const } + await client.send(pending, device) + now += 60_000 + await client.send(pending, device) + expect(fake.requests.map((request) => request.headers['apns-expiration'])).toEqual([ + String(pending.expiresAt / 1000), + String(pending.expiresAt / 1000) + ]) + now = pending.expiresAt + await expect(client.send(pending, device)).resolves.toEqual({ + status: 'error', + reason: 'expired' + }) + expect(fake.requests).toHaveLength(2) +}) diff --git a/cloud/apps/push/src/apns-client.ts b/cloud/apps/push/src/apns-client.ts new file mode 100644 index 00000000000..827efdfea79 --- /dev/null +++ b/cloud/apps/push/src/apns-client.ts @@ -0,0 +1,95 @@ +import type { ApnsEnvironment } from '@orca-cloud/push-contract' +import { ApnsAuthenticationToken } from './apns-authentication-token.js' +import type { ApnsTransport } from './apns-http2-transport.js' +import type { ApnsCredentials } from './config.js' +import type { PushDelivery } from './push-delivery-message.js' +import type { PushProviderOutcome } from './push-provider-outcome.js' + +const APNS_HOSTS: Record = { + production: 'api.push.apple.com', + sandbox: 'api.sandbox.push.apple.com' +} + +const DEAD_TOKEN_REASONS = new Set(['BadDeviceToken', 'Unregistered']) + +export type ApnsClientOptions = { + topic: string + credentials: ApnsCredentials + transport: ApnsTransport + now?: () => number +} + +function readReason(body: string): string { + try { + const parsed = JSON.parse(body) as { reason?: unknown } + return typeof parsed.reason === 'string' ? parsed.reason : 'unknown' + } catch { + return 'unparseable' + } +} + +export function apnsBody(delivery: PushDelivery): string { + return JSON.stringify({ + aps: + delivery.orca.kind === 'dismiss' + ? { 'content-available': 1 } + : { + alert: { title: delivery.title, body: delivery.body }, + ...(delivery.sound === false ? {} : { sound: 'default' }), + 'thread-id': delivery.hostFingerprint + }, + orca: delivery.orca + }) +} + +export class ApnsClient { + private readonly authentication: ApnsAuthenticationToken + private readonly now: () => number + + constructor(private readonly options: ApnsClientOptions) { + this.now = options.now ?? Date.now + this.authentication = new ApnsAuthenticationToken(options.credentials, this.now) + } + + async send( + delivery: PushDelivery, + device: { token: string; apnsEnvironment: ApnsEnvironment } + ): Promise { + const expiration = Math.floor(delivery.expiresAt / 1000) + if (expiration * 1000 <= this.now()) return { status: 'error', reason: 'expired' } + let response + try { + response = await this.options.transport({ + host: APNS_HOSTS[device.apnsEnvironment], + path: `/3/device/${device.token}`, + headers: { + authorization: `bearer ${this.authentication.value()}`, + 'apns-topic': this.options.topic, + 'apns-push-type': delivery.orca.kind === 'dismiss' ? 'background' : 'alert', + 'apns-priority': delivery.orca.kind === 'dismiss' ? '5' : '10', + 'apns-expiration': String(expiration), + ...(delivery.orca.kind === 'dismiss' ? {} : { 'apns-collapse-id': delivery.collapseId }) + }, + body: apnsBody(delivery) + }) + } catch (error) { + return { + status: 'error', + reason: error instanceof Error ? error.name : 'transport_failed', + retryable: true + } + } + if (response.status === 200) return { status: 'sent' } + const reason = readReason(response.body) + if (response.status === 410) return { status: 'dead', reason } + if (response.status === 400 && DEAD_TOKEN_REASONS.has(reason)) { + return { status: 'dead', reason } + } + return { + status: 'error', + reason, + retryable: response.status === 429 || response.status >= 500, + ...(response.retryAfterMs === undefined ? {} : { retryAfterMs: response.retryAfterMs }) + } + } +} diff --git a/cloud/apps/push/src/apns-http2-transport.ts b/cloud/apps/push/src/apns-http2-transport.ts new file mode 100644 index 00000000000..167b4d14e38 --- /dev/null +++ b/cloud/apps/push/src/apns-http2-transport.ts @@ -0,0 +1,50 @@ +import { connect, constants, type ClientHttp2Session } from 'node:http2' +import { readApnsStreamResponse, type ApnsResponse } from './apns-stream-response.js' + +export type ApnsRequest = { + host: string + path: string + headers: Record + body: string +} + +export type { ApnsResponse } +export type ApnsTransport = (request: ApnsRequest) => Promise + +// APNs requires HTTP/2 and rewards a long-lived session per host, so sessions +// are cached and only dropped when the socket itself goes away. +export function createApnsHttp2Transport(): ApnsTransport & { close(): void } { + const sessions = new Map() + + const sessionFor = (host: string): ClientHttp2Session => { + const existing = sessions.get(host) + if (existing && !existing.closed && !existing.destroyed) return existing + const session = connect(`https://${host}`) + const forget = (): void => { + if (sessions.get(host) === session) sessions.delete(host) + } + session.on('error', forget) + session.on('close', forget) + sessions.set(host, session) + return session + } + + const transport = async (request: ApnsRequest): Promise => { + const stream = sessionFor(request.host).request({ + ...request.headers, + [constants.HTTP2_HEADER_METHOD]: 'POST', + [constants.HTTP2_HEADER_PATH]: request.path, + [constants.HTTP2_HEADER_AUTHORITY]: request.host, + 'content-type': 'application/json', + 'content-length': String(Buffer.byteLength(request.body)) + }) + return await readApnsStreamResponse(stream, request.body) + } + + return Object.assign(transport, { + close(): void { + for (const session of sessions.values()) session.close() + sessions.clear() + } + }) +} diff --git a/cloud/apps/push/src/apns-session-replacement.test.ts b/cloud/apps/push/src/apns-session-replacement.test.ts new file mode 100644 index 00000000000..2678732ca94 --- /dev/null +++ b/cloud/apps/push/src/apns-session-replacement.test.ts @@ -0,0 +1,45 @@ +import { EventEmitter } from 'node:events' +import { expect, it, vi } from 'vitest' +const mocks = vi.hoisted(() => ({ + connect: vi.fn(), + read: vi.fn(async () => ({ status: 200, body: '' })) +})) +vi.mock('node:http2', async (original) => ({ + ...(await original()), + connect: mocks.connect +})) +vi.mock('./apns-stream-response.js', () => ({ readApnsStreamResponse: mocks.read })) +import { createApnsHttp2Transport } from './apns-http2-transport.js' + +it('keeps the replacement cached when the draining session closes later', async () => { + const sessions: Array< + EventEmitter & { + closed: boolean + destroyed: boolean + request: ReturnType + close: ReturnType + } + > = [] + mocks.connect.mockImplementation(() => { + const session = Object.assign(new EventEmitter(), { + closed: false, + destroyed: false, + request: vi.fn(() => ({})), + close: vi.fn() + }) + sessions.push(session) + return session + }) + const transport = createApnsHttp2Transport() + const request = { host: 'api.push.apple.com', path: '/synthetic', headers: {}, body: '{}' } + await transport(request) + sessions[0]!.closed = true + await transport(request) + sessions[0]!.emit('close') + sessions[0]!.emit('error', new Error('old-session')) + await transport(request) + expect(sessions).toHaveLength(2) + expect(sessions[1]!.request).toHaveBeenCalledTimes(2) + transport.close() + expect(sessions[1]!.close).toHaveBeenCalledOnce() +}) diff --git a/cloud/apps/push/src/apns-stream-response.test.ts b/cloud/apps/push/src/apns-stream-response.test.ts new file mode 100644 index 00000000000..c87b9031ca1 --- /dev/null +++ b/cloud/apps/push/src/apns-stream-response.test.ts @@ -0,0 +1,82 @@ +import { EventEmitter } from 'node:events' +import { describe, expect, it } from 'vitest' +import { readApnsStreamResponse, type ApnsResponseStream } from './apns-stream-response.js' + +type FakeStream = ApnsResponseStream & { + sentBody: string | null + destroyedWith: Error | null + fireTimeout(): void +} + +function fakeApnsStream(): FakeStream { + const emitter = new EventEmitter() as FakeStream + emitter.sentBody = null + emitter.destroyedWith = null + let onTimeout: (() => void) | null = null + emitter.setTimeout = (_ms, callback) => { + onTimeout = callback + } + emitter.destroy = (error?: Error) => { + emitter.destroyedWith = error ?? null + if (error) emitter.emit('error', error) + } + emitter.end = (body: string) => { + emitter.sentBody = body + } + emitter.fireTimeout = () => onTimeout?.() + return emitter +} + +describe('apns stream response', () => { + it('resolves with the status and the concatenated body', async () => { + const stream = fakeApnsStream() + const pending = readApnsStreamResponse(stream, '{"aps":{}}') + expect(stream.sentBody).toBe('{"aps":{}}') + stream.emit('response', { ':status': '200' }) + stream.emit('data', Buffer.from('{"re')) + stream.emit('data', Buffer.from('ason":"ok"}')) + stream.emit('end') + await expect(pending).resolves.toEqual({ status: 200, body: '{"reason":"ok"}' }) + }) + + it('rejects when the peer resets the stream without an end or an error', async () => { + const stream = fakeApnsStream() + const pending = readApnsStreamResponse(stream, 'body') + stream.emit('response', { ':status': '200' }) + // NGHTTP2_NO_ERROR: node emits only 'close', so nothing else would settle. + stream.emit('close') + await expect(pending).rejects.toThrow('apns_stream_closed') + }) + + it('keeps the resolved response when close follows a completed end', async () => { + const stream = fakeApnsStream() + const pending = readApnsStreamResponse(stream, 'body') + stream.emit('response', { ':status': '410' }) + stream.emit('end') + stream.emit('close') + await expect(pending).resolves.toEqual({ status: 410, body: '' }) + }) + + it('keeps the original error when close follows a stream error', async () => { + const stream = fakeApnsStream() + const pending = readApnsStreamResponse(stream, 'body') + stream.emit('error', new Error('socket_hang_up')) + stream.emit('close') + await expect(pending).rejects.toThrow('socket_hang_up') + }) + + it('destroys the stream on timeout and surfaces the timeout error', async () => { + const stream = fakeApnsStream() + const pending = readApnsStreamResponse(stream, 'body', 10) + stream.fireTimeout() + await expect(pending).rejects.toThrow('apns_timeout') + expect(stream.destroyedWith?.message).toBe('apns_timeout') + }) + + it('reports a missing status header as zero rather than NaN', async () => { + const stream = fakeApnsStream() + const pending = readApnsStreamResponse(stream, 'body') + stream.emit('end') + await expect(pending).resolves.toEqual({ status: 0, body: '' }) + }) +}) diff --git a/cloud/apps/push/src/apns-stream-response.ts b/cloud/apps/push/src/apns-stream-response.ts new file mode 100644 index 00000000000..9fa189304e8 --- /dev/null +++ b/cloud/apps/push/src/apns-stream-response.ts @@ -0,0 +1,53 @@ +import type { EventEmitter } from 'node:events' +import { providerRetryAfter } from './provider-retry-delay.js' +import { constants } from 'node:http2' + +export type ApnsResponse = { status: number; body: string; retryAfterMs?: number } + +// The subset of ClientHttp2Stream this module drives, so a fake emitter can +// stand in for a real APNs stream in tests. +export type ApnsResponseStream = EventEmitter & { + setTimeout(ms: number, callback: () => void): void + destroy(error?: Error): void + end(body: string): void +} + +export const APNS_REQUEST_TIMEOUT_MS = 10_000 + +export function readApnsStreamResponse( + stream: ApnsResponseStream, + body: string, + timeoutMs = APNS_REQUEST_TIMEOUT_MS +): Promise { + return new Promise((resolve, reject) => { + let settled = false + const settle = (run: () => void): void => { + if (settled) return + settled = true + run() + } + let status = 0 + let retryAfterMs: number | undefined + const chunks: Buffer[] = [] + stream.setTimeout(timeoutMs, () => stream.destroy(new Error('apns_timeout'))) + stream.on('response', (headers: Record) => { + status = Number(headers[constants.HTTP2_HEADER_STATUS] ?? 0) + retryAfterMs = providerRetryAfter(String(headers['retry-after'] ?? '')) + }) + stream.on('data', (chunk: Buffer) => chunks.push(chunk)) + stream.on('error', (error: Error) => settle(() => reject(error))) + stream.on('end', () => + settle(() => + resolve({ + status, + body: Buffer.concat(chunks).toString('utf8'), + ...(retryAfterMs === undefined ? {} : { retryAfterMs }) + }) + ) + ) + // A peer reset with NGHTTP2_NO_ERROR emits neither 'end' nor 'error', which + // would leave the worker's delivery pending for the life of the process. + stream.on('close', () => settle(() => reject(new Error('apns_stream_closed')))) + stream.end(body) + }) +} diff --git a/cloud/apps/push/src/apns-topic-recovery.test.ts b/cloud/apps/push/src/apns-topic-recovery.test.ts new file mode 100644 index 00000000000..583d4536b5a --- /dev/null +++ b/cloud/apps/push/src/apns-topic-recovery.test.ts @@ -0,0 +1,48 @@ +import { expect, it } from 'vitest' +import { createPushHostKeypair } from './host-challenge-answering.test-fixture.js' +import { + APNS_TOKEN, + createPushServerHarness, + notification +} from './push-server-harness.test-fixture.js' + +it('delivers again after a topic error without re-registering the phone', async () => { + const h = await createPushServerHarness() + try { + const token = await h.signIn(createPushHostKeypair(71)) + const registered = await h.post( + '/v1/devices', + { + v: 1, + deviceId: 'phone', + platform: 'ios', + token: APNS_TOKEN, + apnsEnvironment: 'sandbox' + }, + token + ) + const { registrationId } = (await registered.json()) as { registrationId: string } + h.setApnsResponse({ status: 400, body: JSON.stringify({ reason: 'DeviceTokenNotForTopic' }) }) + for (const seq of [1, 2]) { + const sent = await h.post( + '/v1/send', + { + v: 1, + registrationIds: [registrationId], + notification: notification({ notificationId: `topic-${seq}`, notificationSeq: seq }) + }, + token + ) + expect(await sent.json()).toEqual({ results: [{ registrationId, status: 'queued' }] }) + await h.flushDeliveries() + if (seq === 1) { + expect(h.server.observability.consume().delivery_error).toBe(1) + h.setApnsResponse({ status: 200, body: '' }) + } + } + expect(h.apnsRequests).toHaveLength(2) + expect(h.server.observability.consume().delivery_sent).toBe(1) + } finally { + await h.close() + } +}) diff --git a/cloud/apps/push/src/canonical-base64.ts b/cloud/apps/push/src/canonical-base64.ts new file mode 100644 index 00000000000..e13ea982cb6 --- /dev/null +++ b/cloud/apps/push/src/canonical-base64.ts @@ -0,0 +1,9 @@ +// Rejects the many base64 spellings of the same bytes: a non-canonical +// encoding would change the transcript the host signs without changing the key. +export function decodeCanonicalBase64(value: string, expectedBytes: number): Buffer | null { + if (!/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(value)) return null + const decoded = Buffer.from(value, 'base64') + return decoded.byteLength === expectedBytes && decoded.toString('base64') === value + ? decoded + : null +} diff --git a/cloud/apps/push/src/client-ip-rate-limit.test.ts b/cloud/apps/push/src/client-ip-rate-limit.test.ts new file mode 100644 index 00000000000..85a88f7d1cd --- /dev/null +++ b/cloud/apps/push/src/client-ip-rate-limit.test.ts @@ -0,0 +1,169 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { Hono } from 'hono' +import { describe, expect, it, vi } from 'vitest' +import { ClientIpRateLimiter, clientIpRateLimit } from './client-ip-rate-limit.js' + +const CAPACITY = PUSH_LIMITS.unauthenticatedRequestsPerMinutePerIp + +function limiterApp(limiter: ClientIpRateLimiter, trustedProxyHops = 0): Hono { + const app = new Hono() + app.post('/probe', clientIpRateLimit(limiter, { trustedProxyHops }), (context) => + context.json({ ok: true }) + ) + return app +} + +describe('client ip rate limiter', () => { + it('admits exactly the per-minute allowance and refuses the next request', () => { + const limiter = new ClientIpRateLimiter({ now: () => 1_000 }) + for (let index = 0; index < CAPACITY; index++) { + expect(limiter.allow('203.0.113.7')).toBe(true) + } + expect(limiter.allow('203.0.113.7')).toBe(false) + }) + + it('keeps one client ip from spending another one budget', () => { + const limiter = new ClientIpRateLimiter({ now: () => 1_000 }) + for (let index = 0; index < CAPACITY; index++) limiter.allow('203.0.113.7') + expect(limiter.allow('203.0.113.7')).toBe(false) + expect(limiter.allow('198.51.100.9')).toBe(true) + }) + + it('refills over the window rather than resetting on a boundary', () => { + let clock = 1_000 + const limiter = new ClientIpRateLimiter({ now: () => clock }) + for (let index = 0; index < CAPACITY; index++) limiter.allow('203.0.113.7') + expect(limiter.allow('203.0.113.7')).toBe(false) + + // Half a window buys back half the allowance, no more. + clock += 30_000 + for (let index = 0; index < CAPACITY / 2; index++) { + expect(limiter.allow('203.0.113.7')).toBe(true) + } + expect(limiter.allow('203.0.113.7')).toBe(false) + }) + + it('bounds what it remembers when a flood of distinct ips arrives', () => { + let clock = 1_000 + const limiter = new ClientIpRateLimiter({ now: () => clock, maxTrackedIps: 8 }) + for (let index = 0; index < 200; index++) { + clock += 1 + limiter.allow(`198.51.100.${index}`) + } + expect(limiter.trackedIpCount()).toBeLessThanOrEqual(8) + }) + + it('evicts the least recently used bucket without scanning the map', () => { + const limiter = new ClientIpRateLimiter({ capacity: 1, maxTrackedIps: 2, now: () => 1_000 }) + limiter.allow('old') + limiter.allow('recent') + expect(limiter.allow('old')).toBe(false) + const entries = vi.spyOn(Map.prototype, 'entries') + const iterator = vi.spyOn(Map.prototype, Symbol.iterator) + try { + limiter.allow('new') + expect(entries).not.toHaveBeenCalled() + expect(iterator).not.toHaveBeenCalled() + } finally { + entries.mockRestore() + iterator.mockRestore() + } + expect(limiter.available('old')).toBe(false) + expect(limiter.available('recent')).toBe(true) + expect(limiter.trackedIpCount()).toBe(2) + }) + + it('answers 429 with a rate_limited body once the bucket is empty', async () => { + const app = limiterApp(new ClientIpRateLimiter({ now: () => 1_000 })) + const headers = { 'x-forwarded-for': '10.0.0.1, 10.0.0.2, 203.0.113.7' } + for (let index = 0; index < CAPACITY; index++) { + expect((await app.request('/probe', { method: 'POST', headers })).status).toBe(200) + } + const limited = await app.request('/probe', { method: 'POST', headers }) + expect(limited.status).toBe(429) + expect(await limited.json()).toEqual({ error: 'rate_limited' }) + }) + + it('buckets on the last forwarded hop, the only one the platform appended', async () => { + const app = limiterApp(new ClientIpRateLimiter({ now: () => 1_000 })) + for (let index = 0; index < CAPACITY; index++) { + await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': `10.0.0.${index}, 203.0.113.7` } + }) + } + const sameClient = await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': '10.9.9.9, 203.0.113.7' } + }) + expect(sameClient.status).toBe(429) + const otherClient = await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': '10.0.0.1, 198.51.100.9' } + }) + expect(otherClient.status).toBe(200) + }) + + it('gives a spoofed left-most hop no escape from the caller own bucket', async () => { + const app = limiterApp(new ClientIpRateLimiter({ now: () => 1_000 })) + // A caller that rewrites its own x-forwarded-for on every request still ends + // up behind the one value Cloud Run appended. + for (let index = 0; index < CAPACITY; index++) { + const allowed = await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': `198.51.100.${index}, 203.0.113.7` } + }) + expect(allowed.status).toBe(200) + } + const spoofed = await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': '198.51.100.250, 10.1.1.1, 203.0.113.7' } + }) + expect(spoofed.status).toBe(429) + }) + + it('skips the configured trusted proxies when counting from the right', async () => { + const app = limiterApp(new ClientIpRateLimiter({ now: () => 1_000, capacity: 1 }), 1) + // , , : one trusted hop after the client. + const headers = { 'x-forwarded-for': '203.0.113.7, 10.0.0.1' } + expect((await app.request('/probe', { method: 'POST', headers })).status).toBe(200) + expect((await app.request('/probe', { method: 'POST', headers })).status).toBe(429) + expect( + ( + await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': '198.51.100.9, 10.0.0.1' } + }) + ).status + ).toBe(200) + }) + + it('trusts nothing when the header is shorter than the configured depth', async () => { + const app = limiterApp(new ClientIpRateLimiter({ now: () => 1_000, capacity: 1 }), 1) + // Only one hop, so the client value the depth points at does not exist. + const headers = { 'x-forwarded-for': '203.0.113.7' } + expect((await app.request('/probe', { method: 'POST', headers })).status).toBe(200) + expect( + ( + await app.request('/probe', { + method: 'POST', + headers: { 'x-forwarded-for': '198.51.100.9' } + }) + ).status + ).toBe(429) + }) + + it('ignores spoofable x-real-ip and uses a single shared bucket', async () => { + const app = limiterApp(new ClientIpRateLimiter({ now: () => 1_000, capacity: 1 })) + expect( + (await app.request('/probe', { method: 'POST', headers: { 'x-real-ip': '198.51.100.9' } })) + .status + ).toBe(200) + expect( + (await app.request('/probe', { method: 'POST', headers: { 'x-real-ip': '203.0.113.7' } })) + .status + ).toBe(429) + expect((await app.request('/probe', { method: 'POST' })).status).toBe(429) + expect((await app.request('/probe', { method: 'POST' })).status).toBe(429) + }) +}) diff --git a/cloud/apps/push/src/client-ip-rate-limit.ts b/cloud/apps/push/src/client-ip-rate-limit.ts new file mode 100644 index 00000000000..18518a2d316 --- /dev/null +++ b/cloud/apps/push/src/client-ip-rate-limit.ts @@ -0,0 +1,95 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import type { Context, MiddlewareHandler } from 'hono' + +const REFILL_WINDOW_MS = 60_000 +const MAX_TRACKED_IPS = 10_000 +const UNKNOWN_CLIENT_IP = 'unknown' + +export type ClientIpRateLimiterOptions = { + capacity?: number + windowMs?: number + maxTrackedIps?: number + now?: () => number +} + +type Bucket = { tokens: number; updatedAt: number } + +// Read x-forwarded-for from the right. Cloud Run appends the connecting peer, +// so the last value is the only one it wrote; everything to its left is +// whatever the caller sent and can be a fresh forgery on every request. +// trustedProxyHops is how many appenders sit between Cloud Run and the client +// (0 today, 1 once a load balancer fronts it). A header too short for that +// depth is not trusted at all and falls through to the shared bucket, which +// throttles rather than opens. +export function readClientIp(context: Context, trustedProxyHops = 0): string { + const hops = + context.req + .header('x-forwarded-for') + ?.split(',') + .map((hop) => hop.trim()) + .filter((hop) => hop.length > 0) ?? [] + const client = hops[hops.length - 1 - trustedProxyHops] + return client ?? UNKNOWN_CLIENT_IP +} + +// Per-instance admission avoids a database round trip; capacity scales with instance count. +export class ClientIpRateLimiter { + private readonly buckets = new Map() + private readonly capacity: number + private readonly windowMs: number + private readonly maxTrackedIps: number + private readonly now: () => number + + constructor(options: ClientIpRateLimiterOptions = {}) { + this.capacity = options.capacity ?? PUSH_LIMITS.unauthenticatedRequestsPerMinutePerIp + this.windowMs = options.windowMs ?? REFILL_WINDOW_MS + this.maxTrackedIps = options.maxTrackedIps ?? MAX_TRACKED_IPS + this.now = options.now ?? Date.now + } + + available(clientIp: string): boolean { + return this.tokensAt(this.buckets.get(clientIp), this.now()) >= 1 + } + + allow(clientIp: string): boolean { + const now = this.now() + const tokens = this.tokensAt(this.buckets.get(clientIp), now) + this.buckets.delete(clientIp) + this.buckets.set(clientIp, { tokens: tokens < 1 ? tokens : tokens - 1, updatedAt: now }) + if (this.buckets.size > this.maxTrackedIps) { + const oldest = this.buckets.keys().next().value + if (oldest !== undefined) this.buckets.delete(oldest) + } + return tokens >= 1 + } + + trackedIpCount(): number { + return this.buckets.size + } + + private tokensAt(bucket: Bucket | undefined, now: number): number { + if (!bucket) return this.capacity + const refilled = ((now - bucket.updatedAt) * this.capacity) / this.windowMs + return Math.min(this.capacity, bucket.tokens + Math.max(0, refilled)) + } +} + +export type ClientIpRateLimitOptions = { + trustedProxyHops?: number + onLimited?: () => void +} + +export function clientIpRateLimit( + limiter: ClientIpRateLimiter, + options: ClientIpRateLimitOptions = {} +): MiddlewareHandler { + const trustedProxyHops = options.trustedProxyHops ?? 0 + return async (context, next) => { + if (!limiter.allow(readClientIp(context, trustedProxyHops))) { + options.onLimited?.() + return context.json({ error: 'rate_limited' }, 429) + } + await next() + return + } +} diff --git a/cloud/apps/push/src/config.test.ts b/cloud/apps/push/src/config.test.ts new file mode 100644 index 00000000000..84efe7db7d7 --- /dev/null +++ b/cloud/apps/push/src/config.test.ts @@ -0,0 +1,109 @@ +import { generateKeyPairSync } from 'node:crypto' +import { PUSH_DEFAULTS } from '@orca-cloud/push-contract' +import { describe, expect, it } from 'vitest' +import { loadPushConfig, PUSH_DATABASE_POOL_MAX } from './config.js' + +function apnsKeyPem(): string { + return generateKeyPairSync('ec', { + namedCurve: 'P-256', + privateKeyEncoding: { type: 'pkcs8', format: 'pem' }, + publicKeyEncoding: { type: 'spki', format: 'pem' } + }).privateKey +} + +const MINIMAL = { + ORCA_PUSH_PUBLIC_URL: 'https://push.onorca.dev', + ORCA_PUSH_FCM_PROJECT_ID: 'onorca-cloud' +} + +describe('push gateway config', () => { + it('applies the documented defaults', () => { + expect(loadPushConfig(MINIMAL)).toEqual({ + mode: 'active', + port: 8080, + publicUrl: 'https://push.onorca.dev', + databaseUrl: undefined, + dataDir: './data/push', + databasePoolMax: PUSH_DATABASE_POOL_MAX, + apns: undefined, + apnsTopic: PUSH_DEFAULTS.apnsTopic, + fcmProjectId: 'onorca-cloud', + trustedProxyHops: 0 + }) + }) + + it('reads a full APNs credential and the overridable knobs', () => { + const keyPem = apnsKeyPem() + const config = loadPushConfig({ + ...MINIMAL, + PORT: '9090', + ORCA_PUSH_DATABASE_URL: 'postgres://localhost/orca_push', + ORCA_PUSH_DATA_DIR: '/var/lib/push', + ORCA_PUSH_APNS_KEY: keyPem, + ORCA_PUSH_APNS_KEY_ID: 'ABCDE12345', + ORCA_PUSH_APPLE_TEAM_ID: 'TEAM123456', + ORCA_PUSH_APNS_TOPIC: 'com.stably.orca.mobile.dev', + ORCA_PUSH_FCM_PROJECT_ID: 'onorca-staging', + ORCA_PUSH_TRUSTED_PROXY_HOPS: '1' + }) + expect(config).toMatchObject({ + port: 9090, + databaseUrl: 'postgres://localhost/orca_push', + dataDir: '/var/lib/push', + apns: { keyPem, keyId: 'ABCDE12345', teamId: 'TEAM123456' }, + apnsTopic: 'com.stably.orca.mobile.dev', + trustedProxyHops: 1, + fcmProjectId: 'onorca-staging' + }) + }) + + it('requires an explicit FCM project instead of silently targeting production', () => { + expect(() => loadPushConfig({ ...MINIMAL, ORCA_PUSH_FCM_PROJECT_ID: undefined })).toThrow() + expect(() => loadPushConfig({ ...MINIMAL, ORCA_PUSH_FCM_PROJECT_ID: ' ' })).toThrow() + }) + + it('refuses a partial APNs credential', () => { + expect(() => loadPushConfig({ ...MINIMAL, ORCA_PUSH_APNS_KEY: apnsKeyPem() })).toThrow( + 'configured together' + ) + expect(() => + loadPushConfig({ + ...MINIMAL, + ORCA_PUSH_APNS_KEY: 'not-a-pem', + ORCA_PUSH_APNS_KEY_ID: 'ABCDE12345', + ORCA_PUSH_APPLE_TEAM_ID: 'TEAM123456' + }) + ).toThrow('PEM text') + }) + + it('requires a canonical HTTPS origin outside loopback', () => { + expect(() => + loadPushConfig({ ...MINIMAL, ORCA_PUSH_PUBLIC_URL: 'https://push.onorca.dev/v1' }) + ).toThrow('must be an origin') + expect(() => + loadPushConfig({ ...MINIMAL, ORCA_PUSH_PUBLIC_URL: 'http://push.onorca.dev' }) + ).toThrow('must use HTTPS') + expect( + loadPushConfig({ ...MINIMAL, ORCA_PUSH_PUBLIC_URL: 'http://localhost:8080' }).publicUrl + ).toBe('http://localhost:8080') + }) + + it('treats an empty optional variable as unset', () => { + expect( + loadPushConfig({ ...MINIMAL, ORCA_PUSH_DATABASE_URL: '', ORCA_PUSH_APNS_KEY_ID: '' }) + ).toMatchObject({ databaseUrl: undefined, apns: undefined }) + }) +}) + +it('treats blank defaulted environment settings as absent', () => { + const blanks = Object.fromEntries( + [ + 'PORT', + 'ORCA_PUSH_DATA_DIR', + 'ORCA_PUSH_APNS_TOPIC', + 'ORCA_PUSH_DATABASE_POOL_MAX', + 'ORCA_PUSH_TRUSTED_PROXY_HOPS' + ].map((key) => [key, ' ']) + ) + expect(loadPushConfig({ ...MINIMAL, ...blanks })).toEqual(loadPushConfig(MINIMAL)) +}) diff --git a/cloud/apps/push/src/config.ts b/cloud/apps/push/src/config.ts new file mode 100644 index 00000000000..2ed0d2111f1 --- /dev/null +++ b/cloud/apps/push/src/config.ts @@ -0,0 +1,108 @@ +import { PUSH_DEFAULTS } from '@orca-cloud/push-contract' +import { z } from 'zod' + +export const PUSH_DATABASE_POOL_MAX = 10 + +const OptionalTextSchema = z.preprocess( + (value) => (value === '' ? undefined : value), + z.string().min(1).optional() +) + +const EnvSchema = z.object({ + ORCA_PUSH_MODE: z.enum(['active', 'validation']).default('active'), + PORT: z.coerce.number().int().positive().default(8080), + ORCA_PUSH_PUBLIC_URL: z.string().url(), + ORCA_PUSH_DATABASE_URL: OptionalTextSchema, + ORCA_PUSH_DATA_DIR: z.string().min(1).default('./data/push'), + ORCA_PUSH_DATABASE_POOL_MAX: z.coerce.number().int().positive().max(100).optional(), + ORCA_PUSH_APNS_KEY: OptionalTextSchema, + ORCA_PUSH_APNS_KEY_ID: z.preprocess( + (value) => (value === '' ? undefined : value), + z + .string() + .regex(/^[A-Z0-9]{10}$/) + .optional() + ), + ORCA_PUSH_APPLE_TEAM_ID: z.preprocess( + (value) => (value === '' ? undefined : value), + z + .string() + .regex(/^[A-Z0-9]{10}$/) + .optional() + ), + ORCA_PUSH_APNS_TOPIC: z.string().min(1).max(255).default(PUSH_DEFAULTS.apnsTopic), + ORCA_PUSH_FCM_PROJECT_ID: z.string().regex(/^[a-z0-9-]{4,64}$/), + // How many proxies append to x-forwarded-for after the client. 0 is Cloud Run + // alone; raise it to 1 when a load balancer fronts the service. + ORCA_PUSH_TRUSTED_PROXY_HOPS: z.coerce.number().int().nonnegative().max(8).default(0) +}) + +export type ApnsCredentials = { keyPem: string; keyId: string; teamId: string } + +export type PushConfig = { + mode: 'active' | 'validation' + port: number + publicUrl: string + databaseUrl?: string + dataDir: string + databasePoolMax: number + apns?: ApnsCredentials + apnsTopic: string + fcmProjectId: string + trustedProxyHops: number +} + +function canonicalOrigin(value: string, name: string): string { + const url = new URL(value) + if (url.origin !== value || url.pathname !== '/') throw new Error(`${name} must be an origin`) + const loopback = ['127.0.0.1', 'localhost', '::1', '[::1]'].includes(url.hostname) + if (url.protocol !== 'https:' && !(loopback && url.protocol === 'http:')) { + throw new Error(`${name} must use HTTPS outside loopback development`) + } + return value +} + +// The APNs key, key id, and team id are one credential; a partial set would +// pass startup and then fail every iOS send at runtime. +function readApnsCredentials(parsed: z.infer): ApnsCredentials | undefined { + const parts = [ + parsed.ORCA_PUSH_APNS_KEY, + parsed.ORCA_PUSH_APNS_KEY_ID, + parsed.ORCA_PUSH_APPLE_TEAM_ID + ] + const present = parts.filter((value) => value !== undefined).length + if (present === 0) return undefined + if (present !== parts.length) { + throw new Error('APNs key, key id, and team id must be configured together') + } + const keyPem = parsed.ORCA_PUSH_APNS_KEY! + if (!keyPem.includes('-----BEGIN')) throw new Error('ORCA_PUSH_APNS_KEY must be PEM text') + return { + keyPem, + keyId: parsed.ORCA_PUSH_APNS_KEY_ID!, + teamId: parsed.ORCA_PUSH_APPLE_TEAM_ID! + } +} + +export function loadPushConfig(env: NodeJS.ProcessEnv = process.env): PushConfig { + const parsed = EnvSchema.parse( + Object.fromEntries( + Object.entries(env).map(([key, value]) => [ + key, + key !== 'ORCA_PUSH_MODE' && value?.trim() === '' ? undefined : value + ]) + ) + ) + return { + mode: parsed.ORCA_PUSH_MODE, + port: parsed.PORT, + publicUrl: canonicalOrigin(parsed.ORCA_PUSH_PUBLIC_URL, 'ORCA_PUSH_PUBLIC_URL'), + databaseUrl: parsed.ORCA_PUSH_DATABASE_URL, + dataDir: parsed.ORCA_PUSH_DATA_DIR, + databasePoolMax: parsed.ORCA_PUSH_DATABASE_POOL_MAX ?? PUSH_DATABASE_POOL_MAX, + apns: readApnsCredentials(parsed), + apnsTopic: parsed.ORCA_PUSH_APNS_TOPIC, + fcmProjectId: parsed.ORCA_PUSH_FCM_PROJECT_ID, + trustedProxyHops: parsed.ORCA_PUSH_TRUSTED_PROXY_HOPS + } +} diff --git a/cloud/apps/push/src/desktop-host-proof-interop.test.ts b/cloud/apps/push/src/desktop-host-proof-interop.test.ts new file mode 100644 index 00000000000..654423b0de8 --- /dev/null +++ b/cloud/apps/push/src/desktop-host-proof-interop.test.ts @@ -0,0 +1,47 @@ +import { describe, expect, it } from 'vitest' +import { createHmac } from 'node:crypto' +import vector from '../../../packages/push-contract/src/push-host-proof-vector.json' with { type: 'json' } +import { answerPushHostChallenge, createPushHostKeypair } from './host-challenge-answering.test-fixture.js' +import { PushHostChallengeStore } from './host-challenge-store.js' +import { deriveHostFingerprint } from './host-fingerprint.js' +import { openInMemoryPushDatabase } from './push-database.js' + +// Why: the desktop answers challenges in a workspace this one cannot import. +// Both sides replay the same checked-in vector, so a transcript drift on +// either side fails in that side's own suite. +describe('desktop host proof interop', () => { + it('the checked-in vector answers to the same proof the fixture host computes', () => { + const secretKey = new Uint8Array(Buffer.from(vector.hostSecretKeyB64, 'base64')) + const keypair = { publicKey: new Uint8Array(Buffer.from(vector.hostPublicKeyB64, 'base64')), secretKey } + expect(deriveHostFingerprint(keypair.publicKey)).toBe(vector.hostFingerprint) + const proof = answerPushHostChallenge(vector.challenge, { + gatewayOrigin: vector.gatewayOrigin, + keypair, + now: () => vector.issuedAt + 1_000 + }) + const expected = createHmac('sha256', Buffer.from(vector.challengeSecretB64, 'base64')) + .update(Buffer.from('orca-push-host-proof/v1\0ack\0')) + .update(Buffer.from(vector.transcriptB64, 'base64')) + .digest('base64') + expect(proof).toBe(expected) + }) + + it('a live challenge from the store round-trips through the fixture host once', async () => { + const database = await openInMemoryPushDatabase() + const store = new PushHostChallengeStore(database, vector.gatewayOrigin) + const keypair = createPushHostKeypair(11) + const challenge = await store.issue(Buffer.from(keypair.publicKey).toString('base64')) + expect(challenge).not.toBeNull() + const proof = answerPushHostChallenge(challenge!, { gatewayOrigin: vector.gatewayOrigin, keypair }) + expect(proof).not.toBeNull() + expect(await store.verify(challenge!.challengeId, proof!)).toEqual({ + ok: true, + hostFingerprint: deriveHostFingerprint(keypair.publicKey) + }) + expect(await store.verify(challenge!.challengeId, proof!)).toEqual({ + ok: false, + reason: 'already_consumed' + }) + await database.close() + }) +}) diff --git a/cloud/apps/push/src/device-registration-delete-race.test.ts b/cloud/apps/push/src/device-registration-delete-race.test.ts new file mode 100644 index 00000000000..2fe51e30516 --- /dev/null +++ b/cloud/apps/push/src/device-registration-delete-race.test.ts @@ -0,0 +1,88 @@ +import { expect, it, vi } from 'vitest' +import { openPushDatabase, type PushDatabase } from './push-database.js' +import { PushDeviceRegistryStore } from './device-registry-store.js' + +const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL +it.skipIf(!databaseUrl)( + 'serializes deletion with a registration that has already read its row', + async () => { + if (!process.env.CI && new URL(databaseUrl!).port !== '55440') + throw new Error('isolated_postgres_port_required') + const database = await openPushDatabase({ + databaseUrl, + dataDir: '', + poolMax: 4, + applicationName: 'push-delete-race' + }) + let release!: () => void + const paused = new Promise((resolve) => { + release = resolve + }) + let read = false + let pause = false + const wrapped: PushDatabase = { + dialect: database.dialect, + query: database.query.bind(database), + close: database.close.bind(database), + lockQuotaScope: database.lockQuotaScope.bind(database), + transaction: (run) => + database.transaction((tx) => + run({ + dialect: tx.dialect, + close: tx.close.bind(tx), + transaction: tx.transaction.bind(tx), + lockQuotaScope: tx.lockQuotaScope.bind(tx), + query: async (sql, params) => { + const rows = await tx.query(sql, params) + if (pause && sql.startsWith('SELECT registration_id FROM push_devices')) { + read = true + await paused + } + return rows + } + }) + ) + } + const devices = new PushDeviceRegistryStore(wrapped) + const input = { + hostFingerprint: 'delete-race-host', + deviceId: 'phone', + platform: 'android' as const, + token: 'synthetic' + } + let registration: Promise | undefined + let deletion: Promise | undefined + try { + await database.query('DELETE FROM push_devices WHERE host_fingerprint = ?', [ + input.hostFingerprint + ]) + const first = await devices.upsert(input) + if (!first.ok) throw new Error('registration refused') + pause = true + registration = devices.upsert(input) + await vi.waitFor(() => expect(read).toBe(true)) + let deleted = false + deletion = devices.deleteOwned(input.hostFingerprint, first.registrationId).then((value) => { + deleted = true + return value + }) + await vi.waitFor(async () => { + const rows = await database.query( + "SELECT 1 FROM pg_stat_activity WHERE application_name = 'push-delete-race' AND wait_event_type = 'Lock'" + ) + expect(deleted || rows.length > 0).toBe(true) + }) + expect(deleted).toBe(false) + release() + expect(await registration).toEqual(first) + expect(await deletion).toBe(true) + } finally { + release() + await Promise.allSettled([registration, deletion]) + await database.query('DELETE FROM push_devices WHERE host_fingerprint = ?', [ + input.hostFingerprint + ]) + await database.close() + } + } +) diff --git a/cloud/apps/push/src/device-registry-store.test.ts b/cloud/apps/push/src/device-registry-store.test.ts new file mode 100644 index 00000000000..f325d7611b1 --- /dev/null +++ b/cloud/apps/push/src/device-registry-store.test.ts @@ -0,0 +1,191 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { PushDeviceRegistryStore, type PushDeviceUpsert } from './device-registry-store.js' +import { openInMemoryPushDatabase, type PushDatabase } from './push-database.js' + +const OWNER = 'abcdefghijklmnop' +const OTHER = 'ponmlkjihgfedcba' + +describe('push device registry store', () => { + let database: PushDatabase + let clock = 1_700_000_000_000 + let devices: PushDeviceRegistryStore + + beforeEach(async () => { + database = await openInMemoryPushDatabase() + clock = 1_700_000_000_000 + devices = new PushDeviceRegistryStore(database, () => clock) + }) + + afterEach(async () => { + await database.close() + }) + + async function upsertOk(input: PushDeviceUpsert): Promise { + const result = await devices.upsert(input) + if (!result.ok) throw new Error(`unexpected upsert refusal: ${result.reason}`) + return result.registrationId + } + + function androidDevice(deviceId: string): PushDeviceUpsert { + return { + hostFingerprint: OWNER, + deviceId, + platform: 'android', + token: `token-${deviceId}` + } + } + + it('keeps one registration per host and device while replacing the token', async () => { + const first = await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'device-1', + platform: 'ios', + token: 'a'.repeat(64), + apnsEnvironment: 'sandbox' + }) + clock += 1_000 + const second = await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'device-1', + platform: 'ios', + token: 'b'.repeat(64), + apnsEnvironment: 'production' + }) + expect(second).toBe(first) + const registration = await devices.findById(first) + expect(registration).toMatchObject({ + token: 'b'.repeat(64), + apnsEnvironment: 'production', + dead: false + }) + expect(await devices.list(OWNER)).toHaveLength(1) + }) + + it('revives a registration that a re-registered token replaces', async () => { + const registrationId = await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'device-1', + platform: 'android', + token: 'token-one' + }) + await devices.markDead((await devices.findById(registrationId))!) + expect((await devices.findById(registrationId))?.dead).toBe(true) + await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'device-1', + platform: 'android', + token: 'token-two' + }) + expect(await devices.findById(registrationId)).toMatchObject({ + token: 'token-two', + dead: false + }) + }) + + it('lets only the owning host delete a registration', async () => { + const registrationId = await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'device-1', + platform: 'android', + token: 'token-one' + }) + expect(await devices.deleteOwned(OTHER, registrationId)).toBe(false) + expect(await devices.findById(registrationId)).not.toBeNull() + expect(await devices.deleteOwned(OWNER, registrationId)).toBe(true) + expect(await devices.findById(registrationId)).toBeNull() + }) + + it('scopes lookups and listings to the owning host', async () => { + const owned = await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'device-1', + platform: 'android', + token: 'token-one' + }) + const foreign = await upsertOk({ + hostFingerprint: OTHER, + deviceId: 'device-2', + platform: 'android', + token: 'token-two' + }) + const found = await devices.findOwned(OWNER, [owned, foreign]) + expect([...found.keys()]).toEqual([owned]) + expect(await devices.list(OTHER)).toEqual([ + { registrationId: foreign, deviceId: 'device-2', platform: 'android', dead: false } + ]) + expect(await devices.findOwned(OWNER, [])).toEqual(new Map()) + }) + + it('refuses a new device once the host reaches its registration cap', async () => { + for (let index = 0; index < PUSH_LIMITS.maxDevicesPerHost; index++) { + await upsertOk(androidDevice(`device-${index}`)) + } + expect(await devices.upsert(androidDevice('one-too-many'))).toEqual({ + ok: false, + reason: 'too_many_devices' + }) + expect(await devices.list(OWNER)).toHaveLength(PUSH_LIMITS.maxDevicesPerHost) + }) + + it('still lets a capped host re-register a device it already owns', async () => { + for (let index = 0; index < PUSH_LIMITS.maxDevicesPerHost; index++) { + await upsertOk(androidDevice(`device-${index}`)) + } + const rotated = await devices.upsert({ ...androidDevice('device-0'), token: 'rotated-token' }) + expect(rotated.ok).toBe(true) + expect(await devices.list(OWNER)).toHaveLength(PUSH_LIMITS.maxDevicesPerHost) + }) + + it('frees a slot when a registration is deleted', async () => { + const first = await upsertOk(androidDevice('device-0')) + for (let index = 1; index < PUSH_LIMITS.maxDevicesPerHost; index++) { + await upsertOk(androidDevice(`device-${index}`)) + } + expect((await devices.upsert(androidDevice('extra'))).ok).toBe(false) + expect(await devices.deleteOwned(OWNER, first)).toBe(true) + expect((await devices.upsert(androidDevice('extra'))).ok).toBe(true) + }) + + it('counts the cap per host, not across the whole table', async () => { + for (let index = 0; index < PUSH_LIMITS.maxDevicesPerHost; index++) { + await upsertOk(androidDevice(`device-${index}`)) + } + expect((await devices.upsert(androidDevice('extra'))).ok).toBe(false) + expect( + (await devices.upsert({ ...androidDevice('device-0'), hostFingerprint: OTHER })).ok + ).toBe(true) + }) + + it('bounds list reads to the host device allowance', async () => { + // Straight past the per-host cap, so only the query LIMIT can bound this. + const rows = PUSH_LIMITS.maxDevicesPerHost + 5 + for (let index = 0; index < rows; index++) { + await database.query( + `INSERT INTO push_devices (registration_id, host_fingerprint, device_id, platform, token, + created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?)`, + [`reg-${index}`, OWNER, `device-${index}`, 'android', 'token', clock + index, clock] + ) + } + expect(await devices.list(OWNER)).toHaveLength(PUSH_LIMITS.maxDevicesPerHost) + }) + + it('separates the same device id registered against two hosts', async () => { + const first = await upsertOk({ + hostFingerprint: OWNER, + deviceId: 'shared-device', + platform: 'ios', + token: 'a'.repeat(64), + apnsEnvironment: 'sandbox' + }) + const second = await upsertOk({ + hostFingerprint: OTHER, + deviceId: 'shared-device', + platform: 'ios', + token: 'c'.repeat(64), + apnsEnvironment: 'sandbox' + }) + expect(first).not.toBe(second) + }) +}) diff --git a/cloud/apps/push/src/device-registry-store.ts b/cloud/apps/push/src/device-registry-store.ts new file mode 100644 index 00000000000..1729f7d1b82 --- /dev/null +++ b/cloud/apps/push/src/device-registry-store.ts @@ -0,0 +1,177 @@ +import { randomUUID } from 'node:crypto' +import { + PUSH_LIMITS, + type ApnsEnvironment, + type PushDeviceSummary, + type PushPlatform +} from '@orca-cloud/push-contract' +import type { PushDatabase, SqlRow } from './push-database.js' + +const DEVICE_CAP_LOCK_PREFIX = 'orca-push-device-cap:' + +export type PushDeviceRegistration = { + registrationId: string + hostFingerprint: string + deviceId: string + platform: PushPlatform + token: string + apnsEnvironment?: ApnsEnvironment + dead: boolean +} + +export type PushDeviceUpsertResult = + | { ok: true; registrationId: string } + | { ok: false; reason: 'too_many_devices' } + +export type PushDeviceUpsert = { + hostFingerprint: string + deviceId: string + platform: PushPlatform + token: string + apnsEnvironment?: ApnsEnvironment +} + +function toRegistration(row: SqlRow): PushDeviceRegistration { + const apnsEnvironment = row.apns_environment + return { + registrationId: String(row.registration_id), + hostFingerprint: String(row.host_fingerprint), + deviceId: String(row.device_id), + platform: String(row.platform) as PushPlatform, + token: String(row.token), + ...(apnsEnvironment === null || apnsEnvironment === undefined + ? {} + : { apnsEnvironment: String(apnsEnvironment) as ApnsEnvironment }), + dead: row.dead_at !== null && row.dead_at !== undefined + } +} + +export class PushDeviceRegistryStore { + constructor( + private readonly database: PushDatabase, + private readonly now: () => number = Date.now + ) {} + + // The registration id is stable for a (host, device) pair so a re-registered + // phone keeps the id the desktop already persisted; only the token rotates. + async upsert(input: PushDeviceUpsert): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + // deviceId is caller-chosen, so counting and inserting must not interleave + // or a burst of new ids would walk straight past the cap. + await transaction.lockQuotaScope(`${DEVICE_CAP_LOCK_PREFIX}${input.hostFingerprint}`) + const [existing] = await transaction.query( + 'SELECT registration_id FROM push_devices WHERE host_fingerprint = ? AND device_id = ?', + [input.hostFingerprint, input.deviceId] + ) + if (existing) { + const registrationId = String(existing.registration_id) + await transaction.query( + `UPDATE push_devices + SET platform = ?, token = ?, apns_environment = ?, + dead_at = NULL, updated_at = ? + WHERE registration_id = ?`, + [input.platform, input.token, input.apnsEnvironment ?? null, now, registrationId] + ) + return { ok: true, registrationId } + } + const [countRow] = await transaction.query( + 'SELECT COUNT(*) AS devices FROM push_devices WHERE host_fingerprint = ?', + [input.hostFingerprint] + ) + if (Number(countRow?.devices ?? 0) >= PUSH_LIMITS.maxDevicesPerHost) { + return { ok: false, reason: 'too_many_devices' } + } + const registrationId = randomUUID() + await transaction.query( + `INSERT INTO push_devices + (registration_id, host_fingerprint, device_id, platform, token, apns_environment, + dead_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, NULL, ?, ?)`, + [ + registrationId, + input.hostFingerprint, + input.deviceId, + input.platform, + input.token, + input.apnsEnvironment ?? null, + now, + now + ] + ) + return { ok: true, registrationId } + }) + } + + async deleteOwned(hostFingerprint: string, registrationId: string): Promise { + return this.database.transaction(async (transaction) => { + await transaction.lockQuotaScope(`${DEVICE_CAP_LOCK_PREFIX}${hostFingerprint}`) + const [result] = await transaction.query( + 'DELETE FROM push_devices WHERE registration_id = ? AND host_fingerprint = ?', + [registrationId, hostFingerprint] + ) + return Number(result?.changes ?? 0) > 0 + }) + } + + async list(hostFingerprint: string): Promise { + const rows = await this.database.query( + // Bounded by the device-list response limit, so an + // oversized table degrades to a truncated list instead of a 500. + `SELECT registration_id, device_id, platform, dead_at + FROM push_devices WHERE host_fingerprint = ? ORDER BY created_at ASC LIMIT ?`, + [hostFingerprint, PUSH_LIMITS.maxDevicesPerHost] + ) + return rows.map((row) => ({ + registrationId: String(row.registration_id), + deviceId: String(row.device_id), + platform: String(row.platform) as PushPlatform, + dead: row.dead_at !== null && row.dead_at !== undefined + })) + } + + async findOwned( + hostFingerprint: string, + registrationIds: readonly string[] + ): Promise> { + if (registrationIds.length === 0) return new Map() + const placeholders = registrationIds.map(() => '?').join(', ') + const rows = await this.database.query( + `SELECT registration_id, host_fingerprint, device_id, platform, token, apns_environment, + dead_at + FROM push_devices + WHERE host_fingerprint = ? AND registration_id IN (${placeholders})`, + [hostFingerprint, ...registrationIds] + ) + return new Map( + rows.map((row) => { + const registration = toRegistration(row) + return [registration.registrationId, registration] + }) + ) + } + + async findById(registrationId: string): Promise { + const [row] = await this.database.query( + `SELECT registration_id, host_fingerprint, device_id, platform, token, apns_environment, + dead_at + FROM push_devices WHERE registration_id = ?`, + [registrationId] + ) + return row ? toRegistration(row) : null + } + + async markDead(observed: PushDeviceRegistration): Promise { + await this.database.query( + `UPDATE push_devices SET dead_at = ?, updated_at = ? WHERE registration_id = ? AND token = ? AND platform = ? AND COALESCE(apns_environment, '') = ?`, + [ + this.now(), + this.now(), + observed.registrationId, + observed.token, + observed.platform, + observed.apnsEnvironment ?? '' + ] + ) + } +} diff --git a/cloud/apps/push/src/durable-push-schema.ts b/cloud/apps/push/src/durable-push-schema.ts new file mode 100644 index 00000000000..770c2a1fbce --- /dev/null +++ b/cloud/apps/push/src/durable-push-schema.ts @@ -0,0 +1,45 @@ +export const DURABLE_PUSH_SCHEMA = ` +CREATE TABLE IF NOT EXISTS push_dismissed_events ( + host_fingerprint TEXT NOT NULL, + notification_epoch TEXT NOT NULL, + notification_id TEXT NOT NULL, + notification_seq BIGINT NOT NULL, + created_at BIGINT NOT NULL, + PRIMARY KEY(host_fingerprint, notification_epoch, notification_id) +); +CREATE INDEX IF NOT EXISTS push_dismissed_retention ON push_dismissed_events(created_at); +CREATE TABLE IF NOT EXISTS push_events ( + event_id TEXT PRIMARY KEY, + host_fingerprint TEXT NOT NULL, + kind TEXT NOT NULL, + fingerprint TEXT NOT NULL, + created_at BIGINT NOT NULL, + expires_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS push_events_quota ON push_events(host_fingerprint, kind, created_at); +CREATE TABLE IF NOT EXISTS push_event_recipients ( + event_id TEXT NOT NULL, + registration_id TEXT NOT NULL, + created_at BIGINT NOT NULL, + PRIMARY KEY(event_id, registration_id) +); +CREATE TABLE IF NOT EXISTS push_delivery_batches ( + batch_id TEXT PRIMARY KEY, + host_fingerprint TEXT NOT NULL, + registration_id TEXT NOT NULL, + kind TEXT NOT NULL, + payload_json TEXT NOT NULL, + state TEXT NOT NULL, + due_at BIGINT NOT NULL, + expires_at BIGINT NOT NULL, + lease_token TEXT, + lease_until BIGINT NOT NULL, + attempts BIGINT NOT NULL, + created_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS push_events_retention ON push_events(created_at); +CREATE INDEX IF NOT EXISTS push_recipients_retention ON push_event_recipients(created_at); +CREATE INDEX IF NOT EXISTS push_batches_expiry ON push_delivery_batches(expires_at); +CREATE INDEX IF NOT EXISTS push_batches_due ON push_delivery_batches(state, due_at); +CREATE INDEX IF NOT EXISTS push_batches_registration ON push_delivery_batches(registration_id, state); +` diff --git a/cloud/apps/push/src/durable-push-store.test.ts b/cloud/apps/push/src/durable-push-store.test.ts new file mode 100644 index 00000000000..136c0832211 --- /dev/null +++ b/cloud/apps/push/src/durable-push-store.test.ts @@ -0,0 +1,289 @@ +import { randomUUID } from 'node:crypto' +import pg from 'pg' +import { afterEach, describe, expect, it } from 'vitest' +import { openInMemoryPushDatabase, openPushDatabase, type PushDatabase } from './push-database.js' +import { DurablePushStore, DELIVERY_LEASE_MS } from './durable-push-store.js' +import type { PushNotification } from '@orca-cloud/push-contract' + +const cleanups: (() => Promise)[] = [] +afterEach(async () => { + await Promise.all(cleanups.splice(0).map((cleanup) => cleanup())) +}) +const notification = (seq: number, kind: 'alert' | 'dismiss' = 'alert'): PushNotification => ({ + notificationId: `notification-${seq}`, + notificationEpoch: 'epoch', + notificationSeq: seq, + source: 'agent-task-complete', + agentState: 'finished', + title: 'Done', + body: '', + kind +}) +async function fixture() { + const databaseUrl = + process.env.ORCA_PUSH_DURABLE_TEST_POSTGRES_URL ?? process.env.ORCA_PUSH_TEST_DATABASE_URL + if (databaseUrl && !process.env.CI && new URL(databaseUrl).port !== '55440') + throw new Error('isolated_postgres_port_required') + let db: PushDatabase + if (databaseUrl) { + const admin = new pg.Client({ connectionString: databaseUrl }) + await admin.connect() + const schema = `durable_${randomUUID().replaceAll('-', '')}` + let scoped: PushDatabase | undefined + cleanups.push(async () => { + try { + await scoped?.close() + } finally { + try { + await admin.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + } finally { + await admin.end() + } + } + }) + await admin.query(`CREATE SCHEMA ${schema}`) + const url = new URL(databaseUrl) + url.searchParams.set('options', `-c search_path=${schema}`) + db = scoped = await openPushDatabase({ databaseUrl: url.toString(), dataDir: '', poolMax: 4 }) + } else { + db = await openInMemoryPushDatabase() + cleanups.push(() => db.close()) + } + let now = 1_000_000 + const clock = () => now + return { + db, + store: new DurablePushStore(db, clock), + clock, + advance: (ms: number) => { + now += ms + } + } +} + +describe('durable push acceptance', () => { + it('counts a logical event once across phones and separates the 300/15min dismissal budget', async () => { + const { store, advance } = await fixture() + for (let i = 0; i < 300; i++) { + expect(await store.accept('host', 'phone1', notification(i))).toBe('queued') + expect(await store.accept('host', 'phone2', notification(i))).toBe('queued') + expect(await store.accept('host', 'phone1', notification(i, 'dismiss'))).toBe('queued') + } + expect(await store.accept('host', 'phone1', notification(300))).toBe('rate_limited') + expect(await store.accept('host', 'phone1', notification(300, 'dismiss'))).toBe('rate_limited') + expect(await store.accept('another-host', 'phone3', notification(300))).toBe('queued') + advance(15 * 60_000) + expect(await store.accept('host', 'phone1', notification(301))).toBe('queued') + }) + + it('queues one delivery per event and recovers work across service instances', async () => { + const { db, store, clock, advance } = await fixture() + await store.accept('host', 'phone', notification(1)) + const restarted = new DurablePushStore(db, clock) + await restarted.accept('host', 'phone', notification(1)) + advance(1) + await restarted.accept('host', 'phone', notification(2)) + const rows = await db.query( + "SELECT payload_json, due_at, created_at FROM push_delivery_batches WHERE registration_id = ? AND state = 'pending' ORDER BY created_at, batch_id", + ['phone'] + ) + expect(rows).toHaveLength(2) + expect(rows.map((row) => JSON.parse(String(row.payload_json)))).toEqual([ + notification(1), + notification(2) + ]) + expect(rows.every((row) => Number(row.due_at) >= Number(row.created_at))).toBe(true) + const delivery = await restarted.claim() + expect(delivery?.notification.notificationSeq).toBe(1) + expect(await store.claim()).toBeNull() + advance(DELIVERY_LEASE_MS) + const reclaimed = await store.claim() + expect(reclaimed?.id).toBe(delivery?.id) + expect(reclaimed?.lease).not.toBe(delivery?.lease) + await restarted.finish(delivery!) + expect(await store.claim()).toBeNull() + await store.finish(reclaimed!) + const second = await restarted.claim() + expect(second?.notification.notificationSeq).toBe(2) + await restarted.finish(second!) + expect(await restarted.claim()).toBeNull() + }) + + it('never extends expiry and refuses conflicting duplicate content', async () => { + const { store, advance } = await fixture() + await store.accept('host', 'phone', notification(1)) + expect(await store.accept('host', 'phone', { ...notification(1), body: 'changed' })).toBe( + 'error' + ) + const delivery = (await store.claim())! + await store.finish(delivery, 10 * 60_000) + advance(60_000) + expect(await store.claim()).toBeNull() + advance(5 * 60_000) + expect(await store.accept('host', 'phone', notification(1))).toBe('error') + }) + + it('orders a due retry before a fresh first attempt without delaying the retry', async () => { + const { store, advance } = await fixture() + await store.accept('host', 'phone', notification(1)) + const first = (await store.claim())! + await store.finish(first, 1000) + expect(await store.claim()).toBeNull() + + advance(1000) + await store.accept('host', 'phone', notification(2)) + const retry = (await store.claim())! + expect(retry.notification.notificationSeq).toBe(1) + await store.finish(retry) + const fresh = (await store.claim())! + expect(fresh?.notification.notificationSeq).toBe(2) + await store.finish(fresh!) + }) + + it('orders an expired first-attempt lease by creation time after a retry becomes due', async () => { + const { db, store, clock, advance } = await fixture() + await store.accept('host', 'phone', notification(1)) + const retry = (await store.claim())! + await store.finish(retry, 1000) + + advance(2000) + await db.query( + `INSERT INTO push_delivery_batches(batch_id, host_fingerprint, registration_id, kind, payload_json, state, due_at, expires_at, lease_until, attempts, created_at) + VALUES ('crashed-singleton', 'host', 'phone', 'alert', ?, 'pending', ?, ?, 0, 1, ?)`, + [JSON.stringify(notification(2)), clock() - 1, clock() + 300_000, clock()] + ) + const reclaimedRetry = (await store.claim())! + expect(reclaimedRetry.notification.notificationSeq).toBe(1) + await store.finish(reclaimedRetry) + const reclaimedCrash = (await store.claim())! + expect(reclaimedCrash.notification.notificationSeq).toBe(2) + await store.finish(reclaimedCrash) + }) + + it('rolls quota and payload back together if persistence fails', async () => { + const { db, store } = await fixture() + await db.query('ALTER TABLE push_delivery_batches RENAME TO push_delivery_batches_unavailable') + try { + const databaseUrl = + process.env.ORCA_PUSH_DURABLE_TEST_POSTGRES_URL ?? process.env.ORCA_PUSH_TEST_DATABASE_URL + if (databaseUrl) { + const concurrent = await openPushDatabase({ databaseUrl, dataDir: '' }) + try { + await expect( + concurrent.query('SELECT COUNT(*) FROM push_delivery_batches') + ).resolves.toHaveLength(1) + } finally { + await concurrent.close() + } + } + await expect(store.accept('host', 'phone', notification(1))).rejects.toThrow() + expect(await db.query('SELECT * FROM push_events')).toEqual([]) + expect(await db.query('SELECT * FROM push_event_recipients')).toEqual([]) + } finally { + await db.query( + 'ALTER TABLE push_delivery_batches_unavailable RENAME TO push_delivery_batches' + ) + } + }) +}) + +it('serializes concurrent instances at the quota boundary', async () => { + const { db, store, clock } = await fixture() + for (let seq = 0; seq < 299; seq++) await store.accept('host', 'phone', notification(seq)) + const second = new DurablePushStore(db, clock) + const results = await Promise.all( + Array.from({ length: 6 }, (_, index) => + (index % 2 ? store : second).accept('host', 'phone', notification(400 + index)) + ) + ) + expect(results.filter((result) => result === 'queued')).toHaveLength(1) + expect(results.filter((result) => result === 'rate_limited')).toHaveLength(5) +}) + +it('cancels unsent alerts and prevents an older replay after dismissal', async () => { + const { store } = await fixture() + const alert = notification(1) + await store.accept('host', 'phone', alert) + await store.accept('host', 'phone', { + ...notification(2, 'dismiss'), + notificationId: alert.notificationId + }) + const delivery = (await store.claim())! + expect(delivery.notification.kind).toBe('dismiss') + await store.finish(delivery) + expect(await store.claim()).toBeNull() + await store.accept('host', 'another-phone', alert) + expect(await store.claim()).toBeNull() +}) + +it('does not resurrect an in-flight alert after a dismissal and transient provider failure', async () => { + const { store, advance } = await fixture() + await store.accept('host', 'phone', notification(1)) + const inFlight = (await store.claim())! + await store.accept('host', 'phone', { + ...notification(2, 'dismiss'), + notificationId: notification(1).notificationId + }) + await store.finish(inFlight, 1000) + const dismissal = (await store.claim())! + expect(dismissal.notification.kind).toBe('dismiss') + await store.finish(dismissal) + advance(1000) + expect(await store.claim()).toBeNull() + expect(await store.pendingCount('phone')).toBe(0) +}) + +it.each([false, true])( + 'normalizes default alert kind (explicit first: %s)', + async (explicitFirst) => { + const { db, store } = await fixture() + const { kind: _kind, ...implicit } = notification(1) + const explicit = { kind: 'alert' as const, ...implicit } + for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) { + expect(await store.accept('host', 'phone', event)).toBe('queued') + } + expect(await store.pendingCount('phone')).toBe(1) + expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(1) + expect(await store.accept('host', 'phone', { ...explicit, body: 'changed' })).toBe('error') + expect(await store.accept('host', 'phone', { ...implicit, kind: 'dismiss' })).toBe('queued') + expect(await db.query('SELECT event_id FROM push_events')).toHaveLength(2) + } +) + +it('fences late renew and finish after an expired claim is dismissed', async () => { + const { db, store, advance } = await fixture() + const alert = notification(1) + await store.accept('host', 'phone', alert) + const stale = (await store.claim())! + advance(DELIVERY_LEASE_MS) + await store.accept('host', 'phone', { + ...notification(2, 'dismiss'), + notificationId: alert.notificationId + }) + const read = async () => + ( + await db.query( + 'SELECT state, payload_json, lease_until FROM push_delivery_batches WHERE batch_id = ?', + [stale.id] + ) + )[0] + const cancelled = await read() + expect(cancelled).toMatchObject({ state: 'dismissed', payload_json: '{}' }) + await store.renew(stale) + expect(await read()).toEqual(cancelled) + await store.finish(stale, 1000) + expect(await read()).toEqual(cancelled) + await store.finish(stale) + expect(await read()).toEqual(cancelled) + const dismissal = (await store.claim())! + expect(dismissal.notification.kind).toBe('dismiss') + await store.finish(dismissal) + await store.accept('host', 'phone', notification(3)) + const fresh = (await store.claim())! + await store.finish(fresh, 1000) + advance(1000) + const retry = (await store.claim())! + expect(retry.id).toBe(fresh.id) + await store.finish(retry) + expect(await store.claim()).toBeNull() +}) diff --git a/cloud/apps/push/src/durable-push-store.ts b/cloud/apps/push/src/durable-push-store.ts new file mode 100644 index 00000000000..3003cc6d384 --- /dev/null +++ b/cloud/apps/push/src/durable-push-store.ts @@ -0,0 +1,194 @@ +import { isDismissedAlert, reconcileQueuedDismissal } from './push-queued-dismissal.js' +import { parsePushDeliveryPayload } from './push-delivery-payload.js' +import { createHash, randomUUID } from 'node:crypto' +import { PUSH_LIMITS, type PushNotification } from '@orca-cloud/push-contract' +import type { PushDatabase, SqlRow } from './push-database.js' + +const RETENTION_MS = 24 * 60 * 60_000 +export const DELIVERY_LEASE_MS = 30_000 +export type QueuedPushDelivery = { + id: string + registrationId: string + hostFingerprint: string + notification: PushNotification + expiresAt: number + lease: string + attempts: number +} + +export class DurablePushStore { + constructor( + private readonly database: PushDatabase, + private readonly now = Date.now + ) {} + + async accept( + host: string, + registrationId: string, + notification: PushNotification + ): Promise<'queued' | 'rate_limited' | 'error'> { + const now = this.now() + const kind = notification.kind ?? 'alert' + const eventId = createHash('sha256') + .update( + JSON.stringify([host, kind, notification.notificationEpoch, notification.notificationSeq]) + ) + .digest('hex') + const { sound: _sound, kind: _kind, ...content } = notification + const fingerprint = createHash('sha256') + .update(JSON.stringify({ kind, ...content })) + .digest('hex') + return this.database.transaction(async (tx) => { + await tx.lockQuotaScope(`push-events:${host}`) + const [existing] = await tx.query('SELECT * FROM push_events WHERE event_id = ?', [eventId]) + if (existing && existing.fingerprint !== fingerprint) return 'error' + const expiresAt = existing + ? Number(existing.expires_at) + : Math.min( + notification.expiresAt ?? Infinity, + now + PUSH_LIMITS.notificationTtlSeconds * 1000 + ) + if (expiresAt <= now) return 'error' + if (!existing) { + const [count] = await tx.query( + 'SELECT COUNT(*) AS total FROM push_events WHERE host_fingerprint = ? AND kind = ? AND created_at > ?', + [host, kind, now - PUSH_LIMITS.eventQuotaWindowMs] + ) + if (Number(count?.total ?? 0) >= PUSH_LIMITS.hostEventsPerWindow) return 'rate_limited' + await tx.query( + 'INSERT INTO push_events(event_id, host_fingerprint, kind, fingerprint, created_at, expires_at) VALUES (?, ?, ?, ?, ?, ?)', + [eventId, host, kind, fingerprint, now, expiresAt] + ) + } + const [recipient] = await tx.query( + 'SELECT event_id FROM push_event_recipients WHERE event_id = ? AND registration_id = ?', + [eventId, registrationId] + ) + if (recipient) return 'queued' + if (await reconcileQueuedDismissal(tx, host, registrationId, notification, now)) + return 'queued' + await tx.query( + `INSERT INTO push_delivery_batches(batch_id, host_fingerprint, registration_id, kind, payload_json, state, due_at, expires_at, lease_until, attempts, created_at) + VALUES (?, ?, ?, ?, ?, 'pending', ?, ?, 0, 0, ?)`, + [ + randomUUID(), + host, + registrationId, + kind, + JSON.stringify(notification), + now, + expiresAt, + now + ] + ) + await tx.query( + 'INSERT INTO push_event_recipients(event_id, registration_id, created_at) VALUES (?, ?, ?)', + [eventId, registrationId, now] + ) + return 'queued' + }) + } + + async claim(): Promise { + return this.database.transaction(async (tx) => { + await tx.lockQuotaScope('push-worker-claim') + const now = this.now() + const params = [now, now, now, now] + const predicate = + "state = 'pending' AND lease_until <= ? AND expires_at > ? AND due_at <= ? AND NOT EXISTS (SELECT 1 FROM push_delivery_batches busy WHERE busy.registration_id = push_delivery_batches.registration_id AND busy.lease_until > ?)" + let [row] = await tx.query( + `SELECT * FROM push_delivery_batches WHERE ${predicate} ORDER BY due_at, created_at, batch_id LIMIT 1`, + params + ) + if (!row) return null + await tx.lockQuotaScope(`push-events:${String(row.host_fingerprint)}`) + ;[row] = await tx.query('SELECT * FROM push_delivery_batches WHERE batch_id = ?', [ + row.batch_id + ]) + if (!row || row.state !== 'pending' || Number(row.expires_at) <= now) return null + const notification = parsePushDeliveryPayload(String(row.payload_json)) + if (await isDismissedAlert(tx, String(row.host_fingerprint), notification)) { + await tx.query( + "UPDATE push_delivery_batches SET state = 'dismissed', payload_json = '{}' WHERE batch_id = ?", + [row.batch_id] + ) + return null + } + const lease = randomUUID() + await tx.query( + 'UPDATE push_delivery_batches SET lease_token = ?, lease_until = ?, attempts = attempts + 1 WHERE batch_id = ?', + [lease, now + DELIVERY_LEASE_MS, row.batch_id] + ) + return this.delivery(row, lease) + }) + } + + private delivery(row: SqlRow, lease: string): QueuedPushDelivery { + return { + id: String(row.batch_id), + registrationId: String(row.registration_id), + hostFingerprint: String(row.host_fingerprint), + notification: parsePushDeliveryPayload(String(row.payload_json)), + expiresAt: Number(row.expires_at), + lease, + attempts: Number(row.attempts) + 1 + } + } + + async renew(delivery: QueuedPushDelivery): Promise { + await this.database.query( + "UPDATE push_delivery_batches SET lease_until = ? WHERE batch_id = ? AND lease_token = ? AND state = 'pending'", + [this.now() + DELIVERY_LEASE_MS, delivery.id, delivery.lease] + ) + } + + async finish( + delivery: QueuedPushDelivery, + retryAfterMs?: number, + outcome = 'done' + ): Promise { + const now = this.now() + const retryAt = retryAfterMs === undefined ? Infinity : now + Math.max(1000, retryAfterMs) + const retry = retryAt < delivery.expiresAt + await this.database.query( + `UPDATE push_delivery_batches SET state = ?, payload_json = ?, due_at = ?, lease_until = 0, lease_token = NULL + WHERE batch_id = ? AND lease_token = ? AND state = 'pending'`, + [ + retry ? 'pending' : retryAfterMs !== undefined ? 'expired' : outcome, + retry ? JSON.stringify(delivery.notification) : '{}', + retry ? retryAt : now, + delivery.id, + delivery.lease + ] + ) + } + + async pendingCount(registrationId: string): Promise { + const [row] = await this.database.query( + "SELECT COUNT(*) AS total FROM push_delivery_batches WHERE registration_id = ? AND state = 'pending'", + [registrationId] + ) + return Number(row?.total ?? 0) + } + + async prune(): Promise { + const now = this.now() + await this.database.query( + "UPDATE push_delivery_batches SET state = 'expired', payload_json = '{}' WHERE expires_at <= ? AND state = 'pending'", + [now] + ) + await this.database.query('DELETE FROM push_dismissed_events WHERE created_at < ?', [ + now - RETENTION_MS + ]) + await this.database.query('DELETE FROM push_delivery_batches WHERE expires_at < ?', [ + now - RETENTION_MS + ]) + await this.database.query('DELETE FROM push_event_recipients WHERE created_at < ?', [ + now - RETENTION_MS + ]) + const [result] = await this.database.query('DELETE FROM push_events WHERE created_at < ?', [ + now - RETENTION_MS + ]) + return Number(result?.changes ?? 0) + } +} diff --git a/cloud/apps/push/src/durable-push-worker.test.ts b/cloud/apps/push/src/durable-push-worker.test.ts new file mode 100644 index 00000000000..7cbd6074b36 --- /dev/null +++ b/cloud/apps/push/src/durable-push-worker.test.ts @@ -0,0 +1,221 @@ +import { afterEach, expect, it, vi } from 'vitest' +import type { PushNotification } from '@orca-cloud/push-contract' +import { DurablePushStore } from './durable-push-store.js' +import { DurablePushWorker } from './durable-push-worker.js' +import { PushDispatcher } from './push-dispatcher.js' +import { PushDeviceRegistryStore } from './device-registry-store.js' +import { openInMemoryPushDatabase } from './push-database.js' +import type { PushDelivery } from './push-delivery-message.js' +import type { PushProviderOutcome } from './push-provider-outcome.js' + +const cleanups: (() => Promise)[] = [] +afterEach(async () => { + for (const cleanup of cleanups.splice(0)) await cleanup() + vi.useRealTimers() + vi.restoreAllMocks() +}) +const note = (seq: number, overrides: Partial = {}): PushNotification => ({ + notificationId: `note-${seq}`, + notificationSeq: seq, + notificationEpoch: 'epoch', + source: 'agent-task-complete', + agentState: 'finished', + title: 'Done', + body: 'Finished task', + ...overrides +}) +async function fixture() { + const db = await openInMemoryPushDatabase() + let time = 1_000_000 + const now = () => time + const store = new DurablePushStore(db, now) + const devices = new PushDeviceRegistryStore(db, now) + const device = await devices.upsert({ + hostFingerprint: 'host', + deviceId: 'phone', + platform: 'android', + token: 'test-token' + }) + if (!device.ok) throw new Error('registration failed') + const send = vi.fn(async (_delivery: PushDelivery): Promise => ({ + status: 'sent' + })) + const onRetry = vi.fn() + const dispatcher = new PushDispatcher({ devices, fcm: { send } as never }) + const worker = new DurablePushWorker(store, dispatcher, { now, onRetry }) + cleanups.push(async () => { + await worker.stop() + await db.close() + }) + vi.spyOn(console, 'warn').mockImplementation(() => {}) + return { + db, + store, + devices, + worker, + dispatcher, + send, + onRetry, + now, + registrationId: device.registrationId, + accept: (notification: PushNotification) => + store.accept('host', device.registrationId, notification), + advance: (ms: number) => { + time += ms + } + } +} + +it('sends every burst event immediately with its original content and identity', async () => { + const h = await fixture() + await h.accept(note(1)) + await h.accept( + note(2, { agentState: 'needs-input', title: 'Answer needed', body: 'Please respond' }) + ) + await h.worker.runDue() + expect(h.send).toHaveBeenCalledTimes(2) + const first = h.send.mock.calls.find(([delivery]) => delivery.orca.notificationSeq === 1)![0] + const second = h.send.mock.calls.find(([delivery]) => delivery.orca.notificationSeq === 2)![0] + expect(first).toMatchObject({ + title: 'Done', + body: 'Finished task', + orca: { + notificationId: 'note-1', + notificationSeq: 1 + } + }) + expect(second).toMatchObject({ + title: 'Answer needed', + body: 'Please respond', + orca: { notificationId: 'note-2', notificationSeq: 2 } + }) + expect(first.collapseId).not.toBe(second.collapseId) + expect(h.send.mock.calls.every(([delivery]) => !('coalescedCount' in delivery.orca))).toBe(true) + expect(h.send.mock.calls.every(([delivery]) => !('summaryMembers' in delivery.orca))).toBe(true) + expect(h.onRetry).not.toHaveBeenCalled() +}) + +it('keeps untrackable bells and per-phone deliveries individually replaceable', async () => { + const h = await fixture() + const other = await h.devices.upsert({ + hostFingerprint: 'host', + deviceId: 'phone2', + platform: 'android', + token: 'other-token' + }) + if (!other.ok) throw new Error('registration failed') + await h.accept(note(1, { notificationId: undefined, source: 'terminal-bell', agentState: null })) + await h.accept(note(2)) + await h.accept(note(3)) + await h.store.accept('host', other.registrationId, note(2)) + await h.worker.runDue() + expect(h.send).toHaveBeenCalledTimes(4) + const deliveries = h.send.mock.calls.map(([delivery]) => delivery) + const primary = deliveries.filter((delivery) => delivery.registrationId === h.registrationId) + expect(primary).toHaveLength(3) + expect(new Set(primary.map((delivery) => delivery.collapseId)).size).toBe(3) + expect( + deliveries.find((delivery) => delivery.registrationId === other.registrationId) + ).toMatchObject({ orca: { notificationId: 'note-2', notificationSeq: 2 } }) +}) + +it('persists provider retry delay and resumes it through a new worker', async () => { + const h = await fixture() + h.send.mockResolvedValueOnce({ + status: 'error', + reason: 'busy', + retryable: true, + retryAfterMs: 10000 + }) + await h.accept(note(1)) + await h.worker.runDue() + expect(h.send).toHaveBeenCalledOnce() + await h.worker.stop() + const restarted = new DurablePushWorker(h.store, h.dispatcher, { now: h.now, onRetry: h.onRetry }) + h.advance(9999) + await restarted.runDue() + expect(h.send).toHaveBeenCalledOnce() + h.advance(1) + await restarted.runDue() + expect(h.send).toHaveBeenCalledTimes(2) + expect(h.send.mock.calls.map(([delivery]) => delivery.expiresAt)).toEqual([1_300_000, 1_300_000]) + expect(h.onRetry).toHaveBeenCalledOnce() + expect(await h.store.pendingCount(h.registrationId)).toBe(0) + await restarted.stop() +}) + +it('expires instead of shortening a provider delay beyond the delivery lifetime', async () => { + const h = await fixture() + h.send.mockResolvedValue({ + status: 'error', + reason: 'busy', + retryable: true, + retryAfterMs: 600000 + }) + await h.accept(note(1)) + await h.worker.runDue() + h.advance(600000) + await h.worker.runDue() + expect(h.send).toHaveBeenCalledOnce() + expect(await h.store.pendingCount(h.registrationId)).toBe(0) + expect(h.onRetry).not.toHaveBeenCalled() +}) + +it('rechecks the device before a persisted retry and does not send after unregistration', async () => { + const h = await fixture() + h.send.mockResolvedValue({ status: 'error', reason: 'timeout', retryable: true }) + await h.accept(note(1)) + await h.worker.runDue() + await h.devices.deleteOwned('host', h.registrationId) + h.advance(3000) + await h.worker.runDue() + expect(h.send).toHaveBeenCalledOnce() + expect(await h.store.pendingCount(h.registrationId)).toBe(0) +}) + +it('joins active work on shutdown and leaves unclaimed work for the next instance', async () => { + const h = await fixture() + let finish!: (outcome: PushProviderOutcome) => void + let started!: () => void + const entered = new Promise((resolve) => { + started = resolve + }) + h.send.mockImplementationOnce(() => { + started() + return new Promise((resolve) => { + finish = resolve + }) + }) + await h.accept(note(1)) + const pending = h.worker.runDue() + await entered + await h.accept(note(2)) + let stopped = false + const stopping = h.worker.stop().then(() => { + stopped = true + }) + await Promise.resolve() + expect(stopped).toBe(false) + finish({ status: 'sent' }) + await Promise.all([pending, stopping]) + expect(stopped).toBe(true) + expect(h.send).toHaveBeenCalledOnce() + const resumed = new DurablePushWorker(h.store, h.dispatcher, { now: h.now }) + await resumed.runDue() + expect(h.send).toHaveBeenCalledTimes(2) + await resumed.stop() +}) + +it('runs due work on its timer and releases the timer on stop', async () => { + const h = await fixture() + vi.useFakeTimers() + await h.accept(note(1)) + h.worker.start() + h.worker.start() + expect(vi.getTimerCount()).toBe(1) + await vi.advanceTimersByTimeAsync(1000) + await h.worker.runDue() + expect(h.send).toHaveBeenCalledOnce() + await h.worker.stop() + expect(vi.getTimerCount()).toBe(0) +}) diff --git a/cloud/apps/push/src/durable-push-worker.ts b/cloud/apps/push/src/durable-push-worker.ts new file mode 100644 index 00000000000..4efe329bfda --- /dev/null +++ b/cloud/apps/push/src/durable-push-worker.ts @@ -0,0 +1,89 @@ +import { buildPushDelivery } from './push-delivery-message.js' +import type { PushDispatcher } from './push-dispatcher.js' +import type { DurablePushStore } from './durable-push-store.js' + +export class DurablePushWorker { + private timer?: NodeJS.Timeout + private running: Promise | null = null + private stopped = false + constructor( + private readonly store: DurablePushStore, + private readonly dispatcher: PushDispatcher, + private readonly options: { now?: () => number; onRetry?: () => void } = {} + ) {} + + start(): void { + if (this.timer) return + this.stopped = false + this.timer = setInterval(() => { + void this.runDue().catch(() => { + console.warn(JSON.stringify({ event: 'orca_push_worker_failed' })) + }) + }, 1000) + this.timer.unref() + } + + async runDue(): Promise { + if (this.running) { + await this.running + return + } + if (this.stopped) return + const pending = Promise.allSettled(Array.from({ length: 4 }, () => this.drain())).then( + (results) => { + const failure = results.find((result) => result.status === 'rejected') + if (failure?.status === 'rejected') throw failure.reason + } + ) + this.running = pending + try { + await pending + } finally { + this.running = null + } + } + + private async drain(): Promise { + for (let count = 0; count < 25 && !this.stopped; count++) { + const queued = await this.store.claim() + if (!queued) return + const delivery = buildPushDelivery({ + expiresAt: queued.expiresAt, + registrationId: queued.registrationId, + hostFingerprint: queued.hostFingerprint, + notification: queued.notification + }) + if ((this.options.now ?? Date.now)() >= queued.expiresAt) { + await this.store.finish(queued) + continue + } + const heartbeat = setInterval(() => { + void this.store.renew(queued).catch(() => {}) + }, 10_000) + heartbeat.unref() + try { + if (queued.attempts > 1) this.options.onRetry?.() + const outcome = await this.dispatcher.sendOnce(delivery) + const retryAfterMs = + outcome.status === 'error' && outcome.retryable + ? Math.max( + outcome.retryAfterMs ?? 0, + Math.min(30_000, 1000 * 2 ** Math.min(queued.attempts, 5)) + ) + : undefined + await this.store.finish(queued, retryAfterMs, outcome.status) + } catch { + await this.store.finish(queued, 5000) + } finally { + clearInterval(heartbeat) + } + } + } + + async stop(): Promise { + this.stopped = true + if (this.timer) clearInterval(this.timer) + this.timer = undefined + await this.running + } +} diff --git a/cloud/apps/push/src/fcm-access-token.ts b/cloud/apps/push/src/fcm-access-token.ts new file mode 100644 index 00000000000..542e0e8d0ed --- /dev/null +++ b/cloud/apps/push/src/fcm-access-token.ts @@ -0,0 +1,15 @@ +import { GoogleAuth } from 'google-auth-library' +import { FCM_SCOPE } from './fcm-client.js' + +// Resolves the runtime service account credential from the GCE metadata server +// in Cloud Run and from GOOGLE_APPLICATION_CREDENTIALS locally; the library +// caches and refreshes the token itself. +export function createFcmAccessTokenProvider(): () => Promise { + const auth = new GoogleAuth({ scopes: [FCM_SCOPE] }) + return async () => { + const client = await auth.getClient() + const token = await client.getAccessToken() + if (!token.token) throw new Error('fcm_access_token_unavailable') + return token.token + } +} diff --git a/cloud/apps/push/src/fcm-client.test.ts b/cloud/apps/push/src/fcm-client.test.ts new file mode 100644 index 00000000000..8843e7aab95 --- /dev/null +++ b/cloud/apps/push/src/fcm-client.test.ts @@ -0,0 +1,229 @@ +import { createHash } from 'node:crypto' +import { describe, expect, it } from 'vitest' +import { FcmClient, type FcmRequest, type FcmResponse } from './fcm-client.js' +import { buildPushDelivery } from './push-delivery-message.js' + +const NOW = 1_700_000_000_000 +const HOST = 'abcdefghijklmnop' +const TOKEN = 'cQ1abcDEF_gh:APA91bZZ-zz0123456789abcdefghijklmnopqrstuvwxyz' + +function delivery(agentState: 'needs-input' | null = 'needs-input') { + return buildPushDelivery({ + expiresAt: NOW + 300_000, + registrationId: 'reg-1', + hostFingerprint: HOST, + notification: { + notificationId: 'note-1', + notificationSeq: 7, + notificationEpoch: 'epoch-1', + source: 'agent-task-complete', + agentState, + title: 'Agent needs input', + body: 'Waiting on your answer', + paneKey: 'tab-b:pane-1', + worktreeId: 'wt-1' + } + }) +} + +function fakeTransport(response: FcmResponse) { + const requests: FcmRequest[] = [] + return { + requests, + transport: async (request: FcmRequest): Promise => { + requests.push(request) + return response + } + } +} + +function client(response: FcmResponse) { + const fake = fakeTransport(response) + return { + fake, + client: new FcmClient({ + projectId: 'onorca-cloud', + now: () => NOW, + accessToken: async () => 'access-token', + transport: fake.transport + }) + } +} + +describe('fcm client', () => { + it('posts the v1 send payload for the configured project', async () => { + const { fake, client: fcm } = client({ status: 200, body: '{"name":"projects/x/messages/1"}' }) + await expect(fcm.send(delivery(), { token: TOKEN })).resolves.toEqual({ status: 'sent' }) + const request = fake.requests[0]! + expect(request.url).toBe('https://fcm.googleapis.com/v1/projects/onorca-cloud/messages:send') + expect(request.accessToken).toBe('access-token') + expect(JSON.parse(request.body)).toEqual({ + message: { + token: TOKEN, + android: { priority: 'HIGH', ttl: '300s' }, + data: { + title: 'Agent needs input', + message: 'Waiting on your answer', + tag: delivery().collapseId, + channelId: 'orca-desktop', + hostFingerprint: HOST, + paneKey: 'tab-b:pane-1', + worktreeId: 'wt-1', + notificationId: 'note-1', + notificationSeq: '7', + notificationEpoch: 'epoch-1', + source: 'agent-task-complete', + agentState: 'needs-input' + } + } + }) + }) + + it('carries every data value as a string and omits a null agent state', async () => { + const { fake, client: fcm } = client({ status: 200, body: '{}' }) + await fcm.send(delivery(null), { token: TOKEN }) + const message = JSON.parse(fake.requests[0]!.body) as { + message: { + android: Record + data: Record + } + } + expect(Object.values(message.message.data).every((value) => typeof value === 'string')).toBe( + true + ) + expect(message.message.data.agentState).toBeUndefined() + const tag = createHash('sha256') + .update(JSON.stringify([HOST, 'note-1'])) + .digest('hex') + expect(message.message.data.coalescedCount).toBeUndefined() + expect(message.message.data.tag).toBe(tag) + expect(message.message.android).not.toHaveProperty('collapse_key') + expect(message.message).not.toHaveProperty('notification') + expect(message.message.data).not.toHaveProperty('body') + }) + + it('marks an unregistered token dead from the status or the error detail', async () => { + const byStatus = client({ + status: 404, + body: JSON.stringify({ error: { status: 'UNREGISTERED', message: 'not registered' } }) + }) + await expect(byStatus.client.send(delivery(), { token: TOKEN })).resolves.toEqual({ + status: 'dead', + reason: 'UNREGISTERED' + }) + const byDetail = client({ + status: 404, + body: JSON.stringify({ + error: { + status: 'NOT_FOUND', + message: 'Requested entity was not found.', + details: [{ errorCode: 'UNREGISTERED' }] + } + }) + }) + await expect(byDetail.client.send(delivery(), { token: TOKEN })).resolves.toEqual({ + status: 'dead', + reason: 'UNREGISTERED' + }) + }) + + it('marks an invalid-argument that names the token dead, and others an error', async () => { + const named = client({ + status: 400, + body: JSON.stringify({ + error: { status: 'INVALID_ARGUMENT', message: 'The registration token is not valid.' } + }) + }) + await expect(named.client.send(delivery(), { token: TOKEN })).resolves.toEqual({ + status: 'dead', + reason: 'INVALID_ARGUMENT' + }) + const unnamed = client({ + status: 400, + body: JSON.stringify({ + error: { status: 'INVALID_ARGUMENT', message: 'Invalid value at message.android.ttl' } + }) + }) + await expect(unnamed.client.send(delivery(), { token: TOKEN })).resolves.toEqual({ + status: 'error', + reason: 'INVALID_ARGUMENT', + retryable: false, + retryAfterMs: 10000 + }) + }) + + it('treats a server fault and a transport failure as errors', async () => { + const faulted = client({ + status: 503, + body: JSON.stringify({ error: { status: 'UNAVAILABLE', message: 'backend busy' } }) + }) + await expect(faulted.client.send(delivery(), { token: TOKEN })).resolves.toEqual({ + status: 'error', + reason: 'UNAVAILABLE', + retryable: true, + retryAfterMs: 10000 + }) + const broken = new FcmClient({ + projectId: 'onorca-cloud', + now: () => NOW, + accessToken: async () => 'access-token', + transport: async () => { + throw new Error('ECONNRESET') + } + }) + await expect(broken.send(delivery(), { token: TOKEN })).resolves.toEqual({ + status: 'error', + reason: 'Error', + retryable: true + }) + }) +}) + +it('does not send when credential refresh crosses the absolute expiry', async () => { + let now = 1000 + const fake = fakeTransport({ status: 200, body: '{}' }) + const fcm = new FcmClient({ + projectId: 'test', + now: () => now, + accessToken: async () => { + now = 3000 + return 'test-token' + }, + transport: fake.transport + }) + await expect(fcm.send({ ...delivery(), expiresAt: 2000 }, { token: TOKEN })).resolves.toEqual({ + status: 'error', + reason: 'expired' + }) + expect(fake.requests).toHaveLength(0) +}) + +it('decreases retry TTL and refuses expired delivery before refreshing credentials', async () => { + let now = NOW + let refreshes = 0 + const fake = fakeTransport({ status: 503, body: '{}' }) + const fcm = new FcmClient({ + projectId: 'test', + now: () => now, + accessToken: async () => { + refreshes++ + return 'test-token' + }, + transport: fake.transport + }) + const pending = delivery() + await fcm.send(pending, { token: TOKEN }) + now += 60_000 + await fcm.send(pending, { token: TOKEN }) + expect(fake.requests.map((request) => JSON.parse(request.body).message.android.ttl)).toEqual([ + '300s', + '240s' + ]) + now = pending.expiresAt + await expect(fcm.send(pending, { token: TOKEN })).resolves.toEqual({ + status: 'error', + reason: 'expired' + }) + expect(fake.requests).toHaveLength(2) + expect(refreshes).toBe(2) +}) diff --git a/cloud/apps/push/src/fcm-client.ts b/cloud/apps/push/src/fcm-client.ts new file mode 100644 index 00000000000..0b58aae80f5 --- /dev/null +++ b/cloud/apps/push/src/fcm-client.ts @@ -0,0 +1,139 @@ +import { providerRetryAfter } from './provider-retry-delay.js' +import { PUSH_DEFAULTS } from '@orca-cloud/push-contract' +import { orcaDataStrings, type PushDelivery } from './push-delivery-message.js' +import type { PushProviderOutcome } from './push-provider-outcome.js' + +export const FCM_SCOPE = 'https://www.googleapis.com/auth/firebase.messaging' + +export type FcmRequest = { url: string; accessToken: string; body: string } +export type FcmResponse = { status: number; body: string; retryAfterMs?: number } +export type FcmTransport = (request: FcmRequest) => Promise + +export type FcmClientOptions = { + projectId: string + accessToken: () => Promise + transport: FcmTransport + channelId?: string + now?: () => number +} + +type FcmErrorBody = { + error?: { status?: unknown; message?: unknown; details?: { errorCode?: unknown }[] } +} + +export function fcmMessageBody(input: { + delivery: PushDelivery + token: string + channelId: string + now?: number +}): string { + const { delivery } = input + const now = input.now ?? Date.now() + return JSON.stringify({ + message: { + token: input.token, + android: { + priority: 'HIGH', + ttl: `${Math.max(0, Math.ceil((delivery.expiresAt - now) / 1000))}s` + }, + // Notification payloads collapse offline; Expo renders these data messages natively. + data: { + ...orcaDataStrings(delivery.orca), + ...(delivery.orca.kind === 'dismiss' + ? {} + : { + title: delivery.title, + message: delivery.body, + tag: delivery.collapseId, + channelId: delivery.sound === false ? `${input.channelId}-silent` : input.channelId, + ...(delivery.sound === false ? { sound: '' } : {}) + }) + } + } + }) +} + +function readFcmError(body: string): { status: string; message: string; errorCodes: string[] } { + try { + const parsed = JSON.parse(body) as FcmErrorBody + return { + status: typeof parsed.error?.status === 'string' ? parsed.error.status : 'unknown', + message: typeof parsed.error?.message === 'string' ? parsed.error.message : '', + errorCodes: (parsed.error?.details ?? []) + .map((detail) => detail.errorCode) + .filter((code): code is string => typeof code === 'string') + } + } catch { + return { status: 'unparseable', message: '', errorCodes: [] } + } +} + +export class FcmClient { + private readonly channelId: string + + constructor(private readonly options: FcmClientOptions) { + this.channelId = options.channelId ?? PUSH_DEFAULTS.androidChannelId + } + + async send(delivery: PushDelivery, device: { token: string }): Promise { + if (delivery.expiresAt <= (this.options.now ?? Date.now)()) + return { status: 'error', reason: 'expired' } + let response: FcmResponse + try { + const accessToken = await this.options.accessToken() + const now = (this.options.now ?? Date.now)() + if (delivery.expiresAt <= now) return { status: 'error', reason: 'expired' } + response = await this.options.transport({ + url: `https://fcm.googleapis.com/v1/projects/${this.options.projectId}/messages:send`, + accessToken, + body: fcmMessageBody({ + delivery, + token: device.token, + channelId: this.channelId, + now + }) + }) + } catch (error) { + return { + status: 'error', + reason: error instanceof Error ? error.name : 'transport_failed', + retryable: true + } + } + if (response.status >= 200 && response.status < 300) return { status: 'sent' } + const failure = readFcmError(response.body) + if (failure.status === 'UNREGISTERED' || failure.errorCodes.includes('UNREGISTERED')) { + return { status: 'dead', reason: 'UNREGISTERED' } + } + // A revoked token also surfaces as INVALID_ARGUMENT naming the token field. + if (failure.status === 'INVALID_ARGUMENT' && /\btoken\b/i.test(failure.message)) { + return { status: 'dead', reason: 'INVALID_ARGUMENT' } + } + return { + status: 'error', + reason: failure.status, + retryable: response.status === 429 || response.status >= 500, + retryAfterMs: Math.max(response.status === 429 ? 60_000 : 10_000, response.retryAfterMs ?? 0) + } + } +} + +export function createFcmFetchTransport(fetchImpl: typeof fetch = fetch): FcmTransport { + return async (request) => { + const response = await fetchImpl(request.url, { + method: 'POST', + headers: { + authorization: `Bearer ${request.accessToken}`, + 'content-type': 'application/json' + }, + body: request.body, + redirect: 'error', + signal: AbortSignal.timeout(10_000) + }) + return { + status: response.status, + body: await response.text(), + retryAfterMs: providerRetryAfter(response.headers.get('retry-after') ?? undefined) + } + } +} diff --git a/cloud/apps/push/src/host-challenge-answering.test-fixture.ts b/cloud/apps/push/src/host-challenge-answering.test-fixture.ts new file mode 100644 index 00000000000..4dec1e48c5b --- /dev/null +++ b/cloud/apps/push/src/host-challenge-answering.test-fixture.ts @@ -0,0 +1,163 @@ +import { createHmac, timingSafeEqual } from 'node:crypto' +import { + PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN, + PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN, + PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT, + PUSH_LIMITS +} from '@orca-cloud/push-contract' +import nacl from 'tweetnacl' +import { decodeCanonicalBase64 } from './canonical-base64.js' +import { deriveHostFingerprint } from './host-fingerprint.js' + +// The desktop side of the push challenge, written the way the shipped host +// will answer it, so the gateway is exercised against a real box-opening peer. +const textEncoder = new TextEncoder() +const textDecoder = new TextDecoder() + +export type PushHostKeypair = { publicKey: Uint8Array; secretKey: Uint8Array } + +export type PushChallengeWire = { + challengeId: string + gatewayEphemeralPublicKeyB64: string + nonceB64: string + ciphertextB64: string + expiresAt: number +} + +export function createPushHostKeypair(seed?: number): PushHostKeypair { + const pair = + seed === undefined + ? nacl.box.keyPair() + : nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(seed)) + return { publicKey: pair.publicKey, secretKey: pair.secretKey } +} + +export function hostPublicKeyB64(keypair: PushHostKeypair): string { + return Buffer.from(keypair.publicKey).toString('base64') +} + +function equal(left: Uint8Array | undefined, right: Uint8Array): boolean { + return Boolean(left && left.byteLength === right.byteLength && timingSafeEqual(left, right)) +} + +function uint64(value: number): Uint8Array { + const bytes = new Uint8Array(8) + new DataView(bytes.buffer).setBigUint64(0, BigInt(value), false) + return bytes +} + +function parseTranscript(transcript: Uint8Array): Map | null { + const fields = new Map() + const view = new DataView(transcript.buffer, transcript.byteOffset, transcript.byteLength) + let offset = 0 + try { + while (offset < transcript.byteLength) { + const nameLength = view.getUint32(offset, false) + offset += 4 + const name = textDecoder.decode(transcript.slice(offset, offset + nameLength)) + offset += nameLength + const valueLength = view.getUint32(offset, false) + offset += 4 + if (fields.has(name) || offset + valueLength > transcript.byteLength) return null + fields.set(name, transcript.slice(offset, offset + valueLength)) + offset += valueLength + } + } catch { + return null + } + return offset === transcript.byteLength ? fields : null +} + +function readUint64(value: Uint8Array | undefined): number | null { + if (!value || value.byteLength !== 8) return null + const parsed = new DataView(value.buffer, value.byteOffset, value.byteLength).getBigUint64(0, false) + return parsed <= BigInt(Number.MAX_SAFE_INTEGER) ? Number(parsed) : null +} + +export type PushHostProofContext = { + gatewayOrigin: string + keypair: PushHostKeypair + now?: () => number + onInvalid?: (reason: string) => void +} + +function validateTranscript( + transcript: Uint8Array, + challenge: PushChallengeWire, + context: PushHostProofContext, + gatewayKey: Uint8Array, + nonce: Uint8Array +): boolean { + const fields = parseTranscript(transcript) + if (!fields || fields.size !== PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT) { + context.onInvalid?.('transcript-structure') + return false + } + const now = (context.now ?? Date.now)() + const issuedAt = readUint64(fields.get('issuedAt')) + const expiresAt = readUint64(fields.get('expiresAt')) + const fingerprint = deriveHostFingerprint(context.keypair.publicKey) + const checks: [string, boolean][] = [ + ['issuedAt-readable', issuedAt !== null], + [ + 'issuedAt-not-future', + issuedAt === null || issuedAt - PUSH_LIMITS.clockSkewToleranceMs <= now + ], + ['not-expired', now - PUSH_LIMITS.clockSkewToleranceMs <= challenge.expiresAt], + ['issuedAt-before-expiry', issuedAt === null || issuedAt <= challenge.expiresAt], + [ + 'window', + issuedAt === null || challenge.expiresAt - issuedAt <= PUSH_LIMITS.challengeTtlMs + ], + ['expiry-consistent', expiresAt === challenge.expiresAt], + ['protocol', equal(fields.get('protocol'), textEncoder.encode(PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN))], + ['version', equal(fields.get('version'), new Uint8Array([1]))], + ['gatewayOrigin', equal(fields.get('gatewayOrigin'), textEncoder.encode(context.gatewayOrigin))], + ['gatewayEphemeralPublicKey', equal(fields.get('gatewayEphemeralPublicKey'), gatewayKey)], + ['challengeNonce', equal(fields.get('challengeNonce'), nonce)], + ['challengeId', equal(fields.get('challengeId'), textEncoder.encode(challenge.challengeId))], + ['hostFingerprint', equal(fields.get('hostFingerprint'), textEncoder.encode(fingerprint))], + ['hostPublicKey', equal(fields.get('hostPublicKey'), context.keypair.publicKey)], + ['issuedAt-value', issuedAt === null || uint64(issuedAt).byteLength === 8] + ] + const failed = checks.filter(([, ok]) => !ok).map(([name]) => name) + if (failed.length === 0) return true + context.onInvalid?.(`transcript:${failed.join('+')}`) + return false +} + +export function answerPushHostChallenge( + challenge: PushChallengeWire, + context: PushHostProofContext +): string | null { + const gatewayKey = decodeCanonicalBase64(challenge.gatewayEphemeralPublicKeyB64, 32) + const nonce = decodeCanonicalBase64(challenge.nonceB64, 24) + const ciphertext = Buffer.from(challenge.ciphertextB64, 'base64') + if (!gatewayKey || !nonce || ciphertext.toString('base64') !== challenge.ciphertextB64) return null + const plaintext = nacl.box.open(ciphertext, nonce, gatewayKey, context.keypair.secretKey) + if (!plaintext) { + context.onInvalid?.('challenge-box-open') + return null + } + const domain = textEncoder.encode(`${PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`) + if ( + !equal(plaintext.slice(0, domain.byteLength), domain) || + plaintext.byteLength < domain.byteLength + 36 + ) { + return null + } + const transcriptLength = new DataView( + plaintext.buffer, + plaintext.byteOffset + domain.byteLength, + 4 + ).getUint32(0, false) + const transcriptStart = domain.byteLength + 4 + const secretStart = transcriptStart + transcriptLength + if (secretStart + 32 !== plaintext.byteLength) return null + const transcript = plaintext.slice(transcriptStart, secretStart) + if (!validateTranscript(transcript, challenge, context, gatewayKey, nonce)) return null + return createHmac('sha256', plaintext.slice(secretStart)) + .update(textEncoder.encode(`${PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN}\0ack\0`)) + .update(transcript) + .digest('base64') +} diff --git a/cloud/apps/push/src/host-challenge-store.test.ts b/cloud/apps/push/src/host-challenge-store.test.ts new file mode 100644 index 00000000000..c5aa7928ba9 --- /dev/null +++ b/cloud/apps/push/src/host-challenge-store.test.ts @@ -0,0 +1,191 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + answerPushHostChallenge, + createPushHostKeypair, + hostPublicKeyB64 +} from './host-challenge-answering.test-fixture.js' +import { PushHostChallengeStore } from './host-challenge-store.js' +import { deriveHostFingerprint } from './host-fingerprint.js' +import { openInMemoryPushDatabase, type PushDatabase } from './push-database.js' + +const GATEWAY_ORIGIN = 'https://push.onorca.dev' + +describe('push host challenge store', () => { + let database: PushDatabase + let clock = 1_700_000_000_000 + let store: PushHostChallengeStore + + beforeEach(async () => { + database = await openInMemoryPushDatabase() + clock = 1_700_000_000_000 + store = new PushHostChallengeStore(database, GATEWAY_ORIGIN, () => clock) + }) + + afterEach(async () => { + await database.close() + }) + + it('completes a challenge, proof, and consume round trip', async () => { + const host = createPushHostKeypair(1) + const challenge = await store.issue(hostPublicKeyB64(host)) + expect(challenge).not.toBeNull() + expect(challenge!.expiresAt).toBe(clock + PUSH_LIMITS.challengeTtlMs) + expect(challenge!.hostFingerprint).toBe(deriveHostFingerprint(host.publicKey)) + + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + }) + expect(proof).not.toBeNull() + await expect(store.verify(challenge!.challengeId, proof!)).resolves.toEqual({ + ok: true, + hostFingerprint: deriveHostFingerprint(host.publicKey) + }) + }) + + it('never stores material that reproduces the proof', async () => { + const host = createPushHostKeypair(2) + const challenge = await store.issue(hostPublicKeyB64(host)) + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + }) + const [row] = await database.query('SELECT secret_hash FROM push_challenges') + expect(String(row?.secret_hash)).not.toBe(proof) + expect(Buffer.from(String(row?.secret_hash), 'base64url').byteLength).toBe(32) + }) + + it('rejects a replayed challenge', async () => { + const host = createPushHostKeypair(3) + const challenge = await store.issue(hostPublicKeyB64(host)) + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + })! + await expect(store.verify(challenge!.challengeId, proof)).resolves.toMatchObject({ ok: true }) + await expect(store.verify(challenge!.challengeId, proof)).resolves.toEqual({ + ok: false, + reason: 'already_consumed' + }) + }) + + it('rejects a challenge the moment its own ttl elapses', async () => { + const host = createPushHostKeypair(4) + const challenge = await store.issue(hostPublicKeyB64(host)) + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + })! + clock += PUSH_LIMITS.challengeTtlMs + 1 + await expect(store.verify(challenge!.challengeId, proof)).resolves.toEqual({ + ok: false, + reason: 'expired' + }) + }) + + it('spends no skew tolerance on its own expiry, so the ttl is the whole window', async () => { + const host = createPushHostKeypair(5) + const challenge = await store.issue(hostPublicKeyB64(host)) + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + })! + // A proof that the host would still consider in-window is refused here: the + // gateway issued expires_at against this clock and needs no allowance. + clock += PUSH_LIMITS.challengeTtlMs + PUSH_LIMITS.clockSkewToleranceMs - 1 + await expect(store.verify(challenge!.challengeId, proof)).resolves.toEqual({ + ok: false, + reason: 'expired' + }) + }) + + it('accepts a proof that lands just inside the ttl', async () => { + const host = createPushHostKeypair(26) + const challenge = await store.issue(hostPublicKeyB64(host)) + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + })! + clock += PUSH_LIMITS.challengeTtlMs + await expect(store.verify(challenge!.challengeId, proof)).resolves.toMatchObject({ ok: true }) + }) + + it('keeps an expired row long enough to answer expired rather than unknown', async () => { + const host = createPushHostKeypair(27) + const challenge = await store.issue(hostPublicKeyB64(host)) + const proof = answerPushHostChallenge(challenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: host, + now: () => clock + })! + clock += PUSH_LIMITS.challengeTtlMs + 1 + expect(await store.pruneExpired()).toBe(0) + await expect(store.verify(challenge!.challengeId, proof)).resolves.toEqual({ + ok: false, + reason: 'expired' + }) + }) + + it('refuses a wrong host: the box will not open and a foreign proof will not match', async () => { + const owner = createPushHostKeypair(6) + const intruder = createPushHostKeypair(7) + const ownerChallenge = await store.issue(hostPublicKeyB64(owner)) + expect( + answerPushHostChallenge(ownerChallenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: intruder, + now: () => clock + }) + ).toBeNull() + + const intruderChallenge = await store.issue(hostPublicKeyB64(intruder)) + const intruderProof = answerPushHostChallenge(intruderChallenge!, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair: intruder, + now: () => clock + })! + await expect(store.verify(ownerChallenge!.challengeId, intruderProof)).resolves.toEqual({ + ok: false, + reason: 'proof_mismatch' + }) + }) + + it('rejects a proof bound to a different gateway origin', async () => { + const host = createPushHostKeypair(8) + const challenge = await store.issue(hostPublicKeyB64(host)) + const reasons: string[] = [] + expect( + answerPushHostChallenge(challenge!, { + gatewayOrigin: 'https://push.example.test', + keypair: host, + now: () => clock, + onInvalid: (reason) => reasons.push(reason) + }) + ).toBeNull() + expect(reasons.join()).toContain('gatewayOrigin') + }) + + it('rejects an unknown challenge id and a malformed public key', async () => { + await expect(store.verify('missing', Buffer.alloc(32, 9).toString('base64'))).resolves.toEqual({ + ok: false, + reason: 'unknown_challenge' + }) + await expect(store.issue('not-base64!!')).resolves.toBeNull() + await expect(store.issue(Buffer.alloc(31, 1).toString('base64'))).resolves.toBeNull() + }) + + it('prunes challenges that fell out of the skew window', async () => { + const host = createPushHostKeypair(9) + await store.issue(hostPublicKeyB64(host)) + expect(await store.pruneExpired()).toBe(0) + clock += PUSH_LIMITS.challengeTtlMs + PUSH_LIMITS.clockSkewToleranceMs + 1 + expect(await store.pruneExpired()).toBe(1) + }) +}) diff --git a/cloud/apps/push/src/host-challenge-store.ts b/cloud/apps/push/src/host-challenge-store.ts new file mode 100644 index 00000000000..fd12af85983 --- /dev/null +++ b/cloud/apps/push/src/host-challenge-store.ts @@ -0,0 +1,135 @@ +import { createHash, createHmac, randomBytes, randomUUID, timingSafeEqual } from 'node:crypto' +import { + buildPushHostChallengePlaintext, + buildPushHostProofMacInput, + buildPushHostProofTranscript, + PUSH_LIMITS +} from '@orca-cloud/push-contract' +import nacl from 'tweetnacl' +import { decodeCanonicalBase64 } from './canonical-base64.js' +import { deriveHostFingerprint } from './host-fingerprint.js' +import type { PushDatabase } from './push-database.js' + +export type IssuedPushChallenge = { + challengeId: string + gatewayEphemeralPublicKeyB64: string + nonceB64: string + ciphertextB64: string + expiresAt: number + hostFingerprint: string +} + +export type PushProofVerification = + | { ok: true; hostFingerprint: string } + | { ok: false; reason: 'unknown_challenge' | 'already_consumed' | 'expired' | 'proof_mismatch' } + +function sha256(value: Uint8Array): string { + return createHash('sha256').update(value).digest('base64url') +} + +function equalDigest(left: string, right: string): boolean { + const leftBytes = Buffer.from(left) + const rightBytes = Buffer.from(right) + return leftBytes.length === rightBytes.length && timingSafeEqual(leftBytes, rightBytes) +} + +export class PushHostChallengeStore { + constructor( + private readonly database: PushDatabase, + private readonly gatewayOrigin: string, + private readonly now: () => number = Date.now + ) {} + + async issue(hostPublicKeyB64: string): Promise { + const hostPublicKey = decodeCanonicalBase64(hostPublicKeyB64, 32) + if (!hostPublicKey) return null + const hostFingerprint = deriveHostFingerprint(hostPublicKey) + const ephemeral = nacl.box.keyPair() + const challengeNonce = randomBytes(nacl.box.nonceLength) + const challengeSecret = randomBytes(32) + const challengeId = randomUUID() + const issuedAt = this.now() + const expiresAt = issuedAt + PUSH_LIMITS.challengeTtlMs + const transcript = buildPushHostProofTranscript({ + gatewayOrigin: this.gatewayOrigin, + gatewayEphemeralPublicKey: ephemeral.publicKey, + challengeNonce, + challengeId, + issuedAt, + expiresAt, + hostFingerprint, + hostPublicKey + }) + const ciphertext = nacl.box( + buildPushHostChallengePlaintext(transcript, challengeSecret), + challengeNonce, + hostPublicKey, + ephemeral.secretKey + ) + const expectedProof = createHmac('sha256', challengeSecret) + .update(buildPushHostProofMacInput(transcript)) + .digest() + await this.database.query( + `INSERT INTO push_challenges + (challenge_id, host_fingerprint, secret_hash, expires_at, + consumed_at) + VALUES (?, ?, ?, ?, NULL)`, + [ + challengeId, + hostFingerprint, + // The stored digest is of the ack the secret produces, never of the + // secret itself: a database reader must not be able to forge a proof. + sha256(expectedProof), + expiresAt + ] + ) + return { + challengeId, + gatewayEphemeralPublicKeyB64: Buffer.from(ephemeral.publicKey).toString('base64'), + nonceB64: Buffer.from(challengeNonce).toString('base64'), + ciphertextB64: Buffer.from(ciphertext).toString('base64'), + expiresAt, + hostFingerprint + } + } + + async verify(challengeId: string, proofB64: string): Promise { + const proof = decodeCanonicalBase64(proofB64, 32) + return await this.database.transaction(async (transaction) => { + const [row] = await transaction.query( + `SELECT host_fingerprint, secret_hash, expires_at, consumed_at + FROM push_challenges WHERE challenge_id = ?`, + [challengeId] + ) + if (!row) return { ok: false, reason: 'unknown_challenge' } + if (row.consumed_at !== null && row.consumed_at !== undefined) { + return { ok: false, reason: 'already_consumed' } + } + const now = this.now() + // No skew allowance here: the gateway set expires_at from this same clock. + // The tolerance belongs to the host, which validates a foreign timestamp. + if (now > Number(row.expires_at)) return { ok: false, reason: 'expired' } + if (!proof || !equalDigest(sha256(proof), String(row.secret_hash))) { + return { ok: false, reason: 'proof_mismatch' } + } + // Consume under the same predicate the read used, so two concurrent + // proofs for one challenge cannot both mint a session. + const [consumed] = await transaction.query( + 'UPDATE push_challenges SET consumed_at = ? WHERE challenge_id = ? AND consumed_at IS NULL', + [now, challengeId] + ) + if (Number(consumed?.changes ?? 0) !== 1) return { ok: false, reason: 'already_consumed' } + return { ok: true, hostFingerprint: String(row.host_fingerprint) } + }) + } + + // Rows outlive the expiry check by the skew tolerance so a late proof reads + // as 'expired' rather than as an unknown challenge. + async pruneExpired(): Promise { + const cutoff = this.now() - PUSH_LIMITS.clockSkewToleranceMs + const [result] = await this.database.query('DELETE FROM push_challenges WHERE expires_at < ?', [ + cutoff + ]) + return Number(result?.changes ?? 0) + } +} diff --git a/cloud/apps/push/src/host-fingerprint.ts b/cloud/apps/push/src/host-fingerprint.ts new file mode 100644 index 00000000000..955b1ac8ecb --- /dev/null +++ b/cloud/apps/push/src/host-fingerprint.ts @@ -0,0 +1,16 @@ +import { createHash } from 'node:crypto' +import { PUSH_HOST_FINGERPRINT_LENGTH } from '@orca-cloud/push-contract' + +// Identical derivation to deriveRelayHostId on the desktop, so a host and a +// phone reach the same fingerprint from the same X25519 public key. +export function deriveHostFingerprint(hostPublicKey: Uint8Array): string { + return createHash('sha256') + .update(hostPublicKey) + .digest('base64url') + .slice(0, PUSH_HOST_FINGERPRINT_LENGTH) +} + +// Logs may carry at most this much of a fingerprint. +export function fingerprintLogPrefix(hostFingerprint: string): string { + return hostFingerprint.slice(0, 4) +} diff --git a/cloud/apps/push/src/host-proof-concurrency.test.ts b/cloud/apps/push/src/host-proof-concurrency.test.ts new file mode 100644 index 00000000000..031ca61c7fd --- /dev/null +++ b/cloud/apps/push/src/host-proof-concurrency.test.ts @@ -0,0 +1,47 @@ +import { expect, it } from 'vitest' +import { openInMemoryPushDatabase, openPushDatabase } from './push-database.js' +import { PushHostChallengeStore } from './host-challenge-store.js' +import { + answerPushHostChallenge, + createPushHostKeypair, + hostPublicKeyB64 +} from './host-challenge-answering.test-fixture.js' + +it('accepts independent proofs but consumes each challenge only once under concurrency', async () => { + const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL + if (databaseUrl && !process.env.CI && new URL(databaseUrl).port !== '55440') { + throw new Error('isolated_postgres_port_required') + } + const db = databaseUrl + ? await openPushDatabase({ databaseUrl, dataDir: '', poolMax: 4 }) + : await openInMemoryPushDatabase() + const host = createPushHostKeypair() + const origin = 'https://push.onorca.dev' + const store = new PushHostChallengeStore(db, origin) + const challenges = await Promise.all([ + store.issue(hostPublicKeyB64(host)), + store.issue(hostPublicKeyB64(host)) + ]) + try { + const proofs = challenges.map((challenge) => + answerPushHostChallenge(challenge!, { gatewayOrigin: origin, keypair: host })! + ) + const results = await Promise.all( + challenges.flatMap((challenge, index) => + Array.from({ length: 5 }, () => store.verify(challenge!.challengeId, proofs[index]!)) + ) + ) + expect(results.filter((result) => result.ok)).toEqual([ + { ok: true, hostFingerprint: challenges[0]!.hostFingerprint }, + { ok: true, hostFingerprint: challenges[0]!.hostFingerprint } + ]) + expect(results.filter((result) => !result.ok)).toEqual( + Array.from({ length: 8 }, () => ({ ok: false, reason: 'already_consumed' })) + ) + } finally { + for (const challenge of challenges) { + await db.query('DELETE FROM push_challenges WHERE challenge_id = ?', [challenge!.challengeId]) + } + await db.close() + } +}) diff --git a/cloud/apps/push/src/host-session-store.test.ts b/cloud/apps/push/src/host-session-store.test.ts new file mode 100644 index 00000000000..129dba2134c --- /dev/null +++ b/cloud/apps/push/src/host-session-store.test.ts @@ -0,0 +1,70 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { PushHostSessionStore } from './host-session-store.js' +import { openInMemoryPushDatabase, type PushDatabase } from './push-database.js' + +const HOST = 'abcdefghijklmnop' + +describe('push host session store', () => { + let database: PushDatabase + let clock = 1_700_000_000_000 + let sessions: PushHostSessionStore + + beforeEach(async () => { + database = await openInMemoryPushDatabase() + clock = 1_700_000_000_000 + sessions = new PushHostSessionStore(database, () => clock) + }) + + afterEach(async () => { + await database.close() + }) + + it('mints a 24 hour session and stores only its hash', async () => { + const session = await sessions.create(HOST) + expect(session.expiresAt).toBe(clock + PUSH_LIMITS.sessionTtlMs) + expect(Buffer.from(session.sessionToken, 'base64url').byteLength).toBe(32) + const [row] = await database.query('SELECT token_hash FROM push_sessions') + expect(String(row?.token_hash)).not.toBe(session.sessionToken) + await expect(sessions.resolve(session.sessionToken)).resolves.toMatchObject({ + ok: true, + hostFingerprint: HOST + }) + }) + + it('reports expiry separately from an unknown token', async () => { + const session = await sessions.create(HOST) + clock += PUSH_LIMITS.sessionTtlMs + 1 + await expect(sessions.resolve(session.sessionToken)).resolves.toEqual({ + ok: false, + reason: 'session_expired' + }) + await expect(sessions.resolve('not-a-session')).resolves.toEqual({ + ok: false, + reason: 'unknown_session' + }) + }) + + it('accepts a session on its final millisecond', async () => { + const session = await sessions.create(HOST) + clock += PUSH_LIMITS.sessionTtlMs + await expect(sessions.resolve(session.sessionToken)).resolves.toMatchObject({ ok: true }) + }) + + it('keeps one live session per host and prunes it once expired', async () => { + const first = await sessions.create(HOST) + const second = await sessions.create(HOST) + // The earlier session is gone the moment its host proves again, so a flood + // of proofs leaves one row per host rather than one per proof. + await expect(sessions.resolve(first.sessionToken)).resolves.toEqual({ + ok: false, + reason: 'unknown_session' + }) + await expect(sessions.resolve(second.sessionToken)).resolves.toMatchObject({ ok: true }) + const other = await sessions.create('ponmlkjihgfedcba') + await expect(sessions.resolve(second.sessionToken)).resolves.toMatchObject({ ok: true }) + clock += PUSH_LIMITS.sessionTtlMs + 1 + expect(await sessions.pruneExpired()).toBe(2) + await expect(sessions.resolve(other.sessionToken)).resolves.toMatchObject({ ok: false }) + }) +}) diff --git a/cloud/apps/push/src/host-session-store.ts b/cloud/apps/push/src/host-session-store.ts new file mode 100644 index 00000000000..899bacabcc8 --- /dev/null +++ b/cloud/apps/push/src/host-session-store.ts @@ -0,0 +1,65 @@ +import { createHash, randomBytes } from 'node:crypto' +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import type { PushDatabase } from './push-database.js' + +export type IssuedPushSession = { + sessionToken: string + expiresAt: number + hostFingerprint: string +} + +export type PushSessionLookup = + | { ok: true; hostFingerprint: string; expiresAt: number } + | { ok: false; reason: 'unknown_session' | 'session_expired' } + +function hashSessionToken(sessionToken: string): string { + return createHash('sha256').update(sessionToken).digest('base64url') +} + +export class PushHostSessionStore { + constructor( + private readonly database: PushDatabase, + private readonly now: () => number = Date.now + ) {} + + async create(hostFingerprint: string): Promise { + const sessionToken = randomBytes(32).toString('base64url') + const createdAt = this.now() + const expiresAt = createdAt + PUSH_LIMITS.sessionTtlMs + await this.database.transaction(async (transaction) => { + // Why: a desktop holds one session at a time and only re-proves once it is + // gone, so an earlier row is dead weight. It also bounds the table to one + // row per host however many proofs a self-minted identity answers. + await transaction.lockQuotaScope(`orca-push-session:${hostFingerprint}`) + await transaction.query('DELETE FROM push_sessions WHERE host_fingerprint = ?', [ + hostFingerprint + ]) + await transaction.query( + `INSERT INTO push_sessions (token_hash, host_fingerprint, expires_at, created_at) + VALUES (?, ?, ?, ?)`, + [hashSessionToken(sessionToken), hostFingerprint, expiresAt, createdAt] + ) + }) + return { sessionToken, expiresAt, hostFingerprint } + } + + async resolve(sessionToken: string): Promise { + const [row] = await this.database.query( + 'SELECT host_fingerprint, expires_at FROM push_sessions WHERE token_hash = ?', + [hashSessionToken(sessionToken)] + ) + if (!row) return { ok: false, reason: 'unknown_session' } + const expiresAt = Number(row.expires_at) + // No skew grace here: a 24h session that just expired should be re-minted + // through the challenge, which is cheap and already handled by the host. + if (this.now() > expiresAt) return { ok: false, reason: 'session_expired' } + return { ok: true, hostFingerprint: String(row.host_fingerprint), expiresAt } + } + + async pruneExpired(): Promise { + const [result] = await this.database.query('DELETE FROM push_sessions WHERE expires_at < ?', [ + this.now() + ]) + return Number(result?.changes ?? 0) + } +} diff --git a/cloud/apps/push/src/index.ts b/cloud/apps/push/src/index.ts new file mode 100644 index 00000000000..d44148874c7 --- /dev/null +++ b/cloud/apps/push/src/index.ts @@ -0,0 +1,55 @@ +import { startPushBackground } from './push-background.js' +import { loadPushConfig } from './config.js' +import { openPushDatabase } from './push-database.js' +import { createPushServer } from './push-server.js' + +const config = loadPushConfig() +const database = await openPushDatabase({ + ...(config.databaseUrl === undefined ? {} : { databaseUrl: config.databaseUrl }), + dataDir: config.dataDir, + poolMax: config.databasePoolMax, + applicationName: 'orca-push', + readOnly: config.mode === 'validation' +}) +const { + server, + challenges, + sessions, + deliveryStore, + worker, + observability, + closeTransports, + requestDrain +} = createPushServer(config, database) + +const stopBackground = startPushBackground(config, { challenges, sessions, deliveryStore, worker }) +observability.start() + +server.listen(config.port, () => { + console.log(`[orca-push] listening on ${config.publicUrl} (port ${config.port})`) +}) + +let stopping = false +const shutdown = (): void => { + if (stopping) return + stopping = true + // Cloud Run sends SIGKILL after ten seconds; leave time for explicit cleanup. + const deadline = setTimeout(() => process.exit(1), 9_000) + deadline.unref() + const requests = requestDrain.begin() + const deliveries = stopBackground() + const connections = new Promise((resolve) => server.close(() => resolve())) + void Promise.all([requests, connections, deliveries]) + .then(async () => { + closeTransports() + await database.close() + observability.stop() + clearTimeout(deadline) + }) + .catch(() => { + console.warn(JSON.stringify({ event: 'orca_push_shutdown_failed' })) + process.exitCode = 1 + }) +} +process.once('SIGTERM', shutdown) +process.once('SIGINT', shutdown) diff --git a/cloud/apps/push/src/provider-retry-delay.ts b/cloud/apps/push/src/provider-retry-delay.ts new file mode 100644 index 00000000000..4c77b3c6dc7 --- /dev/null +++ b/cloud/apps/push/src/provider-retry-delay.ts @@ -0,0 +1,9 @@ +export function providerRetryAfter( + value: string | undefined, + now = Date.now() +): number | undefined { + if (!value) return undefined + const seconds = Number(value) + const delay = Number.isFinite(seconds) ? seconds * 1000 : Date.parse(value) - now + return Number.isFinite(delay) ? Math.max(0, delay) : undefined +} diff --git a/cloud/apps/push/src/push-auth-admission.ts b/cloud/apps/push/src/push-auth-admission.ts new file mode 100644 index 00000000000..cbc951abe25 --- /dev/null +++ b/cloud/apps/push/src/push-auth-admission.ts @@ -0,0 +1,21 @@ +// Bound unauthenticated database lookups independently of Cloud Run HTTP concurrency. +export class PushAuthAdmission { + private active = 0 + private readonly waiting: (() => void)[] = [] + + async run(operation: () => Promise): Promise { + if (this.active >= 4) { + if (this.waiting.length >= 32) return null + await new Promise((resolve) => this.waiting.push(resolve)) + } else { + this.active++ + } + try { + return await operation() + } finally { + const next = this.waiting.shift() + if (next) next() + else this.active-- + } + } +} diff --git a/cloud/apps/push/src/push-authenticated-ip-limit.test.ts b/cloud/apps/push/src/push-authenticated-ip-limit.test.ts new file mode 100644 index 00000000000..a626b80060a --- /dev/null +++ b/cloud/apps/push/src/push-authenticated-ip-limit.test.ts @@ -0,0 +1,50 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { expect, it } from 'vitest' +import { createPushHostKeypair } from './host-challenge-answering.test-fixture.js' +import { createPushServerHarness, FCM_TOKEN } from './push-server-harness.test-fixture.js' + +it('bounds valid hosts together across routes without letting key rotation reset the IP budget', async () => { + const harness = await createPushServerHarness() + const headers = { 'x-forwarded-for': '203.0.113.7' } + try { + const hostCount = + PUSH_LIMITS.authenticatedRequestsPerMinutePerIp / + PUSH_LIMITS.authenticatedRequestsPerMinutePerHost + for (let host = 0; host < hostCount; host++) { + const token = await harness.signIn(createPushHostKeypair(host + 1)) + for ( + let request = 0; + request < PUSH_LIMITS.authenticatedRequestsPerMinutePerHost; + request++ + ) { + expect((await harness.authorized('/v1/devices', { headers }, token)).status).toBe(200) + } + } + const token = await harness.signIn(createPushHostKeypair(99)) + const registration = { + method: 'POST', + headers: { ...headers, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, deviceId: 'phone', platform: 'android', token: FCM_TOKEN }) + } + expect((await harness.authorized('/v1/devices', registration, token)).status).toBe(429) + expect((await harness.authorized('/v1/send', { method: 'POST', headers }, token)).status).toBe( + 429 + ) + expect( + ( + await harness.authorized( + '/v1/devices', + { + ...registration, + headers: { ...registration.headers, 'x-forwarded-for': '198.51.100.9' } + }, + token + ) + ).status + ).toBe(200) + harness.advanceClock(60_000) + expect((await harness.authorized('/v1/devices', registration, token)).status).toBe(200) + } finally { + await harness.close() + } +}, 30_000) diff --git a/cloud/apps/push/src/push-background.ts b/cloud/apps/push/src/push-background.ts new file mode 100644 index 00000000000..8925fd41fed --- /dev/null +++ b/cloud/apps/push/src/push-background.ts @@ -0,0 +1,43 @@ +import type { PushConfig } from './config.js' +import type { createPushServer } from './push-server.js' + +const CHALLENGE_PRUNE_INTERVAL_MS = 60_000 +const SESSION_PRUNE_INTERVAL_MS = 10 * 60_000 +const DELIVERY_PRUNE_INTERVAL_MS = 60_000 + +function prune(label: string, run: () => Promise, intervalMs: number): NodeJS.Timeout { + const timer = setInterval(() => { + void run().catch((error: unknown) => { + console.warn( + JSON.stringify({ + event: 'orca_push_prune_failed', + target: label, + error: error instanceof Error ? error.name : 'unknown' + }) + ) + }) + }, intervalMs) + timer.unref() + return timer +} + +export function startPushBackground( + config: Pick, + runtime: Pick< + ReturnType, + 'challenges' | 'sessions' | 'deliveryStore' | 'worker' + > +): () => Promise { + if (config.mode === 'validation') return async () => {} + const { challenges, sessions, deliveryStore, worker } = runtime + const timers = [ + prune('challenges', () => challenges.pruneExpired(), CHALLENGE_PRUNE_INTERVAL_MS), + prune('sessions', () => sessions.pruneExpired(), SESSION_PRUNE_INTERVAL_MS), + prune('deliveries', () => deliveryStore.prune(), DELIVERY_PRUNE_INTERVAL_MS) + ] + worker.start() + return async () => { + for (const timer of timers) clearInterval(timer) + await worker.stop() + } +} diff --git a/cloud/apps/push/src/push-database-postgres-startup.test.ts b/cloud/apps/push/src/push-database-postgres-startup.test.ts new file mode 100644 index 00000000000..b6fd23071fc --- /dev/null +++ b/cloud/apps/push/src/push-database-postgres-startup.test.ts @@ -0,0 +1,121 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { randomUUID } from 'node:crypto' + +const fakes = vi.hoisted(() => ({ + configs: [] as Array>, + lifecycle: [] as string[], + query: vi.fn(async (_sql: string) => ({ rows: [], rowCount: 0 })), + release: vi.fn() +})) + +vi.mock('pg', () => ({ + default: { + Pool: class { + on = vi.fn() + connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release })) + private readonly label: string + + constructor(config: Record) { + fakes.configs.push(config) + this.label = `max=${String(config.max)} statement_timeout=${String(config.statement_timeout)}` + fakes.lifecycle.push(`open ${this.label}`) + } + + async end(): Promise { + fakes.lifecycle.push(`end ${this.label}`) + } + } + } +})) + +import { openPushDatabase } from './push-database.js' +import { pushSchemaStatements } from './push-schema.js' + +describe('PostgreSQL push gateway startup', () => { + beforeEach(() => { + fakes.configs.length = 0 + fakes.lifecycle.length = 0 + fakes.query.mockClear() + }) + + afterEach(() => { + vi.restoreAllMocks() + }) + + const socketPassword = `${randomUUID()}@/` + const socketUrl = `postgresql://push:${encodeURIComponent(socketPassword)}@/orca_push?host=/cloudsql/test:region:instance` + + it('passes the Terraform socket URL unchanged to both active pools', async () => { + const database = await openPushDatabase({ databaseUrl: socketUrl, dataDir: '/unused' }) + expect(fakes.configs.map((config) => config.connectionString)).toEqual([socketUrl, socketUrl]) + await database.close() + }) + + it('parses socket credentials and query options before enforcing validation read-only', async () => { + const options = '-c search_path=validation -c default_transaction_read_only=off' + const database = await openPushDatabase({ + databaseUrl: `${socketUrl}&port=5433&sslmode=disable&options=${encodeURIComponent(options)}`, + dataDir: '/unused', + readOnly: true + }) + expect(fakes.configs).toHaveLength(1) + expect(fakes.configs[0]).toMatchObject({ + host: '/cloudsql/test:region:instance', + user: 'push', + password: socketPassword, + database: 'orca_push', + port: 5433, + ssl: false, + options: `${options} -c default_transaction_read_only=on` + }) + expect(fakes.configs[0]).not.toHaveProperty('connectionString') + expect(fakes.query).not.toHaveBeenCalled() + await database.close() + }) + + // Why: a CREATE INDEX on a grown table can outlive the 5s request deadline, + // and a schema that inherits it fails every startup at the same statement. + it('applies the schema on an untimed pool that is gone before the serving pool opens', async () => { + const database = await openPushDatabase({ + databaseUrl: 'postgresql://push@localhost:55440/orca_push', + dataDir: '/unused', + poolMax: 2, + applicationName: 'orca-push' + }) + expect(fakes.lifecycle).toEqual([ + 'open max=1 statement_timeout=0', + 'end max=1 statement_timeout=0', + 'open max=2 statement_timeout=5000' + ]) + expect(fakes.configs[0]).toMatchObject({ + application_name: 'orca-push/schema', + lock_timeout: 1_000, + idle_in_transaction_session_timeout: 5_000 + }) + expect( + fakes.query.mock.calls.map(([sql]) => sql).slice(0, pushSchemaStatements().length) + ).toEqual(pushSchemaStatements()) + await database.close() + }) + + it('retries a transaction the pool statement_timeout aborted', async () => { + const database = await openPushDatabase({ + databaseUrl: 'postgresql://push@localhost:55440/orca_push', + dataDir: '/unused' + }) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + let attempts = 0 + const result = await database.transaction(async () => { + attempts += 1 + if (attempts === 1) throw Object.assign(new Error('canceling statement'), { code: '57014' }) + return 'done' + }) + expect(result).toBe('done') + expect(attempts).toBe(2) + expect(warn.mock.calls.map(([line]) => String(line))).toEqual([ + expect.stringContaining('"code":"57014"') + ]) + warn.mockRestore() + await database.close() + }) +}) diff --git a/cloud/apps/push/src/push-database.ts b/cloud/apps/push/src/push-database.ts new file mode 100644 index 00000000000..a2ed2bce9d2 --- /dev/null +++ b/cloud/apps/push/src/push-database.ts @@ -0,0 +1,283 @@ +import { mkdirSync } from 'node:fs' +import { join } from 'node:path' +import { DatabaseSync } from 'node:sqlite' +import pg from 'pg' +import { parseIntoClientConfig } from 'pg-connection-string' +import { applyPostgresSchema } from '@orca-cloud/postgres-schema' +import { pushSchemaStatements } from './push-schema.js' + +const POSTGRES_LOCK_TIMEOUT_MS = 1_000 +const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000 +const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 +const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000 +const POSTGRES_TRANSACTION_ATTEMPTS = 3 +const POSTGRES_RETRY_MAX_DELAY_MS = 25 + +export type SqlRow = Record + +export interface PushDatabase { + readonly dialect: 'sqlite' | 'postgres' + query(sql: string, params?: unknown[]): Promise + transaction(operation: (transaction: PushDatabase) => Promise): Promise + // Serializes every transaction that reads then writes the same identity's + // quota rows. Must be called inside a transaction; it releases at commit. + lockQuotaScope(key: string): Promise + close(): Promise +} + +function postgresSql(sql: string): string { + let index = 0 + return sql.replace(/\?/g, () => `$${++index}`) +} + +function returnsRows(sql: string): boolean { + return /^\s*(select|with)/i.test(sql) || /returning/i.test(sql) +} + +class SqliteTransaction implements PushDatabase { + readonly dialect = 'sqlite' as const + + constructor(protected readonly database: DatabaseSync) {} + + async query(sql: string, params: unknown[] = []): Promise { + const statement = this.database.prepare(sql) + const bound = params.map((value) => (value === undefined ? null : value)) as never[] + if (returnsRows(sql)) return statement.all(...bound) as SqlRow[] + const result = statement.run(...bound) + return [{ changes: Number(result.changes) }] + } + + async transaction(operation: (transaction: PushDatabase) => Promise): Promise { + return await operation(this) + } + + // BEGIN IMMEDIATE already holds the single writer lock for the whole + // transaction, so there is nothing narrower left to take. + async lockQuotaScope(): Promise {} + + async close(): Promise {} +} + +class SqliteDatabase extends SqliteTransaction { + // node:sqlite is synchronous and has no nested transactions, so overlapping + // callers are serialized behind one tail promise instead of racing BEGIN. + private tail: Promise = Promise.resolve() + + override async query(sql: string, params: unknown[] = []): Promise { + await this.tail + return await super.query(sql, params) + } + + override async transaction(operation: (transaction: PushDatabase) => Promise): Promise { + const previous = this.tail + let release!: () => void + this.tail = new Promise((resolve) => (release = resolve)) + await previous + this.database.exec('BEGIN IMMEDIATE') + const transaction = new SqliteTransaction(this.database) + try { + const result = await operation(transaction) + this.database.exec('COMMIT') + return result + } catch (error) { + this.database.exec('ROLLBACK') + throw error + } finally { + release() + } + } + + override async close(): Promise { + await this.tail + this.database.close() + } +} + +class PostgresTransaction implements PushDatabase { + readonly dialect = 'postgres' as const + + constructor(private readonly client: pg.PoolClient) {} + + async query(sql: string, params: unknown[] = []): Promise { + const result = await this.client.query(postgresSql(sql), params) + return returnsRows(sql) ? (result.rows as SqlRow[]) : [{ changes: result.rowCount ?? 0 }] + } + + async transaction(operation: (transaction: PushDatabase) => Promise): Promise { + return await operation(this) + } + + // READ COMMITTED lets a concurrent count-then-insert read the same + // under-quota total, so the identity is serialized for the whole transaction. + async lockQuotaScope(key: string): Promise { + await this.query('SELECT pg_advisory_xact_lock(hashtext(?::text))', [key]) + } + + async close(): Promise {} +} + +function retryablePostgresTransactionError(error: unknown): boolean { + const code = String((error as { code?: unknown }).code) + // 57014 is the pool statement_timeout firing. It aborts the transaction the + // same way a lock timeout does, so it takes the bounded retry path too. + return code === '40P01' || code === '40001' || code === '55P03' || code === '57014' +} + +async function waitForPostgresRetry(): Promise { + const delayMs = Math.floor(Math.random() * (POSTGRES_RETRY_MAX_DELAY_MS + 1)) + await new Promise((resolve) => setTimeout(resolve, delayMs)) +} + +class PostgresDatabase implements PushDatabase { + readonly dialect = 'postgres' as const + + constructor(private readonly pool: pg.Pool) {} + + async query(sql: string, params: unknown[] = []): Promise { + const client = await this.pool.connect() + try { + const result = await client.query(postgresSql(sql), params) + return returnsRows(sql) ? (result.rows as SqlRow[]) : [{ changes: result.rowCount ?? 0 }] + } finally { + client.release() + } + } + + async transaction(operation: (transaction: PushDatabase) => Promise): Promise { + for (let attempt = 1; attempt <= POSTGRES_TRANSACTION_ATTEMPTS; attempt++) { + const client = await this.pool.connect() + try { + await client.query('BEGIN') + const result = await operation(new PostgresTransaction(client)) + await client.query('COMMIT') + return result + } catch (error) { + await client.query('ROLLBACK').catch(() => undefined) + if ( + !retryablePostgresTransactionError(error) || + attempt === POSTGRES_TRANSACTION_ATTEMPTS + ) { + throw error + } + console.warn( + JSON.stringify({ + event: 'orca_push_postgres_transaction_retry', + code: String((error as { code?: unknown }).code), + attempt + }) + ) + } finally { + client.release() + } + // A PostgreSQL transaction is unusable after an abort, so retry all work + // on a fresh pooled client with a small full-jitter delay. + await waitForPostgresRetry() + } + throw new Error('postgres_transaction_retry_exhausted') + } + + // An advisory transaction lock taken outside a transaction is released by the + // implicit commit before the caller reads anything, which protects nothing. + async lockQuotaScope(): Promise { + throw new Error('lock_quota_scope_requires_transaction') + } + + async close(): Promise { + await this.pool.end() + } +} + +async function applySchema(database: PushDatabase): Promise { + for (const statement of pushSchemaStatements()) await database.query(statement) +} + +// Why: DDL is not a request. A CREATE INDEX on a grown table can legitimately +// outlive the request statement_timeout, and inheriting it would fail every +// startup at the same statement instead of finishing once. One connection of +// its own, closed before the serving pool opens, keeps the untimed session off +// the request path entirely. +async function applySchemaOnUntimedPool( + databaseUrl: string, + applicationName: string | undefined +): Promise { + const pool = new pg.Pool({ + connectionString: databaseUrl, + max: 1, + application_name: applicationName ? `${applicationName}/schema` : undefined, + connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, + statement_timeout: 0, + lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, + idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS + }) + absorbPostgresIdleClientErrors(pool) + const database = new PostgresDatabase(pool) + try { + await applyPostgresSchema(pushSchemaStatements(), (statement) => database.query(statement), { + eventPrefix: 'orca_push_postgres_schema' + }) + } finally { + await database.close().catch(() => undefined) + } +} + +export function absorbPostgresIdleClientErrors(pool: Pick): void { + pool.on('error', () => { + // node-postgres removes failed idle clients itself; an unhandled 'error' + // would crash the service and turn a SQL blip into a restart loop. + console.warn('[orca-push] idle PostgreSQL client failed') + }) +} + +export async function openPushDatabase(input: { + databaseUrl?: string + dataDir: string + poolMax?: number + applicationName?: string + readOnly?: boolean +}): Promise { + let database: PushDatabase + if (input.databaseUrl) { + if (!input.readOnly) await applySchemaOnUntimedPool(input.databaseUrl, input.applicationName) + let connection: pg.ClientConfig = { connectionString: input.databaseUrl } + if (input.readOnly) { + connection = parseIntoClientConfig(input.databaseUrl) + // A URL parameter must not trigger a second parse that overrides read-only options. + delete connection.connectionString + connection.options = `${connection.options ?? ''} -c default_transaction_read_only=on`.trim() + } + const pool = new pg.Pool({ + ...connection, + max: input.poolMax ?? 10, + application_name: input.applicationName, + connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, + statement_timeout: POSTGRES_STATEMENT_TIMEOUT_MS, + lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, + idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS + }) + absorbPostgresIdleClientErrors(pool) + database = new PostgresDatabase(pool) + } else { + mkdirSync(input.dataDir, { recursive: true }) + const sqlite = new DatabaseSync(join(input.dataDir, 'orca-push.sqlite'), { + readOnly: input.readOnly ?? false + }) + if (!input.readOnly) sqlite.exec('PRAGMA journal_mode = WAL; PRAGMA foreign_keys = ON;') + database = new SqliteDatabase(sqlite) + } + if (database.dialect === 'postgres' || input.readOnly) return database + try { + await applySchema(database) + return database + } catch (error) { + await database.close().catch(() => undefined) + throw error + } +} + +export async function openInMemoryPushDatabase(): Promise { + const sqlite = new DatabaseSync(':memory:') + sqlite.exec('PRAGMA foreign_keys = ON;') + const database = new SqliteDatabase(sqlite) + await applySchema(database) + return database +} diff --git a/cloud/apps/push/src/push-delivery-lifecycle.test.ts b/cloud/apps/push/src/push-delivery-lifecycle.test.ts new file mode 100644 index 00000000000..c94108e459c --- /dev/null +++ b/cloud/apps/push/src/push-delivery-lifecycle.test.ts @@ -0,0 +1,85 @@ +import { afterEach, expect, it, vi } from 'vitest' +import { Hono } from 'hono' +import { PushRequestDrain } from './push-request-drain.js' +import { PushDispatcher } from './push-dispatcher.js' +import { PushDeviceRegistryStore } from './device-registry-store.js' +import { openInMemoryPushDatabase, type PushDatabase } from './push-database.js' +import { buildPushDelivery } from './push-delivery-message.js' +import { PushNotificationSchema } from '@orca-cloud/push-contract' +import { notification } from './push-server-harness.test-fixture.js' + +const databases: PushDatabase[] = [] +afterEach(async () => { + await Promise.all(databases.splice(0).map((db) => db.close())) + vi.restoreAllMocks() +}) +const note = PushNotificationSchema.parse(notification()) +const tick = () => new Promise((resolve) => setImmediate(resolve)) +function deferred() { + let resolve!: () => void + const promise = new Promise((done) => { + resolve = done + }) + return { promise, resolve } +} +async function registered() { + const db = await openInMemoryPushDatabase() + databases.push(db) + const devices = new PushDeviceRegistryStore(db) + const input = { + hostFingerprint: 'abcdefghijklmnop', + deviceId: 'device', + platform: 'android' as const, + token: 'old-token' + } + const row = await devices.upsert(input) + if (!row.ok) throw new Error('registration failed') + const delivery = buildPushDelivery({ + expiresAt: Date.now() + 300_000, + registrationId: row.registrationId, + hostFingerprint: input.hostFingerprint, + notification: note + }) + return { db, devices, input, delivery } +} + +it('does not retire a refreshed token after the old token fails', async () => { + const h = await registered() + const gate = deferred() + const send = vi.fn(async () => { + await gate.promise + return { status: 'dead', reason: 'UNREGISTERED' } + }) + vi.spyOn(console, 'warn').mockImplementation(() => {}) + const dispatcher = new PushDispatcher({ devices: h.devices, fcm: { send } as never }) + const pending = dispatcher.sendOnce(h.delivery) + await tick() + await h.devices.upsert({ ...h.input, token: 'replacement-token' }) + gate.resolve() + await pending + expect(await h.devices.findById(h.delivery.registrationId)).toMatchObject({ + token: 'replacement-token', + dead: false + }) +}) + +it('rejects new requests during drain and waits for an admitted handler', async () => { + const gate = deferred() + const requests = new PushRequestDrain() + const app = new Hono().use('*', requests.middleware).post('/send', async (c) => { + await gate.promise + return c.json({ queued: true }) + }) + const pending = app.request('/send', { method: 'POST' }) + await tick() + let drained = false + const drain = requests.begin().then(() => { + drained = true + }) + expect((await app.request('/send', { method: 'POST' })).status).toBe(503) + expect(drained).toBe(false) + gate.resolve() + expect((await pending).status).toBe(200) + await drain + expect(drained).toBe(true) +}) diff --git a/cloud/apps/push/src/push-delivery-message.ts b/cloud/apps/push/src/push-delivery-message.ts new file mode 100644 index 00000000000..3bd2dae6e7c --- /dev/null +++ b/cloud/apps/push/src/push-delivery-message.ts @@ -0,0 +1,77 @@ +import { createHash } from 'node:crypto' +import { type PushNotification } from '@orca-cloud/push-contract' + +export type PushOrcaData = { + kind?: 'alert' | 'dismiss' + hostFingerprint: string + worktreeId?: string + paneKey?: string + notificationId?: string + notificationSeq: number + notificationEpoch: string + source: string + agentState: string | null +} + +export type PushDelivery = { + expiresAt: number + sound?: boolean + registrationId: string + hostFingerprint: string + title: string + body: string + collapseId: string + orca: PushOrcaData +} + +export function collapseIdFor(notification: PushNotification, hostFingerprint: string): string { + const identity = + notification.notificationId === undefined + ? [notification.notificationEpoch, notification.notificationSeq] + : notification.notificationId + return createHash('sha256') + .update(JSON.stringify([hostFingerprint, identity])) + .digest('hex') +} + +export function buildPushDelivery(input: { + expiresAt: number + registrationId: string + hostFingerprint: string + notification: PushNotification +}): PushDelivery { + const { notification, hostFingerprint } = input + return { + ...(notification.sound === false ? { sound: false } : {}), + expiresAt: input.expiresAt, + registrationId: input.registrationId, + hostFingerprint, + title: notification.title, + body: notification.body, + collapseId: collapseIdFor(notification, hostFingerprint), + orca: { + ...(notification.kind ? { kind: notification.kind } : {}), + hostFingerprint, + ...(notification.paneKey === undefined ? {} : { paneKey: notification.paneKey }), + ...(notification.worktreeId === undefined ? {} : { worktreeId: notification.worktreeId }), + ...(notification.notificationId === undefined + ? {} + : { notificationId: notification.notificationId }), + notificationSeq: notification.notificationSeq, + notificationEpoch: notification.notificationEpoch, + source: notification.source, + agentState: notification.agentState + } + } +} + +export function orcaDataStrings(orca: PushOrcaData): Record { + return Object.fromEntries( + Object.entries(orca) + .filter(([, value]) => value !== undefined && value !== null) + .map(([key, value]) => [ + key, + typeof value === 'object' ? JSON.stringify(value) : String(value) + ]) + ) +} diff --git a/cloud/apps/push/src/push-delivery-payload.ts b/cloud/apps/push/src/push-delivery-payload.ts new file mode 100644 index 00000000000..825286125b7 --- /dev/null +++ b/cloud/apps/push/src/push-delivery-payload.ts @@ -0,0 +1,8 @@ +import type { PushNotification } from '@orca-cloud/push-contract' + +export function parsePushDeliveryPayload(payload: string): PushNotification { + const value: unknown = JSON.parse(payload) + if (!value || typeof value !== 'object' || Array.isArray(value)) + throw new Error('invalid_push_delivery_payload') + return value as PushNotification +} diff --git a/cloud/apps/push/src/push-deploy-preflight.test.ts b/cloud/apps/push/src/push-deploy-preflight.test.ts new file mode 100644 index 00000000000..4652ef8edaa --- /dev/null +++ b/cloud/apps/push/src/push-deploy-preflight.test.ts @@ -0,0 +1,17 @@ +import { readFileSync } from 'node:fs' +import { expect, it } from 'vitest' +import { loadPushConfig } from './config.js' + +it('runs the deployment image preflight against the actual config loader', () => { + const workflow = readFileSync( + new URL('../../../../.github/workflows/cloud-push-deploy.yml', import.meta.url), + 'utf8' + ) + const step = workflow.split('- name: Require image support for inert validation')[1]! + const script = step.match(/--input-type=module -e '([\s\S]*?)'/)?.[1] + expect(script).toBeDefined() + const run = new Function('loadPushConfig', script!.replace(/import .*?;/, '')) + expect(() => run(loadPushConfig)).not.toThrow() + expect(() => run(() => ({ mode: 'active' }))).toThrow('validation_mode_unsupported') + expect(() => run(() => ({ mode: 'validation' }))).toThrow('validation_mode_not_fail_closed') +}) diff --git a/cloud/apps/push/src/push-dismissal-provider.test.ts b/cloud/apps/push/src/push-dismissal-provider.test.ts new file mode 100644 index 00000000000..65572e8401c --- /dev/null +++ b/cloud/apps/push/src/push-dismissal-provider.test.ts @@ -0,0 +1,31 @@ +import { expect, it } from 'vitest' +import { apnsBody } from './apns-client.js' +import { fcmMessageBody } from './fcm-client.js' +import { buildPushDelivery } from './push-delivery-message.js' + +it('dismissal provider payloads cannot display a new alert or play a sound', () => { + const delivery = buildPushDelivery({ + expiresAt: Date.now() + 300_000, + registrationId: 'reg', + hostFingerprint: 'host', + notification: { + kind: 'dismiss', + notificationId: 'note', + notificationSeq: 2, + notificationEpoch: 'epoch', + source: 'agent-task-complete', + agentState: null, + title: 'Orca', + body: '' + } + }) + expect(JSON.parse(apnsBody(delivery)).aps).toEqual({ 'content-available': 1 }) + const android = JSON.parse(fcmMessageBody({ delivery, token: 'test', channelId: 'test' })).message + expect(android).not.toHaveProperty('notification') + expect(android.android).not.toHaveProperty('notification') + expect(android.android).not.toHaveProperty('collapse_key') + expect(android.data).not.toHaveProperty('title') + expect(android.data).not.toHaveProperty('message') + expect(android.data).not.toHaveProperty('sound') + expect(android.data.kind).toBe('dismiss') +}) diff --git a/cloud/apps/push/src/push-dispatcher.ts b/cloud/apps/push/src/push-dispatcher.ts new file mode 100644 index 00000000000..ca9006bc435 --- /dev/null +++ b/cloud/apps/push/src/push-dispatcher.ts @@ -0,0 +1,52 @@ +import type { ApnsClient } from './apns-client.js' +import type { PushDeviceRegistryStore } from './device-registry-store.js' +import type { FcmClient } from './fcm-client.js' +import { fingerprintLogPrefix } from './host-fingerprint.js' +import type { PushDelivery } from './push-delivery-message.js' +import type { PushProviderOutcome } from './push-provider-outcome.js' + +export type PushDispatcherOptions = { + devices: PushDeviceRegistryStore + apns?: ApnsClient + fcm?: FcmClient + onOutcome?: (outcome: PushProviderOutcome['status']) => void +} + +// Retires the registration when the provider says the token is gone. +export class PushDispatcher { + constructor(private readonly options: PushDispatcherOptions) {} + + async sendOnce(delivery: PushDelivery): Promise { + const device = await this.options.devices.findById(delivery.registrationId) + if (!device || device.dead) return { status: 'dead', reason: 'registration_unavailable' } + let outcome: PushProviderOutcome + if (device.platform === 'ios') { + outcome = this.options.apns + ? await this.options.apns.send(delivery, { + token: device.token, + apnsEnvironment: device.apnsEnvironment ?? 'production' + }) + : { status: 'error', reason: 'apns_not_configured' } + } else { + outcome = this.options.fcm + ? await this.options.fcm.send(delivery, { token: device.token }) + : { status: 'error', reason: 'fcm_not_configured' } + } + this.options.onOutcome?.(outcome.status) + if (outcome.status === 'dead') { + await this.options.devices.markDead(device) + } + if (outcome.status !== 'sent') { + console.warn( + JSON.stringify({ + event: 'orca_push_delivery_failed', + platform: device.platform, + status: outcome.status, + reason: outcome.reason, + host: fingerprintLogPrefix(delivery.hostFingerprint) + }) + ) + } + return outcome + } +} diff --git a/cloud/apps/push/src/push-notification-sound.test.ts b/cloud/apps/push/src/push-notification-sound.test.ts new file mode 100644 index 00000000000..ede4dcd3291 --- /dev/null +++ b/cloud/apps/push/src/push-notification-sound.test.ts @@ -0,0 +1,33 @@ +import { expect, it } from 'vitest' +import { apnsBody } from './apns-client.js' +import { fcmMessageBody } from './fcm-client.js' +import { buildPushDelivery } from './push-delivery-message.js' +import { PushNotificationSchema } from '@orca-cloud/push-contract' + +it('carries a silent preference through validation to APNs and Android payloads', () => { + const notification = PushNotificationSchema.parse({ + notificationSeq: 1, + notificationEpoch: 'epoch', + source: 'terminal-bell', + agentState: null, + title: 'Bell', + body: '', + sound: false + }) + const delivery = buildPushDelivery({ + expiresAt: Date.now() + 300_000, + registrationId: 'reg', + hostFingerprint: 'host', + notification + }) + expect(JSON.parse(apnsBody(delivery)).aps).not.toHaveProperty('sound') + expect( + JSON.parse(fcmMessageBody({ delivery, token: 'test-token', channelId: 'orca-desktop' })).message + .data.channelId + ).toBe('orca-desktop-silent') + expect( + JSON.parse(fcmMessageBody({ delivery, token: 'test-token', channelId: 'orca-desktop' })).message + .data.sound + ).toBe('') + expect(JSON.parse(apnsBody({ ...delivery, sound: undefined })).aps.sound).toBe('default') +}) diff --git a/cloud/apps/push/src/push-observability.ts b/cloud/apps/push/src/push-observability.ts new file mode 100644 index 00000000000..6f0dccb8a09 --- /dev/null +++ b/cloud/apps/push/src/push-observability.ts @@ -0,0 +1,56 @@ +const COUNTER_NAMES = [ + 'ip_rate_limited', + 'request_error', + 'challenge_issued', + 'challenge_rejected', + 'session_issued', + 'session_rejected', + 'device_registered', + 'device_rejected', + 'device_deleted', + 'send_queued', + 'send_dead', + 'send_rate_limited', + 'send_error', + 'delivery_sent', + 'delivery_dead', + 'delivery_error', + 'delivery_retry' +] as const + +type PushCounterName = (typeof COUNTER_NAMES)[number] + +// Aggregate counters only. Nothing here may accept a token, a title, a body, +// or more than the first four characters of a host fingerprint. +export class PushObservability { + private counters = new Map() + private timer: NodeJS.Timeout | null = null + + record(name: PushCounterName, delta = 1): void { + this.counters.set(name, (this.counters.get(name) ?? 0) + delta) + } + + consume(): Record { + const snapshot = Object.fromEntries( + COUNTER_NAMES.map((name) => [name, this.counters.get(name) ?? 0]) + ) as Record + this.counters = new Map() + return snapshot + } + + start(intervalMs = 60_000): void { + if (this.timer) return + this.timer = setInterval(() => { + const counters = this.consume() + if (Object.values(counters).every((value) => value === 0)) return + console.warn(JSON.stringify({ event: 'orca_push_counters', ...counters })) + }, intervalMs) + this.timer.unref() + } + + stop(): void { + if (!this.timer) return + clearInterval(this.timer) + this.timer = null + } +} diff --git a/cloud/apps/push/src/push-pane-routing.test.ts b/cloud/apps/push/src/push-pane-routing.test.ts new file mode 100644 index 00000000000..66ce0b3a38c --- /dev/null +++ b/cloud/apps/push/src/push-pane-routing.test.ts @@ -0,0 +1,27 @@ +import { expect, it } from 'vitest' +import { PushNotificationSchema } from '@orca-cloud/push-contract' +import { buildPushDelivery, orcaDataStrings } from './push-delivery-message.js' + +it('preserves pane identity for both APNs and FCM, and accepts older workspace-only messages', () => { + const base = { + notificationSeq: 1, + notificationEpoch: 'epoch', + source: 'agent-task-complete', + agentState: 'finished', + title: 'Done', + body: '', + worktreeId: 'folder:/work' + } + const paneKey = 'tab-b:11111111-1111-4111-8111-111111111111' + for (const extra of [{}, { paneKey }]) { + const notification = PushNotificationSchema.parse({ ...base, ...extra }) + const delivery = buildPushDelivery({ + notification, + hostFingerprint: 'host', + registrationId: 'phone', + expiresAt: Date.now() + 300000 + }) + expect(delivery.orca.paneKey).toBe('paneKey' in extra ? paneKey : undefined) + expect(orcaDataStrings(delivery.orca).paneKey).toBe('paneKey' in extra ? paneKey : undefined) + } +}) diff --git a/cloud/apps/push/src/push-provider-outcome.ts b/cloud/apps/push/src/push-provider-outcome.ts new file mode 100644 index 00000000000..bc65d10c175 --- /dev/null +++ b/cloud/apps/push/src/push-provider-outcome.ts @@ -0,0 +1,6 @@ +// What a provider send resolved to, before the send route maps it onto the +// contract's queued / dead / rate_limited / error statuses. +export type PushProviderOutcome = + | { status: 'sent' } + | { status: 'dead'; reason: string } + | { status: 'error'; reason: string; retryable?: boolean; retryAfterMs?: number } diff --git a/cloud/apps/push/src/push-queued-dismissal.ts b/cloud/apps/push/src/push-queued-dismissal.ts new file mode 100644 index 00000000000..3ca5e8c7e66 --- /dev/null +++ b/cloud/apps/push/src/push-queued-dismissal.ts @@ -0,0 +1,57 @@ +import type { PushNotification } from '@orca-cloud/push-contract' +import type { PushDatabase } from './push-database.js' +import { parsePushDeliveryPayload } from './push-delivery-payload.js' + +export async function reconcileQueuedDismissal( + tx: PushDatabase, + host: string, + registrationId: string, + notification: PushNotification, + now: number +): Promise { + if (!notification.notificationId) return false + const key = [host, notification.notificationEpoch, notification.notificationId] + const [dismissed] = await tx.query( + 'SELECT notification_seq FROM push_dismissed_events WHERE host_fingerprint = ? AND notification_epoch = ? AND notification_id = ?', + key + ) + if (notification.kind !== 'dismiss') + return Number(dismissed?.notification_seq ?? -1) >= notification.notificationSeq + await tx.query( + `INSERT INTO push_dismissed_events(host_fingerprint, notification_epoch, notification_id, notification_seq, created_at) + VALUES (?, ?, ?, ?, ?) ON CONFLICT(host_fingerprint, notification_epoch, notification_id) + DO UPDATE SET notification_seq = CASE WHEN push_dismissed_events.notification_seq > excluded.notification_seq THEN push_dismissed_events.notification_seq ELSE excluded.notification_seq END, created_at = excluded.created_at`, + [...key, notification.notificationSeq, now] + ) + const deliveries = await tx.query( + "SELECT batch_id, payload_json FROM push_delivery_batches WHERE host_fingerprint = ? AND registration_id = ? AND kind = 'alert' AND state = 'pending' AND lease_until <= ?", + [host, registrationId, now] + ) + for (const delivery of deliveries) { + const queued = parsePushDeliveryPayload(String(delivery.payload_json)) + if ( + queued.notificationEpoch !== notification.notificationEpoch || + queued.notificationId !== notification.notificationId || + queued.notificationSeq > notification.notificationSeq + ) + continue + await tx.query( + 'UPDATE push_delivery_batches SET payload_json = ?, state = ? WHERE batch_id = ?', + ['{}', 'dismissed', delivery.batch_id] + ) + } + return false +} + +export async function isDismissedAlert( + tx: PushDatabase, + host: string, + notification: PushNotification +): Promise { + if (notification.kind === 'dismiss' || !notification.notificationId) return false + const rows = await tx.query( + 'SELECT notification_seq FROM push_dismissed_events WHERE host_fingerprint = ? AND notification_epoch = ? AND notification_id = ?', + [host, notification.notificationEpoch, notification.notificationId] + ) + return Number(rows[0]?.notification_seq ?? -1) >= notification.notificationSeq +} diff --git a/cloud/apps/push/src/push-readiness.test.ts b/cloud/apps/push/src/push-readiness.test.ts new file mode 100644 index 00000000000..45a57da4330 --- /dev/null +++ b/cloud/apps/push/src/push-readiness.test.ts @@ -0,0 +1,26 @@ +import { expect, it, vi } from 'vitest' +import { createPushReadiness } from './push-readiness.js' +import type { PushDatabase } from './push-database.js' + +it('shares a slow check and caches failures before retrying', async () => { + let clock = 0 + let reject!: (error: Error) => void + const query = vi.fn( + () => + new Promise((_, fail) => { + reject = fail + }) + ) + const ready = createPushReadiness({ query } as unknown as PushDatabase, { now: () => clock }) + const checks = Array.from({ length: 100 }, () => ready()) + expect(query).toHaveBeenCalledTimes(1) + reject(new Error('offline')) + expect(await Promise.all(checks)).toEqual(Array(100).fill(false)) + expect(await ready()).toBe(false) + expect(query).toHaveBeenCalledTimes(1) + clock = 10_000 + const retry = ready() + expect(query).toHaveBeenCalledTimes(2) + reject(new Error('offline')) + expect(await retry).toBe(false) +}) diff --git a/cloud/apps/push/src/push-readiness.ts b/cloud/apps/push/src/push-readiness.ts new file mode 100644 index 00000000000..c071a0750e4 --- /dev/null +++ b/cloud/apps/push/src/push-readiness.ts @@ -0,0 +1,39 @@ +import type { PushDatabase } from './push-database.js' + +export type PushReadinessOptions = { + cacheMs?: number + now?: () => number +} + +// The gateway holds no JWKS dependency, so readiness is exactly "can we reach +// the database": /health stays unconditional for the container probe. +export function createPushReadiness( + database: PushDatabase, + options: PushReadinessOptions = {} +): () => Promise { + const cacheMs = options.cacheMs ?? 10_000 + const now = options.now ?? Date.now + let cachedAt = Number.NEGATIVE_INFINITY + let cached = false + + let pending: Promise | null = null + + async function check(): Promise { + try { + await database.query('SELECT 1 AS ready') + cached = true + } catch { + cached = false + } + cachedAt = now() + return cached + } + + return async () => { + if (now() - cachedAt < cacheMs) return cached + pending ??= check().finally(() => { + pending = null + }) + return pending + } +} diff --git a/cloud/apps/push/src/push-request-drain.ts b/cloud/apps/push/src/push-request-drain.ts new file mode 100644 index 00000000000..4acaf09ca67 --- /dev/null +++ b/cloud/apps/push/src/push-request-drain.ts @@ -0,0 +1,28 @@ +import type { MiddlewareHandler } from 'hono' + +export class PushRequestDrain { + private draining = false + private active = 0 + private readonly waiters = new Set<() => void>() + + readonly middleware: MiddlewareHandler = async (context, next) => { + if (this.draining) return context.json({ error: 'shutting_down' }, 503) + this.active++ + try { + await next() + } finally { + this.active-- + if (this.active === 0) { + for (const resolve of this.waiters) resolve() + this.waiters.clear() + } + } + } + + begin(): Promise { + this.draining = true + return this.active === 0 + ? Promise.resolve() + : new Promise((resolve) => this.waiters.add(resolve)) + } +} diff --git a/cloud/apps/push/src/push-schema.ts b/cloud/apps/push/src/push-schema.ts new file mode 100644 index 00000000000..7375d475703 --- /dev/null +++ b/cloud/apps/push/src/push-schema.ts @@ -0,0 +1,46 @@ +import { DURABLE_PUSH_SCHEMA } from './durable-push-schema.js' +// Applied at startup for both dialects, including additive queue tables, +// so every column type has to read the same in SQLite and PostgreSQL. +const PUSH_SCHEMA = ` +CREATE TABLE IF NOT EXISTS push_challenges ( + challenge_id TEXT PRIMARY KEY, + host_fingerprint TEXT NOT NULL, + secret_hash TEXT NOT NULL, + expires_at BIGINT NOT NULL, + consumed_at BIGINT +); +CREATE INDEX IF NOT EXISTS push_challenges_expires_at ON push_challenges(expires_at); + +CREATE TABLE IF NOT EXISTS push_sessions ( + token_hash TEXT PRIMARY KEY, + host_fingerprint TEXT NOT NULL, + expires_at BIGINT NOT NULL, + created_at BIGINT NOT NULL +); +CREATE UNIQUE INDEX IF NOT EXISTS push_sessions_host ON push_sessions(host_fingerprint); +CREATE INDEX IF NOT EXISTS push_sessions_expires_at ON push_sessions(expires_at); + +CREATE TABLE IF NOT EXISTS push_devices ( + registration_id TEXT PRIMARY KEY, + host_fingerprint TEXT NOT NULL, + device_id TEXT NOT NULL, + platform TEXT NOT NULL, + token TEXT NOT NULL, + apns_environment TEXT, + dead_at BIGINT, + created_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); +CREATE UNIQUE INDEX IF NOT EXISTS push_devices_host_device + ON push_devices(host_fingerprint, device_id); +` + +export function pushSchemaStatements(): string[] { + // Comments are stripped before the split so a ';' inside one cannot cut a + // statement in half and hand SQLite an "incomplete input" fragment. + return (PUSH_SCHEMA + DURABLE_PUSH_SCHEMA) + .replace(/--[^\n]*/g, '') + .split(';') + .map((statement) => statement.trim()) + .filter((statement) => statement.length > 0) +} diff --git a/cloud/apps/push/src/push-send-idempotency.test.ts b/cloud/apps/push/src/push-send-idempotency.test.ts new file mode 100644 index 00000000000..1a687a35449 --- /dev/null +++ b/cloud/apps/push/src/push-send-idempotency.test.ts @@ -0,0 +1,64 @@ +import { afterEach, expect, it } from 'vitest' +import { createPushServerHarness, notification } from './push-server-harness.test-fixture.js' +import { createPushHostKeypair } from './host-challenge-answering.test-fixture.js' +const harnesses: Awaited>[] = [] +afterEach(async () => { + await Promise.all(harnesses.splice(0).map((h) => h.close())) +}) + +it('returns queued for concurrent retries without double quota or delivery', async () => { + const h = await createPushServerHarness() + harnesses.push(h) + const token = await h.signIn(createPushHostKeypair(2)) + const registrationId = await h.registerAndroid(token) + const body = { v: 1, registrationIds: [registrationId], notification: notification() } + const responses = await Promise.all( + Array.from({ length: 10 }, () => h.post('/v1/send', body, token)) + ) + for (const response of responses) + expect(await response.json()).toEqual({ results: [{ registrationId, status: 'queued' }] }) + expect(await h.server.deliveryStore.pendingCount(registrationId)).toBe(1) + await h.flushDeliveries() + await h.post('/v1/send', body, token) + await h.flushDeliveries() + expect(h.fcmRequests).toHaveLength(1) + expect(JSON.parse(h.fcmRequests[0]!.body).message.data.coalescedCount).toBeUndefined() + expect( + Number((await h.database.query('SELECT COUNT(*) AS count FROM push_events'))[0]?.count) + ).toBe(1) + await h.post( + '/v1/send', + { ...body, notification: notification({ notificationEpoch: 'new-epoch' }) }, + token + ) + await h.flushDeliveries() + expect(h.fcmRequests).toHaveLength(2) +}) + +it.each([false, true])( + 'accepts default alert kind equivalently through the API (explicit first: %s)', + async (explicitFirst) => { + const h = await createPushServerHarness() + harnesses.push(h) + const token = await h.signIn(createPushHostKeypair(3)) + const registrationId = await h.registerAndroid(token) + const implicit = notification() + const explicit = { kind: 'alert', ...implicit } + for (const event of explicitFirst ? [explicit, implicit] : [implicit, explicit]) { + const response = await h.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: event }, + token + ) + expect(await response.json()).toEqual({ results: [{ registrationId, status: 'queued' }] }) + } + await h.flushDeliveries() + expect(h.fcmRequests).toHaveLength(1) + const changed = await h.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: { ...explicit, body: 'changed' } }, + token + ) + expect(await changed.json()).toEqual({ results: [{ registrationId, status: 'error' }] }) + } +) diff --git a/cloud/apps/push/src/push-server-auth.test.ts b/cloud/apps/push/src/push-server-auth.test.ts new file mode 100644 index 00000000000..a66a0c3bbe6 --- /dev/null +++ b/cloud/apps/push/src/push-server-auth.test.ts @@ -0,0 +1,162 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { createPushHostKeypair } from './host-challenge-answering.test-fixture.js' +import type { PushDatabase } from './push-database.js' +import { createPushServer } from './push-server.js' +import { + createPushServerHarness, + testPushConfig +} from './push-server-harness.test-fixture.js' + +describe('push gateway authentication and device routes', () => { + let harness: Awaited> + + beforeEach(async () => { + harness = await createPushServerHarness() + }) + + afterEach(async () => { + await harness.close() + }) + + it('answers health unconditionally and ready from the database', async () => { + expect((await harness.server.app.request('/health')).status).toBe(200) + expect((await harness.server.app.request('/ready')).status).toBe(200) + }) + + it('reports not ready when the database is unreachable', async () => { + const unreachable: PushDatabase = { + dialect: 'sqlite', + query: async () => { + throw new Error('no connection') + }, + transaction: async (operation) => await operation(unreachable), + lockQuotaScope: async () => undefined, + close: async () => undefined + } + const broken = createPushServer(testPushConfig(), unreachable, { + fcmAccessToken: async () => 'token', + fcmTransport: async () => ({ status: 200, body: '{}' }) + }) + expect((await broken.app.request('/health')).status).toBe(200) + expect((await broken.app.request('/ready')).status).toBe(503) + await broken.worker.stop() + }) + + it('completes challenge, session, register, list, delete', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(11)) + const registrationId = await harness.registerAndroid(sessionToken) + + const list = await harness.authorized('/v1/devices', {}, sessionToken) + expect(await list.json()).toEqual({ + devices: [{ registrationId, deviceId: 'device-1', platform: 'android', dead: false }] + }) + + const deleted = await harness.authorized( + `/v1/devices/${registrationId}`, + { method: 'DELETE' }, + sessionToken + ) + expect(deleted.status).toBe(204) + expect(await harness.server.devices.findById(registrationId)).toBeNull() + }) + + it('refuses a request with no bearer, a bogus bearer, and an expired session', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(12)) + expect((await harness.server.app.request('/v1/devices')).status).toBe(401) + const bogus = await harness.authorized('/v1/devices', {}, 'nonsense') + expect(bogus.status).toBe(401) + expect(await bogus.json()).toEqual({ error: 'invalid_token' }) + + harness.advanceClock(PUSH_LIMITS.sessionTtlMs + 1) + const expired = await harness.authorized('/v1/devices', {}, sessionToken) + expect(expired.status).toBe(401) + expect(await expired.json()).toEqual({ error: 'session_expired' }) + }) + + it('refuses a replayed proof and an unknown challenge', async () => { + const host = createPushHostKeypair(13) + const challenge = await harness.issueChallenge(host) + const proof = harness.answer(challenge, host) + expect( + ( + await harness.post('/v1/host/session', { + v: 1, + challengeId: challenge.challengeId, + proofB64: proof + }) + ).status + ).toBe(200) + + const replay = await harness.post('/v1/host/session', { + v: 1, + challengeId: challenge.challengeId, + proofB64: proof + }) + expect(replay.status).toBe(401) + expect(await replay.json()).toEqual({ error: 'invalid_proof' }) + + const unknown = await harness.post('/v1/host/session', { + v: 1, + challengeId: 'no-such-challenge', + proofB64: proof + }) + expect(await unknown.json()).toEqual({ error: 'invalid_challenge' }) + }) + + it('never returns the host fingerprint on the challenge itself', async () => { + const challenge = await harness.issueChallenge(createPushHostKeypair(22)) + expect(Object.keys(challenge).sort()).toEqual([ + 'challengeId', + 'ciphertextB64', + 'expiresAt', + 'gatewayEphemeralPublicKeyB64', + 'nonceB64' + ]) + }) + + it('lets only the owning host delete a registration', async () => { + const ownerToken = await harness.signIn(createPushHostKeypair(14)) + const intruderToken = await harness.signIn(createPushHostKeypair(15)) + const registrationId = await harness.registerAndroid(ownerToken) + + const forbidden = await harness.authorized( + `/v1/devices/${registrationId}`, + { method: 'DELETE' }, + intruderToken + ) + expect(forbidden.status).toBe(404) + expect(await forbidden.json()).toEqual({ error: 'not_found' }) + expect(await harness.server.devices.findById(registrationId)).not.toBeNull() + }) + + it('replaces the token on a re-registration and keeps one registration id', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(23)) + const first = await harness.registerAndroid(sessionToken) + const again = await harness.post( + '/v1/devices', + { + v: 1, + deviceId: 'device-1', + platform: 'android', + token: 'rotated_token:APA91b-newnewnewnewnewnewnewnewnewnew' + }, + sessionToken + ) + expect(await again.json()).toEqual({ registrationId: first }) + expect(await harness.server.devices.findById(first)).toMatchObject({ + token: 'rotated_token:APA91b-newnewnewnewnewnewnewnewnewnew' + }) + }) + + it('rejects a malformed registration body', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(16)) + const bad = await harness.post( + '/v1/devices', + { v: 1, deviceId: 'device-1', platform: 'ios', token: 'not-hex' }, + sessionToken + ) + expect(bad.status).toBe(400) + expect(await bad.json()).toEqual({ error: 'invalid_request' }) + }) +}) diff --git a/cloud/apps/push/src/push-server-harness.test-fixture.ts b/cloud/apps/push/src/push-server-harness.test-fixture.ts new file mode 100644 index 00000000000..f1567dd9ab7 --- /dev/null +++ b/cloud/apps/push/src/push-server-harness.test-fixture.ts @@ -0,0 +1,162 @@ +import { generateKeyPairSync } from 'node:crypto' +import { expect } from 'vitest' +import type { ApnsRequest, ApnsResponse } from './apns-http2-transport.js' +import type { PushConfig } from './config.js' +import type { FcmRequest, FcmResponse } from './fcm-client.js' +import { + answerPushHostChallenge, + hostPublicKeyB64, + type PushHostKeypair +} from './host-challenge-answering.test-fixture.js' +import { openInMemoryPushDatabase, type PushDatabase } from './push-database.js' +import { createPushServer } from './push-server.js' + +export const GATEWAY_ORIGIN = 'https://push.onorca.dev' +export const APNS_TOKEN = 'a'.repeat(64) +export const FCM_TOKEN = 'cQ1abcDEF_gh:APA91bZZ-zz0123456789abcdefghijklmnopqrstuvwxyz' + +export function notification(overrides: Record = {}): Record { + return { + notificationId: 'note-1', + notificationSeq: 1, + notificationEpoch: 'epoch-1', + source: 'agent-task-complete', + agentState: 'needs-input', + title: 'Agent needs input', + body: 'Waiting on your answer', + worktreeId: 'wt-1', + ...overrides + } +} + +export function testPushConfig(): PushConfig { + const { privateKey } = generateKeyPairSync('ec', { + namedCurve: 'P-256', + privateKeyEncoding: { type: 'pkcs8', format: 'pem' }, + publicKeyEncoding: { type: 'spki', format: 'pem' } + }) + return { + mode: 'active', + port: 0, + publicUrl: GATEWAY_ORIGIN, + dataDir: './data/push-test', + databasePoolMax: 10, + apns: { keyPem: privateKey, keyId: 'ABCDE12345', teamId: 'TEAM123456' }, + apnsTopic: 'com.stably.orca.mobile', + fcmProjectId: 'onorca-cloud', + trustedProxyHops: 0 + } +} + +type ChallengeWire = { + challengeId: string + gatewayEphemeralPublicKeyB64: string + nonceB64: string + ciphertextB64: string + expiresAt: number +} + +export async function createPushServerHarness() { + const database: PushDatabase = await openInMemoryPushDatabase() + let clock = 1_700_000_000_000 + const apnsRequests: ApnsRequest[] = [] + const fcmRequests: FcmRequest[] = [] + let apnsResponse: ApnsResponse = { status: 200, body: '' } + let fcmResponse: FcmResponse = { status: 200, body: '{}' } + const server = createPushServer(testPushConfig(), database, { + now: () => clock, + apnsTransport: async (request) => { + apnsRequests.push(request) + return apnsResponse + }, + fcmTransport: async (request) => { + fcmRequests.push(request) + return fcmResponse + }, + fcmAccessToken: async () => 'access-token' + }) + + const post = async (path: string, body: unknown, token?: string): Promise => + await server.app.request(path, { + method: 'POST', + headers: { + 'content-type': 'application/json', + ...(token ? { authorization: `Bearer ${token}` } : {}) + }, + body: JSON.stringify(body) + }) + + const issueChallenge = async (keypair: PushHostKeypair): Promise => { + const response = await post('/v1/host/challenge', { + v: 1, + hostPublicKeyB64: hostPublicKeyB64(keypair) + }) + expect(response.status).toBe(200) + return (await response.json()) as ChallengeWire + } + + const answer = (challenge: ChallengeWire, keypair: PushHostKeypair): string => { + const proof = answerPushHostChallenge(challenge, { + gatewayOrigin: GATEWAY_ORIGIN, + keypair, + now: () => clock + }) + expect(proof).not.toBeNull() + return proof! + } + + return { + server, + database, + apnsRequests, + fcmRequests, + post, + issueChallenge, + answer, + now: () => clock, + flushDeliveries: async (): Promise => { + await server.worker.runDue() + }, + advanceClock: (deltaMs: number): void => { + clock += deltaMs + }, + setApnsResponse: (response: ApnsResponse): void => { + apnsResponse = response + }, + setFcmResponse: (response: FcmResponse): void => { + fcmResponse = response + }, + authorized: async (path: string, init: RequestInit = {}, token?: string): Promise => + await server.app.request(path, { + ...init, + headers: { + ...(init.headers as Record | undefined), + ...(token ? { authorization: `Bearer ${token}` } : {}) + } + }), + signIn: async (keypair: PushHostKeypair): Promise => { + const challenge = await issueChallenge(keypair) + const response = await post('/v1/host/session', { + v: 1, + challengeId: challenge.challengeId, + proofB64: answer(challenge, keypair) + }) + expect(response.status).toBe(200) + return ((await response.json()) as { sessionToken: string }).sessionToken + }, + registerAndroid: async (token: string, deviceId = 'device-1'): Promise => { + const response = await post( + '/v1/devices', + { v: 1, deviceId, platform: 'android', token: FCM_TOKEN }, + token + ) + expect(response.status).toBe(200) + return ((await response.json()) as { registrationId: string }).registrationId + }, + close: async (): Promise => { + await server.worker.stop() + // A test may close the database itself to provoke a route failure. + await database.close().catch(() => undefined) + } + } +} diff --git a/cloud/apps/push/src/push-server-limits.test.ts b/cloud/apps/push/src/push-server-limits.test.ts new file mode 100644 index 00000000000..d5931d3ed35 --- /dev/null +++ b/cloud/apps/push/src/push-server-limits.test.ts @@ -0,0 +1,273 @@ +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { createPushHostKeypair, hostPublicKeyB64 } from './host-challenge-answering.test-fixture.js' +import { + createPushServerHarness, + FCM_TOKEN, + notification +} from './push-server-harness.test-fixture.js' + +const CLIENT_IP = '203.0.113.7' +const OTHER_CLIENT_IP = '198.51.100.9' + +function oversizedChallengeBody(): string { + return JSON.stringify({ v: 1, filler: 'x'.repeat(PUSH_LIMITS.maxHttpBodyBytes) }) +} + +function chunkedRequest(path: string, body: string): Request { + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode(body)) + controller.close() + } + }) + return new Request(`http://push.test${path}`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: stream, + duplex: 'half' + } as RequestInit) +} + +describe('push gateway request limits', () => { + let harness: Awaited> + + beforeEach(async () => { + harness = await createPushServerHarness() + }) + + afterEach(async () => { + await harness.close() + }) + + it('refuses an oversized chunked body that declares no content length', async () => { + const request = chunkedRequest('/v1/host/challenge', oversizedChallengeBody()) + expect(request.headers.get('content-length')).toBeNull() + + const response = await harness.server.app.request(request) + expect(response.status).toBe(413) + expect(await response.json()).toEqual({ error: 'request_too_large' }) + }) + + it('still refuses an oversized body that declares a content length', async () => { + const body = oversizedChallengeBody() + const response = await harness.server.app.request('/v1/host/challenge', { + method: 'POST', + headers: { + 'content-type': 'application/json', + 'content-length': String(Buffer.byteLength(body)) + }, + body + }) + expect(response.status).toBe(413) + expect(await response.json()).toEqual({ error: 'request_too_large' }) + }) + + it('lets a chunked body under the cap through to schema validation', async () => { + const response = await harness.server.app.request( + chunkedRequest( + '/v1/host/challenge', + JSON.stringify({ v: 1, hostPublicKeyB64: hostPublicKeyB64(createPushHostKeypair(60)) }) + ) + ) + expect(response.status).toBe(200) + }) + + it('caps an authenticated oversized send as well', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(61)) + const response = await harness.server.app.request( + new Request('http://push.test/v1/send', { + method: 'POST', + headers: { + 'content-type': 'application/json', + authorization: `Bearer ${sessionToken}` + }, + body: new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode(oversizedChallengeBody())) + controller.close() + } + }), + duplex: 'half' + } as RequestInit) + ) + expect(response.status).toBe(413) + expect(await response.json()).toEqual({ error: 'request_too_large' }) + }) + + it('rate limits one client ip across both unauthenticated routes', async () => { + const body = JSON.stringify({ + v: 1, + hostPublicKeyB64: hostPublicKeyB64(createPushHostKeypair(62)) + }) + // Cloud Run appends the peer, so the caller's own IP is the last value. + const headers = { + 'content-type': 'application/json', + 'x-forwarded-for': `10.0.0.1, ${CLIENT_IP}` + } + for (let index = 0; index < PUSH_LIMITS.unauthenticatedRequestsPerMinutePerIp; index++) { + const allowed = await harness.server.app.request('/v1/host/challenge', { + method: 'POST', + headers, + body + }) + expect(allowed.status).toBe(200) + } + + const limited = await harness.server.app.request('/v1/host/challenge', { + method: 'POST', + headers, + body + }) + expect(limited.status).toBe(429) + expect(await limited.json()).toEqual({ error: 'rate_limited' }) + + // The session route draws on the same bucket, so a flood cannot simply move. + const session = await harness.server.app.request('/v1/host/session', { + method: 'POST', + headers, + body: JSON.stringify({ v: 1, challengeId: 'anything', proofB64: 'x'.repeat(44) }) + }) + expect(session.status).toBe(429) + + const other = await harness.server.app.request('/v1/host/challenge', { + method: 'POST', + headers: { ...headers, 'x-forwarded-for': `10.0.0.1, ${OTHER_CLIENT_IP}` }, + body + }) + expect(other.status).toBe(200) + + // A caller rewriting the left of the chain lands in its own bucket anyway. + const spoofed = await harness.server.app.request('/v1/host/challenge', { + method: 'POST', + headers: { ...headers, 'x-forwarded-for': `198.51.100.250, ${CLIENT_IP}` }, + body + }) + expect(spoofed.status).toBe(429) + }) + + it('lets a throttled client back in once the window refills', async () => { + const body = JSON.stringify({ + v: 1, + hostPublicKeyB64: hostPublicKeyB64(createPushHostKeypair(63)) + }) + const headers = { 'content-type': 'application/json', 'x-forwarded-for': CLIENT_IP } + for (let index = 0; index < PUSH_LIMITS.unauthenticatedRequestsPerMinutePerIp; index++) { + await harness.server.app.request('/v1/host/challenge', { method: 'POST', headers, body }) + } + expect( + (await harness.server.app.request('/v1/host/challenge', { method: 'POST', headers, body })) + .status + ).toBe(429) + + harness.advanceClock(60_000) + expect( + (await harness.server.app.request('/v1/host/challenge', { method: 'POST', headers, body })) + .status + ).toBe(200) + }) + + it('limits authenticated hosts independently behind the same IP', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(64)) + const headers = { 'x-forwarded-for': CLIENT_IP } + for (let index = 0; index < 600; index++) { + const listed = await harness.authorized('/v1/devices', { headers }, sessionToken) + expect(listed.status).toBe(200) + } + const limited = await harness.authorized('/v1/devices', { headers }, sessionToken) + expect(limited.status).toBe(429) + const otherToken = await harness.signIn(createPushHostKeypair(68)) + expect((await harness.authorized('/v1/devices', { headers }, otherToken)).status).toBe(200) + // The handshake bucket is untouched by any of that. + const challenge = await harness.server.app.request('/v1/host/challenge', { + method: 'POST', + headers: { ...headers, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, hostPublicKeyB64: hostPublicKeyB64(createPushHostKeypair(67)) }) + }) + expect(challenge.status).toBe(200) + }) + + it('stops repeated forged bearers after the invalid-auth budget is exhausted', async () => { + const headers = { 'x-forwarded-for': CLIENT_IP } + const [before] = await harness.database.query('SELECT COUNT(*) AS sessions FROM push_sessions') + for (let index = 0; index < PUSH_LIMITS.unauthenticatedRequestsPerMinutePerIp; index++) { + const refused = await harness.authorized('/v1/send', { method: 'POST', headers }, 'forged') + expect(refused.status).toBe(401) + } + const limited = await harness.authorized('/v1/send', { method: 'POST', headers }, 'forged') + expect(limited.status).toBe(429) + expect(await limited.json()).toEqual({ error: 'rate_limited' }) + expect(harness.server.unauthenticatedIps.trackedIpCount()).toBe(0) + const [after] = await harness.database.query('SELECT COUNT(*) AS sessions FROM push_sessions') + expect(Number(after?.sessions)).toBe(Number(before?.sessions)) + }) + + it('answers 409 once a host has registered its device allowance', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(66)) + for (let index = 0; index < PUSH_LIMITS.maxDevicesPerHost; index++) { + const accepted = await harness.post( + '/v1/devices', + { + v: 1, + deviceId: `device-${index}`, + platform: 'android', + token: FCM_TOKEN + }, + sessionToken + ) + expect(accepted.status).toBe(200) + } + + const refused = await harness.post( + '/v1/devices', + { v: 1, deviceId: 'one-too-many', platform: 'android', token: FCM_TOKEN }, + sessionToken + ) + expect(refused.status).toBe(409) + expect(await refused.json()).toEqual({ error: 'too_many_devices' }) + + const listed = await harness.authorized('/v1/devices', {}, sessionToken) + expect(((await listed.json()) as { devices: unknown[] }).devices).toHaveLength( + PUSH_LIMITS.maxDevicesPerHost + ) + }) + + // Why: a database error carries the failing row in its message. The response + // and the log must both stop at the error's name. + it('answers an unexpected route failure with a bare 500 and logs only the name', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(66)) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + await harness.database.close() + const response = await harness.authorized('/v1/devices', {}, sessionToken) + expect(response.status).toBe(500) + expect(await response.json()).toEqual({ error: 'internal' }) + const logged = warn.mock.calls.map((call) => String(call[0])).join('\n') + expect(logged).toContain('"event":"orca_push_request_failed"') + expect(logged).not.toContain('SELECT') + expect(logged).not.toContain('push_devices') + expect(harness.server.observability.consume().request_error).toBe(1) + } finally { + warn.mockRestore() + } + }) + + it('charges a repeated registration id once and returns one result', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(65)) + const registrationId = await harness.registerAndroid(sessionToken) + + const response = await harness.post( + '/v1/send', + { + v: 1, + registrationIds: [registrationId, registrationId, registrationId], + notification: notification() + }, + sessionToken + ) + expect(await response.json()).toEqual({ results: [{ registrationId, status: 'queued' }] }) + expect(await harness.server.deliveryStore.pendingCount(registrationId)).toBe(1) + const [row] = await harness.database.query('SELECT COUNT(*) AS sends FROM push_events') + expect(Number(row?.sends)).toBe(1) + }) +}) diff --git a/cloud/apps/push/src/push-server-send.test.ts b/cloud/apps/push/src/push-server-send.test.ts new file mode 100644 index 00000000000..4b14b77eaec --- /dev/null +++ b/cloud/apps/push/src/push-server-send.test.ts @@ -0,0 +1,229 @@ +import { PushNotificationSchema } from '@orca-cloud/push-contract' +import { PUSH_LIMITS } from '@orca-cloud/push-contract' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { createPushHostKeypair } from './host-challenge-answering.test-fixture.js' +import { + APNS_TOKEN, + createPushServerHarness, + FCM_TOKEN, + notification +} from './push-server-harness.test-fixture.js' + +describe('push gateway send route', () => { + let harness: Awaited> + + beforeEach(async () => { + harness = await createPushServerHarness() + }) + + afterEach(async () => { + await harness.close() + }) + + it('rejects a batch over the registration cap', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(16)) + const oversized = await harness.post( + '/v1/send', + { + v: 1, + registrationIds: Array.from( + { length: PUSH_LIMITS.maxRegistrationIdsPerSend + 1 }, + (_, index) => `reg-${index}` + ), + notification: notification() + }, + sessionToken + ) + expect(oversized.status).toBe(400) + expect(await oversized.json()).toEqual({ error: 'invalid_request' }) + }) + + it('queues a send, delivers it to fcm, and reports a dead token on the next send', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(17)) + const registrationId = await harness.registerAndroid(sessionToken) + + const queued = await harness.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: notification() }, + sessionToken + ) + expect(await queued.json()).toEqual({ results: [{ registrationId, status: 'queued' }] }) + + harness.setFcmResponse({ + status: 404, + body: JSON.stringify({ error: { status: 'UNREGISTERED', message: 'gone' } }) + }) + await harness.flushDeliveries() + expect(harness.fcmRequests).toHaveLength(1) + expect(JSON.parse(harness.fcmRequests[0]!.body)).toMatchObject({ + message: { token: FCM_TOKEN, data: { title: 'Agent needs input' } } + }) + + const afterDeath = await harness.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: notification() }, + sessionToken + ) + expect(await afterDeath.json()).toEqual({ results: [{ registrationId, status: 'dead' }] }) + + const listed = await harness.authorized('/v1/devices', {}, sessionToken) + expect(await listed.json()).toEqual({ + devices: [{ registrationId, deviceId: 'device-1', platform: 'android', dead: true }] + }) + }) + + it('leaves a live registration alone when the provider reports a transient failure', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(24)) + const registrationId = await harness.registerAndroid(sessionToken) + await harness.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: notification() }, + sessionToken + ) + harness.setFcmResponse({ + status: 503, + body: JSON.stringify({ error: { status: 'UNAVAILABLE', message: 'backend busy' } }) + }) + await harness.flushDeliveries() + expect(await harness.server.devices.findById(registrationId)).toMatchObject({ dead: false }) + }) + + it('reports retries from the durable worker after the provider delay', async () => { + const token = await harness.signIn(createPushHostKeypair(26)) + const registrationId = await harness.registerAndroid(token) + await harness.post( + '/v1/send', + { + v: 1, + registrationIds: [registrationId], + notification: notification() + }, + token + ) + harness.setFcmResponse({ status: 503, body: '{}' }) + await harness.flushDeliveries() + expect(harness.server.observability.consume()).toMatchObject({ + delivery_error: 1, + delivery_retry: 0 + }) + harness.advanceClock(10_000) + harness.setFcmResponse({ status: 200, body: '{}' }) + await harness.server.worker.runDue() + expect(harness.server.observability.consume()).toMatchObject({ + delivery_sent: 1, + delivery_retry: 1 + }) + }) + + it('sends a burst as individual APNs alerts grouped by the host thread', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(18)) + const registration = await harness.post( + '/v1/devices', + { + v: 1, + deviceId: 'iphone-1', + platform: 'ios', + token: APNS_TOKEN, + apnsEnvironment: 'sandbox' + }, + sessionToken + ) + const { registrationId } = (await registration.json()) as { registrationId: string } + for (const seq of [1, 2, 3]) { + await harness.post( + '/v1/send', + { + v: 1, + registrationIds: [registrationId], + notification: notification({ notificationId: `note-${seq}`, notificationSeq: seq }) + }, + sessionToken + ) + } + await harness.flushDeliveries() + expect(harness.apnsRequests).toHaveLength(3) + const bodies = harness.apnsRequests.map( + (request) => + JSON.parse(request.body) as { + aps: { alert: { title: string; body: string }; 'thread-id': string } + orca: Record & { notificationSeq: number } + } + ) + expect( + harness.apnsRequests.every((request) => request.host === 'api.sandbox.push.apple.com') + ).toBe(true) + expect(bodies.map((body) => body.aps.alert)).toEqual( + Array.from({ length: 3 }, () => ({ + title: 'Agent needs input', + body: 'Waiting on your answer' + })) + ) + expect(new Set(bodies.map((body) => body.aps['thread-id'])).size).toBe(1) + expect(bodies.map((body) => body.orca.notificationSeq).sort((a, b) => a - b)).toEqual([1, 2, 3]) + expect(bodies.every((body) => !('coalescedCount' in body.orca))).toBe(true) + expect(bodies.every((body) => !('summaryMembers' in body.orca))).toBe(true) + expect( + new Set(harness.apnsRequests.map((request) => request.headers['apns-collapse-id'])).size + ).toBe(3) + }) + + it('sends a lone event through unchanged with its own collapse id', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(25)) + const registrationId = await harness.registerAndroid(sessionToken) + await harness.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: notification() }, + sessionToken + ) + await harness.flushDeliveries() + const message = JSON.parse(harness.fcmRequests[0]!.body) as { + message: { android: { notification: { tag: string } }; data: Record } + } + expect(message.message.data.tag).toMatch(/^[a-f0-9]{64}$/) + expect(message.message.data.coalescedCount).toBeUndefined() + }) + + it('reports an error for a registration the host does not own', async () => { + const ownerToken = await harness.signIn(createPushHostKeypair(19)) + const intruderToken = await harness.signIn(createPushHostKeypair(20)) + const registrationId = await harness.registerAndroid(ownerToken) + + const foreign = await harness.post( + '/v1/send', + { v: 1, registrationIds: [registrationId, 'made-up'], notification: notification() }, + intruderToken + ) + expect(await foreign.json()).toEqual({ + results: [ + { registrationId, status: 'error' }, + { registrationId: 'made-up', status: 'error' } + ] + }) + expect(await harness.server.deliveryStore.pendingCount(registrationId)).toBe(0) + }) + + it('rate limits a host that exhausted its 15-minute allowance', async () => { + const sessionToken = await harness.signIn(createPushHostKeypair(21)) + const registrationId = await harness.registerAndroid(sessionToken) + const hostFingerprint = (await harness.server.devices.findById(registrationId))!.hostFingerprint + for (let index = 0; index < PUSH_LIMITS.hostEventsPerWindow; index++) { + expect( + await harness.server.deliveryStore.accept( + hostFingerprint, + registrationId, + PushNotificationSchema.parse( + notification({ notificationId: `note-${index + 1000}`, notificationSeq: index + 1000 }) + ) + ) + ).toBe('queued') + } + const limited = await harness.post( + '/v1/send', + { v: 1, registrationIds: [registrationId], notification: notification() }, + sessionToken + ) + expect(limited.status).toBe(200) + expect(await limited.json()).toEqual({ results: [{ registrationId, status: 'rate_limited' }] }) + expect(await harness.server.deliveryStore.pendingCount(registrationId)).toBe(300) + }) +}) diff --git a/cloud/apps/push/src/push-server.ts b/cloud/apps/push/src/push-server.ts new file mode 100644 index 00000000000..911f7890155 --- /dev/null +++ b/cloud/apps/push/src/push-server.ts @@ -0,0 +1,298 @@ +import { PushAuthAdmission } from './push-auth-admission.js' +import { createAdaptorServer } from '@hono/node-server' +import { + PUSH_LIMITS, + PushDeviceRegistrationRequestSchema, + PushHostChallengeRequestSchema, + PushHostSessionRequestSchema, + PushSendRequestSchema, + type PushSendResult +} from '@orca-cloud/push-contract' +import { Hono, type MiddlewareHandler } from 'hono' +import { bodyLimit } from 'hono/body-limit' +import { ApnsClient } from './apns-client.js' +import { createApnsHttp2Transport, type ApnsTransport } from './apns-http2-transport.js' +import { clientIpRateLimit, ClientIpRateLimiter, readClientIp } from './client-ip-rate-limit.js' +import { DurablePushStore } from './durable-push-store.js' +import { DurablePushWorker } from './durable-push-worker.js' +import type { PushConfig } from './config.js' +import { PushDeviceRegistryStore } from './device-registry-store.js' +import { createFcmAccessTokenProvider } from './fcm-access-token.js' +import { createFcmFetchTransport, FcmClient, type FcmTransport } from './fcm-client.js' +import { PushHostChallengeStore } from './host-challenge-store.js' +import { PushHostSessionStore } from './host-session-store.js' +import type { PushDatabase } from './push-database.js' +import { PushDispatcher } from './push-dispatcher.js' +import { PushObservability } from './push-observability.js' +import { createPushReadiness } from './push-readiness.js' +import { PushRequestDrain } from './push-request-drain.js' + +export type PushServerOptions = { + now?: () => number + apnsTransport?: ApnsTransport + fcmTransport?: FcmTransport + fcmAccessToken?: () => Promise +} + +type PushVariables = { hostFingerprint: string } + +export function readBearer(header: string | undefined): string | null { + if (!header) return null + const [scheme, ...rest] = header.split(' ') + const token = rest.join(' ').trim() + return scheme?.toLowerCase() === 'bearer' && token.length > 0 ? token : null +} + +// Hono's body limit, not a Content-Length check: a chunked body declares no +// length, and req.json() would buffer all of it before any handler ran. +const limitBody = bodyLimit({ + maxSize: PUSH_LIMITS.maxHttpBodyBytes, + onError: (context) => context.json({ error: 'request_too_large' }, 413) +}) + +export function createPushServer( + config: PushConfig, + database: PushDatabase, + options: PushServerOptions = {} +) { + const now = options.now ?? Date.now + const observability = new PushObservability() + const challenges = new PushHostChallengeStore(database, config.publicUrl, now) + const sessions = new PushHostSessionStore(database, now) + const devices = new PushDeviceRegistryStore(database, now) + const deliveryStore = new DurablePushStore(database, now) + const apnsTransport = options.apnsTransport ?? (config.apns ? createApnsHttp2Transport() : null) + const dispatcher = new PushDispatcher({ + devices, + ...(config.apns && apnsTransport + ? { + apns: new ApnsClient({ + topic: config.apnsTopic, + credentials: config.apns, + transport: apnsTransport, + now + }) + } + : {}), + fcm: new FcmClient({ + now, + projectId: config.fcmProjectId, + accessToken: options.fcmAccessToken ?? createFcmAccessTokenProvider(), + transport: options.fcmTransport ?? createFcmFetchTransport() + }), + onOutcome: (status) => + observability.record( + status === 'sent' ? 'delivery_sent' : status === 'dead' ? 'delivery_dead' : 'delivery_error' + ) + }) + const worker = new DurablePushWorker(deliveryStore, dispatcher, { + now, + onRetry: () => observability.record('delivery_retry') + }) + const ready = createPushReadiness(database, { now }) + const unauthenticatedIps = new ClientIpRateLimiter({ now }) + const limitUnauthenticatedIp = clientIpRateLimit(unauthenticatedIps, { + trustedProxyHops: config.trustedProxyHops, + onLimited: () => observability.record('ip_rate_limited') + }) + const limitAuthenticatedIp = clientIpRateLimit( + new ClientIpRateLimiter({ + now, + capacity: PUSH_LIMITS.authenticatedRequestsPerMinutePerIp + }), + { + trustedProxyHops: config.trustedProxyHops, + onLimited: () => observability.record('ip_rate_limited') + } + ) + const authAdmission = new PushAuthAdmission() + const invalidBearerIps = new ClientIpRateLimiter({ now }) + const authenticatedHosts = new ClientIpRateLimiter({ + now, + capacity: PUSH_LIMITS.authenticatedRequestsPerMinutePerHost + }) + const app = new Hono<{ Variables: PushVariables }>() + const requestDrain = new PushRequestDrain() + app.use('*', requestDrain.middleware) + // Hono's default handler prints the whole error, and a pg error carries the + // offending row in `detail`. Only the error's name may reach the logs. + app.onError((error, context) => { + observability.record('request_error') + console.warn( + JSON.stringify({ + event: 'orca_push_request_failed', + error: error instanceof Error ? error.name : 'unknown' + }) + ) + return context.json({ error: 'internal' }, 500) + }) + + app.get('/health', (context) => + context.json({ ok: true, pushProtocol: 1, deliveryProtocol: 2, mode: config.mode }) + ) + app.get('/ready', limitUnauthenticatedIp, async (context) => + (await ready()) + ? context.json({ ok: true }) + : context.json({ error: 'dependency_unavailable' }, 503) + ) + + if (config.mode === 'validation') { + app.use('*', async (context) => context.json({ error: 'validation_only' }, 503)) + } + + const bearerSession: MiddlewareHandler<{ Variables: PushVariables }> = async (context, next) => { + const ip = readClientIp(context, config.trustedProxyHops) + if (!invalidBearerIps.available(ip)) return context.json({ error: 'rate_limited' }, 429) + const bearer = readBearer(context.req.header('authorization')) + if (!bearer) { + invalidBearerIps.allow(ip) + return context.json({ error: 'invalid_token' }, 401) + } + const session = await authAdmission.run(async () => { + if (!invalidBearerIps.available(ip)) return null + const result = await sessions.resolve(bearer) + if (!result.ok) invalidBearerIps.allow(ip) + return result + }) + if (!session) { + context.header('Retry-After', '1') + return context.json({ error: 'busy' }, 503) + } + if (!session.ok) { + return context.json( + { error: session.reason === 'session_expired' ? 'session_expired' : 'invalid_token' }, + 401 + ) + } + if (!authenticatedHosts.allow(session.hostFingerprint)) + return context.json({ error: 'rate_limited' }, 429) + context.set('hostFingerprint', session.hostFingerprint) + await next() + return + } + // `/v1/devices/*` matches `/v1/devices` itself; a second registration for the + // bare path would run both middlewares twice on it. + app.use('/v1/devices/*', limitAuthenticatedIp, bearerSession) + app.use('/v1/send', limitAuthenticatedIp, bearerSession) + + app.post('/v1/host/challenge', limitUnauthenticatedIp, limitBody, async (context) => { + const body = PushHostChallengeRequestSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + const issued = await challenges.issue(body.data.hostPublicKeyB64) + if (!issued) { + observability.record('challenge_rejected') + return context.json({ error: 'invalid_request' }, 400) + } + observability.record('challenge_issued') + const { hostFingerprint: _bound, ...response } = issued + return context.json(response) + }) + + app.post('/v1/host/session', limitUnauthenticatedIp, limitBody, async (context) => { + const body = PushHostSessionRequestSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + const verification = await challenges.verify(body.data.challengeId, body.data.proofB64) + if (!verification.ok) { + observability.record('session_rejected') + return context.json( + { + error: verification.reason === 'unknown_challenge' ? 'invalid_challenge' : 'invalid_proof' + }, + 401 + ) + } + observability.record('session_issued') + return context.json(await sessions.create(verification.hostFingerprint)) + }) + + app.post('/v1/devices', limitBody, async (context) => { + const body = PushDeviceRegistrationRequestSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + const registered = await devices.upsert({ + hostFingerprint: context.get('hostFingerprint'), + deviceId: body.data.deviceId, + platform: body.data.platform, + token: body.data.token, + ...(body.data.apnsEnvironment === undefined + ? {} + : { apnsEnvironment: body.data.apnsEnvironment }) + }) + if (!registered.ok) { + observability.record('device_rejected') + return context.json({ error: 'too_many_devices' }, 409) + } + observability.record('device_registered') + return context.json({ registrationId: registered.registrationId }) + }) + + app.delete('/v1/devices/:registrationId', async (context) => { + const deleted = await devices.deleteOwned( + context.get('hostFingerprint'), + context.req.param('registrationId') + ) + if (!deleted) return context.json({ error: 'not_found' }, 404) + observability.record('device_deleted') + return context.body(null, 204) + }) + + app.get('/v1/devices', async (context) => + context.json({ devices: await devices.list(context.get('hostFingerprint')) }) + ) + + app.post('/v1/send', limitBody, async (context) => { + const body = PushSendRequestSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + const hostFingerprint = context.get('hostFingerprint') + const owned = await devices.findOwned(hostFingerprint, body.data.registrationIds) + const results: PushSendResult[] = [] + for (const registrationId of body.data.registrationIds) { + const device = owned.get(registrationId) + if (!device) { + observability.record('send_error') + results.push({ registrationId, status: 'error' }) + continue + } + if (device.dead) { + observability.record('send_dead') + results.push({ registrationId, status: 'dead' }) + continue + } + const reservation = await deliveryStore.accept( + hostFingerprint, + registrationId, + body.data.notification + ) + if (reservation !== 'queued') { + observability.record(reservation === 'rate_limited' ? 'send_rate_limited' : 'send_error') + results.push({ registrationId, status: reservation }) + continue + } + observability.record('send_queued') + results.push({ registrationId, status: 'queued' }) + } + return context.json({ results }) + }) + + return { + app, + requestDrain, + server: createAdaptorServer(app), + challenges, + sessions, + devices, + deliveryStore, + unauthenticatedIps, + worker, + observability, + ready, + closeTransports: (): void => { + if (apnsTransport && 'close' in apnsTransport) { + ;(apnsTransport as { close: () => void }).close() + } + } + } +} diff --git a/cloud/apps/push/src/push-session-concurrency.test.ts b/cloud/apps/push/src/push-session-concurrency.test.ts new file mode 100644 index 00000000000..4c6944255f8 --- /dev/null +++ b/cloud/apps/push/src/push-session-concurrency.test.ts @@ -0,0 +1,65 @@ +import { randomUUID } from 'node:crypto' +import { tmpdir } from 'node:os' +import { afterEach, describe, expect, it } from 'vitest' +import { openInMemoryPushDatabase, openPushDatabase, type PushDatabase } from './push-database.js' +import { PushHostSessionStore } from './host-session-store.js' +const databases: PushDatabase[] = [] +afterEach(async () => { + await Promise.all(databases.splice(0).map((db) => db.close())) +}) + +async function concurrentSessions(db: PushDatabase) { + databases.push(db) + const host = randomUUID() + const store = new PushHostSessionStore(db) + try { + const sessions = await Promise.all(Array.from({ length: 20 }, () => store.create(host))) + const decisions = await Promise.all( + sessions.map((session) => store.resolve(session.sessionToken)) + ) + expect(decisions.filter((decision) => decision.ok)).toHaveLength(1) + const [row] = await db.query( + 'SELECT COUNT(*) AS count FROM push_sessions WHERE host_fingerprint = ?', + [host] + ) + expect(Number(row?.count)).toBe(1) + } finally { + await db.query('DELETE FROM push_sessions WHERE host_fingerprint = ?', [host]) + } +} +it('serializes sessions on SQLite', async () => { + await concurrentSessions(await openInMemoryPushDatabase()) +}) + +it('enforces one session per host directly in the schema', async () => { + const db = await openInMemoryPushDatabase() + databases.push(db) + await db.query('INSERT INTO push_sessions VALUES (?, ?, ?, ?)', ['first', 'host', 100, 1]) + await expect( + db.query('INSERT INTO push_sessions VALUES (?, ?, ?, ?)', ['second', 'host', 100, 2]) + ).rejects.toThrow() + expect(await db.query('SELECT token_hash FROM push_sessions')).toEqual([{ token_hash: 'first' }]) +}) + +describe.skipIf(!process.env.ORCA_PUSH_TEST_DATABASE_URL)('PostgreSQL push sessions', () => { + it('leaves exactly one live token after concurrent creates', async () => { + await concurrentSessions( + await openPushDatabase({ + databaseUrl: process.env.ORCA_PUSH_TEST_DATABASE_URL!, + dataDir: tmpdir() + }) + ) + }) + it('allows concurrent schema startup', async () => { + const opened = await Promise.all( + Array.from({ length: 4 }, () => + openPushDatabase({ + databaseUrl: process.env.ORCA_PUSH_TEST_DATABASE_URL!, + dataDir: tmpdir() + }) + ) + ) + databases.push(...opened) + for (const db of opened) expect(await db.query('SELECT 1 AS ok')).toEqual([{ ok: 1 }]) + }) +}) diff --git a/cloud/apps/push/src/push-validation.test.ts b/cloud/apps/push/src/push-validation.test.ts new file mode 100644 index 00000000000..afb6906e423 --- /dev/null +++ b/cloud/apps/push/src/push-validation.test.ts @@ -0,0 +1,90 @@ +import { expect, it, vi } from 'vitest' +import { loadPushConfig } from './config.js' +import { openPushDatabase } from './push-database.js' +import { createPushServer } from './push-server.js' +import { startPushBackground } from './push-background.js' + +it('fails closed on an invalid validation mode', () => { + for (const mode of ['typo', '', ' ']) { + expect(() => + loadPushConfig({ + ORCA_PUSH_FCM_PROJECT_ID: 'onorca-cloud', + ORCA_PUSH_PUBLIC_URL: 'https://push.onorca.dev', + ORCA_PUSH_MODE: mode + }) + ).toThrow() + } +}) + +const databaseUrl = process.env.ORCA_PUSH_TEST_DATABASE_URL +it.skipIf(!databaseUrl)( + 'validation cannot write PostgreSQL and starts no consumers or pruners', + async () => { + if (!process.env.CI && new URL(databaseUrl!).port !== '55440') + throw new Error('isolated_postgres_port_required') + const active = await openPushDatabase({ databaseUrl, dataDir: '' }) + const schema = `validation_${Date.now()}` + await active.query(`CREATE SCHEMA ${schema}`) + const isolatedUrl = new URL(databaseUrl!) + isolatedUrl.searchParams.set( + 'options', + `-c search_path=${schema} -c default_transaction_read_only=off` + ) + isolatedUrl.searchParams.set('host', isolatedUrl.hostname) + isolatedUrl.searchParams.set('port', isolatedUrl.port) + const hostlessUrl = `postgresql://${isolatedUrl.username}:${isolatedUrl.password}@${isolatedUrl.pathname}?${isolatedUrl.searchParams}` + const database = await openPushDatabase({ + databaseUrl: hostlessUrl, + dataDir: '', + readOnly: true + }) + const config = loadPushConfig({ + ORCA_PUSH_FCM_PROJECT_ID: 'onorca-cloud', + ORCA_PUSH_PUBLIC_URL: 'https://push.onorca.dev', + ORCA_PUSH_MODE: 'validation' + }) + const runtime = createPushServer(config, database) + let stop: (() => Promise) | undefined + try { + const [setting] = await database.query( + "SELECT current_setting('default_transaction_read_only') AS default_transaction_read_only" + ) + expect(setting!.default_transaction_read_only).toBe('on') + await expect( + database.query(`CREATE TABLE ${schema}.forbidden (id integer)`) + ).rejects.toMatchObject({ code: '25006' }) + // An empty schema stays empty: validation must not run startup DDL. + expect( + await active.query('SELECT tablename FROM pg_tables WHERE schemaname = ?', [schema]) + ).toEqual([]) + const calls = vi.spyOn(database, 'query') + const claim = vi.spyOn(runtime.deliveryStore, 'claim') + const send = vi.spyOn(runtime.worker, 'start') + vi.useFakeTimers() + stop = startPushBackground(config, runtime) + await vi.advanceTimersByTimeAsync(31 * 60_000) + expect(calls).not.toHaveBeenCalled() + expect(claim).not.toHaveBeenCalled() + expect(send).not.toHaveBeenCalled() + vi.useRealTimers() + expect(await (await runtime.app.request('/health')).json()).toMatchObject({ + mode: 'validation' + }) + expect((await runtime.app.request('/ready')).status).toBe(200) + expect((await runtime.app.request('/v1/host/challenge', { method: 'POST' })).status).toBe(503) + expect((await runtime.app.request('/v1/send', { method: 'POST' })).status).toBe(503) + expect(calls.mock.calls.map(([sql]) => sql)).toEqual(['SELECT 1 AS ready']) + await expect( + database.query('DELETE FROM public.push_challenges WHERE false') + ).rejects.toMatchObject({ code: '25006' }) + } finally { + vi.useRealTimers() + await stop?.() + runtime.closeTransports() + await database.close() + await active.query(`DROP SCHEMA ${schema} CASCADE`) + await active.close() + vi.restoreAllMocks() + } + } +) diff --git a/cloud/apps/push/tsconfig.build.json b/cloud/apps/push/tsconfig.build.json new file mode 100644 index 00000000000..5e71eb0f951 --- /dev/null +++ b/cloud/apps/push/tsconfig.build.json @@ -0,0 +1,10 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "declaration": true, + "noEmit": false, + "outDir": "dist", + "rootDir": "src" + }, + "exclude": ["src/**/*.test.ts", "src/**/*.test-fixture.ts"] +} diff --git a/cloud/apps/push/tsconfig.json b/cloud/apps/push/tsconfig.json new file mode 100644 index 00000000000..a552e34dbe9 --- /dev/null +++ b/cloud/apps/push/tsconfig.json @@ -0,0 +1,5 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "noEmit": true }, + "include": ["src/**/*.ts"] +} diff --git a/cloud/apps/push/vitest.config.ts b/cloud/apps/push/vitest.config.ts new file mode 100644 index 00000000000..bffcc30e39e --- /dev/null +++ b/cloud/apps/push/vitest.config.ts @@ -0,0 +1,5 @@ +import { defineConfig } from 'vitest/config' + +export default defineConfig({ + test: { name: 'push', include: ['src/**/*.test.ts'], testTimeout: 15_000, hookTimeout: 15_000 } +}) diff --git a/cloud/apps/relay/Dockerfile b/cloud/apps/relay/Dockerfile index 12516cbf749..f0abcf9f5b3 100644 --- a/cloud/apps/relay/Dockerfile +++ b/cloud/apps/relay/Dockerfile @@ -3,11 +3,13 @@ WORKDIR /app RUN corepack enable COPY package.json pnpm-lock.yaml pnpm-workspace.yaml tsconfig.base.json ./ COPY packages/relay-contract/package.json packages/relay-contract/package.json +COPY packages/postgres-schema/package.json packages/postgres-schema/package.json COPY apps/relay/package.json apps/relay/package.json RUN pnpm install --frozen-lockfile COPY packages/relay-contract packages/relay-contract COPY apps/relay apps/relay -RUN pnpm --filter @orca-cloud/relay-contract build && pnpm --filter @orca-cloud/relay build +COPY packages/postgres-schema packages/postgres-schema +RUN pnpm --filter @orca-cloud/postgres-schema build && pnpm --filter @orca-cloud/relay-contract build && pnpm --filter @orca-cloud/relay build FROM node:24-alpine AS runtime ENV NODE_ENV=production @@ -16,8 +18,10 @@ WORKDIR /app RUN corepack enable COPY package.json pnpm-lock.yaml pnpm-workspace.yaml ./ COPY packages/relay-contract/package.json packages/relay-contract/package.json +COPY packages/postgres-schema/package.json packages/postgres-schema/package.json COPY apps/relay/package.json apps/relay/package.json COPY --from=build /app/packages/relay-contract/dist packages/relay-contract/dist +COPY --from=build /app/packages/postgres-schema/dist packages/postgres-schema/dist COPY --from=build /app/apps/relay/dist apps/relay/dist RUN pnpm install --prod --frozen-lockfile --filter @orca-cloud/relay... USER node diff --git a/cloud/apps/relay/package.json b/cloud/apps/relay/package.json index de30d66e413..824513ea446 100644 --- a/cloud/apps/relay/package.json +++ b/cloud/apps/relay/package.json @@ -9,13 +9,14 @@ "clean": "node -e \"require('fs').rmSync('dist', { recursive: true, force: true })\"", "dev": "tsx watch src/index.ts", "lint": "tsc -p tsconfig.json --noEmit", - "pretest": "pnpm --filter @orca-cloud/relay-contract build", + "pretest": "pnpm --filter @orca-cloud/postgres-schema build && pnpm --filter @orca-cloud/relay-contract build", "start": "node dist/index.js", "test": "vitest run", "typecheck": "tsc -p tsconfig.json --noEmit" }, "dependencies": { "@hono/node-server": "^1.19.17", + "@orca-cloud/postgres-schema": "workspace:*", "@orca-cloud/relay-contract": "workspace:*", "hono": "^4.13.7", "jose": "^6.1.3", diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index 9ead45df22e..824cb1e0b2f 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -362,6 +362,9 @@ export const REGIONAL_REHOME_QUARANTINE_FAILURES = 3 export const REGIONAL_REHOME_QUARANTINE_MS = 15 * 60_000 const REGIONAL_REHOME_QUARANTINE_EXCLUSION_LIMIT = 50 const REGIONAL_REHOME_QUARANTINE_MEMORY_LIMIT = 1_000 +// Consecutive drain-dispatch failures that latch the durable control off. Any +// drain receipt and any enable reset it, so it reads "dispatch is broken now". +const REGIONAL_REHOME_FAILURE_BUDGET = 3 const REGIONAL_REHOME_OBSERVATION_MS = 24 * 60 * 60_000 const ASSIGNMENT_LOCK_RETRY_MAX_DELAY_MS = 50 type AssignmentInventoryScope = 'none' | 'general' | 'all' @@ -4992,6 +4995,20 @@ export class RelayAssignmentStore { now ] ) + if (input.enabled) { + // A budget spent under a previous enable is not evidence about this one. + // Without this an old counter latches the fresh enable straight back off + // on its first transient failure. + await transaction.query( + `INSERT INTO relay_region_rehome_worker_state + (worker_id, next_dispatch_at, paused_until, consecutive_failures, updated_at) + VALUES ('global', 0, 0, 0, ?) + ON CONFLICT (worker_id) DO UPDATE + SET paused_until = 0, consecutive_failures = 0, + updated_at = excluded.updated_at`, + [now] + ) + } const updated = ( await transaction.query( `SELECT * FROM relay_region_rehome_control WHERE control_id = 'global'` @@ -5940,7 +5957,11 @@ export class RelayAssignmentStore { async recordRegionalRehomeDispatchFailure(attemptId: string): Promise { const now = this.now() - await this.database.transaction(async (transaction) => { + const disableLog = await this.database.transaction(async (transaction) => { + // Match claim and enable ordering before a spent budget updates the control. + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_control WHERE control_id = 'global'` + ) const worker = ( await transaction.queryLocked( `SELECT * FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` @@ -5952,49 +5973,44 @@ export class RelayAssignmentStore { [attemptId] ) )[0] - if (!worker || !attempt) return - await this.incrementRegionalRehomeWorkerFailure(transaction, worker, now) - }) - } - - async recordRegionalRehomeWorkerFailure(): Promise { - const now = this.now() - await this.database.transaction(async (transaction) => { - await transaction.query( - `INSERT INTO relay_region_rehome_worker_state - (worker_id, next_dispatch_at, paused_until, consecutive_failures, updated_at) - VALUES ('global', 0, 0, 0, ?) - ON CONFLICT (worker_id) DO NOTHING`, - [now] - ) - const worker = ( - await transaction.queryLocked( - `SELECT * FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` - ) - )[0]! - await this.incrementRegionalRehomeWorkerFailure(transaction, worker, now) + if (!worker || !attempt) return null + return await this.incrementRegionalRehomeWorkerFailure(transaction, worker, now) }) + // Logged after the commit so a rollback cannot fabricate the record. + if (disableLog) console.warn(JSON.stringify(disableLog)) } + // Returns the durable disable this failure caused, for the caller to log once + // its transaction commits; null when the budget survives or was already spent. private async incrementRegionalRehomeWorkerFailure( transaction: RelayDatabase, worker: SqlRow, now: number - ): Promise { + ): Promise | null> { const failures = integer(worker, 'consecutive_failures') + 1 + const spent = failures >= REGIONAL_REHOME_FAILURE_BUDGET await transaction.query( `UPDATE relay_region_rehome_worker_state SET consecutive_failures = ?, paused_until = ?, updated_at = ? WHERE worker_id = 'global'`, - [failures, failures >= 3 ? now + 5 * 60_000 : 0, now] + [failures, spent ? now + 5 * 60_000 : 0, now] ) - if (failures >= 3) { - await transaction.query( - `UPDATE relay_region_rehome_control - SET generation = generation + 1, enabled = 0, updated_at = ? - WHERE control_id = 'global' AND enabled = 1`, - [now] - ) + if (!spent) return null + const disabled = await transaction.query( + `UPDATE relay_region_rehome_control + SET generation = generation + 1, enabled = 0, updated_at = ? + WHERE control_id = 'global' AND enabled = 1 + RETURNING generation`, + [now] + ) + // The disable is otherwise invisible: inspection only shows enabled=false and + // nothing records that the failure budget, not an operator, turned it off. + if (disabled.length === 0) return null + return { + event: 'orca_relay_regional_rehome_failure_budget_disabled', + controlGeneration: integer(disabled[0]!, 'generation'), + consecutiveFailures: failures, + now } } diff --git a/cloud/apps/relay/src/postgres-schema-startup.ts b/cloud/apps/relay/src/postgres-schema-startup.ts index 22a75cd9465..3a3428eda32 100644 --- a/cloud/apps/relay/src/postgres-schema-startup.ts +++ b/cloud/apps/relay/src/postgres-schema-startup.ts @@ -1,120 +1 @@ -const RETRYABLE_SCHEMA_CODES = new Set(['55P03', '57014']) -const DEFAULT_RETRY_DEADLINE_MS = 30_000 -const RETRY_BASE_DELAY_MS = 250 -const RETRY_MAX_DELAY_MS = 2_000 - -type SchemaStartupOptions = { - now?: () => number - random?: () => number - retryDeadlineMs?: number - wait?: (delayMs: number) => Promise -} - -function retryDelayMs(attempt: number, random: () => number): number { - const ceiling = Math.min( - RETRY_BASE_DELAY_MS * 2 ** (attempt - 1), - RETRY_MAX_DELAY_MS - ) - return Math.ceil(ceiling * (0.5 + random() * 0.5)) -} - -function wait(delayMs: number): Promise { - return new Promise((resolve) => setTimeout(resolve, delayMs)) -} - -const CREATE_TABLE_IF_NOT_EXISTS = /^\s*CREATE\s+TABLE\s+IF\s+NOT\s+EXISTS\b/i -const CREATE_INDEX_IF_NOT_EXISTS = /^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+IF\s+NOT\s+EXISTS\b/i - -// `IF NOT EXISTS` only checks the name before the catalog inserts, so the loser of a concurrent -// CREATE can fail on the catalog unique index (23505) or, when the winner has already committed by -// the time the loser reaches TypeCreate/heap_create_with_catalog, on the name check those routines -// repeat (42710 duplicate type, 42P07 duplicate relation). Each is a no-op on the next attempt. -function concurrentCreateCollision( - value: { code?: unknown; constraint?: unknown }, - statement: string -): boolean { - if (CREATE_TABLE_IF_NOT_EXISTS.test(statement)) { - return ( - (value.code === '23505' && value.constraint === 'pg_type_typname_nsp_index') || - value.code === '42710' || - value.code === '42P07' - ) - } - if (CREATE_INDEX_IF_NOT_EXISTS.test(statement)) { - return ( - (value.code === '23505' && value.constraint === 'pg_class_relname_nsp_index') || - value.code === '42P07' - ) - } - return false -} - -const ALTER_TABLE_ADD_CONSTRAINT = - /^\s*ALTER\s+TABLE\s+\S+\s+ADD\s+CONSTRAINT\b/i - -// Postgres has no `ADD CONSTRAINT IF NOT EXISTS`, so a re-run and a concurrent -// startup both land on 42710 once the constraint exists. Unlike a CREATE race -// this is terminal, not transient: retrying only repeats it, so the statement -// counts as applied. -function constraintAlreadyApplied(error: unknown, statement: string): boolean { - return ( - ALTER_TABLE_ADD_CONSTRAINT.test(statement) && - (error as { code?: unknown }).code === '42710' - ) -} - -function retryableSchemaError(error: unknown, statement: string): boolean { - const value = error as { code?: unknown; constraint?: unknown } - return ( - RETRYABLE_SCHEMA_CODES.has(String(value.code)) || concurrentCreateCollision(value, statement) - ) -} - -export async function applyPostgresSchema( - statements: string[], - query: (statement: string) => Promise, - options: SchemaStartupOptions = {} -): Promise { - const now = options.now ?? Date.now - const random = options.random ?? Math.random - const pause = options.wait ?? wait - const deadlineAt = now() + (options.retryDeadlineMs ?? DEFAULT_RETRY_DEADLINE_MS) - - for (const statement of statements) { - let attempt = 1 - while (true) { - try { - await query(statement) - break - } catch (error) { - if (constraintAlreadyApplied(error, statement)) break - const code = String((error as { code?: unknown }).code) - const remainingMs = deadlineAt - now() - const retryable = retryableSchemaError(error, statement) - if (!retryable || remainingMs <= 0) { - if (retryable) { - console.warn( - JSON.stringify({ - event: 'orca_relay_postgres_schema_retry_exhausted', - code, - attempts: attempt - }) - ) - } - throw error - } - const delayMs = Math.min(remainingMs, retryDelayMs(attempt, random)) - console.warn( - JSON.stringify({ - event: 'orca_relay_postgres_schema_retry', - code, - attempt, - delayMs - }) - ) - await pause(delayMs) - attempt += 1 - } - } - } -} +export { applyPostgresSchema } from '@orca-cloud/postgres-schema' diff --git a/cloud/apps/relay/src/regional-rehome-postgres.test.ts b/cloud/apps/relay/src/regional-rehome-postgres.test.ts index 44f3b3434af..fdefda54401 100644 --- a/cloud/apps/relay/src/regional-rehome-postgres.test.ts +++ b/cloud/apps/relay/src/regional-rehome-postgres.test.ts @@ -1,4 +1,4 @@ -import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest' +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' import { RelayAssignmentStore } from './assignment-store.js' import { openRelayDatabase, type RelayDatabase } from './database.js' import { @@ -267,6 +267,91 @@ describePostgres('PostgreSQL regional rehoming', () => { )).toEqual([{ count: '1' }]) }) + it('serializes an enable with a budget-exhausting failure without retries', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + const locked = Promise.withResolvers() + const release = Promise.withResolvers() + const primaryTransaction = primary.transaction.bind(primary) + const secondaryTransaction = secondary.transaction.bind(secondary) + let enableTransactions = 0 + let failureTransactions = 0 + let enablePid = 0 + let failurePid = 0 + const enableSpy = vi.spyOn(primary, 'transaction').mockImplementation((operation, options) => + primaryTransaction(async (transaction) => { + enableTransactions++ + enablePid = Number((await transaction.query('SELECT pg_backend_pid() AS pid'))[0]!.pid) + return await operation({ + dialect: 'postgres', + query: transaction.query.bind(transaction), + queryLocked: async (sql, params, lockOptions) => { + const rows = await transaction.queryLocked(sql, params, lockOptions) + if (sql.includes('FROM relay_region_rehome_control')) { + locked.resolve() + await release.promise + } + return rows + }, + transaction: transaction.transaction.bind(transaction), + close: transaction.close.bind(transaction) + }) + }, options) + ) + const failureSpy = vi.spyOn(secondary, 'transaction').mockImplementation((operation, options) => + secondaryTransaction(async (transaction) => { + failureTransactions++ + failurePid = Number((await transaction.query('SELECT pg_backend_pid() AS pid'))[0]!.pid) + return await operation(transaction) + }, options) + ) + const enable = context.store.applyRegionalRehomeControl({ + expectedGeneration: 1, + enabled: true, + notBefore: context.now(), + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + hostCooldownMs: 7 * 24 * 60 * 60_000, + drainGraceMs: 60_000 + }) + let failure: Promise | undefined + let outcomes: PromiseSettledResult[] = [] + try { + await Promise.race([ + locked.promise, + enable.then(() => { + throw new Error('enable completed before the control lock') + }) + ]) + failure = context.competingStore.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + // Observe the actual PostgreSQL wait before letting enable acquire the worker row. + await vi.waitFor(async () => { + expect(failurePid).not.toBe(0) + const rows = await primary.query('SELECT pg_blocking_pids(?) AS blockers', [failurePid]) + expect(rows[0]!.blockers).toContain(enablePid) + }, { interval: 10, timeout: 800 }) + } finally { + release.resolve() + outcomes = await Promise.allSettled([enable, ...(failure ? [failure] : [])]) + enableSpy.mockRestore() + failureSpy.mockRestore() + } + expect(outcomes.map((outcome) => outcome.status)).toEqual(['fulfilled', 'fulfilled']) + expect({ enableTransactions, failureTransactions }).toEqual({ + enableTransactions: 1, + failureTransactions: 1 + }) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: true + }) + expect(await primary.query( + `SELECT consecutive_failures, paused_until FROM relay_region_rehome_worker_state` + )).toEqual([{ consecutive_failures: '1', paused_until: '0' }]) + }) + it('increments the disable generation once across competing directors', async () => { const context = await fixture() const disabled = await Promise.all([ diff --git a/cloud/apps/relay/src/regional-rehome-store.test.ts b/cloud/apps/relay/src/regional-rehome-store.test.ts index 2c1c8132266..662876ef66e 100644 --- a/cloud/apps/relay/src/regional-rehome-store.test.ts +++ b/cloud/apps/relay/src/regional-rehome-store.test.ts @@ -1709,6 +1709,77 @@ describe('regional rehome assignment state', () => { expect(await context.store.claimRegionalRehome()).toBeNull() await context.database.close() }) + + it('clears a stale failure budget when the control is enabled again', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + const attempt = await context.store.claimRegionalRehome() + for (let index = 0; index < 3; index++) { + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + } + expect(await workerState(context)).toMatchObject({ consecutiveFailures: 3 }) + const latched = await context.store.inspectRegionalRehomeControl() + expect(latched).toMatchObject({ generation: 2, enabled: false }) + + await context.store.applyRegionalRehomeControl({ + expectedGeneration: latched.generation, + enabled: true, + notBefore: context.now(), + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + hostCooldownMs: 7 * 24 * 60 * 60_000, + drainGraceMs: 60 * 60_000 + }) + + // A budget spent under the previous enable is not evidence about this one. + expect(await workerState(context)).toMatchObject({ + consecutiveFailures: 0, + pausedUntil: 0 + }) + // One transient failure must not latch the fresh enable straight back off. + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 3, + enabled: true + }) + await context.database.close() + }) + + it('reports the durable disable when the failure budget latches the control off', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + const attempt = await context.store.claimRegionalRehome() + const warnings = collectEventWarnings( + 'orca_relay_regional_rehome_failure_budget_disabled' + ) + try { + for (let index = 0; index < 5; index++) { + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + } + } finally { + warnings.restore() + } + + // Only the transition is reported; later failures find the control already off. + expect(warnings.entries).toEqual([ + expect.objectContaining({ + event: 'orca_relay_regional_rehome_failure_budget_disabled', + controlGeneration: 2, + consecutiveFailures: 3 + }) + ]) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + await context.database.close() + }) }) class TransactionCountingDatabase implements RelayDatabase { @@ -2066,3 +2137,18 @@ class CellInventoryLockProbe { return decorate(database) } } + +async function workerState( + context: Context +): Promise<{ consecutiveFailures: number; pausedUntil: number }> { + const row = ( + await context.database.query( + `SELECT consecutive_failures, paused_until + FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` + ) + )[0]! + return { + consecutiveFailures: Number(row.consecutive_failures), + pausedUntil: Number(row.paused_until) + } +} diff --git a/cloud/apps/relay/src/regional-rehome-worker.test.ts b/cloud/apps/relay/src/regional-rehome-worker.test.ts index af905bb9ab2..33e7f01f737 100644 --- a/cloud/apps/relay/src/regional-rehome-worker.test.ts +++ b/cloud/apps/relay/src/regional-rehome-worker.test.ts @@ -30,11 +30,9 @@ describe('regional rehome worker', () => { } const claimRegionalRehome = vi.fn().mockResolvedValueOnce(null).mockResolvedValue(attempt) const recordRegionalRehomeDrainReceipt = vi.fn().mockResolvedValue(true) - const recordRegionalRehomeWorkerFailure = vi.fn().mockResolvedValue(undefined) const assignments = { claimRegionalRehome, - recordRegionalRehomeDrainReceipt, - recordRegionalRehomeWorkerFailure + recordRegionalRehomeDrainReceipt } as unknown as RelayAssignmentStore const requests: Array<{ url: string; init?: RequestInit }> = [] const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) @@ -99,8 +97,7 @@ describe('regional rehome worker', () => { const claimRegionalRehome = vi.fn().mockResolvedValueOnce(null).mockResolvedValue(attempt) const assignments = { claimRegionalRehome, - recordRegionalRehomeDispatchFailure: vi.fn().mockResolvedValue(undefined), - recordRegionalRehomeWorkerFailure: vi.fn().mockResolvedValue(undefined) + recordRegionalRehomeDispatchFailure: vi.fn().mockResolvedValue(undefined) } as unknown as RelayAssignmentStore const worker = startRegionalRehomeWorker(config(), assignments, { now: () => now, @@ -118,7 +115,36 @@ describe('regional rehome worker', () => { expect(assignments.recordRegionalRehomeDispatchFailure).toHaveBeenCalledWith( '11111111-1111-4111-8111-111111111111' ) - expect(assignments.recordRegionalRehomeWorkerFailure).not.toHaveBeenCalled() + }) + + it('keeps a failed poll out of the durable dispatch-failure budget', async () => { + let now = 0 + const claimRegionalRehome = vi + .fn() + .mockResolvedValueOnce(null) + .mockRejectedValue(new Error('Connection terminated due to connection timeout')) + const recordRegionalRehomeDispatchFailure = vi.fn().mockResolvedValue(undefined) + const assignments = { + claimRegionalRehome, + recordRegionalRehomeDispatchFailure + } as unknown as RelayAssignmentStore + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const worker = startRegionalRehomeWorker(config(), assignments, { + now: () => now, + safetySnapshot: () => safety(now), + intervalMs: 60_000 + })! + await settleWorker() + now = 1_000 + await expect(worker.run()).resolves.toBeUndefined() + worker.stop() + + // The poll never claimed an attempt, so nothing was drained and nothing may + // be charged to the budget that latches the durable control off. + expect(recordRegionalRehomeDispatchFailure).not.toHaveBeenCalled() + expect(warn.mock.calls.map((call) => JSON.parse(String(call[0])).event)).toEqual([ + 'orca_relay_regional_rehome_poll_failed' + ]) }) it('passes unsafe process telemetry to the durable claim gate', async () => { diff --git a/cloud/apps/relay/src/regional-rehome-worker.ts b/cloud/apps/relay/src/regional-rehome-worker.ts index 47a2748cff4..4d8fa694afd 100644 --- a/cloud/apps/relay/src/regional-rehome-worker.ts +++ b/cloud/apps/relay/src/regional-rehome-worker.ts @@ -97,13 +97,19 @@ export function startRegionalRehomeWorker( }) ) } catch (error) { - await (attemptId - ? assignments.recordRegionalRehomeDispatchFailure(attemptId) - : assignments.recordRegionalRehomeWorkerFailure() - ).catch(() => undefined) + // Only a claimed attempt was drained. A poll that failed before the claim + // - a pool timeout on the once-a-second control read - dispatched nothing, + // so it must not spend the budget that latches the durable control off. + if (attemptId) { + await assignments + .recordRegionalRehomeDispatchFailure(attemptId) + .catch(() => undefined) + } console.warn( JSON.stringify({ - event: 'orca_relay_regional_rehome_dispatch_failed', + event: attemptId + ? 'orca_relay_regional_rehome_dispatch_failed' + : 'orca_relay_regional_rehome_poll_failed', reason: error instanceof Error ? error.message : 'unknown' }) ) diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json index dd6f6944322..da91ef44a19 100644 --- a/cloud/dev/fixtures/terraform-root-partition/families.json +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -86,17 +86,21 @@ "relay": [ "google_artifact_registry_repository_iam_member.github_production_relay_staging_mirror_writer", "google_artifact_registry_repository_iam_member.github_production_relay_writer", + "google_artifact_registry_repository_iam_member.github_push_artifact_writer", "google_artifact_registry_repository_iam_member.github_relay_asia_topology_artifact_reader", "google_artifact_registry_repository_iam_member.github_staging_relay_deploy_artifact_reader", "google_certificate_manager_certificate.relay_gce", "google_certificate_manager_certificate_map.relay_gce", "google_certificate_manager_certificate_map_entry.relay_gce", "google_certificate_manager_dns_authorization.relay_gce", + "google_cloud_run_domain_mapping.push", "google_cloud_run_domain_mapping.relay", "google_cloud_run_domain_mapping.relay_cell", + "google_cloud_run_v2_service.push", "google_cloud_run_v2_service.relay", "google_cloud_run_v2_service.relay_cell", "google_cloud_run_v2_service.relay_fence_broker", + "google_cloud_run_v2_service_iam_member.github_production_push_developer", "google_cloud_run_v2_service_iam_member.github_production_relay_director_developer", "google_cloud_run_v2_service_iam_member.github_production_relay_fence_broker_developer", "google_cloud_run_v2_service_iam_member.github_staging_relay_capacity_developer", @@ -125,6 +129,7 @@ "google_iam_workload_identity_pool_provider.github_fence", "google_iam_workload_identity_pool_provider.github_monitor", "google_iam_workload_identity_pool_provider.github_production_relay_capacity", + "google_iam_workload_identity_pool_provider.github_push", "google_iam_workload_identity_pool_provider.github_relay_asia_proof", "google_iam_workload_identity_pool_provider.github_relay_asia_topology", "google_iam_workload_identity_pool_provider.github_staging_relay_capacity", @@ -172,6 +177,9 @@ "google_project_iam_member.github_staging_relay_capacity_viewer", "google_project_iam_member.github_staging_relay_deploy_compute_viewer", "google_project_iam_member.github_staging_relay_power", + "google_project_iam_member.push_runtime_cloudsql_client", + "google_project_iam_member.push_runtime_fcm_admin", + "google_project_iam_member.push_runtime_service_usage_consumer", "google_project_iam_member.relay_director_runtime_cloudsql_client", "google_project_iam_member.relay_fence_broker_artifact_reader", "google_project_iam_member.relay_fence_broker_compute_viewer", @@ -180,9 +188,13 @@ "google_project_iam_member.relay_runtime_artifact_reader", "google_project_iam_member.relay_runtime_cloudsql_client", "google_project_iam_member.relay_runtime_log_writer", + "google_secret_manager_secret.push_dedicated_database_url", + "google_secret_manager_secret.push_provider", "google_secret_manager_secret.relay_assignment_signing_key", "google_secret_manager_secret.relay_database_url", "google_secret_manager_secret.relay_regional_placement_enabled", + "google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor", + "google_secret_manager_secret_iam_member.push_provider_runtime_accessor", "google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor", "google_secret_manager_secret_iam_member.relay_assignment_signing_key_director_accessor", "google_secret_manager_secret_iam_member.relay_database_url_accessor", @@ -192,24 +204,30 @@ "google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer", "google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor", "google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor", + "google_secret_manager_secret_version.push_dedicated_database_url", "google_secret_manager_secret_version.relay_assignment_signing_key", "google_secret_manager_secret_version.relay_database_url", "google_secret_manager_secret_version.relay_regional_placement_enabled", "google_service_account.github_fence", "google_service_account.github_monitor", "google_service_account.github_production_relay_capacity", + "google_service_account.github_push_deploy", "google_service_account.github_relay_asia_proof", "google_service_account.github_relay_asia_topology", "google_service_account.github_staging_relay_capacity", "google_service_account.github_staging_relay_deploy", + "google_service_account.push_runtime", "google_service_account.relay_director_runtime", "google_service_account.relay_fence_broker", "google_service_account.relay_runtime", "google_service_account_iam_member.github_accepted_repository_workload_identity_user", "google_service_account_iam_member.github_fence_workload_identity_user", "google_service_account_iam_member.github_monitor_workload_identity_user", + "google_service_account_iam_member.github_production_push_runtime_token_creator", + "google_service_account_iam_member.github_production_push_runtime_user", "google_service_account_iam_member.github_production_relay_capacity_runtime_user", "google_service_account_iam_member.github_production_relay_capacity_workload_identity_user", + "google_service_account_iam_member.github_push_workload_identity_user", "google_service_account_iam_member.github_relay_asia_proof_workload_identity_user", "google_service_account_iam_member.github_relay_asia_topology_runtime_user", "google_service_account_iam_member.github_relay_asia_topology_workload_identity_user", @@ -221,9 +239,13 @@ "google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user", "google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user", "google_service_account_iam_member.relay_fence_broker_requester_token_creator", + "google_sql_database.push_dedicated", "google_sql_database.relay", + "google_sql_database_instance.push_dedicated", + "google_sql_user.push_dedicated", "google_sql_user.relay", "google_storage_bucket_iam_member.github_production_relay_capacity_state", + "google_storage_bucket_iam_member.github_push_rollout_lease", "google_storage_bucket_iam_member.github_relay_asia_topology_state", "google_storage_bucket_iam_member.github_relay_asia_topology_state_list", "google_storage_bucket_iam_member.github_staging_relay_capacity_state", @@ -231,6 +253,7 @@ "google_storage_bucket_iam_member.github_staging_relay_deploy_state_list", "google_storage_bucket_iam_member.relay_fence_broker_bucket_reader", "google_storage_bucket_iam_member.relay_fence_broker_state_objects", + "random_password.push_dedicated_database", "random_password.relay_assignment_signing_key", "random_password.relay_database" ], diff --git a/cloud/dev/scripts/push-cloud-run-model.mjs b/cloud/dev/scripts/push-cloud-run-model.mjs new file mode 100644 index 00000000000..fea782e625d --- /dev/null +++ b/cloud/dev/scripts/push-cloud-run-model.mjs @@ -0,0 +1,140 @@ +import { readFileSync, writeFileSync } from 'node:fs' + +// Executable fake gcloud: revision deletion obeys the platform's latest/traffic constraints. +const path = process.env.MODEL_STATE +const state = JSON.parse(readFileSync(path, 'utf8')) +const args = process.argv.slice(2) +const option = (name) => args[args.indexOf(name) + 1] +const has = (name) => args.includes(name) +const fail = (message) => { + throw new Error(message) +} +const persist = () => writeFileSync(path, JSON.stringify(state)) +const output = (value) => console.log(typeof value === 'string' ? value : JSON.stringify(value)) +const revision = (name) => state.revisions[name] ?? fail(`missing revision ${name}`) +const traffic = () => [ + { revisionName: state.serving, percent: 100 }, + ...Object.entries(state.tags).map(([tag, name]) => ({ + tag, + revisionName: name, + url: `https://${tag}.test` + })) +] +state.trace.push(args.join(' ')) +try { + if (args[0] === 'curl') { + if (state.failure === 'public' && args.some((arg) => arg.includes('https://public.test'))) { + fail('public check failed') + } + const url = args.find((arg) => arg.startsWith('https://')) + const tag = new URL(url).hostname.split('.')[0] + const name = state.tags[tag] ?? state.serving + if (has('-w')) { + output('200') + } else { + output({ + ok: true, + deliveryProtocol: 2, + mode: revision(name).spec.containers[0].env.some((entry) => entry.value === 'validation') + ? 'validation' + : 'active' + }) + } + } else if (args.slice(0, 2).join(' ') === 'run deploy') { + const name = `${option('deploy')}-${option('--revision-suffix')}` + if (state.failure === 'deploy-before') { + fail('deploy failed before create') + } + const item = structuredClone(revision(state.latest)) + item.metadata.name = name + item.spec.containers[0].image = option('--image') + item.status.imageDigest = option('--image') + item.spec.containers[0].env = item.spec.containers[0].env.filter( + (entry) => entry.name !== 'ORCA_PUSH_MODE' + ) + if (has('--update-env-vars')) { + item.spec.containers[0].env.push({ name: 'ORCA_PUSH_MODE', value: 'validation' }) + } + state.revisions[name] = item + state.latest = name + if (has('--tag')) { + state.tags[option('--tag')] = name + } + state.peak = Math.max(state.peak, Object.keys(state.revisions).length) + if (state.peak > 3) { + fail('three-revision budget exceeded') + } + if (state.failure === 'deploy-after') { + fail('deploy failed after create') + } + } else if (args.slice(0, 3).join(' ') === 'run services describe') { + if (state.failure === 'describe') { + fail('describe failed') + } + if (args.some((arg) => arg.includes('value(status.latestCreatedRevisionName)'))) { + output(state.latest) + } else { + const template = structuredClone(revision(state.latest)) + delete template.spec.containers[0].name + output({ + spec: { template }, + status: { latestCreatedRevisionName: state.latest, traffic: traffic() } + }) + } + } else if (args.slice(0, 3).join(' ') === 'run services update-traffic') { + if (has('--to-revisions')) { + const name = option('--to-revisions').split('=')[0] + revision(name) + state.serving = name + } + if (has('--remove-tags')) { + for (const tag of option('--remove-tags').split(',')) { + delete state.tags[tag] + } + } + if (has('--clear-tags')) { + state.tags = {} + } + if (state.failure === 'traffic-after') { + fail('traffic changed but response failed') + } + } else if (args.slice(0, 3).join(' ') === 'run revisions list') { + const names = Object.keys(state.revisions) + output( + (has('--filter') + ? names.filter((name) => name === option('--filter').split('=')[1]) + : names + ).join('\n') + ) + } else if (args.slice(0, 3).join(' ') === 'run revisions describe') { + const item = revision(args[3]) + const format = args.find((arg) => arg.startsWith('--format=')) ?? option('--format') + if (format.includes('minScale')) { + output(item.metadata.annotations['autoscaling.knative.dev/minScale']) + } else if (format.includes('maxScale')) { + output(item.metadata.annotations['autoscaling.knative.dev/maxScale']) + } else { + output(item) + } + } else if (args.slice(0, 3).join(' ') === 'run revisions delete') { + const name = args[3] + if (name === state.latest) { + fail('FAILED_PRECONDITION: latest created Revision cannot be directly deleted') + } + if (name === state.serving || Object.values(state.tags).includes(name)) { + fail('revision has traffic or tags') + } + if (state.failure === 'delete') { + fail('delete failed') + } + revision(name) + delete state.revisions[name] + } else { + fail(`unmodeled gcloud call ${args.join(' ')}`) + } +} catch (error) { + console.error(error.message) + process.exitCode = 1 +} finally { + persist() +} diff --git a/cloud/dev/scripts/push-gateway-recovery.test.mjs b/cloud/dev/scripts/push-gateway-recovery.test.mjs new file mode 100644 index 00000000000..97d9c4b5539 --- /dev/null +++ b/cloud/dev/scripts/push-gateway-recovery.test.mjs @@ -0,0 +1,373 @@ +import assert from 'node:assert/strict' +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { spawnSync } from 'node:child_process' +import test from 'node:test' +import { readRelayWorkflow } from './relay-repository.mjs' + +const workflow = readRelayWorkflow('push-deploy.yml') +function step(name) { + const start = workflow.indexOf(` - name: ${name}\n`) + assert.notEqual(start, -1) + const end = workflow.indexOf('\n - ', start + 1) + const block = workflow.slice(start, end === -1 ? undefined : end) + return block + .slice(block.indexOf(' run: |\n') + ' run: |\n'.length) + .split('\n') + .filter((line) => line.startsWith(' ')) + .map((line) => line.slice(10)) + .join('\n') +} +const names = { + preflight: 'Record the serving revision and require its Terraform-owned scaling', + candidate: 'Deploy the candidate revision with no traffic', + activate: 'Retire inert validation and activate the verified image', + shift: 'Shift all traffic to the verified candidate', + public: 'Verify the public origin after the shift', + rollback: 'Roll traffic back to the previous revision', + restore: 'Restore the known-good service template', + promoteRecovery: 'Promote and verify the known-good recovery revision', + cleanup: 'Delete the rejected candidate revision', + retire: 'Retire previous consumers after public checks' +} +const image = `registry/push@sha256:${'a'.repeat(64)}` +const spec = { + serviceAccountName: 'runtime@test', + containerConcurrency: 40, + containers: [ + { + image, + name: 'push-test-1', + env: [ + { + name: 'ORCA_PUSH_DATABASE_URL', + valueFrom: { secretKeyRef: { name: 'database', key: '7' } } + }, + { name: 'ORCA_PUSH_DATABASE_POOL_MAX', value: '2' } + ] + } + ] +} +const prior = { + metadata: { + name: 'push-test-old', + annotations: { + 'autoscaling.knative.dev/minScale': '1', + 'autoscaling.knative.dev/maxScale': '2' + } + }, + spec, + status: { imageDigest: image } +} +const model = fileURLToPath(new URL('./push-cloud-run-model.mjs', import.meta.url)) +const options = { skip: process.platform === 'win32' } +function exercise(callback) { + const dir = mkdtempSync(join(tmpdir(), 'push-workflow-')) + const statePath = join(dir, 'state.json') + writeFileSync( + statePath, + JSON.stringify({ + revisions: { 'push-test-old': prior }, + latest: 'push-test-old', + serving: 'push-test-old', + tags: {}, + peak: 1, + trace: [] + }) + ) + writeFileSync(join(dir, 'env'), '') + const env = { + ...process.env, + SERVICE_NAME: 'push-test', + GCP_PROJECT_ID: 'test', + GCP_REGION: 'test', + GITHUB_RUN_ID: '123', + GITHUB_RUN_ATTEMPT: '1', + IMAGE: `registry/push@sha256:${'b'.repeat(64)}`, + PUSH_MIN_INSTANCES: '1', + PUSH_MAX_INSTANCES: '2', + PUSH_RUNTIME_SERVICE_ACCOUNT: 'runtime@test', + PUSH_ORIGIN: 'https://public.test', + MODEL_STATE: statePath, + MODEL_SCRIPT: model, + RUNNER_TEMP: dir, + GITHUB_ENV: join(dir, 'env'), + GITHUB_STEP_SUMMARY: join(dir, 'summary') + } + const state = () => JSON.parse(readFileSync(statePath, 'utf8')) + const change = (edit) => { + const value = state() + edit(value) + writeFileSync(statePath, JSON.stringify(value)) + } + const run = (key, ok = true, extra = '') => { + const result = spawnSync( + 'bash', + [ + '-c', + ` + set -a + source "$GITHUB_ENV" + gcloud() { node "$MODEL_SCRIPT" "$@"; } + curl() { node "$MODEL_SCRIPT" curl "$@"; } + sleep() { :; } + ${extra} + ${names[key] ? step(names[key]) : key} + ` + ], + { cwd: dir, env, encoding: 'utf8', timeout: 30000 } + ) + assert.equal(result.status === 0, ok, `${key}: ${result.stderr}\n${result.stdout}`) + return result + } + try { + callback({ run, state, change, dir }) + } finally { + rmSync(dir, { recursive: true, force: true }) + } +} +function recover(h) { + h.change((state) => { + delete state.failure + }) + h.run('restore') + h.run('promoteRecovery') + h.run('cleanup') + h.run('retire') + const state = h.state() + assert.equal(state.serving, 'push-test-r123-1') + assert.equal(state.latest, state.serving) + assert.deepEqual(Object.keys(state.revisions), [state.serving]) + assert.equal(state.revisions[state.serving].spec.containers[0].image, image) + assert.ok(state.peak <= 3) +} + +test( + 'the executable Cloud Run model rejects deleting latest even without tags or traffic', + options, + () => + exercise((h) => { + h.run('preflight') + h.run('candidate') + h.run('gcloud run services update-traffic "$SERVICE_NAME" --clear-tags') + const result = h.run('gcloud run revisions delete "$CANDIDATE_REVISION"', false) + assert.match(result.stderr, /FAILED_PRECONDITION: latest created Revision/) + }) +) + +test( + 'success creates successor before retirement and repeated rollouts retain one consumer', + options, + () => + exercise((h) => { + for (const attempt of ['1', '2']) { + if (attempt === '2') { + writeFileSync(join(h.dir, 'env'), 'GITHUB_RUN_ATTEMPT=2\n') + } + for (const key of ['preflight', 'candidate', 'activate', 'shift', 'public', 'retire']) { + h.run(key) + } + const state = h.state() + assert.equal(state.serving, `push-test-a123-${attempt}`) + assert.deepEqual(Object.keys(state.revisions), [state.serving]) + assert.equal(state.peak, 3) + } + }) +) + +for (const failure of ['deploy-before', 'deploy-after', 'describe']) { + test(`validation ${failure} recovers without deleting latest`, options, () => + exercise((h) => { + h.run('preflight') + h.change((state) => { + state.failure = failure + }) + h.run('candidate', false) + recover(h) + }) + ) +} +for (const failure of ['deploy-before', 'deploy-after', 'delete', 'describe']) { + test( + `activation ${failure} frees validation slot before recovery and stays within three`, + options, + () => + exercise((h) => { + h.run('preflight') + h.run('candidate') + h.change((state) => { + state.failure = failure + }) + h.run('activate', false) + recover(h) + }) + ) +} + +test( + 'ambiguous traffic shift records intent before mutation, rolls back and recovers', + options, + () => + exercise((h) => { + h.run('preflight') + h.run('candidate') + h.run('activate') + h.change((state) => { + state.failure = 'traffic-after' + }) + h.run('shift', false) + assert.match(readFileSync(join(h.dir, 'env'), 'utf8'), /TRAFFIC_SHIFT_ATTEMPTED=true/) + h.change((state) => { + delete state.failure + }) + h.run('rollback') + recover(h) + }) +) + +test('failed public check rolls back and recovers', options, () => + exercise((h) => { + for (const key of ['preflight', 'candidate', 'activate', 'shift']) { + h.run(key) + } + h.change((state) => { + state.failure = 'public' + }) + h.run('public', false) + h.change((state) => { + delete state.failure + }) + h.run('rollback') + recover(h) + }) +) + +for (const defect of [ + 'runtime', + 'secret', + 'mode', + 'image', + 'traffic', + 'scaling', + 'deploy-before', + 'deploy-after' +]) { + test(`recovery rejects ${defect} and preserves partial-create state`, options, () => + exercise((h) => { + h.run('preflight') + h.run('candidate') + h.change((state) => { + const revision = state.revisions[state.latest] + if (defect === 'runtime') { + revision.spec.serviceAccountName = 'wrong@test' + } + if (defect === 'secret') { + revision.spec.containers[0].env[0].valueFrom.secretKeyRef.key = '8' + } + if (defect === 'scaling') { + revision.metadata.annotations['autoscaling.knative.dev/maxScale'] = '3' + } + if (defect === 'traffic') { + state.serving = state.latest + } + if (defect.startsWith('deploy-')) { + state.failure = defect + } + }) + // Corrupt the recovery response after the modeled deploy while keeping real jq assertions. + const extra = ['mode', 'image'].includes(defect) + ? ` + gcloud() { + node "$MODEL_SCRIPT" "$@" > "$RUNNER_TEMP/out" || return $? + if [[ "$*" == 'run services describe '* && "$*" == *'--format=json'* ]]; then + jq '${defect === 'mode' ? '.spec.template.spec.containers[0].env += [{name:"ORCA_PUSH_MODE",value:"validation"}]' : '.spec.template.spec.containers[0].image = "wrong"'}' "$RUNNER_TEMP/out" + else cat "$RUNNER_TEMP/out"; fi + }` + : '' + h.run('restore', false, extra) + const recorded = readFileSync(join(h.dir, 'env'), 'utf8') + assert.match(recorded, /TEMPLATE_RECOVERY_REVISION=push-test-r123-1/) + assert.doesNotMatch(recorded, /TEMPLATE_RESTORED=true/) + assert.ok(h.state().peak <= 3) + }) + ) +} + +test('failed validation retirement blocks a fourth revision during recovery', options, () => + exercise((h) => { + h.run('preflight') + h.run('candidate') + h.change((state) => { + state.failure = 'delete' + }) + h.run('activate', false) + h.run('restore', false) + assert.equal(h.state().peak, 3) + assert.equal(h.state().revisions['push-test-r123-1'], undefined) + }) +) + +test( + 'failed retirement after public checks leaves verified serving and blocks the next run', + options, + () => + exercise((h) => { + for (const key of ['preflight', 'candidate', 'activate', 'shift', 'public']) { + h.run(key) + } + h.change((state) => { + state.failure = 'delete' + }) + h.run('retire', false) + assert.equal(h.state().serving, 'push-test-a123-1') + h.run('preflight', false) + assert.match(workflow, /env.ROLLOUT_VERIFIED != 'true'/) + }) +) + +test( + 'recovery promotion failure keeps consumers for operator diagnosis and blocks new rollout', + options, + () => + exercise((h) => { + h.run('preflight') + h.run('candidate') + h.run('restore') + h.change((state) => { + state.failure = 'public' + }) + h.run('promoteRecovery', false) + assert.doesNotMatch(readFileSync(join(h.dir, 'env'), 'utf8'), /RECOVERY_VERIFIED=true/) + h.run('preflight', false) + }) +) + +const capability = step('Require image support for inert validation') +for (const [label, source, ok] of [ + ['old image', 'export function loadPushConfig() { return {}; }', false], + [ + 'invalid mode accepted', + 'export function loadPushConfig(env) { return { mode: env.ORCA_PUSH_MODE }; }', + false + ], + [ + 'validation supported', + `export function loadPushConfig(env) { + if (env.ORCA_PUSH_MODE !== 'validation') throw new Error('invalid mode'); + return { mode: 'validation' }; + }`, + true + ] +]) { + test(`pre-production image smoke: ${label}`, options, () => + exercise((h) => { + const dist = join(h.dir, 'apps', 'push', 'dist') + mkdirSync(dist, { recursive: true }) + writeFileSync(join(h.dir, 'package.json'), '{"type":"module"}') + writeFileSync(join(dist, 'config.js'), source) + h.run(capability, ok, 'docker() { node "${@: -3}"; }') + }) + ) +} diff --git a/cloud/dev/scripts/push-gateway-workflow.test.mjs b/cloud/dev/scripts/push-gateway-workflow.test.mjs new file mode 100644 index 00000000000..3ee4fabe410 --- /dev/null +++ b/cloud/dev/scripts/push-gateway-workflow.test.mjs @@ -0,0 +1,334 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { + concurrencyBlocks, + jobIf, + jobs, + LEASE_ACTION, + leaseSteps +} from './cloud-sql-rollout-lock-census.mjs' +import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs' + +// Why: the push gateway holds the APNs key and is the only thing standing between a paired +// phone and a silent notification pipeline. Its deploy is a blue/green rollout against the +// dedicated Cloud SQL instance, and each of the guarantees below is one careless edit from gone. +const WORKFLOW = 'push-deploy.yml' +const workflow = readRelayWorkflow(WORKFLOW) +const deploy = () => { + const job = jobs(workflow).find((entry) => entry.id === 'deploy') + assert.ok(job, 'the workflow no longer declares a deploy job') + return job +} + +function terraform(file) { + return readFileSync(new URL(`../../infra/terraform/${file}`, import.meta.url), 'utf8') +} + +// The ordered step names; every assertion below reads positions out of this list rather than +// restating them, so a reordering that breaks the no-traffic guarantee fails here. +const stepNames = () => [...workflow.matchAll(/^ {6}- name: (.+)$/gm)].map((match) => match[1]) + +const indexOfStep = (name) => { + const index = stepNames().indexOf(name) + assert.notEqual(index, -1, `the workflow no longer has a "${name}" step`) + return index +} + +test('the whole surface stays inert until the owner enables cloud operations', () => { + const guard = jobIf(deploy().text) + assert.ok(guard.includes("vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true'"), guard) + assert.ok(guard.includes("github.ref == 'refs/heads/main'"), guard) + assert.equal(jobs(workflow).length, 1, 'a second job would need its own gate') +}) + +test('it authenticates through Workload Identity and holds no repository secret', () => { + assert.match(workflow, /uses: google-github-actions\/auth@v2/) + assert.match(workflow, /workload_identity_provider: \$\{\{ vars\.PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER \}\}/) + assert.match(workflow, /service_account: \$\{\{ vars\.PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT \}\}/) + assert.match(workflow, /environment: production/) + for (const [, name] of workflow.matchAll(/secrets\.([A-Za-z_][A-Za-z0-9_]*)/g)) { + assert.equal(name, 'GITHUB_TOKEN', `the workflow reads secrets.${name}`) + } +}) + +// Why: Terraform trusts exact workflow filenames, not a prefix. A rename here without the +// matching tfvars-independent list entry would fail authentication at dispatch time only. +test('Terraform trusts this exact workflow file on the production deploy provider', () => { + assert.match(terraform('push-deploy-identity.tf'), /push-deploy\.yml@refs\/heads\/main/) + assert.doesNotMatch(terraform('relay-github-actions.tf'), /push-deploy\.yml/) + assert.equal(relayWorkflowFile(WORKFLOW), 'cloud-push-deploy.yml') +}) + +test('the rollout is serialized and leases its dedicated push rollout lock', () => { + const blocks = concurrencyBlocks(workflow) + assert.equal(blocks.length, 1) + assert.equal(blocks[0].group, 'production-push-rollout') + assert.equal(blocks[0].cancelInProgress, 'false') + const steps = leaseSteps(workflow) + assert.equal(steps.length, 1, 'exactly one lease step, held for the whole run') + assert.equal(steps[0].bucket, 'onorca-cloud-terraform-state') + assert.equal(steps[0].object, 'terraform/state/push-rollout/production.lock') + assert.equal(steps[0].release, undefined, 'release stays at its default for a single-job run') +}) + +// Why: the ops guardrail is that a piped command only fails the step when pipefail is set, and +// pipefail only applies under an explicit bash shell. Every multi-line body here opts in. +test('every multi-line command runs under bash with pipefail', () => { + const bodies = [...workflow.matchAll(/^ {8}(shell: bash\n {8})?run: \|\n((?: {10}.*\n|\n)+)/gm)] + assert.ok(bodies.length >= 8, `only ${bodies.length} multi-line commands were found`) + for (const match of bodies) { + assert.ok(match[1], `a multi-line command does not declare shell: bash:\n${match[2].slice(0, 120)}`) + assert.match(match[2], /^ {10}set -euo pipefail$/m) + } +}) + +test('the candidate revision takes no traffic and is addressed by its own tag', () => { + assert.match(workflow, /gcloud run deploy "\$\{SERVICE_NAME\}"/) + assert.match(workflow, /^ {12}--no-traffic \\$/m) + assert.match(workflow, /--tag "\$\{tag\}"/) + assert.match(workflow, /test "\$\{CANDIDATE_REVISION\}" != "\$\{ROLLBACK_REVISION\}"/) + assert.ok( + indexOfStep('Record the serving revision and require its Terraform-owned scaling') < + indexOfStep('Deploy the candidate revision with no traffic'), + 'the rollback target must be captured before the candidate exists' + ) +}) + +// Why: scaling is a Terraform-owned field that `lifecycle.ignore_changes` does not cover, so a +// deploy that passed --max-instances would revert a later push_max_instances raise on every run. +// The workflow asserts the shape instead of writing it, on the serving revision before the +// candidate exists and on the candidate that inherits it. +test('the deploy asserts the Terraform-owned scaling instead of mutating it', () => { + assert.doesNotMatch(workflow, /--max-instances/, 'the deploy must not write a scaling field') + assert.doesNotMatch(workflow, /--min-instances "/, 'the deploy must not write a scaling field') + // The floor is the variables.tf default; production.tfvars overrides only the ceiling, down to + // the two instances the Cloud SQL connection budget leaves room for. + assert.match(workflow, /PUSH_MIN_INSTANCES: 1$/m) + assert.match(workflow, /PUSH_MAX_INSTANCES: 2$/m) + assert.match(terraform('variables.tf'), /variable "push_min_instances"[\s\S]*?default {5}= 1/) + assert.match(terraform('environments/production.tfvars'), /^push_max_instances {9}= 2$/m) + const gate = indexOfStep('Record the serving revision and require its Terraform-owned scaling') + assert.ok(gate < indexOfStep('Deploy the candidate revision with no traffic')) + assert.match(workflow, /autoscaling\.knative\.dev\/minScale/) + assert.match(workflow, /\[\[ "\$\{floor:-0\}" -lt "\$\{PUSH_MIN_INSTANCES\}" \]\]/) + assert.match(workflow, /test "\$\{ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/) + assert.match(workflow, /test "\$\{candidate_ceiling\}" = "\$\{PUSH_MAX_INSTANCES\}"/) +}) + +// Why: the image build is not a Cloud SQL operation, and the lease is a global serialization +// point. A build inside it blocks every relay deploy and rehome for its duration. +test('the image is built before the rollout lease is taken', () => { + const lease = workflow.indexOf(`- uses: ${LEASE_ACTION}`) + assert.notEqual(lease, -1) + const build = workflow.indexOf('- name: Build and publish the immutable gateway image') + const deployCandidate = workflow.indexOf('- name: Deploy the candidate revision with no traffic') + assert.ok(build < lease, 'the build must finish before the run takes the lease') + assert.ok(lease < deployCandidate, 'the lease must still cover the deploy, probe, and shift') +}) + +// Why: the gateway's Cloud SQL draw is instances x pool, and the root that takes the rollout +// lease can only account for a pool it declares. Leaving it at the application default hid it. +test('the database pool size is Terraform-owned and bounded at plan time', () => { + const source = terraform('push-gateway.tf') + assert.match(source, /name {2}= "ORCA_PUSH_DATABASE_POOL_MAX"/) + assert.match(source, /value = tostring\(var\.push_database_pool_max\)/) + assert.match(terraform('variables.tf'), /variable "push_database_pool_max"[\s\S]*?default {5}= 2/) + const block = /resource "google_cloud_run_v2_service" "push"[\s\S]*?\n lifecycle \{([\s\S]*?)\n \}/.exec(source) + assert.ok(block, 'the push service no longer declares a lifecycle block') + assert.match( + block[1], + /var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/, + 'instances x pool must be bounded at plan time' + ) + assert.match( + readFileSync(new URL('../../apps/push/src/config.ts', import.meta.url), 'utf8'), + /ORCA_PUSH_DATABASE_POOL_MAX/, + 'the gateway must read the variable Terraform sets' + ) +}) + +test('the candidate is probed on its own URL before any traffic moves', () => { + const probe = indexOfStep('Probe the candidate readiness endpoint') + assert.ok(probe > indexOfStep('Deploy the candidate revision with no traffic')) + assert.ok(probe < indexOfStep('Shift all traffic to the verified candidate')) + assert.match(workflow, /"\$\{CANDIDATE_URL\}\/ready"/) + assert.match(workflow, /test "\$\{code\}" = 200/) + assert.ok(workflow.indexOf('${CANDIDATE_URL}/ready') < workflow.indexOf('${CANDIDATE_URL}/health')) + assert.match(workflow, /\.deliveryProtocol == 2/, 'verify the durable gateway after readiness') +}) + +// Why: a gateway that answers /ready can still hold no usable FCM credential. The probe must be +// validate-only, must use a token that cannot exist, and must treat a denied credential as the +// failure. Accepting PERMISSION_DENIED would make the whole step decorative. +test('the FCM probe is validate-only and separates a bad token from a bad credential', () => { + const fcm = indexOfStep('Prove the runtime identity can reach FCM') + assert.ok(fcm > indexOfStep('Probe the candidate readiness endpoint')) + assert.ok(fcm < indexOfStep('Shift all traffic to the verified candidate')) + assert.match(workflow, /"validate_only":true/) + assert.match(workflow, /https:\/\/fcm\.googleapis\.com\/v1\/projects\/\$\{GCP_PROJECT_ID\}\/messages:send/) + assert.match(workflow, /GCP_PROJECT_ID: onorca-cloud$/m) + assert.match(workflow, /orca-push-deploy-probe-invalid-token/) + assert.match(workflow, /test "\$\{status\}" = INVALID_ARGUMENT/) + assert.match(workflow, /test "\$\{status\}" = PERMISSION_DENIED/) + // Only those four answers are conclusive; a 429 or a 5xx says nothing about the credential, so + // it is retried rather than read as either verdict. A denied credential still fails at once. + assert.match(workflow, /for attempt in \$\(seq 1 5\); do/) + const probe = workflow.slice( + workflow.indexOf('- name: Prove the runtime identity can reach FCM'), + workflow.indexOf('- name: Shift all traffic to the verified candidate') + ) + assert.match(probe, /for attempt in \$\(seq 1 5\); do/) + assert.match(probe, /test "\$\{code\}" = 401 \|\| test "\$\{code\}" = 403; then\n {14}break/) + assert.match( + workflow, + /--impersonate-service-account "\$\{PUSH_RUNTIME_SERVICE_ACCOUNT\}"/, + 'the probe must exercise the runtime credential, not the deploy identity' + ) + // Why: that token reads the Apple signing key. Masking it means a later `set -x` or a + // debug re-run cannot print it into a public log. + assert.match( + probe, + /test -n "\$\{token\}"\n {10}echo "::add-mask::\$\{token\}"/, + 'the impersonated token must be masked before anything else runs' + ) + assert.match(workflow, /PUSH_RUNTIME_SERVICE_ACCOUNT: orca-cloud-push@onorca-cloud\.iam\.gserviceaccount\.com/) +}) + +// Why: a deploy ends with traffic pinned to an exact revision, and a rollback pins it to the +// previous one. Terraform reverting the service to 100% LATEST would undo either silently. +test('Terraform does not own the image or the traffic split', () => { + const source = terraform('push-gateway.tf') + const block = /resource "google_cloud_run_v2_service" "push"[\s\S]*?\n lifecycle \{([\s\S]*?)\n \}/.exec(source) + assert.ok(block, 'the push service no longer declares a lifecycle block') + assert.match(block[1], /template\[0\]\.containers\[0\]\.image/) + assert.match(block[1], /^\s*traffic$/m) +}) + +test('impersonating the runtime identity is a Terraform-declared grant', () => { + const source = terraform('push-gateway.tf') + assert.match(source, /resource "google_service_account_iam_member" "github_production_push_runtime_token_creator"/) + assert.match(source, /role\s+= "roles\/iam\.serviceAccountTokenCreator"/) + assert.match(source, /resource "google_cloud_run_v2_service_iam_member" "github_production_push_developer"/) +}) + +test('the traffic shift is all-or-nothing and is verified after the fact', () => { + const shift = indexOfStep('Shift all traffic to the verified candidate') + assert.match(workflow, /gcloud run services update-traffic "\$\{SERVICE_NAME\}"/) + assert.match(workflow, /--to-revisions "\$\{CANDIDATE_REVISION\}=100"/) + assert.match(workflow, /test "\$\{serving\}" = "\$\{CANDIDATE_REVISION\}"/) + assert.ok(shift < indexOfStep('Verify the public origin after the shift')) + assert.match(workflow, /PUSH_ORIGIN: https:\/\/push\.onorca\.dev/) + assert.match(workflow, /"\$\{PUSH_ORIGIN\}\/ready"/) +}) + +// Why: the origin can lag the traffic move by seconds, and a single unlucky curl would otherwise +// roll a healthy deploy back. It retries on the same schedule as the candidate probe. +test('the post-shift origin check retries like the candidate probe', () => { + const check = workflow.slice( + workflow.indexOf('- name: Verify the public origin after the shift'), + workflow.indexOf('- name: Roll traffic back to the previous revision') + ) + assert.match(check, /for attempt in \$\(seq 1 30\); do/) + assert.match(check, /sleep 5/) + assert.match(check, /test "\$\{code\}" = 200/) +}) + +// Why: the summary carries the rollback target. Writing it after the origin check meant the one +// run that needed it, the run whose check failed, was the one run that never got it. +test('the summary is written before anything that can fail after the shift', () => { + const summary = indexOfStep('Publish the rollout summary') + assert.ok(summary > indexOfStep('Shift all traffic to the verified candidate')) + assert.ok(summary < indexOfStep('Verify the public origin after the shift')) + assert.match(workflow, /Known-good image:/) + assert.match(workflow, /GITHUB_STEP_SUMMARY/) +}) + +// Why: everything after the shift runs with production on the candidate, so a failure there is a +// live gateway that has to go back. The marker is what separates that case from a failure before +// the shift, where production never moved and the candidate is the thing to clean up. +test('a failure after the shift rolls production back automatically', () => { + const rollback = indexOfStep('Roll traffic back to the previous revision') + assert.ok(rollback > indexOfStep('Verify the public origin after the shift')) + assert.match(workflow, /echo "TRAFFIC_SHIFTED=true" >> "\$\{GITHUB_ENV\}"/) + const shift = workflow.indexOf('- name: Shift all traffic to the verified candidate') + assert.ok( + workflow.indexOf('echo "TRAFFIC_SHIFTED=true"') > shift, + 'the success marker follows the shift step' + ) + const body = workflow.slice( + workflow.indexOf('- name: Roll traffic back to the previous revision'), + workflow.indexOf('- name: Delete the rejected candidate revision') + ) + assert.match( + body, + /if: \$\{\{ \(failure\(\) \|\| cancelled\(\)\) && env\.TRAFFIC_SHIFT_ATTEMPTED == 'true' && env\.ROLLOUT_VERIFIED != 'true' \}\}/, + 'the rollback must be conditioned on both failure and the shift marker' + ) + assert.match(body, /test -n "\$\{ROLLBACK_REVISION:-\}"/) + assert.match(body, /--to-revisions "\$\{ROLLBACK_REVISION\}=100"/) + assert.match(body, /test "\$\{serving\}" = "\$\{ROLLBACK_REVISION\}"/) + assert.match(body, /GITHUB_STEP_SUMMARY/, 'the rollback must be reported in the summary') +}) + +// Why: a candidate that never took traffic still holds a warm instance and a Cloud SQL pool. Its +// tag comes off first, because Cloud Run refuses to delete a revision a traffic target names. +test('verified recovery authorizes rejected candidate deletion', () => { + const body = workflow.slice( + workflow.indexOf('- name: Delete the rejected candidate revision'), + workflow.indexOf('- name: Drop the candidate traffic tag') + ) + assert.match( + body, + /env\.RECOVERY_VERIFIED == 'true'/, + 'cleanup must wait for verified recovery traffic and public checks' + ) + assert.match(body, /if test -z "\$\{CANDIDATE_REVISION:-\}"; then/) + assert.ok( + body.indexOf('--remove-tags') < body.indexOf('gcloud run revisions delete'), + 'the tag must come off before the revision is deleted' + ) + assert.match(body, /echo "CANDIDATE_TAG=" >> "\$\{GITHUB_ENV\}"/) +}) + +test('the run always drops its traffic tag', () => { + const cleanup = indexOfStep('Drop the candidate traffic tag') + assert.equal(cleanup, stepNames().length - 1, 'tag cleanup must be the last step') + assert.match(workflow, /--remove-tags "\$\{CANDIDATE_TAG\}"/) + const body = workflow.slice(workflow.indexOf('- name: Drop the candidate traffic tag')) + assert.match(body, /if: always\(\)/) + assert.match(body, /test -n "\$\{CANDIDATE_TAG:-\}" \|\| exit 0/) +}) + +test('push credentials cannot assume the shared Relay deploy identity', () => { + const source = terraform('push-deploy-identity.tf') + assert.match(source, /"attribute.push_deploy"\s*=\s*"'production'"/) + assert.doesNotMatch(source, /"attribute.repository"\s*=/) + assert.match(source, /attribute\.push_deploy\/production/) + assert.doesNotMatch(workflow, /PRODUCTION_GCP_RELAY_DEPLOY_/) + assert.doesNotMatch(terraform('push-gateway.tf'), /member\s*=\s*local\.relay_github_deploy_service_account_member/) +}) + +// A latest revision needs a successor even when validation is inert. +test('dedicated database admits three simultaneous revision pools', () => { + assert.match(terraform('push-gateway.tf'), /var\.push_max_instances \* var\.push_database_pool_max \* 3 <= 64/) +}) + +test('push has only a dedicated database attachment and a narrowly scoped deployment lease', () => { + const service = terraform('push-gateway.tf') + const database = terraform('push-dedicated-database.tf') + assert.match(service, /instances = \[google_sql_database_instance\.push_dedicated\[0\]\.connection_name\]/) + assert.match(service, /secret\s*= google_secret_manager_secret\.push_dedicated_database_url\[0\]\.secret_id/) + assert.match(service, /version = google_secret_manager_secret_version\.push_dedicated_database_url\[0\]\.version/) + assert.doesNotMatch(service + database, /push_dedicated_database_(?:active|enabled)|local\.relay_database_connection_name|resource "google_sql_database" "push"/) + assert.match(database, /tier\s*= "db-custom-2-7680"/) + assert.match(database, /availability_type = "REGIONAL"/) + assert.match(database, /deletion_protection\s*= true/) + assert.match(database, /deletion_protection_enabled = true/) + const identity = terraform('push-deploy-identity.tf') + const lease = identity.match(/resource "google_storage_bucket_iam_member" "github_push_rollout_lease" \{([\s\S]*?)\n\}/)?.[1] + assert.ok(lease) + assert.match(lease, /member = local\.push_deploy_member/) + assert.match(lease, /role\s*= "roles\/storage.objectAdmin"/) + assert.match(lease, /resource.name == 'projects\/_\/buckets\/\$\{var.project_id\}-terraform-state\/objects\/terraform\/state\/push-rollout\/production.lock'/) +}) diff --git a/cloud/dev/scripts/push-validation-workflow.test.mjs b/cloud/dev/scripts/push-validation-workflow.test.mjs new file mode 100644 index 00000000000..e26769cfeb3 --- /dev/null +++ b/cloud/dev/scripts/push-validation-workflow.test.mjs @@ -0,0 +1,48 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { readRelayWorkflow } from './relay-repository.mjs' + +const workflow = readRelayWorkflow('push-deploy.yml') +const position = (name) => { + const index = workflow.indexOf(`- name: ${name}`) + assert.notEqual(index, -1) + return index +} +const capability = position('Require image support for inert validation') +const deploy = position('Deploy the candidate revision with no traffic') +const activation = position('Retire inert validation and activate the verified image') +const shift = position('Shift all traffic to the verified candidate') + +test('the exact build digest must support validation before production boot', () => { + assert.match(workflow, /docker buildx build --push --platform linux\/amd64 --provenance=false --metadata-file/) + assert.match(workflow, /containerimage\.digest/) + assert.doesNotMatch(workflow, /gcloud artifacts docker images describe/) + assert.ok(capability < deploy) + const preflight = workflow.slice(capability, deploy) + assert.match(preflight, /docker run --rm --network none --entrypoint node "\$\{IMAGE\}"/) + assert.match(preflight, /loadPushConfig\(env\)\.mode !== "validation"/) + assert.match(preflight, /validation_mode_not_fail_closed/) +}) + +test('inert validation and credential checks precede deliberate activation of the same digest', () => { + assert.match(workflow.slice(deploy, activation), /--update-env-vars ORCA_PUSH_MODE=validation/) + assert.match(workflow.slice(deploy, activation), /\.mode == "validation"/) + assert.ok(position('Prove the runtime identity can reach FCM') < activation) + const active = workflow.slice(activation, shift) + assert.ok(active.indexOf('gcloud run deploy') < active.indexOf('gcloud run revisions delete')) + assert.match(active, /--image "\$\{IMAGE\}"/) + assert.match(active, /--remove-env-vars ORCA_PUSH_MODE/) + assert.match(active, /\.spec\.containers\[0\]\.image == \$image/) + assert.match(active, /\.spec\.serviceAccountName == \$account/) + assert.match(active, /\.mode == "active"/) + assert.ok(active.indexOf('ACTIVATION_ATTEMPTED=true') < active.indexOf('gcloud run deploy')) + assert.match(workflow, /deletion below must stop its workers/) +}) + +test('production startup connects read-only and gates all background work in validation', () => { + const entry = readFileSync(new URL('../../apps/push/src/index.ts', import.meta.url), 'utf8') + assert.match(entry, /readOnly: config\.mode === 'validation'/) + assert.match(entry, /startPushBackground\(config,/) + assert.doesNotMatch(entry, /worker\.start\(/) +}) diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs index a26d24c274d..89d40954f7d 100644 --- a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs @@ -6,7 +6,7 @@ import { readRelayCloudSqlConnectionBudget } from './relay-cloud-sql-connection-budget.mjs' -test('production plus three Asia pools preserves allowance and reserve below the ceiling', () => { +test('production shared consumers keep allowance and reserve below the ceiling', () => { const report = readRelayCloudSqlConnectionBudget() assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 }) @@ -63,7 +63,11 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => { } } `, - terraformVariables: 'variable "relay_director_database_pool_max" { default = 3 }', + terraformVariables: [ + 'variable "relay_director_database_pool_max" { default = 3 }', + 'variable "push_max_instances" { default = 1 }', + 'variable "push_database_pool_max" { default = 2 }' + ].join('\n'), relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10' }, maxConnections: 100, @@ -76,6 +80,37 @@ test('excludes fenced cell pools and reads per-cell pool overrides', () => { assert.equal(report.budgetedTotal, 47) }) +test('dedicated push scaling does not consume shared capacity', () => { + const report = readRelayCloudSqlConnectionBudget({ + proposedAsiaCellCount: 1, + appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 }, + sources: { + productionTfvars: ` + relay_max_instances = 1 + push_max_instances = 3 + relay_gce_fenced_cells = [] + relay_gce_cells = { + "production-gce-c2" = { database_pool_max = 4 + } + } + `, + terraformVariables: [ + 'variable "relay_director_database_pool_max" { default = 3 }', + 'variable "push_max_instances" { default = 1 }', + 'variable "push_database_pool_max" { default = 2 }' + ].join('\n'), + relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10' + }, + maxConnections: 100, + maintenanceAdminAllowance: 1, + explicitReserve: 1 + }) + + assert.equal(report.consumers.push, undefined) + assert.equal(report.rolloutOverlap.pushCandidate, undefined) + assert.equal(report.operatingMaximum, 46) +}) + test('requires strict headroom below the physical ceiling', () => { const report = calculateRelayCloudSqlConnectionBudget({ cellPoolTotal: 20, diff --git a/cloud/dev/scripts/render-workload-identity-conditions.mjs b/cloud/dev/scripts/render-workload-identity-conditions.mjs index 9a7c2e5cc09..815ba1b2c17 100644 --- a/cloud/dev/scripts/render-workload-identity-conditions.mjs +++ b/cloud/dev/scripts/render-workload-identity-conditions.mjs @@ -18,6 +18,7 @@ const TERRAFORM_ROOTS = { 'infra/terraform/relay-shared.tf', 'infra/terraform/relay-github-workflow-trust.tf', 'infra/terraform/relay-github-actions.tf', + 'infra/terraform/push-deploy-identity.tf', 'infra/terraform/relay-staging-deploy-iam.tf', 'infra/terraform/relay-asia-topology-iam.tf', 'infra/terraform/relay-asia-proof-iam.tf' @@ -475,7 +476,7 @@ function collectTfvars(source, variables) { offset += line.length + 1 continue } - const structured = /^([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(?=[[{])/.exec(line) + const structured = /^([A-Za-z_][A-Za-z0-9_]*)\s*=\s*/.exec(line) if (structured) { try { variables[structured[1]] = parseValueAt(source, offset + structured[0].length) diff --git a/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs b/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs index 1d3f3ce4d79..64791605549 100644 --- a/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs +++ b/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs @@ -30,6 +30,9 @@ const EXPECTED_CONDITIONS = { }, production: { relay: { + github_push: + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && assertion.event_name == 'workflow_dispatch' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-push-deploy.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-push-deploy.yml@refs/heads/main'", + github: "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-fence-broker.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-director.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-publish-relay-production.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && (assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main' || assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main')))", github_monitor: diff --git a/cloud/docs/push-database-cutover.md b/cloud/docs/push-database-cutover.md new file mode 100644 index 00000000000..e2b0be3970d --- /dev/null +++ b/cloud/docs/push-database-cutover.md @@ -0,0 +1,112 @@ +# Dedicated push database operations + +Push attaches only to its dedicated PostgreSQL 17 instance: regional HA, 2 vCPU, +7.5 GiB RAM, 50 GiB SSD with automatic growth, seven retained backups and seven-day +point-in-time recovery. Cloud SQL and Terraform deletion protections remain enabled. +The Cloud SQL connector uses the dedicated URL secret pinned to its managed version. +There is no shared-storage fallback or provision/activate switch. + +## Existing-resource cleanup: operator prerequisite + +This is a plan/runbook, not authorization to apply or delete resources. Preserve the +shared Orca instance, dedicated push instance, all dedicated data and identities, and +unrelated resources. The dedicated resource addresses remain unchanged: + +- `google_sql_database_instance.push_dedicated[0]` +- `google_sql_database.push_dedicated[0]` +- `random_password.push_dedicated_database[0]` +- `google_sql_user.push_dedicated[0]` +- `google_secret_manager_secret.push_dedicated_database_url[0]` +- `google_secret_manager_secret_version.push_dedicated_database_url[0]` +- `google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor[0]` + +The relay state may still own these six obsolete shared-store resources, whose +configuration is removed. An untargeted plan would propose deleting them; do not apply it: + +- `google_sql_database.push[0]` +- `google_sql_user.push[0]` +- `random_password.push_database[0]` +- `google_secret_manager_secret.push_database_url[0]` +- `google_secret_manager_secret_version.push_database_url[0]` +- `google_secret_manager_secret_iam_member.push_database_url_runtime_accessor[0]` + +1. Use the production backend in `infra/terraform/README.md`. Inspect state addresses and + the live service attachment, pinned secret reference and revision resources without + printing credentials. Require the dedicated attachment and no old shared-store consumers; + source connection drain needs an authorized operator's read-only observation. +2. Have the shared database owner adopt the six legacy resources in an explicitly owned + archival configuration before retiring their relay-state ownership. Retain the former + database's `prevent_destroy` protection and secret versions; do not disable protection, + drop databases, rotate passwords or introduce a second runtime attachment. A reviewed + exact-address state transfer must preserve remote IDs and secret material in approved + Terraform storage, with no credential exports to local files or terminal output. +3. Require the owner's import/ownership plan to preserve existing resources and then an + empty plan for those addresses. Only after adoption is proven may the operator remove + precisely the six former addresses from relay state under backend locking. Do not + automate this via `removed` blocks, broad `state rm`, force, or an untargeted apply. +4. Review a fresh relay plan. Reject every delete or replace affecting either SQL instance, + dedicated databases/users/secrets, or unrelated resources. Target only the intended push + service and lease IAM grant for rollout; review their dependency closure too. Existing + unrelated drift must be handled by its owner, outside this cleanup. + +No data transfer, dedicated database reset, or phone re-registration is part of this cleanup. + +## Schema prerequisite for existing internal test databases + +New schemas omit `push_hosts` and the unused `host_public_key` and `transcript` columns +on `push_challenges`. Authentication still verifies the encrypted transcript and consumes +its challenge digest once; sessions and device ownership are unchanged. No compatibility +migration for unpublished builds runs at application startup. + +Before deploying onto an older internal schema, an operator must arrange a separately +reviewed schema-preparation job through the approved database execution path. Its entire +scope is dropping `push_hosts` (including its index) and those two unused challenge columns; +preserve challenge digest/expiry/consumption fields and every session, device and delivery +table. Verify that the old NOT NULL columns are absent before admitting the new image. +Do not hand-edit production SQL or reset the dedicated database to satisfy this prerequisite. +Until that job is reviewed and executed, the new image is not ready for an existing schema. + +## Deployment serialization transition + +Finish all old push workflow runs before changing the workflow's lock namespace. An old +shared-lock push run and a new push-lock run do not exclude each other. Hold off new push +dispatches while preparing the following exact changes: + +1. Review the relay-root plan for + `google_storage_bucket_iam_member.github_push_rollout_lease[0]`. It grants only + `roles/storage.objectAdmin` on + `projects/_/buckets/onorca-cloud-terraform-state/objects/terraform/state/push-rollout/production.lock` + to the dedicated push deploy account. The lease action uses object GET/upload/delete, + so no bucket-wide listing or Terraform-state access is needed. +2. After approval, apply only the reviewed IAM/dependency plan. Verify the exact condition + and principal independently. If foundation still grants push membership in the old + `cloud_sql_rollout_lease_members`, its owner removes only that push member; keep Relay's + existing members and permissions. Do not mutate foundation through the relay root. +3. Publish the reviewed workflow on main with `production-push-rollout`, cancellation + disabled, and the existing lease action pointed at the dedicated object. The durable + lease covers admission, candidate validation, activation, traffic changes and recovery. + A stale/conflicting lease stops the run; it is never stolen or force-deleted. +4. Deploy the reviewed image through `cloud-push-deploy.yml`. Preserve candidate readiness, + runtime-provider validation, exact digest/configuration checks, and explicit activation. + Verify the public origin and real notification delivery/dismissal afterward. + +## Recovery and capacity + +Activation starts schema writes and workers before HTTP promotion. Traffic rollback cannot +undo queue or schema changes. Cloud Run cannot delete its latest revision, so failed +activation creates a known-good successor, verifies it, promotes it, then retires rejected +and previous revisions. When partial activation leaves three resources, retire non-latest +inert validation before creating recovery. Failed retirement stops automation. Admission +requires one serving revision resource; retire historical leftovers under the push lease. +Keep the dedicated attachment for application rollback and retain an immutable compatible +image. An image requiring the removed challenge columns needs separate schema review. + +The two-instance ceiling and two-connection pool draw four configured connections, twelve +across three simultaneous revision resources. Terraform caps instances × pool × 3 at 64 +for serving, validation/rejected and active/recovery pools. Push does not draw from Relay's +shared connection budget. Source connections must have drained before treating that old +allocation as free. Increase capacity only after measuring deployed contention. + +Cloud SQL resizing can interrupt connections despite HA. Durable accepted events remain in +SQL; workers retry within each event's original five-minute deadline. Schedule resizes and +verify reconnection, queue recovery, readiness and real delivery afterward. diff --git a/cloud/docs/push-gateway.md b/cloud/docs/push-gateway.md new file mode 100644 index 00000000000..9183f693a87 --- /dev/null +++ b/cloud/docs/push-gateway.md @@ -0,0 +1,392 @@ +# Orca mobile push gateway + +`orca-cloud-push` is a public Cloud Run service in `onorca-cloud` that turns a desktop +notification into an APNs or FCM push for a paired phone. The desktop registers each phone's +native token with it and calls `POST /v1/send` after the socket fan-out it already does; the +phone treats APNs/FCM as the sole ordinary OS-banner path. The notification socket is retained only +for live dismissal and reconnect tray reconciliation; it does not create or recover banners. Desktop +notification categories remain authoritative. The service is the only place the Apple +`.p8` signing key is readable, which is the reason it exists as a service at all. + +The request schemas live in `packages/push-contract/src/`. This document covers Terraform +ownership, deployment, credential rotation, and recovery. + +**There is no staging push gateway.** That is a decision, not an omission. `push_gateway_enabled` +is false in `environments/staging.tfvars` and true in `environments/production.tfvars`, and every +resource in `infra/terraform/push-gateway.tf` is behind it. A staging gateway would be a tfvars +edit plus a second set of Apple credentials. + +## Shape + +| Setting | Value | Where | +| ----------------- | ------------------------------------------------------ | -------------------------------------------- | +| Cloud Run service | `orca-cloud-push` | `push_cloud_run_service_name` | +| Region | `us-central1` | `region` | +| Instances | min 1, max 2 | `push_min_instances`, `push_max_instances` | +| Database pool | 2 per instance | `push_database_pool_max` | +| Concurrency | 80 | `push_concurrency` | +| Ingress | all | `INGRESS_TRAFFIC_ALL` | +| Invoker | IAM disabled | `invoker_iam_disabled = true` on the service | +| Runtime identity | `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com` | `google_service_account.push_runtime` | +| Database | `orca_push` on dedicated HA PostgreSQL 17 | `google_sql_database.push_dedicated` | +| Hostname | `push.onorca.dev` | `push_base_url` | + +The minimum of one instance is deliberate and did not move when the ceiling came down to two. A +cold start delays a notification past the point where it is worth showing, so the floor is what +keeps a notification prompt. The +ceiling is a different question, answered below. + +Push uses its approved dedicated two-vCPU HA database. Two instances with a two-connection +pool draw four connections; three simultaneous revision resources draw twelve. Tagged +candidates can run outside the service-wide cap, so Terraform bounds instances × pool × 3 +at 64 connections, leaving dedicated capacity for maintenance and operators. Increase pool +sizes only after measuring contention. The shared Relay budget excludes push entirely. + +Authentication is the host proof in `POST /v1/host/challenge`, not Cloud Run IAM, so the service +opts out of invoker IAM with `invoker_iam_disabled = true`, exactly as the relay director does. +The project's domain-restricted-sharing policy refuses an `allUsers` invoker binding, so that is +the only way to reach an open service here. + +## Environment + +Set on the container by Terraform: + +| Variable | Source | +| ----------------------------- | -------------------------------------------------------- | +| `PORT` | Cloud Run, container port 8080 | +| `ORCA_PUSH_PUBLIC_URL` | `push_base_url` | +| `ORCA_PUSH_FCM_PROJECT_ID` | `project_id` (required for standalone runtime) | +| `ORCA_PUSH_DATABASE_URL` | Secret `orca-cloud-push-dedicated-database-url`, pinned version | +| `ORCA_PUSH_DATABASE_POOL_MAX` | `push_database_pool_max`, 2 per instance | +| `ORCA_PUSH_APNS_KEY` | Secret `orca-cloud-push-apns-key`, version `latest` | +| `ORCA_PUSH_APNS_KEY_ID` | Secret `orca-cloud-push-apns-key-id`, version `latest` | +| `ORCA_PUSH_APPLE_TEAM_ID` | Secret `orca-cloud-push-apple-team-id`, version `latest` | + +`ORCA_PUSH_APNS_TOPIC` is left to its application default (`com.stably.orca.mobile`). Add it here +only when it has to differ from the code default, so that a code-side change stays visible rather +than silently overridden. + +Terraform owns the three Apple secret **names, labels, and replication, and never a version.** +The `.p8` is issued by the Apple developer portal, so a Terraform-managed version would put the +private key in state and would fight the rotation below. The database URL secret is different: +Terraform generates that password, so it owns that version, exactly as `relay-database.tf` does. +That puts the generated password and the full database URL in the state bucket, which the shared +deploy identity can read; the Apple key never appears there. The three Apple secrets and the +`orca_push` database carry `prevent_destroy`, so disabling the gateway fails the plan instead +of deleting the only copy of the signing key or every live device token. + +## Importing what already exists + +The runtime account, the three Apple secrets, and their accessor bindings were created out of +band alongside the Apple credentials. They are declared so a plan is clean, and imported once. +Run these from `cloud/` after `pnpm infra:init --env production`, review the resulting plan, and +expect the imported resources to show no changes. + +```sh +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_service_account.push_runtime[0]' \ + projects/onorca-cloud/serviceAccounts/orca-cloud-push@onorca-cloud.iam.gserviceaccount.com + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_project_iam_member.push_runtime_fcm_admin[0]' \ + 'onorca-cloud roles/firebasecloudmessaging.admin serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com' + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_project_iam_member.push_runtime_service_usage_consumer[0]' \ + 'onorca-cloud roles/serviceusage.serviceUsageConsumer serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com' + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_secret_manager_secret.push_provider["orca-cloud-push-apns-key"]' \ + projects/onorca-cloud/secrets/orca-cloud-push-apns-key + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_secret_manager_secret.push_provider["orca-cloud-push-apns-key-id"]' \ + projects/onorca-cloud/secrets/orca-cloud-push-apns-key-id + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_secret_manager_secret.push_provider["orca-cloud-push-apple-team-id"]' \ + projects/onorca-cloud/secrets/orca-cloud-push-apple-team-id + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_secret_manager_secret_iam_member.push_provider_runtime_accessor["orca-cloud-push-apns-key"]' \ + 'projects/onorca-cloud/secrets/orca-cloud-push-apns-key roles/secretmanager.secretAccessor serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com' + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_secret_manager_secret_iam_member.push_provider_runtime_accessor["orca-cloud-push-apns-key-id"]' \ + 'projects/onorca-cloud/secrets/orca-cloud-push-apns-key-id roles/secretmanager.secretAccessor serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com' + +terraform -chdir=infra/terraform import -var-file=environments/production.tfvars \ + 'google_secret_manager_secret_iam_member.push_provider_runtime_accessor["orca-cloud-push-apple-team-id"]' \ + 'projects/onorca-cloud/secrets/orca-cloud-push-apple-team-id roles/secretmanager.secretAccessor serviceAccount:orca-cloud-push@onorca-cloud.iam.gserviceaccount.com' +``` + +The push resources already exist in production. Preserve their addresses, dedicated database +and identities; review the [database cleanup runbook](./push-database-cutover.md) before applying +changes. This root has unrelated standing drift, so an untargeted apply is never automatic. + +Two things this root does **not** declare, because the carve assigns them elsewhere. Neither +affects whether this root's plan is clean, since an undeclared resource is invisible to it. + +- `firebase.googleapis.com` and `fcm.googleapis.com` are project service enablement, which is + `google_project_service.required` in the foundation root. They are already enabled; add them + to the foundation root's list so a foundation plan stays clean. +- The Firebase attachment on `onorca-cloud` is project-level and belongs with foundation for the + same reason. It exists already. + +## Deploying + +`Deploy Push Gateway Production` (`.github/workflows/cloud-push-deploy.yml`) is the only +supported path. Like every `cloud-*` workflow it does nothing until `ORCA_CLOUD_OPERATIONS_ENABLED` +is `true`, it runs only on `main`, and it needs the confirmation string `DEPLOY_PUSH_GATEWAY`. + +It authenticates as the dedicated `orca-cloud-gha-push` identity through +`PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER` and +`PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT`. `push-deploy-identity.tf` restricts Workload Identity +to this exact dispatch workflow on main in the production environment. Its distinct principal +attribute cannot assume the shared Relay deploy identity. + +The account can write images to Artifact Registry, deploy the push service, impersonate only +the push runtime account, and manage exactly `terraform/state/push-rollout/production.lock` +in the production state bucket. The relay root owns that conditional lease grant. It grants +no Terraform-state object access. Publish `github_push_workload_identity_provider` and +`github_push_deploy_service_account` as the production-environment variables above. + +The workflow uses the `production-push-rollout` concurrency group with cancellation disabled +and the existing durable lease action on the push-specific object. Push and Relay deploy +independently; two push deploys cannot race traffic changes. Finish every old shared-lock push +run before enabling the new workflow and lease grant. See the cleanup runbook for the bounded +IAM transition and removal of any obsolete foundation-owned push membership. + +The run builds the reviewed `source_sha` while the workflow stays on `main`. Buildx returns +its own pushed digest (no mutable-tag lookup); every subsequent check and deployment uses that +same digest. Before any production boot, a network-isolated container checks that the image +recognizes `ORCA_PUSH_MODE=validation` and rejects invalid modes. Older images that lack this +capability are refused before they can connect to production. + +Under the production push rollout lease, it records the serving rollback revision and +asserts Terraform-owned scaling. It deploys a tagged, zero-traffic validation revision: + +- Validation opens PostgreSQL with `default_transaction_read_only=on` and skips schema setup. +- No delivery worker or challenge, session, or delivery pruner starts. +- Only `/health` and `/ready` are available; all application routes return 503. +- `/health` attests `mode: validation`; `/ready` checks database connectivity only. It does not + prove schema compatibility, provider delivery, or active-worker readiness. Container probes + can still use `/health` without treating an inert process as unhealthy. + +The build explicitly targets `linux/amd64` with provenance disabled so build metadata records a +single manifest digest, rather than an OCI index that Cloud Run resolves to a different digest. +The workflow verifies the exact image and scaling, probes readiness and mode, and checks the +runtime identity with a validate-only FCM request. Cloud Run rejects deletion of the latest +created revision even when it has no tag or traffic. Activation therefore creates a successor +before removing the validation tag and deleting validation. The dedicated 64-connection budget +reserves three simultaneous revision pools: serving, validation/rejected, +and active/recovery successor (12 configured pool connections at the current two-by-two shape). +Revision deletion is not proof of physical SQL session drain; verify termination and SQL sessions +in controlled rollout acceptance. There is no shutdown sleep used as a drain gate. + +**Activation deliberately starts production effects.** The distinct active revision uses the exact +validated digest with the validation override removed. Schema setup runs on its existing +one-connection untimed pool, followed by workers and pruners, before HTTP promotion. The workflow +checks digest, full runtime spec and secret-reference shape, scaling, readiness and active mode, +then moves HTTP traffic and checks the public origin. Those checks commit the new serving revision; +subsequent retirement failures do not trigger rollback to a possibly deleted previous revision. +The previous consumer is retired and all tags are cleared. Retain the previous immutable image +from the summary: later recovery redeploys that digest, because the previous revision is deleted. + +Before any candidate creation, the workflow requires exactly one revision resource, the sole HTTP +serving revision. Existing historical revisions or leftovers from interrupted runs require explicit +operator review and cleanup under the lease first; the workflow does not blindly delete them. +This gate and retirement after every successful rollout prevent repeated runs accumulating workers. +Terraform still owns configuration and scaling; removing validation mode adds no ignored field. + +On failure before public checks pass, any attempted traffic shift is first rolled back and verified. +If partial activation created a successor, recovery retires non-latest validation first; deletion +failure stops recovery before a fourth resource can be created. Recovery then deploys the captured +known-good digest as a tagged, zero-traffic successor with normal mode. It verifies template shape, +secret references and scaling, probes tagged readiness and active mode, promotes the recovery +revision, verifies traffic and public health, and only then deletes rejected and previous revisions. +The latest recovery revision remains serving. Known-good recovery schema and workers can execute +before promotion; neither recovery nor traffic rollback undoes schema changes or sent notifications. + +Partial creates record deterministic names before mutation. Failed recovery or deletion requires +operator cleanup under the lease; the next automated run refuses leftover resources. A canceled +runner can require the same intervention. Traffic restoration alone does not stop queue consumers. + +Manual recovery must preserve the three-resource bound and keep the successor serving: + +```sh +# Hold the rollout lease; inspect latest, traffic, tags and existing revisions first. +# If three resources remain after partial activation, retire non-latest inert validation first. +# Restore previous traffic if its revision still exists and a failed candidate took traffic. +gcloud run deploy orca-cloud-push \ + --project onorca-cloud --region us-central1 --image \ + --remove-env-vars ORCA_PUSH_MODE --no-traffic \ + --tag --revision-suffix +# Verify exact digest, template spec/secret references/scaling, tagged /ready and active /health. +gcloud run services update-traffic orca-cloud-push \ + --project onorca-cloud --region us-central1 --to-revisions =100 +# Verify traffic and public /ready and /health before retiring old consumers. +gcloud run services update-traffic orca-cloud-push \ + --project onorca-cloud --region us-central1 --clear-tags +gcloud run revisions delete \ + --project onorca-cloud --region us-central1 +# Repeat only for reviewed obsolete revisions; retain the latest serving recovery revision. +``` + +Never merely remove validation mode while the template still holds a rejected image. Terraform +owns environment configuration but ignores the image, so that would activate rejected code. +Remove a tag only if it remains present. Verify the recovery revision is serving, the template is safe, +and obsolete revision deletion and connection drain completed; +already accepted provider sends cannot be undone. Activation-time schema changes must be additive +and compatible with the rollback image: rollback does not reverse migrations or queue mutations. +The inert phase intentionally cannot validate a new schema by applying it to production. Review +migrations and validate them against isolated PostgreSQL before dispatch. No actual Cloud Run +rollout, provider delivery or physical-device acceptance is implied by local contract tests. + +### Incompatible queue rollout prerequisite + +The queue stores one notification object per delivery. Before deploying a revision that changes this +format, stop every older push gateway revision and clear only unpublished push delivery fixtures from +the push database. This is an unpublished feature, so do not preserve or migrate queued fixtures; no +production mutation is implied by this prerequisite. + +### Why the FCM probe impersonates the runtime account + +A gateway that boots and answers `/ready` can still be unable to send: the FCM grant lives on +the runtime service account, not on anything the readiness check touches. The probe therefore +mints an access token for `orca-cloud-push@onorca-cloud.iam.gserviceaccount.com` and posts +`validate_only: true` with a token that cannot exist. `validate_only` stops Google before any +delivery, and a healthy credential answers `INVALID_ARGUMENT` because the device token is +garbage. `PERMISSION_DENIED`, `401`, and `403` are the failures the step exists to catch, and +they fail the run immediately, before traffic moves. Those four answers are the only conclusive +ones: a `429`, a `5xx`, or a transport failure says nothing about the credential, so the send is +retried up to five times at five-second intervals rather than read as either verdict. Probing as the deploy identity instead would prove +something true about the wrong account. + +## Rotating the APNs key + +Apple keys do not expire, so this is for a suspected compromise or a routine rotation. Order +matters: the new key must be serving before the old one is revoked, or every iOS push fails in +the window between. + +1. In the Apple developer portal, create a **new** APNs authentication key. Download the `.p8` + once; Apple will not show it again. Note the new key ID. A team may hold two APNs keys at a + time, which is what makes this overlap possible. +2. Add a version to each changed secret, without printing the value: + + ```sh + gcloud secrets versions add orca-cloud-push-apns-key \ + --project onorca-cloud --data-file /path/to/AuthKey_NEW.p8 + printf '%s' '' | gcloud secrets versions add orca-cloud-push-apns-key-id \ + --project onorca-cloud --data-file=- + ``` + + The team ID does not change, so `orca-cloud-push-apple-team-id` is untouched. + +3. Dispatch `Deploy Push Gateway Production`. The container reads `latest` at start, so only a + new revision picks the key up; there is no in-place reload. +4. Verify from a real device that an iOS notification still arrives. The workflow's FCM probe + covers Android only, and APNs has no validate-only equivalent. +5. Only then revoke the old key in the Apple portal, and disable the superseded secret versions: + + ```sh + gcloud secrets versions disable \ + --project onorca-cloud --secret orca-cloud-push-apns-key + ``` + + Disable rather than destroy, so a rollback to the previous revision still works. Destroy + after the next clean deploy. + +Delete the downloaded `.p8` from disk when you are done. It is the whole credential. + +## Dead tokens + +A push token stops working when the app is uninstalled, when the user restores to a new device, +or when iOS reissues it. Both providers report this, and the shapes differ: + +- APNs: HTTP 410, or 400 with `BadDeviceToken` or `Unregistered`. + `DeviceTokenNotForTopic` is a provider configuration error and leaves the registration live. + Check the APNs topic and environment; future notifications can resume after correction without + phone re-registration. The failed notification is not retried for this non-transient error. +- FCM: `UNREGISTERED`, or `INVALID_ARGUMENT` whose message names the token. + +The gateway marks the registration `dead_at` and returns `status: "dead"` for it, and the +desktop drops the registration when it sees that. Nothing here retries a dead token. A phone +that comes back re-registers the same host/device pair, retaining its `registrationId` and +clearing `dead_at`. The per-minute `delivery_dead` counter measures delivery outcomes, not +currently dead registrations. A spike across many hosts warrants checking credentials and topics. + +## Quotas + +Two independent limits, both enforced in the gateway and both returning HTTP 200 with +`status: "rate_limited"` per result rather than failing the request: + +| Limit | Scope | +| --------------------------------------------- | --------------------------------------- | +| 300 logical alerts per rolling 15 minutes | per `hostFingerprint` | +| 300 logical dismissals per rolling 15 minutes | per `hostFingerprint`, separate budget | +| 20 `registrationIds` | per request, hard cap, HTTP 400 over it | + +Fanout to several phones counts one logical event; there is no per-phone daily allowance. +Unauthenticated handshakes and invalid bearer attempts have separate 30/minute IP buckets. +Authenticated routes use a 600/minute host bucket and a shared 6,000/minute client-IP bucket +per instance. The IP budget cannot be reset by generating another host key. It is shared by +clients behind one NAT and is an abuse safeguard, not a global provider-spending cap. Auth database lookup concurrency +and waiting work are bounded independently of HTTP concurrency. + +`push_events` backs quota accounting. `push_event_recipients` deduplicates fanout and +`push_delivery_batches` retains its historical name and persists individual deliveries, worker +leases, retries and outcomes. Identity metadata +is retained for 24 hours. Payloads expire within five minutes and are cleared on completion or by +minute-level expiry cleanup. FCM project-level provider quotas remain independent of host limits. + +Logging is aggregate counters only. Never log a token, a title, a body, or a full fingerprint; +the first four characters of a fingerprint are the most that may appear. + +## DNS: one hand-managed record + +The Cloud Run domain mapping is created here, and Google issues and renews the certificate. The +`onorca.dev` zone is not in this root: it is a Cloudflare zone whose Terraform-managed records +live in the apps root in `stablyai/orca-cloud`, and whose relay and auth records are managed by +hand. The push record follows the relay's precedent and was created by hand on 2026-09-04: + +```text +push.onorca.dev. CNAME ghs.googlehosted.com. (DNS only, not proxied) +``` + +`terraform -chdir=infra/terraform output push_dns_record` prints the same three fields. If the +record is ever lost, recreate it exactly like that; Cloudflare proxying blocks certificate +issuance and breaks Cloud Run host routing. + +### Recovery and delivery guarantees + +Candidate tags and deterministic revision names are recorded before deployment. Promotion intent is +recorded before changing traffic, so a failed verification or ambiguous mutation result still triggers +rollback. A known-good successor must exist before the rejected latest revision can be deleted. +After verified recovery promotion and public checks, rejected and previous consumers are retired; +the recovery revision remains serving. Failed cleanup blocks subsequent rollout admission. +The summary runs even if candidate discovery or traffic verification fails. + +Push uses the relay's schema-startup retry implementation through `@orca-cloud/postgres-schema`. +Session replacement is serialized per host and a unique host index upgrades older databases by +retaining their newest session. Cloud Verify runs push concurrency tests against PostgreSQL. + +Accepted sends commit quota and pending work together before returning `queued`. Workers resume +unfinished deliveries after restarts without relying on desktop retries. The durable queue and +expiring leases coordinate replicas. All provider attempts retain the original five-minute deadline +and respect provider backoff; no retry extends alert life. Silent dismissal messages have their own +quota and cancel matching unsent alerts. Mobile OS delivery/execution is not guaranteed. + +Shutdown stops admission and new claims; unfinished leases remain recoverable. Provider acceptance +and SQL completion cannot be atomic, so repeated transport delivery remains possible after a crash. +Stable per-event replacement identities reduce duplicates without promising exactly-once visible +delivery. FCM notification messages are inherently collapsible while offline and support only a +small number of concurrent collapse keys per device, so excess pending messages may be discarded and +every offline alert is not guaranteed to appear. Socket reconnect reconciles dismissals against the +current native tray; it has no stored replay watermark and never recovers a missed OS banner. + +### Dedicated database operations + +Push has one dedicated database attachment, with stable Terraform addresses and deletion +protection. There is no switch to shared storage. Follow the [database operations runbook](./push-database-cutover.md) +for deployment prerequisites, legacy resource ownership, capacity and recovery. diff --git a/cloud/docs/relay-workflows.md b/cloud/docs/relay-workflows.md index 14bb2a25d7c..54c0a681f2c 100644 --- a/cloud/docs/relay-workflows.md +++ b/cloud/docs/relay-workflows.md @@ -400,3 +400,30 @@ after checkout and authentication, before package installation, revision checks, Their typed confirmations are `PAUSE_REGIONAL_REHOMING` and `DISABLE_REGIONAL_REHOMING`. Keep the default 3,600,000 ms drain grace so existing splices can finish. The job summary contains only fresh aggregate active, receipt, registration, completion, and abort counts. + +## Mobile push gateway + +`Deploy Push Gateway Production` (`.github/workflows/cloud-push-deploy.yml`) is the deploy path +for `orca-cloud-push`, the mobile push gateway. It is the one `cloud-*` workflow that is not a +relay operation, and it is here because it shares the Artifact Registry repository and rollout +lease. Push uses a dedicated Cloud SQL instance. + +It authenticates through `PRODUCTION_GCP_PUSH_DEPLOY_WORKLOAD_IDENTITY_PROVIDER` and +`PRODUCTION_GCP_PUSH_DEPLOY_SERVICE_ACCOUNT`. The dedicated identity has Artifact Registry writer, +Cloud Run developer on the push service, and impersonation of only the push runtime account. +Its provider pins the repository, production environment, main branch, and exact dispatch workflow; +its distinct principal attribute cannot assume the shared Relay deploy account. + +Foundation grants the dedicated account access to the rollout-lock prefix and bucket metadata. +Apply that companion grant and publish the identity outputs before running the workflow. See +[push gateway deployment setup](./push-gateway.md#deploying) for the activation steps. + +The run builds the exact reviewed image digest before taking the lease and rejects images +without validation-mode support using a network-isolated container. Under the lease it boots +an inert, read-only validation revision, checks readiness, mode, scaling and FCM credentials, +then deletes it before deliberately activating the same digest in a new revision. Activation +starts schema writes, pruners and queue consumers before HTTP promotion. Rollback requires +restoring traffic, deleting the rejected active revision, and restoring the service template to +the known-good image in normal mode. The untagged template-recovery revision can run known-good +workers and is retired before lease release; see the +[deployment and rollback contract](./push-gateway.md#deploying). diff --git a/cloud/infra/terraform/environments/production.tfvars b/cloud/infra/terraform/environments/production.tfvars index e1522b3827e..925a171719d 100644 --- a/cloud/infra/terraform/environments/production.tfvars +++ b/cloud/infra/terraform/environments/production.tfvars @@ -412,3 +412,12 @@ relay_region_rehome_source_cell_ids = [ # Slack #orca-relay-alerts, created out of band on 2026-08-05. Declared here because an apply # was otherwise going to strip it from every policy, leaving the alerts firing at nobody. relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels/4879431412695417284"] + +# Mobile push gateway. Production is the only environment that runs one; the runtime account, +# the three Apple secrets, and their accessor bindings already exist and are imported once +# (see docs/push-gateway.md). +push_gateway_enabled = true +push_base_url = "https://push.onorca.dev" +# Dedicated push pools allow three revision resources during validation and recovery. +push_max_instances = 2 +manage_push_domain_mapping = true diff --git a/cloud/infra/terraform/environments/staging.tfvars b/cloud/infra/terraform/environments/staging.tfvars index 4a32458fcd5..1cb260e11ea 100644 --- a/cloud/infra/terraform/environments/staging.tfvars +++ b/cloud/infra/terraform/environments/staging.tfvars @@ -81,3 +81,6 @@ relay_gce_cells = { } relay_region_rehome_source_cell_ids = ["staging-gce-c2", "staging-gce-c3"] + +# Push is currently provisioned only in production. +push_gateway_enabled = false diff --git a/cloud/infra/terraform/outputs.tf b/cloud/infra/terraform/outputs.tf index 220aa5cf94f..5fd3b647b76 100644 --- a/cloud/infra/terraform/outputs.tf +++ b/cloud/infra/terraform/outputs.tf @@ -189,3 +189,27 @@ output "relay_gce_cell_deployments" { error_message = "relay_gce_fenced_cells may contain only configured relay_gce_cells keys." } } + +output "push_cloud_run_service_uri" { + value = try(google_cloud_run_v2_service.push[0].uri, null) + description = "Default push gateway service URI for pre-domain smoke tests." +} + +output "push_runtime_service_account" { + value = try(google_service_account.push_runtime[0].email, null) + description = "Runtime identity that holds the APNs key and sends through FCM." +} + +output "push_database_name" { + value = try(google_sql_database.push_dedicated[0].name, null) + description = "Database isolated for durable push gateway state." +} + +output "push_dns_record" { + value = var.push_gateway_enabled ? { + name = local.push_fqdn + type = "CNAME" + data = "ghs.googlehosted.com." + } : null + description = "Record the stablyai/orca-cloud apps root must publish in the onorca.dev zone." +} diff --git a/cloud/infra/terraform/push-dedicated-database.tf b/cloud/infra/terraform/push-dedicated-database.tf new file mode 100644 index 00000000000..9a4b9e4c69f --- /dev/null +++ b/cloud/infra/terraform/push-dedicated-database.tf @@ -0,0 +1,111 @@ +resource "google_sql_database_instance" "push_dedicated" { + count = local.push_gateway_count + + project = var.project_id + name = "${var.name_prefix}-push-db" + region = var.region + database_version = "POSTGRES_17" + deletion_protection = true + + settings { + tier = "db-custom-2-7680" + availability_type = "REGIONAL" + edition = "ENTERPRISE" + disk_type = "PD_SSD" + disk_size = 50 + disk_autoresize = true + user_labels = local.relay_shared_labels + + backup_configuration { + enabled = true + point_in_time_recovery_enabled = true + transaction_log_retention_days = 7 + start_time = "05:00" + backup_retention_settings { + retained_backups = 7 + } + } + + ip_configuration { + ipv4_enabled = true + ssl_mode = "ENCRYPTED_ONLY" + } + + maintenance_window { + day = 7 + hour = 6 + update_track = "stable" + } + + deletion_protection_enabled = true + } + + lifecycle { + prevent_destroy = true + } +} + +resource "google_sql_database" "push_dedicated" { + count = local.push_gateway_count + + project = var.project_id + name = "orca_push" + instance = google_sql_database_instance.push_dedicated[0].name + + lifecycle { + prevent_destroy = true + } +} + +resource "random_password" "push_dedicated_database" { + count = local.push_gateway_count + length = 32 + special = false +} + +resource "google_sql_user" "push_dedicated" { + count = local.push_gateway_count + + project = var.project_id + name = "orca_push" + instance = google_sql_database_instance.push_dedicated[0].name + password = random_password.push_dedicated_database[0].result +} + +resource "google_secret_manager_secret" "push_dedicated_database_url" { + count = local.push_gateway_count + + project = var.project_id + secret_id = "${var.name_prefix}-push-dedicated-database-url" + labels = local.relay_shared_labels + + replication { + auto {} + } + + lifecycle { + prevent_destroy = true + } +} + +resource "google_secret_manager_secret_version" "push_dedicated_database_url" { + count = local.push_gateway_count + + secret = google_secret_manager_secret.push_dedicated_database_url[0].id + secret_data = format( + "postgresql://%s:%s@/%s?host=/cloudsql/%s", + google_sql_user.push_dedicated[0].name, + random_password.push_dedicated_database[0].result, + google_sql_database.push_dedicated[0].name, + google_sql_database_instance.push_dedicated[0].connection_name + ) +} + +resource "google_secret_manager_secret_iam_member" "push_dedicated_database_url_accessor" { + count = local.push_gateway_count + + project = var.project_id + secret_id = google_secret_manager_secret.push_dedicated_database_url[0].secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.push_runtime[0].member +} diff --git a/cloud/infra/terraform/push-deploy-identity.tf b/cloud/infra/terraform/push-deploy-identity.tf new file mode 100644 index 00000000000..057c8fb8bee --- /dev/null +++ b/cloud/infra/terraform/push-deploy-identity.tf @@ -0,0 +1,82 @@ +locals { + # Deployment uses its own production-only identity. + push_gateway_deploy_count = ( + var.push_gateway_enabled && local.relay_create_production_ops_identity ? 1 : 0 + ) + github_push_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "assertion.workflow_ref == '${prefix}push-deploy.yml@refs/heads/main' && assertion.job_workflow_ref == '${prefix}push-deploy.yml@refs/heads/main'" + ] + push_deploy_member = one(google_service_account.github_push_deploy[*].member) +} + +resource "google_service_account" "github_push_deploy" { + count = local.push_gateway_deploy_count + account_id = "${var.name_prefix}-gha-push" + display_name = "Orca push production deploy" +} + +resource "google_iam_workload_identity_pool_provider" "github_push" { + count = local.push_gateway_deploy_count + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-push-deploy" + display_name = "GitHub push production deploy" + # Do not map attribute.repository: that principal set can assume the shared deploy account. + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.push_deploy" = "'production'" + } + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'production'", + "assertion.event_name == 'workflow_dispatch'", + local.relay_github_workflow_conditions["github_push"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_service_account_iam_member" "github_push_workload_identity_user" { + count = local.push_gateway_deploy_count + + service_account_id = google_service_account.github_push_deploy[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.push_deploy/production" +} + +resource "google_artifact_registry_repository_iam_member" "github_push_artifact_writer" { + count = local.push_gateway_deploy_count + + project = var.project_id + location = var.region + repository = var.artifact_repository_id + role = "roles/artifactregistry.writer" + member = local.push_deploy_member +} + +output "github_push_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_push[0].name, null) +} + +output "github_push_deploy_service_account" { + value = try(google_service_account.github_push_deploy[0].email, null) +} + + +resource "google_storage_bucket_iam_member" "github_push_rollout_lease" { + count = local.push_gateway_deploy_count + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectAdmin" + member = local.push_deploy_member + + condition { + title = "push_rollout_lease" + description = "Limits push deployment coordination to its own lease object." + expression = "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/push-rollout/production.lock'" + } +} diff --git a/cloud/infra/terraform/push-gateway.tf b/cloud/infra/terraform/push-gateway.tf new file mode 100644 index 00000000000..10164d1911e --- /dev/null +++ b/cloud/infra/terraform/push-gateway.tf @@ -0,0 +1,305 @@ +# Orca mobile push gateway (`cloud/apps/push`). +# +# One public Cloud Run service that holds the APNs key and sends through APNs and FCM V1 on +# behalf of paired phones. Operations: `docs/push-gateway.md`. +# +# There is no staging push gateway by decision, so every resource here is behind +# `var.push_gateway_enabled`, which only `environments/production.tfvars` sets true. The file +# still reads every environment-shaped value from a variable, like the rest of this root, so a +# future staging gateway is a tfvars edit rather than a rewrite. +# +# Several resources below already exist in `onorca-cloud`; they are declared so a plan is clean +# and imported once. `docs/push-gateway.md` carries the exact `terraform import` commands. + +locals { + push_gateway_count = var.push_gateway_enabled ? 1 : 0 + + # The runtime account, the three provider secrets, and their accessor bindings already exist in + # production and were created out of band with the Apple credentials. + push_runtime_service_account_id = "${var.name_prefix}-push" + + # Secret Manager holds the Apple credentials. Terraform owns the secret names, labels, and + # replication; it never owns a version. The `.p8` is issued by the Apple developer portal and + # rotated by `docs/push-gateway.md`, so a Terraform-managed version would either put the key in + # state or fight the rotation. `ignore_changes` on the whole resource is not available, so the + # versions are simply not declared and every consumer reads `latest`. + push_provider_secret_ids = var.push_gateway_enabled ? toset([ + "${var.name_prefix}-push-apns-key", + "${var.name_prefix}-push-apns-key-id", + "${var.name_prefix}-push-apple-team-id" + ]) : toset([]) + + push_provider_secret_env = { + "${var.name_prefix}-push-apns-key" = "ORCA_PUSH_APNS_KEY" + "${var.name_prefix}-push-apns-key-id" = "ORCA_PUSH_APNS_KEY_ID" + "${var.name_prefix}-push-apple-team-id" = "ORCA_PUSH_APPLE_TEAM_ID" + } + + push_fqdn = replace(replace(var.push_base_url, "https://", ""), "http://", "") + + +} + +# --- Runtime identity --------------------------------------------------------------------- + +resource "google_service_account" "push_runtime" { + count = local.push_gateway_count + + project = var.project_id + account_id = local.push_runtime_service_account_id + display_name = "Orca mobile push gateway" + description = "Runtime identity for the Orca mobile push gateway; sends through FCM V1." +} + +# FCM V1 sends are authorized by the runtime account's own metadata-server token. +resource "google_project_iam_member" "push_runtime_fcm_admin" { + count = local.push_gateway_count + + project = var.project_id + role = "roles/firebasecloudmessaging.admin" + member = google_service_account.push_runtime[0].member +} + +# The FCM V1 endpoint bills against the caller's project quota, which the caller must consume. +resource "google_project_iam_member" "push_runtime_service_usage_consumer" { + count = local.push_gateway_count + + project = var.project_id + role = "roles/serviceusage.serviceUsageConsumer" + member = google_service_account.push_runtime[0].member +} + +resource "google_project_iam_member" "push_runtime_cloudsql_client" { + count = local.push_gateway_count + + project = var.project_id + role = "roles/cloudsql.client" + member = google_service_account.push_runtime[0].member +} + +# --- Apple credentials ---------------------------------------------------------------------- + +resource "google_secret_manager_secret" "push_provider" { + for_each = local.push_provider_secret_ids + + project = var.project_id + secret_id = each.value + labels = local.relay_shared_labels + + replication { + auto {} + } + + # Why: Apple issues a `.p8` once and Secret Manager has no undelete. Turning the gateway off + # must fail the plan rather than destroy the only copy of the signing key. + lifecycle { + prevent_destroy = true + } +} + +resource "google_secret_manager_secret_iam_member" "push_provider_runtime_accessor" { + for_each = local.push_provider_secret_ids + + project = var.project_id + secret_id = google_secret_manager_secret.push_provider[each.value].secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.push_runtime[0].member +} + +# --- Service -------------------------------------------------------------------------------- + +resource "google_cloud_run_v2_service" "push" { + count = local.push_gateway_count + + project = var.project_id + name = var.push_cloud_run_service_name + location = var.region + ingress = "INGRESS_TRAFFIC_ALL" + # Why: the host proof in `POST /v1/host/challenge` is the authentication, not Cloud Run IAM. + # The project's domain-restricted-sharing policy refuses an `allUsers` invoker binding, so the + # service opts out of invoker IAM exactly as the relay director does. + invoker_iam_disabled = true + deletion_protection = var.environment == "production" + labels = local.relay_shared_labels + + template { + service_account = google_service_account.push_runtime[0].email + timeout = "${var.push_request_timeout_seconds}s" + max_instance_request_concurrency = var.push_concurrency + + scaling { + min_instance_count = var.push_min_instances + max_instance_count = var.push_max_instances + } + + volumes { + name = "cloudsql" + + cloud_sql_instance { + instances = [google_sql_database_instance.push_dedicated[0].connection_name] + } + } + + containers { + image = var.push_cloud_run_image + + ports { + container_port = 8080 + } + + volume_mounts { + name = "cloudsql" + mount_path = "/cloudsql" + } + + env { + name = "ORCA_PUSH_PUBLIC_URL" + value = var.push_base_url + } + + env { + name = "ORCA_PUSH_FCM_PROJECT_ID" + value = var.project_id + } + + # Bound the declared pool against the dedicated database rollout budget. + env { + name = "ORCA_PUSH_DATABASE_POOL_MAX" + value = tostring(var.push_database_pool_max) + } + + env { + name = "ORCA_PUSH_DATABASE_URL" + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.push_dedicated_database_url[0].secret_id + version = google_secret_manager_secret_version.push_dedicated_database_url[0].version + } + } + } + + # Rotation adds a new version and redeploys; `latest` is what the redeploy picks up. + dynamic "env" { + for_each = local.push_provider_secret_env + + content { + name = env.value + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.push_provider[env.key].secret_id + version = "latest" + } + } + } + } + + resources { + limits = { + cpu = var.push_cloud_run_cpu + memory = var.push_cloud_run_memory + } + + cpu_idle = false + } + + startup_probe { + failure_threshold = 12 + initial_delay_seconds = 0 + period_seconds = 5 + timeout_seconds = 2 + + http_get { + path = "/health" + port = 8080 + } + } + } + } + + # Deploys update the immutable image and shift traffic; Terraform owns the shape and IAM. + # + # `traffic` is ignored as well as the image. A deploy ends with traffic pinned to an exact + # revision and a rollback pins it to the previous one; an apply that reset the service to + # 100% LATEST would silently undo either, and this root carries unrelated standing drift, so + # that apply need not be a push change at all. + lifecycle { + precondition { + condition = var.push_max_instances * var.push_database_pool_max * 3 <= 64 + error_message = "Dedicated push serving, validation/rejected and successor pools must fit the 64-connection rollout budget." + } + + ignore_changes = [ + client, + client_version, + template[0].containers[0].image, + traffic + ] + } + + depends_on = [ + data.google_artifact_registry_repository.relay_images, + google_project_iam_member.push_runtime_cloudsql_client, + google_secret_manager_secret_iam_member.push_provider_runtime_accessor, + google_secret_manager_secret_version.push_dedicated_database_url, + google_secret_manager_secret_iam_member.push_dedicated_database_url_accessor + ] +} + +# Google issues and renews the certificate for the mapping. The DNS record itself is a +# hand-managed Cloudflare CNAME to ghs.googlehosted.com, like relay.onorca.dev; this root has no +# Cloudflare surface by design. `terraform output push_dns_record` prints the record. +resource "google_cloud_run_domain_mapping" "push" { + count = var.push_gateway_enabled && var.manage_push_domain_mapping ? 1 : 0 + + location = var.region + name = local.push_fqdn + + metadata { + namespace = var.project_id + } + + spec { + route_name = google_cloud_run_v2_service.push[0].name + } + + # Same reason as relay-dns.tf: a gcloud-created mapping reports an empty legacy + # certificate_mode, and replacing it would reset issuance for no behavioral change. + lifecycle { + ignore_changes = [spec[0].certificate_mode] + } +} + +# --- Deploy identity grants ------------------------------------------------------------------- +# Push deploy authority is isolated from Relay; foundation grants its rollout-lock access. + +resource "google_cloud_run_v2_service_iam_member" "github_production_push_developer" { + count = local.push_gateway_deploy_count + + project = var.project_id + location = var.region + name = google_cloud_run_v2_service.push[0].name + role = "roles/run.developer" + member = local.push_deploy_member +} + +resource "google_service_account_iam_member" "github_production_push_runtime_user" { + count = local.push_gateway_deploy_count + + service_account_id = google_service_account.push_runtime[0].name + role = "roles/iam.serviceAccountUser" + member = local.push_deploy_member +} + +# Why: the deploy workflow's validate-only FCM send has to exercise the credential the gateway +# will actually use. Impersonating the runtime account proves its firebasecloudmessaging grant; +# granting the deploy account FCM admin outright would prove nothing about the runtime account +# and would widen a project-level role on the shared identity. +resource "google_service_account_iam_member" "github_production_push_runtime_token_creator" { + count = local.push_gateway_deploy_count + + service_account_id = google_service_account.push_runtime[0].name + role = "roles/iam.serviceAccountTokenCreator" + member = local.push_deploy_member +} diff --git a/cloud/infra/terraform/relay-github-actions.tf b/cloud/infra/terraform/relay-github-actions.tf index 450ea64cc0a..186d8942f3d 100644 --- a/cloud/infra/terraform/relay-github-actions.tf +++ b/cloud/infra/terraform/relay-github-actions.tf @@ -20,6 +20,7 @@ locals { "deploy-relay-production.yml", "operate-relay-asia-admission.yml", "publish-relay-production.yml" + ] github_production_relay_capacity_workflow_file = "deploy-relay-production-capacity.yml" github_production_relay_capacity_job_workflow_file = "deploy-relay-production-capacity-job.yml" diff --git a/cloud/infra/terraform/relay-github-workflow-trust.tf b/cloud/infra/terraform/relay-github-workflow-trust.tf index 9cf57f05198..431408f0d9a 100644 --- a/cloud/infra/terraform/relay-github-workflow-trust.tf +++ b/cloud/infra/terraform/relay-github-workflow-trust.tf @@ -10,6 +10,7 @@ locals { relay_github_workflow_clauses = { github = local.github_production_relay_workflow_clauses + github_push = local.github_push_workflow_clauses github_monitor = local.github_monitor_workflow_clauses github_fence = local.github_fence_workflow_clauses github_production_relay_capacity = local.github_production_relay_capacity_workflow_clauses diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index 57d73fe75b4..b4360bb6a8a 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -484,3 +484,95 @@ variable "relay_gce_cloud_sql_proxy_image" { error_message = "relay_gce_cloud_sql_proxy_image must be pinned by sha256 digest." } } + +# --- Mobile push gateway --------------------------------------------------------------------- +# There is no staging push gateway by decision, so this defaults false and only +# environments/production.tfvars turns it on. Everything in push-gateway.tf is behind it. +variable "push_gateway_enabled" { + type = bool + description = "Create the Orca mobile push gateway, its database, secrets, and identity." + default = false +} + +variable "push_base_url" { + type = string + description = "Public TLS origin of the mobile push gateway." + default = "https://push.onorca.dev" + + validation { + condition = can(regex("^https://[^/]+$", var.push_base_url)) + error_message = "push_base_url must be an HTTPS origin with no path." + } +} + +variable "push_cloud_run_service_name" { + type = string + description = "Cloud Run service name for the mobile push gateway." + default = "orca-cloud-push" +} + +variable "push_cloud_run_image" { + type = string + description = "Initial image for the Terraform-created push gateway service; deploys own it after." + default = "us-docker.pkg.dev/cloudrun/container/hello" +} + +variable "push_cloud_run_cpu" { + type = string + description = "CPU limit for the push gateway container." + default = "1" +} + +variable "push_cloud_run_memory" { + type = string + description = "Memory limit for the push gateway container." + default = "512Mi" +} + +# Keep a warm instance to run durable delivery retries without incoming requests. +variable "push_min_instances" { + type = number + description = "Minimum instances for the push gateway." + default = 1 +} + +variable "push_max_instances" { + type = number + description = "Maximum instances for the push gateway." + default = 4 + + validation { + condition = var.push_max_instances >= 1 + error_message = "The push gateway needs at least one instance." + } +} + +# The dedicated database budget counts pools across all three rollout revision resources. +variable "push_database_pool_max" { + type = number + description = "Push gateway database pool size per instance; instances x pool is its Cloud SQL draw." + default = 2 + + validation { + condition = var.push_database_pool_max >= 1 && var.push_database_pool_max <= 100 + error_message = "The push gateway pool must hold at least one connection and stay under the per-service bound." + } +} + +variable "push_concurrency" { + type = number + description = "Cloud Run concurrency for short-lived push gateway HTTP requests." + default = 80 +} + +variable "push_request_timeout_seconds" { + type = number + description = "Cloud Run timeout for push gateway requests; every route is short-lived." + default = 30 +} + +variable "manage_push_domain_mapping" { + type = bool + description = "Manage the push gateway Cloud Run domain mapping; the DNS record stays in the apps root." + default = false +} diff --git a/cloud/package.json b/cloud/package.json index 242bbbd824c..c2c4b8bd025 100644 --- a/cloud/package.json +++ b/cloud/package.json @@ -21,7 +21,7 @@ "load:relay:recovery-gate": "node dev/scripts/run-relay-recovery-wave-gate.mjs", "ops:relay": "pnpm --filter @orca-cloud/relay-ops dev", "pretest": "node --test dev/scripts/capture-terraform-plan-baseline.test.mjs dev/scripts/operate-relay-asia-admission.test.mjs dev/scripts/prepare-relay-asia-director-cells.test.mjs dev/scripts/prepare-relay-asia-topology-input.test.mjs dev/scripts/production-cloud-sql-rollout-lock.test.mjs dev/scripts/read-relay-serving-regional-placement-version.test.mjs dev/scripts/relay-asia-admission-workflow.test.mjs dev/scripts/relay-asia-rollout-evidence.test.mjs dev/scripts/relay-asia-topology-workflow.test.mjs dev/scripts/relay-cloud-sql-connection-budget.test.mjs dev/scripts/relay-load-reader-evidence.test.mjs dev/scripts/relay-region-hint-metrics.test.mjs dev/scripts/relay-staging-deploy-identity.test.mjs dev/scripts/sanitize-relay-asia-admission-result.test.mjs dev/scripts/terraform-root-partition.test.mjs dev/scripts/validate-relay-asia-topology-plan.test.mjs ../.github/actions/cloud-sql-rollout-lease/action-contract.test.mjs ../.github/actions/cloud-sql-rollout-lease/storage-lease.test.mjs", - "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs dev/scripts/relay-admin-transient-retry.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-same-cap-script-census.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", + "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/push-gateway-workflow.test.mjs dev/scripts/push-gateway-recovery.test.mjs dev/scripts/push-validation-workflow.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admin-endpoint-retry-workflow.test.mjs dev/scripts/relay-admin-transient-retry.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-same-cap-script-census.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", "typecheck": "pnpm -r typecheck" }, "devDependencies": { diff --git a/cloud/packages/postgres-schema/package.json b/cloud/packages/postgres-schema/package.json new file mode 100644 index 00000000000..e170973cf2b --- /dev/null +++ b/cloud/packages/postgres-schema/package.json @@ -0,0 +1,20 @@ +{ + "name": "@orca-cloud/postgres-schema", + "version": "0.0.0", + "private": true, + "type": "module", + "main": "dist/index.js", + "types": "dist/index.d.ts", + "scripts": { + "build": "tsc -p tsconfig.build.json", + "clean": "node -e \"require('fs').rmSync('dist', { recursive: true, force: true })\"", + "lint": "tsc -p tsconfig.json --noEmit", + "test": "pnpm build", + "typecheck": "tsc -p tsconfig.json --noEmit" + }, + "devDependencies": { + "@types/node": "^24.10.0", + "typescript": "^5.9.3", + "vitest": "^4.0.8" + } +} diff --git a/cloud/packages/postgres-schema/src/index.ts b/cloud/packages/postgres-schema/src/index.ts new file mode 100644 index 00000000000..9981a03ead5 --- /dev/null +++ b/cloud/packages/postgres-schema/src/index.ts @@ -0,0 +1,113 @@ +const RETRYABLE_SCHEMA_CODES = new Set(['55P03', '57014']) +const DEFAULT_RETRY_DEADLINE_MS = 30_000 +const RETRY_BASE_DELAY_MS = 250 +const RETRY_MAX_DELAY_MS = 2_000 + +type SchemaStartupOptions = { + eventPrefix?: string + now?: () => number + random?: () => number + retryDeadlineMs?: number + wait?: (delayMs: number) => Promise +} + +function retryDelayMs(attempt: number, random: () => number): number { + const ceiling = Math.min(RETRY_BASE_DELAY_MS * 2 ** (attempt - 1), RETRY_MAX_DELAY_MS) + return Math.ceil(ceiling * (0.5 + random() * 0.5)) +} + +function wait(delayMs: number): Promise { + return new Promise((resolve) => setTimeout(resolve, delayMs)) +} + +const CREATE_TABLE_IF_NOT_EXISTS = /^\s*CREATE\s+TABLE\s+IF\s+NOT\s+EXISTS\b/i +const CREATE_INDEX_IF_NOT_EXISTS = /^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+IF\s+NOT\s+EXISTS\b/i + +// `IF NOT EXISTS` only checks the name before the catalog inserts, so the loser of a concurrent +// CREATE can fail on the catalog unique index (23505) or, when the winner has already committed by +// the time the loser reaches TypeCreate/heap_create_with_catalog, on the name check those routines +// repeat (42710 duplicate type, 42P07 duplicate relation). Each is a no-op on the next attempt. +function concurrentCreateCollision( + value: { code?: unknown; constraint?: unknown }, + statement: string +): boolean { + if (CREATE_TABLE_IF_NOT_EXISTS.test(statement)) { + return ( + (value.code === '23505' && value.constraint === 'pg_type_typname_nsp_index') || + value.code === '42710' || + value.code === '42P07' + ) + } + if (CREATE_INDEX_IF_NOT_EXISTS.test(statement)) { + return ( + (value.code === '23505' && value.constraint === 'pg_class_relname_nsp_index') || + value.code === '42P07' + ) + } + return false +} + +const ALTER_TABLE_ADD_CONSTRAINT = /^\s*ALTER\s+TABLE\s+\S+\s+ADD\s+CONSTRAINT\b/i + +function constraintAlreadyApplied(error: unknown, statement: string): boolean { + return ( + ALTER_TABLE_ADD_CONSTRAINT.test(statement) && + (error as { code?: unknown }).code === '42710' + ) +} + +function retryableSchemaError(error: unknown, statement: string): boolean { + const value = error as { code?: unknown; constraint?: unknown } + return ( + RETRYABLE_SCHEMA_CODES.has(String(value.code)) || concurrentCreateCollision(value, statement) + ) +} + +export async function applyPostgresSchema( + statements: string[], + query: (statement: string) => Promise, + options: SchemaStartupOptions = {} +): Promise { + const now = options.now ?? Date.now + const random = options.random ?? Math.random + const pause = options.wait ?? wait + const deadlineAt = now() + (options.retryDeadlineMs ?? DEFAULT_RETRY_DEADLINE_MS) + + for (const statement of statements) { + let attempt = 1 + while (true) { + try { + await query(statement) + break + } catch (error) { + if (constraintAlreadyApplied(error, statement)) break + const code = String((error as { code?: unknown }).code) + const remainingMs = deadlineAt - now() + const retryable = retryableSchemaError(error, statement) + if (!retryable || remainingMs <= 0) { + if (retryable) { + console.warn( + JSON.stringify({ + event: `${options.eventPrefix ?? 'orca_relay_postgres_schema'}_retry_exhausted`, + code, + attempts: attempt + }) + ) + } + throw error + } + const delayMs = Math.min(remainingMs, retryDelayMs(attempt, random)) + console.warn( + JSON.stringify({ + event: `${options.eventPrefix ?? 'orca_relay_postgres_schema'}_retry`, + code, + attempt, + delayMs + }) + ) + await pause(delayMs) + attempt += 1 + } + } + } +} diff --git a/cloud/packages/postgres-schema/tsconfig.build.json b/cloud/packages/postgres-schema/tsconfig.build.json new file mode 100644 index 00000000000..94c84b60803 --- /dev/null +++ b/cloud/packages/postgres-schema/tsconfig.build.json @@ -0,0 +1,11 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "declaration": true, + "emitDeclarationOnly": false, + "noEmit": false, + "outDir": "dist", + "rootDir": "src" + }, + "exclude": ["src/**/*.test.ts"] +} diff --git a/cloud/packages/postgres-schema/tsconfig.json b/cloud/packages/postgres-schema/tsconfig.json new file mode 100644 index 00000000000..a552e34dbe9 --- /dev/null +++ b/cloud/packages/postgres-schema/tsconfig.json @@ -0,0 +1,5 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "noEmit": true }, + "include": ["src/**/*.ts"] +} diff --git a/cloud/packages/push-contract/package.json b/cloud/packages/push-contract/package.json new file mode 100644 index 00000000000..072b5e7193f --- /dev/null +++ b/cloud/packages/push-contract/package.json @@ -0,0 +1,23 @@ +{ + "name": "@orca-cloud/push-contract", + "private": true, + "version": "0.0.0", + "type": "module", + "main": "dist/index.js", + "types": "dist/index.d.ts", + "scripts": { + "build": "pnpm clean && tsc -p tsconfig.build.json", + "clean": "node -e \"require('fs').rmSync('dist', { recursive: true, force: true })\"", + "lint": "tsc -p tsconfig.json --noEmit", + "test": "vitest run", + "typecheck": "tsc -p tsconfig.json --noEmit" + }, + "dependencies": { + "zod": "^3.25.76" + }, + "devDependencies": { + "@types/node": "^24.10.0", + "typescript": "^5.9.3", + "vitest": "^4.0.8" + } +} diff --git a/cloud/packages/push-contract/src/apns-token-length.test.ts b/cloud/packages/push-contract/src/apns-token-length.test.ts new file mode 100644 index 00000000000..d85413a22ef --- /dev/null +++ b/cloud/packages/push-contract/src/apns-token-length.test.ts @@ -0,0 +1,26 @@ +import { expect, it } from 'vitest' +import { PushDeviceRegistrationRequestSchema } from './device-registration-messages.js' + +const registration = (token: string) => ({ + v: 1, + deviceId: 'qa-device', + platform: 'ios', + token, + apnsEnvironment: 'sandbox' +}) + +it.each([32, 64, 160, 256])( + 'accepts variable-length APNs device tokens (%i hex characters)', + (length) => { + expect( + PushDeviceRegistrationRequestSchema.safeParse(registration('aB'.repeat(length / 2))).success + ).toBe(true) + } +) + +it.each(['', 'abc', 'not-hex', 'ab cd', 'ab'.repeat(2049)])( + 'rejects malformed or oversized APNs tokens', + (token) => { + expect(PushDeviceRegistrationRequestSchema.safeParse(registration(token)).success).toBe(false) + } +) diff --git a/cloud/packages/push-contract/src/contract.test.ts b/cloud/packages/push-contract/src/contract.test.ts new file mode 100644 index 00000000000..edfb915e72d --- /dev/null +++ b/cloud/packages/push-contract/src/contract.test.ts @@ -0,0 +1,162 @@ +import { describe, expect, it } from 'vitest' +import { + ApnsEnvironmentSchema, + PushDeviceRegistrationRequestSchema +} from './device-registration-messages.js' +import { + PushHostChallengeRequestSchema, + PushHostChallengeResponseSchema, + PushHostSessionRequestSchema, + PushHostSessionResponseSchema +} from './host-auth-messages.js' +import { PUSH_DEFAULTS, PUSH_LIMITS } from './push-limits.js' + +const KEY_B64 = Buffer.alloc(32, 1).toString('base64') +const NONCE_B64 = Buffer.alloc(24, 2).toString('base64') +const SESSION_TOKEN = Buffer.alloc(32, 3).toString('base64url') +const FINGERPRINT = 'abcdefghijklmnop' +const APNS_TOKEN = 'a'.repeat(64) +const FCM_TOKEN = 'cQ1abcDEF_gh:APA91bZZ-zz0123456789abcdefghijklmnopqrstuvwxyz' + +function notification(): Record { + return { + notificationId: 'note-1', + notificationSeq: 4, + notificationEpoch: '5c9e9a1e-0000-4000-8000-000000000000', + source: 'agent-task-complete', + agentState: 'needs-input', + title: 'Agent needs input', + body: 'Waiting on your answer', + worktreeId: 'wt-1' + } +} + +describe('push contract limits', () => { + it('locks the normative limits the desktop and gateway both assume', () => { + expect(PUSH_LIMITS).toMatchObject({ + titleMaxChars: 80, + bodyMaxChars: 180, + maxRegistrationIdsPerSend: 20, + maxDevicesPerHost: 64, + hostEventsPerWindow: 300, + eventQuotaWindowMs: 900_000, + challengeTtlMs: 10_000, + clockSkewToleranceMs: 30_000, + sessionTtlMs: 86_400_000, + notificationTtlSeconds: 300, + unauthenticatedRequestsPerMinutePerIp: 30, + authenticatedRequestsPerMinutePerIp: 6_000, + authenticatedRequestsPerMinutePerHost: 600 + }) + expect(PUSH_DEFAULTS.apnsTopic).toBe('com.stably.orca.mobile') + expect(PUSH_DEFAULTS.androidChannelId).toBe('orca-desktop') + }) +}) + +describe('host authentication schemas', () => { + it('accepts a well formed challenge round trip', () => { + expect( + PushHostChallengeRequestSchema.safeParse({ v: 1, hostPublicKeyB64: KEY_B64 }).success + ).toBe(true) + expect( + PushHostChallengeResponseSchema.safeParse({ + challengeId: 'challenge-1', + gatewayEphemeralPublicKeyB64: KEY_B64, + nonceB64: NONCE_B64, + ciphertextB64: Buffer.alloc(96, 5).toString('base64'), + expiresAt: 1_700_000_010_000 + }).success + ).toBe(true) + expect( + PushHostSessionRequestSchema.safeParse({ + v: 1, + challengeId: 'challenge-1', + proofB64: KEY_B64 + }).success + ).toBe(true) + expect( + PushHostSessionResponseSchema.safeParse({ + sessionToken: SESSION_TOKEN, + expiresAt: 1_700_086_400_000, + hostFingerprint: FINGERPRINT + }).success + ).toBe(true) + }) + + it('rejects unknown keys, wrong versions, and mis-sized keys', () => { + expect( + PushHostChallengeRequestSchema.safeParse({ + v: 1, + hostPublicKeyB64: KEY_B64, + extra: true + }).success + ).toBe(false) + expect( + PushHostChallengeRequestSchema.safeParse({ v: 2, hostPublicKeyB64: KEY_B64 }).success + ).toBe(false) + expect( + PushHostChallengeRequestSchema.safeParse({ + v: 1, + hostPublicKeyB64: Buffer.alloc(31, 1).toString('base64') + }).success + ).toBe(false) + expect( + PushHostSessionResponseSchema.safeParse({ + sessionToken: SESSION_TOKEN, + expiresAt: 1_700_086_400_000, + hostFingerprint: 'short' + }).success + ).toBe(false) + }) +}) + +describe('device registration schemas', () => { + it('requires an apns environment and a hex token for ios', () => { + expect( + PushDeviceRegistrationRequestSchema.safeParse({ + v: 1, + deviceId: 'device-1', + platform: 'ios', + token: APNS_TOKEN, + apnsEnvironment: 'sandbox' + }).success + ).toBe(true) + expect( + PushDeviceRegistrationRequestSchema.safeParse({ + v: 1, + deviceId: 'device-1', + platform: 'ios', + token: APNS_TOKEN + }).success + ).toBe(false) + expect( + PushDeviceRegistrationRequestSchema.safeParse({ + v: 1, + deviceId: 'device-1', + platform: 'ios', + token: 'not-hex', + apnsEnvironment: 'production' + }).success + ).toBe(false) + }) + + it('rejects an apns environment on android and accepts an fcm token', () => { + expect( + PushDeviceRegistrationRequestSchema.safeParse({ + v: 1, + deviceId: 'device-2', + platform: 'android', + token: FCM_TOKEN + }).success + ).toBe(true) + expect( + PushDeviceRegistrationRequestSchema.safeParse({ + v: 1, + deviceId: 'device-2', + platform: 'android', + token: FCM_TOKEN, + apnsEnvironment: 'sandbox' + }).success + ).toBe(false) + }) +}) diff --git a/cloud/packages/push-contract/src/device-registration-messages.ts b/cloud/packages/push-contract/src/device-registration-messages.ts new file mode 100644 index 00000000000..37af2cc9d68 --- /dev/null +++ b/cloud/packages/push-contract/src/device-registration-messages.ts @@ -0,0 +1,74 @@ +import { z } from 'zod' +import { OpaqueIdSchema } from './wire-scalars.js' + +export const PushPlatformSchema = z.enum(['ios', 'android']) +export const ApnsEnvironmentSchema = z.enum(['sandbox', 'production']) +export const PushNotificationSourceSchema = z.enum([ + 'agent-task-complete', + 'terminal-bell', + 'plugin' +]) +export const PushAgentStateSchema = z.enum(['needs-input', 'finished']) + +// APNs tokens are variable-length byte strings, including longer simulator tokens. +const APNS_TOKEN_PATTERN = /^(?:[0-9a-fA-F]{2})+$/ +const FCM_TOKEN_PATTERN = /^[A-Za-z0-9_:.\-]{32,4096}$/ + +export const PushDeviceRegistrationRequestSchema = z + .object({ + v: z.literal(1), + deviceId: OpaqueIdSchema, + platform: PushPlatformSchema, + token: z.string().min(1).max(4096), + apnsEnvironment: ApnsEnvironmentSchema.optional() + }) + .strict() + .superRefine((value, context) => { + if (value.platform === 'ios') { + if (value.apnsEnvironment === undefined) { + context.addIssue({ + code: 'custom', + path: ['apnsEnvironment'], + message: 'apnsEnvironment is required for ios' + }) + } + if (!APNS_TOKEN_PATTERN.test(value.token)) { + context.addIssue({ + code: 'custom', + path: ['token'], + message: 'ios token must be hex-encoded bytes' + }) + } + return + } + if (value.apnsEnvironment !== undefined) { + context.addIssue({ + code: 'custom', + path: ['apnsEnvironment'], + message: 'apnsEnvironment is ios only' + }) + } + if (!FCM_TOKEN_PATTERN.test(value.token)) { + context.addIssue({ + code: 'custom', + path: ['token'], + message: 'android token must be an FCM registration string' + }) + } + }) + +export const PushDeviceSummarySchema = z + .object({ + registrationId: OpaqueIdSchema, + deviceId: OpaqueIdSchema, + platform: PushPlatformSchema, + dead: z.boolean() + }) + .strict() + +export type PushPlatform = z.infer +export type ApnsEnvironment = z.infer +export type PushNotificationSource = z.infer +export type PushAgentState = z.infer +export type PushDeviceRegistrationRequest = z.infer +export type PushDeviceSummary = z.infer diff --git a/cloud/packages/push-contract/src/host-auth-messages.ts b/cloud/packages/push-contract/src/host-auth-messages.ts new file mode 100644 index 00000000000..deb9ffbfb42 --- /dev/null +++ b/cloud/packages/push-contract/src/host-auth-messages.ts @@ -0,0 +1,41 @@ +import { z } from 'zod' +import { + Base6432ByteSchema, + Base64Raw24ByteSchema, + Base64Url32ByteSchema, + BoundedCiphertextSchema, + EpochMsSchema, + OpaqueIdSchema, + PushHostFingerprintSchema +} from './wire-scalars.js' + +export const PushHostChallengeRequestSchema = z + .object({ v: z.literal(1), hostPublicKeyB64: Base6432ByteSchema }) + .strict() + +export const PushHostChallengeResponseSchema = z + .object({ + challengeId: OpaqueIdSchema, + gatewayEphemeralPublicKeyB64: Base6432ByteSchema, + nonceB64: Base64Raw24ByteSchema, + ciphertextB64: BoundedCiphertextSchema, + expiresAt: EpochMsSchema + }) + .strict() + +export const PushHostSessionRequestSchema = z + .object({ v: z.literal(1), challengeId: OpaqueIdSchema, proofB64: Base6432ByteSchema }) + .strict() + +export const PushHostSessionResponseSchema = z + .object({ + sessionToken: Base64Url32ByteSchema, + expiresAt: EpochMsSchema, + hostFingerprint: PushHostFingerprintSchema + }) + .strict() + +export type PushHostChallengeRequest = z.infer +export type PushHostChallengeResponse = z.infer +export type PushHostSessionRequest = z.infer +export type PushHostSessionResponse = z.infer diff --git a/cloud/packages/push-contract/src/index.ts b/cloud/packages/push-contract/src/index.ts new file mode 100644 index 00000000000..3bd8a871f28 --- /dev/null +++ b/cloud/packages/push-contract/src/index.ts @@ -0,0 +1,6 @@ +export * from './device-registration-messages.js' +export * from './host-auth-messages.js' +export * from './push-host-proof-transcript.js' +export * from './push-limits.js' +export * from './send-messages.js' +export * from './wire-scalars.js' diff --git a/cloud/packages/push-contract/src/notification-identity-limits.test.ts b/cloud/packages/push-contract/src/notification-identity-limits.test.ts new file mode 100644 index 00000000000..e19fd93140a --- /dev/null +++ b/cloud/packages/push-contract/src/notification-identity-limits.test.ts @@ -0,0 +1,32 @@ +import { expect, it } from 'vitest' +import { PushNotificationSchema } from './send-messages.js' +const base = { + source: 'agent-task-complete', + agentState: 'finished', + notificationSeq: 1, + notificationEpoch: 'epoch', + title: 'Done', + body: '' +} +it.each([ + 'repo::/Users/developer/orca/workspaces/monorepo/packages/desktop/integrations/feature-mobile-background-notifications', + 'repo::C:\\Users\\developer\\Documents\\projects\\monorepo\\packages\\desktop\\feature-mobile-notifications', + 'folder::/home/developer/projects/通知/作業ディレクトリ/機能', + 'ssh:host::/home/developer/workspaces/monorepo/packages/desktop/feature-mobile-background-notifications' +])('preserves long desktop identities: %s', (path) => { + const worktreeId = `12345678-1234-1234-1234-123456789012::${path}` + const notificationId = [ + 'agent', + encodeURIComponent(worktreeId), + encodeURIComponent('12345678-1234-1234-1234-123456789012:87654321-4321-4321-4321-210987654321'), + '1780000000123' + ].join(':') + const result = PushNotificationSchema.parse({ ...base, worktreeId, notificationId }) + expect(result.worktreeId).toBe(worktreeId) + expect(result.notificationId).toBe(notificationId) +}) +it('rejects oversized provider data by UTF-8 bytes instead of truncating identities', () => { + expect(PushNotificationSchema.safeParse({ ...base, worktreeId: '界'.repeat(1100) }).success).toBe( + false + ) +}) diff --git a/cloud/packages/push-contract/src/push-host-proof-transcript.test.ts b/cloud/packages/push-contract/src/push-host-proof-transcript.test.ts new file mode 100644 index 00000000000..34423beaf6d --- /dev/null +++ b/cloud/packages/push-contract/src/push-host-proof-transcript.test.ts @@ -0,0 +1,106 @@ +import { describe, expect, it } from 'vitest' +import { + buildPushHostChallengePlaintext, + buildPushHostProofMacInput, + buildPushHostProofTranscript, + PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN, + PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN, + PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT +} from './push-host-proof-transcript.js' +import { PUSH_LIMITS } from './push-limits.js' + +const transcriptInput = { + gatewayOrigin: 'https://push.onorca.dev', + gatewayEphemeralPublicKey: new Uint8Array(32).fill(7), + challengeNonce: new Uint8Array(24).fill(9), + challengeId: 'challenge-1', + issuedAt: 1_700_000_000_000, + expiresAt: 1_700_000_000_000 + PUSH_LIMITS.challengeTtlMs, + hostFingerprint: 'abcdefghijklmnop', + hostPublicKey: new Uint8Array(32).fill(4) +} + +describe('push host proof transcript', () => { + it('is deterministic and order dependent', () => { + const first = buildPushHostProofTranscript(transcriptInput) + const second = buildPushHostProofTranscript({ ...transcriptInput }) + expect(Buffer.from(first).equals(Buffer.from(second))).toBe(true) + const different = buildPushHostProofTranscript({ + ...transcriptInput, + challengeId: 'challenge-2' + }) + expect(Buffer.from(first).equals(Buffer.from(different))).toBe(false) + }) + + it('encodes exactly the ten specified fields in order', () => { + const transcript = buildPushHostProofTranscript(transcriptInput) + const view = new DataView(transcript.buffer, transcript.byteOffset, transcript.byteLength) + const names: string[] = [] + let offset = 0 + while (offset < transcript.byteLength) { + const nameLength = view.getUint32(offset, false) + offset += 4 + names.push(Buffer.from(transcript.slice(offset, offset + nameLength)).toString('utf8')) + offset += nameLength + offset += 4 + view.getUint32(offset, false) + } + expect(names).toEqual([ + 'protocol', + 'version', + 'gatewayOrigin', + 'gatewayEphemeralPublicKey', + 'challengeNonce', + 'challengeId', + 'issuedAt', + 'expiresAt', + 'hostFingerprint', + 'hostPublicKey' + ]) + expect(names).toHaveLength(PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT) + expect(offset).toBe(transcript.byteLength) + }) + + it('rejects mis-sized key material', () => { + expect(() => + buildPushHostProofTranscript({ + ...transcriptInput, + hostPublicKey: new Uint8Array(31) + }) + ).toThrow('hostPublicKey must be 32 bytes') + expect(() => + buildPushHostProofTranscript({ ...transcriptInput, challengeNonce: new Uint8Array(23) }) + ).toThrow('challengeNonce must be 24 bytes') + }) + + it('frames the challenge plaintext as domain, length, transcript, secret', () => { + const transcript = buildPushHostProofTranscript(transcriptInput) + const secret = new Uint8Array(32).fill(11) + const plaintext = buildPushHostChallengePlaintext(transcript, secret) + const domain = Buffer.from(`${PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`, 'utf8') + expect(Buffer.from(plaintext.slice(0, domain.byteLength)).equals(domain)).toBe(true) + const declared = new DataView( + plaintext.buffer, + plaintext.byteOffset + domain.byteLength, + 4 + ).getUint32(0, false) + expect(declared).toBe(transcript.byteLength) + expect(plaintext.byteLength).toBe(domain.byteLength + 4 + transcript.byteLength + 32) + expect( + Buffer.from(plaintext.slice(plaintext.byteLength - 32)).equals(Buffer.from(secret)) + ).toBe(true) + expect(() => buildPushHostChallengePlaintext(transcript, new Uint8Array(16))).toThrow( + 'challengeSecret must be 32 bytes' + ) + }) + + it('separates the ack mac input from the challenge domain', () => { + const transcript = buildPushHostProofTranscript(transcriptInput) + const macInput = buildPushHostProofMacInput(transcript) + expect(Buffer.from(macInput).toString('utf8')).toContain( + `${PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN}\0ack\0` + ) + expect(macInput.byteLength).toBe( + Buffer.byteLength(`${PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN}\0ack\0`) + transcript.byteLength + ) + }) +}) diff --git a/cloud/packages/push-contract/src/push-host-proof-transcript.ts b/cloud/packages/push-contract/src/push-host-proof-transcript.ts new file mode 100644 index 00000000000..a375b18ca76 --- /dev/null +++ b/cloud/packages/push-contract/src/push-host-proof-transcript.ts @@ -0,0 +1,90 @@ +const textEncoder = new TextEncoder() + +export const PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN = 'orca-push-host-proof/v1' +export const PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN = 'orca-push-host-challenge/v1' +export const PUSH_HOST_CHALLENGE_BOX_ALGORITHM = 'Curve25519-XSalsa20-Poly1305' +export const PUSH_HOST_PROOF_ALGORITHM = 'HMAC-SHA-256' + +export interface PushHostProofTranscriptInput { + gatewayOrigin: string + gatewayEphemeralPublicKey: Uint8Array + challengeNonce: Uint8Array + challengeId: string + issuedAt: number + expiresAt: number + hostFingerprint: string + hostPublicKey: Uint8Array +} + +export const PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT = 10 + +function uint32(value: number): Uint8Array { + const bytes = new Uint8Array(4) + new DataView(bytes.buffer).setUint32(0, value, false) + return bytes +} + +function uint64(value: number): Uint8Array { + const bytes = new Uint8Array(8) + new DataView(bytes.buffer).setBigUint64(0, BigInt(value), false) + return bytes +} + +function concat(parts: readonly Uint8Array[]): Uint8Array { + const output = new Uint8Array(parts.reduce((total, part) => total + part.byteLength, 0)) + let offset = 0 + for (const part of parts) { + output.set(part, offset) + offset += part.byteLength + } + return output +} + +function field(name: string, value: Uint8Array): Uint8Array { + const encodedName = textEncoder.encode(name) + return concat([uint32(encodedName.byteLength), encodedName, uint32(value.byteLength), value]) +} + +function text(value: string): Uint8Array { + return textEncoder.encode(value) +} + +function requireByteLength(value: Uint8Array, expected: number, name: string): void { + if (value.byteLength !== expected) throw new Error(`${name} must be ${expected} bytes`) +} + +export function buildPushHostProofTranscript(input: PushHostProofTranscriptInput): Uint8Array { + requireByteLength(input.gatewayEphemeralPublicKey, 32, 'gatewayEphemeralPublicKey') + requireByteLength(input.challengeNonce, 24, 'challengeNonce') + requireByteLength(input.hostPublicKey, 32, 'hostPublicKey') + return concat([ + field('protocol', text(PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN)), + field('version', new Uint8Array([1])), + field('gatewayOrigin', text(input.gatewayOrigin)), + field('gatewayEphemeralPublicKey', input.gatewayEphemeralPublicKey), + field('challengeNonce', input.challengeNonce), + field('challengeId', text(input.challengeId)), + field('issuedAt', uint64(input.issuedAt)), + field('expiresAt', uint64(input.expiresAt)), + field('hostFingerprint', text(input.hostFingerprint)), + field('hostPublicKey', input.hostPublicKey) + ]) +} + +export function buildPushHostChallengePlaintext( + transcript: Uint8Array, + challengeSecret: Uint8Array +): Uint8Array { + if (challengeSecret.byteLength !== 32) throw new Error('challengeSecret must be 32 bytes') + // Why: the encrypted random secret makes the public transcript insufficient to forge the ack. + return concat([ + text(`${PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`), + uint32(transcript.byteLength), + transcript, + challengeSecret + ]) +} + +export function buildPushHostProofMacInput(transcript: Uint8Array): Uint8Array { + return concat([text(`${PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN}\0ack\0`), transcript]) +} diff --git a/cloud/packages/push-contract/src/push-host-proof-vector.json b/cloud/packages/push-contract/src/push-host-proof-vector.json new file mode 100644 index 00000000000..128eba46980 --- /dev/null +++ b/cloud/packages/push-contract/src/push-host-proof-vector.json @@ -0,0 +1,16 @@ +{ + "hostSecretKeyB64": "BwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwc=", + "hostPublicKeyB64": "E75P6uryBMf9M1j8nAByGIHRdCeBKCJ+xnTzf3/pe20=", + "hostFingerprint": "D20lU_8MD0R64gLt", + "gatewayOrigin": "https://push.onorca.dev", + "challenge": { + "challengeId": "vector-challenge-1", + "gatewayEphemeralPublicKeyB64": "V9tLNZ8jrl4Ubk4lEgVnBHIlBjSMFQwUdT0Mkz0E1CE=", + "nonceB64": "AwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMD", + "ciphertextB64": "znNOCR0fq0KKa5dwfTAwbhE6GmfC4TUjgB5n+/0BXrrG0A9oKjo38uvUY3VoBvTfCvlkLOmI2bu8kGN/yAHmMz6jhY77FIztAywVQ1WfBlu/tbxgiK/9QHxydUQwTAjc2vGjgPENC2EPH2VYZWEB10a6p6nlV3uezJda2exBLbJE/hPZGUkRJVedSa0WlQQpro/FwYqcqmI2iSpJ28nIQHn1wylc/Vgv7xw+/EBY39SzuR7HpY48h1MU0lzlsS1wcO2c/F7xEFYWUtfkbZGxET+b/eF6tzdLM5/MPJr8ibiwcPwfFfLnaYJYHpsFP0Tpu/ZQ3lLblX5Gqjf0vPn0MXB45RR/ZcMds1UUfC1WtDkFd2Z74xnN7GHTXNPYZwRChNC6TCxtK83UvqRfUqydzpTL5Z3R+zsunmSJvV8xONjW/ikwOqitjrMiqlnNGf7dFh4FC2vOfgg7HxwVQd8VumWeW2oT3WCcQH4FkxM2LjAvej34vE4WGPw9s6vcKoP4ESMG34TTVBz6Tyjm4oZv9ylLFrFISSkaZoZ5smKi/F0/xOscHKg4u4Sfz7wK+8Ve3Uc5eTos9yBkf1Ydbht7mbWqBSQTMC9BazmRZ5UlrM+GzGgI", + "expiresAt": 1800000010000 + }, + "issuedAt": 1800000000000, + "challengeSecretB64": "BQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQU=", + "transcriptB64": "AAAACHByb3RvY29sAAAAF29yY2EtcHVzaC1ob3N0LXByb29mL3YxAAAAB3ZlcnNpb24AAAABAQAAAA1nYXRld2F5T3JpZ2luAAAAF2h0dHBzOi8vcHVzaC5vbm9yY2EuZGV2AAAAGWdhdGV3YXlFcGhlbWVyYWxQdWJsaWNLZXkAAAAgV9tLNZ8jrl4Ubk4lEgVnBHIlBjSMFQwUdT0Mkz0E1CEAAAAOY2hhbGxlbmdlTm9uY2UAAAAYAwMDAwMDAwMDAwMDAwMDAwMDAwMDAwMDAAAAC2NoYWxsZW5nZUlkAAAAEnZlY3Rvci1jaGFsbGVuZ2UtMQAAAAhpc3N1ZWRBdAAAAAgAAAGjGFxQAAAAAAlleHBpcmVzQXQAAAAIAAABoxhcdxAAAAAPaG9zdEZpbmdlcnByaW50AAAAEEQyMGxVXzhNRDBSNjRnTHQAAAANaG9zdFB1YmxpY0tleQAAACATvk/q6vIEx/0zWPycAHIYgdF0J4EoIn7GdPN/f+l7bQ==" +} diff --git a/cloud/packages/push-contract/src/push-limits.ts b/cloud/packages/push-contract/src/push-limits.ts new file mode 100644 index 00000000000..0a4b9b25866 --- /dev/null +++ b/cloud/packages/push-contract/src/push-limits.ts @@ -0,0 +1,28 @@ +export const PUSH_LIMITS = { + titleMaxChars: 80, + bodyMaxChars: 180, + maxRegistrationIdsPerSend: 20, + // A host pairs phones, not a fleet. The cap bounds what one session can write + // through a caller-chosen deviceId. + maxDevicesPerHost: 64, + maxHttpBodyBytes: 16 * 1024, + hostEventsPerWindow: 300, + eventQuotaWindowMs: 15 * 60 * 1000, + challengeTtlMs: 10_000, + // Covers routine NTP drift without extending the signed challenge window. + clockSkewToleranceMs: 30_000, + sessionTtlMs: 24 * 60 * 60 * 1000, + notificationTtlSeconds: 5 * 60, + // The challenge and session routes are the only unauthenticated writes, so + // they are capped per client IP before any key material is generated. + unauthenticatedRequestsPerMinutePerIp: 30, + authenticatedRequestsPerMinutePerIp: 6_000, + authenticatedRequestsPerMinutePerHost: 600 +} as const + +export const PUSH_DEFAULTS = { + apnsTopic: 'com.stably.orca.mobile', + androidChannelId: 'orca-desktop' +} as const + +export const PUSH_HOST_FINGERPRINT_LENGTH = 16 diff --git a/cloud/packages/push-contract/src/send-messages.test.ts b/cloud/packages/push-contract/src/send-messages.test.ts new file mode 100644 index 00000000000..a9e14da79d9 --- /dev/null +++ b/cloud/packages/push-contract/src/send-messages.test.ts @@ -0,0 +1,115 @@ +import { describe, expect, it } from 'vitest' +import { PUSH_LIMITS } from './push-limits.js' +import { + PushSendRequestSchema, + PushSendStatusSchema +} from './send-messages.js' + +function notification(): Record { + return { + notificationId: 'note-1', + notificationSeq: 4, + notificationEpoch: '5c9e9a1e-0000-4000-8000-000000000000', + source: 'agent-task-complete', + agentState: 'needs-input', + title: 'Agent needs input', + body: 'Waiting on your answer', + worktreeId: 'wt-1' + } +} + +describe('send schemas', () => { + it('accepts a batch at the registration cap and a terminal bell without an id', () => { + const ids = Array.from({ length: PUSH_LIMITS.maxRegistrationIdsPerSend }, (_, i) => `reg-${i}`) + expect( + PushSendRequestSchema.safeParse({ v: 1, registrationIds: ids, notification: notification() }) + .success + ).toBe(true) + const { notificationId: _dropped, ...bell } = notification() + expect( + PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-1'], + notification: { ...bell, source: 'terminal-bell', agentState: null } + }).success + ).toBe(true) + }) + + it('rejects an oversized batch, over-long copy, and unknown notification keys', () => { + const ids = Array.from( + { length: PUSH_LIMITS.maxRegistrationIdsPerSend + 1 }, + (_, i) => `reg-${i}` + ) + expect( + PushSendRequestSchema.safeParse({ v: 1, registrationIds: ids, notification: notification() }) + .success + ).toBe(false) + expect( + PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-1'], + notification: { ...notification(), title: 'x'.repeat(PUSH_LIMITS.titleMaxChars + 1) } + }).success + ).toBe(false) + expect( + PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-1'], + notification: { ...notification(), body: 'x'.repeat(PUSH_LIMITS.bodyMaxChars + 1) } + }).success + ).toBe(false) + expect( + PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-1'], + notification: { ...notification(), coalescedCount: 2 } + }).success + ).toBe(false) + expect(PushSendRequestSchema.safeParse({ v: 1, registrationIds: [], notification: notification() }).success) + .toBe(false) + }) + + it('rejects a notification id that could not be sent as a collapse header', () => { + for (const notificationId of ['line\nbreak', 'nul\0byte', 'émoji', '\t']) { + expect( + PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-1'], + notification: { ...notification(), notificationId } + }).success + ).toBe(false) + } + expect( + PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-1'], + notification: { + ...notification(), + notificationId: 'agent:repo%3A%3A%2FUsers%2Fme:pane-1:1700000000000' + } + }).success + ).toBe(true) + }) + + it('dedupes repeated registration ids and keeps the first-seen order', () => { + const parsed = PushSendRequestSchema.safeParse({ + v: 1, + registrationIds: ['reg-b', 'reg-a', 'reg-b', 'reg-c', 'reg-a'], + notification: notification() + }) + expect(parsed.success).toBe(true) + expect(parsed.success && parsed.data.registrationIds).toEqual(['reg-b', 'reg-a', 'reg-c']) + }) + + it('counts duplicates against the batch cap before deduping them', () => { + const ids = Array.from({ length: PUSH_LIMITS.maxRegistrationIdsPerSend + 1 }, () => 'reg-1') + expect( + PushSendRequestSchema.safeParse({ v: 1, registrationIds: ids, notification: notification() }) + .success + ).toBe(false) + }) + + it('locks the send result statuses', () => { + expect(PushSendStatusSchema.options).toEqual(['queued', 'dead', 'rate_limited', 'error']) + }) +}) diff --git a/cloud/packages/push-contract/src/send-messages.ts b/cloud/packages/push-contract/src/send-messages.ts new file mode 100644 index 00000000000..a8959c18b05 --- /dev/null +++ b/cloud/packages/push-contract/src/send-messages.ts @@ -0,0 +1,66 @@ +import { z } from 'zod' +import { + PushAgentStateSchema, + PushNotificationSourceSchema +} from './device-registration-messages.js' +import { PUSH_LIMITS } from './push-limits.js' +import { OpaqueIdSchema, SequenceSchema } from './wire-scalars.js' + +export const PushNotificationSchema = z + .object({ + // Absent for terminal-bell, which the desktop raises without a notification record. + // Printable ASCII only: the id becomes the APNs collapse header, and the + // desktop builds it from URL-encoded parts, so anything else is not Orca's. + notificationId: z + .string() + .min(1) + .max(2048) + .regex(/^[\x20-\x7e]+$/) + .optional(), + notificationSeq: SequenceSchema, + notificationEpoch: OpaqueIdSchema, + source: PushNotificationSourceSchema, + sound: z.boolean().optional(), + kind: z.enum(['alert', 'dismiss']).optional(), + expiresAt: z.number().int().positive().optional(), + agentState: PushAgentStateSchema.nullable(), + title: z.string().min(1).max(PUSH_LIMITS.titleMaxChars), + body: z.string().max(PUSH_LIMITS.bodyMaxChars), + worktreeId: z.string().min(1).max(2048).optional(), + paneKey: z.string().min(1).max(2048).optional() + }) + .strict() + .refine( + (notification) => notification.kind !== 'dismiss' || Boolean(notification.notificationId), + { message: 'dismiss requires notificationId' } + ) + .refine( + (notification) => new TextEncoder().encode(JSON.stringify(notification)).byteLength <= 3000, + { + message: 'notification exceeds provider payload budget' + } + ) + +export const PushSendRequestSchema = z + .object({ + v: z.literal(1), + // Deduped before the gateway sees it so a repeated id cannot reserve quota twice. + registrationIds: z + .array(OpaqueIdSchema) + .min(1) + .max(PUSH_LIMITS.maxRegistrationIdsPerSend) + .transform((ids) => [...new Set(ids)]), + notification: PushNotificationSchema + }) + .strict() + +export const PushSendStatusSchema = z.enum(['queued', 'dead', 'rate_limited', 'error']) + +export const PushSendResultSchema = z + .object({ registrationId: OpaqueIdSchema, status: PushSendStatusSchema }) + .strict() + +export type PushNotification = z.infer +export type PushSendRequest = z.infer +export type PushSendStatus = z.infer +export type PushSendResult = z.infer diff --git a/cloud/packages/push-contract/src/wire-scalars.ts b/cloud/packages/push-contract/src/wire-scalars.ts new file mode 100644 index 00000000000..f190aef8354 --- /dev/null +++ b/cloud/packages/push-contract/src/wire-scalars.ts @@ -0,0 +1,16 @@ +import { z } from 'zod' + +// Copied from relay-contract rather than imported: the push gateway ships as a +// standalone image and must not pull the relay wire contract into its closure. +export const Base64Url32ByteSchema = z.string().regex(/^[A-Za-z0-9_-]{43}$/) +export const Base6432ByteSchema = z.string().regex(/^(?:[A-Za-z0-9+/]{4}){10}[A-Za-z0-9+/]{3}=$/) +export const Base64Raw24ByteSchema = z.string().regex(/^(?:[A-Za-z0-9+/]{4}){8}$/) +export const PushHostFingerprintSchema = z.string().regex(/^[A-Za-z0-9_-]{16}$/) +export const OpaqueIdSchema = z.string().min(1).max(128) +export const EpochMsSchema = z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER) +export const SequenceSchema = z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER) +export const BoundedCiphertextSchema = z + .string() + .min(1) + .max(16 * 1024) + .regex(/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/) diff --git a/cloud/packages/push-contract/tsconfig.build.json b/cloud/packages/push-contract/tsconfig.build.json new file mode 100644 index 00000000000..94c84b60803 --- /dev/null +++ b/cloud/packages/push-contract/tsconfig.build.json @@ -0,0 +1,11 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "declaration": true, + "emitDeclarationOnly": false, + "noEmit": false, + "outDir": "dist", + "rootDir": "src" + }, + "exclude": ["src/**/*.test.ts"] +} diff --git a/cloud/packages/push-contract/tsconfig.json b/cloud/packages/push-contract/tsconfig.json new file mode 100644 index 00000000000..a552e34dbe9 --- /dev/null +++ b/cloud/packages/push-contract/tsconfig.json @@ -0,0 +1,5 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "noEmit": true }, + "include": ["src/**/*.ts"] +} diff --git a/cloud/pnpm-lock.yaml b/cloud/pnpm-lock.yaml index 3598ae54930..aa8afb0da67 100644 --- a/cloud/pnpm-lock.yaml +++ b/cloud/pnpm-lock.yaml @@ -21,11 +21,60 @@ importers: specifier: ^4.0.8 version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + apps/push: + dependencies: + '@hono/node-server': + specifier: ^1.19.17 + version: 1.19.17(hono@4.13.7) + '@orca-cloud/postgres-schema': + specifier: workspace:* + version: link:../../packages/postgres-schema + '@orca-cloud/push-contract': + specifier: workspace:* + version: link:../../packages/push-contract + google-auth-library: + specifier: ^10.5.0 + version: 10.9.1 + hono: + specifier: ^4.13.7 + version: 4.13.7 + pg: + specifier: ^8.22.0 + version: 8.22.0 + pg-connection-string: + specifier: 2.14.0 + version: 2.14.0 + tweetnacl: + specifier: ^1.0.3 + version: 1.0.3 + zod: + specifier: ^3.25.76 + version: 3.25.76 + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + '@types/pg': + specifier: ^8.20.0 + version: 8.20.0 + tsx: + specifier: ^4.21.0 + version: 4.22.4 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + apps/relay: dependencies: '@hono/node-server': specifier: ^1.19.17 version: 1.19.17(hono@4.13.7) + '@orca-cloud/postgres-schema': + specifier: workspace:* + version: link:../../packages/postgres-schema '@orca-cloud/relay-contract': specifier: workspace:* version: link:../../packages/relay-contract @@ -117,6 +166,34 @@ importers: specifier: ^4.0.8 version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + packages/postgres-schema: + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + + packages/push-contract: + dependencies: + zod: + specifier: ^3.25.76 + version: 3.25.76 + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + packages/relay-contract: dependencies: zod: @@ -463,10 +540,23 @@ packages: '@vitest/utils@4.1.9': resolution: {integrity: sha512-A51o8ymO5PpqlWNnBP9ZHPXDIpuMtTLlGSjN7la4US+LJzoUMyhwjA5QXlm39JexgwHKW4Xjs8Z2d3dLCXOeuA==} + agent-base@7.1.4: + resolution: {integrity: sha512-MnA+YT8fwfJPgBx3m60MNqakm30XOkyIoH1y6huTQvC0PwZG7ki8NacLBcrPbNoo8vEZy7Jpuk7+jMO+CUovTQ==} + engines: {node: '>= 14'} + assertion-error@2.0.1: resolution: {integrity: sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA==} engines: {node: '>=12'} + base64-js@1.5.1: + resolution: {integrity: sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==} + + bignumber.js@9.3.1: + resolution: {integrity: sha512-Ko0uX15oIUS7wJ3Rb30Fs6SkVbLmPBAKdlm7q9+ak9bbIeFf0MwuBsQV6z7+X768/cHsfg+WlysDWJcmthjsjQ==} + + buffer-equal-constant-time@1.0.1: + resolution: {integrity: sha512-zRpUiDwd/xk6ADqPMATG8vc9VPrkck7T07OIx0gnjmJAnHnTVXNQG3vfvWNuiZIkwu9KrKdA1iJKfsfTVxE6NA==} + chai@6.2.2: resolution: {integrity: sha512-NUPRluOfOiTKBKvWPtSD4PhFvWCqOi0BGStNWs57X9js7XGTprSmFoz5F0tWhR4WPjNeR9jXqdC7/UpSJTnlRg==} engines: {node: '>=18'} @@ -474,10 +564,26 @@ packages: convert-source-map@2.0.0: resolution: {integrity: sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg==} + data-uri-to-buffer@4.0.1: + resolution: {integrity: sha512-0R9ikRb668HB7QDxT1vkpuUBtqc53YyAwMwGeUFKRojY/NWKvdZ+9UYtRfGmhqNbRkTSVpMbmyhXipFFv2cb/A==} + engines: {node: '>= 12'} + + debug@4.4.3: + resolution: {integrity: sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA==} + engines: {node: '>=6.0'} + peerDependencies: + supports-color: '*' + peerDependenciesMeta: + supports-color: + optional: true + detect-libc@2.1.2: resolution: {integrity: sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==} engines: {node: '>=8'} + ecdsa-sig-formatter@1.0.11: + resolution: {integrity: sha512-nagl3RYrbNv6kQkeJIpt6NJZy8twLB/2vtz6yN9Z4vRKHN4/QZJIEbqohALSgwKdnksuY3k5Addp5lg8sVoVcQ==} + es-module-lexer@2.1.0: resolution: {integrity: sha512-n27zTYMjYu1aj4MjCWzSP7G9r75utsaoc8m61weK+W8JMBGGQybd43GstCXZ3WNmSFtGT9wi59qQTW6mhTR5LQ==} @@ -493,6 +599,9 @@ packages: resolution: {integrity: sha512-knvyeauYhqjOYvQ66MznSMs83wmHrCycNEN6Ao+2AeYEfxUIkuiVxdEa1qlGEPK+We3n0THiDciYSsCcgW/DoA==} engines: {node: '>=12.0.0'} + extend@3.0.2: + resolution: {integrity: sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g==} + fdir@6.5.0: resolution: {integrity: sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==} engines: {node: '>=12.0.0'} @@ -502,18 +611,55 @@ packages: picomatch: optional: true + fetch-blob@3.2.0: + resolution: {integrity: sha512-7yAQpD2UMJzLi1Dqv7qFYnPbaPx7ZfFK6PiIxQ4PfkGPyNyl2Ugx+a/umUonmKqjhM4DnfbMvdX6otXq83soQQ==} + engines: {node: ^12.20 || >= 14.13} + + formdata-polyfill@4.0.10: + resolution: {integrity: sha512-buewHzMvYL29jdeQTVILecSaZKnt/RJWjoZCF5OW60Z67/GmSLBkOFM7qh1PI3zFNtJbaZL5eQu1vLfazOwj4g==} + engines: {node: '>=12.20.0'} + fsevents@2.3.3: resolution: {integrity: sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==} engines: {node: ^8.16.0 || ^10.6.0 || >=11.0.0} os: [darwin] + gaxios@7.3.1: + resolution: {integrity: sha512-kB3rzJV7d9juLZh8/56QTXCwQfxyhdOMdyYk1HdQKFtF8TJTDTZQJtixWIwXdE9Jji91mC41DUNpjleo4L4eAQ==} + engines: {node: '>=18'} + + gcp-metadata@8.1.2: + resolution: {integrity: sha512-zV/5HKTfCeKWnxG0Dmrw51hEWFGfcF2xiXqcA3+J90WDuP0SvoiSO5ORvcBsifmx/FoIjgQN3oNOGaQ5PhLFkg==} + engines: {node: '>=18'} + + google-auth-library@10.9.1: + resolution: {integrity: sha512-i1ydyHrqcIxXkWh/uBmVkzCvIuq5yiK2ATndIe5XxKholrG/MTYP9xGYka4sQhrbIAgGjL2B6NOE7rFaiF3fXw==} + engines: {node: '>=18'} + + google-logging-utils@1.1.3: + resolution: {integrity: sha512-eAmLkjDjAFCVXg7A1unxHsLf961m6y17QFqXqAXGj/gVkKFrEICfStRfwUlGNfeCEjNRa32JEWOUTlYXPyyKvA==} + engines: {node: '>=14'} + hono@4.13.7: resolution: {integrity: sha512-c8/gF9ac8Y78/agExVocyLevgR+JlpNB444Py0FSX8pJoPdYUfUzRcXtYEYGwt6l19qIlVZPN5Mfsw9jFShmQQ==} engines: {node: '>=16.9.0'} + https-proxy-agent@7.0.6: + resolution: {integrity: sha512-vK9P5/iUfdl95AI+JVyUuIcVtd4ofvtrOr3HNtM2yxC9bnMbEdp3x01OhQNnjb8IJYi38VlTE3mBXwcfvywuSw==} + engines: {node: '>= 14'} + jose@6.2.3: resolution: {integrity: sha512-YYVDInQKFJfR/xa3ojUTl8c2KoTwiL1R5Wg9YCydwH0x0B9grbzlg5HC7mMjCtUJjbQ/YnGEZIhI5tCgfTb4Hw==} + json-bigint@1.0.0: + resolution: {integrity: sha512-SiPv/8VpZuWbvLSMtTDU8hEfrZWg/mH/nV/b4o0CYbSxu1UIQPLdwKOCIyLQX+VIPO5vrLX3i8qtqFyhdPSUSQ==} + + jwa@2.0.1: + resolution: {integrity: sha512-hRF04fqJIP8Abbkq5NKGN0Bbr3JxlQ+qhZufXVr0DvujKy93ZCbXZMHDL4EOtodSbCWxOqR8MS1tXA5hwqCXDg==} + + jws@4.0.1: + resolution: {integrity: sha512-EKI/M/yqPncGUUh44xz0PxSidXFr/+r0pA70+gIYhjv+et7yxM+s29Y+VGDkovRofQem0fs7Uvf4+YmAdyRduA==} + lightningcss-android-arm64@1.32.0: resolution: {integrity: sha512-YK7/ClTt4kAK0vo6w3X+Pnm0D2cf2vPHbhOXdoNti1Ga0al1P4TBZhwjATvjNwLEBCnKvjJc2jQgHXH0NEwlAg==} engines: {node: '>= 12.0.0'} @@ -587,11 +733,23 @@ packages: magic-string@0.30.21: resolution: {integrity: sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ==} - nanoid@3.3.18: - resolution: {integrity: sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w==} + ms@2.1.3: + resolution: {integrity: sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==} + + nanoid@3.3.13: + resolution: {integrity: sha512-sPdqC6ByMVVGvF1ynvvMo0/o+oD1VX7DaHhijt1bFgjvBkHBib4t49GoNDhf2NDta4oeUNlaGbSt5K7qjZ955Q==} engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} hasBin: true + node-domexception@1.0.0: + resolution: {integrity: sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ==} + engines: {node: '>=10.5.0'} + deprecated: Use your platform's native DOMException instead + + node-fetch@3.3.2: + resolution: {integrity: sha512-dRB78srN/l6gqWulah9SrxeYnxeddIG30+GOqK/9OlLVyLg3HPnr6SqOWTWOXKRwC2eGYCkZ59NNuSgvSrpgOA==} + engines: {node: ^12.20.0 || ^14.13.1 || >=16.0.0} + obug@2.1.3: resolution: {integrity: sha512-9miFgM2OFba7hB+pRgvtV84pYTBaoTHohvmIgiRt6dRIzbwEOIaNaP+dIlGs2fNFoB0SeISs0Jz5WFVRid6Xyg==} engines: {node: '>=12.20.0'} @@ -640,8 +798,8 @@ packages: resolution: {integrity: sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==} engines: {node: '>=12'} - postcss@8.5.28: - resolution: {integrity: sha512-RRuzqDtt5Y9h3quz5hWhK+TPnsmVs6WwSU6LkJMeY4HstUEDuYTG8UJSdawMRzmzAtV+KEoG8N3Qg2qLy5vM/A==} + postcss@8.5.15: + resolution: {integrity: sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A==} engines: {node: ^10 || ^12 || >=14} postgres-array@2.0.0: @@ -665,6 +823,9 @@ packages: engines: {node: ^20.19.0 || >=22.12.0} hasBin: true + safe-buffer@5.2.1: + resolution: {integrity: sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==} + siginfo@2.0.0: resolution: {integrity: sha512-ybx0WO1/8bSBLEWXZvEd7gMW3Sn3JFlW3TvX1nREbDLRNQNaeNN8WK0meBwPdAaOI7TtRRRJn/Es1zhrrCHu7g==} @@ -800,6 +961,10 @@ packages: jsdom: optional: true + web-streams-polyfill@3.3.3: + resolution: {integrity: sha512-d2JWLCivmZYTSIoge9MsgFCZrt571BikcWGYkjC1khllbTeDlGqZ2D8vD8E/lJa8WGWbb7Plm8/XJYV7IJHZZw==} + engines: {node: '>= 8'} + why-is-node-running@2.3.0: resolution: {integrity: sha512-hUrmaWBdVDcxvYqnyh09zunKzROWjbZTiNy8dBEjkS7ehEDQibXJ7XvlmtbwuTclUiIyN+CyXQD4Vmko8fNm8w==} engines: {node: '>=8'} @@ -1057,14 +1222,32 @@ snapshots: convert-source-map: 2.0.0 tinyrainbow: 3.1.0 + agent-base@7.1.4: {} + assertion-error@2.0.1: {} + base64-js@1.5.1: {} + + bignumber.js@9.3.1: {} + + buffer-equal-constant-time@1.0.1: {} + chai@6.2.2: {} convert-source-map@2.0.0: {} + data-uri-to-buffer@4.0.1: {} + + debug@4.4.3: + dependencies: + ms: 2.1.3 + detect-libc@2.1.2: {} + ecdsa-sig-formatter@1.0.11: + dependencies: + safe-buffer: 5.2.1 + es-module-lexer@2.1.0: {} esbuild@0.28.1: @@ -1102,17 +1285,79 @@ snapshots: expect-type@1.3.0: {} + extend@3.0.2: {} + fdir@6.5.0(picomatch@4.0.4): optionalDependencies: picomatch: 4.0.4 + fetch-blob@3.2.0: + dependencies: + node-domexception: 1.0.0 + web-streams-polyfill: 3.3.3 + + formdata-polyfill@4.0.10: + dependencies: + fetch-blob: 3.2.0 + fsevents@2.3.3: optional: true + gaxios@7.3.1: + dependencies: + extend: 3.0.2 + https-proxy-agent: 7.0.6 + node-fetch: 3.3.2 + transitivePeerDependencies: + - supports-color + + gcp-metadata@8.1.2: + dependencies: + gaxios: 7.3.1 + google-logging-utils: 1.1.3 + json-bigint: 1.0.0 + transitivePeerDependencies: + - supports-color + + google-auth-library@10.9.1: + dependencies: + base64-js: 1.5.1 + ecdsa-sig-formatter: 1.0.11 + gaxios: 7.3.1 + gcp-metadata: 8.1.2 + google-logging-utils: 1.1.3 + jws: 4.0.1 + transitivePeerDependencies: + - supports-color + + google-logging-utils@1.1.3: {} + hono@4.13.7: {} + https-proxy-agent@7.0.6: + dependencies: + agent-base: 7.1.4 + debug: 4.4.3 + transitivePeerDependencies: + - supports-color + jose@6.2.3: {} + json-bigint@1.0.0: + dependencies: + bignumber.js: 9.3.1 + + jwa@2.0.1: + dependencies: + buffer-equal-constant-time: 1.0.1 + ecdsa-sig-formatter: 1.0.11 + safe-buffer: 5.2.1 + + jws@4.0.1: + dependencies: + jwa: 2.0.1 + safe-buffer: 5.2.1 + lightningcss-android-arm64@1.32.0: optional: true @@ -1166,7 +1411,17 @@ snapshots: dependencies: '@jridgewell/sourcemap-codec': 1.5.5 - nanoid@3.3.18: {} + ms@2.1.3: {} + + nanoid@3.3.13: {} + + node-domexception@1.0.0: {} + + node-fetch@3.3.2: + dependencies: + data-uri-to-buffer: 4.0.1 + fetch-blob: 3.2.0 + formdata-polyfill: 4.0.10 obug@2.1.3: {} @@ -1211,9 +1466,9 @@ snapshots: picomatch@4.0.4: {} - postcss@8.5.28: + postcss@8.5.15: dependencies: - nanoid: 3.3.18 + nanoid: 3.3.13 picocolors: 1.1.1 source-map-js: 1.2.1 @@ -1248,6 +1503,8 @@ snapshots: '@rolldown/binding-win32-arm64-msvc': 1.0.3 '@rolldown/binding-win32-x64-msvc': 1.0.3 + safe-buffer@5.2.1: {} + siginfo@2.0.0: {} source-map-js@1.2.1: {} @@ -1288,7 +1545,7 @@ snapshots: dependencies: lightningcss: 1.32.0 picomatch: 4.0.4 - postcss: 8.5.28 + postcss: 8.5.15 rolldown: 1.0.3 tinyglobby: 0.2.17 optionalDependencies: @@ -1324,6 +1581,8 @@ snapshots: transitivePeerDependencies: - msw + web-streams-polyfill@3.3.3: {} + why-is-node-running@2.3.0: dependencies: siginfo: 2.0.0 diff --git a/config/electron-builder.config.cjs b/config/electron-builder.config.cjs index 7e0009b3a24..6b36938a3d1 100644 --- a/config/electron-builder.config.cjs +++ b/config/electron-builder.config.cjs @@ -279,16 +279,6 @@ module.exports = { } }, afterPack: async (context) => { - // Why: a Linux runner-image glibc bump silently shipped a node-pty pty.node - // requiring GLIBC_2.34, crashing the app on startup on Ubuntu 20.04 (#9902). - // Fail packaging if any bundled native binary exceeds the supported floor. - if (context.electronPlatformName === 'linux') { - // Why the arch is passed: symbol-version checks pass happily on a wrong-architecture binary, - // so a cross-built slice could ship the host's pty.node and only fail at runtime. - verifyLinuxGlibcFloor(context.appOutDir, { - targetArch: { 1: 'x64', 3: 'arm64' }[context.arch] - }) - } const resourcesDir = context.electronPlatformName === 'darwin' ? join( @@ -326,6 +316,19 @@ module.exports = { } stampPackagedCliVersion(resourcesDir, context.packager.appInfo.version) prunePackagedRuntimeNodeModules(resourcesDir, context.electronPlatformName, context.arch) + // Why: a Linux runner-image glibc bump silently shipped a node-pty pty.node + // requiring GLIBC_2.34, crashing the app on startup on Ubuntu 20.04 (#9902). + // Fail packaging if any bundled native binary exceeds the supported floor. + // Why after the prune: cross-builds intentionally install every optional + // native variant, so an arm64 slice still carries the x64 @parcel/watcher + // until prunePackagedRuntimeNodeModules drops it. + if (context.electronPlatformName === 'linux') { + // Why the arch is passed: symbol-version checks pass happily on a wrong-architecture binary, + // so a cross-built slice could ship the host's pty.node and only fail at runtime. + verifyLinuxGlibcFloor(context.appOutDir, { + targetArch: { 1: 'x64', 3: 'arm64' }[context.arch] + }) + } verifyPackagedMainRuntimeDeps(resourcesDir) // Why: boot the packaged daemon-entry under plain Node, but only for the // slice matching the packaging host's arch — daemon-entry.js is JS, yet it diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index bd1b3c20ad6..4e1f44bd7c1 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -10,6 +10,252 @@ } }, "gates": [ + { + "id": "runtime.connection-owned-host-status", + "title": "Host status recovers with its owning connection", + "maturity": "experimental", + "protection": "partial", + "owner": "runtime", + "layer": "service-integration-and-e2e", + "surfaces": [ + "sidebar host status", + "desktop runtime connection", + "browser primary connection" + ], + "platforms": ["macos", "linux", "windows"], + "providers": ["remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["remote-runtime"], + "coverageNotes": "Real authenticated sockets plus isolated desktop and headless hosts with desktop and browser viewers; deterministic lifecycle tests cover stale results and reader deadlines.", + "motivatingLinks": ["https://github.com/stablyai/orca/pull/19163"], + "invariant": "Failed bootstrap and authenticated reconnect converge without UI triggers; one connection owner publishes verified status, with no independent healthy status polling.", + "oracle": "Observe automatic recovery, retained runtime identity on failure, ordered publications, exact request counts, isolated viewer outages, and retirement on disconnect.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/runtime-host-status-owner.test.ts src/main/ipc/runtime-environment-status-recovery.test.ts src/main/ipc/runtime-environment-status-connection.test.ts src/renderer/src/store/slices/runtime-status-snapshot.test.ts src/renderer/src/web/web-runtime-status-owner.test.ts", + "ORCA_BACKGROUND_LAUNCH=1 ORCA_E2E_WEB_CLIENT=1 pnpm exec playwright test tests/e2e/runtime-host-status-recovery.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1" + ], + "testFiles": [ + "src/shared/runtime-host-status-owner.test.ts", + "src/main/ipc/runtime-environment-status-recovery.test.ts", + "src/main/ipc/runtime-environment-status-connection.test.ts", + "src/renderer/src/store/slices/runtime-status-snapshot.test.ts", + "src/renderer/src/web/web-runtime-status-owner.test.ts", + "tests/e2e/runtime-host-status-recovery.spec.ts" + ], + "assertionRefs": [ + { + "file": "src/main/ipc/runtime-environment-status-recovery.test.ts", + "assertions": [ + "recovers a saved host after its first status check fails, without another UI request" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-10", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 ORCA_E2E_WEB_CLIENT=1 pnpm exec playwright test tests/e2e/runtime-host-status-recovery.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", + "result": "passed", + "summary": "Desktop-host and headless-host journeys passed with desktop and browser viewers.", + "durationSeconds": 31.3 + } + ], + "runtimeBudget": { + "p95Seconds": 180, + "scope": "Target excluding builds; measured p95 not established." + }, + "flakeHistory": { + "status": "soaking", + "evidence": "Local candidate runs passed; no sustained CI history yet." + }, + "redGreenEvidence": { + "status": "partial", + "evidence": "First-status-failure oracle failed on main 58ff95becb40 (one request instead of two) and passes on the candidate. E2E verifies candidate recovery, not a baseline comparison." + }, + "performanceBudget": { + "required": false, + "evidence": "Deterministic tests assert one shared request and no healthy owner polling." + }, + "promotionCriteria": ["Collect repeated CI runs without unexplained flakes."], + "knownGaps": [ + "No live Linux, Windows, SSH, or mixed-version pair validation.", + "TCP interruption exercises reconnect, not a full real host process restart.", + "The outage begins on the first saved-host check, not by relaunching a preseeded desktop profile." + ], + "demotionRule": "Keep experimental until repeated runs establish reliability; preserve request-count and lifecycle assertions." + }, + { + "id": "mobile-push.headless-startup-and-policy", + "title": "Headless push lifecycle and mobile delivery policy", + "maturity": "experimental", + "protection": "partial", + "owner": "runtime", + "layer": "service-integration", + "surfaces": ["headless startup", "push registration", "native push delivery policy"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "remote-runtime"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local", "remote-runtime"], + "coverageNotes": "Actual startOrcad entry with mocked daemon/RPC startup boundaries; real controller, push service, and persisted device registry. Gateway send is stubbed.", + "motivatingLinks": ["https://github.com/stablyai/orca/pull/19204"], + "invariant": "Headless startup installs and disposes push delivery; desktop notification categories remain authoritative, the three-minute away policy is preserved, and host activity cannot extend the seven-day mobile lease.", + "oracle": "Require registration after RPC identity initialization and shutdown cleanup; idle 179/180/0 yields false/true/false in retained event metadata and exactly one gateway push. Legacy socket subscriptions preserve notification filtering, while opted-in push clients can request dismissal reconciliation. Desktop categories remain authoritative. Persisted lease expires exactly at seven days and only explicit registration renews it.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/mobile-notification-dismissal-store.test.ts src/renderer/src/hooks/useAutoAckViewedAgent.away.test.ts", + "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/orcad/orcad-push-startup.test.ts src/main/runtime/push/push-policy-pipeline.integration.test.ts" + ], + "testFiles": [ + "src/main/orcad/orcad-push-startup.test.ts", + "src/main/runtime/push/push-policy-pipeline.integration.test.ts", + "src/main/runtime/mobile-notification-dismissal-store.test.ts", + "src/renderer/src/hooks/useAutoAckViewedAgent.away.test.ts" + ], + "assertionRefs": [ + { + "file": "src/main/orcad/orcad-push-startup.test.ts", + "assertions": [ + "starts push after RPC identity is available and stops dispatch on shutdown" + ] + }, + { + "file": "src/main/runtime/push/push-policy-pipeline.integration.test.ts", + "assertions": [ + "carries the native idle boundary through replay and push dispatch", + "expires persisted registration at seven days despite host activity and renews explicitly" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-07", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm exec vitest run --config config/vitest.config.ts src/main/orcad/orcad-push-startup.test.ts src/main/runtime/push/push-policy-pipeline.integration.test.ts", + "result": "passed", + "summary": "Four tests passed across two files.", + "durationSeconds": 0.407 + } + ], + "runtimeBudget": { + "p95Seconds": 10, + "scope": "Target budget; measured p95 not established" + }, + "flakeHistory": { + "status": "soaking", + "evidence": "Focused local run passed; repeated CI history not established." + }, + "redGreenEvidence": { + "status": "partial", + "evidence": "Headless startup oracle reproduced missing registrar before the startup fix and passed afterward. Policy tests cover candidate behavior." + }, + "performanceBudget": { + "required": false, + "evidence": "Lifecycle and policy coverage; asserts exact gateway send counts and zero remaining dispatch listeners after shutdown." + }, + "promotionCriteria": ["Collect repeated CI runs without unexplained failures."], + "knownGaps": [ + "Does not prove APNs silent background wakeup or actual operating-system idle transitions.", + "Rendererless agent/bell event generation remains outside the documented feature contract.", + "No live Windows or Linux policy evidence.", + "Mobile native presentation and dismissal integration belongs to the subsequent mobile PR." + ], + "demotionRule": "Keep experimental if lifecycle or policy assertions fail; do not weaken them to bypass platform delivery gaps." + }, + { + "id": "agent-session.completed-turn-duration", + "title": "Completed turn duration survives client recovery and history pagination", + "maturity": "experimental", + "protection": "partial", + "owner": "agent-session-runtime", + "layer": "shared-and-renderer-unit", + "surfaces": ["native chat", "structured agent history", "structured agent subscriptions"], + "platforms": ["macos", "linux", "windows"], + "providers": ["local", "ssh", "remote-runtime", "mobile"], + "coveredPlatforms": ["macos"], + "coveredProviders": ["local"], + "coverageNotes": "Deterministic provider, journal recovery, RPC projection and shared renderer tests run on macOS. Shared logic serves desktop and mobile; these tests do not prove native mobile or real SSH transport. Local/daemon PTY input, geometry and WSL process launch are unaffected.", + "motivatingLinks": ["https://github.com/stablyai/orca/pull/19695"], + "invariant": "An observed completed turn retains the same host-recorded duration and initiating user-message attribution through client reload, settlement retry and loading older history; older clients receive the compatible item carrier, and a host-recorded unknown duration suppresses local estimates.", + "oracle": "Render a host-settled duration without observing completion locally; preserve an exit at 2000 through failed settlement at 60000 and later retry; retain a seven-second older turn beyond 256 submitted turns; prune aliases with removed user items and preserve alias identity during item-only streaming; render no completed duration for an unverifiable turn both before and after client reload.", + "commands": [ + "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/native-chat-unverifiable-turn-status.test.ts src/main/codex/codex-requested-close-turn-timing.test.ts src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts src/shared/structured-agent-session-turn-timing-retention.test.ts src/shared/structured-agent-session-turn-timing.test.ts src/main/codex/codex-structured-journal-translation-turn-lifecycle.test.ts src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.test.ts src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx" + ], + "testFiles": [ + "src/shared/native-chat-unverifiable-turn-status.test.ts", + "src/main/codex/codex-requested-close-turn-timing.test.ts", + "src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts", + "src/shared/structured-agent-session-turn-timing-retention.test.ts", + "src/shared/structured-agent-session-turn-timing.test.ts", + "src/main/codex/codex-structured-journal-translation-turn-lifecycle.test.ts", + "src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.test.ts", + "src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx" + ], + "assertionRefs": [ + { + "file": "src/shared/native-chat-unverifiable-turn-status.test.ts", + "assertions": [ + "does not convert a running turn to local Worked for after unverifiable recovery (start %s)" + ] + }, + { + "file": "src/main/codex/codex-requested-close-turn-timing.test.ts", + "assertions": ["keeps the first exit receipt when retry requestedClose=%s"] + }, + { + "file": "src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts", + "assertions": ["keeps exit receipt %s on retry"] + }, + { + "file": "src/shared/structured-agent-session-turn-timing-retention.test.ts", + "assertions": [ + "keeps an older page duration after the recent submission budget fills", + "drops an old alias once rewind removes its user item" + ] + }, + { + "file": "src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx", + "assertions": ["renders a host-settled duration without ever clocking the turn locally"] + } + ], + "evidenceRuns": [ + { + "date": "2026-09-09", + "runner": "local", + "platform": "macos", + "command": "ORCA_BACKGROUND_LAUNCH=1 pnpm test src/shared/native-chat-unverifiable-turn-status.test.ts src/main/codex/codex-requested-close-turn-timing.test.ts src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts src/shared/structured-agent-session-turn-timing-retention.test.ts src/shared/structured-agent-session-turn-timing.test.ts src/main/codex/codex-structured-journal-translation-turn-lifecycle.test.ts src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.test.ts src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx", + "result": "passed", + "durationSeconds": 4.65, + "summary": "58 tests passed across eight files at source commit 5acb404bf244cdc0bd3643c4e79beb189ce9a0cd; the following manifest-only commit records this run without changing tested source." + } + ], + "runtimeBudget": { + "p95Seconds": 30, + "scope": "Deterministic focused tests without an app launch; initial target budget pending CI soak." + }, + "flakeHistory": { + "status": "not-started", + "evidence": "No CI soak history for this combined gate." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "Before fixes, failed settlement stored 60000 instead of observed 2000 and older-page duration was undefined instead of seven seconds after 257 submissions. Mounted unverifiable recovery also displayed a local 59 seconds versus no duration after reload, including an older-host row without a start; the focused regressions pass after fixes." + }, + "performanceBudget": { + "required": true, + "evidence": "No polling, process fanout or new timer. Alias retention follows loaded user items plus 256 recent records; lookup uses a Set, removal/reset prunes aliases, and item-only stream batches retain the submission array identity." + }, + "knownGaps": [ + "This deterministic gate does not launch Electron or prove a full app restart; exact-head manual screenshots are separate PR evidence.", + "Native mobile, live SSH/relay transport and Windows/Linux execution are not exercised by this command.", + "Accumulated-history wall-clock profiling and CI soak remain unmeasured." + ], + "promotionCriteria": [ + "Complete CI soak without unexplained flakes and retain the value-based retry, pagination and rendering oracles." + ], + "demotionRule": "Keep experimental until soak and platform evidence support promotion; do not relax value or identity assertions to hide failures." + }, + { "id": "agent-session.history-forward-read-budget", "title": "Journal catch-up reads only the next page and one lookahead row", diff --git a/config/scripts/capture-agent-pty-transcript.mjs b/config/scripts/capture-agent-pty-transcript.mjs new file mode 100644 index 00000000000..a60d0adbdd5 --- /dev/null +++ b/config/scripts/capture-agent-pty-transcript.mjs @@ -0,0 +1,283 @@ +/** + * Records a live agent CLI session through a real PTY into a test fixture, bytes intact. + * + * Why a PTY and not `agy | tee`: a pipe is not a terminal, so the CLI renders its + * non-interactive path — no alternate screen, no caret, no dialogs. The detector under + * test only ever sees the PTY shape, so that is the only shape worth capturing. + * + * Nothing here strips escapes, folds CRs, or rewraps lines: the transcript is written + * exactly as the terminal received it. See docs/reference/agent-pty-transcript-capture.md. + */ +import { createWriteStream, mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { dirname, join, resolve } from 'node:path' +import { pathToFileURL } from 'node:url' +import { + formatFindings, + redactTranscript, + scanTranscriptForSecrets +} from './pty-transcript-secret-scan.mjs' + +const REPO_ROOT = resolve(import.meta.dirname, '..', '..') +const FIXTURE_DIR = join(REPO_ROOT, 'src', 'main', 'runtime', '__fixtures__') +const STOP_KEY = 0x1d // Ctrl-], consumed by the recorder and never forwarded to the agent. +const NAME_RE = /^[a-z0-9][a-z0-9-]*$/ + +const USAGE = `Capture a raw agent PTY transcript into src/main/runtime/__fixtures__/. + + node config/scripts/capture-agent-pty-transcript.mjs --name [options] -- [args...] + node config/scripts/capture-agent-pty-transcript.mjs --scan [--redact] + +Options + --name Output fixture name, e.g. antigravity-ready-personal-non-gemini + --out Write somewhere other than the fixture directory + --cols --rows Pin the PTY size (default: this terminal's size, else 120x40) + --duration Stop unattended after N seconds + --send ":" Type into the PTY at (repeatable; \\r \\n \\t \\e escapes) + --note "" Recorded in the .meta.json sidecar + --scan Scan existing transcripts for identifiers/credentials and exit + --redact With --scan: rewrite each finding as a same-length placeholder + +Press Ctrl-] to end a capture. That key is consumed here, so the agent keeps whatever +dialog it is showing — which is the only way to capture a dialog that owns the screen.` + +function parseArgs(argv) { + const options = { cols: null, rows: null, duration: null, scan: [], sends: [], redact: false } + const command = [] + let cursor = 0 + let afterSeparator = false + while (cursor < argv.length) { + const arg = argv[cursor] + if (afterSeparator) { + command.push(arg) + cursor += 1 + continue + } + if (arg === '--') { + afterSeparator = true + } else if (arg === '--redact') { + options.redact = true + } else if (arg === '--help' || arg === '-h') { + options.help = true + } else if (arg === '--scan') { + while (cursor + 1 < argv.length && !argv[cursor + 1].startsWith('--')) { + cursor += 1 + options.scan.push(argv[cursor]) + } + } else if (arg === '--send') { + cursor += 1 + options.sends.push(parseSend(argv[cursor])) + } else if (arg.startsWith('--')) { + const key = arg.slice(2) + cursor += 1 + options[key] = argv[cursor] + } + cursor += 1 + } + for (const key of ['cols', 'rows', 'duration']) { + options[key] = options[key] == null ? null : Number(options[key]) + } + return { options, command } +} + +// String.fromCharCode, not a literal: the formatter rewrites an escape sequence into a raw +// control byte in source, which is unreadable and survives badly in diffs. +const ESC = String.fromCharCode(27) +const SEND_ESCAPES = { r: '\r', n: '\n', t: '\t', e: ESC, '\\': '\\' } + +/** `":"` — a keystroke to deliver at a fixed offset, for an unattended dialog capture. */ +function parseSend(value) { + const separator = String(value ?? '').indexOf(':') + if (separator === -1) { + throw new Error(`--send expects ":", got ${String(value)}`) + } + const atMs = Number(value.slice(0, separator)) + if (!Number.isFinite(atMs)) { + throw new Error( + `--send delay must be a number of milliseconds, got ${value.slice(0, separator)}` + ) + } + const text = value + .slice(separator + 1) + .replace(/\\(.)/g, (whole, code) => SEND_ESCAPES[code] ?? whole) + return { atMs, text } +} + +function runScan(files, redact) { + let failed = false + for (const file of files) { + const path = resolve(file) + const text = readFileSync(path, 'utf8') + if (redact) { + const { text: redacted, redacted: count } = redactTranscript(text) + writeFileSync(path, redacted) + console.log(`${file}: redacted ${count} span(s) in place, same length each.`) + continue + } + const findings = scanTranscriptForSecrets(text) + console.log(formatFindings(file, findings)) + failed ||= findings.length > 0 + } + return failed ? 1 : 0 +} + +function resolveSpawn(command) { + // node-pty cannot run a .cmd/.bat shim directly on Windows; those need cmd.exe. + if (process.platform === 'win32' && /\.(cmd|bat)$/i.test(command[0])) { + return { file: 'cmd.exe', args: ['/c', `"${command[0]}"`, ...command.slice(1)] } + } + return { file: command[0], args: command.slice(1) } +} + +async function runCapture(options, command) { + const name = options.name + if (typeof name === 'string' && !NAME_RE.test(name)) { + console.error(`--name must be lowercase kebab-case; got ${name}`) + return 2 + } + const outPath = options.out ? resolve(options.out) : join(FIXTURE_DIR, `${name}.txt`) + mkdirSync(dirname(outPath), { recursive: true }) + + const pty = await import('node-pty').catch((error) => { + console.error( + `node-pty failed to load. Build it for plain node first: + node config/scripts/ensure-native-runtime.mjs --runtime=node +${String(error)}` + ) + return null + }) + if (pty === null) { + return 2 + } + + const cols = options.cols ?? process.stdout.columns ?? 120 + const rows = options.rows ?? process.stdout.rows ?? 40 + const { file, args } = resolveSpawn(command) + const term = pty.spawn(file, args, { + name: 'xterm-256color', + cols, + rows, + cwd: process.cwd(), + env: { ...process.env, TERM: 'xterm-256color' }, + encoding: null + }) + + const sink = createWriteStream(outPath) + let recording = true + term.onData((chunk) => { + const bytes = typeof chunk === 'string' ? Buffer.from(chunk, 'utf8') : chunk + // Why recording stops before the kill: an agent repaints an idle frame on its way out, so + // a transcript that keeps writing through shutdown ends on that frame instead of on the + // state you stopped to capture. A mid-turn or dialog capture cannot survive that. + if (recording) { + sink.write(bytes) + } + process.stdout.write(bytes) + }) + + const wasRaw = process.stdin.isTTY === true && process.stdin.isRaw === true + if (process.stdin.isTTY) { + process.stdin.setRawMode(true) + } + process.stdin.resume() + let stopping = false + const stop = () => { + if (stopping) { + return + } + stopping = true + recording = false + try { + term.kill() + } catch { + // The agent may have exited on its own; the transcript is already on disk. + } + } + process.stdin.on('data', (chunk) => { + if (chunk.includes(STOP_KEY)) { + stop() + return + } + term.write(chunk.toString('binary')) + }) + // Why scripted input: a dialog capture has to be driven, and CI (or an agent) has no TTY to + // type into. The keystrokes ride the same PTY a human's would, so the capture is unchanged. + const sendTimers = options.sends.map((send) => setTimeout(() => term.write(send.text), send.atMs)) + const durationTimer = options.duration === null ? null : setTimeout(stop, options.duration * 1000) + + const exitCode = await new Promise((resolveExit) => { + term.onExit(({ exitCode: code }) => resolveExit(code ?? 0)) + }) + for (const timer of sendTimers) { + clearTimeout(timer) + } + if (durationTimer !== null) { + clearTimeout(durationTimer) + } + if (process.stdin.isTTY) { + process.stdin.setRawMode(wasRaw) + } + process.stdin.pause() + await new Promise((done) => sink.end(done)) + + writeMeta(outPath, { command, cols, rows, note: options.note ?? null, exitCode }) + const findings = scanTranscriptForSecrets(readFileSync(outPath, 'utf8')) + console.log(`\nTranscript: ${outPath}`) + console.log(formatFindings('scrub check', findings)) + if (findings.length > 0) { + console.log( + `Scrub with: + node config/scripts/capture-agent-pty-transcript.mjs --scan ${outPath} --redact` + ) + } + return 0 +} + +function writeMeta(outPath, details) { + const metaPath = outPath.replace(/\.txt$/, '.meta.json') + writeFileSync( + metaPath, + `${JSON.stringify( + { + capturedAt: new Date().toISOString(), + platform: process.platform, + command: details.command, + cols: details.cols, + rows: details.rows, + note: details.note, + exitCode: details.exitCode + }, + null, + 2 + )}\n` + ) +} + +async function main() { + const { options, command } = parseArgs(process.argv.slice(2)) + if (options.help === true) { + console.log(USAGE) + return 0 + } + if (options.scan.length > 0) { + return runScan(options.scan, options.redact) + } + if (command.length === 0 || (options.name === undefined && options.out === undefined)) { + console.error(USAGE) + return 2 + } + return runCapture(options, command) +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + main().then( + (code) => { + process.exitCode = code + }, + (error) => { + console.error(error) + process.exitCode = 1 + } + ) +} + +export { parseArgs, resolveSpawn } diff --git a/config/scripts/electron-builder-runtime-resources.test.mjs b/config/scripts/electron-builder-runtime-resources.test.mjs index 5a93ec12c25..45145572b80 100644 --- a/config/scripts/electron-builder-runtime-resources.test.mjs +++ b/config/scripts/electron-builder-runtime-resources.test.mjs @@ -2,7 +2,7 @@ import { readFileSync, readdirSync } from 'node:fs' import { cp, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from 'node:fs/promises' import { createRequire } from 'node:module' import { tmpdir } from 'node:os' -import { dirname, join, relative, resolve } from 'node:path' +import { delimiter, dirname, join, relative, resolve } from 'node:path' import { describe, expect, it } from 'vitest' const require = createRequire(import.meta.url) @@ -387,6 +387,72 @@ describe('packaged runtime resources', () => { } }) + it.skipIf(process.platform === 'win32')( + 'prunes non-target native packages before the Linux glibc gate', + async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-after-pack-prune-order-')) + const previousPath = process.env.PATH + try { + const appOutDir = join(root, 'linux-unpacked') + const resourcesDir = join(appOutDir, 'resources') + await cp( + join(process.cwd(), 'resources', 'plugins', 'launch'), + join(resourcesDir, 'plugins', 'launch'), + { recursive: true } + ) + + const unpackedMainDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'main') + await mkdir(unpackedMainDir, { recursive: true }) + await writeFile(join(unpackedMainDir, 'daemon-entry.js'), '', 'utf8') + await writeFile( + join(resourcesDir, 'app.asar.unpacked', 'out', 'package.json'), + `${JSON.stringify({ name: 'orca-compiled-output', type: 'commonjs', private: true })}\n`, + 'utf8' + ) + + const unpackedCliDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'cli') + await mkdir(join(unpackedCliDir, 'handlers'), { recursive: true }) + await writeFile(join(unpackedCliDir, 'handlers', 'skills.js'), '', 'utf8') + await writeFile(join(unpackedCliDir, 'index.js'), '', 'utf8') + + const target = + process.arch === 'x64' + ? { electronArch: 3, machine: 0xb7, nonTarget: 'x64' } + : { electronArch: 1, machine: 0x3e, nonTarget: 'arm64' } + const wrongArchPackage = join( + resourcesDir, + 'node_modules', + '@parcel', + `watcher-linux-${target.nonTarget}-glibc` + ) + await mkdir(wrongArchPackage, { recursive: true }) + const wrongArchElf = Buffer.alloc(20) + wrongArchElf.set([0x7f, 0x45, 0x4c, 0x46]) + wrongArchElf[5] = 1 + wrongArchElf.writeUInt16LE(target.machine, 18) + await writeFile(join(wrongArchPackage, 'watcher.node'), wrongArchElf) + + const stubBinDir = join(root, 'bin') + await mkdir(stubBinDir) + await writeFile(join(stubBinDir, 'objdump'), '#!/bin/sh\nexit 0\n', { mode: 0o755 }) + process.env.PATH = `${stubBinDir}${delimiter}${previousPath ?? ''}` + + await expect( + electronBuilderConfig.afterPack({ + appOutDir, + electronPlatformName: 'linux', + arch: target.electronArch, + packager: { appInfo: { version: '9.9.9' } } + }) + ).resolves.toBeUndefined() + await expect(stat(wrongArchPackage)).rejects.toMatchObject({ code: 'ENOENT' }) + } finally { + process.env.PATH = previousPath + await rm(root, { recursive: true, force: true }) + } + } + ) + it.skipIf(process.platform === 'win32')( 'marks packaged Unix CLI launchers executable', async () => { diff --git a/config/scripts/generate-rpc-params-catalog.mjs b/config/scripts/generate-rpc-params-catalog.mjs new file mode 100644 index 00000000000..acf31aae6a1 --- /dev/null +++ b/config/scripts/generate-rpc-params-catalog.mjs @@ -0,0 +1,249 @@ +// Why: the host registry is the only place that binds a method name to its params +// schema. Reading it back — instead of hand-listing 600 methods — is what keeps the +// shared catalog and the dispatcher from drifting apart. +import { execFileSync } from 'node:child_process' +import { + existsSync, + globSync, + mkdirSync, + mkdtempSync, + readFileSync, + rmSync, + writeFileSync +} from 'node:fs' +import { createRequire } from 'node:module' +import path from 'node:path' +import process from 'node:process' +import * as esbuild from 'esbuild' +import { resolveOxcCliInvocation } from './oxc-cli-invocation.mjs' + +const REPO_ROOT = path.resolve(import.meta.dirname, '..', '..') +const SHARED_DIR = path.join(REPO_ROOT, 'src', 'shared') +const CONTRACT_DIR = path.join(SHARED_DIR, 'rpc-contract') +const RPC_DIR = path.join(REPO_ROOT, 'src', 'main', 'runtime', 'rpc') +const REGISTRY_ENTRY = path.join(RPC_DIR, 'methods', 'index.ts') +const OUTPUT_PATH = path.join(CONTRACT_DIR, 'rpc-params-catalog.generated.ts') + +// Why mkdirSync first: out/ is gitignored and absent on a fresh checkout, so +// mkdtempSync threw ENOENT and took `pnpm lint` down with it. Why not os.tmpdir(): +// the bundle keeps its node_modules deps external and oxfmt reads .oxfmtrc.json by +// walking up, so both scratch files have to sit under the repo to resolve at all. +function scratchDir(prefix) { + const root = path.join(REPO_ROOT, 'out') + mkdirSync(root, { recursive: true }) + return mkdtempSync(path.join(root, prefix)) +} + +const posix = (value) => value.split(path.sep).join('/') +const repoPath = (absolute) => posix(path.relative(REPO_ROOT, absolute)) + +// Every module the catalog may import from: the extracted params modules plus the +// pre-existing src/shared schemas the RPC methods already bind directly. +function indexableModules() { + const modules = new Set( + globSync('*.ts', { cwd: CONTRACT_DIR }).map((name) => path.join(CONTRACT_DIR, name)) + ) + modules.delete(OUTPUT_PATH) + for (const file of globSync('**/*.ts', { cwd: RPC_DIR })) { + if (file.endsWith('.test.ts')) { + continue + } + const source = readFileSync(path.join(RPC_DIR, file), 'utf8') + for (const [, specifier] of source.matchAll(/from\s+'(\.[^']+)'/g)) { + const resolved = `${path.resolve(path.dirname(path.join(RPC_DIR, file)), specifier)}.ts` + if (resolved.startsWith(`${SHARED_DIR}${path.sep}`) && existsSync(resolved)) { + modules.add(resolved) + } + } + } + return [...modules].sort() +} + +// Why: one bundle keeps the registry and the shared modules on the same module +// instances, so schema object identity is what maps a method to its export. +function loadRegistryAndSchemas(modules) { + const buildDir = scratchDir('rpc-params-catalog-') + try { + const entry = path.join(buildDir, 'entry.ts') + const importOf = (file) => JSON.stringify(posix(path.relative(buildDir, file))) + writeFileSync( + entry, + [ + `export { ALL_RPC_METHODS } from ${importOf(REGISTRY_ENTRY)}`, + 'export const SCHEMA_MODULES = {', + ...modules.map( + (file) => ` ${JSON.stringify(repoPath(file))}: require(${importOf(file)}),` + ), + '}' + ].join('\n') + ) + const outfile = path.join(buildDir, 'bundle.cjs') + esbuild.buildSync({ + entryPoints: [entry], + bundle: true, + platform: 'node', + format: 'cjs', + outfile, + logLevel: 'error', + packages: 'external' + }) + const loaded = createRequire(import.meta.url)(outfile) + return { methods: loaded.ALL_RPC_METHODS, schemaModules: loaded.SCHEMA_MODULES } + } finally { + rmSync(buildDir, { recursive: true, force: true }) + } +} + +// Why: schema objects are compared by identity, not by shape — two structurally +// identical schemas are still two different wire contracts. +function buildSchemaIndex(schemaModules) { + const index = new Map() + for (const [modulePath, moduleExports] of Object.entries(schemaModules)) { + for (const [exportName, value] of Object.entries(moduleExports)) { + if (!value || typeof value !== 'object' || typeof value.safeParse !== 'function') { + continue + } + if (index.has(value)) { + continue + } + index.set(value, { modulePath, exportName }) + } + } + return index +} + +function localNameFor(origin, taken) { + if (!taken.has(origin.exportName)) { + return origin.exportName + } + const hint = path + .basename(origin.modulePath, '.ts') + .split('-') + .map((part) => part.charAt(0).toUpperCase() + part.slice(1)) + .join('') + let candidate = `${origin.exportName}Of${hint}` + let suffix = 2 + while (taken.has(candidate)) { + candidate = `${origin.exportName}Of${hint}${suffix++}` + } + return candidate +} + +function render({ methods, schemaModules }) { + const index = buildSchemaIndex(schemaModules) + const entries = [] + const uncataloged = [] + const imports = new Map() + const taken = new Set() + + for (const method of [...methods].sort((left, right) => (left.name < right.name ? -1 : 1))) { + if (method.params === null) { + entries.push(` '${method.name}': null`) + continue + } + const origin = index.get(method.params) + if (!origin) { + uncataloged.push(method.name) + continue + } + const key = `${origin.modulePath}#${origin.exportName}` + let local = imports.get(key) + if (!local) { + local = localNameFor(origin, taken) + taken.add(local) + imports.set(key, local) + } + entries.push(` '${method.name}': ${local}`) + } + + const byModule = new Map() + for (const [key, local] of imports) { + const [modulePath, exportName] = key.split('#') + if (!byModule.has(modulePath)) { + byModule.set(modulePath, []) + } + byModule.get(modulePath).push(local === exportName ? exportName : `${exportName} as ${local}`) + } + const importLines = [...byModule] + .sort(([left], [right]) => (left < right ? -1 : 1)) + .map(([modulePath, names]) => { + let specifier = posix(path.relative(CONTRACT_DIR, path.join(REPO_ROOT, modulePath))).replace( + /\.ts$/, + '' + ) + if (!specifier.startsWith('.')) { + specifier = `./${specifier}` + } + return `import { ${names.sort().join(', ')} } from '${specifier}'` + }) + + return `// GENERATED by config/scripts/generate-rpc-params-catalog.mjs. Do not edit; +// run \`pnpm run generate:rpc-params-catalog\`. +import type { z } from 'zod' +${importLines.join('\n')} + +// Why: the host parses params with these schemas, so a client that matches this map +// matches the dispatcher. Clients must import it for types only — parsing a params +// schema client-side runs the coercing transforms and rewrites the wire bytes. +export const RPC_PARAMS_BY_METHOD = { +${entries.join(',\n')} +} as const + +// Why: these methods bind a schema the shared contract cannot hold because its value +// graph reaches into src/main. Listing them keeps the gap visible instead of absent. +export const RPC_METHODS_WITHOUT_SHARED_PARAMS: readonly string[] = [ +${uncataloged.map((name) => ` '${name}'`).join(',\n')} +] + +export type RpcMethodName = keyof typeof RPC_PARAMS_BY_METHOD + +// Why: z.output is the post-parse shape the handler receives. z.input is not a +// send-side type here — requiredString is z.unknown().transform(...), so its input +// admits any value and loses optional/default semantics. +export type RpcParams = + (typeof RPC_PARAMS_BY_METHOD)[Method] extends z.ZodType + ? z.output<(typeof RPC_PARAMS_BY_METHOD)[Method]> + : void +` +} + +// Why: the drift gate compares bytes, so the generator must emit exactly what the +// formatter would produce or every run would look like drift. +function formatted(source) { + const buildDir = scratchDir('rpc-params-catalog-fmt-') + try { + const file = path.join(buildDir, 'rpc-params-catalog.generated.ts') + writeFileSync(file, source) + const { command, prefixArgs } = resolveOxcCliInvocation('oxfmt', 'oxfmt', REPO_ROOT) + execFileSync(command, [...prefixArgs, '--write', file], { + stdio: 'ignore', + windowsHide: true + }) + return readFileSync(file, 'utf8') + } finally { + rmSync(buildDir, { recursive: true, force: true }) + } +} + +function main() { + const check = process.argv.includes('--check') + const generated = formatted(render(loadRegistryAndSchemas(indexableModules()))) + const current = existsSync(OUTPUT_PATH) ? readFileSync(OUTPUT_PATH, 'utf8') : null + if (generated === current) { + if (!check) { + console.log(`rpc params catalog already up to date: ${repoPath(OUTPUT_PATH)}`) + } + return + } + if (check) { + console.error( + `${repoPath(OUTPUT_PATH)} is out of date. Run \`pnpm run generate:rpc-params-catalog\`.` + ) + process.exitCode = 1 + return + } + writeFileSync(OUTPUT_PATH, generated) + console.log(`wrote ${repoPath(OUTPUT_PATH)}`) +} + +main() diff --git a/config/scripts/orchestration-skill-guidance.test.mjs b/config/scripts/orchestration-skill-guidance.test.mjs index c15e3e93ea8..dee2fd75fb2 100644 --- a/config/scripts/orchestration-skill-guidance.test.mjs +++ b/config/scripts/orchestration-skill-guidance.test.mjs @@ -381,6 +381,9 @@ describe('owned orchestration references', () => { expect(reference).toContain(group) } expect(reference).toContain('Dispatch lifecycle messages never target groups') + expect(squash(reference)).toContain("means the live Dispatches of the sender's own Run.") + expect(squash(reference)).toContain('A sender bound to no Run is refused') + expect(squash(reference)).toContain('A Run group excludes its owning coordinator') expect(reference).toContain('gate-create --task ') expect(reference).toContain("Do not create a gate merely to answer a worker's `ask`") expect(reference).toContain('successful `send` proves durable enqueue') diff --git a/config/scripts/oxc-cli-invocation.mjs b/config/scripts/oxc-cli-invocation.mjs new file mode 100644 index 00000000000..4bf17b5c994 --- /dev/null +++ b/config/scripts/oxc-cli-invocation.mjs @@ -0,0 +1,23 @@ +import { createRequire } from 'node:module' +import path from 'node:path' +import process from 'node:process' + +// Why not `pnpm exec ` / `node_modules/.bin/.cmd`: both land on a Windows +// .cmd shim, and Node >= 20 refuses to spawn one without `shell: true` (the +// CVE-2024-27980 mitigation), so every gate that took that route died with EINVAL +// before doing any work. The oxc bins are plain Node scripts, so run them under this +// process's own node — no shim, no shell, no quoting question. +export function resolveOxcCliInvocation(packageName, binName, root = process.cwd()) { + const requireFromRoot = createRequire(path.join(root, 'package.json')) + // The oxc packages' "exports" hide ./bin, so read the manifest and walk to its bin entry. + const manifestPath = requireFromRoot.resolve(`${packageName}/package.json`) + const binField = requireFromRoot(`${packageName}/package.json`).bin + const binEntry = typeof binField === 'string' ? binField : binField?.[binName] + if (!binEntry) { + throw new Error(`${packageName} package.json declares no "${binName}" bin entry.`) + } + return { + command: process.execPath, + prefixArgs: [path.resolve(path.dirname(manifestPath), binEntry)] + } +} diff --git a/config/scripts/oxc-cli-invocation.test.mjs b/config/scripts/oxc-cli-invocation.test.mjs new file mode 100644 index 00000000000..5776580dbfe --- /dev/null +++ b/config/scripts/oxc-cli-invocation.test.mjs @@ -0,0 +1,40 @@ +import { spawnSync } from 'node:child_process' +import { existsSync } from 'node:fs' +import path from 'node:path' +import process from 'node:process' +import { describe, expect, it } from 'vitest' +import { resolveOxcCliInvocation } from './oxc-cli-invocation.mjs' + +const repoRoot = path.resolve(import.meta.dirname, '..', '..') + +describe('resolveOxcCliInvocation', () => { + it('runs oxfmt under this process node, never through a shim', () => { + const { command, prefixArgs } = resolveOxcCliInvocation('oxfmt', 'oxfmt', repoRoot) + + expect(command).toBe(process.execPath) + expect(prefixArgs).toHaveLength(1) + // The params-catalog generator spawned node_modules/.bin/oxfmt, which is a .cmd on + // Windows — Node >= 20 refuses it without shell:true and dies with EINVAL. + expect(prefixArgs[0]).not.toMatch(/\.(cmd|bat)$/i) + expect(existsSync(prefixArgs[0])).toBe(true) + }) + + it('spawns oxfmt without a shell', () => { + const { command, prefixArgs } = resolveOxcCliInvocation('oxfmt', 'oxfmt', repoRoot) + const result = spawnSync(command, [...prefixArgs, '--help'], { + cwd: repoRoot, + encoding: 'utf8', + shell: false, + windowsHide: true + }) + + expect(result.error).toBeUndefined() + expect(result.stdout).toContain('oxfmt') + }) + + it('names the package and bin it could not find', () => { + expect(() => resolveOxcCliInvocation('oxfmt', 'nope', repoRoot)).toThrow( + 'oxfmt package.json declares no "nope" bin entry.' + ) + }) +}) diff --git a/config/scripts/oxlint-cli-invocation.mjs b/config/scripts/oxlint-cli-invocation.mjs index 605aa33c686..92e6f55fb95 100644 --- a/config/scripts/oxlint-cli-invocation.mjs +++ b/config/scripts/oxlint-cli-invocation.mjs @@ -1,23 +1,6 @@ -import { createRequire } from 'node:module' -import path from 'node:path' import process from 'node:process' +import { resolveOxcCliInvocation } from './oxc-cli-invocation.mjs' -// Why not `pnpm exec oxlint` / `node_modules/.bin/oxlint.cmd`: both land on a -// Windows .cmd shim, and Node >= 20 refuses to spawn one without `shell: true` -// (the CVE-2024-27980 mitigation), so every lint gate died with EINVAL before -// linting anything. Oxlint's bin is a plain Node script, so run it under this -// process's own node — no shim, no shell, no quoting question. export function resolveOxlintInvocation(root = process.cwd()) { - const requireFromRoot = createRequire(path.join(root, 'package.json')) - // Oxlint's "exports" hides ./bin, so read the manifest and walk to its bin entry. - const manifestPath = requireFromRoot.resolve('oxlint/package.json') - const binField = requireFromRoot('oxlint/package.json').bin - const binEntry = typeof binField === 'string' ? binField : binField?.oxlint - if (!binEntry) { - throw new Error('oxlint package.json declares no "oxlint" bin entry.') - } - return { - command: process.execPath, - prefixArgs: [path.resolve(path.dirname(manifestPath), binEntry)] - } + return resolveOxcCliInvocation('oxlint', 'oxlint', root) } diff --git a/config/scripts/pty-transcript-secret-scan.mjs b/config/scripts/pty-transcript-secret-scan.mjs new file mode 100644 index 00000000000..1d93204ccda --- /dev/null +++ b/config/scripts/pty-transcript-secret-scan.mjs @@ -0,0 +1,135 @@ +// Finds account identifiers and credentials in a captured PTY transcript before it is committed. +import os from 'node:os' + +// Why same-length replacements: a transcript's value is its exact wrapping and column +// alignment. Shortening a redacted span reflows the screen and destroys the evidence. +const EMAIL_DOMAIN = '@example.com' +const PLACEHOLDER_UUID = '00000000-0000-4000-8000-000000000000' + +/** Ordered most-specific first; the first pattern to claim a span owns it. */ +function buildPatterns() { + const username = os.userInfo().username + const hostname = os.hostname() + const patterns = [ + { kind: 'jwt', re: /\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{4,}/g }, + { kind: 'google-api-key', re: /\bAIza[0-9A-Za-z_-]{20,}/g }, + { kind: 'google-refresh-token', re: /\b1\/\/[0-9A-Za-z_-]{20,}/g }, + { kind: 'vendor-key', re: /\b(?:sk-|ghp_|gho_|github_pat_|xoxb-|xoxp-)[A-Za-z0-9_-]{16,}/g }, + { kind: 'bearer-token', re: /\bBearer\s+[A-Za-z0-9._~+/=-]{16,}/gi }, + { kind: 'email', re: /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}/g }, + // Why a UUID counts: agy prints a resumable conversation id on exit, and installation and + // project ids look the same. They identify the operator's session, not just its shape. + { kind: 'uuid', re: /\b[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b/gi }, + { kind: 'opaque-token', re: /\b[A-Za-z0-9_-]{40,}\b/g } + ] + if (username.length >= 3) { + patterns.splice(5, 0, { kind: 'local-username', re: literalPattern(username) }) + } + if (hostname.length >= 3) { + patterns.splice(5, 0, { kind: 'local-hostname', re: literalPattern(hostname) }) + } + return patterns +} + +function literalPattern(value) { + return new RegExp(value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'), 'g') +} + +/** + * @param {string} text raw transcript, escapes intact + * @returns {{kind: string, line: number, column: number, index: number, match: string}[]} + */ +export function scanTranscriptForSecrets(text) { + const claimed = [] + const findings = [] + for (const { kind, re } of buildPatterns()) { + re.lastIndex = 0 + let match = re.exec(text) + while (match !== null) { + const start = match.index + const end = start + match[0].length + if (!claimed.some(([from, to]) => start < to && end > from)) { + claimed.push([start, end]) + if (!isAlreadyScrubbed(kind, match[0])) { + findings.push({ kind, index: start, match: match[0], ...locate(text, start) }) + } + } + match = re.exec(text) + } + } + return findings.sort((left, right) => left.index - right.index) +} + +// Why: a scrubbed fixture must verify clean, so this scanner has to recognise its own +// placeholders — otherwise "prove it's gone" can never pass and the check gets ignored. +const PLACEHOLDER_DOMAIN_RE = /@(?:example\.(?:com|org|net)|localhost)$/i + +function isAlreadyScrubbed(kind, match) { + if (kind === 'email') { + return PLACEHOLDER_DOMAIN_RE.test(match) + } + if (kind === 'uuid') { + return match.toLowerCase() === PLACEHOLDER_UUID + } + return /^(.)\1*$/.test(match) +} + +function locate(text, index) { + let line = 1 + let lineStart = 0 + for (let cursor = 0; cursor < index; cursor += 1) { + if (text.charCodeAt(cursor) === 10) { + line += 1 + lineStart = cursor + 1 + } + } + return { line, column: index - lineStart + 1 } +} + +/** Same-length stand-in so redaction cannot reflow the captured screen. */ +export function placeholderFor(kind, length) { + if (kind === 'uuid' && length === PLACEHOLDER_UUID.length) { + return PLACEHOLDER_UUID + } + if (kind === 'email' && length > EMAIL_DOMAIN.length) { + return 'u'.repeat(length - EMAIL_DOMAIN.length) + EMAIL_DOMAIN + } + return kind === 'local-username' || kind === 'local-hostname' + ? 'x'.repeat(length) + : 'X'.repeat(length) +} + +/** @returns {{text: string, redacted: number}} */ +export function redactTranscript(text) { + const findings = scanTranscriptForSecrets(text) + let out = '' + let cursor = 0 + for (const finding of findings) { + out += text.slice(cursor, finding.index) + out += placeholderFor(finding.kind, finding.match.length) + cursor = finding.index + finding.match.length + } + return { text: out + text.slice(cursor), redacted: findings.length } +} + +export function formatFindings(label, findings) { + if (findings.length === 0) { + return `${label}: clean — no account identifier or credential shapes found.` + } + const rows = findings.map( + (finding) => ` ${finding.line}:${finding.column} ${finding.kind} ${preview(finding.match)}` + ) + return [`${label}: ${findings.length} finding(s) — scrub before committing.`, ...rows].join('\n') +} + +// Why a codepoint test and not a character class: a control-byte range written as an escape is +// folded back into raw 0x00-0x1f bytes by the formatter, which makes this file binary to the VCS +// and leaves the one file gating real PTY data into history unreviewable in a diff. +function preview(value) { + const head = value.length <= 24 ? value : `${value.slice(0, 21)}...` + let printable = '' + for (const char of head) { + printable += (char.codePointAt(0) ?? 0) < 0x20 ? '?' : char + } + return printable +} diff --git a/config/scripts/pty-transcript-secret-scan.test.mjs b/config/scripts/pty-transcript-secret-scan.test.mjs new file mode 100644 index 00000000000..2d3cd894da0 --- /dev/null +++ b/config/scripts/pty-transcript-secret-scan.test.mjs @@ -0,0 +1,133 @@ +// The scrub gate is the only thing standing between a live agent transcript and a +// committed account identifier, so it is pinned on the shapes those transcripts carry. +import { readdirSync, readFileSync } from 'node:fs' +import os from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { + formatFindings, + placeholderFor, + redactTranscript, + scanTranscriptForSecrets +} from './pty-transcript-secret-scan.mjs' +import { parseArgs, resolveSpawn } from './capture-agent-pty-transcript.mjs' + +describe('pty transcript secret scan', () => { + it('finds the account row of a ready screen', () => { + const findings = scanTranscriptForSecrets('Antigravity CLI 1.1.17\njin.woo@acme.dev (Business)') + expect(findings).toHaveLength(1) + expect(findings[0]).toMatchObject({ kind: 'email', line: 2, column: 1 }) + }) + + it('finds credentials an agent may echo while signing in', () => { + const kinds = scanTranscriptForSecrets( + [ + 'token: eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dBjftJeZ4CVP', + 'key: AIzaSyA1234567890abcdefghijklmnopqrstu', + 'refresh: 1//0gLm34XyZabcdefghijklmnopqrstuvwx', + 'Authorization: Bearer abcdefghijklmnopqrstuvwxyz012345' + ].join('\n') + ).map((finding) => finding.kind) + expect(kinds).toEqual(['jwt', 'google-api-key', 'google-refresh-token', 'bearer-token']) + }) + + it('flags this machine’s own username, which a prompt line leaks', () => { + const username = os.userInfo().username + const findings = scanTranscriptForSecrets(`~/Users/${username}/orca/repo\n> `) + expect(findings.some((finding) => finding.kind === 'local-username')).toBe(true) + }) + + it('finds the resumable conversation id agy prints on exit', () => { + const findings = scanTranscriptForSecrets( + 'Resume with -c (or command below):\nagy --conversation=26dc1986-9eec-456a-a534-d93e5c1076c2' + ) + expect(findings).toHaveLength(1) + expect(findings[0].kind).toBe('uuid') + expect(placeholderFor('uuid', findings[0].match.length)).toMatch( + /^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[0-9a-f]{4}-[0-9a-f]{12}$/ + ) + }) + + it('reports a clean transcript as clean', () => { + const findings = scanTranscriptForSecrets('Antigravity CLI 1.1.17\nSonnet 4.6 (High)\n> ') + expect(findings).toEqual([]) + expect(formatFindings('fixture', findings)).toContain('clean') + }) + + it('claims a span once, so a token inside an email is not double-reported', () => { + const findings = scanTranscriptForSecrets('longlivedaccountname@corp.internal') + expect(findings).toHaveLength(1) + }) + + it('passes a fixture that is already scrubbed, so "prove it is gone" can succeed', () => { + const scrubbed = `uuuu@example.com\n${'X'.repeat(44)}` + expect(scanTranscriptForSecrets(scrubbed)).toEqual([]) + }) +}) + +describe('redaction', () => { + it('replaces every finding with the same number of characters', () => { + // Why length matters: the fixture's value is its exact wrapping. A shorter + // replacement reflows the screen and invalidates the capture. + const text = 'Antigravity CLI 1.1.17\njin.woo@acme.dev (Antigravity Business)\n> ' + const { text: redacted, redacted: count } = redactTranscript(text) + expect(count).toBe(1) + expect(redacted).toHaveLength(text.length) + expect(redacted).not.toContain('jin.woo@acme.dev') + expect(scanTranscriptForSecrets(redacted)).toEqual([]) + expect(redactTranscript(redacted).redacted).toBe(0) + }) + + it('keeps a redacted email shaped like an email', () => { + expect(placeholderFor('email', 'a@b.example.com'.length)).toMatch(/^u+@example\.com$/) + }) + + it('leaves the rest of the screen byte-for-byte untouched', () => { + const text = 'line one\nuser@corp.io\nline three' + expect(redactTranscript(text).text.split('\n')[2]).toBe('line three') + }) +}) + +describe('committed transcripts', () => { + // Why in CI and not just in the recorder: a transcript is committed once and read forever. + // The capture-time warning is skippable; this is not. + const fixtureDir = join(import.meta.dirname, '..', '..', 'src', 'main', 'runtime', '__fixtures__') + const transcripts = readdirSync(fixtureDir).filter((entry) => entry.endsWith('.txt')) + + it.each(transcripts)('%s carries no account identifier or credential', (name) => { + const findings = scanTranscriptForSecrets(readFileSync(join(fixtureDir, name), 'utf8')) + expect(formatFindings(name, findings)).toContain('clean') + }) +}) + +describe('capture argv', () => { + it('splits recorder options from the agent command', () => { + const { options, command } = parseArgs([ + '--name', + 'antigravity-ready-personal-non-gemini', + '--cols', + '120', + '--', + 'agy', + '--model', + 'sonnet' + ]) + expect(options.name).toBe('antigravity-ready-personal-non-gemini') + expect(options.cols).toBe(120) + expect(command).toEqual(['agy', '--model', 'sonnet']) + }) + + it('collects a multi-file scan list', () => { + const { options } = parseArgs(['--scan', 'a.txt', 'b.txt', '--redact']) + expect(options.scan).toEqual(['a.txt', 'b.txt']) + expect(options.redact).toBe(true) + }) + + it('routes a Windows shim through cmd.exe, which node-pty cannot spawn directly', () => { + expect(resolveSpawn(['agy.cmd', '--model', 'sonnet'])).toEqual( + process.platform === 'win32' + ? { file: 'cmd.exe', args: ['/c', '"agy.cmd"', '--model', 'sonnet'] } + : { file: 'agy.cmd', args: ['--model', 'sonnet'] } + ) + }) +}) diff --git a/config/scripts/session-search-retention-benchmark.ts b/config/scripts/session-search-retention-benchmark.ts new file mode 100644 index 00000000000..095eb5f29b0 --- /dev/null +++ b/config/scripts/session-search-retention-benchmark.ts @@ -0,0 +1,148 @@ +import assert from 'node:assert/strict' +import { mkdtemp, rm, stat } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { setImmediate as yieldToEventLoop } from 'node:timers/promises' +import { + syntheticCandidate, + syntheticSession, + userMessages +} from '../../src/main/ai-vault-search/session-search-index-test-fixture' +import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' +import SyncDatabase from '../../src/main/sqlite/sync-database' + +// Bundle with esbuild --bundle --platform=node, then run on the host under test. +// Every mode seeds through SessionSearchStore so the three arms are comparable; +// only `whole-file` leaves the shipped path, because it is the baseline the +// batched purge exists to replace. + +const ROWS = 60_000 + +/** The purge yields with `setImmediate` between chunks, so a peer chain samples each gap. */ +async function sampleLoopStalls(running: () => boolean, intervals: number[]): Promise { + let previous = performance.now() + while (running()) { + await yieldToEventLoop() + const now = performance.now() + intervals.push(now - previous) + previous = now + } +} + +/** What a search would still return: rows whose session row is still there. */ +function visibleRows(db: SyncDatabase): number { + return ( + db + .prepare(`SELECT count(*) AS n FROM messages m JOIN sessions s ON s.id = m.session_row_id`) + .get() as { n: number } + ).n +} + +const root = await mkdtemp(join(tmpdir(), 'orca-search-retention-bench-')) +try { + for (const mode of ['whole-file', 'batched', 'batched-pinned-reader']) { + const path = join(root, `${mode}.sqlite`) + const errors: unknown[] = [] + const store = new SessionSearchStore(path, (error) => errors.push(error)) + let reader: SyncDatabase | null = null + try { + const write = store.beginWrite(syntheticCandidate(), 'replace', 0)! + for (const message of userMessages( + 'synthetic benchmark needle repeated context for a representative coding conversation with commands and paths src/example.ts', + ROWS + )) { + write.add(message) + } + assert.equal( + write.commit({ + session: syntheticSession(), + byteOffset: 4096, + incomplete: false + }), + true + ) + assert.deepEqual(errors, []) + // Truncating first is what makes walBytes below the purge's own growth. + const checkpoint = new SyncDatabase(path) + checkpoint.pragma('wal_checkpoint(TRUNCATE)') + checkpoint.close() + if (mode === 'batched-pinned-reader') { + reader = new SyncDatabase(path, { readonly: true }) + reader.exec('BEGIN') + reader.prepare('SELECT count(*) FROM messages').get() + } + const probe = new SyncDatabase(path, { readonly: true }) + const intervals: number[] = [] + const started = performance.now() + if (mode === 'whole-file') { + const raw = new SyncDatabase(path) + try { + raw.exec('BEGIN IMMEDIATE') + const ids = raw.prepare('SELECT id FROM messages').all() as { + id: number + }[] + for (const { id } of ids) { + raw.prepare('DELETE FROM messages_fts WHERE rowid=?').run(id) + } + raw.exec('DELETE FROM messages; DELETE FROM sessions; DELETE FROM files; COMMIT') + } finally { + raw.close() + } + intervals.push(performance.now() - started) + } else { + let purging = true + const purge = store.purgeOlderThan(Date.now() + 60_000) + // Hiding is immediate: cutting the session loose from its file is the + // first transaction, so a read one turn in already sees nothing, long + // before the rows are gone. + const hiddenEarly = yieldToEventLoop().then(() => visibleRows(probe)) + const sampler = sampleLoopStalls(() => purging, intervals) + await purge + purging = false + await sampler + assert.equal(await hiddenEarly, 0) + assert.deepEqual(errors, []) + } + const wallMs = performance.now() - started + probe.close() + reader?.exec('COMMIT') + reader?.close() + reader = null + const after = new SyncDatabase(path, { readonly: true }) + try { + for (const table of ['messages_fts']) { + assert.equal( + ( + after.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { + n: number + } + ).n, + 0 + ) + } + } finally { + after.close() + } + const walBytes = (await stat(`${path}-wal`)).size + intervals.sort((a, b) => a - b) + console.log( + JSON.stringify({ + mode, + platform: process.platform, + node: process.version, + rows: ROWS, + wallMs: Math.round(wallMs), + samples: intervals.length, + maxStepMs: Math.round(intervals.at(-1) ?? 0), + p95StepMs: Math.round(intervals[Math.floor(intervals.length * 0.95)] ?? 0), + walBytes + }) + ) + } finally { + reader?.close() + store.close() + } + } +} finally { + await rm(root, { recursive: true, force: true }) +} diff --git a/config/scripts/session-search-write-benchmark.ts b/config/scripts/session-search-write-benchmark.ts new file mode 100644 index 00000000000..db09275cd98 --- /dev/null +++ b/config/scripts/session-search-write-benchmark.ts @@ -0,0 +1,266 @@ +import assert from 'node:assert/strict' +import { rm, stat } from 'node:fs/promises' +import { join } from 'node:path' +import { setImmediate as yieldToEventLoop } from 'node:timers/promises' +import { + createSessionParseStats, + parseAgentSessionFileCached, + resetSessionParseCacheForTests +} from '../../src/main/ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../../src/main/ai-vault/session-transcript-consumers' +import { requestWholeTranscriptRead } from '../../src/main/ai-vault/session-transcript-reader' +import { registerSessionSearchIndexConsumer } from '../../src/main/ai-vault-search/session-search-index-consumer' +import { SessionSearchStore } from '../../src/main/ai-vault-search/session-search-store' +import { writeSyntheticTranscriptCorpus } from '../../src/main/ai-vault-search/session-search-synthetic-corpus' +import { sessionCandidate } from '../../src/main/ai-vault-search/session-search-transcript-fixtures' +import SyncDatabase from '../../src/main/sqlite/sync-database' + +// Measures the real transcript reader and search store over a synthetic corpus. +// Never point this at a real transcript tree. + +/** + * How long the longest single transaction held the process. + * + * With one transaction per file that is the whole stall a file costs, so it is + * the number the commit ceiling exists to bound. Measured by wrapping `exec`, + * because the writer's transactions are the only ones this benchmark runs. + */ +function recordTransactionDurations(durations: number[]): () => void { + const exec = SyncDatabase.prototype.exec + let started = 0 + SyncDatabase.prototype.exec = function (this: SyncDatabase, sql: string): void { + if (sql === 'BEGIN IMMEDIATE') { + started = performance.now() + } + exec.call(this, sql) + if (sql === 'COMMIT' && started > 0) { + durations.push(performance.now() - started) + started = 0 + } + } + return () => { + SyncDatabase.prototype.exec = exec + } +} + +/** The writer commits synchronously, so a peer chain samples the gap each read leaves. */ +async function sampleLoopStalls(running: () => boolean, stalls: number[]): Promise { + let previous = performance.now() + while (running()) { + await yieldToEventLoop() + const now = performance.now() + stalls.push(now - previous) + previous = now + } +} + +function tableBytes(db: SyncDatabase): Record { + const rows = db.prepare('SELECT name, sum(pgsize) AS bytes FROM dbstat GROUP BY name').all() as { + name: string + bytes: number + }[] + const group = (prefix: string): number => + rows + .filter((row) => row.name === prefix || row.name.startsWith(`${prefix}_`)) + .reduce((sum, row) => sum + row.bytes, 0) + return { + messagesFts: group('messages_fts'), + messages: group('messages') - group('messages_fts'), + sessions: group('sessions'), + total: rows.reduce((sum, row) => sum + row.bytes, 0) + } +} + +function assertIndexedMessages(db: SyncDatabase, expected: number): number { + const { n } = db + .prepare('SELECT count(*) AS n FROM messages m JOIN sessions s ON s.id = m.session_row_id') + .get() as { n: number } + assert.equal(n, expected, 'indexed message count') + return n +} + +async function checkpointedFileBytes(db: SyncDatabase, path: string): Promise { + // Flush committed WAL pages before reporting the final database footprint. + const [checkpoint] = db.pragma('wal_checkpoint(TRUNCATE)') as { busy: number }[] + assert.equal(checkpoint?.busy, 0, 'storage measurement requires a completed checkpoint') + return (await stat(path)).size +} + +// The default corpus puts tool output at about half the message text; set this +// far higher to price the tool-row cap against the real 80-97 % band. +const toolResultWords = Number(process.env.ORCA_SEARCH_BENCH_TOOL_WORDS ?? 200) +const corpus = await writeSyntheticTranscriptCorpus({ toolResultWords }) +const indexPath = join(corpus.root, 'index.sqlite') +try { + const errors: unknown[] = [] + const store = new SessionSearchStore(indexPath, (error) => errors.push(error)) + const unregister = registerSessionSearchIndexConsumer(store) + const stalls: number[] = [] + const transactions: number[] = [] + const restoreExec = recordTransactionDurations(transactions) + let indexing = true + try { + const stats = createSessionParseStats() + const started = performance.now() + const sampler = sampleLoopStalls(() => indexing, stalls) + for (const path of corpus.files) { + await parseAgentSessionFileCached( + await sessionCandidate('claude', path), + process.platform, + stats + ) + } + indexing = false + await sampler + restoreExec() + const rebuildMs = performance.now() - started + assert.deepEqual(errors, []) + + const reader = new SyncDatabase(indexPath, { readonly: true }) + try { + const rows = assertIndexedMessages(reader, corpus.messageCount) + const sessions = ( + reader.prepare('SELECT count(*) AS n FROM sessions').get() as { + n: number + } + ).n + assert.equal(sessions, corpus.files.length) + const bytes = tableBytes(reader) + const perMb = (value: number): number => + Math.round((value / (corpus.transcriptBytes / (1024 * 1024))) * 10) / 10 + const fileBytes = await checkpointedFileBytes(store.connection, indexPath) + stalls.sort((a, b) => a - b) + transactions.sort((a, b) => a - b) + console.log( + JSON.stringify( + { + platform: process.platform, + node: process.version, + transcriptMb: Math.round((corpus.transcriptBytes / (1024 * 1024)) * 100) / 100, + toolResultWords, + sessions, + rows, + rebuildMs: Math.round(rebuildMs), + rowsPerSecond: Math.round(rows / (rebuildMs / 1000)), + transcriptMbPerSecond: + Math.round((corpus.transcriptBytes / (1024 * 1024) / (rebuildMs / 1000)) * 100) / 100, + bytesPerTranscriptMb: { + messagesFts: perMb(bytes.messagesFts), + messages: perMb(bytes.messages), + sessions: perMb(bytes.sessions), + total: perMb(bytes.total) + }, + writeAmplification: Math.round((bytes.total / corpus.transcriptBytes) * 100) / 100, + fileWriteAmplification: Math.round((fileBytes / corpus.transcriptBytes) * 100) / 100, + transactions: transactions.length, + maxTransactionMs: Math.round((transactions.at(-1) ?? 0) * 100) / 100, + maxLoopStallMs: Math.round(stalls.at(-1) ?? 0), + p95LoopStallMs: Math.round(stalls[Math.floor(stalls.length * 0.95)] ?? 0), + loopStallSamples: stalls.length, + parseStats: stats + }, + null, + 2 + ) + ) + } finally { + reader.close() + } + } finally { + indexing = false + restoreExec() + unregister() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store.close() + } +} finally { + await rm(corpus.root, { recursive: true, force: true }) +} + +// Phase two: one transcript far larger than any real one, to price the ceiling +// that decides whether a file commits once or in chunks. +const largeTurns = Number(process.env.ORCA_SEARCH_BENCH_LARGE_TURNS ?? 23_000) +const large = await writeSyntheticTranscriptCorpus({ + sessions: 1, + turnsPerSession: largeTurns, + seed: 2 +}) +const largeIndexPath = join(large.root, 'index.sqlite') +try { + const errors: unknown[] = [] + const store = new SessionSearchStore(largeIndexPath, (error) => errors.push(error)) + const unregister = registerSessionSearchIndexConsumer(store) + const transactions: number[] = [] + const restoreExec = recordTransactionDurations(transactions) + try { + const stats = createSessionParseStats() + const started = performance.now() + await parseAgentSessionFileCached( + await sessionCandidate('claude', large.files[0]!), + process.platform, + stats + ) + const indexMs = performance.now() - started + restoreExec() + assert.deepEqual(errors, []) + assertIndexedMessages(store.connection, large.messageCount) + transactions.sort((a, b) => a - b) + + // The same file again, over a generation the index already holds. That is + // the pass a growing transcript really costs, and the one whose transaction + // used to be sized by the old session rather than by the chunk being + // written. The drain that reclaims the cut-loose generation runs after the + // commit, so its bounded batches are in `replaceTransactions` too. + const replaceTransactions: number[] = [] + const restoreReplaceExec = recordTransactionDurations(replaceTransactions) + requestWholeTranscriptRead(large.files[0]!) + const replaceStarted = performance.now() + await parseAgentSessionFileCached( + await sessionCandidate('claude', large.files[0]!), + process.platform, + stats + ) + const replaceMs = performance.now() - replaceStarted + // Finishes whatever the scheduled drain has not reached, so the reclaim is + // priced rather than left half done under the next measurement. + const reclaimStarted = performance.now() + await store.purgeOlderThan(null) + const reclaimMs = performance.now() - reclaimStarted + restoreReplaceExec() + assert.deepEqual(errors, []) + assertIndexedMessages(store.connection, large.messageCount) + replaceTransactions.sort((a, b) => a - b) + + console.log( + JSON.stringify( + { + phase: 'single-large-file', + transcriptMb: Math.round((large.transcriptBytes / (1024 * 1024)) * 100) / 100, + indexMs: Math.round(indexMs), + transactions: transactions.length, + maxTransactionMs: Math.round(transactions.at(-1) ?? 0), + replaceMs: Math.round(replaceMs), + replaceTransactions: replaceTransactions.length, + maxReplaceTransactionMs: Math.round(replaceTransactions.at(-1) ?? 0), + reclaimMs: Math.round(reclaimMs), + indexMb: + Math.round( + ((await checkpointedFileBytes(store.connection, largeIndexPath)) / (1024 * 1024)) * + 100 + ) / 100 + }, + null, + 2 + ) + ) + } finally { + restoreExec() + unregister() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store.close() + } +} finally { + await rm(large.root, { recursive: true, force: true }) +} diff --git a/config/scripts/telemetry-bundle-constant-patterns.mjs b/config/scripts/telemetry-bundle-constant-patterns.mjs index 04944b7b156..87c2ae43cf7 100644 --- a/config/scripts/telemetry-bundle-constant-patterns.mjs +++ b/config/scripts/telemetry-bundle-constant-patterns.mjs @@ -1,2 +1,6 @@ -export const BUILD_IDENTITY_RE = /\b(?:const|let|var)\s+BUILD_IDENTITY\s*=\s*"(rc|stable)"/ -export const WRITE_KEY_RE = /\b(?:const|let|var)\s+WRITE_KEY\s*=\s*"(phc_[A-Za-z0-9_-]+)"/ +// The unminified bundle keeps these names. Production minification may rename +// them, but the injected identity and key remain adjacent in the declaration. +export const BUILD_IDENTITY_RE = /\b(?:const|let|var)\s+BUILD_IDENTITY\s*=\s*["`](rc|stable)["`]/ +export const WRITE_KEY_RE = /\b(?:const|let|var)\s+WRITE_KEY\s*=\s*["`](phc_[A-Za-z0-9_-]+)["`]/ +export const MINIFIED_TELEMETRY_RE = + /\b(?:const|let|var)\s+[$\w]+\s*=\s*["'`](rc|stable)["'`][\s\S]{0,200}?[,$]\s*[$\w]+\s*=\s*["'`](phc_[A-Za-z0-9_-]+)["'`]/ diff --git a/config/scripts/telemetry-bundle-constant-patterns.test.mjs b/config/scripts/telemetry-bundle-constant-patterns.test.mjs index b8df5ec3d0f..ca87e3ee971 100644 --- a/config/scripts/telemetry-bundle-constant-patterns.test.mjs +++ b/config/scripts/telemetry-bundle-constant-patterns.test.mjs @@ -1,5 +1,9 @@ import { describe, expect, it } from 'vitest' -import { BUILD_IDENTITY_RE, WRITE_KEY_RE } from './telemetry-bundle-constant-patterns.mjs' +import { + BUILD_IDENTITY_RE, + MINIFIED_TELEMETRY_RE, + WRITE_KEY_RE +} from './telemetry-bundle-constant-patterns.mjs' describe('telemetry bundle constant patterns', () => { it.each(['const', 'let', 'var'])('accepts %s declarations', (declaration) => { @@ -13,4 +17,9 @@ describe('telemetry bundle constant patterns', () => { expect('const WRITE_KEY = null').not.toMatch(WRITE_KEY_RE) expect('const WRITE_KEY = "example-key"').not.toMatch(WRITE_KEY_RE) }) + + it('accepts minified adjacent declarations', () => { + const bundle = 'var dde=`stable`,fde=`phc_example-key_123`,pde=(dde===`stable`)' + expect(bundle).toMatch(MINIFIED_TELEMETRY_RE) + }) }) diff --git a/config/scripts/verify-telemetry-constants.mjs b/config/scripts/verify-telemetry-constants.mjs index 3bdcd8b3ec6..836a3e4546f 100644 --- a/config/scripts/verify-telemetry-constants.mjs +++ b/config/scripts/verify-telemetry-constants.mjs @@ -38,7 +38,11 @@ import { join, resolve } from 'node:path' // `node_modules`). If electron-builder ever drops it, promote this to a // direct devDependency in package.json. import { extractFile, listPackage } from '@electron/asar' -import { BUILD_IDENTITY_RE, WRITE_KEY_RE } from './telemetry-bundle-constant-patterns.mjs' +import { + BUILD_IDENTITY_RE, + MINIFIED_TELEMETRY_RE, + WRITE_KEY_RE +} from './telemetry-bundle-constant-patterns.mjs' // Why resolve from import.meta.url instead of cwd: a release runner (or a // developer debugging locally) may invoke this script from a non-root cwd. @@ -156,8 +160,14 @@ function verifyAsar(asarPath) { const buildIdentityMatch = BUILD_IDENTITY_RE.exec(indexJs) const writeKeyMatch = WRITE_KEY_RE.exec(indexJs) + const minifiedTelemetryMatch = MINIFIED_TELEMETRY_RE.exec(indexJs) - if (!buildIdentityMatch) { + // Rolldown renames module-local constants in production output. In that + // form, verify the adjacent injected identity/key declaration instead. + const verifiedIdentity = buildIdentityMatch?.[1] ?? minifiedTelemetryMatch?.[1] + const verifiedWriteKey = writeKeyMatch?.[1] ?? minifiedTelemetryMatch?.[2] + + if (!verifiedIdentity) { console.error(`::error::BUILD_IDENTITY constant missing or unexpected value in ${asarPath}`) const sample = indexJs.match(/.{0,80}BUILD_IDENTITY.{0,80}/g)?.slice(0, 5) ?? [] for (const line of sample) { @@ -165,7 +175,7 @@ function verifyAsar(asarPath) { } return null } - if (!writeKeyMatch) { + if (!verifiedWriteKey) { console.error(`::error::PostHog WRITE_KEY missing from ${asarPath}`) const sample = indexJs.match(/.{0,80}WRITE_KEY.{0,80}/g)?.slice(0, 5) ?? [] for (const line of sample) { @@ -174,7 +184,7 @@ function verifyAsar(asarPath) { return null } - return { asarPath, buildIdentity: buildIdentityMatch[1], writeKey: writeKeyMatch[1] } + return { asarPath, buildIdentity: verifiedIdentity, writeKey: verifiedWriteKey } } // Why verify every match (not just the first): macOS dual-arch produces one diff --git a/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs b/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs index a8c2cb3f4e7..253605781cf 100644 --- a/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs +++ b/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs @@ -46,6 +46,7 @@ const WINDOWS_SHIM_SPAWN_ALLOWLIST = [ 'config/scripts/electron-builder-config.test.mjs', 'config/scripts/ensure-native-runtime.test.mjs', 'config/scripts/live-remote-freeze-rpc.mjs', + 'config/scripts/pty-transcript-secret-scan.test.mjs', 'config/scripts/remote-agent-session-authority-repro.mjs', // Platform-local build paths; the win32 branch is dead code on both. 'config/scripts/build-mac-local.mjs', diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index 39008eaa963..dce3559fd10 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 45m + + downloads: 48m @@ -15,7 +15,7 @@ downloads downloads - 45m - 45m + 48m + 48m diff --git a/docs/reference/agent-pty-transcript-capture.md b/docs/reference/agent-pty-transcript-capture.md new file mode 100644 index 00000000000..934f0028a93 --- /dev/null +++ b/docs/reference/agent-pty-transcript-capture.md @@ -0,0 +1,129 @@ +# Capturing an agent PTY transcript + +Orca's readiness and blocked-prompt rules are text rules over what an agent CLI paints on a +terminal. They are only as good as the screens they were written against. This is how to record +one, byte for byte, so a rule can be pinned to evidence instead of to a remembered screen. + +Related: [`antigravity-readiness-evidence.md`](./antigravity-readiness-evidence.md) names the +specific Antigravity transcripts that are still missing and what each one decides. + +## The recorder + +``` +node config/scripts/capture-agent-pty-transcript.mjs --name [options] -- [args...] +``` + +It allocates a real PTY, spawns the agent inside it, mirrors the session to your terminal so you +can drive it by hand, and appends every byte it receives to +`src/main/runtime/__fixtures__/.txt`. It does not strip escapes, fold `\r`, rewrap +lines, or normalise anything — the file is what the terminal received. + +- **Ending a capture:** press Ctrl+]. The recorder consumes that key and + never forwards it, which is the only way to end a capture _while a dialog still owns the + screen_. Quitting the agent instead would first dismiss the dialog you came to record. +- `--cols N --rows M` pin the PTY size (default: your terminal's). Wrapping is part of the + evidence, so record the size — the sidecar does it for you. +- `--duration S` stops unattended after S seconds, for a screen that needs no interaction. +- `--send ":"` types into the PTY at a fixed offset, repeatable, with `\r` `\n` `\t` `\e` + escapes. A dialog capture has to be driven, and an unattended run (CI, or an agent) has no TTY to + type into; the keystrokes ride the same PTY a human's would. For example, the committed + `antigravity-dialog-model-picker.txt` was recorded with + `--duration 24 --send "14000:/model" --send "16000:\r"`, which leaves the picker owning the + screen when the capture stops. +- `--note ""` records the account type, plan, model and CLI version in the sidecar. +- `--out ` writes outside the fixture directory (use it for a first dry run). + +Each capture also writes `.meta.json` with the timestamp, platform, command, +PTY size, note and exit code. Commit it with the transcript; the version and account type behind +a screen are not recoverable from the bytes. + +**Prerequisite:** `node-pty` must be built for plain Node: + +``` +node config/scripts/ensure-native-runtime.mjs --runtime=node +``` + +Orca itself does not need to be running, and the recorder never touches Orca state. + +### Platform notes + +- **macOS / Linux:** nothing special. `TERM=xterm-256color` is set for the child. +- **Windows:** run it from Windows Terminal / PowerShell, not a Git Bash (MSYS) pane — MSYS + rewrites arguments that start with `/`, which mangles the `cmd.exe /c` hand-off. A `.cmd` or + `.bat` agent shim cannot be spawned by node-pty directly, so the recorder routes those through + `cmd.exe` for you. +- **WSL:** capture _inside_ the distro (run the recorder from the distro's checkout). Recording + `wsl.exe` from the Windows side adds the login-shell banner to the transcript. +- **SSH:** record on the execution host. A transcript recorded locally is not evidence about what + a remote agent prints. + +## Privacy: scrub before committing + +A live agent screen routinely contains things that must not enter git history: + +| Scrub | Why | +| ---------------------------------------------------------------------- | ---------------------------------------------------- | +| Account email / sign-in identifier | The account row on a ready screen prints it verbatim | +| Org, tenant or team name | Identifies a customer | +| Machine hostname and OS username | Appear in prompts, paths and the OSC title | +| Absolute home paths (`/Users/`, `C:\Users\`) | Contain the username | +| JWTs, `AIza…` keys, `1//…` refresh tokens, `Bearer …`, `sk-…`, `ghp_…` | Live credentials; a sign-in screen can echo one | +| Private repo, branch and ticket names | Leak roadmap detail | +| Anything you pasted into the agent during the capture | You typed it; it is in the transcript | + +The recorder scans the file as soon as the capture ends and prints every hit with a line and +column. To scrub: + +``` +node config/scripts/capture-agent-pty-transcript.mjs --scan src/main/runtime/__fixtures__/.txt --redact +``` + +Redaction replaces each finding with a **same-length** placeholder (`u…u@example.com`, `XXXX…`). +Length matters: a transcript's value is its exact wrapping and column alignment, and a shorter +replacement reflows the screen and destroys the evidence. + +### Verify it is gone + +1. `node config/scripts/capture-agent-pty-transcript.mjs --scan src/main/runtime/__fixtures__/.txt` + must print `clean` and exit `0`. It recognises its own placeholders, so a scrubbed file passes. +2. Grep for the specifics the scanner cannot know: + `rg -n -i -- "$(whoami)|||" src/main/runtime/__fixtures__/.txt` +3. Read it once with escapes visible: `LC_ALL=C cat -v src/main/runtime/__fixtures__/.txt`. + The scanner matches shapes; only a human catches a project name. +4. Check the sidecar too — `--note` text is free-form and is committed. + +`config/scripts/pty-transcript-secret-scan.test.mjs` re-scans every committed +`__fixtures__/*.txt`, so a transcript that skips step 1 fails the suite. + +## Consuming a transcript in a test + +Feed the raw bytes through the runtime rather than into a matcher directly: escape handling, +tail retention and title tracking all live in `onPtyData`, and a rule tested on pre-normalised +text is tested on something no pane ever sees. + +`src/main/runtime/agent-transcript-pane-test-harness.ts` builds the pane; +`src/main/runtime/terminal-interactive-wait-visibility.test.ts` (cursor-agent) and +`src/main/runtime/antigravity-readiness-transcripts.test.ts` (Antigravity) are the two consumers. + +## Worked example: the Antigravity captures + +The six committed `antigravity-*.txt` fixtures were recorded this way on macOS against +`agy` 1.1.25. Two points generalise: + +- **Reach a state without mutating the operator's config.** The ready-screen captures ran in a + directory the CLI already trusted, so no trust answer was written. Where a dialog could only be + reached by signing the operator out or deleting their settings, it was left uncaptured and + recorded as such rather than forced. +- **An environment variable is a legitimate capture knob** where a setting is not. + `AGY_CLI_HIDE_ACCOUNT_INFO=1` produced a second ready screen with no account row, which is + evidence no amount of reasoning about the first screen could have supplied. It changes nothing + on disk. + +## Known gap in the existing captures + +The three `cursor-agent-*.txt` fixtures contain **no escape bytes and no carriage returns**. +Whatever produced them went through a renderer and a clipboard, so they preserve wording and +box-drawing glyphs but not the caret, the cursor moves, the repaints, or whether the CLI uses the +alternate screen buffer. They are good enough for the wording-based rules built on them and are +not evidence for anything else. New captures made with this recorder keep those bytes; the +Antigravity scaffold asserts their presence so a pasted screen cannot pass as a capture. diff --git a/docs/reference/agent-status-store.md b/docs/reference/agent-status-store.md new file mode 100644 index 00000000000..f1068a0d993 --- /dev/null +++ b/docs/reference/agent-status-store.md @@ -0,0 +1,255 @@ +# Agent status store + +## Status + +Proposed on 2026-09-09 as the follow-up to #19217. It lands in four steps, in +this order, each independently shippable: + +1. main-only: every producer writes into one store and `worktree ps` reads it, + split into 1a (structured sessions join the store) and 1b (the runtime's + duplicate retained store is deleted); +2. renderer: the sidebar becomes a subscriber and stops re-deriving rows; +3. shared: one worktree-status rollup and one freshness rule for every reader. + +The PR that carries this document is PR 1a. Sections below are grouped under +the step that delivers them; only PR 1a has landed. + +## The problem this solves + +Orca shows "what is this agent doing" in four places: the desktop sidebar, the +`orca worktree ps` command, the mobile app, and the agent dashboard. Before +#19217 those readers did not even share their inputs. After #19217 they share +the structured-session mapping and nothing else. + +An audit on 2026-09-09 found six producers and three consumers, and three +separate copies of the same row inside the main process alone: + +| Main-process copy | Keyed by | Owned by | Persisted | Evicted | +| -------------------------------------- | --------- | ---------------------------------------------------------- | ------------------- | ----------------------------- | +| hook server `lastStatusByPaneKey` | paneKey | `src/main/agent-hooks/server.ts` | `last-status.json` | tab close, pty exit, hydrate | +| runtime `RuntimeAgentRowStore` | paneKey | `src/main/runtime/runtime-agent-row-store.ts` | no | pty exit only | +| structured feed `published` | sessionId | `src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts` | no | never (a broadcast cache) | + +The second copy is a duplicate write: the OSC status parsed in main is +forwarded to the hook server _and_ retained in the runtime store from the same +call (`orca-runtime-create-terminal-side-effect-command-code-detector.ts`). +The third copy is keyed differently and never reaches the hook server at all, +which is why `worktree ps` grew its own adapter for it in #19217. + +Each reader then applies its own precedence and freshness rules, so the same +pane can legitimately read differently on the desktop, on the phone, and in +the CLI. + +## The rule + +**The execution host owns agent status, in one store, and every reader +subscribes to it.** This follows the boundary in +[`ssh-execution-boundary.md`](./ssh-execution-boundary.md): the host that runs +the process is the only party that can observe it, and the client is never +authoritative for execution state. + +Three consequences: + +- One store per execution host. A remote host keeps its own store and the + client mirrors it down, as the web-session mirror already does. Mirroring is + not merging: a client never writes its observations back to a host. +- Precedence is decided once, at write time, with provenance recorded on the + row. Readers never re-adjudicate hook versus terminal versus structured. +- Readers keep only presentation policy and user facts: the 30-minute display + decay, acknowledgements, dismissals, unread. Those stay reader-side but + become one shared implementation (PR 3). + +## The store already exists + +The hook server's state is that store today for every PTY-based agent. The +audit established: + +- hook HTTP posts, the WSL and SSH relay receivers, and main's own OSC parse + all converge on the same `applyNormalizedStatus` path, stamped with the + authority id `main-agent-hooks`; +- it alone holds pane authority: launch tokens and their hashed commitments, + retired-pane fences, pane-key aliases, per-connection ordering watermarks, + and the evidence-age map that must outlive a transport clear; +- it alone persists, with a seven-day hydrate window and the + `restoredUnconfirmed` stamp that keeps a hydrated row from ever reading as + live truth; +- it already fans out to both renderer windows over `agentStatus:set` and + `agentStatus:clear`, and serves `agentStatus:getSnapshot`. + +Nothing else in main carries those guarantees, and building a second store +with them would be the wrong direction. So the design is not "add a store". It +is: **route the two producers that bypass the hook server through it, then +delete the copies.** + +## PR 1a: structured sessions publish into the store + +No renderer behavior changes. The sidebar keeps receiving the same IPC events +it receives today, plus structured-session rows it currently derives itself. + +### Structured sessions publish into the hook server + +The structured feed keeps its job of projecting a session's journal into a +summary and streaming it to subscribers. On every publish it additionally +ingests the summary into the hook server as a status row: + +| Row field | From | +| ----------------- | ------------------------------------------------------------- | +| `paneKey` | `structuredAgentSessionPaneKey(tabId, sessionId)`, the key the renderer already uses; its leaf is UUID-shaped so pane-key validation accepts it | +| `tabId` | `structuredAgentSessionTabId(sessionId)` | +| `worktreeId` | `summary.workspaceId` (a folder workspace id is a valid value) | +| `state` | `structuredAgentSessionStatusState(summary.status)`, the mapping #19217 shared | +| `structuredHost` | `'owned'` while `summary.hostExecutionOwned` is set, otherwise `'held'`; `worktree ps` derives its row's `structuredHostOwned` from it | +| prompt, tool, last message, model, provider session | the summary's fields | + +Sessions with no persisted turn (`status === null`) produce no row, matching +what the chat shows. When the host revokes live ownership the row is re-set +without the flag; when the host closes or evicts the session the row is +dropped. Both already exist as feed events (`revokeLive` and the roster +filter in `liveSessionSummaries`); PR 1 turns them into store writes. + +Dropping the session from the host's map and dropping its row are one +operation, `forgetStructuredAgentSession`. The store keeps a row until told, +and a host-owned row bypasses the staleness check, so a deletion path that +forgot the row would strand a permanently working-looking agent. + +Two rules the ingest must keep: + +- **Never persist a structured row.** The journal is the durable truth for a + structured session and the host republishes on restore. A structured row in + `last-status.json` would hydrate as `restoredUnconfirmed` and then fight the + live republish. The serializer skips rows carrying `structuredHost`, and + hydrate drops any such row found on disk. Applying one therefore also skips + the persist schedule: the walk and stringify could only reproduce the file + that is already on disk, once per debounce window for every streaming chat. +- **Never let it fight a hook row.** A structured session has no PTY, so no + hook or OSC event carries its pane key. The ingest still goes through the + disposition gate so a retired pane key is refused like any other. + +Applying one does still run both status fan-outs, and that is intended rather +than incidental. `notifyStatusChangeListeners` is what feeds +`agentAwakeService`'s power-save blocker, and `subscribeEnrichedStatus` is what +feeds `AgentSessionTransitionRecorder`'s stats, so joining the store enrolls +native chats in both. A working native chat is real work and should hold the +machine awake exactly like a PTY agent does. + +The drop side routes through `dropStatusEntry`, not `clearPaneState`: a +pane-status-clear reaches the renderer, and until PR 2 the renderer's own feed +bridge is that pane key's writer. It also passes `preserveResumeIdentity: +false` — the `providerSessionOnly` remnant a dismissed pane keeps exists so the +agent can be resumed in that pane, and a structured session has no pane and +keeps its resume identity in the record store. Like every other +`dropStatusEntry` caller, it emits no pane clear, so a session dropped +mid-`working` leaves `AgentSessionTransitionRecorder` holding an open stats +session until its LRU evicts it; that gap is shared with the user-dismissal +path and is not specific to structured rows. + +The ingest lives in the feed, not in `structured-agent-session-host.ts`, which +sits at the file-length cap. + +### `worktree ps` becomes a reader + +The structured adapter added in #19217 is deleted, and structured rows reach +`worktree ps` through the same snapshot as every other row. The +retained-versus-hook reconciliation in `collectRuntimeWorktreePtyAgentSources` +stays until PR 1b removes the store that feeds it. What this step settles is +the admission gate that decides which rows a worktree listing may show: + +- a hook or OSC row needs its tab mirrored or a connected pty, as today, and + SSH rows stay exempt because their tabs may exist only remotely; +- a row carrying `structuredHost` is admitted while the host holds the session, and + the host's drop on close is what removes it. No tab-mirror requirement: a + structured session's tab lives in the renderer's own tab state, and a + headless host has no renderer to mirror it from. That argument only holds if + the headless host is itself wired to the store, which is a separate + obligation per entry point: the Electron hosts (desktop and `orca serve`) + share `main-process-runtime-service.ts`, and `orcad` constructs its own + runtime in `src/main/orcad/orcad-entry.ts`. A host missing that wiring lists + no agents at all, not just no structured ones, because `worktree ps` reads + the same snapshot for every row. + +The freshness bypass for host-owned structured rows already exists in +`isFreshNonDoneAgentStatus`; with the flag now on the row it becomes the only +path, and the hand-rolled check in `runtime-worktree-agent-rows.ts` goes. + +### Wire compatibility + +`AgentStatusIpcPayload` gains one optional field, `structuredHost`, and the +`worktree ps` row gains `structuredHostOwned`. Under rule 1 of +[`remote-wire-compatibility.md`](./remote-wire-compatibility.md) both are safe: +an old client ignores them. `worktree ps` rows keep their shape and vocabulary, +so the mobile app sees no change. + +Until PR 2 the main process does not forward structured rows to the renderer +over `agentStatus:set` or `agentStatus:getSnapshot`. The renderer's feed +bridge still writes those rows itself, and forwarding them too would give one +pane key two writers. Removing that filter is the first step of PR 2. + +## PR 1b: the runtime's retained row store is deleted + +Not yet implemented; `RuntimeAgentRowStore` and the retained-versus-hook +reconciliation it feeds are both still in place after PR 1a. + +`RuntimeAgentRowStore` keeps the same payload the hook server already holds. +Its only extra is the pty id, used to clear rows on exit and as a fallback key +for the mobile projection. PR 1b will stamp `terminalHandle` on OSC-ingested +rows from the runtime event's `ptyId`, and rewrite the three readers over the +hook server's snapshot: + +- `worktree ps` reads `getStatusSnapshot()` directly; +- `getFreshExplicit` already consults hook rows; it drops the retained input; +- `getFreshForMobile` matches on pane key, then on `terminalHandle`. + +One behavior change will follow and is intended: a row the user dismisses on +the desktop disappears from `worktree ps` and the phone at the same time, +instead of lingering until the pty exits. + +## PR 2: the renderer subscribes + +With structured rows arriving over `agentStatus:set`, the renderer's +`StructuredAgentSessionStatusBridge` no longer needs to write status; its +unmount cleanup becomes a tab-close signal to the host. The IPC applicator is +the single writer for observed status. The 2026-09-09 audit sorted the other +writers: + +| Writer | Disposition | +| --------------------------------------------------------------- | -------------------------------------------------- | +| Command Code output seeds, parked-pane seeds, pty-exit removal | delete; main already emits the same facts | +| structured bridge status writes | delete; main now publishes the row | +| launch placeholder seeds (a user launched an agent with a prompt) | keep for now; main holds the launch config and can seed later | +| dismissal, acknowledgement, unmount | keep; user facts and component lifecycle | +| remote-runtime OSC parse (bytes never transit local main) | keep, fenced behind the host's published row once the host is new enough; rule 3 of the wire doc applies | +| web-session mirror receipt clock | keep; the decay rule needs both clocks from one machine | + +The Command Code done-settle window is renderer policy with no main +equivalent. PR 2 either moves it into main's detector or leaves it, and says +which. + +## PR 3: one rollup, one clock + +The worktree card status is derived three times: `lib/worktree-status.ts` in +the renderer, `runtime-worktree-status-projection.ts` in main, and +`agent-row-display.ts` in mobile, which hand-copies the 30-minute constant. +PR 3 moves the rollup and the decay into `src/shared` and makes all three +call it. + +## What does not change + +- The hook scripts, the OSC 9999 wire format, and the relay protocol. +- The status vocabulary. `working / blocked / done` for rows, + `working / attention / idle` for structured summaries, mapped once. +- The `live / unverifiable / exited` verdicts for remote work. Loss of contact + clears nothing; the SSH exemptions in the admission gate stay. +- Hydration honesty: a restored non-done row is `restoredUnconfirmed` and is + never fresh. + +## Verification + +- Unit: ingest a structured summary and read it back through + `getStatusSnapshot`, `worktree ps`, and the mobile projection; assert the + serializer never writes a row carrying `structuredHost`; assert a hydrated + file that somehow contains one is dropped. +- Unit: the existing `worktree ps` suites pass unchanged, which is the + characterization that will show PR 1b's deletion of the retained store + changed no listing. +- Live: the parity check from #19217 (working, done, close, reload) repeated + against the merged store, with both surfaces read from the one row. diff --git a/docs/reference/antigravity-readiness-evidence.md b/docs/reference/antigravity-readiness-evidence.md new file mode 100644 index 00000000000..0010fa76ded --- /dev/null +++ b/docs/reference/antigravity-readiness-evidence.md @@ -0,0 +1,263 @@ +# Antigravity readiness: what the transcripts show + +`findAntigravityReadyPromptIndex` in `src/main/runtime/terminal-wait-detection.ts` decides whether +an Antigravity pane is ready for a prompt. It has been written five times, each version tuned +against a five-line screen typed from memory into a `.spec.ts` fixture. Three of the first four +were found worse than the bug they replaced, and the fifth was reverted. + +Real transcripts now exist. They were recorded from a live `agy` on macOS with +[`agent-pty-transcript-capture.md`](./agent-pty-transcript-capture.md) and are committed under +`src/main/runtime/__fixtures__/`. `src/main/runtime/antigravity-readiness-transcripts.test.ts` +replays them through the runtime. + +**Headline: on real output the current detector is inverted.** It refuses a genuinely ready screen +and accepts a live model picker. The five attempts argued about which extra condition to add; none +of them had noticed that the condition they all shared — a line beginning with the model name — +never matches a real Antigravity ready screen at all. + +## Versions + +| Thing | Value | +| ------------------------- | ----------------------------- | +| `agy --version` | `1.1.25` | +| Banner printed by the TUI | `Antigravity CLI 1.2.0` | +| Captured | 2026-09-10, macOS, 120x40 PTY | + +The binary and its own banner disagree. Any rule keyed to a version string must read the banner, +not `--version`, and must tolerate the two disagreeing. + +## What the captures are + +| Fixture | What it is | +| -------------------------------------------- | --------------------------------------------------------- | +| `antigravity-ready-api-key-gemini-model.txt` | Ready screen, API-key identity, Gemini 3.7 Flash (Low) | +| `antigravity-ready-account-info-hidden.txt` | The same ready screen with `AGY_CLI_HIDE_ACCOUNT_INFO=1` | +| `antigravity-dialog-trust-workspace.txt` | Workspace trust dialog, live and unanswered | +| `antigravity-dialog-model-picker.txt` | `/model` picker, live and unanswered | +| `antigravity-dialog-command-palette.txt` | Slash-command palette, live and unanswered | +| `antigravity-dialog-dismissed.txt` | `/model` picker dismissed with esc, then settled | +| `antigravity-busy-mid-turn.txt` | A real turn, recording stopped while the spinner was live | +| `antigravity-busy-turn-ended.txt` | The same turn after it ended and the composer returned | + +## What could not be captured, and why + +Nothing below was faked. Each is a case the recorder could not reach without changing the +operator's account state or configuration, which is out of bounds. + +| Missing | Why | +| ------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `antigravity-ready-business-non-gemini.txt` | This machine has no OAuth session — the CLI prints _"You are currently not signed in"_ and authenticates from `GEMINI_API_KEY`. Reaching a Business ready screen means signing someone in. | +| A non-Gemini model on any ready screen | `agy models` offers 11 models, all Gemini, and `settings.json` pins `modelProvider: gemini`. A non-Gemini row is not reachable from this account. | +| `antigravity-dialog-sign-in.txt` | Unsetting `GEMINI_API_KEY` does not reach the sign-in dialog; the CLI refuses to start because `modelProvider` is pinned. Reaching it means editing the operator's `settings.json`. | +| `antigravity-dialog-theme-picker.txt` | There is no `/theme` command in 1.2.0 (`Unknown command: /theme`). The picker appears only in first-run onboarding, which means deleting the operator's config. | +| `antigravity-dialog-privacy-notice.txt` | First-run onboarding, as above. | +| `antigravity-dialog-update-banner.txt` | Cannot be forced; no update was pending during the session. | + +Each remains as a named, skipping case in the suite so it is visible rather than forgotten. + +## What the transcripts show + +### 1. The ready screen's model row is not at the start of a line + +The ready screen prints a block-glyph logo down the left, and the identity, model and path rows are +painted **on the same physical lines as the logo**. What Orca derives is: + +``` +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) +▄▀▀ ▀▀▄ ~ +``` + +The detector requires `normalized.startsWith('gemini', trimmedStart)` on a trimmed line. The +trimmed line starts with `▀`. It never matches. Measured three ways on the real screen: + +| Input | `isKnownReadyPromptPreview` | +| ------------------------------------------------------ | --------------------------- | +| Real ready screen | `false` | +| The same screen with the logo glyphs stripped | `true` | +| Real ready screen followed by the live `/model` picker | `true` | + +So the logo — decoration, and suppressible with `AGY_CLI_HIDE_LOGO` — is what decides readiness +today, and the live dialog is what supplies the model line the ready screen could not. + +### 2. The dialog is what satisfies the model rule + +`/model` prints its options one per line: + +``` +Gemini 3.8 Flash +> Gemini 3.7 Flash (current) +Gemini 3.1 Pro +``` + +Those lines _do_ begin with `Gemini`, and a bare `>` composer line sits earlier in the same tail +from before the picker opened. Both halves of the rule are satisfied **while a dialog owns the +screen**, and the pane reads ready. This is the false-ready hazard the last three attempts were +each trying to close, reproduced from a real capture. + +### 3. `>` is the dialog selection marker, not only the composer caret + +Every dialog uses `>` to mark the highlighted row: `> Yes, I trust this folder`, +`> Gemini 3.7 Flash (current)`, `> /add-dir`. The idle composer is a line whose whole trimmed +content is `>`. That distinction is the only thing separating them, which means the relaxation +proposed in PRs #15840 and #15852 — accept any line _beginning_ with `>` — would make the trust +dialog and the model picker read as ready. On 1.2.0 the idle composer is a bare `>`; those PRs' +1.1.17 mode-banner claim could not be reproduced here and may be mode-specific. + +### 4. There is no email account row, and the row can be switched off entirely + +For an API-key user the identity row reads literally `Gemini API key`. There is no `@`, no +domain, nothing an account-row rule can key on. Separately, `AGY_CLI_HIDE_ACCOUNT_INFO=1` — a +supported environment variable in the binary — removes the row from a fully ready screen, which +`antigravity-ready-account-info-hidden.txt` captures. + +### 5. Dialogs are drawn two different ways, and the banner is never reprinted + +The trust dialog and the sign-in splash take the **alternate screen** (`ESC[?1049h` … `ESC[?1049l`). +The model picker and command palette are drawn **in place on the main screen** with erase-to-EOL. +After dismissal the CLI prints `⎿ Exited /model command` and redraws the composer — it does **not** +reprint the banner. The header stays where it was at startup. + +### 6. Rows are positioned with cursor addressing, not newlines + +The status row is written with absolute and relative moves (`ESC[13;99H`, `ESC[83X ESC[83C`), so +`? for shortcuts` and `Gemini 3.7 Flash · low` end up on one derived line. Any rule that assumes +one screen row equals one `\n`-delimited line is reading a different document than the user sees. + +## 8. Busy frames park the caret exactly like idle frames — the spinner is what differs + +The frame that ends a turn-in-progress and the frame that ends an idle screen park the cursor with +the **same bytes**. Only the hint row differs, and the park erases it: + +``` +idle: ? for shortcuts ESC[83X ESC[83C Gemini 3.7 Flash · low CR ESC[2A ESC[2C ESC[?25h +busy: esc to cancel ESC[85X ESC[85C Gemini 3.7 Flash · low CR ESC[2A ESC[2C ESC[?25h +``` + +So a rule that keys on "the caret is the last thing in the tail" cannot tell busy from idle **on the +frame alone**. What saves it is what comes next. Each spinner tick is its own repaint with its own +park, two rows higher than the frame's: + +``` +ESC[?25l CR ESC[2A ⣯ Generating ESC[11D ESC[?25h +ESC[?25l CR ESC[2A ⣟ Generating. ESC[12D ESC[?25h +``` + +That second `CR ESC[2A` splices the composer row away, so the retained tail during a live turn ends +on the spinner row, not on the caret. Measured on `antigravity-busy-mid-turn.txt`: + +| Capture | last retained line | bare `>` line present | +| -------------------------------------------- | ------------------ | --------------------- | +| `antigravity-ready-api-key-gemini-model.txt` | `>` | **yes** | +| `antigravity-busy-mid-turn.txt` | `⣟ Generating...` | **no** | + +**Consequence for a caret-based rule:** it already answers "not ready" for a real mid-turn capture, +because there is no bare caret in the tail to match. A constructed input that keeps the park bytes +and only edits the status text is not faithful to a live turn — a live turn has a spinner row +repainting _below_ the composer. + +**The residual window, and the clause it implies.** Between a frame park and the next spinner tick +the tail does end on the bare caret and is indistinguishable from idle. The gap is one tick +interval. Any readiness path gated on sustained quiescence is safe, because ticks keep arriving and +the pane is never quiet; a path that only inspects retained text is not. For those paths the +evidence supports one clause, and only one: + +> **A braille glyph (U+2800–U+28FF) on the last visible line of the retained tail means working.** + +That predicate already exists in this file for cursor-agent (`CURSOR_BUSY_SPINNER_RE`) and should be +reused rather than reinvented. It must be scoped to the **last visible line**, not the whole tail: +a first-run transcript prints `⠾ Signing in...` during startup, which would otherwise pin a ready +screen as busy forever. + +Nothing else in the capture distinguishes the two states. The hint row (`esc to cancel` versus +`? for shortcuts`) is erased by the park in both cases, the park offsets are identical, and +`ESC[?25l`/`ESC[?25h` fencing appears around every repaint, idle or busy. + +## Confirmed / refuted, by attempt + +Evidence column names the fixture; all quoted text is from the committed transcripts. + +### Attempt 1 — the rule at HEAD + +| # | Claim | Verdict | Evidence | +| ---- | -------------------------------------------------------- | --------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 1.1 | A ready screen prints the banner `Antigravity CLI` | **Confirmed** | `Antigravity CLI 1.2.0` in both ready fixtures | +| 1.1b | …and its last occurrence in the tail is the live one | **Refuted** | The trust dialog's own body says _"Antigravity CLI requires permission to read, edit, and execute files here"_, so `lastIndexOf` lands inside the dialog | +| 1.2 | The model row begins with the vendor word `Gemini` | **Refuted** | `▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low)` — the logo precedes it; never at line start | +| 1.3 | The caret line's whole trimmed content is `>` | **Confirmed** on 1.2.0 idle | bare `>` in both ready fixtures | +| 1.3b | …and only the composer prints `>` | **Refuted** | `> Yes, I trust this folder`, `> Gemini 3.7 Flash (current)`, `> /add-dir` | +| 1.4 | A ready screen prints the workspace path on its own line | **Refuted** | the path shares its line with logo glyphs (`▄▀▀ ▀▀▄ ~`) | + +### Attempt 2 (loop 1) — blacklist the model line + +| # | Claim | Verdict | Evidence | +| --- | ------------------------------------------ | ----------- | ---------------------------------------------------------------------------------------------------------------- | +| 2.1 | Dialog model-row wording is enumerable | **Refuted** | the palette lists 50+ commands with free-form descriptions; the picker prints whatever models the account offers | +| 2.2 | A dialog never reproduces a real model row | **Refuted** | the `/model` picker prints four real model rows, one per line, at line start | + +### Attempt 3 (loop 2) — structural ordering on `headerIndex` + +| # | Claim | Verdict | Evidence | +| --- | -------------------------------------------------- | ---------------------------------- | ---------------------------------------------------------------------------------------------------- | +| 3.1 | A live dialog is printed below the ready chrome | **Confirmed** for in-place dialogs | picker and palette append below the composer | +| 3.2 | The banner is reprinted when a dialog is dismissed | **Refuted** | `antigravity-dialog-dismissed.txt` shows `⎿ Exited /model command` and a redrawn composer, no banner | +| 3.3 | Antigravity does not use the alternate screen | **Refuted** | `ESC[?1049h` opens the trust dialog and the sign-in splash | +| 3.4 | No full repaint per keystroke | **Partly refuted** | typing `/mod` repaints the palette region on each keystroke with `ESC[K` | + +Because of 3.2, `headerIndex` cannot be the anchor: it never advances. Ordering can only be +expressed against the model/caret positions, which is what 1.2 and 1.3b just invalidated. + +### Attempt 4 (loop 3) — require a positive account row + +| # | Claim | Verdict | Evidence | +| --- | ---------------------------------------------------- | ---------------------- | --------------------------------------------------------------------------------------------------------------------------- | +| 4.1 | Every ready screen prints an account row | **Refuted, twice** | API-key identity prints `Gemini API key` (no `@`); `AGY_CLI_HIDE_ACCOUNT_INFO=1` removes the row entirely | +| 4.2 | A startup dialog never contains an `@`-and-`.` token | **Not reachable here** | none of the captured dialogs contains one, but the palette shows free-form skill descriptions, which are user-authored text | +| 4.3 | The account row is distinguishable from prose | **Refuted** | the row is not a distinct line; it shares one with the logo | + +### Attempt 5 (PR #19749, reverted) — ordering + account row + +| # | Claim | Verdict | Evidence | +| --- | -------------------------------------------------------- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 5.1 | Ordering plus an account row separates ready from dialog | **Refuted** | the account row is optional (4.1) and the ordering anchor never moves (3.2) | +| 5.2 | Executing both builds was sufficient verification | **Refuted** | the executed input was the hand-written fixture, so the check reproduced the fixture's assumptions. The real screen disagrees with that fixture on the model row, the path row and the account row | +| 5.3 | The wedge is a model-name problem | **Refuted** | it is a line-start problem. Even `Gemini 3.7 Flash (Low)` — a Gemini model — fails, because a logo glyph precedes it | + +### Cross-cutting + +| # | Question | Answer | +| --- | ---------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | +| X1 | Does `agy` set an OSC title distinguishing busy from idle? | **No.** Not one OSC title sequence appears in any capture. Title-based readiness is unavailable for this agent | +| X2 | Does it repaint with bare `\r`? | **Yes**, constantly, plus `ESC[K` and absolute cursor moves | +| X3 | Does the caret survive in the tail? | **Yes** — a bare `>` line is present in every ready capture | +| X4 | Banner-to-caret distance | ~8 derived lines on a 120x40 PTY; the banner falls outside the 6-line preview window, so only the full retained tail can see it | +| X5 | Pane title on the trust screen versus ready | Identical: none | + +## Can attempt six be written? + +Yes — but not as a variation on any of the five. Every one of them refined a predicate over +`\n`-delimited lines, and that is the layer where the evidence says the information is not. + +What the captures support: + +- **The one stable, dialog-free ready marker is a line whose entire trimmed content is `>`.** It is + present in every ready capture and absent from every dialog capture, because a dialog's `>` always + carries its selected row's label. This is a much narrower rule than any attempt used, and it is + the only one that survived contact with the transcripts. +- **Drop the model-row requirement.** It matches dialogs and not ready screens. Keeping it inverted + the detector. +- **Do not require an account row.** It is optional by environment variable and carries no email for + API-key users. +- **Do not anchor on `headerIndex`.** The banner is printed once and never reprinted. +- **The blocked-signal path already works** for the trust dialog: `antigravity-dialog-trust-workspace.txt` + is correctly refused today, by wording, not by structure. + +What is still unknown and should be captured before shipping: the sign-in, theme, privacy and +update dialogs, and any ready screen where the composer is not idle (accept-edits and plan mode, +which PRs #15840 and #15852 describe from a screenshot). A bare-`>` rule is only as good as the +claim that those modes still end on a bare `>`; that claim is untested. + +The honest summary is that this is a screen-shaped problem being solved with line-shaped tools. A +rule over the derived tail can be made much better than what ships today, but the durable fix is to +ask the terminal emulator what the bottom row of the screen actually is, rather than inferring it +from a byte stream that was written with cursor addressing. diff --git a/docs/reference/headless-linux-server.md b/docs/reference/headless-linux-server.md index 50a38cf446e..a6a13489f2c 100644 --- a/docs/reference/headless-linux-server.md +++ b/docs/reference/headless-linux-server.md @@ -390,6 +390,10 @@ its own `orca`. `ws://` through an HTTPS-only endpoint. - Hostnames, IPv4, bracketed IPv6, and raw IPv6 literals are supported. IPv6 still requires an IPv6-reachable listener/network path. +- Background push notifications to a paired phone do not fire from a headless + server: agent-completion detection runs in the desktop renderer, which is not started in serve + mode, so nothing reaches the push gateway even though the phone + registers successfully. - `xvfb-run` and `dbus-run-session -- xvfb-run` remain valid diagnostic launch shapes, but neither should be needed when `Xvfb` is installed and no display is configured. Repeated D-Bus messages without a ready block indicate startup diff --git a/docs/reference/remote-wire-compatibility.md b/docs/reference/remote-wire-compatibility.md index 13741c03789..fb43d98ac12 100644 --- a/docs/reference/remote-wire-compatibility.md +++ b/docs/reference/remote-wire-compatibility.md @@ -180,6 +180,33 @@ An old client against a new host ignores the key, as Rule 1 allows. New members `RuntimeTerminalWaitBlockedReason` are also Rule 1: no consumer switches exhaustively on it, and both the CLI and worker-start interpolate it as an opaque string. +## Worked example: the `turn` journal item and its transitional downgrade + +The structured chat journal records a turn as a first-class item, +`{ kind: 'turn', turnId, state, userItemId?, startedAt?, completedAt?, durationMs? }`, where it +used to write `{ kind: 'status', text, turnLifecycle }`. Nothing in the codec moves, but it is +Rule 3: a client that predates the item does not know the kind and renders it as a text bubble +with no text. So the item is gated on a client capability, `agent-session.turn-item.v1`. + +The gate lives at the RPC boundary only, in +`src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.ts`, composed +around `agentSession.history` and `agentSession.subscribe` next to the background-task +projection. A client that does not advertise the capability receives every `turn` item +rewritten to the legacy status form with the full lifecycle under `turnLifecycle`; a client that +advertises it, and any in-process caller, receives the canonical body. The journal, the status +feed, and every host-side reader keep the `turn` item; `readAgentJournalTurn` in +`src/shared/agent-session-turn-record.ts` reads either form, so a new client against an old +host that still writes the status row also works. + +An old client against a new host sees the status row it always did. A new client against an old +host advertises a capability the host ignores and reads the status row through the shared +reader. The downgrade is transitional: once no supported release lacks the capability, delete +the projection module and the capability check, and leave the reader. + +The cross-version suite derives the old client's list by removing this capability from the +baseline's own list, per the rule above, so the downgrade stays exercised after a release ships +it. + ## Known debt: JSON-RPC errors drop Node's string code An error raised on an SSH host crosses the relay as JSON-RPC, and diff --git a/docs/site/content/docs/browser/profiles.mdx b/docs/site/content/docs/browser/profiles.mdx index 102dd1e443d..fd9af7176e2 100644 --- a/docs/site/content/docs/browser/profiles.mdx +++ b/docs/site/content/docs/browser/profiles.mdx @@ -9,7 +9,9 @@ Browser-use profiles let you run the Orca browser with a specific identity — a 1. Open [Settings → Browser → Profiles](/docs/settings). 1. Click **Add profile**, give it a name. 1. Optionally seed it with cookies, a user-agent, and a viewport size. -1. Every profile presents Electron's own user agent. Orca no longer rewrites it to look like Chrome, because Cloudflare Turnstile rejects a Chrome-shaped UA that sends no client hints and accepts a declared Electron client. The only exception is Google's sign-in hosts, where Orca presents a Firefox identity so Google issues cookies bound to the embedded browser. A **native user agent** profile (`orca tab profile create --no-ua-spoof`) also skips that Google exception. +1. Default profiles remove Orca and Electron tokens from the browser engine's user agent, preserving the Chrome-shaped identity expected by imported sessions. This focused compatibility measure does not make the embedded browser identical to Chrome. Google sign-in hosts use a scoped Firefox identity. If a site rejects the cleaned identity, including some Cloudflare-protected sites, create a profile that keeps the **native Electron user agent** instead. + +You can also create a no-spoof profile from the CLI with `orca tab profile create --no-ua-spoof` when you script browser setup. ## Cookie import and Google sign-in diff --git a/docs/site/content/docs/cli/orchestration.mdx b/docs/site/content/docs/cli/orchestration.mdx index a8db0b06782..adad28a76dc 100644 --- a/docs/site/content/docs/cli/orchestration.mdx +++ b/docs/site/content/docs/cli/orchestration.mdx @@ -116,12 +116,14 @@ orca orchestration dispatch --task --to --inject --json - Default `check` is the bound Run's oldest unacked Delivery (FIFO). Replay until `--ack`. - `--peek` / `--all` do not consume mail. - Group addresses: `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, `@gemini`, `@droid`, `@grok`, `@cursor`, `@worktree:` — never for `worker_done` / heartbeat. +- Every group except `@worktree:` means the live Dispatches of the sender's own Run, delivered to their Dispatch mailboxes (or child Run mailboxes for nested coordinators). A sender in no Run is refused; `--run` must match the audience and never grants membership. +- Run groups exclude their owning coordinator. A worker raising a blocker sends to `run:`. `@worktree:` includes coordinators in that workspace. - Quote PowerShell group addresses: `--to "@all"`. ```bash orca orchestration send --to @all --subject "Heads up" --body "Pausing dispatches for a review." --json orca orchestration send --to @idle --subject "Anyone free?" --json -orca orchestration send --to @codex --subject "Codex agents only" --json +orca orchestration send --to @codex --subject "Codex workers in this Run only" --json ``` While a wait is active, the CLI emits small JSON heartbeat lines to stderr every 15 seconds. Stdout remains the final command result. diff --git a/mobile/app/about.tsx b/mobile/app/about.tsx index 70a1effa44b..7328e6d9b8d 100644 --- a/mobile/app/about.tsx +++ b/mobile/app/about.tsx @@ -1,11 +1,7 @@ -import { View, Text, StyleSheet, Pressable, Linking, Platform } from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { Linking, Platform } from 'react-native' import { useRouter } from 'expo-router' -import { ChevronLeft, Globe } from 'lucide-react-native' -import Svg, { Path } from 'react-native-svg' import Constants from 'expo-constants' -import { OrcaLogo } from '../src/components/OrcaLogo' -import { colors, spacing, typography } from '../src/theme/mobile-theme' +import AboutScreen from '../src/settings/about-screen' // Why: read version + native build identifier from expo-constants at // runtime so the About screen never drifts out of sync with app.json. @@ -20,148 +16,13 @@ function getVersionLabel(): string { return build ? `v${version} (${build})` : `v${version}` } -function GithubIcon({ size = 16, color = colors.textSecondary }) { - return ( - - - - ) -} - -function XIcon({ size = 16, color = colors.textSecondary }) { - return ( - - - - ) -} - -export default function AboutScreen() { +export default function NativeAboutRoute() { const router = useRouter() - const insets = useSafeAreaInsets() - return ( - - - router.back()}> - - - About - - - - - Orca - Open-source agent IDE for 100x builders - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => void Linking.openURL('https://onOrca.dev')} - > - - onOrca.dev - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => void Linking.openURL('https://github.com/stablyai/orca')} - > - - stablyai/orca - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => void Linking.openURL('https://x.com/orca_build')} - > - - @orca_build - - - - {getVersionLabel()} - + router.back()} + openExternal={(url) => Linking.openURL(url)} + versionLabel={getVersionLabel()} + /> ) } - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - padding: spacing.lg - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginBottom: spacing.xl - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - brand: { - alignItems: 'center', - paddingVertical: spacing.xl, - marginBottom: spacing.lg - }, - brandName: { - fontSize: 22, - fontWeight: '800', - color: colors.textPrimary, - marginTop: spacing.sm - }, - brandSub: { - fontSize: 13, - color: colors.textMuted, - marginTop: spacing.xs - }, - section: { - backgroundColor: colors.bgPanel, - borderRadius: 12, - overflow: 'hidden' - }, - row: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - rowPressed: { - backgroundColor: colors.bgRaised - }, - rowLabel: { - flex: 1, - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - rowValue: { - flex: 1, - textAlign: 'right', - fontSize: typography.bodySize, - color: colors.textSecondary - }, - separator: { - height: StyleSheet.hairlineWidth, - backgroundColor: colors.borderSubtle, - marginHorizontal: spacing.md - }, - versionText: { - marginTop: spacing.lg, - textAlign: 'center', - fontSize: typography.metaSize, - color: colors.textMuted - } -}) diff --git a/mobile/app/browser-settings.tsx b/mobile/app/browser-settings.tsx index a8e9c518fe4..d47944e5501 100644 --- a/mobile/app/browser-settings.tsx +++ b/mobile/app/browser-settings.tsx @@ -1,163 +1 @@ -import { useCallback, useEffect, useState } from 'react' -import { Pressable, ScrollView, StyleSheet, Text, View } from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' -import { useRouter } from 'expo-router' -import { ChevronLeft, ChevronRight, Globe } from 'lucide-react-native' -import { PickerModal, type PickerOption } from '../src/components/PickerModal' -import { - loadTerminalLinkOpenMode, - saveTerminalLinkOpenMode, - type MobileTerminalLinkOpenMode -} from '../src/storage/preferences' -import { colors, radii, spacing, typography } from '../src/theme/mobile-theme' - -const LINK_MODE_OPTIONS: PickerOption[] = [ - { - value: 'orca-browser', - label: 'Orca browser on desktop', - subtitle: 'Open in the streamed browser from your paired desktop.' - }, - { - value: 'phone-browser', - label: 'Phone browser', - subtitle: 'Open in Safari, Chrome, or another browser on this phone.' - } -] - -function linkModeLabel(mode: MobileTerminalLinkOpenMode): string { - return ( - LINK_MODE_OPTIONS.find((option) => option.value === mode)?.label ?? LINK_MODE_OPTIONS[0]!.label - ) -} - -export default function BrowserSettingsScreen(): React.JSX.Element { - const router = useRouter() - const insets = useSafeAreaInsets() - const [linkMode, setLinkMode] = useState('orca-browser') - const [pickerOpen, setPickerOpen] = useState(false) - - useEffect(() => { - void loadTerminalLinkOpenMode().then(setLinkMode) - }, []) - - const selectLinkMode = useCallback((mode: MobileTerminalLinkOpenMode) => { - setLinkMode(mode) - void saveTerminalLinkOpenMode(mode) - }, []) - - return ( - - - router.back()}> - - - Browser - - - - LINKS - - Choose where HTTP(S) links tapped in terminal output open. - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => setPickerOpen(true)} - > - - - Open terminal links - {linkModeLabel(linkMode)} - - - - - - - - visible={pickerOpen} - title="Open terminal links" - options={LINK_MODE_OPTIONS} - selected={linkMode} - onSelect={selectLinkMode} - onClose={() => setPickerOpen(false)} - /> - - ) -} - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - paddingHorizontal: spacing.lg, - paddingTop: 0 - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginTop: spacing.sm, - marginBottom: spacing.lg - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - scrollContent: { - paddingBottom: spacing.xl - }, - groupHeading: { - fontSize: 11, - fontWeight: '600', - color: colors.textMuted, - letterSpacing: 0.5, - marginBottom: spacing.xs, - paddingHorizontal: spacing.xs - }, - groupDescription: { - fontSize: typography.bodySize - 1, - color: colors.textSecondary, - lineHeight: 20, - paddingHorizontal: spacing.xs - }, - section: { - backgroundColor: colors.bgPanel, - borderRadius: radii.card, - overflow: 'hidden' - }, - sectionTopGap: { - marginTop: spacing.sm - }, - row: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - rowPressed: { - backgroundColor: colors.bgRaised - }, - rowContent: { - flex: 1 - }, - rowLabel: { - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - rowSublabel: { - fontSize: typography.bodySize - 2, - color: colors.textSecondary, - marginTop: 2 - } -}) +export { default } from '../src/settings/browser-settings-screen' diff --git a/mobile/app/connection-log.tsx b/mobile/app/connection-log.tsx index 1ab79c50269..f7a1e80c231 100644 --- a/mobile/app/connection-log.tsx +++ b/mobile/app/connection-log.tsx @@ -1,12 +1,7 @@ import { useCallback, useEffect, useMemo, useState, useSyncExternalStore } from 'react' -import { View, Text, StyleSheet, Pressable, Platform } from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { View, Text, Pressable } from 'react-native' import { useLocalSearchParams, useRouter } from 'expo-router' import * as Clipboard from 'expo-clipboard' -import Constants from 'expo-constants' -import { ChevronLeft, Copy, Check, Send } from 'lucide-react-native' -import { colors, spacing, typography } from '../src/theme/mobile-theme' -import { ConnectionLog } from '../src/components/ConnectionLog' import { loadHosts } from '../src/transport/host-store' import { connectionLogStore } from '../src/transport/persisted-connection-log-store' import { useHostClient, useRpcClientContext } from '../src/transport/client-context' @@ -14,22 +9,15 @@ import { useConnectionPathStatus, useReconnectAttempt } from '../src/transport/client-context-connection-metrics' -import { buildConnectionDiagnosticsReport } from '../src/diagnostics/connection-diagnostics-report' -import { - diagnoseConnection, - getReportableConnectionIncidentId -} from '../src/diagnostics/connection-diagnostics-analysis' -import { submitConnectionDiagnostics } from '../src/diagnostics/connection-diagnostics-submission' +import { useHostStatusGates } from '../src/transport/host-status-gates' +import { ConnectionDiagnosticsScreen } from '../src/diagnostics/connection-diagnostics-screen' +import { createNativeDiagnosticsOperations } from '../src/diagnostics/native-diagnostics-operations' import { readHydratedConnectionLog, - readConnectionDiagnosticsSnapshot, resolveDiagnosticsHostId, - getDiagnosticsSubmissionState, - updateDiagnosticsSubmissionState, - type DiagnosticsSubmissionStates + type DiagnosticsHostSelection } from '../src/diagnostics/connection-diagnostics-screen-data' -import { useHostStatusGates } from '../src/transport/host-status-gates' -import { loadHostAppVersion } from '../src/transport/host-app-version-store' +import { connectionDiagnosticsScreenStyles as styles } from '../src/diagnostics/connection-diagnostics-screen-styles' import type { ConnectionLogEntry, HostProfile } from '../src/transport/types' // Why: getSnapshot must be referentially stable when there's no data — @@ -37,30 +25,22 @@ import type { ConnectionLogEntry, HostProfile } from '../src/transport/types' const EMPTY_ENTRIES: readonly ConnectionLogEntry[] = [] // Why: reading the log is most needed while a host is failing, so this -// screen also *acquires* the host client — opening it kicks a dial and the +// route also *acquires* the host client — opening it kicks a dial and the // log fills live instead of showing a stale tail. -export default function ConnectionLogScreen() { +export default function NativeConnectionLogRoute() { const clientContext = useRpcClientContext() const router = useRouter() const params = useLocalSearchParams<{ hostId?: string }>() - const insets = useSafeAreaInsets() const routeKey = useMemo(() => ({}), [params.hostId]) const [hosts, setHosts] = useState([]) - const [manualSelection, setManualSelection] = useState<{ - hostId: string - requestedHostId: string | undefined - routeKey: object - } | null>(null) - const [copiedHostId, setCopiedHostId] = useState(null) - const [submissionStates, setSubmissionStates] = useState({}) + const [manualSelection, setManualSelection] = useState(null) useEffect(() => { let stale = false void loadHosts().then((loaded) => { - if (stale) { - return + if (!stale) { + setHosts(loaded) } - setHosts(loaded) }) return () => { stale = true @@ -68,9 +48,9 @@ export default function ConnectionLogScreen() { }, []) const selectedId = resolveDiagnosticsHostId(hosts, params.hostId, manualSelection, routeKey) - const selected = hosts.find((h) => h.id === selectedId) ?? null + const selected = hosts.find((host) => host.id === selectedId) ?? null const { client, state } = useHostClient(selected?.id) - const { desktopAppVersion: liveDesktopAppVersion } = useHostStatusGates({ + const { desktopAppVersion } = useHostStatusGates({ hostId: selected?.id, client, connState: state @@ -94,314 +74,47 @@ export default function ConnectionLogScreen() { [selectedId] ) const entries = useSyncExternalStore(subscribe, getSnapshot) - const diagnosis = selected - ? diagnoseConnection({ endpoint: selected.endpoint, state, activePath, pendingPath, entries }) - : null - const incidentId = selected - ? getReportableConnectionIncidentId({ - endpoint: selected.endpoint, - state, - activePath, - pendingPath, - entries - }) - : null - const submissionKey = selected && incidentId ? `${selected.id}:${incidentId}` : null - const submissionState = getDiagnosticsSubmissionState(submissionStates, submissionKey) - const copied = copiedHostId === selectedId - - const copyDiagnostics = useCallback(async () => { - if (!selected) { - return - } - const desktopAppVersion = liveDesktopAppVersion ?? (await loadHostAppVersion(selected.id)) - const snapshot = await readConnectionDiagnosticsSnapshot( - clientContext, - connectionLogStore, - selected.id - ) - const report = buildConnectionDiagnosticsReport({ - hostName: selected.name, - endpoint: selected.endpoint, - state: snapshot.state, - reconnectAttempts: snapshot.reconnectAttempts, - lastConnectedAt: snapshot.lastConnectedAt, - platform: `${Platform.OS} ${Platform.Version ?? ''}`.trim(), - appVersion: Constants.expoConfig?.version ?? 'unknown', - desktopAppVersion, - entries: snapshot.entries, - activePath: snapshot.activePath, - pendingPath: snapshot.pendingPath - }) - await Clipboard.setStringAsync(report) - setCopiedHostId(selected.id) - setTimeout(() => setCopiedHostId((hostId) => (hostId === selected.id ? null : hostId)), 2000) - }, [selected, liveDesktopAppVersion, clientContext]) - - const sendDiagnostics = useCallback(async () => { - if (!selected || !submissionKey || submissionState === 'sending') { - return - } - const startedKey = submissionKey - setSubmissionStates((states) => updateDiagnosticsSubmissionState(states, startedKey, 'sending')) - const appVersion = Constants.expoConfig?.version ?? 'unknown' - const platform = `${Platform.OS} ${Platform.Version ?? ''}`.trim() - const desktopAppVersion = liveDesktopAppVersion ?? (await loadHostAppVersion(selected.id)) - const snapshot = await readConnectionDiagnosticsSnapshot( - clientContext, - connectionLogStore, - selected.id - ) - const currentIncidentId = getReportableConnectionIncidentId({ - endpoint: selected.endpoint, - state: snapshot.state, - activePath: snapshot.activePath, - pendingPath: snapshot.pendingPath, - entries: snapshot.entries - }) - if (`${selected.id}:${currentIncidentId ?? ''}` !== startedKey) { - setSubmissionStates((states) => updateDiagnosticsSubmissionState(states, startedKey, null)) - return - } - const report = buildConnectionDiagnosticsReport({ - hostName: selected.name, - endpoint: selected.endpoint, - state: snapshot.state, - reconnectAttempts: snapshot.reconnectAttempts, - lastConnectedAt: snapshot.lastConnectedAt, - platform, - appVersion, - desktopAppVersion, - entries: snapshot.entries, - activePath: snapshot.activePath, - pendingPath: snapshot.pendingPath - }) - const result = await submitConnectionDiagnostics({ report, appVersion, platform }) - setSubmissionStates((states) => - updateDiagnosticsSubmissionState(states, startedKey, result.ok ? 'sent' : 'failed') - ) - }, [selected, submissionKey, submissionState, liveDesktopAppVersion, clientContext]) + const device = useMemo( + () => + selected + ? createNativeDiagnosticsOperations(selected, clientContext, desktopAppVersion) + : null, + [selected, clientContext, desktopAppVersion] + ) return ( - - - router.back()}> - - - Network diagnostics - - - {hosts.length > 1 && ( - - {hosts.map((host) => ( - - setManualSelection({ hostId: host.id, requestedHostId: params.hostId, routeKey }) - } - > - Clipboard.setStringAsync(report)} + onBack={() => router.back()} + hostPicker={ + hosts.length > 1 ? ( + + {hosts.map((host) => ( + + setManualSelection({ hostId: host.id, requestedHostId: params.hostId, routeKey }) + } > - {host.name} - - - ))} - - )} - - {selected ? ( - <> - - - {state} - {reconnectAttempts > 0 ? ` · attempt ${reconnectAttempts}` : ''} - - void copyDiagnostics()}> - {copied ? ( - - ) : ( - - )} - {copied ? 'Copied' : 'Copy report'} - + + {host.name} + + + ))} - {diagnosis && ( - - What this suggests - {diagnosis.likelyCause} - {diagnosis.nextStep} - {diagnosis.reportability === 'orca-relay' && ( - <> - - Sends a size-limited redacted report including host name, endpoint, versions, - connection state, and events—never terminal contents or credentials. - - void sendDiagnostics()} - disabled={submissionState === 'sending'} - > - {submissionState === 'sent' ? ( - - ) : ( - - )} - - {submissionState === 'sending' - ? 'Sending…' - : submissionState === 'sent' - ? 'Diagnostics sent' - : submissionState === 'failed' - ? 'Retry sending' - : 'Send diagnostics to Orca'} - - - - )} - - )} - {entries.length > 0 ? ( - - ) : ( - - No connection events yet. Events appear as the app dials this host. - - )} - - ) : ( - No paired hosts. - )} - + ) : null + } + /> ) } - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - padding: spacing.lg - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginBottom: spacing.lg - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - hostPicker: { - flexDirection: 'row', - flexWrap: 'wrap', - gap: spacing.sm, - marginBottom: spacing.md - }, - hostChip: { - paddingVertical: spacing.xs + 2, - paddingHorizontal: spacing.md, - borderRadius: 16, - backgroundColor: colors.bgRaised - }, - hostChipActive: { - backgroundColor: colors.bgPanel, - borderWidth: 1, - borderColor: colors.borderSubtle - }, - hostChipText: { - fontSize: typography.metaSize, - color: colors.textSecondary, - maxWidth: 160 - }, - hostChipTextActive: { - color: colors.textPrimary, - fontWeight: '600' - }, - statusRow: { - flexDirection: 'row', - alignItems: 'center', - justifyContent: 'space-between', - marginBottom: spacing.sm - }, - statusText: { - fontSize: typography.metaSize, - color: colors.textSecondary - }, - diagnosisCard: { - backgroundColor: colors.bgPanel, - borderWidth: StyleSheet.hairlineWidth, - borderColor: colors.borderSubtle, - borderRadius: 10, - padding: spacing.md, - marginBottom: spacing.md - }, - diagnosisHeading: { - fontSize: typography.metaSize, - fontWeight: '600', - color: colors.textPrimary, - marginBottom: spacing.xs - }, - diagnosisText: { - fontSize: typography.metaSize, - color: colors.textPrimary, - lineHeight: 18 - }, - diagnosisNext: { - fontSize: typography.metaSize, - color: colors.textSecondary, - lineHeight: 18, - marginTop: spacing.xs - }, - privacyHint: { - marginTop: spacing.sm, - fontSize: 11, - lineHeight: 15, - color: colors.textMuted - }, - sendButton: { - marginTop: spacing.md, - alignSelf: 'flex-start', - flexDirection: 'row', - alignItems: 'center', - gap: spacing.xs, - paddingVertical: spacing.sm, - paddingHorizontal: spacing.md, - borderRadius: 8, - backgroundColor: colors.bgRaised - }, - sendButtonText: { - fontSize: typography.metaSize, - fontWeight: '600', - color: colors.textPrimary - }, - copyButton: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.xs + 2, - paddingVertical: spacing.xs + 2, - paddingHorizontal: spacing.md, - borderRadius: 8, - backgroundColor: colors.bgRaised - }, - copyButtonText: { - fontSize: typography.metaSize, - fontWeight: '600', - color: colors.textPrimary - }, - emptyText: { - fontSize: typography.metaSize, - color: colors.textMuted, - lineHeight: 18 - } -}) diff --git a/mobile/app/native-chat-settings.tsx b/mobile/app/native-chat-settings.tsx index 1e2ebadd9ef..dc428e78809 100644 --- a/mobile/app/native-chat-settings.tsx +++ b/mobile/app/native-chat-settings.tsx @@ -1,126 +1 @@ -import { View, Text, StyleSheet, Pressable, ScrollView, Switch } from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' -import { useRouter } from 'expo-router' -import { ChevronLeft } from 'lucide-react-native' -import { colors, radii, spacing, typography } from '../src/theme/mobile-theme' -import { useMobileDefaultSessionViewPreference } from '../src/session/use-mobile-default-session-view-preference' - -export default function NativeChatSettingsScreen() { - const router = useRouter() - const insets = useSafeAreaInsets() - - const { defaultView, setDefaultView } = useMobileDefaultSessionViewPreference() - const chatDefault = defaultView === 'chat' - - return ( - - - router.back()} - > - - - Chat UI - - - - DEFAULT VIEW - - Choose how supported agent sessions (Claude, Codex, and other chat-capable agents) open on - this device. Terminal shows the raw CLI; Chat UI shows a chat interface like the desktop - app. You can still switch any individual session from its long-press menu. - - - - - Open sessions in Chat UI - {chatDefault ? 'On' : 'Off'} - - setDefaultView(next ? 'chat' : 'terminal')} - trackColor={{ false: colors.bgRaised, true: colors.textSecondary }} - thumbColor={colors.textPrimary} - /> - - - - - ) -} - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - paddingHorizontal: spacing.lg - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginTop: spacing.sm, - marginBottom: spacing.lg - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - groupHeading: { - fontSize: 11, - fontWeight: '600', - color: colors.textMuted, - letterSpacing: 0.5, - marginBottom: spacing.xs, - paddingHorizontal: spacing.xs - }, - groupDescription: { - fontSize: typography.bodySize - 1, - color: colors.textSecondary, - lineHeight: 20, - paddingHorizontal: spacing.xs - }, - section: { - backgroundColor: colors.bgPanel, - borderRadius: radii.card, - overflow: 'hidden' - }, - sectionTopGap: { - marginTop: spacing.sm - }, - row: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - rowContent: { - flex: 1 - }, - rowLabel: { - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - rowSublabel: { - fontSize: typography.bodySize - 2, - color: colors.textSecondary, - marginTop: 2 - } -}) +export { default } from '../src/settings/native-chat-settings-screen' diff --git a/mobile/app/notifications.tsx b/mobile/app/notifications.tsx index d9696251a94..d6566f66ac7 100644 --- a/mobile/app/notifications.tsx +++ b/mobile/app/notifications.tsx @@ -1,178 +1,12 @@ -import { useState, useCallback, useEffect } from 'react' -import { AppState, Linking, View, Text, StyleSheet, Pressable, Switch } from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' -import { useRouter, useFocusEffect } from 'expo-router' -import { ChevronLeft } from 'lucide-react-native' -import { colors, spacing, typography } from '../src/theme/mobile-theme' -import { - loadPushNotificationsEnabled, - savePushNotificationsEnabled -} from '../src/storage/preferences' -import { - ensureNotificationPermissions, - getNotificationPermissionState, - type NotificationPermissionState -} from '../src/notifications/mobile-notifications' - -const DEFAULT_PERMISSION_STATE: NotificationPermissionState = { - granted: false, - status: 'undetermined', - canAskAgain: true, - authorizationReflectsUserChoice: false -} - -export default function NotificationsScreen() { +import { useRouter } from 'expo-router' +import NotificationsScreen from '../src/settings/notification-settings-screen' +import { nativeNotificationSettingsOperations } from '../src/settings/native-notification-settings-operations' +export default function NativeNotificationsRoute() { const router = useRouter() - const insets = useSafeAreaInsets() - const [pushEnabled, setPushEnabled] = useState(false) - const [permissionState, setPermissionState] = useState(DEFAULT_PERMISSION_STATE) - - const refreshSettings = useCallback(async () => { - const [enabled, permission] = await Promise.all([ - loadPushNotificationsEnabled(), - getNotificationPermissionState() - ]) - setPushEnabled(enabled) - setPermissionState(permission) - }, []) - - useFocusEffect( - useCallback(() => { - void refreshSettings() - }, [refreshSettings]) - ) - - useEffect(() => { - const subscription = AppState.addEventListener('change', (state) => { - if (state === 'active') { - void refreshSettings() - } - }) - return () => subscription.remove() - }, [refreshSettings]) - - const togglePush = async (value: boolean) => { - if (value) { - const granted = await ensureNotificationPermissions() - const permission = await getNotificationPermissionState() - setPermissionState(permission) - if (!granted) { - setPushEnabled(false) - await savePushNotificationsEnabled(false) - return - } - } - setPushEnabled(value) - await savePushNotificationsEnabled(value) - } - - const switchEnabled = pushEnabled && permissionState.granted - const notificationsBlocked = permissionState.status === 'denied' - const hint = notificationsBlocked - ? 'Notifications are disabled in system settings.' - : 'Get notified on this device when an agent needs your input or finishes a task.' - return ( - - - router.back()}> - - - Notifications - - - - - Agent notifications - void togglePush(v)} - trackColor={{ false: colors.bgRaised, true: colors.textSecondary }} - thumbColor={colors.textPrimary} - /> - - {hint} - {notificationsBlocked && ( - [ - styles.settingsButton, - pressed && styles.settingsButtonPressed - ]} - onPress={() => void Linking.openSettings()} - > - Open Settings - - )} - - + router.back()} + /> ) } - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - padding: spacing.lg - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginBottom: spacing.xl - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - section: { - backgroundColor: colors.bgPanel, - borderRadius: 12, - overflow: 'hidden' - }, - row: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - rowLabel: { - flex: 1, - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - hint: { - fontSize: typography.metaSize, - color: colors.textMuted, - lineHeight: 18, - paddingHorizontal: spacing.md + 2, - paddingBottom: spacing.md - }, - settingsButton: { - alignSelf: 'flex-start', - marginHorizontal: spacing.md + 2, - marginBottom: spacing.md, - paddingVertical: spacing.xs, - paddingHorizontal: spacing.sm, - borderRadius: 8, - backgroundColor: colors.bgRaised - }, - settingsButtonPressed: { - opacity: 0.6 - }, - settingsButtonText: { - color: colors.textPrimary, - fontSize: typography.metaSize, - fontWeight: '600' - } -}) diff --git a/mobile/app/settings.tsx b/mobile/app/settings.tsx index 0e74b516a87..eb954a8e83d 100644 --- a/mobile/app/settings.tsx +++ b/mobile/app/settings.tsx @@ -1,321 +1,16 @@ -import { useCallback, useRef, useState } from 'react' -import { - View, - Text, - StyleSheet, - Pressable, - Linking, - ActivityIndicator, - ScrollView -} from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' -import { useFocusEffect, useRouter } from 'expo-router' -import { - ChevronLeft, - ChevronRight, - Info, - Bell, - Wrench, - Shield, - LifeBuoy, - Mic, - Globe, - MessageSquare, - Terminal as TerminalIcon, - KeyRound -} from 'lucide-react-native' -import { colors, radii, spacing, typography } from '../src/theme/mobile-theme' -import { - loadPendingHostCredentialCleanup, - subscribePendingHostCredentialCleanup -} from '../src/transport/host-credential-cleanup' -import { retryPendingHostCredentialCleanup } from '../src/transport/host-store' +import { Linking } from 'react-native' +import { useRouter } from 'expo-router' +import SettingsMenuScreen from '../src/settings/settings-menu-screen' +import { PendingCredentialCleanupCard } from '../src/settings/pending-credential-cleanup-card' -export default function SettingsScreen() { +export default function NativeSettingsRoute() { const router = useRouter() - const insets = useSafeAreaInsets() - const [pendingCredentialIds, setPendingCredentialIds] = useState([]) - const [credentialStorageUnreadable, setCredentialStorageUnreadable] = useState(false) - const [retryingCredentialCleanup, setRetryingCredentialCleanup] = useState(false) - const [credentialRetryFailed, setCredentialRetryFailed] = useState(false) - const credentialRefreshGenerationRef = useRef(0) - - useFocusEffect( - useCallback(() => { - let active = true - setCredentialRetryFailed(false) - const refresh = () => { - const generation = ++credentialRefreshGenerationRef.current - void loadPendingHostCredentialCleanup().then((state) => { - if (active && generation === credentialRefreshGenerationRef.current) { - setPendingCredentialIds(state.ids) - setCredentialStorageUnreadable(state.storageUnreadable) - // Why: neutral copy once the queue is confirmed empty so a later - // pending set does not inherit a previous Retry failure message. - if (state.ids.length === 0 && !state.storageUnreadable) { - setCredentialRetryFailed(false) - } - } - }) - } - const unsubscribe = subscribePendingHostCredentialCleanup(refresh) - refresh() - return () => { - active = false - credentialRefreshGenerationRef.current += 1 - unsubscribe() - } - }, []) - ) - - const retryCredentialCleanup = useCallback(async () => { - if (retryingCredentialCleanup) { - return - } - setCredentialRetryFailed(false) - setRetryingCredentialCleanup(true) - try { - const result = await retryPendingHostCredentialCleanup() - setPendingCredentialIds(result.remainingIds) - setCredentialStorageUnreadable(result.storageUnreadable) - setCredentialRetryFailed(result.remainingIds.length > 0 || result.storageUnreadable) - } catch { - setCredentialRetryFailed(true) - } finally { - setRetryingCredentialCleanup(false) - } - }, [retryingCredentialCleanup]) - - const pendingCredentialCount = pendingCredentialIds.length - // Why: show the cleanup card whenever cleanup is pending OR the durable queue - // is unreadable — an unreadable queue can hide an orphaned token, so keep a - // retry affordance rather than a silently-empty (hidden) section. - const showCredentialCleanup = pendingCredentialCount > 0 || credentialStorageUnreadable - return ( - - - router.back()}> - - - Settings - - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/terminal-settings')} - > - - Terminal - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/native-chat-settings')} - > - - Chat UI - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/browser-settings')} - > - - Browser - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/voice-settings')} - > - - Voice - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/notifications')} - > - - Notifications - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/troubleshoot')} - > - - Troubleshooting - - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => router.push('/about')} - > - - About - - - - - {showCredentialCleanup ? ( - - - - - Pairing credential cleanup - - {credentialRetryFailed - ? "Cleanup still couldn't be confirmed. Try again later." - : pendingCredentialCount > 0 - ? `Couldn't confirm cleanup for ${pendingCredentialCount} credential${pendingCredentialCount === 1 ? '' : 's'} on this device.` - : "Couldn't check cleanup status on this device. Retry to be safe."} - - - [ - styles.retryButton, - pressed && !retryingCredentialCleanup && styles.rowPressed - ]} - onPress={() => void retryCredentialCleanup()} - > - {retryingCredentialCleanup ? ( - - ) : ( - Retry - )} - - - - ) : null} - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => void Linking.openURL('https://www.onorca.dev/privacy')} - > - - Privacy Policy - - - [styles.row, pressed && styles.rowPressed]} - onPress={() => void Linking.openURL('https://github.com/stablyai/orca/issues')} - > - - Support - - - - + router.push(route)} + openExternal={(url) => Linking.openURL(url)} + > + + ) } - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - paddingHorizontal: spacing.lg - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginBottom: spacing.xl - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - section: { - backgroundColor: colors.bgPanel, - borderRadius: 12, - overflow: 'hidden' - }, - sectionSpacer: { - marginTop: spacing.md - }, - row: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - rowPressed: { - backgroundColor: colors.bgRaised - }, - rowLabel: { - flex: 1, - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - credentialCleanupRow: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - credentialCleanupCopy: { - flex: 1, - gap: spacing.xs - }, - credentialCleanupTitle: { - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - rowHint: { - fontSize: typography.metaSize, - color: colors.textSecondary, - lineHeight: 17 - }, - retryButton: { - width: 72, - height: 32, - borderRadius: radii.button, - backgroundColor: colors.bgRaised, - alignItems: 'center', - justifyContent: 'center' - }, - retryButtonText: { - fontSize: typography.metaSize, - fontWeight: '600', - color: colors.textPrimary - }, - separator: { - height: StyleSheet.hairlineWidth, - backgroundColor: colors.borderSubtle, - marginHorizontal: spacing.md - } -}) diff --git a/mobile/app/troubleshoot.tsx b/mobile/app/troubleshoot.tsx index 63fc0ddd477..d07368b21f9 100644 --- a/mobile/app/troubleshoot.tsx +++ b/mobile/app/troubleshoot.tsx @@ -1,276 +1,18 @@ -import { useState, useCallback, useRef } from 'react' -import { View, Text, Pressable, ScrollView, ActivityIndicator, Platform } from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' import { useRouter } from 'expo-router' -import { - ChevronLeft, - ChevronDown, - ChevronUp, - Activity, - CheckCircle2, - ScrollText, - XCircle, - AlertTriangle -} from 'lucide-react-native' -import { colors, spacing } from '../src/theme/mobile-theme' -import { loadHosts } from '../src/transport/host-store' -import { - startDiagnosticFetchTimeout, - type DiagnosticFetchTimeout -} from '../src/diagnostics/diagnostic-fetch-timeout' -import { - formatEndpoint, - testHostReachability, - unreachableHostDetail -} from '../src/diagnostics/host-reachability' -import { troubleshootCommonIssues } from '../src/diagnostics/troubleshoot-common-issues' -import { troubleshootScreenStyles as styles } from '../src/diagnostics/troubleshoot-screen-styles' +import { TroubleshootView } from '../src/diagnostics/troubleshoot-view' +import { useTroubleshootDiagnostics } from '../src/diagnostics/use-troubleshoot-diagnostics' -type DiagnosticStatus = 'idle' | 'running' | 'done' - -type CheckResult = { - label: string - status: 'pass' | 'fail' | 'warn' - detail: string -} - -function StatusIcon({ status }: { status: CheckResult['status'] }) { - switch (status) { - case 'pass': - return - case 'fail': - return - case 'warn': - return - } -} - -export default function TroubleshootScreen() { +export default function NativeTroubleshootRoute() { const router = useRouter() - const insets = useSafeAreaInsets() - const [expandedId, setExpandedId] = useState(null) - const [diagnosticStatus, setDiagnosticStatus] = useState('idle') - const [checks, setChecks] = useState([]) - const abortRef = useRef(false) - const diagnosticRunRef = useRef(0) - const activeInternetCheckRef = useRef(null) - - const setTroubleshootRootRef = useCallback((node: View | null): void => { - if (node !== null) { - return - } - // Why: diagnostics can outlive the screen; cancel the active run when the - // route detaches without a passive cleanup-only Effect. - abortRef.current = true - diagnosticRunRef.current += 1 - activeInternetCheckRef.current?.dispose() - activeInternetCheckRef.current = null - }, []) - - const toggleSection = useCallback((id: string) => { - setExpandedId((prev) => (prev === id ? null : id)) - }, []) - - const runDiagnostics = useCallback(async () => { - const runId = diagnosticRunRef.current + 1 - diagnosticRunRef.current = runId - abortRef.current = false - activeInternetCheckRef.current?.dispose() - activeInternetCheckRef.current = null - setDiagnosticStatus('running') - setChecks([]) - - const results: CheckResult[] = [] - const isCurrentRun = () => !abortRef.current && diagnosticRunRef.current === runId - - try { - const hosts = await loadHosts() - results.push( - hosts.length > 0 - ? { label: 'Paired hosts', status: 'pass', detail: `${hosts.length} paired` } - : { label: 'Paired hosts', status: 'fail', detail: 'None — scan a QR to pair' } - ) - } catch { - results.push({ label: 'Paired hosts', status: 'warn', detail: 'Could not read host data' }) - } - - if (!isCurrentRun()) { - return - } - setChecks([...results]) - - const internetCheck = startDiagnosticFetchTimeout(5000) - activeInternetCheckRef.current = internetCheck - try { - const resp = await fetch('https://dns.google/resolve?name=example.com&type=A', { - signal: internetCheck.signal - }) - if (!isCurrentRun()) { - return - } - results.push( - resp.ok - ? { label: 'Internet', status: 'pass', detail: 'Connected' } - : { label: 'Internet', status: 'warn', detail: 'Unexpected response' } - ) - } catch { - if (!isCurrentRun()) { - return - } - results.push({ label: 'Internet', status: 'fail', detail: 'No connection' }) - } finally { - internetCheck.dispose() - if (activeInternetCheckRef.current === internetCheck) { - activeInternetCheckRef.current = null - } - } - - if (!isCurrentRun()) { - return - } - setChecks([...results]) - - try { - const hosts = await loadHosts() - for (const host of hosts) { - if (!isCurrentRun()) { - return - } - const reachable = await testHostReachability(host.endpoint) - if (!isCurrentRun()) { - return - } - results.push({ - label: host.name, - status: reachable ? 'pass' : 'fail', - detail: reachable - ? `Reachable at ${formatEndpoint(host.endpoint)}` - : unreachableHostDetail(host.endpoint) - }) - setChecks([...results]) - } - } catch { - results.push({ label: 'Hosts', status: 'warn', detail: 'Could not test' }) - } - - if (!isCurrentRun()) { - return - } - - results.push({ - label: 'Platform', - status: 'pass', - detail: `${Platform.OS} ${Platform.Version ?? ''}` - }) - - setChecks([...results]) - setDiagnosticStatus('done') - }, []) - + const { rootRef, diagnosticStatus, checks, runDiagnostics } = useTroubleshootDiagnostics() return ( - - - router.back()}> - - - Troubleshooting - - - - [ - styles.diagnosticButton, - pressed && styles.diagnosticButtonPressed, - diagnosticStatus === 'running' && styles.diagnosticButtonDisabled - ]} - onPress={runDiagnostics} - disabled={diagnosticStatus === 'running'} - > - {diagnosticStatus === 'running' ? ( - - ) : ( - - )} - - {diagnosticStatus === 'running' - ? 'Running…' - : diagnosticStatus === 'done' - ? 'Run again' - : 'Run diagnostics'} - - - - [ - styles.diagnosticButton, - pressed && styles.diagnosticButtonPressed - ]} - onPress={() => router.push('/connection-log')} - > - - View network diagnostics - - - {checks.length > 0 && ( - - {checks.map((check, i) => ( - - {i > 0 && } - - - {check.label} - - {check.detail} - - - - ))} - - )} - - Common issues - - - {troubleshootCommonIssues.map((section, i) => ( - - {i > 0 && } - [styles.accordionHeader, pressed && styles.rowPressed]} - onPress={() => toggleSection(section.id)} - > - {section.icon} - {section.title} - {expandedId === section.id ? ( - - ) : ( - - )} - - {expandedId === section.id && ( - - {section.steps.map((step, j) => ( - - • - {step} - - ))} - - )} - - ))} - - - - - + void runDiagnostics()} + onBack={() => router.back()} + onConnectionLog={() => router.push('/connection-log')} + /> ) } diff --git a/mobile/app/voice-settings.tsx b/mobile/app/voice-settings.tsx index 8648a1d38e5..4ed812e960b 100644 --- a/mobile/app/voice-settings.tsx +++ b/mobile/app/voice-settings.tsx @@ -1,411 +1,25 @@ -import { useCallback, useEffect, useMemo, useState } from 'react' -import { - ActivityIndicator, - Pressable, - ScrollView, - StyleSheet, - Switch, - Text, - View -} from 'react-native' -import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { useEffect, useMemo, useState } from 'react' import { useRouter } from 'expo-router' -import { ChevronLeft, ChevronRight } from 'lucide-react-native' -import { colors, radii, spacing, typography } from '../src/theme/mobile-theme' import { loadHosts } from '../src/transport/host-store' import type { HostProfile } from '../src/transport/types' import { useFocusedSettingsHostClients } from '../src/transport/settings-host-client-connections' -import type { RpcClient } from '../src/transport/rpc-client' -import { BottomDrawer } from '../src/components/BottomDrawer' -import { VoiceModelList } from '../src/components/VoiceModelList' -import { useDictationSetupPoller } from '../src/dictation/use-dictation-setup-poller' -import { - deleteDictationModel, - downloadDictationModel, - fetchDictationSetup, - isModelInFlight, - setDictationConfig, - type MobileSpeechModel, - type MobileSpeechSetup -} from '../src/dictation/mobile-dictation-setup' +import VoiceSettingsScreen from '../src/settings/voice-settings-screen' +import { nativeVoiceSettingsOperations } from '../src/settings/native-voice-settings-operations' -const POLL_INTERVAL_MS = 1500 - -const DICTATION_MODES = [ - { value: 'toggle', label: 'Toggle' }, - { value: 'hold', label: 'Hold' } -] as const - -type ModelBusyAction = { modelId: string; type: 'download' | 'select' | 'delete' } - -export default function VoiceSettingsScreen(): React.JSX.Element { +export default function NativeVoiceSettingsRoute() { const router = useRouter() - const insets = useSafeAreaInsets() - const [hosts, setHosts] = useState([]) useEffect(() => { void loadHosts().then(setHosts) }, []) - const hostIds = useMemo(() => hosts.map((h) => h.id), [hosts]) - const { clients: hostClients, focused: routeFocused } = useFocusedSettingsHostClients(hostIds) - // Voice dictation runs on the paired desktop, so pick the first connected host. - const client: RpcClient | null = useMemo( - () => hostClients.find((entry) => entry.state === 'connected')?.client ?? null, - [hostClients] - ) - - const [setup, setSetup] = useState(null) - const [loading, setLoading] = useState(false) - const [error, setError] = useState(null) - const [busyAction, setBusyAction] = useState(null) - const [modelDrawerOpen, setModelDrawerOpen] = useState(false) - const refresh = useCallback(async (): Promise => { - if (!client) { - return false - } - try { - const next = await fetchDictationSetup(client) - setSetup(next) - setError(null) - return next.models.some(isModelInFlight) - } catch (err) { - setError(err instanceof Error ? err.message : 'Failed to load voice settings') - return undefined - } finally { - setLoading(false) - } - }, [client]) - - const polling = setup?.models.some(isModelInFlight) ?? false - const refreshSetup = useDictationSetupPoller({ - visible: routeFocused && client !== null, - polling, - refresh, - intervalMs: POLL_INTERVAL_MS - }) - - useEffect(() => { - if (routeFocused && client && setup === null) { - setLoading(true) - } - }, [routeFocused, client, setup]) - - const handleToggleEnabled = useCallback( - async (enabled: boolean) => { - if (!client) { - return - } - setError(null) - // Optimistic flip so the switch responds instantly; reconcile below. - setSetup((prev) => (prev ? { ...prev, enabled } : prev)) - try { - setSetup(await setDictationConfig(client, { enabled })) - } catch (err) { - setError(err instanceof Error ? err.message : 'Could not update') - void refreshSetup() - } - }, - [client, refreshSetup] - ) - - const handleSelectMode = useCallback( - async (dictationMode: 'toggle' | 'hold') => { - if (!client) { - return - } - setError(null) - setSetup((prev) => (prev ? { ...prev, dictationMode } : prev)) - try { - setSetup(await setDictationConfig(client, { dictationMode })) - } catch (err) { - setError(err instanceof Error ? err.message : 'Could not update') - void refreshSetup() - } - }, - [client, refreshSetup] - ) - - const handleUseModel = useCallback( - async (model: MobileSpeechModel) => { - if (!client) { - return - } - setBusyAction({ modelId: model.id, type: 'select' }) - setError(null) - try { - setSetup(await setDictationConfig(client, { enabled: true, modelId: model.id })) - setModelDrawerOpen(false) - } catch (err) { - setError(err instanceof Error ? err.message : 'Could not select model') - } finally { - setBusyAction(null) - } - }, + const hostIds = useMemo(() => hosts.map((host) => host.id), [hosts]) + const { clients, focused } = useFocusedSettingsHostClients(hostIds) + const client = clients.find((entry) => entry.state === 'connected')?.client ?? null + const operations = useMemo( + () => (client ? nativeVoiceSettingsOperations(client) : null), [client] ) - - const handleDownload = useCallback( - async (model: MobileSpeechModel) => { - if (!client) { - return - } - setBusyAction({ modelId: model.id, type: 'download' }) - setError(null) - try { - await downloadDictationModel(client, model.id) - await refreshSetup() - } catch (err) { - setError(err instanceof Error ? err.message : 'Download failed') - } finally { - setBusyAction(null) - } - }, - [client, refreshSetup] - ) - - const handleDelete = useCallback( - async (model: MobileSpeechModel) => { - if (!client) { - return - } - const deletedSelectedModel = setup?.selectedModelId === model.id - setBusyAction({ modelId: model.id, type: 'delete' }) - setError(null) - try { - setSetup(await deleteDictationModel(client, model.id)) - if (deletedSelectedModel) { - setModelDrawerOpen(false) - } - } catch (err) { - setError(err instanceof Error ? err.message : 'Delete failed') - } finally { - setBusyAction(null) - } - }, - [client, setup?.selectedModelId] - ) - - const enabled = setup?.enabled ?? false - const selectedModel = setup?.models.find((m) => m.id === setup.selectedModelId) - const selectedModelLabel = selectedModel?.label ?? 'None selected' - return ( - - - router.back()}> - - - Voice - - - {!client ? ( - - Connect to a desktop to manage voice settings. - - ) : loading && setup === null ? ( - - - - ) : setup === null ? ( - - {error ?? 'Failed to load voice settings.'} - - ) : ( - - DICTATION - - - - Enable Voice Dictation - - Dictate text into any focused pane on your desktop. - - - void handleToggleEnabled(v)} - trackColor={{ false: colors.bgRaised, true: colors.textSecondary }} - thumbColor={colors.textPrimary} - /> - - - - - - - Dictation Mode - - Toggle: press once to start, again to stop. Hold: dictate while held. - - - - {DICTATION_MODES.map((mode) => { - const active = setup.dictationMode === mode.value - return ( - void handleSelectMode(mode.value)} - style={[styles.segment, active && styles.segmentActive]} - > - - {mode.label} - - - ) - })} - - - - - SPEECH MODEL - - [ - styles.row, - !enabled && styles.disabled, - pressed && styles.rowPressed - ]} - disabled={!enabled} - onPress={() => setModelDrawerOpen(true)} - > - - Speech Model - - {selectedModelLabel} - - - - - - - {error ? {error} : null} - - )} - - setModelDrawerOpen(false)}> - Speech Model - {setup ? ( - void handleUseModel(m)} - onDownload={(m) => void handleDownload(m)} - onDelete={(m) => void handleDelete(m)} - /> - ) : null} - - + router.back()} /> ) } - -const styles = StyleSheet.create({ - container: { - flex: 1, - backgroundColor: colors.bgBase, - paddingHorizontal: spacing.lg - }, - topRow: { - flexDirection: 'row', - alignItems: 'center', - marginTop: spacing.sm, - marginBottom: spacing.lg - }, - backButton: { - width: 36, - height: 36, - borderRadius: 18, - alignItems: 'center', - justifyContent: 'center', - marginRight: spacing.sm - }, - heading: { - fontSize: 20, - fontWeight: '700', - color: colors.textPrimary - }, - scrollContent: { - paddingBottom: spacing.xl - }, - loading: { paddingVertical: spacing.xl, alignItems: 'center' }, - groupHeading: { - fontSize: 11, - fontWeight: '600', - color: colors.textMuted, - letterSpacing: 0.5, - marginBottom: spacing.xs, - paddingHorizontal: spacing.xs - }, - section: { - backgroundColor: colors.bgPanel, - borderRadius: radii.card, - overflow: 'hidden' - }, - sectionTopGap: { marginTop: spacing.sm }, - inputGroupGap: { marginTop: spacing.xl }, - disabled: { opacity: 0.5 }, - emptyText: { - fontSize: typography.bodySize, - color: colors.textSecondary, - padding: spacing.md - }, - errorText: { - fontSize: typography.bodySize, - color: colors.statusRed, - padding: spacing.md - }, - row: { - flexDirection: 'row', - alignItems: 'center', - gap: spacing.sm + 2, - paddingVertical: spacing.md, - paddingHorizontal: spacing.md + 2 - }, - rowPressed: { backgroundColor: colors.bgRaised }, - rowContent: { flex: 1 }, - rowLabel: { - fontSize: typography.bodySize, - fontWeight: '500', - color: colors.textPrimary - }, - drawerTitle: { - fontSize: typography.bodySize, - fontWeight: '700', - color: colors.textPrimary, - paddingHorizontal: spacing.md + 2, - paddingTop: spacing.sm, - paddingBottom: spacing.xs - }, - rowSublabel: { - fontSize: typography.bodySize - 2, - color: colors.textSecondary, - marginTop: 2 - }, - separator: { - height: StyleSheet.hairlineWidth, - backgroundColor: colors.borderSubtle, - marginHorizontal: spacing.md - }, - segmented: { - flexDirection: 'row', - alignItems: 'center', - backgroundColor: colors.bgBase, - borderRadius: radii.button, - padding: 2 - }, - segment: { - paddingHorizontal: spacing.md, - paddingVertical: 6, - borderRadius: radii.button - 1 - }, - segmentActive: { backgroundColor: colors.bgRaised }, - segmentText: { fontSize: typography.metaSize, color: colors.textSecondary, fontWeight: '600' }, - segmentTextActive: { color: colors.textPrimary }, - error: { color: colors.statusRed, fontSize: typography.metaSize, marginTop: spacing.md } -}) diff --git a/mobile/package.json b/mobile/package.json index 159028c978d..13f2f98acf5 100644 --- a/mobile/package.json +++ b/mobile/package.json @@ -57,6 +57,7 @@ "react-native-safe-area-context": "^5.7.0", "react-native-screens": "^4.24.0", "react-native-svg": "^15.15.4", + "react-native-uitextview": "2.2.0", "react-native-web": "^0.21.2", "react-native-webview": "13.16.2", "react-native-worklets": "^0.8.3", diff --git a/mobile/pnpm-lock.yaml b/mobile/pnpm-lock.yaml index d44e612bb73..7cebee89eec 100644 --- a/mobile/pnpm-lock.yaml +++ b/mobile/pnpm-lock.yaml @@ -136,6 +136,9 @@ importers: react-native-svg: specifier: ^15.15.4 version: 15.15.4(react-native@0.83.10(patch_hash=44876634a8efbb0f2c3f66cd4332be170ec821d1cbfc0264ac80680983e8513d)(@babel/core@7.29.7)(@react-native/metro-config@0.85.2(@babel/core@7.29.7))(@types/react@19.2.14)(react@19.2.8))(react@19.2.8) + react-native-uitextview: + specifier: 2.2.0 + version: 2.2.0(react-native@0.83.10(patch_hash=44876634a8efbb0f2c3f66cd4332be170ec821d1cbfc0264ac80680983e8513d)(@babel/core@7.29.7)(@react-native/metro-config@0.85.2(@babel/core@7.29.7))(@types/react@19.2.14)(react@19.2.8))(react@19.2.8) react-native-web: specifier: ^0.21.2 version: 0.21.2(react-dom@19.2.8(react@19.2.8))(react@19.2.8) @@ -6158,6 +6161,12 @@ packages: react: '*' react-native: '*' + react-native-uitextview@2.2.0: + resolution: {integrity: sha512-Vbv3cTAuyfkYrfsR2YKFsOd9OYgfys+IX5yvvYY7Wd6TrOd6FSGrX93xtQHjeLXV1ds6fDJcFJsuLi3S4Ypr8A==} + peerDependencies: + react: '*' + react-native: '*' + react-native-web@0.21.2: resolution: {integrity: sha512-SO2t9/17zM4iEnFvlu2DA9jqNbzNhoUP+AItkoCOyFmDMOhUnBBznBDCYN92fGdfAkfQlWzPoez6+zLxFNsZEg==} peerDependencies: @@ -14609,6 +14618,11 @@ snapshots: react-native: 0.83.10(patch_hash=44876634a8efbb0f2c3f66cd4332be170ec821d1cbfc0264ac80680983e8513d)(@babel/core@7.29.7)(@react-native/metro-config@0.85.2(@babel/core@7.29.7))(@types/react@19.2.14)(react@19.2.8) warn-once: 0.1.1 + react-native-uitextview@2.2.0(react-native@0.83.10(patch_hash=44876634a8efbb0f2c3f66cd4332be170ec821d1cbfc0264ac80680983e8513d)(@babel/core@7.29.7)(@react-native/metro-config@0.85.2(@babel/core@7.29.7))(@types/react@19.2.14)(react@19.2.8))(react@19.2.8): + dependencies: + react: 19.2.8 + react-native: 0.83.10(patch_hash=44876634a8efbb0f2c3f66cd4332be170ec821d1cbfc0264ac80680983e8513d)(@babel/core@7.29.7)(@react-native/metro-config@0.85.2(@babel/core@7.29.7))(@types/react@19.2.14)(react@19.2.8) + react-native-web@0.21.2(react-dom@19.2.8(react@19.2.8))(react@19.2.8): dependencies: '@babel/runtime': 7.29.2 diff --git a/mobile/src/components/MobileMarkdown.tsx b/mobile/src/components/MobileMarkdown.tsx index 2f5b52cfe18..7ecfab8c393 100644 --- a/mobile/src/components/MobileMarkdown.tsx +++ b/mobile/src/components/MobileMarkdown.tsx @@ -1,5 +1,22 @@ -import { Fragment, memo, useMemo, type ReactNode } from 'react' -import { Linking, Pressable, ScrollView, Text, View } from 'react-native' +import { MobileSelectableText } from './MobileSelectableText' +import { + Fragment, + createElement, + createContext, + memo, + useContext, + useMemo, + type ComponentType, + type ReactNode +} from 'react' +import { + Linking, + Pressable, + ScrollView, + Text as NativeText, + View, + type TextProps +} from 'react-native' import { normalizeMobileMarkdownPreviewHtml } from './mobile-markdown-preview-html' import { styles } from './mobile-markdown-styles' import { @@ -19,6 +36,8 @@ import { MermaidDiagram } from './pr-sidebar/MermaidDiagram' type Props = { content?: string fallback?: string + /** Enables iOS range selection for native-chat transcript prose. */ + rangeSelectable?: boolean /** Multiplier for prose font size (paragraphs, lists, quotes). Defaults to 1; * the chat view passes >1 so agent prose reads larger than the compact base. */ textScale?: number @@ -33,6 +52,12 @@ const MAX_TABLE_ROWS = 40 const MAX_TABLE_COLUMNS = 8 /** Prose base size — passed to MermaidDiagram fallback mono text. */ const MERMAID_BASE = 13 +const MarkdownTextContext = createContext>(NativeText) + +function MarkdownText(props: TextProps): React.JSX.Element { + const TextComponent = useContext(MarkdownTextContext) + return createElement(TextComponent, props) +} // Web/mail hrefs open the system handler; file-target hrefs (file: URIs and // scheme-less paths — the entire desktop file-link contract) go to onOpenFile. @@ -64,13 +89,13 @@ function renderTextRun( return segments.map((segment, segmentIndex) => { if (segment.type === 'file') { return ( - onOpenFile(segment.path)} > {segment.value} - + ) } return {segment.value} @@ -105,22 +130,34 @@ function renderInline(text: string, onOpenFile?: (pathText: string) => void): Re const link = token.match(/^\[([^\]]+)\]\(([^)]+)\)$/) if (image) { parts.push( - openMarkdownHref(image[2]!, onOpenFile)}> + openMarkdownHref(image[2]!, onOpenFile)} + > {image[1] || 'image'} - + ) } else if (link) { parts.push( - openMarkdownHref(link[2]!, onOpenFile)}> + openMarkdownHref(link[2]!, onOpenFile)} + > {link[1]} - + ) } else if (/^https?:\/\//i.test(token)) { const { url, trailing } = trimAutolinkTrailingPunctuation(token) parts.push( - openMarkdownHref(url, onOpenFile)}> + openMarkdownHref(url, onOpenFile)} + > {url} - + ) if (trailing) { parts.push({trailing}) @@ -129,38 +166,38 @@ function renderInline(text: string, onOpenFile?: (pathText: string) => void): Re const code = token.slice(1, -1) if (onOpenFile && isFilePathCodeSpan(code)) { parts.push( - onOpenFile(normalizeFilePath(code.trim()))} > {code} - + ) } else { parts.push( - + {code} - + ) } } else if (token.startsWith('~~')) { parts.push( - + {renderTextRun(token.slice(2, -2), `${key}i`, onOpenFile)} - + ) } else if (token.startsWith('**') || token.startsWith('__')) { parts.push( - + {renderTextRun(token.slice(2, -2), `${key}i`, onOpenFile)} - + ) } else { parts.push( - + {renderTextRun(token.slice(1, -1), `${key}i`, onOpenFile)} - + ) } } @@ -171,7 +208,13 @@ function renderInline(text: string, onOpenFile?: (pathText: string) => void): Re return parts } -function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile }: Props) { +function MobileMarkdownContent({ + content, + fallback = '', + rangeSelectable = false, + textScale = 1, + onOpenFile +}: Props) { const text = content?.trim() ?? '' const previewText = useMemo(() => normalizeMobileMarkdownPreviewHtml(text), [text]) const blocks = useMemo(() => parseMobileMarkdown(previewText), [previewText]) @@ -181,30 +224,35 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile const proseScale = scaled(13) const listScale = scaled(14) if (!text) { - return fallback ? {fallback} : null + return fallback ? ( + + {fallback} + + ) : null } const mermaidSourceOccurrences = new Map() + // Native-chat range selection is set on each block; nested inline spans inherit it. return ( {blocks.map((block, index) => { if (block.type === 'heading') { return ( - {renderInline(block.text, onOpenFile)} - + ) } if (block.type === 'quote') { return ( - + {renderInline(block.text, onOpenFile)} - + ) } @@ -225,10 +273,12 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile } return ( - {block.language ? {block.language} : null} - + {block.language ? ( + {block.language} + ) : null} + {block.text} - + ) } @@ -239,10 +289,10 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile style={styles.imageFrame} onPress={() => openMarkdownHref(block.url, onOpenFile)} > - {block.alt || 'Open image'} - + {block.alt || 'Open image'} + {block.url} - + ) } @@ -256,26 +306,30 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleHeaders.map((header, cellIndex) => ( - + {renderInline(header, onOpenFile)} - + ))} {visibleRows.map((row, rowIndex) => ( {visibleHeaders.map((_, cellIndex) => ( - + {renderInline(row[cellIndex] ?? '', onOpenFile)} - + ))} ))} {hiddenRows > 0 || hiddenColumns > 0 ? ( - + {hiddenRows > 0 ? `${hiddenRows} more rows` : ''} {hiddenRows > 0 && hiddenColumns > 0 ? ' · ' : ''} {hiddenColumns > 0 ? `${hiddenColumns} more columns` : ''} - + ) : null} @@ -286,7 +340,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {block.items.map((item, itemIndex) => ( - + {item.checked == null ? block.ordered ? `${itemIndex + 1}.` @@ -294,10 +348,10 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile : item.checked ? '[x]' : '[ ]'} - - + + {renderInline(item.text, onOpenFile)} - + ))} @@ -307,18 +361,31 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return } return ( - + {block.text.split('\n').map((line, lineIndex) => ( {lineIndex > 0 ? '\n' : null} {renderInline(line, onOpenFile)} ))} - + ) })} ) } +function MobileMarkdownInner(props: Props): React.JSX.Element | null { + const TextComponent = props.rangeSelectable ? MobileSelectableText : NativeText + return ( + + + + ) +} + export const MobileMarkdown = memo(MobileMarkdownInner) diff --git a/mobile/src/components/MobileRichMarkdownEditor.tsx b/mobile/src/components/MobileRichMarkdownEditor.tsx index 8fbcaa149e7..2ad299170ea 100644 --- a/mobile/src/components/MobileRichMarkdownEditor.tsx +++ b/mobile/src/components/MobileRichMarkdownEditor.tsx @@ -2,7 +2,6 @@ import { forwardRef, memo, useCallback, - useEffect, useImperativeHandle, useMemo, useRef, @@ -29,7 +28,12 @@ import { } from 'lucide-react-native' import WebView, { type WebViewMessageEvent } from 'react-native-webview' import { colors, radii, spacing } from '../theme/mobile-theme' -import { normalizeMobileRichMarkdownKeyboardInset } from './mobile-rich-markdown-editor-keyboard-inset-script' +import type { + MobileRichMarkdownCommand, + MobileRichMarkdownEditorMessage, + MobileRichMarkdownEditorProps +} from './mobile-rich-markdown-editor-contract' +import { useMobileRichMarkdownEditorController } from './use-mobile-rich-markdown-editor-controller' import { buildMobileRichMarkdownEditorHtml, escapeInjectedJavaScriptString @@ -38,67 +42,16 @@ import { const EDITOR_DOCUMENT_ORIGIN = 'https://orca-mobile-editor.invalid' const EDITOR_DOCUMENT_URL = `${EDITOR_DOCUMENT_ORIGIN}/rich-markdown-editor` -function normalizeExternalEditorUrl(value: string): string | null { - const url = value.trim() - if (!url) { - return null - } - for (let index = 0; index < url.length; index += 1) { - const code = url.charCodeAt(index) - if (code <= 32 || code === 127) { - return null - } - } - if (/^mailto:/i.test(url)) { - return url - } - if (!/^https?:\/\//i.test(url)) { - return null - } - try { - const parsed = new URL(url) - return parsed.protocol === 'http:' || parsed.protocol === 'https:' ? parsed.toString() : null - } catch { - return null - } -} - -type RichMarkdownCommand = - | 'paragraph' - | 'heading1' - | 'heading2' - | 'heading3' - | 'bold' - | 'italic' - | 'strike' - | 'bulletList' - | 'orderedList' - | 'taskList' - | 'quote' - | 'inlineCode' - | 'codeBlock' - | 'link' - | 'image' - -type Props = { - content: string - editable: boolean - onChange: (content: string) => void - onKeyboardInsetChange?: (bottom: number) => void +type Props = Omit & { + onOpenLink?: (url: string) => void } export type MobileRichMarkdownEditorHandle = { dismissKeyboard: () => void } -type EditorWebViewMessage = - | { type: 'ready' } - | { type: 'change'; markdown: string; generation: number } - | { type: 'openLink'; url: string } - | { type: 'keyboardInset'; bottom: number } - type ToolbarItem = { - command: RichMarkdownCommand + command: MobileRichMarkdownCommand label: string icon: ComponentType<{ size?: number; color?: string }> } @@ -122,61 +75,55 @@ const TOOLBAR_ITEMS: ToolbarItem[] = [ ] function MobileRichMarkdownEditorInner( - { content, editable, onChange, onKeyboardInsetChange }: Props, + { content, editable, onChange, onKeyboardInsetChange, onOpenLink }: Props, ref: ForwardedRef ) { const webViewRef = useRef(null) - const readyRef = useRef(false) - const documentGenerationRef = useRef(0) - const currentWebViewContentRef = useRef(null) const html = useMemo(() => buildMobileRichMarkdownEditorHtml(), []) const inject = useCallback((script: string) => { webViewRef.current?.injectJavaScript(`${script}\ntrue;`) }, []) - const applyContent = useCallback( - (nextContent: string) => { - documentGenerationRef.current += 1 - currentWebViewContentRef.current = nextContent - inject( - `window.__orcaRichMarkdown && window.__orcaRichMarkdown.setMarkdown(${escapeInjectedJavaScriptString(nextContent)}, ${documentGenerationRef.current});` - ) - }, + const transport = useMemo( + () => ({ + setMarkdown: (markdown: string, generation: number) => + inject( + `window.__orcaRichMarkdown && window.__orcaRichMarkdown.setMarkdown(${escapeInjectedJavaScriptString(markdown)}, ${generation});` + ), + setEditable: (nextEditable: boolean) => + inject( + `window.__orcaRichMarkdown && window.__orcaRichMarkdown.setEditable(${nextEditable ? 'true' : 'false'});` + ), + runCommand: (command: MobileRichMarkdownCommand) => + inject( + `window.__orcaRichMarkdown && window.__orcaRichMarkdown.runCommand(${escapeInjectedJavaScriptString(command)});` + ) + }), [inject] ) - const applyEditable = useCallback( - (nextEditable: boolean) => { - inject( - `window.__orcaRichMarkdown && window.__orcaRichMarkdown.setEditable(${nextEditable ? 'true' : 'false'});` - ) + const openLink = useCallback( + (url: string) => { + if (onOpenLink) { + onOpenLink(url) + return + } + void Linking.openURL(url).catch(() => {}) }, - [inject] + [onOpenLink] ) - useEffect(() => { - if (!readyRef.current) { - return - } - if (currentWebViewContentRef.current !== content) { - applyContent(content) - } - }, [applyContent, content]) + const { handleMessage, runCommand } = useMobileRichMarkdownEditorController({ + content, + editable, + onChange, + onKeyboardInsetChange, + onOpenLink: openLink, + transport + }) - useEffect(() => { - if (readyRef.current) { - applyEditable(editable) - } - }, [applyEditable, editable]) - - // Clear any reported keyboard inset when the editor unmounts so a lifted - // Save/Discard bar settles back once the tab closes. - useEffect(() => { - return () => onKeyboardInsetChange?.(0) - }, [onKeyboardInsetChange]) - - const handleMessage = useCallback( + const handleWebViewMessage = useCallback( (event: WebViewMessageEvent) => { let message: unknown try { @@ -187,37 +134,9 @@ function MobileRichMarkdownEditorInner( if (!message || typeof message !== 'object') { return } - const editorMessage = message as Partial - if ('type' in message && message.type === 'ready') { - readyRef.current = true - applyContent(content) - applyEditable(editable) - return - } - if ( - editorMessage.type === 'change' && - typeof editorMessage.markdown === 'string' && - editorMessage.generation === documentGenerationRef.current - ) { - currentWebViewContentRef.current = editorMessage.markdown - onChange(editorMessage.markdown) - return - } - if (editorMessage.type === 'openLink' && typeof editorMessage.url === 'string') { - const url = normalizeExternalEditorUrl(editorMessage.url) - if (url) { - void Linking.openURL(url).catch(() => {}) - } - return - } - if (editorMessage.type === 'keyboardInset' && typeof editorMessage.bottom === 'number') { - const bottom = normalizeMobileRichMarkdownKeyboardInset(editorMessage.bottom) - if (bottom !== null) { - onKeyboardInsetChange?.(bottom) - } - } + handleMessage(message as Partial) }, - [applyContent, applyEditable, content, editable, onChange, onKeyboardInsetChange] + [handleMessage] ) const handleShouldStartLoadWithRequest = useCallback((request: { url?: string }) => { @@ -230,15 +149,6 @@ function MobileRichMarkdownEditorInner( return isEditorDocument }, []) - const runCommand = useCallback( - (command: RichMarkdownCommand) => { - inject( - `window.__orcaRichMarkdown && window.__orcaRichMarkdown.runCommand(${escapeInjectedJavaScriptString(command)});` - ) - }, - [inject] - ) - const dismissKeyboard = useCallback(() => { // Why: the caret lives in the WebView, so the injected blur is what closes the keyboard; // Keyboard.dismiss only clears a native TextInput that stole focus first. @@ -286,7 +196,7 @@ function MobileRichMarkdownEditorInner( domStorageEnabled={false} hideKeyboardAccessoryView keyboardDisplayRequiresUserAction={false} - onMessage={handleMessage} + onMessage={handleWebViewMessage} onShouldStartLoadWithRequest={handleShouldStartLoadWithRequest} style={styles.webView} scrollEnabled diff --git a/mobile/src/components/MobileSelectableText.ios.tsx b/mobile/src/components/MobileSelectableText.ios.tsx new file mode 100644 index 00000000000..74cb686e26e --- /dev/null +++ b/mobile/src/components/MobileSelectableText.ios.tsx @@ -0,0 +1,38 @@ +import { Children, Fragment, isValidElement, type ReactNode } from 'react' +import { StyleSheet, Text, UIManager, type TextProps } from 'react-native' +import { UITextView } from 'react-native-uitextview' + +// Older development clients can load this bundle before rebuilding their native views. +const hasRangeSelection = UIManager.hasViewManagerConfig('RNUITextView') + +function flattenFragments(children: ReactNode): ReactNode[] { + return ( + Children.map(children, (child) => + isValidElement<{ children?: ReactNode }>(child) && child.type === Fragment + ? flattenFragments(child.props.children) + : child + ) ?? [] + ) +} + +export function MobileSelectableText({ children, style, ...props }: TextProps): React.JSX.Element { + if (!hasRangeSelection) { + return ( + + {children} + + ) + } + + // The native span adapter otherwise maps numeric bold to semibold. + const textStyle = StyleSheet.flatten(style) + const nativeStyle = + textStyle?.fontWeight === '700' || textStyle?.fontWeight === 700 + ? { ...textStyle, fontWeight: 'bold' as const } + : style + return ( + + {flattenFragments(children)} + + ) +} diff --git a/mobile/src/components/MobileSelectableText.tsx b/mobile/src/components/MobileSelectableText.tsx new file mode 100644 index 00000000000..ddfffc37ee9 --- /dev/null +++ b/mobile/src/components/MobileSelectableText.tsx @@ -0,0 +1 @@ +export { Text as MobileSelectableText } from 'react-native' diff --git a/mobile/src/components/NewWorktreeFormSheet.tsx b/mobile/src/components/NewWorktreeFormSheet.tsx index 2a6587ffa87..f202d084c82 100644 --- a/mobile/src/components/NewWorktreeFormSheet.tsx +++ b/mobile/src/components/NewWorktreeFormSheet.tsx @@ -43,6 +43,7 @@ export function NewWorktreeFormSheet(props: { creating: boolean canCreate: boolean onClose: () => void + onOpenExternalUrl: (url: string) => Promise onOpenProject: () => void onOpenRunTarget: () => void onOpenSource: () => void @@ -84,6 +85,7 @@ export function NewWorktreeFormSheet(props: { label={props.selectedRepoIsGit ? "Name or 'Create From'" : 'Workspace name'} disabled={props.sshGate.requiresConnection} interactive={props.interactive} + onOpenExternalUrl={props.onOpenExternalUrl} onBeforeOpen={props.onClearError} onOpenDrawer={props.onOpenSource} /> diff --git a/mobile/src/components/NewWorktreeModal.tsx b/mobile/src/components/NewWorktreeModal.tsx index a986d347782..511b8276be5 100644 --- a/mobile/src/components/NewWorktreeModal.tsx +++ b/mobile/src/components/NewWorktreeModal.tsx @@ -55,8 +55,16 @@ export function NewWorktreeModal(props: NewWorktreeModalProps) { } function NewWorktreeModalContent(props: NewWorktreeModalProps) { - const { visible, client, hostId, existingWorktreePaths, existingWorktrees, onCreated, onClose } = - props + const { + visible, + client, + hostId, + existingWorktreePaths, + existingWorktrees, + openExternalUrl, + onCreated, + onClose + } = props const { repos, selectedRepo, setSelectedRepo, loading } = useNewWorkspaceRepositories({ client, hostId, @@ -217,6 +225,7 @@ function NewWorktreeModalContent(props: NewWorktreeModalProps) { creating={createSubmit.creating} canCreate={canCreate} onClose={onClose} + onOpenExternalUrl={openExternalUrl} onOpenProject={() => openPicker('project')} onOpenRunTarget={() => openPicker('runTarget')} onOpenSource={navigation.openSourceDrawer} diff --git a/mobile/src/components/NewWorktreeModalController.tsx b/mobile/src/components/NewWorktreeModalController.tsx index 9a062692d17..4b6812fc9ff 100644 --- a/mobile/src/components/NewWorktreeModalController.tsx +++ b/mobile/src/components/NewWorktreeModalController.tsx @@ -13,6 +13,7 @@ type Props = { hostId?: string existingWorktreePaths?: readonly string[] existingWorktrees?: readonly { repoId: string; branch: string }[] + openExternalUrl: (url: string) => Promise onVisibleChange?: (visible: boolean) => void onRouteVisibleChange: (visible: boolean) => void onCreated: (worktreeId: string, name: string) => void @@ -26,6 +27,7 @@ export const NewWorktreeModalController = forwardRef diff --git a/mobile/src/components/SmartWorkspaceSourceField.tsx b/mobile/src/components/SmartWorkspaceSourceField.tsx index 9b8972b16be..23eb04ecb33 100644 --- a/mobile/src/components/SmartWorkspaceSourceField.tsx +++ b/mobile/src/components/SmartWorkspaceSourceField.tsx @@ -1,4 +1,4 @@ -import { Linking, Pressable, StyleSheet, Text, TextInput, View } from 'react-native' +import { Pressable, StyleSheet, Text, TextInput, View } from 'react-native' import { CircleDot, ExternalLink, @@ -16,6 +16,7 @@ type Props = { composer: MobileComposerSource label: string disabled?: boolean + onOpenExternalUrl: (url: string) => Promise // Why: only the active form view may focus this field. While the source drawer // is open/closing this stays non-focusable so the drawer's dismiss (which // restores native focus back here) can't re-fire onFocus and reopen the drawer. @@ -44,6 +45,7 @@ export function SmartWorkspaceSourceField({ composer, label, disabled, + onOpenExternalUrl, interactive, onBeforeOpen, onOpenDrawer @@ -71,8 +73,10 @@ export function SmartWorkspaceSourceField({ {selection.url ? ( selection.url && void Linking.openURL(selection.url).catch(() => {})} + onPress={() => selection.url && void onOpenExternalUrl(selection.url).catch(() => {})} > diff --git a/mobile/src/components/codex-reset-credit-capability.ts b/mobile/src/components/codex-reset-credit-capability.ts index 1a32ef37873..8e88f74f2f9 100644 --- a/mobile/src/components/codex-reset-credit-capability.ts +++ b/mobile/src/components/codex-reset-credit-capability.ts @@ -2,6 +2,7 @@ import { useEffect, useState } from 'react' import { CODEX_RESET_CREDIT_RUNTIME_CAPABILITY } from '../../../src/shared/protocol-version' import type { RpcClient } from '../transport/rpc-client' import { startRuntimeCapabilityProbe } from '../transport/runtime-capability-probe' +import { rpcObjectResultOrNull } from '../transport/rpc-acceptance-policies' // Why: source the capability string from the shared contract so a host bump can never // silently drift from the mobile probe. @@ -12,10 +13,7 @@ export async function readCodexResetCreditCapability( ): Promise { try { const response = await client.sendRequest('status.get') - if (!response.ok || !response.result || typeof response.result !== 'object') { - return false - } - const capabilities = (response.result as { capabilities?: unknown }).capabilities + const capabilities = rpcObjectResultOrNull(response)?.capabilities return ( Array.isArray(capabilities) && capabilities.includes(MOBILE_CODEX_RESET_CREDIT_CAPABILITY) ) diff --git a/mobile/src/components/mobile-markdown-selectable.test.tsx b/mobile/src/components/mobile-markdown-selectable.test.tsx new file mode 100644 index 00000000000..2b42fc70715 --- /dev/null +++ b/mobile/src/components/mobile-markdown-selectable.test.tsx @@ -0,0 +1,110 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { MobileMarkdown } from './MobileMarkdown' + +vi.mock('react-native', () => ({ + Linking: { openURL: () => Promise.resolve() }, + Pressable: 'Pressable', + ScrollView: 'ScrollView', + StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 }, + Text: 'Text', + View: 'View' +})) + +vi.mock('./pr-sidebar/MermaidDiagram', () => ({ MermaidDiagram: 'MermaidDiagram' })) + +type TestNode = { + type: string + props: Record + children: (TestNode | string)[] | null +} + +/** Every Text that is not nested inside another Text, paired with the full prose + * it renders. Nested inline spans inherit selection, so only these carry it. */ +function outermostTextNodes( + node: TestNode | string, + insideText = false +): { text: string; selectable: boolean }[] { + if (typeof node === 'string') { + return [] + } + const children = node.children ?? [] + if (node.type === 'Text' && !insideText) { + return [{ text: flattenText(node), selectable: node.props.selectable === true }] + } + return children.flatMap((child) => outermostTextNodes(child, insideText || node.type === 'Text')) +} + +function flattenText(node: TestNode | string): string { + if (typeof node === 'string') { + return node + } + return (node.children ?? []).map(flattenText).join('') +} + +function renderMarkdown(props: Parameters[0]): TestNode { + let renderer: ReactTestRenderer | null = null + act(() => { + renderer = create(createElement(MobileMarkdown, { rangeSelectable: true, ...props })) + }) + const tree = renderer!.toJSON() as unknown as TestNode + act(() => renderer!.unmount()) + return tree +} + +function selectableFor(tree: TestNode, needle: string): boolean { + const match = outermostTextNodes(tree).find((entry) => entry.text.includes(needle)) + if (!match) { + throw new Error(`no Text rendered "${needle}"`) + } + return match.selectable +} + +describe('MobileMarkdown selection', () => { + afterEach(() => vi.clearAllMocks()) + + // Paragraphs are the default block for agent prose, and were the one block + // type left non-selectable when the others gained it. + it.each([ + ['paragraph', 'Paragraph prose here.'], + ['heading', 'Heading prose'], + ['quote', 'Quote prose'], + ['code', 'const code = 1'], + ['list item', 'List item prose'], + ['table header', 'Head A'], + ['table cell', 'Cell A'] + ])('makes %s prose selectable', (_label, needle) => { + const content = [ + '# Heading prose', + '', + 'Paragraph prose here.', + '', + '> Quote prose', + '', + '```ts', + 'const code = 1', + '```', + '', + '- List item prose', + '', + '| Head A | Head B |', + '| --- | --- |', + '| Cell A | Cell B |' + ].join('\n') + expect(selectableFor(renderMarkdown({ content }), needle)).toBe(true) + }) + + it('makes the empty-content fallback selectable', () => { + const tree = renderMarkdown({ content: '', fallback: 'Fallback prose' }) + expect(selectableFor(tree, 'Fallback prose')).toBe(true) + }) + + it('keeps inline spans inside their selectable block rather than splitting it', () => { + const tree = renderMarkdown({ content: 'Prose with `code` and **bold** inline.' }) + const blocks = outermostTextNodes(tree) + expect(blocks).toHaveLength(1) + expect(blocks[0]!.selectable).toBe(true) + expect(blocks[0]!.text).toBe('Prose with code and bold inline.') + }) +}) diff --git a/mobile/src/components/mobile-rich-markdown-editor-contract.ts b/mobile/src/components/mobile-rich-markdown-editor-contract.ts new file mode 100644 index 00000000000..5e133e6e23b --- /dev/null +++ b/mobile/src/components/mobile-rich-markdown-editor-contract.ts @@ -0,0 +1,37 @@ +export type MobileRichMarkdownCommand = + | 'paragraph' + | 'heading1' + | 'heading2' + | 'heading3' + | 'bold' + | 'italic' + | 'strike' + | 'bulletList' + | 'orderedList' + | 'taskList' + | 'quote' + | 'inlineCode' + | 'codeBlock' + | 'link' + | 'image' + +export type MobileRichMarkdownEditorMessage = + | { type: 'ready' } + | { type: 'change'; markdown: string; generation: number } + | { type: 'openLink'; url: string } + | { type: 'keyboardInset'; bottom: number } + +export type MobileRichMarkdownEditorProps = { + content: string + editable: boolean + onChange: (content: string) => void + onKeyboardInsetChange?: (bottom: number) => void + onOpenLink: (url: string) => void +} + +/** How a host delivers a command into whatever surface renders the editor document. */ +export type MobileRichMarkdownEditorTransport = { + setMarkdown: (markdown: string, generation: number) => void + setEditable: (editable: boolean) => void + runCommand: (command: MobileRichMarkdownCommand) => void +} diff --git a/mobile/src/components/mobile-rich-markdown-editor-body-primary.ts b/mobile/src/components/mobile-rich-markdown-editor-document-body.ts similarity index 56% rename from mobile/src/components/mobile-rich-markdown-editor-body-primary.ts rename to mobile/src/components/mobile-rich-markdown-editor-document-body.ts index c4348ad15a2..a381b7d38e4 100644 --- a/mobile/src/components/mobile-rich-markdown-editor-body-primary.ts +++ b/mobile/src/components/mobile-rich-markdown-editor-document-body.ts @@ -1,4 +1,5 @@ -export const MOBILE_RICH_MARKDOWN_EDITOR_BODY_PRIMARY = [ +// Head of the editor document through the editable surface; the script follows it. +export const MOBILE_RICH_MARKDOWN_EDITOR_DOCUMENT_BODY = [ ';', ' --font-mono: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, "Liberation Mono", monospace;', ' --font-sans: Geist, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif;', @@ -181,99 +182,5 @@ export const MOBILE_RICH_MARKDOWN_EDITOR_BODY_PRIMARY = [ ' ', '', '', - '
', - ' ', - '', - '' + ' })();' ].join('\n') diff --git a/mobile/src/components/mobile-rich-markdown-editor-document.test.ts b/mobile/src/components/mobile-rich-markdown-editor-document.test.ts new file mode 100644 index 00000000000..f45d45139c6 --- /dev/null +++ b/mobile/src/components/mobile-rich-markdown-editor-document.test.ts @@ -0,0 +1,19 @@ +import { createHash } from 'node:crypto' +import { describe, expect, it } from 'vitest' +import { buildMobileRichMarkdownEditorHtml } from './mobile-rich-markdown-editor-html' + +// Digest of main's document at e80fae0c4d, captured before the body/script split. Splitting the +// constants must not move a single byte of what the WebView loads. A hash rather than a +// checked-in HTML file, because the formatter would rewrite the file and defeat the check. +const PRE_SPLIT_DOCUMENT_SHA256 = '1ef29c8802170800011e8accf1966bc542cdd7dd5c9600bacb6e0860f77b6df8' +const PRE_SPLIT_DOCUMENT_BYTES = 29852 + +describe('mobile rich markdown editor document', () => { + it('reproduces the pre-split document byte for byte', () => { + const document = buildMobileRichMarkdownEditorHtml() + expect(Buffer.byteLength(document, 'utf8')).toBe(PRE_SPLIT_DOCUMENT_BYTES) + expect(createHash('sha256').update(document, 'utf8').digest('hex')).toBe( + PRE_SPLIT_DOCUMENT_SHA256 + ) + }) +}) diff --git a/mobile/src/components/mobile-rich-markdown-editor-html.ts b/mobile/src/components/mobile-rich-markdown-editor-html.ts index 493fea3dd02..a98a77d6e9d 100644 --- a/mobile/src/components/mobile-rich-markdown-editor-html.ts +++ b/mobile/src/components/mobile-rich-markdown-editor-html.ts @@ -1,17 +1,9 @@ import { colors } from '../theme/mobile-theme' -import { MOBILE_RICH_MARKDOWN_KEYBOARD_DISMISS_SCRIPT } from './mobile-rich-markdown-keyboard-dismiss-script' -import { MOBILE_RICH_MARKDOWN_KEYBOARD_INSET_SCRIPT } from './mobile-rich-markdown-editor-keyboard-inset-script' -import { MOBILE_RICH_MARKDOWN_SELECTION_SCRIPT } from './mobile-rich-markdown-selection-script' -import { MOBILE_RICH_MARKDOWN_EDITOR_BODY_PRIMARY } from './mobile-rich-markdown-editor-body-primary' -import { MOBILE_RICH_MARKDOWN_EDITOR_BODY_SECONDARY } from './mobile-rich-markdown-editor-body-secondary' -import { - MOBILE_RICH_MARKDOWN_EDITOR_AFTER_KEYBOARD_DISMISS, - MOBILE_RICH_MARKDOWN_EDITOR_DOCUMENT_END -} from './mobile-rich-markdown-editor-document-suffix' +import { MOBILE_RICH_MARKDOWN_EDITOR_DOCUMENT_BODY } from './mobile-rich-markdown-editor-document-body' +import { MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT } from './mobile-rich-markdown-editor-script' -export function escapeInjectedJavaScriptString(value: string): string { - return JSON.stringify(value).replace(/<\/script/gi, '<\\/script') -} +export { escapeInjectedJavaScriptString } from './mobile-rich-markdown-editor-script-string' +export { MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT } from './mobile-rich-markdown-editor-script' export function buildMobileRichMarkdownEditorHtml(): string { return ` @@ -30,7 +22,10 @@ export function buildMobileRichMarkdownEditorHtml(): string { --border: ${colors.borderSubtle}; --primary: ${colors.textPrimary}; --primary-foreground: ${colors.bgBase}; - --accent-link: ${colors.accentBlue}${MOBILE_RICH_MARKDOWN_EDITOR_BODY_PRIMARY} -${MOBILE_RICH_MARKDOWN_EDITOR_BODY_SECONDARY}${MOBILE_RICH_MARKDOWN_SELECTION_SCRIPT} -${MOBILE_RICH_MARKDOWN_KEYBOARD_DISMISS_SCRIPT}${MOBILE_RICH_MARKDOWN_EDITOR_AFTER_KEYBOARD_DISMISS}${MOBILE_RICH_MARKDOWN_KEYBOARD_INSET_SCRIPT}${MOBILE_RICH_MARKDOWN_EDITOR_DOCUMENT_END}` + --accent-link: ${colors.accentBlue}${MOBILE_RICH_MARKDOWN_EDITOR_DOCUMENT_BODY} + + +` } diff --git a/mobile/src/components/mobile-rich-markdown-editor-script-primary.ts b/mobile/src/components/mobile-rich-markdown-editor-script-primary.ts new file mode 100644 index 00000000000..1f0e5aa7b3b --- /dev/null +++ b/mobile/src/components/mobile-rich-markdown-editor-script-primary.ts @@ -0,0 +1,95 @@ +export const MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_PRIMARY = [ + ' (function () {', + " var editor = document.getElementById('editor');", + " var lastMarkdown = '';", + ' var inputTimer = null;', + ' var documentGeneration = 0;', + ' var editable = true;', + ' var suppressInput = false;', + '', + ' function post(message) {', + ' window.ReactNativeWebView && window.ReactNativeWebView.postMessage(JSON.stringify(message));', + ' }', + '', + ' function decodeMarkdownEntities(value) {', + ' return String(value).replace(/&(#x[0-9a-f]+|#\\d+|amp|lt|gt|quot|apos);/gi, function (match, entity) {', + ' var lower = String(entity).toLowerCase();', + " if (lower === 'amp') return '&';", + " if (lower === 'lt') return '<';", + " if (lower === 'gt') return '>';", + " if (lower === 'quot') return '\"';", + " if (lower === 'apos') return \"'\";", + " if (lower.indexOf('#x') === 0) {", + ' var hex = Number.parseInt(lower.slice(2), 16);', + ' return Number.isFinite(hex) && hex >= 0 && hex <= 0x10ffff ? String.fromCodePoint(hex) : match;', + ' }', + " if (lower.indexOf('#') === 0) {", + ' var code = Number.parseInt(lower.slice(1), 10);', + ' return Number.isFinite(code) && code >= 0 && code <= 0x10ffff ? String.fromCodePoint(code) : match;', + ' }', + ' return match;', + ' });', + ' }', + '', + ' function escapeHtml(value) {', + ' return decodeMarkdownEntities(value).replace(/[&<>"\']/g, function (char) {', + " return ({ '&': '&', '<': '<', '>': '>', '\"': '"', \"'\": ''' })[char];", + ' });', + ' }', + '', + ' function escapeAttr(value) {', + " return escapeHtml(value).replace(/\\n/g, ' ');", + ' }', + '', + ' function isSafeUrl(value) {', + " var trimmed = String(value || '').trim();", + ' return !/^javascript:/i.test(trimmed);', + ' }', + '', + ' function splitTableRow(line) {', + " return line.trim().replace(/^\\|/, '').replace(/\\|$/, '').split('|').map(function (cell) {", + ' return cell.trim();', + ' });', + ' }', + '', + ' function isTableSeparator(line) {', + ' var cells = splitTableRow(line);', + ' return cells.length > 0 && cells.every(function (cell) {', + ' return /^:?-{3,}:?$/.test(cell);', + ' });', + ' }', + '', + ' function renderInline(text) {', + " var output = '';", + ' var pattern = /(!\\[[^\\]]*\\]\\([^)]+\\)|`[^`]+`|~~[^~]+~~|\\*\\*[^*]+\\*\\*|__[^_]+__|\\*[^*\\n]+\\*|_[^_\\n]+_|\\[[^\\]]+\\]\\([^)]+\\)|https?:\\/\\/[^\\s<]+)/g;', + ' var lastIndex = 0;', + ' var match;', + ' while ((match = pattern.exec(text))) {', + ' output += escapeHtml(text.slice(lastIndex, match.index));', + ' var token = match[0];', + ' var image = token.match(/^!\\[([^\\]]*)\\]\\(([^)]+)\\)$/);', + ' var link = token.match(/^\\[([^\\]]+)\\]\\(([^)]+)\\)$/);', + ' if (image && isSafeUrl(image[2])) {', + " output += '\"'';", + ' } else if (link && isSafeUrl(link[2])) {', + " output += '' + renderInline(link[1]) + '';", + ' } else if (/^https?:\\/\\//i.test(token)) {', + " output += '' + escapeHtml(token) + '';", + " } else if (token.indexOf('`') === 0) {", + " output += '' + escapeHtml(token.slice(1, -1)) + '';", + " } else if (token.indexOf('~~') === 0) {", + " output += '' + renderInline(token.slice(2, -2)) + '';", + " } else if (token.indexOf('**') === 0 || token.indexOf('__') === 0) {", + " output += '' + renderInline(token.slice(2, -2)) + '';", + ' } else {', + " output += '' + renderInline(token.slice(1, -1)) + '';", + ' }', + ' lastIndex = pattern.lastIndex;', + ' }', + ' output += escapeHtml(text.slice(lastIndex));', + ' return output;', + ' }', + '', + ' function isBlockStart(line) {', + ' return /^(```|#{1,6}\\s+|>\\s?|\\s*(?:[-*+]|\\d+[.)])\\s+|\\s*(-{3,}|\\*{3,}|_{3,})\\s*$)/.test(line);' +].join('\n') diff --git a/mobile/src/components/mobile-rich-markdown-editor-body-secondary.ts b/mobile/src/components/mobile-rich-markdown-editor-script-secondary.ts similarity index 99% rename from mobile/src/components/mobile-rich-markdown-editor-body-secondary.ts rename to mobile/src/components/mobile-rich-markdown-editor-script-secondary.ts index 70de1c33692..9578bdc49ea 100644 --- a/mobile/src/components/mobile-rich-markdown-editor-body-secondary.ts +++ b/mobile/src/components/mobile-rich-markdown-editor-script-secondary.ts @@ -1,4 +1,4 @@ -export const MOBILE_RICH_MARKDOWN_EDITOR_BODY_SECONDARY = [ +export const MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_SECONDARY = [ ' }', '', ' function indentationWidth(value) {', diff --git a/mobile/src/components/mobile-rich-markdown-editor-script-string.ts b/mobile/src/components/mobile-rich-markdown-editor-script-string.ts new file mode 100644 index 00000000000..4b5671d0331 --- /dev/null +++ b/mobile/src/components/mobile-rich-markdown-editor-script-string.ts @@ -0,0 +1,3 @@ +export function escapeInjectedJavaScriptString(value: string): string { + return JSON.stringify(value).replace(/<\/script/gi, '<\\/script') +} diff --git a/mobile/src/components/mobile-rich-markdown-editor-script.ts b/mobile/src/components/mobile-rich-markdown-editor-script.ts new file mode 100644 index 00000000000..b8644705463 --- /dev/null +++ b/mobile/src/components/mobile-rich-markdown-editor-script.ts @@ -0,0 +1,14 @@ +import { + MOBILE_RICH_MARKDOWN_EDITOR_AFTER_KEYBOARD_DISMISS, + MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_END +} from './mobile-rich-markdown-editor-document-suffix' +import { MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_PRIMARY } from './mobile-rich-markdown-editor-script-primary' +import { MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_SECONDARY } from './mobile-rich-markdown-editor-script-secondary' +import { MOBILE_RICH_MARKDOWN_KEYBOARD_INSET_SCRIPT } from './mobile-rich-markdown-editor-keyboard-inset-script' +import { MOBILE_RICH_MARKDOWN_KEYBOARD_DISMISS_SCRIPT } from './mobile-rich-markdown-keyboard-dismiss-script' +import { MOBILE_RICH_MARKDOWN_SELECTION_SCRIPT } from './mobile-rich-markdown-selection-script' + +/** The editor's whole program, independent of how a host delivers it to a WebView. */ +export const MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT = `${MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_PRIMARY} +${MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_SECONDARY}${MOBILE_RICH_MARKDOWN_SELECTION_SCRIPT} +${MOBILE_RICH_MARKDOWN_KEYBOARD_DISMISS_SCRIPT}${MOBILE_RICH_MARKDOWN_EDITOR_AFTER_KEYBOARD_DISMISS}${MOBILE_RICH_MARKDOWN_KEYBOARD_INSET_SCRIPT}${MOBILE_RICH_MARKDOWN_EDITOR_SCRIPT_END}` diff --git a/mobile/src/components/mobile-selectable-text-ios.test.tsx b/mobile/src/components/mobile-selectable-text-ios.test.tsx new file mode 100644 index 00000000000..3e9ee710be4 --- /dev/null +++ b/mobile/src/components/mobile-selectable-text-ios.test.tsx @@ -0,0 +1,198 @@ +import { createElement, Fragment } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' + +const native = vi.hoisted(() => ({ available: true })) +vi.mock('react-native', () => ({ + Platform: { OS: 'ios' }, + UIManager: { hasViewManagerConfig: () => native.available }, + Linking: { openURL: vi.fn() }, + Text: 'Text', + View: 'View', + ScrollView: 'ScrollView', + Pressable: 'Pressable', + StyleSheet: { + create: (styles: unknown) => styles, + flatten: (style: unknown): object => + Array.isArray(style) + ? Object.assign({}, ...style.flat(Infinity).filter(Boolean)) + : (style ?? {}), + hairlineWidth: 1 + } +})) +vi.mock('react-native/Libraries/Utilities/codegenNativeComponent', () => ({ + default: (name: string) => name +})) +// Exercise the dependency's real span conversion without a native runtime. +vi.mock('react-native-uitextview', () => import('react-native-uitextview/src/Text')) +vi.mock('./MobileSelectableText', () => import('./MobileSelectableText.ios')) +vi.mock('./pr-sidebar/MermaidDiagram', () => ({ MermaidDiagram: 'MermaidDiagram' })) + +let renderer: ReactTestRenderer | undefined +afterEach(() => { + act(() => renderer?.unmount()) + renderer = undefined + native.available = true + vi.resetModules() + vi.restoreAllMocks() +}) + +function render(element: React.ReactElement): ReactTestRenderer { + act(() => { + renderer = create(element) + }) + return renderer! +} + +function nodes(tree: ReactTestRenderer, name: string) { + return tree.root.findAll((node) => node.type === name) +} + +describe('iOS selectable text boundary', () => { + it('preserves line-scoped keys when repeated inline spans update or disappear', async () => { + const errors = vi.spyOn(console, 'error').mockImplementation(() => {}) + const { MobileMarkdown } = await import('./MobileMarkdown') + const onOpenFile = vi.fn() + const line = '**same** [file](src/main.ts)' + const tree = render( + createElement(MobileMarkdown, { + content: `${line}\n${line}`, + rangeSelectable: true, + onOpenFile + }) + ) + expect( + nodes(tree, 'RNUITextViewChild') + .map((node) => node.props.text) + .join('') + ).toBe('same file\nsame file') + act(() => + tree.update( + createElement(MobileMarkdown, { + content: `${line}\n**changed** [file](src/main.ts)`, + rangeSelectable: true, + onOpenFile + }) + ) + ) + expect( + nodes(tree, 'RNUITextViewChild') + .map((node) => node.props.text) + .join('') + ).toBe('same file\nchanged file') + act(() => + tree.update( + createElement(MobileMarkdown, { content: line, rangeSelectable: true, onOpenFile }) + ) + ) + const spans = nodes(tree, 'RNUITextViewChild') + expect(spans.map((node) => node.props.text).join('')).toBe('same file') + act(() => spans.find((node) => node.props.text === 'file')!.props.onPress()) + expect(onOpenFile).toHaveBeenCalledExactlyOnceWith('src/main.ts') + expect(errors.mock.calls.filter((args) => String(args[0]).includes('same key'))).toEqual([]) + }) + + it.each([ + ['500', 'medium'], + ['600', 'semibold'], + ['700', 'bold'] + ] as const)('preserves font weight %s', async (fontWeight, expected) => { + const { MobileSelectableText: Text } = await import('./MobileSelectableText.ios') + const tree = render(createElement(Text, { selectable: true, style: { fontWeight } }, 'Weight')) + expect(nodes(tree, 'RNUITextViewChild')[0]!.props.style.fontWeight).toBe(expected) + }) + + it('keeps fragments, arrays, newlines and nested styles in one native root', async () => { + const { MobileSelectableText: Text } = await import('./MobileSelectableText.ios') + const tree = render( + createElement( + Text, + { selectable: true, style: { fontSize: 18 } }, + createElement(Fragment, null, 'Before ', ['one', '\n']), + createElement(Text, { style: { fontWeight: '700' } }, 'bold'), + createElement(Text, { style: { color: 'blue' } }, 'nested'), + ' after' + ) + ) + expect(nodes(tree, 'RNUITextView')).toHaveLength(1) + expect(nodes(tree, 'Text')).toHaveLength(0) + const spans = nodes(tree, 'RNUITextViewChild') + expect(spans.map((node) => node.props.text).join('')).toBe('Before one\nboldnested after') + expect(spans.find((node) => node.props.text === 'bold')?.props.style).toMatchObject({ + fontSize: 18, + fontWeight: 'bold' + }) + expect(spans.find((node) => node.props.text === 'nested')?.props.style).toMatchObject({ + fontSize: 18, + color: 'blue' + }) + }) + + it('preserves Markdown text, inline styles and file-link callbacks', async () => { + const { MobileMarkdown } = await import('./MobileMarkdown') + const onOpenFile = vi.fn() + const tree = render( + createElement(MobileMarkdown, { + content: 'Hello 😀 [src/main.ts](src/main.ts) and `code`.\nNext line.', + rangeSelectable: true, + onOpenFile + }) + ) + const spans = nodes(tree, 'RNUITextViewChild') + expect(spans.map((node) => node.props.text).join('')).toBe( + 'Hello 😀 src/main.ts and code.\nNext line.' + ) + const link = spans.find((node) => node.props.text === 'src/main.ts')! + expect(link.props.style.color).toBeDefined() + act(() => link.props.onPress()) + expect(onOpenFile).toHaveBeenCalledExactlyOnceWith('src/main.ts') + expect(nodes(tree, 'RNUITextView')).toHaveLength(1) + }) + + it('keeps ordinary button labels on React Native Text', async () => { + const { MobileSelectableText: Text } = await import('./MobileSelectableText.ios') + const tree = render(createElement(Text, null, 'Submit')) + expect(nodes(tree, 'RNUITextView')).toHaveLength(0) + expect(nodes(tree, 'Text')).toHaveLength(1) + }) + + it('uses native range selection only when Markdown opts in', async () => { + const { MobileMarkdown } = await import('./MobileMarkdown') + const tree = render(createElement(MobileMarkdown, { content: 'Transcript prose' })) + expect(nodes(tree, 'RNUITextView')).toHaveLength(0) + expect( + nodes(tree, 'Text').find((node) => node.children.includes('Transcript prose'))?.props + .selectable + ).toBe(false) + act(() => + tree.update( + createElement(MobileMarkdown, { content: 'Transcript prose', rangeSelectable: true }) + ) + ) + expect(nodes(tree, 'RNUITextView')).toHaveLength(1) + }) + + it('keeps code-language labels on styled React Native Text', async () => { + const { MobileMarkdown } = await import('./MobileMarkdown') + const tree = render( + createElement(MobileMarkdown, { + content: '```ts\nconst value = 1\n```', + rangeSelectable: true + }) + ) + const label = nodes(tree, 'Text').find((node) => node.children.join('') === 'ts')! + expect(label.props.style.textTransform).toBe('uppercase') + expect(nodes(tree, 'RNUITextView')).toHaveLength(1) + }) + + it('falls back for older clients without the native view', async () => { + native.available = false + const { MobileSelectableText: Text } = await import('./MobileSelectableText.ios') + const tree = render( + createElement(Text, { selectable: true }, 'Old client ', createElement(Text, null, 'inline')) + ) + expect(nodes(tree, 'RNUITextView')).toHaveLength(0) + expect(nodes(tree, 'Text')).toHaveLength(2) + expect(nodes(tree, 'Text')[0]!.props.selectable).toBe(true) + }) +}) diff --git a/mobile/src/components/new-worktree-modal-types.ts b/mobile/src/components/new-worktree-modal-types.ts index 4246dd6f29f..8dae250cd71 100644 --- a/mobile/src/components/new-worktree-modal-types.ts +++ b/mobile/src/components/new-worktree-modal-types.ts @@ -23,6 +23,7 @@ export type NewWorktreeModalProps = { hostId?: string existingWorktreePaths?: readonly string[] existingWorktrees?: readonly { repoId: string; branch: string }[] + openExternalUrl: (url: string) => Promise onCreated: (worktreeId: string, name: string) => void onClose: () => void } diff --git a/mobile/src/components/use-mobile-rich-markdown-editor-controller.ts b/mobile/src/components/use-mobile-rich-markdown-editor-controller.ts new file mode 100644 index 00000000000..8b154680b38 --- /dev/null +++ b/mobile/src/components/use-mobile-rich-markdown-editor-controller.ts @@ -0,0 +1,114 @@ +import { useCallback, useEffect, useRef } from 'react' +import { normalizeMobileRichMarkdownKeyboardInset } from './mobile-rich-markdown-editor-keyboard-inset-script' +import type { + MobileRichMarkdownCommand, + MobileRichMarkdownEditorMessage, + MobileRichMarkdownEditorProps, + MobileRichMarkdownEditorTransport +} from './mobile-rich-markdown-editor-contract' + +export function normalizeExternalEditorUrl(value: string): string | null { + const url = value.trim() + if (!url) { + return null + } + for (let index = 0; index < url.length; index += 1) { + const code = url.charCodeAt(index) + if (code <= 32 || code === 127) { + return null + } + } + if (/^mailto:/i.test(url)) { + return url + } + if (!/^https?:\/\//i.test(url)) { + return null + } + try { + const parsed = new URL(url) + return parsed.protocol === 'http:' || parsed.protocol === 'https:' ? parsed.toString() : null + } catch { + return null + } +} + +export function useMobileRichMarkdownEditorController({ + content, + editable, + onChange, + onKeyboardInsetChange, + onOpenLink, + transport +}: MobileRichMarkdownEditorProps & { transport: MobileRichMarkdownEditorTransport }) { + const readyRef = useRef(false) + const documentGenerationRef = useRef(0) + const currentEditorContentRef = useRef(null) + + const applyContent = useCallback( + (nextContent: string) => { + documentGenerationRef.current += 1 + currentEditorContentRef.current = nextContent + transport.setMarkdown(nextContent, documentGenerationRef.current) + }, + [transport] + ) + + useEffect(() => { + if (readyRef.current && currentEditorContentRef.current !== content) { + applyContent(content) + } + }, [applyContent, content]) + + useEffect(() => { + if (readyRef.current) { + transport.setEditable(editable) + } + }, [editable, transport]) + + // Clear any reported keyboard inset when the editor unmounts so a lifted + // Save/Discard bar settles back once the tab closes. + useEffect(() => { + return () => onKeyboardInsetChange?.(0) + }, [onKeyboardInsetChange]) + + const handleMessage = useCallback( + (message: Partial) => { + if (message.type === 'ready') { + readyRef.current = true + applyContent(content) + transport.setEditable(editable) + return + } + if ( + message.type === 'change' && + typeof message.markdown === 'string' && + message.generation === documentGenerationRef.current + ) { + currentEditorContentRef.current = message.markdown + onChange(message.markdown) + return + } + if (message.type === 'openLink' && typeof message.url === 'string') { + const url = normalizeExternalEditorUrl(message.url) + if (url) { + onOpenLink(url) + } + return + } + if (message.type === 'keyboardInset' && typeof message.bottom === 'number') { + const bottom = normalizeMobileRichMarkdownKeyboardInset(message.bottom) + if (bottom !== null) { + onKeyboardInsetChange?.(bottom) + } + } + }, + [applyContent, content, editable, onChange, onKeyboardInsetChange, onOpenLink, transport] + ) + + const runCommand = useCallback( + (command: MobileRichMarkdownCommand) => transport.runCommand(command), + [transport] + ) + + return { handleMessage, runCommand } +} diff --git a/mobile/src/diagnostics/connection-diagnostics-screen-styles.ts b/mobile/src/diagnostics/connection-diagnostics-screen-styles.ts new file mode 100644 index 00000000000..67a08050711 --- /dev/null +++ b/mobile/src/diagnostics/connection-diagnostics-screen-styles.ts @@ -0,0 +1,95 @@ +import { StyleSheet } from 'react-native' +import { colors, spacing, typography } from '../theme/mobile-theme' + +export const connectionDiagnosticsScreenStyles = StyleSheet.create({ + container: { flex: 1, backgroundColor: colors.bgBase, padding: spacing.lg }, + topRow: { flexDirection: 'row', alignItems: 'center', marginBottom: spacing.lg }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { fontSize: 20, fontWeight: '700', color: colors.textPrimary }, + hostPicker: { + flexDirection: 'row', + flexWrap: 'wrap', + gap: spacing.sm, + marginBottom: spacing.md + }, + hostChip: { + paddingVertical: spacing.xs + 2, + paddingHorizontal: spacing.md, + borderRadius: 16, + backgroundColor: colors.bgRaised + }, + hostChipActive: { + backgroundColor: colors.bgPanel, + borderWidth: 1, + borderColor: colors.borderSubtle + }, + hostChipText: { fontSize: typography.metaSize, color: colors.textSecondary, maxWidth: 160 }, + hostChipTextActive: { color: colors.textPrimary, fontWeight: '600' }, + statusRow: { + flexDirection: 'row', + alignItems: 'center', + justifyContent: 'space-between', + marginBottom: spacing.sm + }, + statusText: { fontSize: typography.metaSize, color: colors.textSecondary }, + diagnosisCard: { + backgroundColor: colors.bgPanel, + borderWidth: StyleSheet.hairlineWidth, + borderColor: colors.borderSubtle, + borderRadius: 10, + padding: spacing.md, + marginBottom: spacing.md + }, + diagnosisHeading: { + fontSize: typography.metaSize, + fontWeight: '600', + color: colors.textPrimary, + marginBottom: spacing.xs + }, + diagnosisText: { fontSize: typography.metaSize, color: colors.textPrimary, lineHeight: 18 }, + diagnosisNext: { + fontSize: typography.metaSize, + color: colors.textSecondary, + lineHeight: 18, + marginTop: spacing.xs + }, + privacyHint: { marginTop: spacing.sm, fontSize: 11, lineHeight: 15, color: colors.textMuted }, + sendButton: { + marginTop: spacing.md, + alignSelf: 'flex-start', + flexDirection: 'row', + alignItems: 'center', + gap: spacing.xs, + paddingVertical: spacing.sm, + paddingHorizontal: spacing.md, + borderRadius: 8, + backgroundColor: colors.bgRaised + }, + sendButtonText: { + fontSize: typography.metaSize, + fontWeight: '600', + color: colors.textPrimary + }, + copyButton: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.xs + 2, + paddingVertical: spacing.xs + 2, + paddingHorizontal: spacing.md, + borderRadius: 8, + backgroundColor: colors.bgRaised + }, + copyButtonText: { + fontSize: typography.metaSize, + fontWeight: '600', + color: colors.textPrimary + }, + emptyText: { fontSize: typography.metaSize, color: colors.textMuted, lineHeight: 18 } +}) diff --git a/mobile/src/diagnostics/connection-diagnostics-screen.tsx b/mobile/src/diagnostics/connection-diagnostics-screen.tsx new file mode 100644 index 00000000000..063f2d0dcad --- /dev/null +++ b/mobile/src/diagnostics/connection-diagnostics-screen.tsx @@ -0,0 +1,103 @@ +import { useCallback, useState, type ReactNode } from 'react' +import { ConnectionDiagnosticsView } from './connection-diagnostics-view' +import { + diagnoseConnection, + getReportableConnectionIncidentId +} from './connection-diagnostics-analysis' +import { + getDiagnosticsSubmissionState, + updateDiagnosticsSubmissionState, + type DiagnosticsSubmissionStates +} from './connection-diagnostics-screen-data' +import type { DiagnosticsDeviceOperations } from './diagnostics-device-operations' +import type { + ConnectionLogEntry, + ConnectionState, + MobileConnectionDiagnosticPath +} from '../transport/types' + +export function ConnectionDiagnosticsScreen({ + device, + host, + state, + reconnectAttempts, + activePath, + pendingPath, + entries, + writeClipboard, + onBack, + hostPicker +}: { + device: DiagnosticsDeviceOperations | null + /** Null only when no host is paired; an empty name or endpoint is still a host. */ + host: { id: string; name: string; endpoint: string } | null + state: ConnectionState + reconnectAttempts: number + activePath?: MobileConnectionDiagnosticPath + pendingPath?: MobileConnectionDiagnosticPath | null + entries: readonly ConnectionLogEntry[] + writeClipboard: (report: string) => Promise + onBack: () => void + hostPicker?: ReactNode +}) { + const [copiedHostId, setCopiedHostId] = useState(null) + const [submissionStates, setSubmissionStates] = useState({}) + + const diagnosisArgs = host + ? { endpoint: host.endpoint, state, activePath, pendingPath, entries } + : null + const diagnosis = diagnosisArgs ? diagnoseConnection(diagnosisArgs) : null + const incidentId = diagnosisArgs ? getReportableConnectionIncidentId(diagnosisArgs) : null + const hostId = host?.id ?? null + const submissionKey = hostId && incidentId ? `${hostId}:${incidentId}` : null + const submissionState = getDiagnosticsSubmissionState(submissionStates, submissionKey) + const copied = copiedHostId !== null && copiedHostId === hostId + + const copyDiagnostics = useCallback(async () => { + if (!device || !hostId) { + return + } + const { report } = await device.report() + await writeClipboard(report) + setCopiedHostId(hostId) + setTimeout(() => setCopiedHostId((current) => (current === hostId ? null : current)), 2000) + }, [device, hostId, writeClipboard]) + + const sendDiagnostics = useCallback(async () => { + if (!device || !hostId || !submissionKey || submissionState === 'sending') { + return + } + const startedKey = submissionKey + setSubmissionStates((states) => updateDiagnosticsSubmissionState(states, startedKey, 'sending')) + const fresh = await device.report() + if (`${hostId}:${fresh.incidentId ?? ''}` !== startedKey) { + setSubmissionStates((states) => updateDiagnosticsSubmissionState(states, startedKey, null)) + return + } + const result = await device.submit({ + report: fresh.report, + appVersion: fresh.appVersion, + platform: fresh.platform + }) + setSubmissionStates((states) => + updateDiagnosticsSubmissionState(states, startedKey, result.ok ? 'sent' : 'failed') + ) + }, [device, hostId, submissionKey, submissionState]) + + return ( + + ) +} diff --git a/mobile/src/diagnostics/connection-diagnostics-view.tsx b/mobile/src/diagnostics/connection-diagnostics-view.tsx new file mode 100644 index 00000000000..c084c761e8b --- /dev/null +++ b/mobile/src/diagnostics/connection-diagnostics-view.tsx @@ -0,0 +1,119 @@ +import type { ReactNode } from 'react' +import { View, Text, Pressable } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { ChevronLeft, Copy, Check, Send } from 'lucide-react-native' +import { colors, spacing } from '../theme/mobile-theme' +import { ConnectionLog } from '../components/ConnectionLog' +import { connectionDiagnosticsScreenStyles as styles } from './connection-diagnostics-screen-styles' +import type { ConnectionLogEntry, ConnectionState } from '../transport/types' +import type { ConnectionDiagnosis } from './connection-diagnostics-analysis' +import type { DiagnosticsSubmissionState } from './connection-diagnostics-screen-data' + +export function ConnectionDiagnosticsView({ + hostPicker, + hasHost, + hostName, + state, + reconnectAttempts, + copied, + copyDiagnostics, + diagnosis, + submissionState, + sendDiagnostics, + entries, + onBack +}: { + hostPicker?: ReactNode + hasHost: boolean + hostName: string + state: ConnectionState + reconnectAttempts: number + copied: boolean + copyDiagnostics: () => Promise + diagnosis: ConnectionDiagnosis | null + submissionState: DiagnosticsSubmissionState | 'idle' + sendDiagnostics: () => Promise + entries: readonly ConnectionLogEntry[] + onBack: () => void +}) { + const insets = useSafeAreaInsets() + return ( + + + + + + Network diagnostics + + + {hostPicker} + {hasHost ? ( + <> + + + {state} + {reconnectAttempts > 0 ? ` · attempt ${reconnectAttempts}` : ''} + + void copyDiagnostics()}> + {copied ? ( + + ) : ( + + )} + {copied ? 'Copied' : 'Copy report'} + + + {diagnosis && ( + + What this suggests + {diagnosis.likelyCause} + {diagnosis.nextStep} + {diagnosis.reportability === 'orca-relay' && ( + <> + + Sends a size-limited redacted report including host name, endpoint, versions, + connection state, and events—never terminal contents or credentials. + + void sendDiagnostics()} + disabled={submissionState === 'sending'} + > + {submissionState === 'sent' ? ( + + ) : ( + + )} + + {submissionState === 'sending' + ? 'Sending…' + : submissionState === 'sent' + ? 'Diagnostics sent' + : submissionState === 'failed' + ? 'Retry sending' + : 'Send diagnostics to Orca'} + + + + )} + + )} + {entries.length > 0 ? ( + + ) : ( + + No connection events yet. Events appear as the app dials this host. + + )} + + ) : ( + No paired hosts. + )} + + ) +} diff --git a/mobile/src/diagnostics/diagnostics-device-operations.ts b/mobile/src/diagnostics/diagnostics-device-operations.ts new file mode 100644 index 00000000000..7f85655c723 --- /dev/null +++ b/mobile/src/diagnostics/diagnostics-device-operations.ts @@ -0,0 +1,16 @@ +import type { + ConnectionDiagnosticsSubmission, + ConnectionDiagnosticsSubmissionResult +} from './connection-diagnostics-submission' + +/** A fresh report plus the incident it describes, so a stale send can be dropped. */ +export type ConnectionDiagnosticsReport = ConnectionDiagnosticsSubmission & { + incidentId: string | null +} + +export interface DiagnosticsDeviceOperations { + report(): Promise + submit( + submission: ConnectionDiagnosticsSubmission + ): Promise +} diff --git a/mobile/src/diagnostics/native-diagnostics-operations.ts b/mobile/src/diagnostics/native-diagnostics-operations.ts new file mode 100644 index 00000000000..2b7b7fe6f0b --- /dev/null +++ b/mobile/src/diagnostics/native-diagnostics-operations.ts @@ -0,0 +1,51 @@ +import { Platform } from 'react-native' +import Constants from 'expo-constants' +import type { HostProfile } from '../transport/types' +import type { RpcClientContextValue } from '../transport/rpc-client-context-contract' +import { connectionLogStore } from '../transport/persisted-connection-log-store' +import { loadHostAppVersion } from '../transport/host-app-version-store' +import { readConnectionDiagnosticsSnapshot } from './connection-diagnostics-screen-data' +import { getReportableConnectionIncidentId } from './connection-diagnostics-analysis' +import { buildConnectionDiagnosticsReport } from './connection-diagnostics-report' +import { submitConnectionDiagnostics } from './connection-diagnostics-submission' +import type { DiagnosticsDeviceOperations } from './diagnostics-device-operations' + +export function createNativeDiagnosticsOperations( + host: HostProfile, + context: RpcClientContextValue, + liveDesktopAppVersion?: string | null +): DiagnosticsDeviceOperations { + return { + async report() { + const appVersion = Constants.expoConfig?.version ?? 'unknown' + const platform = `${Platform.OS} ${Platform.Version ?? ''}`.trim() + const desktopAppVersion = liveDesktopAppVersion ?? (await loadHostAppVersion(host.id)) + const snapshot = await readConnectionDiagnosticsSnapshot(context, connectionLogStore, host.id) + return { + report: buildConnectionDiagnosticsReport({ + hostName: host.name, + endpoint: host.endpoint, + state: snapshot.state, + reconnectAttempts: snapshot.reconnectAttempts, + lastConnectedAt: snapshot.lastConnectedAt, + platform, + appVersion, + desktopAppVersion, + entries: snapshot.entries, + activePath: snapshot.activePath, + pendingPath: snapshot.pendingPath + }), + appVersion, + platform, + incidentId: getReportableConnectionIncidentId({ + endpoint: host.endpoint, + state: snapshot.state, + activePath: snapshot.activePath, + pendingPath: snapshot.pendingPath, + entries: snapshot.entries + }) + } + }, + submit: submitConnectionDiagnostics + } +} diff --git a/mobile/src/diagnostics/troubleshoot-view.tsx b/mobile/src/diagnostics/troubleshoot-view.tsx new file mode 100644 index 00000000000..0fd86605f13 --- /dev/null +++ b/mobile/src/diagnostics/troubleshoot-view.tsx @@ -0,0 +1,166 @@ +import { useCallback, useState } from 'react' +import { View, Text, Pressable, ScrollView, ActivityIndicator } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { + ChevronLeft, + ChevronDown, + ChevronUp, + Activity, + CheckCircle2, + ScrollText, + XCircle, + AlertTriangle +} from 'lucide-react-native' +import { colors, spacing } from '../theme/mobile-theme' +import { troubleshootCommonIssues } from './troubleshoot-common-issues' +import { troubleshootScreenStyles as styles } from './troubleshoot-screen-styles' +export type DiagnosticStatus = 'idle' | 'running' | 'done' + +export type CheckResult = { + label: string + status: 'pass' | 'fail' | 'warn' + detail: string +} + +function StatusIcon({ status }: { status: CheckResult['status'] }) { + switch (status) { + case 'pass': + return + case 'fail': + return + case 'warn': + return + } +} + +export function TroubleshootView({ + rootRef, + diagnosticStatus, + checks, + runDiagnostics, + onBack, + onConnectionLog +}: { + rootRef?: (node: View | null) => void + diagnosticStatus: DiagnosticStatus + checks: CheckResult[] + runDiagnostics: () => void + onBack: () => void + onConnectionLog: () => void +}) { + const insets = useSafeAreaInsets() + const [expandedId, setExpandedId] = useState(null) + const toggleSection = useCallback( + (id: string) => setExpandedId((prev) => (prev === id ? null : id)), + [] + ) + return ( + + + + + + Troubleshooting + + + + [ + styles.diagnosticButton, + pressed && styles.diagnosticButtonPressed, + diagnosticStatus === 'running' && styles.diagnosticButtonDisabled + ]} + testID="diagnostics-run" + onPress={runDiagnostics} + disabled={diagnosticStatus === 'running'} + > + {diagnosticStatus === 'running' ? ( + + ) : ( + + )} + + {diagnosticStatus === 'running' + ? 'Running…' + : diagnosticStatus === 'done' + ? 'Run again' + : 'Run diagnostics'} + + + + [ + styles.diagnosticButton, + pressed && styles.diagnosticButtonPressed + ]} + onPress={onConnectionLog} + > + + View network diagnostics + + + {checks.length > 0 && ( + + {checks.map((check, i) => ( + + {i > 0 && } + + + {check.label} + + {check.detail} + + + + ))} + + )} + + Common issues + + + {troubleshootCommonIssues.map((section, i) => ( + + {i > 0 && } + [styles.accordionHeader, pressed && styles.rowPressed]} + onPress={() => toggleSection(section.id)} + > + {section.icon} + {section.title} + {expandedId === section.id ? ( + + ) : ( + + )} + + {expandedId === section.id && ( + + {section.steps.map((step, j) => ( + + • + {step} + + ))} + + )} + + ))} + + + + + + ) +} diff --git a/mobile/src/diagnostics/use-troubleshoot-diagnostics.ts b/mobile/src/diagnostics/use-troubleshoot-diagnostics.ts new file mode 100644 index 00000000000..0645a34331e --- /dev/null +++ b/mobile/src/diagnostics/use-troubleshoot-diagnostics.ts @@ -0,0 +1,128 @@ +import { useCallback, useRef, useState } from 'react' +import type { View } from 'react-native' +import { Platform } from 'react-native' +import { loadHosts } from '../transport/host-store' +import { + startDiagnosticFetchTimeout, + type DiagnosticFetchTimeout +} from './diagnostic-fetch-timeout' +import { formatEndpoint, testHostReachability, unreachableHostDetail } from './host-reachability' +import type { CheckResult, DiagnosticStatus } from './troubleshoot-view' + +export function useTroubleshootDiagnostics() { + const [diagnosticStatus, setDiagnosticStatus] = useState('idle') + const [checks, setChecks] = useState([]) + const abortRef = useRef(false) + const diagnosticRunRef = useRef(0) + const activeInternetCheckRef = useRef(null) + + const rootRef = useCallback((node: View | null): void => { + if (node !== null) { + return + } + // Why: diagnostics can outlive the screen; cancel the active run when the + // route detaches without a passive cleanup-only Effect. + abortRef.current = true + diagnosticRunRef.current += 1 + activeInternetCheckRef.current?.dispose() + activeInternetCheckRef.current = null + }, []) + + const runDiagnostics = useCallback(async () => { + const runId = diagnosticRunRef.current + 1 + diagnosticRunRef.current = runId + abortRef.current = false + activeInternetCheckRef.current?.dispose() + activeInternetCheckRef.current = null + setDiagnosticStatus('running') + setChecks([]) + + const results: CheckResult[] = [] + const isCurrentRun = () => !abortRef.current && diagnosticRunRef.current === runId + + try { + const hosts = await loadHosts() + results.push( + hosts.length > 0 + ? { label: 'Paired hosts', status: 'pass', detail: `${hosts.length} paired` } + : { label: 'Paired hosts', status: 'fail', detail: 'None — scan a QR to pair' } + ) + } catch { + results.push({ label: 'Paired hosts', status: 'warn', detail: 'Could not read host data' }) + } + + if (!isCurrentRun()) { + return + } + setChecks([...results]) + + const internetCheck = startDiagnosticFetchTimeout(5000) + activeInternetCheckRef.current = internetCheck + try { + const resp = await fetch('https://dns.google/resolve?name=example.com&type=A', { + signal: internetCheck.signal + }) + if (!isCurrentRun()) { + return + } + results.push( + resp.ok + ? { label: 'Internet', status: 'pass', detail: 'Connected' } + : { label: 'Internet', status: 'warn', detail: 'Unexpected response' } + ) + } catch { + if (!isCurrentRun()) { + return + } + results.push({ label: 'Internet', status: 'fail', detail: 'No connection' }) + } finally { + internetCheck.dispose() + if (activeInternetCheckRef.current === internetCheck) { + activeInternetCheckRef.current = null + } + } + + if (!isCurrentRun()) { + return + } + setChecks([...results]) + + try { + const hosts = await loadHosts() + for (const host of hosts) { + if (!isCurrentRun()) { + return + } + const reachable = await testHostReachability(host.endpoint) + if (!isCurrentRun()) { + return + } + results.push({ + label: host.name, + status: reachable ? 'pass' : 'fail', + detail: reachable + ? `Reachable at ${formatEndpoint(host.endpoint)}` + : unreachableHostDetail(host.endpoint) + }) + setChecks([...results]) + } + } catch { + results.push({ label: 'Hosts', status: 'warn', detail: 'Could not test' }) + } + + if (!isCurrentRun()) { + return + } + + results.push({ + label: 'Platform', + status: 'pass', + detail: `${Platform.OS} ${Platform.Version ?? ''}` + }) + + setChecks([...results]) + setDiagnosticStatus('done') + }, []) + + return { rootRef, diagnosticStatus, checks, runDiagnostics } +} diff --git a/mobile/src/host-screen/host-screen-overlays.tsx b/mobile/src/host-screen/host-screen-overlays.tsx index 07cbbd58fc5..0f15f9d4e61 100644 --- a/mobile/src/host-screen/host-screen-overlays.tsx +++ b/mobile/src/host-screen/host-screen-overlays.tsx @@ -1,4 +1,4 @@ -import { Pressable, Text, View } from 'react-native' +import { Linking, Pressable, Text, View } from 'react-native' import { Check, Moon } from 'lucide-react-native' import { buildWorktreeNavigationActions } from '../agent-history/worktree-navigation-actions' import { ActionSheetContent } from '../components/ActionSheetModal' @@ -215,6 +215,7 @@ export function HostScreenOverlays({ controller }: { controller: HostScreenContr hostId={hostId} existingWorktreePaths={existingWorktreePaths} existingWorktrees={state.worktrees} + openExternalUrl={(url) => Linking.openURL(url)} onVisibleChange={(visible) => { state.newWorktreeModalVisibleRef.current = visible }} diff --git a/mobile/src/host-screen/use-host-repo-metadata.ts b/mobile/src/host-screen/use-host-repo-metadata.ts index 198efbaf3ea..6840184c3a0 100644 --- a/mobile/src/host-screen/use-host-repo-metadata.ts +++ b/mobile/src/host-screen/use-host-repo-metadata.ts @@ -18,7 +18,7 @@ type SshTargetSummaryRow = { id: string; label: string } async function requestResult(client: RpcClient, method: string): Promise { try { const response = await client.sendRequest(method) - return response.ok ? (response as RpcSuccess).result : null + return response.ok ? response.result : null } catch { // Best-effort: hosts that predate a method still list repos; labels degrade to host ids. return null diff --git a/mobile/src/rpc-params-contract-type-only-boundary.test.ts b/mobile/src/rpc-params-contract-type-only-boundary.test.ts new file mode 100644 index 00000000000..bc9088bea0b --- /dev/null +++ b/mobile/src/rpc-params-contract-type-only-boundary.test.ts @@ -0,0 +1,144 @@ +import { readFileSync, readdirSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { extname, join, relative, resolve } from 'node:path' +import ts from 'typescript' +import { describe, expect, it } from 'vitest' + +// Why: src/shared/rpc-contract/*-params.ts hold the host's zod schemas. Bundling one +// into the app would let client code call parse(), and requiredString is +// z.unknown().transform(...) — it coerces a non-string to '' instead of rejecting it, +// silently changing the bytes the phone puts on the wire. Types only, never values. +const mobileRoot = fileURLToPath(new URL('..', import.meta.url)) +const contractRoot = resolve(mobileRoot, '..', 'src', 'shared', 'rpc-contract') +const scannedRoots = ['app', 'src'].map((directory) => join(mobileRoot, directory)) +const sourceExtensions = new Set(['.js', '.jsx', '.ts', '.tsx']) + +function sourceFiles(directory: string): string[] { + return readdirSync(directory, { withFileTypes: true }).flatMap((entry) => { + const path = join(directory, entry.name) + if (entry.isDirectory()) { + return entry.name === 'node_modules' ? [] : sourceFiles(path) + } + return [path] + }) +} + +function targetsContract(path: string, specifier: string): boolean { + if (!specifier.startsWith('.')) { + return false + } + const resolved = resolve(path, '..', specifier) + return resolved === contractRoot || resolved.startsWith(`${contractRoot}/`) +} + +function parse(path: string, source: string): ts.SourceFile { + const extension = extname(path) + return ts.createSourceFile( + path, + source, + ts.ScriptTarget.Latest, + true, + extension === '.tsx' || extension === '.jsx' ? ts.ScriptKind.TSX : ts.ScriptKind.TS + ) +} + +// Returns the specifiers that would pull contract *values* into the bundle. +export function contractValueImports(path: string, source: string): string[] { + const sourceFile = parse(path, source) + const offenders: string[] = [] + const visit = (node: ts.Node): void => { + if (ts.isImportDeclaration(node) && ts.isStringLiteral(node.moduleSpecifier)) { + const specifier = node.moduleSpecifier.text + if (targetsContract(path, specifier)) { + const clause = node.importClause + const everyNamedIsType = + clause?.isTypeOnly === true || + (clause?.namedBindings !== undefined && + ts.isNamedImports(clause.namedBindings) && + clause.namedBindings.elements.every((element) => element.isTypeOnly)) + // A bare `import './x'` has no clause at all and still emits a require. + if (!everyNamedIsType) { + offenders.push(specifier) + } + } + } + if ( + ts.isExportDeclaration(node) && + node.moduleSpecifier && + ts.isStringLiteral(node.moduleSpecifier) + ) { + const specifier = node.moduleSpecifier.text + if (targetsContract(path, specifier)) { + const everyNamedIsType = + node.isTypeOnly || + (node.exportClause !== undefined && + ts.isNamedExports(node.exportClause) && + node.exportClause.elements.every((element) => element.isTypeOnly)) + if (!everyNamedIsType) { + offenders.push(specifier) + } + } + } + if (ts.isCallExpression(node)) { + const callee = node.expression + const isDynamic = callee.kind === ts.SyntaxKind.ImportKeyword + const isRequire = ts.isIdentifier(callee) && callee.text === 'require' + const argument = node.arguments[0] + if ( + (isDynamic || isRequire) && + argument && + ts.isStringLiteral(argument) && + targetsContract(path, argument.text) + ) { + offenders.push(argument.text) + } + } + ts.forEachChild(node, visit) + } + visit(sourceFile) + return offenders +} + +describe('RPC params contract boundary', () => { + it('flags every shape that would emit a runtime require', () => { + const path = join(mobileRoot, 'src', 'probe.ts') + const contract = '../../src/shared/rpc-contract/repo-params' + expect(contractValueImports(path, `import { RepoSelector } from '${contract}'`)).toEqual([ + contract + ]) + expect(contractValueImports(path, `import '${contract}'`)).toEqual([contract]) + expect(contractValueImports(path, `export { RepoSelector } from '${contract}'`)).toEqual([ + contract + ]) + expect(contractValueImports(path, `const s = require('${contract}')`)).toEqual([contract]) + expect(contractValueImports(path, `const s = await import('${contract}')`)).toEqual([contract]) + expect(contractValueImports(path, `import type { RepoSelector } from '${contract}'`)).toEqual( + [] + ) + expect(contractValueImports(path, `import { type RepoSelector } from '${contract}'`)).toEqual( + [] + ) + expect(contractValueImports(path, `export type { RepoSelector } from '${contract}'`)).toEqual( + [] + ) + expect( + contractValueImports( + path, + `import type { GitHubWorkItem } from '../../src/shared/github/work-item-types'` + ) + ).toEqual([]) + }) + + it('keeps every mobile import of the params contract type-only', () => { + const offenders = scannedRoots + .flatMap(sourceFiles) + .filter((path) => sourceExtensions.has(extname(path))) + .flatMap((path) => + contractValueImports(path, readFileSync(path, 'utf8')).map( + (specifier) => `${relative(mobileRoot, path)} -> ${specifier}` + ) + ) + + expect(offenders).toEqual([]) + }) +}) diff --git a/mobile/src/session/MobileNativeChatComposer.tsx b/mobile/src/session/MobileNativeChatComposer.tsx index c16b69ced89..20c43aa6c8b 100644 --- a/mobile/src/session/MobileNativeChatComposer.tsx +++ b/mobile/src/session/MobileNativeChatComposer.tsx @@ -130,7 +130,7 @@ export function MobileNativeChatComposer({ if (trigger.kind === 'slash') { const commands = structuredCommands !== undefined - ? structuredSlashCommands(structuredCommands) + ? structuredSlashCommands(structuredCommands, agent) : agent ? getVerifiedNativeChatCommands(agent) : [] diff --git a/mobile/src/session/MobileNativeChatMessage.test.ts b/mobile/src/session/MobileNativeChatMessage.test.ts index 677b76d240c..67e90132866 100644 --- a/mobile/src/session/MobileNativeChatMessage.test.ts +++ b/mobile/src/session/MobileNativeChatMessage.test.ts @@ -9,6 +9,7 @@ vi.mock('react-native', async () => { const Text = ({ children, ...props }: { children?: unknown }): unknown => React.createElement('Text', props, children) return { + ActivityIndicator: 'ActivityIndicator', Animated: { Text, Value: class { @@ -106,6 +107,21 @@ describe('MobileNativeChatMessage', () => { expect(texts.some((text) => text.includes('/tmp/host.png'))).toBe(true) }) + it('makes user message text selectable', () => { + const tree = render(userMessage([{ type: 'text', text: 'Prompt I typed' }])) + const text = tree.root + .findAllByType('Text' as never) + .find((node) => String(node.children.join('')) === 'Prompt I typed') + expect(text?.props.selectable).toBe(true) + }) + + it('routes assistant prose through selectable Markdown', () => { + const tree = render(toolMessage([{ type: 'text', text: 'Agent reply prose' }])) + const markdown = tree.root.findByType('MobileMarkdown' as never) + expect(markdown.props.content).toBe('Agent reply prose') + expect(markdown.props.rangeSelectable).toBe(true) + }) + it('labels a tool row with the target path instead of raw input JSON', () => { const tree = render( toolMessage([{ type: 'tool-call', name: 'Read', input: { file_path: 'src/index.ts' } }]), @@ -253,12 +269,12 @@ describe('MobileNativeChatMessage', () => { expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) }) - it('renders the turn status row under a user message', () => { + it('renders the settled turn status row under a user message', () => { const tree = render(userMessage([{ type: 'text', text: 'go' }]), { structuredActivityUi: true, - turnStatus: { startedAt: Date.now(), thinking: true, workedSeconds: null } + turnStatus: { startedAt: Date.now() - 3_000, thinking: false, workedSeconds: 3 } }) - expect(textIn(tree.root)).toContain('Thinking') + expect(textIn(tree.root)).toContain('Worked for 3s') }) it('does not render a turn status row without one', () => { diff --git a/mobile/src/session/MobileNativeChatMessage.tsx b/mobile/src/session/MobileNativeChatMessage.tsx index cc6c086b1f3..9b480fdd7d9 100644 --- a/mobile/src/session/MobileNativeChatMessage.tsx +++ b/mobile/src/session/MobileNativeChatMessage.tsx @@ -1,7 +1,6 @@ -import { memo, useEffect, useRef, useState } from 'react' -import { Image, Pressable, Text, View } from 'react-native' -import * as Clipboard from 'expo-clipboard' -import { ArrowUp, Copy } from 'lucide-react-native' +import { MobileSelectableText as Text } from '../components/MobileSelectableText' +import { memo } from 'react' +import { Image, Text as NativeText, View } from 'react-native' import { splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' import { selectActiveToolCall } from '../../../src/shared/native-chat-tool-activity' import { isImageRefBlock, isTextBlock } from '../../../src/shared/native-chat-types' @@ -10,10 +9,8 @@ import { MobileMarkdown } from '../components/MobileMarkdown' import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' import { ToolRun } from './MobileNativeChatToolRun' import type { NativeChatTurnStatus } from './use-mobile-native-chat-turn-status' -import { colors } from '../theme/mobile-theme' import { isRenderableImageUri } from './mobile-native-chat-image-preview' import { styles, TEXT_SIZE } from './mobile-native-chat-message-styles' -import { nativeChatMessageText } from './mobile-native-chat-message-text' function Prose({ block, @@ -37,7 +34,12 @@ function Prose({ ) } return ( - + ) } if (isImageRefBlock(block)) { @@ -55,53 +57,18 @@ function Prose({ ) } return ( - + 🖼 {block.alt ?? block.path ?? block.url ?? 'image'} - + ) } return null } -/** Subtle top-right controls for an agent message: copy its prose, or scroll so - * this message's top aligns to the top of the viewport. */ -function AgentControls({ - onCopy, - onScrollToTop -}: { - onCopy: () => void - onScrollToTop?: () => void -}): React.JSX.Element { - return ( - - [styles.controlButton, pressed && styles.controlPressed]} - onPress={onCopy} - hitSlop={8} - accessibilityLabel="Copy message" - > - - - {onScrollToTop ? ( - [styles.controlButton, pressed && styles.controlPressed]} - onPress={onScrollToTop} - hitSlop={8} - accessibilityLabel="Scroll this message to top" - > - - - ) : null} - - ) -} - function MobileNativeChatMessageImpl({ message, toolsExpanded = false, fontScale = 1, - messageIndex, - onScrollToMessage, onOpenFile, turnStatus, turnExpanded, @@ -114,12 +81,8 @@ function MobileNativeChatMessageImpl({ toolsExpanded?: boolean /** Multiplies all chat text sizes for pinch-to-zoom (1 = no change). */ fontScale?: number - /** This message's index in the list, paired with onScrollToMessage. */ - messageIndex?: number - /** Ask the list to align this message's top to the top of the viewport. */ - onScrollToMessage?: (index: number) => void onOpenFile?: (relativePath: string) => void - /** This turn's status row, rendered under a user message (desktop parity). */ + /** This settled turn's status row, rendered under its user message. */ turnStatus?: NativeChatTurnStatus | null /** Whether the turn caret has disclosed this turn's activity. */ turnExpanded?: boolean @@ -134,18 +97,6 @@ function MobileNativeChatMessageImpl({ }): React.JSX.Element { const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' - const isAgent = !isUser - // Briefly tint the bubble to confirm a copy landed. - const [copied, setCopied] = useState(false) - const copyTimer = useRef | null>(null) - useEffect( - () => () => { - if (copyTimer.current) { - clearTimeout(copyTimer.current) - } - }, - [] - ) // Separate the agent's words from its tool activity: prose renders first, the // tool calls fold into a collapsible run beneath. The user's own messages get // an inverted (filled accent) bubble so they stand apart from agent prose. @@ -165,42 +116,11 @@ function MobileNativeChatMessageImpl({ !toolsExpanded const showToolRun = tools.length > 0 && !settledToolsHidden - const handleCopy = (): void => { - const text = nativeChatMessageText(message.blocks) - if (!text) { - return - } - void Clipboard.setStringAsync(text) - setCopied(true) - if (copyTimer.current) { - clearTimeout(copyTimer.current) - } - copyTimer.current = setTimeout(() => setCopied(false), 700) - } - - // Copy + scroll-to-top, shown inline with the first tool call (or after the - // prose when there are no tools). - const controls = isAgent ? ( - onScrollToMessage(messageIndex) - : undefined - } - /> - ) : null - return ( <> {prose.map((block, index) => ( - ) : controls ? ( - {controls} ) : null} diff --git a/mobile/src/session/MobileNativeChatOverlay.tsx b/mobile/src/session/MobileNativeChatOverlay.tsx index 357a089466e..389beb8eaad 100644 --- a/mobile/src/session/MobileNativeChatOverlay.tsx +++ b/mobile/src/session/MobileNativeChatOverlay.tsx @@ -71,7 +71,11 @@ export function MobileNativeChatOverlay({ error={session.error} agent={controller.nativeChatAgent} agentWorking={controller.nativeChatAgentWorking} + canStop={controller.nativeChatCanStop} structuredActivityUi={controller.nativeChatStructured} + turnIndicator={controller.nativeChatTurnIndicator} + workingStartedAt={controller.nativeChatWorkingStartedAt} + settledTurns={controller.nativeChatSettledTurns} streaming={streaming} onStop={controller.handleNativeChatStop} ask={controller.nativeChatAsk} diff --git a/mobile/src/session/MobileNativeChatToolRun.tsx b/mobile/src/session/MobileNativeChatToolRun.tsx index ccc732dab88..4dad714957c 100644 --- a/mobile/src/session/MobileNativeChatToolRun.tsx +++ b/mobile/src/session/MobileNativeChatToolRun.tsx @@ -172,7 +172,6 @@ export function ToolRun({ defaultExpanded, expandChildren, activeCall, - trailing, onOpenFile }: { blocks: NativeChatBlock[] @@ -181,7 +180,6 @@ export function ToolRun({ expandChildren: boolean /** The still-running call, when the turn is live (desktop parity). */ activeCall: ReturnType - trailing?: React.ReactNode onOpenFile?: (relativePath: string) => void }): React.JSX.Element { const [open, setOpen] = useState(defaultExpanded) @@ -230,7 +228,6 @@ export function ToolRun({
)} - {trailing} {open ? ( diff --git a/mobile/src/session/MobileNativeChatTurnStatus.test.ts b/mobile/src/session/MobileNativeChatTurnStatus.test.ts index 78ac01e0d37..6b6a1d49e07 100644 --- a/mobile/src/session/MobileNativeChatTurnStatus.test.ts +++ b/mobile/src/session/MobileNativeChatTurnStatus.test.ts @@ -7,18 +7,8 @@ vi.mock('react-native', async () => { const Text = ({ children, ...props }: { children?: unknown }): unknown => React.createElement('Text', props, children) return { - Animated: { - Text, - Value: class { - constructor(private value: number) {} - setValue(next: number): void { - this.value = next - } - }, - loop: (animation: unknown) => animation, - sequence: () => ({ start: vi.fn(), stop: vi.fn() }), - timing: () => ({ start: vi.fn(), stop: vi.fn() }) - }, + ActivityIndicator: (props: Record) => + React.createElement('ActivityIndicator', props), Pressable: ({ children, ...props }: { children?: unknown }) => React.createElement('Pressable', props, children), Text, @@ -49,6 +39,7 @@ describe('MobileNativeChatTurnStatus', () => { startedAt: number | null thinking: boolean workedSeconds?: number | null + activityText?: string | null expanded?: boolean onToggleExpanded?: () => void }): ReactTestRenderer { @@ -61,12 +52,16 @@ describe('MobileNativeChatTurnStatus', () => { const labels = (node: ReactTestInstance): string[] => node.findAllByType('Text' as never).map((text) => String(text.children.join(''))) - it('reads "Thinking" before the turn produces output', () => { + const spinners = (node: ReactTestInstance): ReactTestInstance[] => + node.findAllByType('ActivityIndicator' as never) + + it('reads "Thinking" beside one spinner while the turn reasons', () => { const tree = render({ startedAt: Date.now(), thinking: true }) expect(labels(tree.root)).toEqual(['Thinking']) + expect(spinners(tree.root)).toHaveLength(1) }) - it('counts up once the turn is producing output', () => { + it('counts up on that same single row when the turn is not reasoning', () => { const startedAt = Date.now() const tree = render({ startedAt, thinking: false }) expect(labels(tree.root)).toEqual(['Working for 0s']) @@ -74,6 +69,19 @@ describe('MobileNativeChatTurnStatus', () => { vi.advanceTimersByTime(12_000) }) expect(labels(tree.root)).toEqual(['Working for 12s']) + expect(spinners(tree.root)).toHaveLength(1) + }) + + it('lets provider activity text beat both fallbacks and hold the clock', () => { + const tree = render({ + startedAt: Date.now(), + thinking: true, + activityText: 'Running pnpm test' + }) + expect(labels(tree.root)).toEqual(['Running pnpm test']) + expect(spinners(tree.root)).toHaveLength(1) + // No label consumes the duration, so nothing schedules a tick for it. + expect(vi.getTimerCount()).toBe(0) }) it('settles to a tappable "Worked for" row that toggles the turn', () => { @@ -98,9 +106,10 @@ describe('MobileNativeChatTurnStatus', () => { expect(labels(tree.root)).toEqual(['Worked for 5s']) }) - it('holds no interval once the turn has settled', () => { - render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + it('holds no interval, and no spinner, once the turn has settled', () => { + const tree = render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) expect(vi.getTimerCount()).toBe(0) + expect(spinners(tree.root)).toHaveLength(0) }) it('announces the live row to assistive tech', () => { diff --git a/mobile/src/session/MobileNativeChatTurnStatus.tsx b/mobile/src/session/MobileNativeChatTurnStatus.tsx index 4ce73cdcd38..acf922265f6 100644 --- a/mobile/src/session/MobileNativeChatTurnStatus.tsx +++ b/mobile/src/session/MobileNativeChatTurnStatus.tsx @@ -1,7 +1,8 @@ -import { useEffect, useRef, useState } from 'react' -import { Animated, Pressable, StyleSheet, Text, View } from 'react-native' +import { useEffect, useState } from 'react' +import { ActivityIndicator, Pressable, StyleSheet, Text, View } from 'react-native' import { ChevronRight } from 'lucide-react-native' import { + formatNativeChatActiveTurnLabel, formatNativeChatTurnStatusLabel, NATIVE_CHAT_TURN_STATUS_COPY, nativeChatElapsedSeconds @@ -25,48 +26,38 @@ function useElapsedSeconds(startedAt: number | null, counting: boolean): number return counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 } -/** The per-turn status row — "Thinking", then "Working for 12s" while the turn - * runs, settling to a tappable "Worked for 3m 4s" that discloses the turn's - * tool activity. Desktop parity: `NativeChatWorkingStatus`. */ +/** The per-turn status row. While the turn runs it is the one live indicator — a + * spinner beside what the provider says it is doing, else "Thinking", else + * "Working for 12s". It settles to a tappable "Worked for 3m 4s" that discloses + * the turn's tool activity. Desktop parity: `NativeChatTurnActivityLine` for the + * live row, `NativeChatWorkingStatus` for the settled one. */ export function MobileNativeChatTurnStatus({ startedAt, thinking, workedSeconds, + activityText, expanded = false, onToggleExpanded }: { startedAt: number | null thinking: boolean workedSeconds?: number | null + /** Provider activity copy for a live turn; outranks the other two labels. */ + activityText?: string | null expanded?: boolean onToggleExpanded?: () => void }): React.JSX.Element { - const counting = !thinking && workedSeconds == null + const settled = workedSeconds != null + const counting = !settled && !thinking && !activityText?.trim() const elapsedSeconds = useElapsedSeconds(startedAt, counting) - const label = formatNativeChatTurnStatusLabel({ thinking, workedSeconds, elapsedSeconds }) + const label = settled + ? formatNativeChatTurnStatusLabel({ thinking, workedSeconds, elapsedSeconds }) + : formatNativeChatActiveTurnLabel({ activityText, thinking, elapsedSeconds }) - const pulse = useRef(new Animated.Value(1)).current - useEffect(() => { - if (!thinking) { - pulse.setValue(1) - return - } - const animation = Animated.loop( - Animated.sequence([ - Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), - Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) - ]) - ) - animation.start() - return () => animation.stop() - }, [pulse, thinking]) - - const rowStyle = [styles.row, thinking ? null : styles.rowSettled] - - if (workedSeconds != null && onToggleExpanded) { + if (settled && onToggleExpanded) { return ( [...rowStyle, pressed && styles.pressed]} + style={({ pressed }) => [styles.row, styles.rowSettled, pressed && styles.pressed]} onPress={onToggleExpanded} hitSlop={6} accessibilityRole="button" @@ -83,11 +74,14 @@ export function MobileNativeChatTurnStatus({ return ( - {label} + {settled ? null : } + + {label} + ) } @@ -109,7 +103,8 @@ const styles = StyleSheet.create({ }, label: { color: colors.textMuted, - fontSize: typography.bodySize + fontSize: typography.bodySize, + flexShrink: 1 }, caretOpen: { transform: [{ rotate: '90deg' }] diff --git a/mobile/src/session/MobileNativeChatView.test.ts b/mobile/src/session/MobileNativeChatView.test.ts index d101c3f0ef6..c573d89cc89 100644 --- a/mobile/src/session/MobileNativeChatView.test.ts +++ b/mobile/src/session/MobileNativeChatView.test.ts @@ -73,7 +73,9 @@ type Overrides = { onSend?: (text: string) => Promise pending?: Parameters[0]['pending'] structuredActivityUi?: boolean + turnIndicator?: Parameters[0]['turnIndicator'] agentWorking?: boolean + canStop?: boolean sendSurfaceId?: string } @@ -118,6 +120,18 @@ describe('MobileNativeChatView', () => { } /** Ids of the rows the list is currently rendering. */ + it('keeps Stop hidden during a structured dispatch until a provider turn can be cancelled', async () => { + const props = { structuredActivityUi: true, agentWorking: true, canStop: false } + await render(props) + const stops = () => + renderer!.root.findAll((node) => node.props.accessibilityLabel === 'Stop the agent') + expect(stops()).toHaveLength(0) + await update({ ...props, canStop: true }) + expect(stops()).toHaveLength(1) + await update({ agentWorking: true }) + expect(stops()).toHaveLength(1) + }) + function listIds(): string[] { const list = renderer!.root.find((node) => node.type === 'FlatList') return (list.props.data as { id: string }[]).map((row) => row.id) @@ -260,20 +274,74 @@ describe('MobileNativeChatView', () => { return (renderedRow(id) as { props: Record }).props } + function footerProps(): Record | null { + const list = renderer!.root.find((node) => node.type === 'FlatList') + const footer = list.props.ListFooterComponent as + | { props: Record } + | null + | undefined + return footer?.props ?? null + } + function workingIndicators(): ReactTestInstance[] { return renderer!.root.findAll((node) => node.type === 'WorkingIndicator') } - it('gives the live user turn a status row and drops the three-dot indicator', async () => { - const folded = [userTurn('u1', 'go')] + it('puts the live status at the turn tail and drops the three-dot indicator', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'still working')] await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) const props = rowProps('u1') expect(props.structuredActivityUi).toBe(true) - expect(props.turnStatus).toMatchObject({ thinking: true, workedSeconds: null }) + expect(props.turnStatus).toBeNull() + // Nothing reports reasoning, so the one live footer counts instead of guessing. + expect(footerProps()).toMatchObject({ thinking: false, workedSeconds: null }) + expect(listIds().at(-1)).toBe('a1') expect(props.activeTurnIsWorking).toBe(true) expect(workingIndicators()).toHaveLength(0) }) + it('reports the live turn as thinking only when its journal says it is reasoning', async () => { + const folded = [userTurn('u1', 'go')] + await render({ + messages: folded, + folded, + structuredActivityUi: true, + agentWorking: true, + turnIndicator: { thinking: true, activityText: null } + }) + expect(rowProps('u1').turnStatus).toBeNull() + expect(footerProps()).toMatchObject({ thinking: true, workedSeconds: null }) + }) + + it('hands the live row the provider activity copy that outranks its fallbacks', async () => { + const folded = [userTurn('u1', 'go')] + await render({ + messages: folded, + folded, + structuredActivityUi: true, + agentWorking: true, + turnIndicator: { thinking: true, activityText: 'Running pnpm test' } + }) + expect(footerProps()).toMatchObject({ + thinking: true, + activityText: 'Running pnpm test' + }) + }) + + it('keeps the activity copy on the live footer instead of a historical row', async () => { + const folded = [userTurn('u1', 'go'), userTurn('u2', 'again')] + await render({ + messages: folded, + folded, + structuredActivityUi: true, + agentWorking: true, + turnIndicator: { thinking: false, activityText: 'Running pnpm test' } + }) + expect(rowProps('u1')).not.toHaveProperty('turnActivityText') + expect(rowProps('u2')).not.toHaveProperty('turnActivityText') + expect(footerProps()).toMatchObject({ activityText: 'Running pnpm test' }) + }) + it('keeps the bridge lane on the three-dot indicator with no turn status', async () => { const folded = [userTurn('u1', 'go')] await render({ messages: folded, folded, agentWorking: true }) @@ -281,13 +349,15 @@ describe('MobileNativeChatView', () => { expect(props.structuredActivityUi).toBe(false) expect(props.turnStatus).toBeNull() expect(props.activeTurnIsWorking).toBe(false) + expect(footerProps()).toBeNull() expect(workingIndicators()).toHaveLength(1) }) it('settles the finished turn to a tappable duration', async () => { const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) - expect(rowProps('u1').turnStatus).toMatchObject({ thinking: false, workedSeconds: null }) + expect(rowProps('u1').turnStatus).toBeNull() + expect(footerProps()).toMatchObject({ thinking: false, workedSeconds: null }) await update({ messages: folded, folded, structuredActivityUi: true, agentWorking: false }) const settled = rowProps('u1') expect(settled.turnStatus).toMatchObject({ thinking: false }) @@ -296,6 +366,7 @@ describe('MobileNativeChatView', () => { ) expect(settled.onToggleTurn).toBeTypeOf('function') expect(settled.activeTurnIsWorking).toBe(false) + expect(footerProps()).toBeNull() }) it('hangs no status row on an assistant row', async () => { @@ -304,6 +375,7 @@ describe('MobileNativeChatView', () => { expect(rowProps('a1').turnStatus).toBeNull() // The assistant row still belongs to the live turn, so its tool row stays visible. expect(rowProps('a1').activeTurnIsWorking).toBe(true) + expect(footerProps()).toMatchObject({ workedSeconds: null }) }) it('does not carry a running turn clock across chat surfaces', async () => { @@ -318,7 +390,7 @@ describe('MobileNativeChatView', () => { agentWorking: true, sendSurfaceId: 'host\0worktree\0tab-a' }) - expect(rowProps('u1').turnStatus).toMatchObject({ startedAt: 1_000 }) + expect(footerProps()).toMatchObject({ startedAt: 1_000 }) vi.setSystemTime(12_000) const secondTab = [userTurn('u2', 'second')] @@ -330,7 +402,7 @@ describe('MobileNativeChatView', () => { sendSurfaceId: 'host\0worktree\0tab-b' }) - expect(rowProps('u2').turnStatus).toMatchObject({ startedAt: 12_000 }) + expect(footerProps()).toMatchObject({ startedAt: 12_000 }) } finally { vi.useRealTimers() } diff --git a/mobile/src/session/MobileNativeChatView.tsx b/mobile/src/session/MobileNativeChatView.tsx index 70a67787de0..67fc93506a9 100644 --- a/mobile/src/session/MobileNativeChatView.tsx +++ b/mobile/src/session/MobileNativeChatView.tsx @@ -13,6 +13,10 @@ import { GestureDetector, GestureHandlerRootView } from 'react-native-gesture-ha import { ArrowDown, ChevronsDownUp, ChevronsUpDown, Square } from 'lucide-react-native' import type { AskAnswerSelection, AskPrompt } from '../../../src/shared/native-chat-ask' import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import type { + NativeChatLiveTurnIndicator, + NativeChatSettledTurns +} from '../../../src/shared/native-chat-turn-status' import { colors } from '../theme/mobile-theme' import { styles } from './mobile-native-chat-view-styles' import { @@ -22,6 +26,7 @@ import { } from './mobile-native-chat-render-data' import { useMobileNativeChatPinchGesture } from './use-mobile-native-chat-pinch-gesture' import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' +import { useSettledMobileNativeChatInputLock } from './use-mobile-native-chat-input-lease' import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' import { MobileAgentWorkingIndicator } from './MobileAgentWorkingIndicator' import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment' @@ -33,8 +38,6 @@ import type { MobileNativeChatSessionOptionPickersProps } from './MobileNativeCh import { MobileNativeChatMessage } from './MobileNativeChatMessage' import type { MobileNativeChatStatus } from './use-mobile-native-chat-session' -const INPUT_LOCK_SETTLE_MS = 600 - /** Why the composer input is locked: the transport is disconnected, or the * terminal subscription has not acknowledged its input lease yet. */ export type MobileNativeChatInputLockReason = 'disconnected' | 'waiting' @@ -49,10 +52,17 @@ type Props = { /** Resolved agent for this chat; names the empty-state copy (desktop parity). */ agent?: string | null agentWorking?: boolean + canStop?: boolean /** Structured lane: per-turn "Working for N" status plus live tool progress, * replacing the bridge lane's static three-dot working row (desktop parity). */ structuredActivityUi?: boolean + /** What labels the live turn's one indicator row (structured lane only). */ + turnIndicator?: NativeChatLiveTurnIndicator | null + /** Structured lane: host-recorded turn timing feeding the per-turn status rows. */ + workingStartedAt?: number | null + settledTurns?: NativeChatSettledTurns | null /** Interrupt the agent mid-turn (shown as a Stop button on the working bar). */ + /** Interrupt a provider turn. */ onStop?: () => void /** Live partial assistant text to show as an in-progress bubble, already gated * by the overlay against the transcript catching up. */ @@ -129,7 +139,11 @@ export function MobileNativeChatView({ error, agent, agentWorking, + canStop = agentWorking, structuredActivityUi = false, + turnIndicator = null, + workingStartedAt, + settledTurns, onStop, streaming, hasMore, @@ -251,17 +265,17 @@ export function MobileNativeChatView({ [hasMore, loadingEarlier, onLoadEarlier] ) - // Align a single message's top to the top of the viewport. - const onScrollToMessage = useCallback((index: number) => { - listRef.current?.scrollToIndex({ index, viewPosition: 0, animated: true }) - }, []) - - // Per-turn "Thinking / Working for N / Worked for N" rows. The structured lane - // owns them; the bridge lane keeps its three-dot indicator. + // Per-turn status rows: one live indicator while the turn runs, then a settled + // "Worked for N" row. The structured lane owns them; the bridge lane keeps its + // three-dot indicator. const turns = useMobileNativeChatTurnDisclosure({ messages: data, enabled: structuredActivityUi, isWorking: agentWorking === true, + workingStartedAt, + settledTurns, + thinking: turnIndicator?.thinking === true, + activityText: turnIndicator?.activityText ?? null, scopeKey: sendSurfaceId }) @@ -271,32 +285,19 @@ export function MobileNativeChatView({ message={item} toolsExpanded={toolsExpanded} fontScale={fontScale} - messageIndex={index} - onScrollToMessage={onScrollToMessage} onOpenFile={onOpenFile} structuredActivityUi={structuredActivityUi} onToggleTurn={turns.onToggleTurn} {...turns.resolveRow(index, item)} /> ), - [toolsExpanded, fontScale, onScrollToMessage, onOpenFile, structuredActivityUi, turns] + [toolsExpanded, fontScale, onOpenFile, structuredActivityUi, turns] ) const emptyState = mobileNativeChatEmptyState(status, agent ?? null, error) const showLoading = status === 'loading' && messages.length === 0 - // A dead PTY emits subscribed→end; settle both edges so its false lease cannot flash the composer enabled. - const rawLockReason = inputLockReason ?? null - const rawLockHeld = rawLockReason !== null - const [lockHeld, setLockHeld] = useState(false) - useEffect(() => { - if (rawLockHeld === lockHeld) { - return - } - const timer = setTimeout(() => setLockHeld(rawLockHeld), INPUT_LOCK_SETTLE_MS) - return () => clearTimeout(timer) - }, [lockHeld, rawLockHeld]) - const lockReason = lockHeld ? (rawLockReason ?? 'waiting') : null + const lockReason = useSettledMobileNativeChatInputLock(inputLockReason) return ( @@ -323,21 +324,6 @@ export function MobileNativeChatView({ listRef.current?.scrollToEnd({ animated: false }) } }} - // scrollToIndex can fail before an off-screen row is measured — - // fall back to an estimated offset, then retry once it's laid out. - onScrollToIndexFailed={(info) => { - listRef.current?.scrollToOffset({ - offset: info.averageItemLength * info.index, - animated: true - }) - setTimeout(() => { - listRef.current?.scrollToIndex({ - index: info.index, - viewPosition: 0, - animated: true - }) - }, 120) - }} ListHeaderComponent={ hasMore ? ( ) : null } @@ -372,8 +359,7 @@ export function MobileNativeChatView({ } /> - {/* Jump-to-latest control. The scroll-to-top affordance now lives - per-message (the up-arrow in each agent message's controls). */} + {/* Jump-to-latest control. */} {!atBottom ? ( - {/* Chrome row above the composer: the working indicator and the global - tool-calls expand/collapse toggle on the left, Stop in the far corner. */} {agentWorking && !structuredActivityUi ? : null} @@ -414,7 +398,7 @@ export function MobileNativeChatView({ {toolsExpanded ? 'Collapse' : 'Tools'} - {agentWorking ? ( + {canStop ? ( [styles.stopButton, pressed && styles.pressed]} onPress={onStop} diff --git a/mobile/src/session/mobile-native-chat-controller-contract.ts b/mobile/src/session/mobile-native-chat-controller-contract.ts index 53187e0d6db..36c07215e8e 100644 --- a/mobile/src/session/mobile-native-chat-controller-contract.ts +++ b/mobile/src/session/mobile-native-chat-controller-contract.ts @@ -6,6 +6,10 @@ import type { } from '../../../src/shared/native-chat-ask' import type { detectAgentPermission } from './mobile-native-chat-permission' import type { parseAgentQuestion } from './mobile-native-chat-question' +import type { + NativeChatLiveTurnIndicator, + NativeChatSettledTurns +} from '../../../src/shared/native-chat-turn-status' import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' import type { MobileNativeChatPendingMessage } from './use-mobile-native-chat-drafts' import type { useMobileNativeChatSession } from './use-mobile-native-chat-session' @@ -28,6 +32,12 @@ export type MobileNativeChatController = { /** Structured lane: drives the per-turn status row and live tool progress. */ nativeChatStructured: boolean nativeChatAgentWorking: boolean + /** What labels the live turn's one indicator row; null off the structured lane. */ + nativeChatTurnIndicator: NativeChatLiveTurnIndicator | null + /** Structured lane: host-recorded turn timing for the per-turn status rows. */ + nativeChatWorkingStartedAt: number | null + nativeChatSettledTurns: NativeChatSettledTurns | null + nativeChatCanStop: boolean nativeChatStreamingText?: string /** Agent mid-turn, regardless of whether chat is the visible view. */ nativeChatStreamLive: boolean diff --git a/mobile/src/session/mobile-native-chat-message-styles.ts b/mobile/src/session/mobile-native-chat-message-styles.ts index 7ae1128445a..51ff9ab1f36 100644 --- a/mobile/src/session/mobile-native-chat-message-styles.ts +++ b/mobile/src/session/mobile-native-chat-message-styles.ts @@ -29,23 +29,6 @@ export const styles = StyleSheet.create({ lineHeight: TEXT_SIZE + 6, fontWeight: '500' }, - controls: { - flexDirection: 'row', - justifyContent: 'flex-end', - gap: spacing.xs, - marginBottom: 2, - opacity: 0.7 - }, - controlButton: { - padding: 3 - }, - controlPressed: { - opacity: 0.5 - }, - copied: { - backgroundColor: colors.diffAddedBg, - borderRadius: radii.card - }, reasoning: { opacity: 0.7 }, @@ -64,10 +47,6 @@ export const styles = StyleSheet.create({ gap: spacing.sm, paddingVertical: 3 }, - controlsRow: { - flexDirection: 'row', - justifyContent: 'flex-end' - }, toolRunCount: { color: colors.statusGreen, fontFamily: typography.monoFamily, diff --git a/mobile/src/session/mobile-native-chat-message-text.test.ts b/mobile/src/session/mobile-native-chat-message-text.test.ts index 225904cc5d8..165313c2e43 100644 --- a/mobile/src/session/mobile-native-chat-message-text.test.ts +++ b/mobile/src/session/mobile-native-chat-message-text.test.ts @@ -1,31 +1,5 @@ import { describe, expect, it } from 'vitest' -import type { NativeChatBlock } from '../../../src/shared/native-chat-types' -import { - clampFontScale, - FONT_SCALE_MAX, - FONT_SCALE_MIN, - nativeChatMessageText -} from './mobile-native-chat-message-text' - -describe('nativeChatMessageText', () => { - it('joins text blocks and skips non-text blocks', () => { - const blocks: NativeChatBlock[] = [ - { type: 'text', text: 'Hello' }, - { type: 'tool-call', name: 'Read', input: {} }, - { type: 'text', text: 'World' } - ] - expect(nativeChatMessageText(blocks)).toBe('Hello\n\nWorld') - }) - - it('returns an empty string when there is no prose', () => { - const blocks: NativeChatBlock[] = [{ type: 'tool-call', name: 'Read', input: {} }] - expect(nativeChatMessageText(blocks)).toBe('') - }) - - it('trims surrounding whitespace', () => { - expect(nativeChatMessageText([{ type: 'text', text: ' hi ' }])).toBe('hi') - }) -}) +import { clampFontScale, FONT_SCALE_MAX, FONT_SCALE_MIN } from './mobile-native-chat-message-text' describe('clampFontScale', () => { it('clamps below the minimum', () => { diff --git a/mobile/src/session/mobile-native-chat-message-text.ts b/mobile/src/session/mobile-native-chat-message-text.ts index a94104835a9..84d2f8f1cc9 100644 --- a/mobile/src/session/mobile-native-chat-message-text.ts +++ b/mobile/src/session/mobile-native-chat-message-text.ts @@ -1,15 +1,3 @@ -import { isTextBlock, type NativeChatBlock } from '../../../src/shared/native-chat-types' - -/** Concatenate a message's text blocks into a single copyable string. Tool - * calls/results and image refs are skipped — Copy is for the agent's prose. */ -export function nativeChatMessageText(blocks: readonly NativeChatBlock[]): string { - return blocks - .filter(isTextBlock) - .map((b) => b.text) - .join('\n\n') - .trim() -} - /** Pinch-to-zoom font bounds. Default 1 means no visible change until pinched. */ export const FONT_SCALE_MIN = 0.8 export const FONT_SCALE_MAX = 1.8 diff --git a/mobile/src/session/mobile-structured-agent-session-launch.test.ts b/mobile/src/session/mobile-structured-agent-session-launch.test.ts index 8a020d2eea9..2deb3042ca2 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.test.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.test.ts @@ -228,6 +228,34 @@ describe('mobile structured agent-session launch', () => { }) }) + describe.each(['top-level', 'nested'])('%s refusal messages', (location) => { + function refusalClient(message: unknown) { + const refusal = { code: 'method_not_found', ...(message === undefined ? {} : { message }) } + return clientReturning( + { ok: true, result: { supported: true } }, + location === 'top-level' + ? { ok: false, error: refusal } + : { ok: true, result: { ok: false, refusal } } + ) + } + + it.each( + [undefined, null, 42, false, { text: 'unavailable' }, ['unavailable']].map((message) => ({ + message + })) + )('keeps a malformed message $message unknown', async ({ message }) => { + await expect( + createMobileStructuredAgentSession(refusalClient(message), 'workspace-1', 'codex') + ).resolves.toMatchObject({ kind: 'unknown' }) + }) + + it('preserves the fallback for an empty string message', async () => { + await expect( + createMobileStructuredAgentSession(refusalClient(''), 'workspace-1', 'codex') + ).resolves.toEqual({ kind: 'failed', message: 'Could not open Codex chat.' }) + }) + }) + it.each(['structured_agent_session_unsupported', 'method_not_found'])( 'treats a top-level %s as a definitive refusal', async (code) => { diff --git a/mobile/src/session/mobile-structured-agent-session-launch.ts b/mobile/src/session/mobile-structured-agent-session-launch.ts index 9e26eaab91e..bd15e595736 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.ts @@ -137,18 +137,22 @@ export async function createMobileStructuredAgentSession( } } + // Why: this path distrusts the declared RpcResponse type — a malformed reply must read as + // unconfirmed, not as a refusal we can classify. if (!response || typeof response !== 'object' || typeof response.ok !== 'boolean') { return unknownCreateResult(agent, new Error(unconfirmedMessage(agent))) } if (!response.ok) { + const error = response.error as { code?: unknown; message?: unknown } | null | undefined if ( - !response.error || - typeof response.error !== 'object' || - typeof response.error.code !== 'string' + !error || + typeof error !== 'object' || + typeof error.code !== 'string' || + typeof error.message !== 'string' ) { return unknownCreateResult(agent, new Error(unconfirmedMessage(agent))) } - return classifyCreateRefusal(agent, response.error.code, response.error.message) + return classifyCreateRefusal(agent, error.code, error.message) } const result = response.result as AgentSessionMutationResult if (!result || typeof result !== 'object' || typeof result.ok !== 'boolean') { @@ -158,7 +162,8 @@ export async function createMobileStructuredAgentSession( if ( !result.refusal || typeof result.refusal !== 'object' || - typeof result.refusal.code !== 'string' + typeof result.refusal.code !== 'string' || + typeof result.refusal.message !== 'string' ) { return unknownCreateResult(agent, new Error(unconfirmedMessage(agent))) } diff --git a/mobile/src/session/use-mobile-bridge-chat-prompt-writes.ts b/mobile/src/session/use-mobile-bridge-chat-prompt-writes.ts new file mode 100644 index 00000000000..2dddad48e16 --- /dev/null +++ b/mobile/src/session/use-mobile-bridge-chat-prompt-writes.ts @@ -0,0 +1,65 @@ +import type { MutableRefObject } from 'react' +import type { RpcClient } from '../transport/rpc-client' +import { useMobileNativeChatPermissionSend } from './mobile-native-chat-permission-send' +import { useMobileNativeChatAnswerSend } from './use-mobile-native-chat-answer-send' +import { useMobileNativeChatCancelAsk } from './use-mobile-native-chat-cancel-ask' +import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' +import type { MobileNativeChatAnswerSend } from './use-mobile-native-chat-answer-send' + +/** The bridge lane's four prompt/interrupt write seams. They share one enable + * gate and chain through the answer seam's `cancelPending`, so a caller cannot + * wire one of them to a different lane or forget to drop in-flight answer + * writes before an Escape. The structured lane answers over RPC instead. */ +export function useMobileBridgeChatPromptWrites(args: { + client: RpcClient | null + enabled: boolean + handleRef: MutableRefObject + deviceTokenRef: MutableRefObject + agentRef: MutableRefObject + /** Changes on chat session swap; cancels pending writes when it does. */ + sessionId: string | null + streamIdentity: string + onSendError: (message: string) => void +}): { + answerAsk: MobileNativeChatAnswerSend['answerAsk'] + cancelAsk: () => Promise + respondPermission: (send: string) => Promise + stop: () => void +} { + const { client, enabled, handleRef, deviceTokenRef, streamIdentity, onSendError } = args + const { answerAsk, cancelPending } = useMobileNativeChatAnswerSend({ + client, + enabled, + handleRef, + deviceTokenRef, + agentRef: args.agentRef, + sessionId: args.sessionId, + streamIdentity, + onSendError + }) + const cancelAsk = useMobileNativeChatCancelAsk({ + client, + enabled, + handleRef, + deviceTokenRef, + cancelPending, + onSendError + }) + const respondPermission = useMobileNativeChatPermissionSend({ + client, + enabled, + handleRef, + deviceTokenRef, + onSendError + }) + const stop = useMobileNativeChatStop({ + client, + enabled, + handleRef, + deviceTokenRef, + streamIdentity, + cancelPending, + onSendError + }) + return { answerAsk, cancelAsk, respondPermission, stop } +} diff --git a/mobile/src/session/use-mobile-native-chat-controller.test.ts b/mobile/src/session/use-mobile-native-chat-controller.test.ts index 83f8075d914..c90033e2404 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.test.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.test.ts @@ -55,6 +55,7 @@ const structuredQuestion = { allowOther: true, optionTokens: ['choice-a', 'choice-b'] } +const structuredActivity = { isWorking: false, turnId: null as string | null } const structuredSessionState = { messages: [] as unknown[], status: 'ready', @@ -86,8 +87,7 @@ vi.mock('./use-mobile-native-chat-session', () => ({ vi.mock('./use-mobile-structured-agent-session', () => ({ useMobileStructuredAgentSession: () => ({ session: structuredSessionState, - isWorking: false, - turnId: null, + ...structuredActivity, sendWithOutcome: structuredSendWithOutcome, cancel: structuredCancel, permission: structuredPermission, @@ -341,6 +341,37 @@ describe('useMobileNativeChatController handleNativeChatSend', () => { expect(clientStub.sendRequest).not.toHaveBeenCalled() }) + it('separates structured working status from provider cancellation availability', async () => { + const props = { + tab: { + type: 'agent-session', + id: 'agent-tab-1', + title: 'Chat', + sessionId: 'session-structured', + agent: 'codex', + isActive: true + }, + activeHandle: null, + inputLeaseReady: false + } + structuredActivity.isWorking = true + try { + await act(async () => { + renderer?.update(createElement(Harness, props)) + }) + expect(controller?.nativeChatAgentWorking).toBe(true) + expect(controller?.nativeChatCanStop).toBe(false) + structuredActivity.turnId = 'provider-turn' + await act(async () => { + renderer?.update(createElement(Harness, props)) + }) + expect(controller?.nativeChatCanStop).toBe(true) + } finally { + structuredActivity.isWorking = false + structuredActivity.turnId = null + } + }) + it('exposes structured prompt cards and session options on structured tabs', async () => { await act(async () => { renderer?.update( diff --git a/mobile/src/session/use-mobile-native-chat-controller.ts b/mobile/src/session/use-mobile-native-chat-controller.ts index 729cec302c1..4087694b567 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.ts @@ -2,10 +2,7 @@ import { useLayoutEffect, useRef, type MutableRefObject } from 'react' import type { RpcClient } from '../transport/rpc-client' import type { ConnectionState } from '../transport/types' import type { MobileNativeChatTab } from './mobile-native-chat-eligibility' -import { useMobileNativeChatPermissionSend } from './mobile-native-chat-permission-send' -import { useMobileNativeChatAnswerSend } from './use-mobile-native-chat-answer-send' import { useMobileNativeChatAskDismiss } from './use-mobile-native-chat-ask-dismiss' -import { useMobileNativeChatCancelAsk } from './use-mobile-native-chat-cancel-ask' import { useMobileNativeChatDrafts } from './use-mobile-native-chat-drafts' import { useMobileNativeChatFileSearch } from './use-mobile-native-chat-file-search' import { useMobileNativeChatMessageSend } from './use-mobile-native-chat-message-send' @@ -14,10 +11,10 @@ import { useMobileNativeChatSessionOptionController } from './use-mobile-native- import { useMobileNativeChatSessionLane } from './use-mobile-native-chat-session-lane' import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge' import { useMobileNativeChatPrompts } from './use-mobile-native-chat-prompts' -import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' import { useNativeChatAcceptedAction } from './use-native-chat-action-outcomes' import { useThrottledLatestValue } from './use-throttled-latest-value' import type { MobileNativeChatController } from './mobile-native-chat-controller-contract' +import { useMobileBridgeChatPromptWrites } from './use-mobile-bridge-chat-prompt-writes' import { useMobileNativeChatActiveResolution } from './use-mobile-native-chat-active-resolution' export type { MobileNativeChatController } from './mobile-native-chat-controller-contract' @@ -123,14 +120,13 @@ export function useMobileNativeChatController(args: { transcriptSettled: nativeChatSession.status === 'ready' }) - const nativeChatAgentWorking = activeChatStructured - ? structuredNativeChat.isWorking - : activeChatResolution != null && activeTabAgentWorking // Deliberately not gated on the chat view being visible: the streaming gate // has to tell "hidden mid-turn" from "the turn ended". const nativeChatStreamLive = activeChatStructured ? structuredNativeChat.isWorking : activeTabAgentWorking + const nativeChatAgentWorking = + nativeChatStreamLive && (activeChatStructured || activeChatResolution != null) // Throttle the streaming bubble: OpenCode emits a status frame per streamed // part, and each one re-renders and re-parses the whole accumulated markdown. const nativeChatStreamingText = useThrottledLatestValue( @@ -141,7 +137,7 @@ export function useMobileNativeChatController(args: { ) const { permission: legacyNativeChatPermission, - question: legacyNativeChatQuestion, + question: legacyQuestion, detectedAsk: nativeChatDetectedAsk, ask: nativeChatAskPrompt } = useMobileNativeChatPrompts({ @@ -172,42 +168,19 @@ export function useMobileNativeChatController(args: { ? client != null && activeChatSessionId != null && connState === 'connected' : nativeChatInputLeaseReady && connState === 'connected' - const { answerAsk: handleNativeChatAnswerAsk, cancelPending: cancelNativeChatAnswer } = - useMobileNativeChatAnswerSend({ - client, - enabled: inputSendable && !activeChatStructured, - handleRef: activeHandleRef, - deviceTokenRef, - agentRef: activeChatAgentRef, - sessionId: activeChatSessionId, - streamIdentity, - onSendError - }) - - const handleNativeChatCancelAsk = useMobileNativeChatCancelAsk({ - client, - enabled: inputSendable && !activeChatStructured, - handleRef: activeHandleRef, - deviceTokenRef, - cancelPending: cancelNativeChatAnswer, - onSendError - }) - - const legacyHandleNativeChatRespondPermission = useMobileNativeChatPermissionSend({ - client, - enabled: inputSendable && !activeChatStructured, - handleRef: activeHandleRef, - deviceTokenRef, - onSendError - }) - - const handleNativeChatStop = useMobileNativeChatStop({ + const { + answerAsk: handleNativeChatAnswerAsk, + cancelAsk: handleNativeChatCancelAsk, + respondPermission: legacyHandleNativeChatRespondPermission, + stop: handleNativeChatStop + } = useMobileBridgeChatPromptWrites({ client, enabled: inputSendable && !activeChatStructured, handleRef: activeHandleRef, deviceTokenRef, + agentRef: activeChatAgentRef, + sessionId: activeChatSessionId, streamIdentity, - cancelPending: cancelNativeChatAnswer, onSendError }) @@ -243,6 +216,7 @@ export function useMobileNativeChatController(args: { }) const structuredNativeChatSend = useMobileStructuredNativeChatSendBridge({ + agent: activeChatResolution?.agent === 'claude' ? 'claude' : 'codex', sendStructured: structuredNativeChat.sendWithOutcome, captureSendOrigin, clearDraftForSend, @@ -299,15 +273,19 @@ export function useMobileNativeChatController(args: { /** Structured lane: drives the per-turn status row and live tool progress. */ nativeChatStructured: activeChatStructured, nativeChatAgentWorking, + nativeChatTurnIndicator: activeChatStructured ? structuredNativeChat.turnIndicator : null, + nativeChatWorkingStartedAt: activeChatStructured ? structuredNativeChat.workingStartedAt : null, + nativeChatSettledTurns: activeChatStructured ? structuredNativeChat.settledTurns : null, + nativeChatCanStop: activeChatStructured + ? structuredNativeChat.turnId !== null + : nativeChatAgentWorking, nativeChatStreamingText, nativeChatStreamLive, nativeChatStreamScopeKey: streamScopeKey, nativeChatPermission: activeChatStructured ? structuredNativeChat.permission : legacyNativeChatPermission, - nativeChatQuestion: activeChatStructured - ? structuredNativeChat.question - : legacyNativeChatQuestion, + nativeChatQuestion: activeChatStructured ? structuredNativeChat.question : legacyQuestion, nativeChatAsk: !activeChatStructured && showNativeChatAsk ? nativeChatAskPrompt : null, nativeChatAskKey, dismissNativeChatAsk, diff --git a/mobile/src/session/use-mobile-native-chat-input-lease.test.ts b/mobile/src/session/use-mobile-native-chat-input-lease.test.ts index 7491f74500f..6e4d12a2bcc 100644 --- a/mobile/src/session/use-mobile-native-chat-input-lease.test.ts +++ b/mobile/src/session/use-mobile-native-chat-input-lease.test.ts @@ -1,7 +1,10 @@ import { createElement } from 'react' import { act, create, type ReactTestRenderer } from 'react-test-renderer' -import { afterEach, describe, expect, it } from 'vitest' -import { useMobileNativeChatInputLease } from './use-mobile-native-chat-input-lease' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + useMobileNativeChatInputLease, + useSettledMobileNativeChatInputLock +} from './use-mobile-native-chat-input-lease' type Lease = ReturnType @@ -61,3 +64,40 @@ describe('useMobileNativeChatInputLease', () => { expect(lease?.clear()).toBe(true) }) }) + +describe('useSettledMobileNativeChatInputLock', () => { + let renderer: ReactTestRenderer | null = null + let settled: ReturnType | undefined + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + vi.useRealTimers() + }) + + function Harness({ reason }: { reason: 'waiting' | 'disconnected' | null }): null { + settled = useSettledMobileNativeChatInputLock(reason) + return null + } + + it('holds each edge until the lease has stopped flapping', () => { + vi.useFakeTimers() + act(() => { + renderer = create(createElement(Harness, { reason: 'waiting' })) + }) + expect(settled).toBeNull() + act(() => vi.advanceTimersByTime(600)) + expect(settled).toBe('waiting') + + // A brief unlock that reverts inside the settle window never reaches the composer. + act(() => renderer?.update(createElement(Harness, { reason: null }))) + act(() => vi.advanceTimersByTime(300)) + act(() => renderer?.update(createElement(Harness, { reason: 'disconnected' }))) + act(() => vi.advanceTimersByTime(600)) + expect(settled).toBe('disconnected') + + act(() => renderer?.update(createElement(Harness, { reason: null }))) + act(() => vi.advanceTimersByTime(600)) + expect(settled).toBeNull() + }) +}) diff --git a/mobile/src/session/use-mobile-native-chat-input-lease.ts b/mobile/src/session/use-mobile-native-chat-input-lease.ts index fddf9274573..f1489faa9cd 100644 --- a/mobile/src/session/use-mobile-native-chat-input-lease.ts +++ b/mobile/src/session/use-mobile-native-chat-input-lease.ts @@ -72,3 +72,23 @@ export function useMobileNativeChatInputLease(args: { clear } } + +const INPUT_LOCK_SETTLE_MS = 600 + +/** A dead PTY emits subscribed→end; settle both edges so its false lease cannot + * flash the composer enabled. */ +export function useSettledMobileNativeChatInputLock( + reason: MobileNativeChatInputLockReason | null | undefined +): MobileNativeChatInputLockReason | null { + const rawLockReason = reason ?? null + const rawLockHeld = rawLockReason !== null + const [lockHeld, setLockHeld] = useState(false) + useEffect(() => { + if (rawLockHeld === lockHeld) { + return + } + const timer = setTimeout(() => setLockHeld(rawLockHeld), INPUT_LOCK_SETTLE_MS) + return () => clearTimeout(timer) + }, [lockHeld, rawLockHeld]) + return lockHeld ? (rawLockReason ?? 'waiting') : null +} diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx index 8467684ce16..33e99bfb603 100644 --- a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx @@ -2,6 +2,7 @@ import { createElement } from 'react' import { act, create, type ReactTestRenderer } from 'react-test-renderer' import { afterEach, describe, expect, it, vi } from 'vitest' import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import type { NativeChatSettledTurns } from '../../../src/shared/native-chat-turn-status' import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' function userMessage(id: string): NativeChatMessage { @@ -18,17 +19,20 @@ function Harness({ messages, enabled, isWorking = true, + settledTurns, scopeKey = 'host\0worktree\0tab-a' }: { messages: readonly NativeChatMessage[] enabled: boolean isWorking?: boolean + settledTurns?: NativeChatSettledTurns scopeKey?: string }): React.JSX.Element { const disclosure = useMobileNativeChatTurnDisclosure({ messages, enabled, isWorking, + settledTurns, scopeKey }) return createElement('result', { disclosure }) @@ -116,6 +120,56 @@ describe('useMobileNativeChatTurnDisclosure', () => { } }) + it('shows the host-recorded duration over the locally observed one', () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const messages = [userMessage('u1')] + act(() => { + renderer = create(createElement(Harness, { messages, enabled: true })) + }) + // Locally this turn ran 5s; the host says 3m 17s and the host wins. + vi.setSystemTime(6_000) + const settledTurns = new Map([['u1', { startedAt: 500, workedSeconds: 197 }]]) + act(() => { + renderer?.update( + createElement(Harness, { messages, enabled: true, isWorking: false, settledTurns }) + ) + }) + const row = renderer!.root.findByType('result').props.disclosure.resolveRow(0, messages[0]) + expect(row.turnStatus).toEqual({ startedAt: 500, thinking: false, workedSeconds: 197 }) + expect(row.turnKey).toBe('u1') + } finally { + vi.useRealTimers() + } + }) + + it('suppresses local duration when the host explicitly cannot verify the end', () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const messages = [userMessage('u1')] + act(() => { + renderer = create(createElement(Harness, { messages, enabled: true })) + }) + vi.setSystemTime(60_000) + act(() => { + renderer?.update( + createElement(Harness, { + messages, + enabled: true, + isWorking: false, + settledTurns: new Map([['u1', null]]) + }) + ) + }) + const row = renderer!.root.findByType('result').props.disclosure.resolveRow(0, messages[0]) + expect(row.turnStatus).toBeNull() + } finally { + vi.useRealTimers() + } + }) + it('keeps at most the latest 128 turns expanded', () => { vi.useFakeTimers() try { diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts index 46b58f29cba..a67a6a88663 100644 --- a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts @@ -1,5 +1,6 @@ import { useCallback, useMemo, useState } from 'react' import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import type { NativeChatSettledTurns } from '../../../src/shared/native-chat-turn-status' import { MOBILE_UNANCHORED_TURN_KEY, useMobileNativeChatTurnStatus, @@ -25,17 +26,28 @@ export function useMobileNativeChatTurnDisclosure({ messages, enabled, isWorking, + workingStartedAt, + settledTurns, + thinking = false, + activityText = null, scopeKey }: { messages: readonly NativeChatMessage[] enabled: boolean isWorking: boolean + workingStartedAt?: number | null + /** Host-recorded durations; they outrank whatever this client observed. */ + settledTurns?: NativeChatSettledTurns | null + /** Whether the turn is reasoning right now, derived from its journal content. */ + thinking?: boolean + /** What the provider says the live turn is doing; outranks the other labels. */ + activityText?: string | null /** Host/worktree/tab identity for timing and disclosure isolation. */ scopeKey: string }): { active: NativeChatTurnStatus | null - /** True when the live turn has no user message to hang its status row under. */ - activeTurnIsUnanchored: boolean + /** The live turn's provider activity copy, for the footer row. */ + activeActivityText: string | null onToggleTurn: (turnKey: string) => void resolveRow: (index: number, message: NativeChatMessage) => MobileNativeChatTurnRow } { @@ -43,6 +55,9 @@ export function useMobileNativeChatTurnDisclosure({ messages, enabled, isWorking, + workingStartedAt, + settledTurns, + thinking, scopeKey }) const [expandedTurns, setExpandedTurns] = useState<{ @@ -85,17 +100,16 @@ export function useMobileNativeChatTurnDisclosure({ }, [enabled, messages]) const { active, activeTurnKey, completedByTurn } = turnStatuses + const activeActivityText = enabled && isWorking ? (activityText ?? null) : null const resolveRow = useCallback( (index: number, message: NativeChatMessage): MobileNativeChatTurnRow => { const turnKey = turnKeys[index] const turnStatus = !enabled || message.role !== 'user' ? null - : turnKey === activeTurnKey - ? active - : turnKey - ? (completedByTurn[turnKey] ?? null) - : null + : turnKey + ? (completedByTurn[turnKey] ?? null) + : null return { turnStatus, turnExpanded: turnKey ? expandedTurnIds.has(turnKey) : false, @@ -112,15 +126,14 @@ export function useMobileNativeChatTurnDisclosure({ (turnKey === undefined && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY)) } }, - [turnKeys, enabled, activeTurnKey, active, completedByTurn, expandedTurnIds, isWorking] + [turnKeys, enabled, activeTurnKey, completedByTurn, expandedTurnIds, isWorking] ) return { active, + activeActivityText, /** Stable for a given chat scope, so it never disturbs a row's memo. */ onToggleTurn: toggleExpandedTurn, - activeTurnIsUnanchored: - enabled && active != null && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY, resolveRow } } diff --git a/mobile/src/session/use-mobile-native-chat-turn-status.ts b/mobile/src/session/use-mobile-native-chat-turn-status.ts index 13afbe70c09..f7cbb2dd887 100644 --- a/mobile/src/session/use-mobile-native-chat-turn-status.ts +++ b/mobile/src/session/use-mobile-native-chat-turn-status.ts @@ -1,9 +1,9 @@ import { useEffect, useMemo, useRef, useState } from 'react' import type { NativeChatMessage } from '../../../src/shared/native-chat-types' import { - nativeChatTurnHasResponse, reduceNativeChatTurnTiming, selectNativeChatTurnStatuses, + type NativeChatSettledTurns, type NativeChatTurnStatus, type NativeChatTurnTimingByTurn } from '../../../src/shared/native-chat-turn-status' @@ -25,12 +25,18 @@ export function useMobileNativeChatTurnStatus({ enabled, isWorking, workingStartedAt, + settledTurns, + thinking = false, scopeKey }: { messages: readonly NativeChatMessage[] enabled: boolean isWorking: boolean workingStartedAt?: number | null + /** Host-recorded durations; they outrank whatever this client observed. */ + settledTurns?: NativeChatSettledTurns | null + /** Whether the turn is reasoning right now, derived from its journal content. */ + thinking?: boolean /** Host/worktree/tab identity. Timings never carry across chat surfaces. */ scopeKey: string }): { @@ -41,7 +47,6 @@ export function useMobileNativeChatTurnStatus({ const latestUserIndex = enabled ? messages.findLastIndex((message) => message.role === 'user') : -1 - const hasCurrentTurnResponse = enabled && nativeChatTurnHasResponse(messages, latestUserIndex) const latestUserId = latestUserIndex !== -1 ? (messages[latestUserIndex]?.id ?? null) : null const activeTurnKey = latestUserId ?? MOBILE_UNANCHORED_TURN_KEY const [scopedTiming, setScopedTiming] = useState(() => ({ @@ -91,15 +96,18 @@ export function useMobileNativeChatTurnStatus({ // turn re-renders ~20x/s. Without this, every settled turn's row gets fresh // props each tick and the memoized message rows all re-render. const turnIsWorking = enabled && isWorking + const turnIsThinking = enabled && thinking + const settledByTurn = enabled ? (settledTurns ?? undefined) : undefined const statuses = useMemo( () => selectNativeChatTurnStatuses(timingByTurn, { activeTurnKey, isWorking: turnIsWorking, workingStartedAt, - hasCurrentTurnResponse + thinking: turnIsThinking, + settledByTurn }), - [timingByTurn, activeTurnKey, turnIsWorking, workingStartedAt, hasCurrentTurnResponse] + [timingByTurn, activeTurnKey, turnIsWorking, workingStartedAt, turnIsThinking, settledByTurn] ) return { ...statuses, activeTurnKey } } diff --git a/mobile/src/session/use-mobile-structured-agent-session.ts b/mobile/src/session/use-mobile-structured-agent-session.ts index 271f9143671..938e881aec7 100644 --- a/mobile/src/session/use-mobile-structured-agent-session.ts +++ b/mobile/src/session/use-mobile-structured-agent-session.ts @@ -11,7 +11,12 @@ import { import { encodeNativeChatTranscriptIdentity } from '../../../src/shared/native-chat-transcript-retention' import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' import { projectStructuredAgentSessionMessages } from '../../../src/shared/structured-agent-session-message-projection' -import { activeStructuredAgentSessionTurnId } from '../../../src/shared/structured-agent-session-projection' +import { hasUnansweredStructuredAgentSessionDispatch } from '../../../src/shared/structured-agent-session-projection' +import { + activeStructuredAgentSessionTurnId, + isStructuredAgentSessionThinking +} from '../../../src/shared/structured-agent-session-live-turn' +import { selectStructuredAgentTurnActivity } from '../../../src/shared/native-chat-turn-activity' import { pendingStructuredApproval, pendingStructuredQuestion, @@ -28,28 +33,33 @@ import type { RpcClient } from '../transport/rpc-client' import type { MobileChatPermission } from './mobile-native-chat-permission' import type { MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatSession } from './use-mobile-native-chat-session' +import type { NativeChatLiveTurnIndicator } from '../../../src/shared/native-chat-turn-status' import { useMobileStructuredAgentState } from './use-mobile-structured-agent-state' import { useMobileStructuredPromptResponses } from './use-mobile-structured-prompt-responses' import { useMobileStructuredAgentOptions } from './use-mobile-structured-agent-options' +import { useMobileStructuredAgentTurnTiming } from './use-mobile-structured-agent-turn-timing' type StructuredMobileAttachment = StructuredAgentSessionAttachment & { id?: string } -type StructuredMobileSession = ReturnType & { - session: MobileNativeChatSession - isWorking: boolean - turnId: string | null - sendWithOutcome: ( - text: string, - images?: string[], - deadline?: number, - attachments?: readonly StructuredMobileAttachment[] - ) => Promise - cancel: () => void - permission: MobileChatPermission | null - question: MobileChatQuestion | null - respondPermission: (optionId: string) => Promise - respondQuestion: (answer: string) => Promise -} +type StructuredMobileSession = ReturnType & + ReturnType & { + session: MobileNativeChatSession + isWorking: boolean + turnId: string | null + /** What labels the live turn's one indicator row. */ + turnIndicator: NativeChatLiveTurnIndicator + sendWithOutcome: ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredMobileAttachment[] + ) => Promise + cancel: () => void + permission: MobileChatPermission | null + question: MobileChatQuestion | null + respondPermission: (optionId: string) => Promise + respondQuestion: (answer: string) => Promise + } export function useMobileStructuredAgentSession(args: { client: RpcClient | null @@ -116,15 +126,7 @@ export function useMobileStructuredAgentSession(args: { [client, enabled, onSendError, sessionId, sessionKey] ) - const { - conversationCommands, - optionPickerRequest, - invokeStructuredOption, - optionSnapshot, - optionSurface, - pendingOptionId, - setStructuredOption - } = useMobileStructuredAgentOptions({ + const options = useMobileStructuredAgentOptions({ agent, client, sessionId, @@ -132,6 +134,8 @@ export function useMobileStructuredAgentSession(args: { fence: state.fence, mutate }) + const { conversationCommands, invokeStructuredOption, optionSnapshot, setStructuredOption } = + options const sendWithOutcome = useCallback( async ( @@ -269,6 +273,14 @@ export function useMobileStructuredAgentSession(args: { () => projectStructuredAgentSessionMessages(state.items, [], state.submissions), [state.items, state.submissions] ) + const turnId = activeStructuredAgentSessionTurnId(state.items) + const turnTiming = useMobileStructuredAgentTurnTiming(state, turnId) + const activityText = + selectStructuredAgentTurnActivity(state.items, turnId, state.activity)?.text ?? null + const thinking = isStructuredAgentSessionThinking(state.items) + // Stable while the readings hold, so a streaming turn does not re-render the + // whole chat surface on every journal batch. + const turnIndicator = useMemo(() => ({ thinking, activityText }), [thinking, activityText]) const status = state.status === 'idle' ? 'idle' : state.status const approvalPrompt = useMemo( () => state.items.find(pendingStructuredApproval) ?? null, @@ -280,8 +292,7 @@ export function useMobileStructuredAgentSession(args: { ) return { - conversationCommands, - optionPickerRequest, + ...options, session: { messages, status, @@ -291,18 +302,18 @@ export function useMobileStructuredAgentSession(args: { loadingEarlier: loadingOlder, loadEarlier }, - isWorking: activeStructuredAgentSessionTurnId(state.items) !== null, - turnId: activeStructuredAgentSessionTurnId(state.items), + // A dispatch the provider has not answered yet is already work — see the desktop hook. + isWorking: + turnId !== null || + hasUnansweredStructuredAgentSessionDispatch(state.submissions, state.fence), + turnId, + turnIndicator, + ...turnTiming, sendWithOutcome, cancel, permission: projectStructuredPermission(approvalPrompt), question: projectStructuredQuestion(questionPrompt, groupedDraft), - optionSnapshot, - optionSurface, - pendingOptionId, respondPermission, - respondQuestion, - setStructuredOption, - invokeStructuredOption + respondQuestion } } diff --git a/mobile/src/session/use-mobile-structured-agent-state.ts b/mobile/src/session/use-mobile-structured-agent-state.ts index 52aefab24aa..49426004968 100644 --- a/mobile/src/session/use-mobile-structured-agent-state.ts +++ b/mobile/src/session/use-mobile-structured-agent-state.ts @@ -16,6 +16,8 @@ import type { RpcClient } from '../transport/rpc-client' import { callAgentSession } from './mobile-structured-agent-session-rpc' const MAX_RETAINED_SESSION_STATES = 32 +/** Bounded so a busy stream cannot turn one Load-earlier tap into an endless read chain. */ +const OLDER_PAGE_ANCHOR_ATTEMPTS = 3 function isSubscribeEvent(value: unknown): value is AgentSessionSubscribeEvent { if (typeof value !== 'object' || value === null) { @@ -65,7 +67,7 @@ export function useMobileStructuredAgentState(args: { } setSessionStates((current) => { const previous = current.get(sessionKey) ?? EMPTY_STRUCTURED_AGENT_SESSION - const next = reduceStructuredAgentSession(previous, action) + const next = reduceStructuredAgentSession(previous, action, Date.now()) if (next === previous) { return current } @@ -154,41 +156,45 @@ export function useMobileStructuredAgentState(args: { if (!client || !sessionId || !sessionKey || loadingOlder || !current.hasOlder) { return } - const cursor = oldestStructuredAgentSessionCursor(current) - if (!cursor) { + if (!oldestStructuredAgentSessionCursor(current)) { return } const requestSessionKey = sessionKey const requestGeneration = streamGenerationRef.current + const isCurrentRead = (): boolean => + sessionKeyRef.current === requestSessionKey && + streamGenerationRef.current === requestGeneration setLoadingOlder(true) - void callAgentSession(client, 'agentSession.history', { - sessionId, - direction: 'before', - cursor, - limit: AGENT_SESSION_HISTORY_MAX_LIMIT - }) - .then((result) => { - if ( - result.ok && - sessionKeyRef.current === requestSessionKey && - streamGenerationRef.current === requestGeneration - ) { - apply({ type: 'older-page', requestedEpoch: cursor.epoch, page: result.page }) + void (async () => { + // A live batch can head-trim past the anchor mid-read, and the reducer drops that + // page rather than leave a hole in the transcript. Re-anchor and retry. + for (let attempt = 0; attempt < OLDER_PAGE_ANCHOR_ATTEMPTS; attempt += 1) { + const cursor = oldestStructuredAgentSessionCursor(stateRef.current) + if (!cursor || !isCurrentRead()) { + return } - }) + const result = await callAgentSession( + client, + 'agentSession.history', + { sessionId, direction: 'before', cursor, limit: AGENT_SESSION_HISTORY_MAX_LIMIT } + ) + if (!result.ok || !isCurrentRead()) { + return + } + // The reducer drops a page whose anchor slid, so only an intact anchor lands. + if (oldestStructuredAgentSessionCursor(stateRef.current)?.sequence === cursor.sequence) { + apply({ type: 'older-page', requestedCursor: cursor, page: result.page }) + return + } + } + })() .catch((error: unknown) => { - if ( - sessionKeyRef.current === requestSessionKey && - streamGenerationRef.current === requestGeneration - ) { + if (isCurrentRead()) { apply({ type: 'error', message: error instanceof Error ? error.message : String(error) }) } }) .finally(() => { - if ( - sessionKeyRef.current === requestSessionKey && - streamGenerationRef.current === requestGeneration - ) { + if (isCurrentRead()) { setLoadingOlder(false) } }) diff --git a/mobile/src/session/use-mobile-structured-agent-turn-timing.test.tsx b/mobile/src/session/use-mobile-structured-agent-turn-timing.test.tsx new file mode 100644 index 00000000000..944b1d9be13 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-agent-turn-timing.test.tsx @@ -0,0 +1,165 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalRenderItem, + AgentJournalSubmission, + AgentJournalTurnLifecycle +} from '../../../src/shared/agent-session-journal-types' +import { agentJournalTurnBody } from '../../../src/shared/agent-session-turn-record' +import { useMobileStructuredAgentTurnTiming } from './use-mobile-structured-agent-turn-timing' + +// Host clock sits an hour ahead of the client's so any leak of a host timestamp +// into the local anchor shows up as a huge offset. +const HOST_START = 3_600_000_000 +const CLIENT_NOW = 12_345_000 + +function user(itemId: string, sequence: number): AgentJournalRenderItem { + return { + itemId, + revision: 0, + sequence, + observedAt: HOST_START + sequence, + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: itemId }] } + } +} + +function lifecycle( + turnId: string, + sequence: number, + turn: Omit, + observedAt: number +): AgentJournalRenderItem { + return { + itemId: `lifecycle-${turnId}`, + revision: 1, + sequence, + observedAt, + body: agentJournalTurnBody({ turnId, ...turn }) + } +} + +type Timing = ReturnType +const NO_SUBMISSIONS: readonly AgentJournalSubmission[] = [] + +// The submission the provider acknowledged under the key its lifecycle row cites. +const SUBMISSIONS: AgentJournalSubmission[] = [ + { + clientMessageId: 'first', + fence: 1, + payloadFingerprint: 'fp', + dispatchState: 'accepted', + providerItemId: 'codex:thread:t1:0', + reason: null, + submittedAt: 1, + resolvedAt: 2 + } +] + +describe('useMobileStructuredAgentTurnTiming', () => { + let renderer: ReactTestRenderer | null = null + let timing: Timing | null = null + + function Harness({ + items, + submissions = NO_SUBMISSIONS, + turnId, + hostClock + }: { + items: readonly AgentJournalRenderItem[] + submissions?: readonly AgentJournalSubmission[] + turnId: string | null + hostClock?: { hostNow: number; receivedAt: number } + }): null { + timing = useMobileStructuredAgentTurnTiming({ items, submissions, hostClock }, turnId) + return null + } + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + timing = null + vi.useRealTimers() + }) + + it('hands settled host durations through and anchors the live counter locally, once per turn', () => { + vi.useFakeTimers() + vi.setSystemTime(CLIENT_NOW) + const items = [ + user('u1', 1), + lifecycle( + 't1', + 2, + { + state: 'interrupted', + startedAt: HOST_START, + completedAt: HOST_START + 61_000, + userItemId: 'codex:thread:t1:0' + }, + HOST_START + 5 + ), + user('u2', 3), + // The host appended the row 2.5s after it saw the turn start. + lifecycle( + 't2', + 4, + { state: 'running', startedAt: HOST_START + 100_000 }, + HOST_START + 102_500 + ) + ] + const submissions = SUBMISSIONS + act(() => { + renderer = create(createElement(Harness, { items, submissions, turnId: 't2' })) + }) + expect(timing?.workingStartedAt).toBe(CLIENT_NOW - 2_500) + // The row's provider key resolves through the submission alias, not journal order. + expect([...timing!.settledTurns]).toEqual([ + ['orca:first', { startedAt: HOST_START, workedSeconds: 61 }], + ['u2', null] + ]) + + vi.setSystemTime(CLIENT_NOW + 30_000) + act(() => + renderer?.update(createElement(Harness, { items: [...items], submissions, turnId: 't2' })) + ) + expect(timing?.workingStartedAt).toBe(CLIENT_NOW - 2_500) + + act(() => renderer?.update(createElement(Harness, { items, turnId: null }))) + expect(timing?.workingStartedAt).toBeNull() + + // With a host clock that said the turn was 35s old 5s ago, the anchor sits + // 40s before first sight, wherever the client's absolute clock is. + vi.setSystemTime(CLIENT_NOW + 60_000) + const next = [ + ...items, + user('u3', 5), + lifecycle( + 't3', + 6, + { state: 'running', startedAt: HOST_START + 150_000 }, + HOST_START + 150_100 + ) + ] + act(() => + renderer?.update( + createElement(Harness, { + items: next, + turnId: 't3', + hostClock: { hostNow: HOST_START + 185_000, receivedAt: CLIENT_NOW + 55_000 } + }) + ) + ) + expect(timing?.workingStartedAt).toBe(CLIENT_NOW + 60_000 - 40_000) + }) + + it('leaves the anchor null when an older host records no start', () => { + vi.useFakeTimers() + vi.setSystemTime(CLIENT_NOW) + const items = [user('u1', 1), lifecycle('t1', 2, { state: 'running' }, HOST_START)] + act(() => { + renderer = create(createElement(Harness, { items, turnId: 't1' })) + }) + expect(timing?.workingStartedAt).toBeNull() + expect(timing?.settledTurns.size).toBe(0) + }) +}) diff --git a/mobile/src/session/use-mobile-structured-agent-turn-timing.ts b/mobile/src/session/use-mobile-structured-agent-turn-timing.ts new file mode 100644 index 00000000000..48189811e59 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-agent-turn-timing.ts @@ -0,0 +1,70 @@ +import { useMemo, useState } from 'react' +import type { + AgentJournalRenderItem, + AgentJournalSubmission +} from '../../../src/shared/agent-session-journal-types' +import type { NativeChatSettledTurns } from '../../../src/shared/native-chat-turn-status' +import { + selectStructuredAgentRunningTurnTiming, + selectStructuredAgentSettledTurns, + structuredAgentTurnLocalStartedAt +} from '../../../src/shared/structured-agent-session-turn-timing' + +type TurnAnchor = { turnId: string; startedAt: number | null } + +/** The host's clock as last published, paired with the client clock at receipt. */ +type HostClock = { hostNow: number; receivedAt: number } + +/** The live turn's local-clock anchor. Null when its row carries no host start + * (older hosts), so local observation applies. */ +function anchorRunningTurn( + items: readonly AgentJournalRenderItem[], + turnId: string, + hostClock: HostClock | null | undefined +): TurnAnchor { + const timing = selectStructuredAgentRunningTurnTiming(items, turnId) + if (!timing) { + return { turnId, startedAt: null } + } + const now = Date.now() + // Advance the published host clock by the client time since receipt; both + // terms stay single-clock, so a mid-turn attach counts from the real start. + const hostNow = hostClock ? hostClock.hostNow + (now - hostClock.receivedAt) : undefined + return { turnId, startedAt: structuredAgentTurnLocalStartedAt(timing, now, hostNow) } +} + +/** Host-recorded turn timing for the structured lane: settled durations straight + * off the journal, and a skew-free start for the live counter stamped once per + * turn so re-renders never move it. */ +export function useMobileStructuredAgentTurnTiming( + { + items, + submissions, + hostClock + }: { + items: readonly AgentJournalRenderItem[] + submissions: readonly AgentJournalSubmission[] + hostClock?: HostClock | null + }, + turnId: string | null +): { settledTurns: NativeChatSettledTurns; workingStartedAt: number | null } { + const settledTurns = useMemo( + () => selectStructuredAgentSettledTurns(items, submissions), + [items, submissions] + ) + const [anchor, setAnchor] = useState(null) + // Stamp during render (React's derive-from-props pattern) so the first paint of + // a new turn already counts from the right instant. + if (turnId === null) { + if (anchor !== null) { + setAnchor(null) + } + return { settledTurns, workingStartedAt: null } + } + if (anchor?.turnId !== turnId) { + const next = anchorRunningTurn(items, turnId, hostClock) + setAnchor(next) + return { settledTurns, workingStartedAt: next.startedAt } + } + return { settledTurns, workingStartedAt: anchor.startedAt } +} diff --git a/mobile/src/session/use-mobile-structured-native-chat-send-bridge.test.ts b/mobile/src/session/use-mobile-structured-native-chat-send-bridge.test.ts new file mode 100644 index 00000000000..3cf432381c2 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-native-chat-send-bridge.test.ts @@ -0,0 +1,88 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionHandleProvider } from '../../../src/shared/agent-session-provider-handle' +import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' +import type { MobileNativeChatSendOrigin } from './use-mobile-native-chat-drafts' +import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge' + +const ORIGIN: MobileNativeChatSendOrigin = { + draftKey: 'draft', + draftEditGeneration: 0, + pendingKey: 'pending', + normalizedText: 'command', + baselineOccurrences: 0, + baselineTailMessageId: null, + baselineResolved: true +} + +describe('useMobileStructuredNativeChatSendBridge', () => { + let renderer: ReactTestRenderer | null = null + let sendWithOutcome: (text: string) => Promise + const acceptSend = vi.fn() + const captureSendOrigin = vi.fn(() => ORIGIN) + const clearDraftForSend = vi.fn() + const holdUnconfirmedSend = vi.fn() + const onSendError = vi.fn() + const restoreRejectedDraft = vi.fn() + const sendStructured = vi.fn() + + function Harness({ agent }: { agent: AgentSessionHandleProvider }): null { + sendWithOutcome = useMobileStructuredNativeChatSendBridge({ + agent, + acceptSend, + captureSendOrigin, + clearDraftForSend, + holdUnconfirmedSend, + onSendError, + restoreRejectedDraft, + sendStructured + }).sendWithOutcome + return null + } + + function mount(agent: AgentSessionHandleProvider): void { + act(() => { + renderer = create(createElement(Harness, { agent })) + }) + } + + beforeEach(() => { + vi.clearAllMocks() + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + }) + + it('optimistically echoes accepted commands owned by the active agent', async () => { + sendStructured.mockResolvedValue('accepted') + mount('claude') + + await expect(sendWithOutcome('/init')).resolves.toBe('accepted') + + expect(acceptSend).toHaveBeenCalledWith(ORIGIN, '/init', undefined) + expect(restoreRejectedDraft).not.toHaveBeenCalled() + }) + + it('holds unknown delivery for commands owned by the active agent', async () => { + sendStructured.mockResolvedValue('unknown') + mount('claude') + + await expect(sendWithOutcome('/review')).resolves.toBe('unknown') + + expect(holdUnconfirmedSend).toHaveBeenCalledWith(ORIGIN, '/review', expect.any(Function)) + expect(restoreRejectedDraft).not.toHaveBeenCalled() + }) + + it('keeps host-command reconciliation for Codex', async () => { + sendStructured.mockResolvedValue('unknown') + mount('codex') + + await expect(sendWithOutcome('/review')).resolves.toBe('unknown') + + expect(restoreRejectedDraft).toHaveBeenCalledWith(ORIGIN, '/review') + expect(holdUnconfirmedSend).not.toHaveBeenCalled() + }) +}) diff --git a/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts b/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts index 70261e9df9b..d1cfe3d42f4 100644 --- a/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts +++ b/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts @@ -1,4 +1,5 @@ import { useCallback } from 'react' +import type { AgentSessionHandleProvider } from '../../../src/shared/agent-session-provider-handle' import { isStructuredAgentSessionComposerCommand } from '../../../src/shared/structured-agent-session-composer' import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' import type { MobileNativeChatSendOrigin } from './use-mobile-native-chat-drafts' @@ -10,6 +11,7 @@ type StructuredNativeChatAttachment = { } export function useMobileStructuredNativeChatSendBridge(args: { + agent: AgentSessionHandleProvider sendStructured: ( text: string, images?: string[], @@ -37,6 +39,7 @@ export function useMobileStructuredNativeChatSendBridge(args: { } { const { acceptSend, + agent, captureSendOrigin, clearDraftForSend, holdUnconfirmedSend, @@ -56,6 +59,7 @@ export function useMobileStructuredNativeChatSendBridge(args: { onSendError('Message not sent (disconnected)') return 'rejected' } + const isHostCommand = isStructuredAgentSessionComposerCommand(text, agent) clearDraftForSend(origin, text) const outcome = attachments !== undefined @@ -66,19 +70,13 @@ export function useMobileStructuredNativeChatSendBridge(args: { ? await sendStructured(text, images) : await sendStructured(text) if (outcome === 'accepted') { - if ( - !isStructuredAgentSessionComposerCommand(text, 'codex') && - !isStructuredAgentSessionComposerCommand(text, 'claude') - ) { + if (!isHostCommand) { acceptSend(origin, text.trimEnd(), images) } return 'accepted' } if (outcome === 'unknown') { - if ( - isStructuredAgentSessionComposerCommand(text, 'codex') || - isStructuredAgentSessionComposerCommand(text, 'claude') - ) { + if (isHostCommand) { restoreRejectedDraft(origin, text) return 'unknown' } @@ -92,6 +90,7 @@ export function useMobileStructuredNativeChatSendBridge(args: { }, [ acceptSend, + agent, captureSendOrigin, clearDraftForSend, holdUnconfirmedSend, diff --git a/mobile/src/session/use-mobile-structured-turn-indicator.test.tsx b/mobile/src/session/use-mobile-structured-turn-indicator.test.tsx new file mode 100644 index 00000000000..8369ff03a17 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-turn-indicator.test.tsx @@ -0,0 +1,138 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalRenderItem } from '../../../src/shared/agent-session-journal-types' +import type { AgentSessionSubscribeEvent } from '../../../src/shared/agent-session-wire' +import type { RpcClient } from '../transport/rpc-client' +import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' + +function journalItem( + sequence: number, + body: AgentJournalRenderItem['body'] +): AgentJournalRenderItem { + return { itemId: `item-${sequence}`, revision: 1, sequence, observedAt: sequence, body } +} + +function snapshot(items: AgentJournalRenderItem[], fence: number): AgentSessionSubscribeEvent { + const newest = items.length + return { + type: 'snapshot', + sessionId: 'session-1', + fence, + page: { + sessionId: 'session-1', + epoch: 'epoch-1', + fence, + direction: 'tail', + items, + removedItemIds: [], + submissions: [], + window: { + oldest: { epoch: 'epoch-1', sequence: 1 }, + newest: { epoch: 'epoch-1', sequence: newest }, + nextCursor: { epoch: 'epoch-1', sequence: newest + 1 } + }, + liveCursor: { epoch: 'epoch-1', sequence: newest }, + hasOlder: false, + hasNewer: false + } + } as AgentSessionSubscribeEvent +} + +/** What the one live indicator row reads, resolved off the session journal. */ +describe('useMobileStructuredAgentSession turn indicator', () => { + let renderer: ReactTestRenderer | null = null + let hook: ReturnType | null = null + let listener: ((value: unknown) => void) | null = null + const sendRequest = vi.fn(async (method: string) => ({ + ok: true, + result: + method === 'agentSession.options' + ? { + models: [{ id: 'gpt-fast', label: 'GPT Fast', isDefault: true, efforts: [] }], + current: { model: 'gpt-fast' } + } + : {}, + _meta: { runtimeId: 'r1' } + })) + const subscribe = vi.fn((_method: string, _params: unknown, onData: (value: unknown) => void) => { + listener = onData + return vi.fn() + }) + const client = { sendRequest, subscribe } as unknown as RpcClient + // Stable across renders: a fresh callback would re-run the hold/subscribe effect + // and release the session out from under the test. + const onSendError = vi.fn() + + function Harness(): null { + hook = useMobileStructuredAgentSession({ + client, + sessionId: 'session-1', + sourceIdentity: 'host-a\0workspace-a', + enabled: true, + connected: true, + agent: 'codex', + onSendError + } as never) + return null + } + + beforeEach(() => { + vi.clearAllMocks() + listener = null + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + hook = null + }) + + const runningTurn = journalItem(1, { kind: 'turn', turnId: 'turn-1', state: 'running' }) + const reasoning = journalItem(2, { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Weighing two approaches' }] + }) + + it('reads the live turn as reasoning while reasoning is its newest content', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).not.toBeNull()) + + act(() => { + listener?.(snapshot([runningTurn, reasoning], 3)) + }) + + expect(hook?.turnIndicator).toEqual({ thinking: true, activityText: null }) + }) + + it('hands the row the provider copy once real content ends the reasoning', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).not.toBeNull()) + + act(() => { + listener?.( + snapshot( + [ + runningTurn, + reasoning, + journalItem(3, { + kind: 'tool-call', + name: 'shell', + input: { command: 'pnpm lint' }, + state: 'running' + }), + journalItem(4, { kind: 'status', text: 'Updating the plan' }) + ], + 3 + ) + ) + }) + + expect(hook?.turnIndicator).toEqual({ thinking: false, activityText: 'Updating the plan' }) + }) +}) diff --git a/mobile/src/settings/about-screen.tsx b/mobile/src/settings/about-screen.tsx new file mode 100644 index 00000000000..3212343fee6 --- /dev/null +++ b/mobile/src/settings/about-screen.tsx @@ -0,0 +1,187 @@ +import { useState } from 'react' +import { View, Text, StyleSheet, Pressable } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { ChevronLeft, Globe } from 'lucide-react-native' +import Svg, { Path } from 'react-native-svg' +import { OrcaLogo } from '../components/OrcaLogo' +import { colors, spacing, typography } from '../theme/mobile-theme' + +function GithubIcon({ size = 16, color = colors.textSecondary }) { + return ( + + + + ) +} + +function XIcon({ size = 16, color = colors.textSecondary }) { + return ( + + + + ) +} + +export default function AboutScreen({ + onBack, + openExternal, + versionLabel +}: { + onBack: () => void + openExternal: (url: string) => Promise + versionLabel: string +}) { + const [error, setError] = useState(null) + const openLink = (url: string) => { + setError(null) + void openExternal(url).catch(() => setError('Could not open the link. Try again.')) + } + const insets = useSafeAreaInsets() + + return ( + + + + + + About + + + + + Orca + Open-source agent IDE for 100x builders + + + + [styles.row, pressed && styles.rowPressed]} + accessibilityRole="button" + accessibilityLabel="Orca website" + onPress={() => openLink('https://onOrca.dev')} + > + + onOrca.dev + + + [styles.row, pressed && styles.rowPressed]} + accessibilityRole="button" + accessibilityLabel="Orca source code" + onPress={() => openLink('https://github.com/stablyai/orca')} + > + + stablyai/orca + + + [styles.row, pressed && styles.rowPressed]} + accessibilityRole="button" + accessibilityLabel="Orca on X" + onPress={() => openLink('https://x.com/orca_build')} + > + + @orca_build + + + + {versionLabel} + {error && ( + + {error} + + )} + + ) +} + +const styles = StyleSheet.create({ + container: { + flex: 1, + backgroundColor: colors.bgBase, + padding: spacing.lg + }, + topRow: { + flexDirection: 'row', + alignItems: 'center', + marginBottom: spacing.xl + }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { + fontSize: 20, + fontWeight: '700', + color: colors.textPrimary + }, + brand: { + alignItems: 'center', + paddingVertical: spacing.xl, + marginBottom: spacing.lg + }, + brandName: { + fontSize: 22, + fontWeight: '800', + color: colors.textPrimary, + marginTop: spacing.sm + }, + brandSub: { + fontSize: 13, + color: colors.textMuted, + marginTop: spacing.xs + }, + section: { + backgroundColor: colors.bgPanel, + borderRadius: 12, + overflow: 'hidden' + }, + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + rowPressed: { + backgroundColor: colors.bgRaised + }, + rowLabel: { + flex: 1, + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + rowValue: { + flex: 1, + textAlign: 'right', + fontSize: typography.bodySize, + color: colors.textSecondary + }, + separator: { + height: StyleSheet.hairlineWidth, + backgroundColor: colors.borderSubtle, + marginHorizontal: spacing.md + }, + versionText: { + marginTop: spacing.lg, + textAlign: 'center', + fontSize: typography.metaSize, + color: colors.textMuted + }, + errorText: { + marginTop: spacing.sm, + textAlign: 'center', + fontSize: typography.metaSize, + color: colors.statusRed + } +}) diff --git a/mobile/src/settings/browser-settings-screen.tsx b/mobile/src/settings/browser-settings-screen.tsx new file mode 100644 index 00000000000..924f7a7660b --- /dev/null +++ b/mobile/src/settings/browser-settings-screen.tsx @@ -0,0 +1,199 @@ +import { useCallback, useEffect, useState } from 'react' +import { Pressable, ScrollView, StyleSheet, Text, View } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { useRouter } from 'expo-router' +import { ChevronLeft, ChevronRight, Globe } from 'lucide-react-native' +import { PickerModal, type PickerOption } from '../components/PickerModal' +import { + loadTerminalLinkOpenMode, + saveTerminalLinkOpenMode, + type MobileTerminalLinkOpenMode +} from '../storage/preferences' +import { colors, radii, spacing, typography } from '../theme/mobile-theme' + +const LINK_MODE_OPTIONS: PickerOption[] = [ + { + value: 'orca-browser', + label: 'Orca browser on desktop', + subtitle: 'Open in the streamed browser from your paired desktop.' + }, + { + value: 'phone-browser', + label: 'Phone browser', + subtitle: 'Open in Safari, Chrome, or another browser on this phone.' + } +] + +function linkModeLabel(mode: MobileTerminalLinkOpenMode): string { + return ( + LINK_MODE_OPTIONS.find((option) => option.value === mode)?.label ?? LINK_MODE_OPTIONS[0]!.label + ) +} + +export default function BrowserSettingsScreen({ + onBack +}: { + onBack?: () => void +}): React.JSX.Element { + const router = useRouter() + const insets = useSafeAreaInsets() + const [linkMode, setLinkMode] = useState('orca-browser') + const [pickerOpen, setPickerOpen] = useState(false) + const [error, setError] = useState(null) + + useEffect(() => { + let active = true + void loadTerminalLinkOpenMode().then( + (mode) => { + if (active) { + setLinkMode(mode) + } + }, + () => { + if (active) { + setError('Could not load browser preferences. Try again.') + } + } + ) + return () => { + active = false + } + }, []) + + const selectLinkMode = useCallback((mode: MobileTerminalLinkOpenMode) => { + setError(null) + // Optimistic, as base was: the row shows the tapped mode before the write lands. + setLinkMode(mode) + void saveTerminalLinkOpenMode(mode).catch(() => + setError('Could not save browser preferences. Try again.') + ) + }, []) + + return ( + + + router.back())} + > + + + Browser + + + + LINKS + + Choose where HTTP(S) links tapped in terminal output open. + + {error && ( + + {error} + + )} + + [styles.row, pressed && styles.rowPressed]} + onPress={() => setPickerOpen(true)} + > + + + Open terminal links + {linkModeLabel(linkMode)} + + + + + + + + visible={pickerOpen} + title="Open terminal links" + options={LINK_MODE_OPTIONS} + selected={linkMode} + onSelect={selectLinkMode} + onClose={() => setPickerOpen(false)} + /> + + ) +} + +const styles = StyleSheet.create({ + container: { + flex: 1, + backgroundColor: colors.bgBase, + paddingHorizontal: spacing.lg, + paddingTop: 0 + }, + topRow: { + flexDirection: 'row', + alignItems: 'center', + marginTop: spacing.sm, + marginBottom: spacing.lg + }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { + fontSize: 20, + fontWeight: '700', + color: colors.textPrimary + }, + scrollContent: { + paddingBottom: spacing.xl + }, + groupHeading: { + fontSize: 11, + fontWeight: '600', + color: colors.textMuted, + letterSpacing: 0.5, + marginBottom: spacing.xs, + paddingHorizontal: spacing.xs + }, + groupDescription: { + fontSize: typography.bodySize - 1, + color: colors.textSecondary, + lineHeight: 20, + paddingHorizontal: spacing.xs + }, + section: { + backgroundColor: colors.bgPanel, + borderRadius: radii.card, + overflow: 'hidden' + }, + sectionTopGap: { + marginTop: spacing.sm + }, + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + rowPressed: { + backgroundColor: colors.bgRaised + }, + rowContent: { + flex: 1 + }, + rowLabel: { + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + rowSublabel: { + fontSize: typography.bodySize - 2, + color: colors.textSecondary, + marginTop: 2 + } +}) diff --git a/mobile/src/settings/mobile-settings-menu-items.ts b/mobile/src/settings/mobile-settings-menu-items.ts new file mode 100644 index 00000000000..baf7faf7b3d --- /dev/null +++ b/mobile/src/settings/mobile-settings-menu-items.ts @@ -0,0 +1,14 @@ +import { Bell, Globe, Info, MessageSquare, Mic, Terminal, Wrench } from 'lucide-react-native' +import type { MobileSettingsMenuItem } from './mobile-settings-menu' + +export function mobileSettingsMenuItems(push: (route: string) => void): MobileSettingsMenuItem[] { + return [ + { label: 'Terminal', icon: Terminal, onPress: () => push('/terminal-settings') }, + { label: 'Chat UI', icon: MessageSquare, onPress: () => push('/native-chat-settings') }, + { label: 'Browser', icon: Globe, onPress: () => push('/browser-settings') }, + { label: 'Voice', icon: Mic, onPress: () => push('/voice-settings') }, + { label: 'Notifications', icon: Bell, onPress: () => push('/notifications') }, + { label: 'Troubleshooting', icon: Wrench, onPress: () => push('/troubleshoot') }, + { label: 'About', icon: Info, onPress: () => push('/about') } + ] +} diff --git a/mobile/src/settings/mobile-settings-menu.tsx b/mobile/src/settings/mobile-settings-menu.tsx new file mode 100644 index 00000000000..23f989e073c --- /dev/null +++ b/mobile/src/settings/mobile-settings-menu.tsx @@ -0,0 +1,111 @@ +import type { ReactNode } from 'react' +import { Pressable, ScrollView, StyleSheet, Text, View } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { useRouter } from 'expo-router' +import { ChevronLeft, ChevronRight, type LucideIcon } from 'lucide-react-native' +import { colors, spacing, typography } from '../theme/mobile-theme' + +export function MobileSettingsFrame({ + children, + onBack +}: { + children: ReactNode + onBack?: () => void +}) { + const router = useRouter() + const insets = useSafeAreaInsets() + return ( + + + router.back())} + > + + + Settings + + + {children} + + + ) +} + +export type MobileSettingsMenuItem = { + label: string + icon: LucideIcon + onPress: () => void + external?: boolean + disabled?: boolean +} + +export function MobileSettingsSection({ + items, + spaced = false +}: { + items: MobileSettingsMenuItem[] + spaced?: boolean +}) { + return ( + + {items.map(({ label, icon: Icon, onPress, external, disabled }, index) => ( + + {index > 0 && } + [styles.row, pressed && styles.rowPressed]} + onPress={onPress} + > + + {label} + {!external && } + + + ))} + + ) +} + +const styles = StyleSheet.create({ + container: { flex: 1, backgroundColor: colors.bgBase, paddingHorizontal: spacing.lg }, + topRow: { flexDirection: 'row', alignItems: 'center', marginBottom: spacing.xl }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { fontSize: 20, fontWeight: '700', color: colors.textPrimary }, + section: { backgroundColor: colors.bgPanel, borderRadius: 12, overflow: 'hidden' }, + sectionSpacer: { marginTop: spacing.md }, + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + rowPressed: { backgroundColor: colors.bgRaised }, + rowLabel: { + flex: 1, + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + separator: { + height: StyleSheet.hairlineWidth, + backgroundColor: colors.borderSubtle, + marginHorizontal: spacing.md + } +}) diff --git a/mobile/src/settings/native-chat-settings-screen.tsx b/mobile/src/settings/native-chat-settings-screen.tsx new file mode 100644 index 00000000000..fd7a6ac7f36 --- /dev/null +++ b/mobile/src/settings/native-chat-settings-screen.tsx @@ -0,0 +1,126 @@ +import { View, Text, StyleSheet, Pressable, ScrollView, Switch } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { useRouter } from 'expo-router' +import { ChevronLeft } from 'lucide-react-native' +import { colors, radii, spacing, typography } from '../theme/mobile-theme' +import { useMobileDefaultSessionViewPreference } from '../session/use-mobile-default-session-view-preference' + +export default function NativeChatSettingsScreen({ onBack }: { onBack?: () => void }) { + const router = useRouter() + const insets = useSafeAreaInsets() + + const { defaultView, setDefaultView } = useMobileDefaultSessionViewPreference() + const chatDefault = defaultView === 'chat' + + return ( + + + router.back())} + > + + + Chat UI + + + + DEFAULT VIEW + + Choose how supported agent sessions (Claude, Codex, and other chat-capable agents) open on + this device. Terminal shows the raw CLI; Chat UI shows a chat interface like the desktop + app. You can still switch any individual session from its long-press menu. + + + + + Open sessions in Chat UI + {chatDefault ? 'On' : 'Off'} + + setDefaultView(next ? 'chat' : 'terminal')} + trackColor={{ false: colors.bgRaised, true: colors.textSecondary }} + thumbColor={colors.textPrimary} + /> + + + + + ) +} + +const styles = StyleSheet.create({ + container: { + flex: 1, + backgroundColor: colors.bgBase, + paddingHorizontal: spacing.lg + }, + topRow: { + flexDirection: 'row', + alignItems: 'center', + marginTop: spacing.sm, + marginBottom: spacing.lg + }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { + fontSize: 20, + fontWeight: '700', + color: colors.textPrimary + }, + groupHeading: { + fontSize: 11, + fontWeight: '600', + color: colors.textMuted, + letterSpacing: 0.5, + marginBottom: spacing.xs, + paddingHorizontal: spacing.xs + }, + groupDescription: { + fontSize: typography.bodySize - 1, + color: colors.textSecondary, + lineHeight: 20, + paddingHorizontal: spacing.xs + }, + section: { + backgroundColor: colors.bgPanel, + borderRadius: radii.card, + overflow: 'hidden' + }, + sectionTopGap: { + marginTop: spacing.sm + }, + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + rowContent: { + flex: 1 + }, + rowLabel: { + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + rowSublabel: { + fontSize: typography.bodySize - 2, + color: colors.textSecondary, + marginTop: 2 + } +}) diff --git a/mobile/src/settings/native-notification-settings-operations.ts b/mobile/src/settings/native-notification-settings-operations.ts new file mode 100644 index 00000000000..cd5da9584bf --- /dev/null +++ b/mobile/src/settings/native-notification-settings-operations.ts @@ -0,0 +1,23 @@ +import { Linking } from 'react-native' +import { + ensureNotificationPermissions, + getNotificationPermissionState +} from '../notifications/notification-permissions' +import { loadPushNotificationsEnabled, savePushNotificationsEnabled } from '../storage/preferences' +import type { NotificationSettingsOperations } from './notification-settings-operations' + +export const nativeNotificationSettingsOperations: NotificationSettingsOperations = { + async permission(request) { + if (request) { + await ensureNotificationPermissions() + } + return getNotificationPermissionState() + }, + async preference(enabled) { + if (enabled !== undefined) { + await savePushNotificationsEnabled(enabled) + } + return { enabled: await loadPushNotificationsEnabled() } + }, + openSettings: () => Linking.openSettings() +} diff --git a/mobile/src/settings/native-voice-settings-operations.ts b/mobile/src/settings/native-voice-settings-operations.ts new file mode 100644 index 00000000000..12903c72b85 --- /dev/null +++ b/mobile/src/settings/native-voice-settings-operations.ts @@ -0,0 +1,19 @@ +import type { RpcClient } from '../transport/rpc-client' +import { + fetchDictationSetup, + setDictationConfig, + downloadDictationModel, + deleteDictationModel +} from '../dictation/mobile-dictation-setup' +import type { VoiceSettingsOperations } from './voice-settings-operations' + +export function nativeVoiceSettingsOperations( + client: Pick +): VoiceSettingsOperations { + return { + load: () => fetchDictationSetup(client), + configure: (params) => setDictationConfig(client, params), + download: (modelId) => downloadDictationModel(client, modelId), + delete: (modelId) => deleteDictationModel(client, modelId) + } +} diff --git a/mobile/src/settings/notification-settings-operations.ts b/mobile/src/settings/notification-settings-operations.ts new file mode 100644 index 00000000000..669f8d4b074 --- /dev/null +++ b/mobile/src/settings/notification-settings-operations.ts @@ -0,0 +1,7 @@ +import type { NotificationPermissionState } from '../notifications/notification-permissions' + +export interface NotificationSettingsOperations { + permission(request?: boolean): Promise + preference(enabled?: boolean): Promise<{ enabled: boolean }> + openSettings(): Promise +} diff --git a/mobile/src/settings/notification-settings-screen.tsx b/mobile/src/settings/notification-settings-screen.tsx new file mode 100644 index 00000000000..7da9be8a813 --- /dev/null +++ b/mobile/src/settings/notification-settings-screen.tsx @@ -0,0 +1,196 @@ +import { useState, useCallback, useEffect } from 'react' +import { AppState, View, Text, StyleSheet, Pressable, Switch } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import { useFocusEffect } from 'expo-router' +import type { NotificationSettingsOperations } from './notification-settings-operations' +import { ChevronLeft } from 'lucide-react-native' +import { colors, spacing, typography } from '../theme/mobile-theme' +import type { NotificationPermissionState } from '../notifications/notification-permissions' + +const DEFAULT_PERMISSION_STATE: NotificationPermissionState = { + granted: false, + status: 'undetermined', + canAskAgain: true, + authorizationReflectsUserChoice: false +} + +export default function NotificationsScreen({ + operations, + onBack +}: { + operations: NotificationSettingsOperations + onBack: () => void +}) { + const insets = useSafeAreaInsets() + const [error, setError] = useState(null) + const [pushEnabled, setPushEnabled] = useState(false) + const [permissionState, setPermissionState] = useState(DEFAULT_PERMISSION_STATE) + + const refreshSettings = useCallback(async () => { + const [enabled, permission] = await Promise.all([ + operations.preference(), + operations.permission() + ]) + setPushEnabled(enabled.enabled) + setPermissionState(permission) + setError(null) + }, [operations]) + + useFocusEffect( + useCallback(() => { + void refreshSettings().catch(() => + setError('Could not load notification settings. Try again.') + ) + }, [refreshSettings]) + ) + + useEffect(() => { + const subscription = AppState.addEventListener('change', (state) => { + if (state === 'active') { + void refreshSettings().catch(() => + setError('Could not load notification settings. Try again.') + ) + } + }) + return () => subscription.remove() + }, [refreshSettings]) + + const togglePush = async (value: boolean) => { + setError(null) + try { + const permission = await operations.permission(value) + setPermissionState(permission) + const saved = await operations.preference(value && permission.granted) + setPushEnabled(saved.enabled) + } catch { + setError('Could not save notification settings. Try again.') + } + } + + const switchEnabled = pushEnabled && permissionState.granted + const notificationsBlocked = permissionState.status === 'denied' + const hint = notificationsBlocked + ? 'Notifications are disabled in system settings.' + : 'Get notified on this device when an agent needs your input or finishes a task.' + + return ( + + + + + + Notifications + + + {error && ( + + {error} + + )} + + + Agent notifications + void togglePush(v)} + trackColor={{ false: colors.bgRaised, true: colors.textSecondary }} + thumbColor={colors.textPrimary} + /> + + {hint} + {notificationsBlocked && ( + [ + styles.settingsButton, + pressed && styles.settingsButtonPressed + ]} + testID="notification-system-settings" + onPress={() => + void operations + .openSettings() + .catch(() => setError('Could not open system settings. Try again.')) + } + > + Open Settings + + )} + + + ) +} + +const styles = StyleSheet.create({ + container: { + flex: 1, + backgroundColor: colors.bgBase, + padding: spacing.lg + }, + topRow: { + flexDirection: 'row', + alignItems: 'center', + marginBottom: spacing.xl + }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { + fontSize: 20, + fontWeight: '700', + color: colors.textPrimary + }, + section: { + backgroundColor: colors.bgPanel, + borderRadius: 12, + overflow: 'hidden' + }, + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + rowLabel: { + flex: 1, + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + hint: { + fontSize: typography.metaSize, + color: colors.textMuted, + lineHeight: 18, + paddingHorizontal: spacing.md + 2, + paddingBottom: spacing.md + }, + settingsButton: { + alignSelf: 'flex-start', + marginHorizontal: spacing.md + 2, + marginBottom: spacing.md, + paddingVertical: spacing.xs, + paddingHorizontal: spacing.sm, + borderRadius: 8, + backgroundColor: colors.bgRaised + }, + settingsButtonPressed: { + opacity: 0.6 + }, + settingsButtonText: { + color: colors.textPrimary, + fontSize: typography.metaSize, + fontWeight: '600' + } +}) diff --git a/mobile/src/settings/pending-credential-cleanup-card.tsx b/mobile/src/settings/pending-credential-cleanup-card.tsx new file mode 100644 index 00000000000..48c668d12b8 --- /dev/null +++ b/mobile/src/settings/pending-credential-cleanup-card.tsx @@ -0,0 +1,159 @@ +import { useCallback, useRef, useState } from 'react' +import { ActivityIndicator, Pressable, StyleSheet, Text, View } from 'react-native' +import { useFocusEffect } from 'expo-router' +import { KeyRound } from 'lucide-react-native' +import { colors, radii, spacing, typography } from '../theme/mobile-theme' +import { + loadPendingHostCredentialCleanup, + subscribePendingHostCredentialCleanup +} from '../transport/host-credential-cleanup' +import { retryPendingHostCredentialCleanup } from '../transport/host-store' + +export function PendingCredentialCleanupCard() { + const [pendingCredentialIds, setPendingCredentialIds] = useState([]) + const [credentialStorageUnreadable, setCredentialStorageUnreadable] = useState(false) + const [retryingCredentialCleanup, setRetryingCredentialCleanup] = useState(false) + const [credentialRetryFailed, setCredentialRetryFailed] = useState(false) + const credentialRefreshGenerationRef = useRef(0) + + useFocusEffect( + useCallback(() => { + let active = true + setCredentialRetryFailed(false) + const refresh = () => { + const generation = ++credentialRefreshGenerationRef.current + void loadPendingHostCredentialCleanup().then((state) => { + if (active && generation === credentialRefreshGenerationRef.current) { + setPendingCredentialIds(state.ids) + setCredentialStorageUnreadable(state.storageUnreadable) + // Why: neutral copy once the queue is confirmed empty so a later + // pending set does not inherit a previous Retry failure message. + if (state.ids.length === 0 && !state.storageUnreadable) { + setCredentialRetryFailed(false) + } + } + }) + } + const unsubscribe = subscribePendingHostCredentialCleanup(refresh) + refresh() + return () => { + active = false + credentialRefreshGenerationRef.current += 1 + unsubscribe() + } + }, []) + ) + + const retryCredentialCleanup = useCallback(async () => { + if (retryingCredentialCleanup) { + return + } + setCredentialRetryFailed(false) + setRetryingCredentialCleanup(true) + try { + const result = await retryPendingHostCredentialCleanup() + setPendingCredentialIds(result.remainingIds) + setCredentialStorageUnreadable(result.storageUnreadable) + setCredentialRetryFailed(result.remainingIds.length > 0 || result.storageUnreadable) + } catch { + setCredentialRetryFailed(true) + } finally { + setRetryingCredentialCleanup(false) + } + }, [retryingCredentialCleanup]) + + const pendingCredentialCount = pendingCredentialIds.length + // Why: show the cleanup card whenever cleanup is pending OR the durable queue + // is unreadable — an unreadable queue can hide an orphaned token, so keep a + // retry affordance rather than a silently-empty (hidden) section. + if (pendingCredentialCount === 0 && !credentialStorageUnreadable) { + return null + } + + return ( + + + + + Pairing credential cleanup + + {credentialRetryFailed + ? "Cleanup still couldn't be confirmed. Try again later." + : pendingCredentialCount > 0 + ? `Couldn't confirm cleanup for ${pendingCredentialCount} credential${pendingCredentialCount === 1 ? '' : 's'} on this device.` + : "Couldn't check cleanup status on this device. Retry to be safe."} + + + [ + styles.retryButton, + pressed && !retryingCredentialCleanup && styles.rowPressed + ]} + onPress={() => void retryCredentialCleanup()} + > + {retryingCredentialCleanup ? ( + + ) : ( + Retry + )} + + + + ) +} + +const styles = StyleSheet.create({ + section: { + backgroundColor: colors.bgPanel, + borderRadius: 12, + overflow: 'hidden' + }, + sectionSpacer: { + marginTop: spacing.md + }, + rowPressed: { + backgroundColor: colors.bgRaised + }, + credentialCleanupRow: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + credentialCleanupCopy: { + flex: 1, + gap: spacing.xs + }, + credentialCleanupTitle: { + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + rowHint: { + fontSize: typography.metaSize, + color: colors.textSecondary, + lineHeight: 17 + }, + retryButton: { + width: 72, + height: 32, + borderRadius: radii.button, + backgroundColor: colors.bgRaised, + alignItems: 'center', + justifyContent: 'center' + }, + retryButtonText: { + fontSize: typography.metaSize, + fontWeight: '600', + color: colors.textPrimary + } +}) diff --git a/mobile/src/settings/settings-menu-screen.tsx b/mobile/src/settings/settings-menu-screen.tsx new file mode 100644 index 00000000000..76d2207227e --- /dev/null +++ b/mobile/src/settings/settings-menu-screen.tsx @@ -0,0 +1,42 @@ +import type { ReactNode } from 'react' +import { Shield, LifeBuoy } from 'lucide-react-native' +import { MobileSettingsFrame, MobileSettingsSection } from './mobile-settings-menu' +import { mobileSettingsMenuItems } from './mobile-settings-menu-items' + +export default function SettingsMenuScreen({ + push, + onBack, + openExternal, + children +}: { + push: (route: string) => void + onBack?: () => void + openExternal: (url: string) => Promise + children?: ReactNode +}) { + return ( + + + + {children} + + void openExternal('https://www.onorca.dev/privacy') + }, + { + label: 'Support', + icon: LifeBuoy, + external: true, + onPress: () => void openExternal('https://github.com/stablyai/orca/issues') + } + ]} + /> + + ) +} diff --git a/mobile/src/settings/settings-screen-state.test.tsx b/mobile/src/settings/settings-screen-state.test.tsx new file mode 100644 index 00000000000..ba790d50abc --- /dev/null +++ b/mobile/src/settings/settings-screen-state.test.tsx @@ -0,0 +1,210 @@ +import { createElement, useEffect } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import NotificationsScreen from './notification-settings-screen' +import VoiceSettingsScreen from './voice-settings-screen' +import type { VoiceSettingsOperations } from './voice-settings-operations' + +vi.mock('react-native', () => ({ + View: 'View', + Text: 'Text', + Pressable: 'Pressable', + Switch: 'Switch', + ScrollView: 'ScrollView', + ActivityIndicator: 'ActivityIndicator', + StyleSheet: { create: (value: unknown) => value, hairlineWidth: 1 }, + AppState: { addEventListener: () => ({ remove() {} }) } +})) +vi.mock('react-native-safe-area-context', () => ({ + useSafeAreaInsets: () => ({ top: 0, bottom: 0 }) +})) +vi.mock('expo-router', () => ({ + useFocusEffect: (callback: () => void) => useEffect(callback, [callback]) +})) +vi.mock('lucide-react-native', () => ({ ChevronLeft: 'Icon', ChevronRight: 'Icon' })) +vi.mock('../components/BottomDrawer', () => ({ BottomDrawer: () => null })) +vi.mock('../components/VoiceModelList', () => ({ VoiceModelList: () => null })) +vi.mock('../dictation/use-dictation-setup-poller', () => ({ + useDictationSetupPoller: ({ refresh }: { refresh: () => Promise }) => { + useEffect(() => { + void refresh() + }, [refresh]) + return refresh + } +})) +let renderer: ReactTestRenderer + +afterEach(() => { + act(() => renderer?.unmount()) +}) +describe('shared settings screen state', () => { + it('flips the Voice switch before the desktop replies and surfaces a rejected save', async () => { + const loaded = { + enabled: true, + dictationMode: 'toggle', + selectedModelId: '', + models: [] + } + let rejectConfigure: (error: Error) => void = () => {} + const operations = { + // Why: the reconcile read stays pending so the optimistic value and the + // rejection message are both observable, as they are on a slow desktop. + load: vi + .fn() + .mockResolvedValueOnce(loaded) + .mockImplementation(() => new Promise(() => {})), + configure: vi.fn().mockImplementation( + () => + new Promise((_resolve, reject) => { + rejectConfigure = reject + }) + ), + download: vi.fn(), + delete: vi.fn() + } as VoiceSettingsOperations + await act(async () => { + renderer = create( + createElement(VoiceSettingsScreen, { operations, focused: true, onBack: vi.fn() }) + ) + }) + const switchProps = () => renderer.root.findByProps({ testID: 'voice-enabled' }).props + expect(switchProps().value).toBe(true) + expect(switchProps().disabled).toBeUndefined() + + await act(async () => { + switchProps().onValueChange(false) + }) + // The switch moves on tap, before the desktop has answered. + expect(switchProps().value).toBe(false) + expect(switchProps().disabled).toBeUndefined() + expect(operations.configure).toHaveBeenCalledOnce() + + await act(async () => { + rejectConfigure(new Error('Desktop unavailable')) + await Promise.resolve() + }) + expect(JSON.stringify(renderer.toJSON())).toContain('Desktop unavailable') + expect(operations.load).toHaveBeenCalledTimes(2) + }) + it('shows the spinner, not the error card, while the first voice read is pending', async () => { + const operations = { + load: vi.fn().mockImplementation(() => new Promise(() => {})), + configure: vi.fn(), + download: vi.fn(), + delete: vi.fn() + } as VoiceSettingsOperations + await act(async () => { + renderer = create( + createElement(VoiceSettingsScreen, { operations, focused: true, onBack: vi.fn() }) + ) + }) + expect(JSON.stringify(renderer.toJSON())).not.toContain('Failed to load voice settings') + expect(renderer.root.findAllByType('ActivityIndicator')).toHaveLength(1) + }) + it('drops a poll that resolves after a toggle instead of clobbering it', async () => { + const loaded = { + enabled: true, + dictationMode: 'toggle', + selectedModelId: '', + models: [] + } + let resolveStalePoll: (value: typeof loaded) => void = () => {} + const shared = { + load: vi + .fn() + .mockResolvedValueOnce(loaded) + .mockImplementationOnce( + () => + new Promise((resolve) => { + resolveStalePoll = resolve + }) + ), + configure: vi.fn().mockImplementation(() => new Promise(() => {})), + download: vi.fn(), + delete: vi.fn() + } + const first = { ...shared } as VoiceSettingsOperations + await act(async () => { + renderer = create( + createElement(VoiceSettingsScreen, { operations: first, focused: true, onBack: vi.fn() }) + ) + }) + const switchProps = () => renderer.root.findByProps({ testID: 'voice-enabled' }).props + expect(switchProps().value).toBe(true) + + // A new operations identity restarts the poller, so a read is in flight below. + const second = { ...shared } as VoiceSettingsOperations + await act(async () => { + renderer.update( + createElement(VoiceSettingsScreen, { operations: second, focused: true, onBack: vi.fn() }) + ) + }) + expect(shared.load).toHaveBeenCalledTimes(2) + + await act(async () => { + switchProps().onValueChange(false) + }) + expect(switchProps().value).toBe(false) + + await act(async () => { + resolveStalePoll(loaded) + await Promise.resolve() + }) + // Without the request-epoch fence the stale read would flip the switch back on. + expect(switchProps().value).toBe(false) + }) + it('does not enable notifications after denied OS permission', async () => { + const denied = { + granted: false, + status: 'undetermined', + canAskAgain: true, + authorizationReflectsUserChoice: false + } + const operations = { + permission: vi.fn().mockResolvedValue(denied), + preference: vi.fn().mockResolvedValue({ enabled: false }), + openSettings: vi.fn() + } + await act(async () => { + renderer = create(createElement(NotificationsScreen, { operations, onBack: vi.fn() })) + }) + await act(async () => { + await renderer.root.findByProps({ testID: 'notification-enabled' }).props.onValueChange(true) + }) + expect(operations.permission).toHaveBeenLastCalledWith(true) + expect(operations.preference).toHaveBeenLastCalledWith(false) + expect(renderer.root.findByProps({ testID: 'notification-enabled' }).props.value).toBe(false) + expect(operations.openSettings).not.toHaveBeenCalled() + }) + it('keeps the notification switch live while the read is pending and after it fails', async () => { + let rejectPreference: (error: Error) => void = () => {} + const operations = { + permission: vi.fn().mockResolvedValue({ + granted: true, + status: 'granted', + canAskAgain: true, + authorizationReflectsUserChoice: true + }), + preference: vi.fn().mockImplementation( + () => + new Promise((_resolve, reject) => { + rejectPreference = reject + }) + ), + openSettings: vi.fn() + } + await act(async () => { + renderer = create(createElement(NotificationsScreen, { operations, onBack: vi.fn() })) + }) + const switchProps = () => renderer.root.findByProps({ testID: 'notification-enabled' }).props + // Base gates only on a denied OS permission, so the control is live from first paint. + expect(switchProps().disabled).toBe(false) + + await act(async () => { + rejectPreference(new Error('storage failed')) + await Promise.resolve() + }) + expect(switchProps().disabled).toBe(false) + expect(JSON.stringify(renderer.toJSON())).toContain('Could not load notification settings') + }) +}) diff --git a/mobile/src/settings/voice-settings-operations.ts b/mobile/src/settings/voice-settings-operations.ts new file mode 100644 index 00000000000..fc5fced208d --- /dev/null +++ b/mobile/src/settings/voice-settings-operations.ts @@ -0,0 +1,12 @@ +import type { MobileSpeechSetup } from '../dictation/mobile-dictation-setup' + +export interface VoiceSettingsOperations { + load(): Promise + configure(params: { + enabled?: boolean + modelId?: string + dictationMode?: 'toggle' | 'hold' + }): Promise + download(modelId: string): Promise + delete(modelId: string): Promise +} diff --git a/mobile/src/settings/voice-settings-poller-refresh.test.tsx b/mobile/src/settings/voice-settings-poller-refresh.test.tsx new file mode 100644 index 00000000000..ec3b4370e19 --- /dev/null +++ b/mobile/src/settings/voice-settings-poller-refresh.test.tsx @@ -0,0 +1,162 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import VoiceSettingsScreen from './voice-settings-screen' +import type { VoiceSettingsOperations } from './voice-settings-operations' + +// Why: the sibling state test stubs the poller, so it can only prove the screen's local +// machine. This one runs the real useDictationSetupPoller, whose refreshNow is gated on +// visible && foreground, to prove the rejected-save recovery actually reaches the wire. +vi.mock('react-native', () => ({ + View: 'View', + Text: 'Text', + Pressable: 'Pressable', + Switch: 'Switch', + ScrollView: 'ScrollView', + ActivityIndicator: 'ActivityIndicator', + StyleSheet: { create: (value: unknown) => value, hairlineWidth: 1 }, + AppState: { currentState: 'active', addEventListener: () => ({ remove() {} }) } +})) +vi.mock('react-native-safe-area-context', () => ({ + useSafeAreaInsets: () => ({ top: 0, bottom: 0 }) +})) +vi.mock('lucide-react-native', () => ({ ChevronLeft: 'Icon', ChevronRight: 'Icon' })) +vi.mock('../components/BottomDrawer', () => ({ BottomDrawer: () => null })) +vi.mock('../components/VoiceModelList', () => ({ VoiceModelList: () => null })) + +let renderer: ReactTestRenderer + +afterEach(() => { + act(() => renderer?.unmount()) +}) + +describe('voice settings poller integration', () => { + it('reaches the real poller refresh after a rejected save', async () => { + let rejectConfigure: (error: Error) => void = () => {} + const operations = { + load: vi.fn().mockResolvedValue({ + enabled: true, + dictationMode: 'toggle', + selectedModelId: '', + models: [] + }), + configure: vi.fn().mockImplementation( + () => + new Promise((_resolve, reject) => { + rejectConfigure = reject + }) + ), + download: vi.fn(), + delete: vi.fn() + } as VoiceSettingsOperations + await act(async () => { + renderer = create( + createElement(VoiceSettingsScreen, { operations, focused: true, onBack: vi.fn() }) + ) + }) + const loadsAfterMount = (operations.load as ReturnType).mock.calls.length + expect(loadsAfterMount).toBeGreaterThan(0) + + await act(async () => { + renderer.root.findByProps({ testID: 'voice-enabled' }).props.onValueChange(false) + }) + await act(async () => { + rejectConfigure(new Error('Desktop unavailable')) + await Promise.resolve() + await Promise.resolve() + }) + expect((operations.load as ReturnType).mock.calls.length).toBe( + loadsAfterMount + 1 + ) + }) + + it('shows the spinner, not the error card, when it mounts before focus lands', async () => { + const operations = { + load: vi.fn(), + configure: vi.fn(), + download: vi.fn(), + delete: vi.fn() + } as VoiceSettingsOperations + await act(async () => { + renderer = create( + createElement(VoiceSettingsScreen, { operations, focused: false, onBack: vi.fn() }) + ) + }) + // The poller is gated on focus, so no read runs and only the initial flag decides + // this paint. Base started it false and flashed the error card here. + expect(operations.load).not.toHaveBeenCalled() + expect(renderer.root.findAllByType('ActivityIndicator')).toHaveLength(1) + expect(JSON.stringify(renderer.toJSON())).not.toContain('Failed to load voice settings') + }) + + it('re-shows the spinner when a refocus retries a failed load', async () => { + const operations = { + load: vi + .fn() + .mockRejectedValueOnce(new Error('Desktop unavailable')) + .mockImplementationOnce(() => new Promise(() => {})), + configure: vi.fn(), + download: vi.fn(), + delete: vi.fn() + } as VoiceSettingsOperations + const onBack = vi.fn() + await act(async () => { + renderer = create(createElement(VoiceSettingsScreen, { operations, focused: true, onBack })) + }) + // The failed read leaves the error card up, exactly as base did. + expect(JSON.stringify(renderer.toJSON())).toContain('Desktop unavailable') + expect(renderer.root.findAllByType('ActivityIndicator')).toHaveLength(0) + + await act(async () => { + renderer.update(createElement(VoiceSettingsScreen, { operations, focused: false, onBack })) + }) + await act(async () => { + renderer.update(createElement(VoiceSettingsScreen, { operations, focused: true, onBack })) + }) + expect((operations.load as ReturnType).mock.calls.length).toBe(2) + // Base re-showed the spinner on re-entry; the stale error must not sit there during the retry. + expect(renderer.root.findAllByType('ActivityIndicator')).toHaveLength(1) + expect(JSON.stringify(renderer.toJSON())).not.toContain('Desktop unavailable') + }) + + it('drops the recovery read once the screen is no longer focused', async () => { + let rejectConfigure: (error: Error) => void = () => {} + const operations = { + load: vi.fn().mockResolvedValue({ + enabled: true, + dictationMode: 'toggle', + selectedModelId: '', + models: [] + }), + configure: vi.fn().mockImplementation( + () => + new Promise((_resolve, reject) => { + rejectConfigure = reject + }) + ), + download: vi.fn(), + delete: vi.fn() + } as VoiceSettingsOperations + const onBack = vi.fn() + await act(async () => { + renderer = create(createElement(VoiceSettingsScreen, { operations, focused: true, onBack })) + }) + await act(async () => { + renderer.root.findByProps({ testID: 'voice-enabled' }).props.onValueChange(false) + }) + await act(async () => { + renderer.update(createElement(VoiceSettingsScreen, { operations, focused: false, onBack })) + }) + const loadsBeforeRejection = (operations.load as ReturnType).mock.calls.length + + await act(async () => { + rejectConfigure(new Error('Desktop unavailable')) + await Promise.resolve() + await Promise.resolve() + }) + // The controller's visible && foreground gate is base's behaviour, preserved here. + expect((operations.load as ReturnType).mock.calls.length).toBe( + loadsBeforeRejection + ) + }) +}) diff --git a/mobile/src/settings/voice-settings-screen.tsx b/mobile/src/settings/voice-settings-screen.tsx new file mode 100644 index 00000000000..1e2c33c7a4c --- /dev/null +++ b/mobile/src/settings/voice-settings-screen.tsx @@ -0,0 +1,295 @@ +import { useCallback, useRef, useState } from 'react' +import { ActivityIndicator, Pressable, ScrollView, Switch, Text, View } from 'react-native' +import { useSafeAreaInsets } from 'react-native-safe-area-context' +import type { VoiceSettingsOperations } from './voice-settings-operations' +import { voiceSettingsStyles as styles } from './voice-settings-styles' +import { ChevronLeft, ChevronRight } from 'lucide-react-native' +import { colors, spacing } from '../theme/mobile-theme' +import { BottomDrawer } from '../components/BottomDrawer' +import { VoiceModelList } from '../components/VoiceModelList' +import { useDictationSetupPoller } from '../dictation/use-dictation-setup-poller' +import { + isModelInFlight, + type MobileSpeechModel, + type MobileSpeechSetup +} from '../dictation/mobile-dictation-setup' + +const POLL_INTERVAL_MS = 1500 + +const DICTATION_MODES = [ + { value: 'toggle', label: 'Toggle' }, + { value: 'hold', label: 'Hold' } +] as const + +type ModelBusyAction = { modelId: string; type: 'download' | 'select' | 'delete' } + +export default function VoiceSettingsScreen({ + operations, + focused, + onBack +}: { + operations: VoiceSettingsOperations | null + focused: boolean + onBack: () => void +}): React.JSX.Element { + const insets = useSafeAreaInsets() + const [setup, setSetup] = useState(null) + const [loading, setLoading] = useState(true) + const [error, setError] = useState(null) + const [busyAction, setBusyAction] = useState(null) + const requestEpoch = useRef(0) + const [modelDrawerOpen, setModelDrawerOpen] = useState(false) + const refresh = useCallback(async (): Promise => { + if (!operations) { + return false + } + const epoch = requestEpoch.current + // Own the spinner from the read that clears it, so a retry after a failed load shows + // the spinner again instead of the stale error card. Reads are serialised by + // DictationSetupPollController, so no in-flight read can clear another's flag. + setLoading(true) + try { + const next = await operations.load() + if (epoch !== requestEpoch.current) { + return undefined + } + setSetup(next) + setError(null) + return next.models.some(isModelInFlight) + } catch (err) { + setError(err instanceof Error ? err.message : 'Failed to load voice settings') + return undefined + } finally { + setLoading(false) + } + }, [operations]) + + const polling = setup?.models.some(isModelInFlight) ?? false + const refreshSetup = useDictationSetupPoller({ + visible: focused && operations !== null, + polling, + refresh, + intervalMs: POLL_INTERVAL_MS + }) + + const configure = useCallback( + async (params: Parameters[0]) => { + if (!operations) { + return + } + requestEpoch.current += 1 + setError(null) + // Optimistic flip so the control responds instantly; reconcile below. + const { enabled, dictationMode } = params + setSetup((prev) => + prev + ? { + ...prev, + ...(enabled === undefined ? {} : { enabled }), + ...(dictationMode === undefined ? {} : { dictationMode }) + } + : prev + ) + try { + setSetup(await operations.configure(params)) + } catch (err) { + setError(err instanceof Error ? err.message : 'Could not update') + void refreshSetup() + } + }, + [operations, refreshSetup] + ) + + const handleUseModel = useCallback( + async (model: MobileSpeechModel) => { + if (!operations) { + return + } + requestEpoch.current += 1 + setBusyAction({ modelId: model.id, type: 'select' }) + setError(null) + try { + setSetup(await operations.configure({ enabled: true, modelId: model.id })) + setModelDrawerOpen(false) + } catch (err) { + setError(err instanceof Error ? err.message : 'Could not select model') + } finally { + setBusyAction(null) + } + }, + [operations] + ) + + const handleDownload = useCallback( + async (model: MobileSpeechModel) => { + if (!operations) { + return + } + requestEpoch.current += 1 + setBusyAction({ modelId: model.id, type: 'download' }) + setError(null) + try { + await operations.download(model.id) + await refreshSetup() + } catch (err) { + setError(err instanceof Error ? err.message : 'Download failed') + } finally { + setBusyAction(null) + } + }, + [operations, refreshSetup] + ) + + const handleDelete = useCallback( + async (model: MobileSpeechModel) => { + if (!operations) { + return + } + const deletedSelectedModel = setup?.selectedModelId === model.id + requestEpoch.current += 1 + setBusyAction({ modelId: model.id, type: 'delete' }) + setError(null) + try { + setSetup(await operations.delete(model.id)) + if (deletedSelectedModel) { + setModelDrawerOpen(false) + } + } catch (err) { + setError(err instanceof Error ? err.message : 'Delete failed') + } finally { + setBusyAction(null) + } + }, + [operations, setup?.selectedModelId] + ) + + const enabled = setup?.enabled ?? false + const selectedModel = setup?.models.find((m) => m.id === setup.selectedModelId) + const selectedModelLabel = selectedModel?.label ?? 'None selected' + + return ( + + + + + + Voice + + + {!operations ? ( + + Connect to a desktop to manage voice settings. + + ) : loading && setup === null ? ( + + + + ) : setup === null ? ( + + {error ?? 'Failed to load voice settings.'} + + ) : ( + + DICTATION + + + + Enable Voice Dictation + + Dictate text into any focused pane on your desktop. + + + void configure({ enabled })} + trackColor={{ false: colors.bgRaised, true: colors.textSecondary }} + thumbColor={colors.textPrimary} + /> + + + + + + + Dictation Mode + + Toggle: press once to start, again to stop. Hold: dictate while held. + + + + {DICTATION_MODES.map((mode) => { + const active = setup.dictationMode === mode.value + return ( + void configure({ dictationMode: mode.value })} + style={[styles.segment, active && styles.segmentActive]} + > + + {mode.label} + + + ) + })} + + + + + SPEECH MODEL + + [ + styles.row, + !enabled && styles.disabled, + pressed && styles.rowPressed + ]} + disabled={!enabled} + testID="voice-model-picker" + onPress={() => setModelDrawerOpen(true)} + > + + Speech Model + + {selectedModelLabel} + + + + + + + {error ? {error} : null} + + )} + + setModelDrawerOpen(false)}> + Speech Model + {setup ? ( + void handleUseModel(m)} + onDownload={(m) => void handleDownload(m)} + onDelete={(m) => void handleDelete(m)} + /> + ) : null} + + + ) +} diff --git a/mobile/src/settings/voice-settings-styles.ts b/mobile/src/settings/voice-settings-styles.ts new file mode 100644 index 00000000000..19f3a8f06e8 --- /dev/null +++ b/mobile/src/settings/voice-settings-styles.ts @@ -0,0 +1,107 @@ +import { StyleSheet } from 'react-native' +import { colors, radii, spacing, typography } from '../theme/mobile-theme' + +export const voiceSettingsStyles = StyleSheet.create({ + container: { + flex: 1, + backgroundColor: colors.bgBase, + paddingHorizontal: spacing.lg + }, + topRow: { + flexDirection: 'row', + alignItems: 'center', + marginTop: spacing.sm, + marginBottom: spacing.lg + }, + backButton: { + width: 36, + height: 36, + borderRadius: 18, + alignItems: 'center', + justifyContent: 'center', + marginRight: spacing.sm + }, + heading: { + fontSize: 20, + fontWeight: '700', + color: colors.textPrimary + }, + scrollContent: { + paddingBottom: spacing.xl + }, + loading: { paddingVertical: spacing.xl, alignItems: 'center' }, + groupHeading: { + fontSize: 11, + fontWeight: '600', + color: colors.textMuted, + letterSpacing: 0.5, + marginBottom: spacing.xs, + paddingHorizontal: spacing.xs + }, + section: { + backgroundColor: colors.bgPanel, + borderRadius: radii.card, + overflow: 'hidden' + }, + sectionTopGap: { marginTop: spacing.sm }, + inputGroupGap: { marginTop: spacing.xl }, + disabled: { opacity: 0.5 }, + emptyText: { + fontSize: typography.bodySize, + color: colors.textSecondary, + padding: spacing.md + }, + errorText: { + fontSize: typography.bodySize, + color: colors.statusRed, + padding: spacing.md + }, + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm + 2, + paddingVertical: spacing.md, + paddingHorizontal: spacing.md + 2 + }, + rowPressed: { backgroundColor: colors.bgRaised }, + rowContent: { flex: 1 }, + rowLabel: { + fontSize: typography.bodySize, + fontWeight: '500', + color: colors.textPrimary + }, + drawerTitle: { + fontSize: typography.bodySize, + fontWeight: '700', + color: colors.textPrimary, + paddingHorizontal: spacing.md + 2, + paddingTop: spacing.sm, + paddingBottom: spacing.xs + }, + rowSublabel: { + fontSize: typography.bodySize - 2, + color: colors.textSecondary, + marginTop: 2 + }, + separator: { + height: StyleSheet.hairlineWidth, + backgroundColor: colors.borderSubtle, + marginHorizontal: spacing.md + }, + segmented: { + flexDirection: 'row', + alignItems: 'center', + backgroundColor: colors.bgBase, + borderRadius: radii.button, + padding: 2 + }, + segment: { + paddingHorizontal: spacing.md, + paddingVertical: 6, + borderRadius: radii.button - 1 + }, + segmentActive: { backgroundColor: colors.bgRaised }, + segmentText: { fontSize: typography.metaSize, color: colors.textSecondary, fontWeight: '600' }, + segmentTextActive: { color: colors.textPrimary }, + error: { color: colors.statusRed, fontSize: typography.metaSize, marginTop: spacing.md } +}) diff --git a/mobile/src/terminal/terminal-viewport-refit-state.ts b/mobile/src/terminal/terminal-viewport-refit-state.ts index dfa4daadf4c..766345d8c73 100644 --- a/mobile/src/terminal/terminal-viewport-refit-state.ts +++ b/mobile/src/terminal/terminal-viewport-refit-state.ts @@ -1,4 +1,8 @@ import type { RpcResponse } from '../transport/types' +import { + isMethodNotFoundRefusal, + rpcObjectResultOrNull +} from '../transport/rpc-acceptance-policies' export type TerminalUpdateViewportCapability = 'unknown' | 'supported' | 'unsupported' @@ -14,17 +18,11 @@ export type TerminalViewportRefitTargetState = { } export function isTerminalUpdateViewportUpdated(response: RpcResponse): boolean { - if (!response.ok || typeof response.result !== 'object' || response.result == null) { - return false - } - return (response.result as { updated?: unknown }).updated === true + return rpcObjectResultOrNull(response)?.updated === true } export function isTerminalUpdateViewportApplied(response: RpcResponse): boolean { - if (!response.ok || typeof response.result !== 'object' || response.result == null) { - return false - } - return (response.result as { applied?: unknown }).applied === true + return rpcObjectResultOrNull(response)?.applied === true } export function resolveTerminalUpdateViewportCapability( @@ -33,7 +31,7 @@ export function resolveTerminalUpdateViewportCapability( if (response.ok) { return 'supported' } - return response.error.code === 'method_not_found' ? 'unsupported' : 'unknown' + return isMethodNotFoundRefusal(response) ? 'unsupported' : 'unknown' } // Why: defer height refits while typing, then coalesce every skipped layout diff --git a/mobile/src/transport/mobile-relay-direct-upgrade.ts b/mobile/src/transport/mobile-relay-direct-upgrade.ts index b804d6fb610..ed019262bca 100644 --- a/mobile/src/transport/mobile-relay-direct-upgrade.ts +++ b/mobile/src/transport/mobile-relay-direct-upgrade.ts @@ -21,7 +21,11 @@ import { type MobileRelayDirectUpgradeJournal } from './mobile-relay-direct-upgrade-journal' import type { RpcClient } from './rpc-client' -import type { HostProfile, RpcResponse } from './types' +import type { HostProfile } from './types' +import { + isMethodNotFoundRefusal, + requireRpcResultOrThrowCodedError +} from './rpc-acceptance-policies' export type MobileRelayDirectUpgradeResult = { host: HostProfile @@ -79,11 +83,13 @@ export async function upgradeDirectMobileRelay(args: { reqId: journal.reqId, newResumeTokenHash: journal.pendingResumeTokenHash }) - if (isMethodNotFound(provisionResponse)) { + if (isMethodNotFoundRefusal(provisionResponse)) { await dependencies.clearJournal(args.host.id) return null } - const installed = DeviceCredentialInstalledSchema.parse(requireSuccess(provisionResponse)) + const installed = DeviceCredentialInstalledSchema.parse( + requireRpcResultOrThrowCodedError(provisionResponse) + ) assertDirectInstall(journal, installed) const reconciled = await getEndpoints(args.client, journal.reqId) if (reconciled === 'method-not-found') { @@ -136,10 +142,10 @@ async function getEndpoints( installReqId: string ): Promise { const response = await client.sendRequest('pairing.getEndpoints', { installReqId }) - if (isMethodNotFound(response)) { + if (isMethodNotFoundRefusal(response)) { return 'method-not-found' } - return PairingGetEndpointsResultSchema.parse(requireSuccess(response)) + return PairingGetEndpointsResultSchema.parse(requireRpcResultOrThrowCodedError(response)) } function assertDirectInstall( @@ -162,14 +168,3 @@ function assertCommitted( throw new Error('relay credential install was not authoritatively reconciled') } } - -function requireSuccess(response: RpcResponse): unknown { - if (!response.ok) { - throw new Error(`${response.error.code}: ${response.error.message}`) - } - return response.result -} - -function isMethodNotFound(response: RpcResponse): boolean { - return !response.ok && response.error.code === 'method_not_found' -} diff --git a/mobile/src/transport/mobile-relay-pairing-recovery.ts b/mobile/src/transport/mobile-relay-pairing-recovery.ts index 26dc08620de..b37a18a06c2 100644 --- a/mobile/src/transport/mobile-relay-pairing-recovery.ts +++ b/mobile/src/transport/mobile-relay-pairing-recovery.ts @@ -25,7 +25,8 @@ import { type PairingCandidateClient } from './mobile-relay-physical-client' import { createRecoveringPairingRelayCandidate } from './pairing-relay-candidate' -import type { HostProfile, RpcResponse } from './types' +import type { HostProfile } from './types' +import { requireRpcResultOrThrowCodedError } from './rpc-acceptance-policies' export type MobileRelayPairingRecoveryResult = 'none' | 'recovered' | 'deferred' | 'abandoned' @@ -128,7 +129,7 @@ async function runRecovery( if (credential.kind === 'invite' && endpoints.installStatus?.state === 'not-found') { journal = await transitionToInviteAuthorization(journal, dependencies) const installed = DeviceCredentialInstalledSchema.parse( - requireSuccess( + requireRpcResultOrThrowCodedError( await client.sendRequest('pairing.provisionRelay', { reqId: journal.metadata.installReqId, newResumeTokenHash: journal.metadata.pendingResumeTokenHash @@ -220,7 +221,7 @@ async function getRecoveryStatus( kind: 'resume' | 'invite' ) { return PairingGetEndpointsResultSchema.parse( - requireSuccess( + requireRpcResultOrThrowCodedError( await client.sendRequest('pairing.getEndpoints', { installReqId: journal.metadata.installReqId, ...(kind === 'resume' ? { resumeConfirmReqId: journal.metadata.resumeConfirmReqId } : {}) @@ -294,13 +295,6 @@ function pairingRelay(journal: MobileRelayPairingJournal): PairingRelay { return { ...journal.metadata.relay, inviteToken: journal.secrets.inviteToken } } -function requireSuccess(response: RpcResponse): unknown { - if (!response.ok) { - throw new Error(`${response.error.code}: ${response.error.message}`) - } - return response.result -} - function assertCommitted( endpoints: ReturnType, installed: DeviceCredentialInstalled diff --git a/mobile/src/transport/mobile-relay-rpc-streams.ts b/mobile/src/transport/mobile-relay-rpc-streams.ts index abb2c12564b..3af5b2b3f5e 100644 --- a/mobile/src/transport/mobile-relay-rpc-streams.ts +++ b/mobile/src/transport/mobile-relay-rpc-streams.ts @@ -9,6 +9,7 @@ import { updateTerminalSubscriptionViewport } from './rpc-client-terminal-subscription' import { buildReadyStreamUnsubscribe } from './rpc-client-server-subscription' +import { isStreamingOpenerReply } from './rpc-acceptance-policies' import type { RpcClient } from './rpc-client' import type { RpcResponse, RpcSuccess } from './types' @@ -119,7 +120,7 @@ export class MobileRelayRpcStreams { } } } - if (response.ok && response.streaming !== true) { + if (response.ok && !isStreamingOpenerReply(response)) { this.cancelledSubscriptions.delete(response.id) } return true diff --git a/mobile/src/transport/mobile-runtime-client-capabilities.test.ts b/mobile/src/transport/mobile-runtime-client-capabilities.test.ts index 7a9b2d841ce..bdd1330de1d 100644 --- a/mobile/src/transport/mobile-runtime-client-capabilities.test.ts +++ b/mobile/src/transport/mobile-runtime-client-capabilities.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from 'vitest' import { + AGENT_SESSION_TURN_ITEM_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY @@ -13,12 +14,13 @@ const HOST_CAPABILITY_LIMIT = 64 const HOST_CAPABILITY_NAME_LIMIT = 128 describe('mobile runtime client capabilities', () => { - it('advertises structured agent sessions including the Claude lane', () => { + it('advertises structured agent sessions, the Claude lane, and the turn item', () => { expect(MOBILE_RUNTIME_CLIENT_CAPABILITIES).toEqual( expect.arrayContaining([ STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, - CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY ]) ) }) diff --git a/mobile/src/transport/mobile-runtime-client-capabilities.ts b/mobile/src/transport/mobile-runtime-client-capabilities.ts index 29a9e93b527..30b627a9ef3 100644 --- a/mobile/src/transport/mobile-runtime-client-capabilities.ts +++ b/mobile/src/transport/mobile-runtime-client-capabilities.ts @@ -1,4 +1,6 @@ import { + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY @@ -7,8 +9,11 @@ import { remoteRuntimeClientCapabilities } from '../../../src/shared/remote-runt export const MOBILE_RUNTIME_CLIENT_CAPABILITIES = remoteRuntimeClientCapabilities([ STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, - CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + // Opts into the typed turn record; without it the host sends the legacy status carrier. + AGENT_SESSION_TURN_ITEM_CAPABILITY ]) export const MOBILE_RUNTIME_CLIENT_CAPABILITY_UPDATE_METHOD = diff --git a/mobile/src/transport/pre-profile-pairing-coordinator.ts b/mobile/src/transport/pre-profile-pairing-coordinator.ts index a9b04320471..eb1f524334e 100644 --- a/mobile/src/transport/pre-profile-pairing-coordinator.ts +++ b/mobile/src/transport/pre-profile-pairing-coordinator.ts @@ -7,7 +7,11 @@ import { } from '../../../src/shared/mobile-relay-credential-contract' import { connect, type ConnectOptions } from './rpc-client' import { resolvePairingHostIdentity, saveHost } from './host-store' -import type { HostProfile, PairingOffer, RpcResponse } from './types' +import type { HostProfile, PairingOffer } from './types' +import { + isMethodNotFoundRefusal, + requireRpcResultOrThrowCodedError +} from './rpc-acceptance-policies' import { createMobileRelayPairingJournal, type MobileRelayPairingJournal @@ -219,7 +223,7 @@ async function runPairing( reqId: journal.metadata.installReqId, newResumeTokenHash: journal.metadata.pendingResumeTokenHash }) - if (isMethodNotFound(provision)) { + if (isMethodNotFoundRefusal(provision)) { if (winner.path !== 'direct') { throw new Error('relay pairing RPC unavailable after relay path authentication') } @@ -227,9 +231,11 @@ async function runPairing( await dependencies.clearJournal(journal.metadata.journalId) return { hostId } } - const installed = DeviceCredentialInstalledSchema.parse(requireSuccess(provision)) + const installed = DeviceCredentialInstalledSchema.parse( + requireRpcResultOrThrowCodedError(provision) + ) const endpoints = PairingGetEndpointsResultSchema.parse( - requireSuccess( + requireRpcResultOrThrowCodedError( await winner.client.sendRequest('pairing.getEndpoints', { installReqId: journal.metadata.installReqId }) @@ -283,17 +289,6 @@ function relayWebSocketUrl(relay: MobileRelayEndpoint): string { return url.toString() } -function requireSuccess(response: RpcResponse): unknown { - if (!response.ok) { - throw new Error(`${response.error.code}: ${response.error.message}`) - } - return response.result -} - -function isMethodNotFound(response: RpcResponse): boolean { - return !response.ok && response.error.code === 'method_not_found' -} - function assertCommittedInstall( status: | { state: 'not-found' } diff --git a/mobile/src/transport/rpc-acceptance-policies.test.ts b/mobile/src/transport/rpc-acceptance-policies.test.ts new file mode 100644 index 00000000000..2f2abc90e16 --- /dev/null +++ b/mobile/src/transport/rpc-acceptance-policies.test.ts @@ -0,0 +1,168 @@ +import { describe, expect, it } from 'vitest' +import type { RpcResponse } from './types' +import { + isMethodNotFoundRefusal, + isStreamingOpenerReply, + requireRpcResultOrThrowCodedError, + rpcObjectResultOrNull +} from './rpc-acceptance-policies' + +const meta = { runtimeId: 'runtime-1' } + +function success(result: unknown, streaming?: true): RpcResponse { + return { id: 'rpc-1', ok: true, result, _meta: meta, ...(streaming ? { streaming } : {}) } +} + +function refusal(code: string, message = 'Nope'): RpcResponse { + return { id: 'rpc-1', ok: false, error: { code, message }, _meta: meta } +} + +/** Every result partition a policy has to survive. */ +const resultPartitions: [string, unknown][] = [ + ['object result', { value: 1 }], + ['undefined result', undefined], + ['null result', null], + ['empty object result', {}], + ['numeric result', 7], + ['zero result', 0], + ['string result', 'done'], + ['empty string result', ''], + ['boolean result', false], + ['array result', [1, 2]], + ['empty array result', []] +] + +describe('requireRpcResultOrThrowCodedError', () => { + it.each(resultPartitions)('returns the %s untouched', (_label, result) => { + expect(requireRpcResultOrThrowCodedError(success(result))).toEqual(result) + }) + + it('returns an absent result field as undefined', () => { + const response = { id: 'rpc-1', ok: true, _meta: meta } as unknown as RpcResponse + expect(requireRpcResultOrThrowCodedError(response)).toBeUndefined() + }) + + it('throws a code-prefixed message on refusal', () => { + expect(() => requireRpcResultOrThrowCodedError(refusal('method_not_found', 'no such'))).toThrow( + 'method_not_found: no such' + ) + }) + + it('throws even when the refusal carries an empty message', () => { + expect(() => requireRpcResultOrThrowCodedError(refusal('runtime_error', ''))).toThrow( + 'runtime_error: ' + ) + }) +}) + +describe('rpcObjectResultOrNull', () => { + it('accepts a plain object result', () => { + expect(rpcObjectResultOrNull(success({ value: 1 }))).toEqual({ value: 1 }) + }) + + it('accepts an empty object result', () => { + expect(rpcObjectResultOrNull(success({}))).toEqual({}) + }) + + it('accepts an array result, because arrays are objects', () => { + expect(rpcObjectResultOrNull(success([1, 2]))).toEqual([1, 2]) + }) + + it.each([ + ['null', null], + ['undefined', undefined], + ['numeric', 7], + ['zero', 0], + ['string', 'done'], + ['empty string', ''], + ['boolean', false], + ['true', true] + ])('refuses a %s result', (_label, result) => { + expect(rpcObjectResultOrNull(success(result))).toBeNull() + }) + + it('refuses a refusal regardless of its code', () => { + expect(rpcObjectResultOrNull(refusal('method_not_found'))).toBeNull() + expect(rpcObjectResultOrNull(refusal('runtime_error'))).toBeNull() + }) +}) + +// A refusal that illegally carries success-shaped fields: without the `ok` check each of these +// would read the stray field and answer as if the call had succeeded. +describe('a refusal carrying stray success fields', () => { + const strayRefusal = { + id: 'rpc-1', + ok: false, + error: { code: 'method_not_found', message: 'Nope' }, + result: { value: 1 }, + streaming: true, + _meta: meta + } as unknown as RpcResponse + + it('yields null rather than the stray result', () => { + expect(rpcObjectResultOrNull(strayRefusal)).toBeNull() + }) + + it('is still recognised as method-not-found', () => { + expect(isMethodNotFoundRefusal(strayRefusal)).toBe(true) + }) + + // The mirror case: a success carrying a stray error must not read as a refusal. + it('does not read a success carrying a stray error as a refusal', () => { + const straySuccess = { + id: 'rpc-1', + ok: true, + result: { value: 1 }, + error: { code: 'method_not_found', message: 'Nope' }, + _meta: meta + } as unknown as RpcResponse + expect(isMethodNotFoundRefusal(straySuccess)).toBe(false) + }) +}) + +describe('isMethodNotFoundRefusal', () => { + it('matches only the method_not_found code', () => { + expect(isMethodNotFoundRefusal(refusal('method_not_found'))).toBe(true) + expect(isMethodNotFoundRefusal(refusal('runtime_error'))).toBe(false) + expect(isMethodNotFoundRefusal(refusal('METHOD_NOT_FOUND'))).toBe(false) + }) + + it('never matches a success, including one with a null result', () => { + expect(isMethodNotFoundRefusal(success(null))).toBe(false) + expect(isMethodNotFoundRefusal(success({ code: 'method_not_found' }))).toBe(false) + }) +}) + +describe('isStreamingOpenerReply', () => { + it('accepts a success flagged streaming', () => { + expect(isStreamingOpenerReply(success({ subscriptionId: 's1' }, true))).toBe(true) + }) + + it('refuses a success with no streaming flag', () => { + expect(isStreamingOpenerReply(success({ subscriptionId: 's1' }))).toBe(false) + }) + + // A truthy non-boolean off the wire must not open a stream: the registry would route it to + // handleStreamingResponse and wait for frames that never come. + it.each([['yes'], [1], [{}]])('refuses a truthy non-boolean streaming flag %j', (flag) => { + const response = { + id: 'rpc-1', + ok: true, + result: { subscriptionId: 's1' }, + streaming: flag, + _meta: meta + } as unknown as RpcResponse + expect(isStreamingOpenerReply(response)).toBe(false) + }) + + it('refuses a refusal even when it carries a streaming flag', () => { + const response = { + id: 'rpc-1', + ok: false, + error: { code: 'runtime_error', message: 'Nope' }, + streaming: true, + _meta: meta + } as unknown as RpcResponse + expect(isStreamingOpenerReply(response)).toBe(false) + }) +}) diff --git a/mobile/src/transport/rpc-acceptance-policies.ts b/mobile/src/transport/rpc-acceptance-policies.ts new file mode 100644 index 00000000000..742d94aafa7 --- /dev/null +++ b/mobile/src/transport/rpc-acceptance-policies.ts @@ -0,0 +1,33 @@ +import type { RpcResponse, RpcSuccess } from './types' + +// Named acceptance policies for RPC replies. Call sites used to hand-roll these +// predicates and did not agree with each other; each policy here preserves one +// call site's existing acceptance exactly. Do not merge two policies without +// proving every caller of both tolerates the wider or narrower set. + +/** Throws `code: message` on refusal. Diagnostic text; not user-facing. */ +export function requireRpcResultOrThrowCodedError(response: RpcResponse): unknown { + if (!response.ok) { + throw new Error(`${response.error.code}: ${response.error.message}`) + } + return response.result +} + +/** Accepts only a success whose result is a non-null object. Arrays qualify. */ +export function rpcObjectResultOrNull(response: RpcResponse): Record | null { + if (!response.ok || typeof response.result !== 'object' || response.result === null) { + return null + } + return response.result as Record +} + +export function isMethodNotFoundRefusal(response: RpcResponse): boolean { + return !response.ok && response.error.code === 'method_not_found' +} + +/** A success that opened a stream rather than delivering a terminal result. */ +export function isStreamingOpenerReply( + response: RpcResponse +): response is RpcSuccess & { streaming: true } { + return response.ok && response.streaming === true +} diff --git a/mobile/src/transport/rpc-client-capabilities.test.ts b/mobile/src/transport/rpc-client-capabilities.test.ts index 41bb4091a0b..cc670ab8a41 100644 --- a/mobile/src/transport/rpc-client-capabilities.test.ts +++ b/mobile/src/transport/rpc-client-capabilities.test.ts @@ -92,6 +92,7 @@ describe('mobile rpc-client capabilities', () => { expect(capabilityRequest.params).toMatchObject({ clientCapabilities: expect.arrayContaining([ 'agent-session.structured.v1', + 'agent-session.pending-send-result.v1', 'agent-session.structured.claude.v1' ]) }) diff --git a/mobile/src/transport/rpc-client-stream-registry.ts b/mobile/src/transport/rpc-client-stream-registry.ts index ac20acdecc0..68e50a6e6dd 100644 --- a/mobile/src/transport/rpc-client-stream-registry.ts +++ b/mobile/src/transport/rpc-client-stream-registry.ts @@ -8,6 +8,7 @@ import { updateTerminalSubscriptionViewport } from './rpc-client-terminal-subscription' import { buildReadyStreamUnsubscribe } from './rpc-client-server-subscription' +import { isStreamingOpenerReply } from './rpc-acceptance-policies' import { isStreamingSubscriptionReadyResult, isTerminalSubscribedResult @@ -112,7 +113,7 @@ export class RpcClientStreamRegistry { } handleResponse(response: RpcResponse): boolean { - if (response.ok && response.streaming === true) { + if (isStreamingOpenerReply(response)) { this.handleStreamingResponse(response) return true } diff --git a/mobile/src/transport/rpc-params-contract.ts b/mobile/src/transport/rpc-params-contract.ts new file mode 100644 index 00000000000..fcf3303e965 --- /dev/null +++ b/mobile/src/transport/rpc-params-contract.ts @@ -0,0 +1,8 @@ +// Why: mobile's only entry to the host's params contract, and type-only on purpose. +// The schemas behind these types must never reach the bundle: requiredString is +// z.unknown().transform(...), so a client-side parse coerces a non-string to '' +// instead of rejecting it, silently changing the bytes on the wire. +export type { + RpcMethodName, + RpcParams +} from '../../../src/shared/rpc-contract/rpc-params-catalog.generated' diff --git a/package.json b/package.json index 0536f4fd606..4f03d793aa6 100644 --- a/package.json +++ b/package.json @@ -13,7 +13,7 @@ "audit:perf": "oxlint --config config/oxlint-performance-audit.json --format json src", "test:perf:contracts": "vitest run --config config/vitest.performance.config.ts", "format": "oxfmt --write .", - "lint": "oxlint && pnpm run audit:code-quality:native && pnpm run audit:code-quality:type-aware && pnpm run check:reliability-gates && pnpm run check:max-lines-ratchet && pnpm run check:ts-nocheck-ratchet && pnpm run check:runtime-electron-ratchet && pnpm run verify:bundled-skill-guides && pnpm run verify:skill-bundle-manifest && pnpm run verify:localization-catalog && pnpm run verify:localization-runtime-catalog && pnpm run verify:localization-extraction && pnpm run verify:localization-coverage", + "lint": "oxlint && pnpm run audit:code-quality:native && pnpm run audit:code-quality:type-aware && pnpm run check:reliability-gates && pnpm run check:max-lines-ratchet && pnpm run check:ts-nocheck-ratchet && pnpm run check:runtime-electron-ratchet && pnpm run verify:rpc-params-catalog && pnpm run verify:bundled-skill-guides && pnpm run verify:skill-bundle-manifest && pnpm run verify:localization-catalog && pnpm run verify:localization-runtime-catalog && pnpm run verify:localization-extraction && pnpm run verify:localization-coverage", "audit:code-quality": "pnpm run audit:code-quality:native && pnpm run audit:code-quality:type-aware && pnpm run audit:react-doctor", "audit:code-quality:native": "oxlint --config config/oxlint-code-quality-native-plugins.json src config tests mobile --deny-warnings", "audit:code-quality:type-aware": "oxlint --type-aware --config config/oxlint-code-quality-type-aware.json src config tests --deny-warnings", @@ -29,6 +29,7 @@ "test": "node config/scripts/ensure-native-runtime.mjs --runtime=node && vitest run --config config/vitest.config.ts", "test:skill-sharing:release": "vitest run --config config/vitest.config.ts src/main/skills src/main/runtime/rpc/methods/skills.test.ts src/relay/skill-install-handler.test.ts src/shared/skill-bundle-install-contract.test.ts src/shared/skill-install-contract.test.ts src/shared/skill-install-failure.test.ts src/shared/skill-package-manifest.test.ts", "test:repro:remote-agent-session": "pnpm run build:cli && pnpm run build:electron-vite && node config/scripts/remote-agent-session-authority-repro.mjs", + "capture:agent-transcript": "node config/scripts/ensure-native-runtime.mjs --runtime=node && node config/scripts/capture-agent-pty-transcript.mjs", "check:reliability-gates": "node config/scripts/check-reliability-gates.mjs", "check:max-lines-ratchet": "node config/scripts/check-max-lines-ratchet.mjs", "check:ts-nocheck-ratchet": "node config/scripts/check-ts-nocheck-ratchet.mjs", @@ -38,6 +39,8 @@ "smoke:orcad-terminal": "node config/scripts/ensure-native-runtime.mjs --runtime=node && pnpm run build:cli && pnpm run build:orcad && node config/scripts/runtime-serve-terminal-smoke.mjs --target orcad", "smoke:serve-terminal": "node config/scripts/runtime-serve-terminal-smoke.mjs", "check:feature-wall-assets": "node config/scripts/check-feature-wall-assets.mjs", + "generate:rpc-params-catalog": "node config/scripts/generate-rpc-params-catalog.mjs", + "verify:rpc-params-catalog": "node config/scripts/generate-rpc-params-catalog.mjs --check", "generate:bundled-skill-guides": "node config/scripts/generate-bundled-skill-guides.mjs --write", "verify:bundled-skill-guides": "node config/scripts/generate-bundled-skill-guides.mjs --check", "generate:skill-bundle-manifest": "node config/scripts/generate-skill-bundle-manifest.mjs --write", diff --git a/skill-guides/orchestration/references/messaging-and-gates.md b/skill-guides/orchestration/references/messaging-and-gates.md index b9e4371251e..573ca90e5e7 100644 --- a/skill-guides/orchestration/references/messaging-and-gates.md +++ b/skill-guides/orchestration/references/messaging-and-gates.md @@ -40,8 +40,16 @@ capability arguments in its preamble. `check` is the exception: it identifies its caller with `--terminal`, never `--from`. Group addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`, -`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Use them only for -intentional fan-out status or questions. `worker_done`, heartbeat, and other +`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Every group but +`@worktree:` means the live Dispatches of the sender's own Run. Mail goes +to each `dispatch:` mailbox, except a worker coordinating a child Run +receives it in that `run:` mailbox. A sender bound to no Run is refused; +`--run` must match the group audience and never grants membership. +A Run group excludes its owning coordinator; a worker raising a blocker sends +to `run:`. A worker that created its own Run addresses that Run's workers, +not its siblings. `@worktree:` reaches matching workspace terminals, +including coordinators. Use groups only for intentional fan-out status or +questions. `worker_done`, heartbeat, and other Dispatch lifecycle messages never target groups. ## Questions and gates diff --git a/src/cli/bundled-skill-guides.ts b/src/cli/bundled-skill-guides.ts index 47b5559f01b..2f1efdcd671 100644 --- a/src/cli/bundled-skill-guides.ts +++ b/src/cli/bundled-skill-guides.ts @@ -69,7 +69,7 @@ const ORCA_PER_WORKSPACE_ENV_WINDOWS_SCRIPTS_REFERENCE_MARKDOWN = "# Windows loc const ORCHESTRATION_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nA `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting\nwith `check --wait`. Absence never earns an argv; settlement and pending work still do.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n" // oxfmt-ignore -const ORCHESTRATION_FULL_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nA `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting\nwith `check --wait`. Absence never earns an argv; settlement and pending work still do.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n\n---\n\n# Bundled references\n\nThese references belong to the version-matched guide above. Read only the documents named by its action gates.\n\n\n\n# Coordinator loop\n\nLoad this reference for expanded DAG waves, per-invocation launch preferences,\nsame-terminal reuse, or review ownership. The compact guide remains the source\nof truth for the loop order and completion boundary.\n\n## Ready waves\n\nCreate independent Tasks before the first wait. Encode only real dependencies,\nthen use the ready view as external memory:\n\n```text\nORCA orchestration task-create --spec \"\" --deps --json\nORCA orchestration task-list --ready --brief --json\n```\n\n`--brief` collapses whitespace and caps echoed specs at 160 characters;\n`spec_truncated` identifies shortened rows. Omit it when full specs are needed or\nwhen an older CLI rejects the flag. A nested worker must respect\n`nested_worker_depth_exceeded`; creating another Run does not reset depth.\n\n## Launch preferences\n\nFor a fresh Claude, Codex, or Cursor terminal, `--model` accepts an opaque\nprovider model ID. Pass it only when the user named a model; otherwise omit it\nso the worker inherits the user's configured agent default. Add `--effort` only\nwhen that model supports it:\n\n```text\nORCA orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`; neither option combines with `--terminal`. A\nconnected worker server must advertise launch-preference support before Orca\nforwards either field. Compare `launch.requested` with `launch.effective`; never\nclaim a model or effort from requested arguments alone.\n\n## Reuse after settlement\n\nChoose the terminal's next owner before acknowledging the Delivery. When the\nsame exact agent has immediate follow-up work, recover the proven handle and\ntransfer cleanup ownership to the new Dispatch:\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-start --task --terminal --json\n```\n\nOtherwise explicitly retain or release the settled worker. Do not leave it live\nonly to inspect output; archived output remains available through `worker-read`.\n\n## Review ownership\n\nA review-only `worker_done` authorizes synthesis of findings, not coordinator\nfile edits. Dispatch or hand off fixes unless the user explicitly assigned them\nto the coordinator. If the user's plan names a next owner, post-review fixes and\nPR preparation remain with that owner; the coordinator routes and synthesizes.\n\n\n\n# Legacy contract migration\n\nLoad this reference only for an authority label, adopted Run, compatibility or\nrecovery receipt, or explicit legacy takeover. A newly created attempt always\nuses the current grammar.\n\n## Authority labels\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported\n command printed with the message, using the same selected executable and\n arguments supplied by the original prompt.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded,\n at-least-once cutover replay. Process it idempotently and acknowledge only\n through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or\n lifecycle mutation.\n- An unlabeled current message uses the current guide and grammar.\n\nAn explicitly selected current Run, attested current binding, current Dispatch,\nor federated attachment takes precedence over legacy fallback. A retained\nadoption record alone does not grant mutation authority. If liveness, principal\nownership, capability, or the exact legacy contract is unproven, degrade to\nread-only inspection and never fall back to local execution.\n\nAdoption preserves the live agent process, PTY/session, terminal handle,\ntab/pane, worktree or folder workspace, Task, and Dispatch. It never restarts or\nreplaces the worker and never revives the retired scheduler. Loss of lifecycle\nauthority does not invalidate the existing process, assignment, or filesystem\nwork. Exact recovery may restore the same PTY once in its original inactive\nbackground tab; it must not spawn, write, signal, stop, switch, focus, split, or\ninject a terminal.\n\n## Compatibility recovery\n\nWhen a compatibility response returns structured next-step arguments, execute\nthose exact arguments with the same selected CLI executable. Do not translate\nfrom memory, broaden the recipient, or retry as a current mutation unless the\nreceipt explicitly authorizes it.\n\nA pending ask, reply, final Dispatch settlement, and consuming check have\ndurable recovery identities. Heartbeat and escalation remain at-least-once\nacross a manual contract-boundary retry. If an ask may already have been\nanswered, run the exact non-consuming recovery check printed by Orca before\ncreating any new question. Never guess among identical question threads.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The\ninitial command commits the question, prints its exact\n`ask --resume ` command, and exits with launcher status `75`. Run\nthat exact resume after the launcher or update boundary. For an attested WSL\nlaunch, preserve the printed `orca-ide` executable and distro route. Older WSL\nworkers without launch proof remain lifecycle read-only even while their\nterminal and filesystem work continue.\n\n## Read-only inspection and takeover\n\nRead-only inspection does not consume mail:\n\n```text\nORCA orchestration run-list --json\nORCA orchestration run-show --id run_legacy_local --json\nORCA orchestration run-show --id --json\nORCA orchestration task-list --run --json\nORCA orchestration inbox --full --json\nORCA orchestration check --terminal --peek --format --json\nORCA terminal read --terminal --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\n`run_legacy_local` is an empty audit tombstone after adoption. Find the ordinary\nRun whose objective is `Recovered orchestration work from a contract update`.\n\nOnly when the original coordinator is unavailable or cannot prove retained\nauthority may a new live coordinator take over from its own terminal:\n\n```text\nORCA orchestration run-use --id --takeover-legacy --json\nORCA orchestration check --run --json\n```\n\nTakeover binds the authenticated invoking terminal; `--from` cannot nominate\nanother coordinator. It fences only the old coordinator and moves pending mail\ninto current Run delivery. It preserves live workers, Tasks, Dispatches, processes, and files.\nNever take over while the original coordinator is actively coordinating.\n\nDo not launch a replacement editor merely because Orca updated or authority is\nunclear. Keep the original worker as the only editor until a stable handoff\npoint, then use a fresh current Dispatch in a conflict-free placement.\n\n\n\n# Low-level topology\n\nLoad this reference only when `worker-start` cannot express required custom argv\nor terminal topology. It is not the normal supervised loop and is never a full\nhandoff recipe.\n\n```text\nORCA terminal create --worktree active --title --command \"\" --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\nORCA orchestration dispatch --task --to --inject --json\n```\n\nWait for readiness only when startup could lose injected input. Prefer\nagent-first `worker-start` whenever its argv and topology are sufficient.\n\n`dispatch --inject` creates authoritative Task/Dispatch context but deliberately\nkeeps an operator-created process unsupervised: it creates no supervised worker\nresource row. `worker-show`, `worker-read`, and `worker-list` report the lane as\n`unsupervised`; `worker-stop` and `worker-abandon` do not close that process, and\nsettled retain/release take no process action.\n\nUse `worker-start --terminal ` when lifecycle ownership of an existing\nagent terminal is required. Never imply that low-level dispatch retroactively\nowns a process, never use it to route around the nested-depth limit, and never\nuse it for an ownership handoff.\n\n\n\n# Messaging and gates\n\nLoad this reference for inbox replay, attempt-specific guidance, group\naddresses, blocking questions, or coordinator-managed DAG decisions.\n\nA successful `send` proves durable enqueue. Wake and nudge are best-effort\nattention only: neither proves the recipient read the message, began a turn, or\naccepted steering.\n\n## Coordinator delivery loop\n\n`check` names its caller with `--terminal ` and is the only verb that\nrejects `--from`. Omit `--terminal` inside an Orca terminal, where Orca resolves\nthe caller; pass it explicitly from anywhere else, including a dispatched\nworker reading coordinator follow-ups.\n\nA consuming coordinator `check` returns the bound Run's oldest FIFO Delivery,\nup to 50 messages, and replays that exact batch until acknowledged. Process\nevery row and required terminal ownership decision before `--ack`. Type filters\ndecide when a waiter wakes; they do not authorize skipping older actionable\nmail. A Delivery therefore always carries the whole FIFO batch whatever its\ntypes, and a `check` without `--wait` hands that batch over unfiltered.\n`--peek` and `--all` are read-only inspection, not progress through the\ncoordinator inbox.\n\nAn empty wait or timeout is a checkpoint. Continue rolling waits until every\nexpected Dispatch settles. Heartbeat or visible activity means alive, not done.\n\n## Addresses\n\nUse a stable Dispatch address for attempt-specific coordinator guidance:\n\n```text\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nDo not substitute a remote terminal handle. Omit `--from` for ordinary\ncoordinator calls; a dispatched worker instead copies the exact `--from` and\ncapability arguments in its preamble. `check` is the exception: it identifies\nits caller with `--terminal`, never `--from`.\n\nGroup addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`,\n`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Use them only for\nintentional fan-out status or questions. `worker_done`, heartbeat, and other\nDispatch lifecycle messages never target groups.\n\n## Questions and gates\n\nA worker uses `ask`; its timeout leaves one durable question pending, which the\nworker resumes by message ID. The coordinator answers that message with `reply`.\n\nUse a gate only for a coordinator-owned Task-DAG decision:\n\n```text\nORCA orchestration gate-create --task --question \"\" --options --json\nORCA orchestration gate-resolve --id --resolution \"\" --json\nORCA orchestration gate-list --task --json\n```\n\nPass `json_array` using the quoting rules of the active shell; do not copy POSIX\nsingle-quote syntax into PowerShell or `cmd.exe`.\n\nDo not create a gate merely to answer a worker's `ask`.\n\n\n\n# Placement and remote execution\n\nLoad this reference before creating a new worktree or placing work through SSH,\nWSL, or another connected Orca server.\n\n## Placement choices\n\nA fresh worker means a fresh agent terminal, not a new Git worktree. Use the\ncurrent or an exact existing workspace by default. Create a worktree only when\nthe user requested one or a concrete checkout or filesystem conflict makes\nsharing unsafe.\n\n```text\n# Current workspace; setup is not rerun.\nORCA orchestration worker-start --task --worktree current --agent codex --json\n\n# Stacked child worktree.\nORCA orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n\n# Independent top-level worktree.\nORCA orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nCurrent and exact existing workspaces create a fresh terminal unless\n`--terminal` is explicit. Folder workspaces are first-class; do not invoke Git\nor require worktree lineage when the selected workspace is a folder.\n\nRegister a folder workspace through project setup. `repo add --path `\nrequires a valid Git repository and rejects a plain directory:\n\n```text\nORCA project setup-existing-folder --project --host --path --kind folder --json\n```\n\nThen place work on the returned workspace with an exact selector. A worktree\nselector needs the full `::` value Orca returned, passed as\n`id:`; a bare repo id is not a worktree id. `new-child` and\n`new-top-level` are worktree creation and do not apply to a folder.\n\nNew worktrees use agent-first creation and run setup by default. Preserve the\nrepository's startup policy: `start-immediately` can report setup as `running`,\nwhile `wait-for-setup` gates prompt delivery on success. Orca lineage, Git base,\nfilesystem isolation, coordination parentage, UI grouping, and execution host\nare separate decisions.\n\n## Connected servers\n\nThe Run and Tasks remain authoritative on the current server. `--on` selects\nonly the worker's execution server and appears only on `worker-start`:\n\n```text\nORCA orchestration worker-start --task --on --worktree new-top-level --repo --name --agent codex --setup run --json\n```\n\nRemote `current` and `new-child` are invalid because they are ambiguous across\nservers. Use an exact discovered remote workspace, or `new-top-level` with an\nexact remote repository selector. After start, route every follow-up, read,\nstop, and cleanup by Dispatch ID; never repeat `--on` or substitute a remote\nterminal handle.\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\nORCA orchestration worker-list --run --include-remote --json\n```\n\n`worker-list` reads local fleet state only; enumerate remote workers with\n`--include-remote` or every one of them reads `unverifiable`. Scope every list\nwith `--run `: unscoped, it reports every Dispatch this runtime has\nrecorded, and the workers you are waiting on are lost in that history.\n\n## Execution-host and mixed-version floor\n\nThe execution host owns process, filesystem, transcript, stop, and cleanup\nfacts. Render only `live`, `unverifiable`, or `exited`. Connection loss, relay\nabsence, missing client inventory, or timeout yields `unverifiable`, never\nsynthetic exit and never a client-local substitute action.\n\nClients and servers update independently. Optional response fields may be\nabsent. Forward model/effort, transcript reads, cleanup, or another new remote\noperation only when the peer advertises the relevant capability; unknown stream\nopcodes can be silently dropped. A narrow unsupported response may degrade to a\ndocumented older path, but must not broaden the target or cross the execution\nboundary. Changing host-published content reaches old clients even without a\nwire-shape change, so preserve established semantics or negotiate the behavior.\n\nFor WSL, use the exact executable and arguments returned by Orca so the distro\nand packaged launcher remain bound. Do not translate a printed `orca-ide`\nrecovery command into a PATH-resolved local command.\n\n\n\n# Recovery and cleanup\n\nLoad this reference only after a failed/stopped/unknown attempt, explicit retry\ndecision, stop/abandon request, retention request, or uncertain release.\n\n| Proven state | Safe action |\n| ----------------------- | ------------------------------------------------------------------ |\n| `ready` or active | Keep waiting; optionally read bounded output |\n| `failed` or `stopped` | Start a replacement with `--retry-of`; repeat placement explicitly |\n| `outcome_unknown` | Inspect, then choose `worker-stop` or explicit `worker-abandon` |\n| Accepted `worker_done` | Reuse, retain, or release |\n| Remote contact lost | Preserve `unverifiable`; do not stop or retry from absence alone |\n| `unverifiable` liveness | Keep waiting or inspect; never stop, abandon, retry, or release |\n| Proven `exited` agent | Enumerate with `worker-list`; follow its `nextAction` |\n\n## Inspect before acting\n\n```text\nORCA orchestration worker-list --run --json\nORCA orchestration worker-list --run --include-remote --json\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\n```\n\n`worker-list` is the enumerating command and the authority on agent liveness:\neach row carries `projection.liveness`, `projection.attention.categories`,\n`projection.attention.requiresAction`, and a literal `projection.nextAction`\nargv to run. Always scope it with `--run `; an unscoped list reports\nevery Dispatch this runtime has ever recorded and buries the live ones.\n`worker-show`'s `observation.status` is PTY liveness only, so a `live` terminal\nwhose agent died at a trust prompt still reads `live` there.\n\nWhen the two disagree, the fleet verdict decides — unless the fleet row is\n`unverifiable` for a reason that names a gap on this client rather than a fact\nabout the worker. `missing_status`, `host_unavailable`, and\n`capability_unsupported` are such gaps: the first means this runtime holds no\nstatus row, the second that it could not ask the execution host at all, and the\nthird that a stale peer answered but lacks the fleet-snapshot capability.\nAgainst any of them, a `worker-show` verdict sourced from the execution host is\nthe better evidence and outranks the row. Only `host_unavailable` is contact\nloss; the other two mean the host was never asked or answered without the\ncapability.\n\nThis never promotes absence. `unverifiable` from either command still authorizes\nnothing — only a positive `live` or `exited` verdict does.\n\nA worker started with `--on ` reads `unverifiable` until you\nenumerate with `--include-remote`, which asks its execution host for the\nverdict. Past 100 rows the response pages, so follow `page.nextCursor` with\n`--cursor ` until `page.hasMore` is false.\n\n## Stall needs positive evidence\n\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Only then choose `worker-stop` or `worker-abandon`.\n\n`unverifiable` is always absence — `missing_status`, `stale_status`,\n`restored_unconfirmed`, or a remote worker with no connection — and a null\n`agentWait` or an unchanged `worker-read` tail is that same absence seen again.\nAbsence never authorizes stop, abandon, retry, or release: keep waiting, or\ninspect until you hold one of the positive signals above. A `nextAction` that\nnames an inspecting command is asking for evidence, not for cleanup.\n\n`worker-read --source auto` uses a proven provider transcript when available and\notherwise returns bounded terminal output with a typed `fallbackReason`.\nContinue with its top-level cursor, which is pinned to that source. If Orca\nreports `source_changed`, restart without the old cursor. A bounded initial\ntranscript tail can return an EOF cursor that follows only newly appended records;\nread `contentComplete`, `clipping`, and `warnings` before assuming omitted older\nrecords are pageable. Never guess a provider session ID, transcript path, or\nremote terminal handle.\n\n## Was the mutation applied?\n\nWhen a mutation's response was lost and named no Dispatch, do not replay blind.\nEvery orchestration mutation accepts `--retry-request `, which reuses one\noperation identity so Orca can replay, join, or recover it instead of starting a\nduplicate. Ask what happened first:\n\n```text\nORCA orchestration request-show --request --json\n```\n\n`completed` means the mutation already took effect; read its recorded receipt\ninstead of rerunning. `pending` means the original mutation is still running or\nOrca restarted before recording its outcome; replay the original command with\n`--retry-request `. `absent` means this runtime holds no receipt\nunder your caller identity — that is not proof nothing happened, so inspect the\naffected Task, Dispatch, and terminal before deciding whether to retry.\n\nWhen a worker's terminal accepted input but the submit is unconfirmed, use\n`terminal send --wait-submit `: it observes the accepted prompt for that\nlong and, on timeout, returns the input-accepted receipt without resending.\n\n## Refused starts\n\n`dispatch` and `worker-start` refuse the following preflight cases with a stable\n`error.code`; read it before choosing a recovery, and treat `error.data.nextSteps`\nas the exact recovery text. Older hosts may omit `data`, so treat every field as\noptional.\n\n| Code | Meaning | Recovery |\n| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |\n| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |\n| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |\n| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |\n\n## Retry, stop, and abandon\n\nRetry only a positively proven failed or stopped attempt. Name the failed Task\nwith `--task`, since `--spec` creates a new one. Placement is never silently\ninherited:\n\n```text\nORCA orchestration worker-start --task --retry-of --worktree --agent --json\n```\n\nAfter three consecutive failures for one Task, its dispatch context\ncircuit-breaks and the Task is failed. Do not route around that boundary with a\nnew Run or an unrelated Dispatch.\n\nFor `outcome_unknown`, inspect first, then make an explicit choice:\n\n```text\nORCA orchestration worker-stop --dispatch --json\nORCA orchestration worker-abandon --dispatch --json\n```\n\n`worker-stop` closes only the exact proven supervised agent terminal. It never\ndeletes the worktree, setup terminal, configured tabs, or unrelated processes.\n`worker-abandon` fences orchestration while accepting that resources may remain\nlive; it performs no remote, process, or filesystem action.\n\n## Retain and release\n\n```text\nORCA orchestration worker-retain --dispatch --json\nORCA orchestration worker-release --dispatch --json\n```\n\nRetain only when the user explicitly wants the settled terminal kept live.\nRelease works after succeeded and failed reports, archives readable output, and\ncloses only the exact terminal owned by that settled Dispatch. Replays may call\nrelease again safely. Reused, pre-existing, setup, coordinator, active,\nuser-taken-over, and unproven terminals are retained.\n\nA `worker-start` that failed before its agent was ready still owns the terminal\nit created. Its receipt names `worker-release`, and `worker-list` reports that\nrow as `reclaimable`; release it there rather than closing the terminal by hand.\n\nNever release because of timeout, TUI idle, heartbeat, status, question,\nescalation, or stale/rejected completion. If the receipt says `release_pending`\nor `release_unknown`, follow its exact recovery action. Never substitute\n`terminal close`.\n\n`orchestration reset` is destructive recovery. Do not run it during active\ncoordination unless the user explicitly abandons that state.\n\n\n\n# Worker contract\n\nThe injected preamble is authoritative. Copy its command rather than\nreconstructing flags. In particular, preserve the exact executable, worker\nhandle, Dispatch capability, Task ID, and Dispatch ID.\n\n## Heartbeat\n\nSend heartbeats only at the cadence required by the live preamble. Skip them\nwhile blocked inside `ask` or `check --wait`; those calls are liveness signals.\n\n```text\nORCA orchestration send --from --dispatch-capability --type heartbeat --subject \"alive\" --task-id --dispatch-id --phase \"\"\n```\n\nUse typed lifecycle flags, not a hand-written JSON payload. A heartbeat proves\nliveness, never completion.\n\n## Ask and resume\n\nUse Orca `ask` whenever the coordinator must answer. Never open a local question\nTUI the coordinator cannot answer.\n\n```text\nORCA orchestration ask --from --dispatch-capability --question \"\" --options \",\" --timeout-ms 600000\n\nORCA orchestration ask --from --dispatch-capability --resume --timeout-ms 600000\n```\n\nA timeout or disconnect leaves the original question pending. Resume its\nmessage ID; do not create a duplicate question.\n\n## Reading coordinator follow-ups\n\nThe coordinator steers a running worker with `send --to dispatch:`. That\nenqueue is durable but does not interrupt you, so nothing arrives unless you\nlook:\n\n```text\nORCA orchestration check --terminal --json\n```\n\nRun it at each natural checkpoint — before starting a new file, after a test\nrun — and once more immediately before `worker_done`, so a redirect or a\ncancellation lands before the Task settles. `check` names its caller with\n`--terminal`, never `--from`. Stop checking after `worker_done`.\n\nIf `check` returns `consumer_fenced`, this process no longer owns its Dispatch:\nthe Attempt was re-attached to another worker or settled without you. Stop, do\nnot send `worker_done`, and do not retry the check. An empty `check` never means\nyou were replaced; `consumer_fenced` is the only way you learn that.\n\n## Escalation\n\nEscalate only before completion and only when the coordinator must intervene:\n\n```text\nORCA orchestration send --from --dispatch-capability --type escalation --subject \"Blocked: \" --body \"
\" --task-id --dispatch-id \n```\n\n## Completion\n\nSend exactly one terminal report. `--body` is three sentences: what changed,\nwhat was found, and what remains. Use `--outcome failed` when the requested work\nis not complete; never hide failure in prose or silently exit.\n\nAppend `--files-modified` or `--report-path` only when applicable, using actual\npaths. Do not send documentation placeholders as metadata.\n\n```text\nORCA orchestration send --from --dispatch-capability --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded\n```\n\nAfter `worker_done`, end the dispatched turn and idle. Do not poll, close your\nown terminal, or begin unrelated work. A later direct user instruction is new\nuser-owned work and must not reuse settled lifecycle IDs; a supervised follow-up\narrives with a fresh preamble and Task block.\n" +const ORCHESTRATION_FULL_MARKDOWN = "---\nname: orchestration\ndescription: >-\n Coordinate supervised Orca workers: threaded messages, blocking ask/reply,\n task dispatch, worker_done/escalation waits, task DAGs, decision gates,\n coordinator loops, and decomposing work across agents. Use `orca-cli` for full\n ownership handoffs — \"hand off\", \"handoff\", \"handover\", \"give this to another\n agent\", \"another worktree\" — unless asked to supervise, monitor, or coordinate\n a DAG, and for terminal control, lightweight terminal prompts, shell commands,\n Orca worktree management, and reading or waiting on terminals. Use Computer\n Use for external browser windows, webviews, Orca app UI, or desktop UI outside\n Orca's embedded browser only when the task requires OS/window-level control\n such as focus, menus, dialogs, coordinates, or screenshots. Use `orca-cli` for\n Orca's embedded pages and a page-automation tool such as Playwright or CDP for\n external pages.\n---\n\n# Orca orchestration\n\nOrchestration is Orca's structured coordination layer. It records who owns work,\nwhich attempt is authoritative, and when supervised work has settled.\n\n## Outcome\n\n**Result:** every in-scope Task has one explicit outcome and every settled worker\nterminal has a next owner or cleanup decision. **Next consumer:** the user who\nrequested supervision. **Done:** all expected Dispatches have settled, every\ndelivered message was processed before acknowledgment, each settled worker was\nreused, explicitly retained, or released, and the turn ends only when the report\nto that user names, per Task, its outcome, the evidence behind it, and any\nunresolved blocker.\n\n**Safe failure:** preserve work and authority and report the state as unknown or\n`unverifiable`. Only positive proof of exit authorizes stop, abandon, or retry,\nand only an accepted settlement authorizes release. Every other observation,\nabsence included, is a checkpoint.\n\n## Classify the role\n\n| Current context | Role | Route |\n| ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------ |\n| The user explicitly asks to supervise, monitor, wait for results, track completion, coordinate a DAG, use a decision gate, or manage ask/reply | Coordinator | Use the supervised loop below |\n| The current prompt contains a live injected preamble with Task and Dispatch IDs | Dispatched worker | Follow the preamble and the worker obligations below |\n| The user asks to hand off ownership or start another agent/worktree without supervision | Handoff owner | Use `orca-cli`; create no Run, Task, or Dispatch and do not monitor completion |\n| A message carries a legacy authority label | Compatibility operator | Load the legacy contract reference before any lifecycle mutation |\n| No live preamble and no explicit supervision | Ordinary terminal agent | Do not emit lifecycle messages; use `orca-cli` for terminal/worktree work |\n\nModel or effort selection does not make a handoff supervised. Never substitute a\nnon-Orca subagent tool when Orca orchestration provenance was requested.\n\n## Authority and safety floor\n\n- A Run is a durable namespace and coordinator inbox; it does not schedule or\n place workers. A Task is work. A Dispatch is one authoritative Task attempt.\n- Lifecycle authority comes from the active Dispatch, not a terminal title,\n copied ID, old database row, provider transcript, or visible pane.\n- Workers use the exact executable, handle, capability, Task ID, and Dispatch ID\n in the live preamble. Never reconstruct, translate, or broaden those arguments.\n- After remote start, address the worker by Dispatch ID. The execution host owns\n process, filesystem, transcript, stop, and cleanup facts. Preserve the verdicts\n `live` / `unverifiable` / `exited`; contact loss is not process death.\n- Liveness is layered: `worker-list`'s `projection.liveness` is the fleet verdict\n for the agent; `worker-show`'s `observation.status` is PTY liveness only. A live\n terminal can still hold a dead or stuck agent.\n- Folder workspaces are valid; never require Git or assume a worktree.\n- Clients and remote servers update independently. Treat unknown optional fields\n as absent. A new stream operation requires advertised capability because old\n decoders may silently drop unknown opcodes. Never fall back to local execution\n when remote authority or capability is unproven.\n- Use the executable you used to run `skills get` for the entire run. In the\n examples below, replace `ORCA` with it; do not create a shell variable or run\n `ORCA` literally. If it fails, report that exact error instead of switching.\n- A successful `orchestration send` proves durable enqueue; its wake or nudge is\n best-effort attention only and does not prove the recipient read or accepted it.\n\n## Worker obligations\n\nThe injected preamble is authoritative. A dispatched worker must:\n\n1. Do only the current Task and use the preamble's `ask` command for a blocking\n coordinator question. Never open a local question TUI the coordinator cannot\n answer. Resume the same message ID after an ask timeout.\n2. Send heartbeats only at the cadence in the preamble. A heartbeat proves\n liveness, not completion.\n3. Read coordinator follow-ups at each natural checkpoint — before starting a\n new file, after a test run — and once more immediately before `worker_done`:\n `ORCA orchestration check --terminal --json`.\n4. Send `worker_done` exactly once, from the dispatched terminal, with a\n three-sentence executive summary, both lifecycle IDs, and explicit\n `--outcome succeeded` or `--outcome failed`. Never encode failure only in prose.\n5. Append `--files-modified` and `--report-path` only with real values when\n applicable. After `worker_done`, end the dispatched turn and idle; do not poll\n or start new work.\n\nA direct user instruction after completion starts new user-owned work and takes\nprecedence over the idle rule. Do not reuse the settled lifecycle IDs.\n\n## Canonical supervised loop\n\nConfirm the runtime, bind one Run, and start the full independent wave before\nwaiting. `worker-start --spec` creates the Task and its attempt in one call:\n\n```text\nORCA status --json\nORCA orchestration run-create --objective \"\" --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent codex --json\nORCA orchestration worker-start --spec \"\" --worktree current --agent claude --json\nORCA orchestration check --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nIf `worker-start` exits non-zero, do not relaunch. Read the receipt's\n`failedStage` and `residualResources`, then load\n`references/recovery-and-cleanup.md`.\n\nUse `task-create` plus `worker-start --task ` for planned fan-out with\ndependencies or a retry of a known Task. Use dependencies only for real ordering\nand prefer parallel waves over chains deeper than three or four steps; nested\nworkers obey the depth limit, and a new Run does not reset the caller's depth.\n\nA consuming `check` names its caller with `--terminal `, never `--from`;\nomit it inside the coordinator's own Orca terminal. It returns the bound Run's\noldest FIFO Delivery and replays that batch until acknowledged. Process every\nmessage: reply to questions, validate each `worker_done` against the expected\nactive Dispatch, and decide each settled terminal's next owner before the ack:\n\n```text\nORCA orchestration reply --id --body \"\" --json\nORCA orchestration worker-release --dispatch --json\nORCA orchestration check --ack --wait --types \"worker_done,escalation,question\" --timeout-ms 900000 --json\n```\n\nKeep waiting until every expected Dispatch settles. A timeout or empty result is\na checkpoint, not a failure. Do not stop, retry, release, or launch a duplicate\neditor without the positive proof `## Outcome` requires.\n\nAfter three consecutive empty waits, stop waiting blindly and enumerate with\n`ORCA orchestration worker-list --include-remote --json` (defaults to the bound\nRun; `--run ` overrides; the receipt's `scope` names which), acting on\neach row's `projection.attention` categories, `projection.attention.requiresAction`, and literal `projection.nextAction` argv.\nA `none` `nextAction` has no argv to run: read `liveness.reason` and keep waiting\nwith `check --wait`. Absence never earns an argv; settlement and pending work still do.\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Then load `references/recovery-and-cleanup.md` and choose\n`worker-stop` or `worker-abandon` explicitly. `unverifiable` is absence,\nincluding when `worker-show` reports `agentWait` null. Absence never authorizes\nstop, abandon, retry, or release; keep waiting or inspect.\n\n`worker-start` is the normal path, composing placement, terminal readiness,\nprompt injection, and supervised resource ownership. `dispatch --inject` leaves\nan operator-created process unsupervised and is only for an expressiveness gap.\n\n## Task-spec contract\n\nEvery Task spec must be self-contained and name:\n\n- **Target:** the files, component, or environment in scope.\n- **Change:** the concrete result to produce.\n- **Constraints:** invariants, compatibility rules, and do-not-touch boundaries.\n- **Ownership:** what this worker may edit and any coordination boundary.\n- **Observable acceptance:** the test, output, or evidence that proves completion.\n\n## Completion accounting\n\nAfter an accepted success or failure report, immediately do exactly one:\n\n1. Reuse the same proven agent terminal for an immediate follow-up Dispatch.\n2. Record user-requested retention with `worker-retain`.\n3. Run `worker-release`.\n\nRelease is post-settlement cleanup, not cancellation. Only an accepted\nsettlement authorizes it; no other observation does. If release is uncertain,\nfollow its exact recovery receipt and never substitute `terminal close`.\n\nA valid `worker_done` settles the Task and Dispatch automatically; do not follow\nit with `task-update --status completed`. Enumerate the terminals still owing a\ndecision with `worker-list --run --terminal-state reclaimable --json`,\nand do not end the coordinator turn until it returns none.\n\n## Conditional references\n\nThis compact guide is sufficient for the normal local loop. At an action gate\nbelow, run `ORCA skills get orchestration --reference references/.md` and\nread only that document; `--references` lists the names. If the CLI rejects\n`--reference`, run `ORCA skills get orchestration --full` once instead: it\nreturns this exact kernel and every reference, so read only the named one. If an\nolder CLI rejects `--full`, keep this kernel's safety floor, use that command's\n`--help`, and never guess newer flags.\n\n| Action gate | Bundled reference |\n| ------------------------------------------------------------------------------------------------------------- | ----------------------------------------- |\n| Expanded DAG waves, launch model/effort, same-terminal reuse, or review ownership | `references/coordinator-loop.md` |\n| You are a dispatched worker and the live preamble does not answer your question, or `check` returned an error | `references/worker-contract.md` |\n| New worktree, exact workspace, SSH, WSL, or connected-server placement | `references/placement-and-remote.md` |\n| Inbox replay, follow-up messages, group addresses, or decision gates | `references/messaging-and-gates.md` |\n| Failed/stopped/unknown attempts, retry, stop, abandon, retain, or uncertain release | `references/recovery-and-cleanup.md` |\n| Custom argv or terminal topology that `worker-start` cannot express | `references/low-level-topology.md` |\n| Any legacy label, adopted Run, compatibility receipt, or takeover | `references/legacy-contract-migration.md` |\n\nRetired scheduler commands are not aliases for Run creation. Recovery commands\nmust provide their exact next action; follow it with the same selected executable.\n\n---\n\n# Bundled references\n\nThese references belong to the version-matched guide above. Read only the documents named by its action gates.\n\n\n\n# Coordinator loop\n\nLoad this reference for expanded DAG waves, per-invocation launch preferences,\nsame-terminal reuse, or review ownership. The compact guide remains the source\nof truth for the loop order and completion boundary.\n\n## Ready waves\n\nCreate independent Tasks before the first wait. Encode only real dependencies,\nthen use the ready view as external memory:\n\n```text\nORCA orchestration task-create --spec \"\" --deps --json\nORCA orchestration task-list --ready --brief --json\n```\n\n`--brief` collapses whitespace and caps echoed specs at 160 characters;\n`spec_truncated` identifies shortened rows. Omit it when full specs are needed or\nwhen an older CLI rejects the flag. A nested worker must respect\n`nested_worker_depth_exceeded`; creating another Run does not reset depth.\n\n## Launch preferences\n\nFor a fresh Claude, Codex, or Cursor terminal, `--model` accepts an opaque\nprovider model ID. Pass it only when the user named a model; otherwise omit it\nso the worker inherits the user's configured agent default. Add `--effort` only\nwhen that model supports it:\n\n```text\nORCA orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`; neither option combines with `--terminal`. A\nconnected worker server must advertise launch-preference support before Orca\nforwards either field. Compare `launch.requested` with `launch.effective`; never\nclaim a model or effort from requested arguments alone.\n\n## Reuse after settlement\n\nChoose the terminal's next owner before acknowledging the Delivery. When the\nsame exact agent has immediate follow-up work, recover the proven handle and\ntransfer cleanup ownership to the new Dispatch:\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-start --task --terminal --json\n```\n\nOtherwise explicitly retain or release the settled worker. Do not leave it live\nonly to inspect output; archived output remains available through `worker-read`.\n\n## Review ownership\n\nA review-only `worker_done` authorizes synthesis of findings, not coordinator\nfile edits. Dispatch or hand off fixes unless the user explicitly assigned them\nto the coordinator. If the user's plan names a next owner, post-review fixes and\nPR preparation remain with that owner; the coordinator routes and synthesizes.\n\n\n\n# Legacy contract migration\n\nLoad this reference only for an authority label, adopted Run, compatibility or\nrecovery receipt, or explicit legacy takeover. A newly created attempt always\nuses the current grammar.\n\n## Authority labels\n\n- `[LEGACY COMPATIBILITY]` is live and attested. Run only the exact supported\n command printed with the message, using the same selected executable and\n arguments supplied by the original prompt.\n- `[LEGACY RECOVERY REPLAY — MAY HAVE BEEN SEEN]` is one bounded,\n at-least-once cutover replay. Process it idempotently and acknowledge only\n through the exact displayed guidance.\n- `[LEGACY READ-ONLY]` is inspection-only. It has no reply, acknowledgment, or\n lifecycle mutation.\n- An unlabeled current message uses the current guide and grammar.\n\nAn explicitly selected current Run, attested current binding, current Dispatch,\nor federated attachment takes precedence over legacy fallback. A retained\nadoption record alone does not grant mutation authority. If liveness, principal\nownership, capability, or the exact legacy contract is unproven, degrade to\nread-only inspection and never fall back to local execution.\n\nAdoption preserves the live agent process, PTY/session, terminal handle,\ntab/pane, worktree or folder workspace, Task, and Dispatch. It never restarts or\nreplaces the worker and never revives the retired scheduler. Loss of lifecycle\nauthority does not invalidate the existing process, assignment, or filesystem\nwork. Exact recovery may restore the same PTY once in its original inactive\nbackground tab; it must not spawn, write, signal, stop, switch, focus, split, or\ninject a terminal.\n\n## Compatibility recovery\n\nWhen a compatibility response returns structured next-step arguments, execute\nthose exact arguments with the same selected CLI executable. Do not translate\nfrom memory, broaden the recipient, or retry as a current mutation unless the\nreceipt explicitly authorizes it.\n\nA pending ask, reply, final Dispatch settlement, and consuming check have\ndurable recovery identities. Heartbeat and escalation remain at-least-once\nacross a manual contract-boundary retry. If an ask may already have been\nanswered, run the exact non-consuming recovery check printed by Orca before\ncreating any new question. Never guess among identical question threads.\n\nOn packaged Windows, a legacy ask uses a two-step commit/resume protocol. The\ninitial command commits the question, prints its exact\n`ask --resume ` command, and exits with launcher status `75`. Run\nthat exact resume after the launcher or update boundary. For an attested WSL\nlaunch, preserve the printed `orca-ide` executable and distro route. Older WSL\nworkers without launch proof remain lifecycle read-only even while their\nterminal and filesystem work continue.\n\n## Read-only inspection and takeover\n\nRead-only inspection does not consume mail:\n\n```text\nORCA orchestration run-list --json\nORCA orchestration run-show --id run_legacy_local --json\nORCA orchestration run-show --id --json\nORCA orchestration task-list --run --json\nORCA orchestration inbox --full --json\nORCA orchestration check --terminal --peek --format --json\nORCA terminal read --terminal --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\n```\n\n`run_legacy_local` is an empty audit tombstone after adoption. Find the ordinary\nRun whose objective is `Recovered orchestration work from a contract update`.\n\nOnly when the original coordinator is unavailable or cannot prove retained\nauthority may a new live coordinator take over from its own terminal:\n\n```text\nORCA orchestration run-use --id --takeover-legacy --json\nORCA orchestration check --run --json\n```\n\nTakeover binds the authenticated invoking terminal; `--from` cannot nominate\nanother coordinator. It fences only the old coordinator and moves pending mail\ninto current Run delivery. It preserves live workers, Tasks, Dispatches, processes, and files.\nNever take over while the original coordinator is actively coordinating.\n\nDo not launch a replacement editor merely because Orca updated or authority is\nunclear. Keep the original worker as the only editor until a stable handoff\npoint, then use a fresh current Dispatch in a conflict-free placement.\n\n\n\n# Low-level topology\n\nLoad this reference only when `worker-start` cannot express required custom argv\nor terminal topology. It is not the normal supervised loop and is never a full\nhandoff recipe.\n\n```text\nORCA terminal create --worktree active --title --command \"\" --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\nORCA orchestration dispatch --task --to --inject --json\n```\n\nWait for readiness only when startup could lose injected input. Prefer\nagent-first `worker-start` whenever its argv and topology are sufficient.\n\n`dispatch --inject` creates authoritative Task/Dispatch context but deliberately\nkeeps an operator-created process unsupervised: it creates no supervised worker\nresource row. `worker-show`, `worker-read`, and `worker-list` report the lane as\n`unsupervised`; `worker-stop` and `worker-abandon` do not close that process, and\nsettled retain/release take no process action.\n\nUse `worker-start --terminal ` when lifecycle ownership of an existing\nagent terminal is required. Never imply that low-level dispatch retroactively\nowns a process, never use it to route around the nested-depth limit, and never\nuse it for an ownership handoff.\n\n\n\n# Messaging and gates\n\nLoad this reference for inbox replay, attempt-specific guidance, group\naddresses, blocking questions, or coordinator-managed DAG decisions.\n\nA successful `send` proves durable enqueue. Wake and nudge are best-effort\nattention only: neither proves the recipient read the message, began a turn, or\naccepted steering.\n\n## Coordinator delivery loop\n\n`check` names its caller with `--terminal ` and is the only verb that\nrejects `--from`. Omit `--terminal` inside an Orca terminal, where Orca resolves\nthe caller; pass it explicitly from anywhere else, including a dispatched\nworker reading coordinator follow-ups.\n\nA consuming coordinator `check` returns the bound Run's oldest FIFO Delivery,\nup to 50 messages, and replays that exact batch until acknowledged. Process\nevery row and required terminal ownership decision before `--ack`. Type filters\ndecide when a waiter wakes; they do not authorize skipping older actionable\nmail. A Delivery therefore always carries the whole FIFO batch whatever its\ntypes, and a `check` without `--wait` hands that batch over unfiltered.\n`--peek` and `--all` are read-only inspection, not progress through the\ncoordinator inbox.\n\nAn empty wait or timeout is a checkpoint. Continue rolling waits until every\nexpected Dispatch settles. Heartbeat or visible activity means alive, not done.\n\n## Addresses\n\nUse a stable Dispatch address for attempt-specific coordinator guidance:\n\n```text\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nDo not substitute a remote terminal handle. Omit `--from` for ordinary\ncoordinator calls; a dispatched worker instead copies the exact `--from` and\ncapability arguments in its preamble. `check` is the exception: it identifies\nits caller with `--terminal`, never `--from`.\n\nGroup addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`,\n`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Every group but\n`@worktree:` means the live Dispatches of the sender's own Run. Mail goes\nto each `dispatch:` mailbox, except a worker coordinating a child Run\nreceives it in that `run:` mailbox. A sender bound to no Run is refused;\n`--run` must match the group audience and never grants membership.\nA Run group excludes its owning coordinator; a worker raising a blocker sends\nto `run:`. A worker that created its own Run addresses that Run's workers,\nnot its siblings. `@worktree:` reaches matching workspace terminals,\nincluding coordinators. Use groups only for intentional fan-out status or\nquestions. `worker_done`, heartbeat, and other\nDispatch lifecycle messages never target groups.\n\n## Questions and gates\n\nA worker uses `ask`; its timeout leaves one durable question pending, which the\nworker resumes by message ID. The coordinator answers that message with `reply`.\n\nUse a gate only for a coordinator-owned Task-DAG decision:\n\n```text\nORCA orchestration gate-create --task --question \"\" --options --json\nORCA orchestration gate-resolve --id --resolution \"\" --json\nORCA orchestration gate-list --task --json\n```\n\nPass `json_array` using the quoting rules of the active shell; do not copy POSIX\nsingle-quote syntax into PowerShell or `cmd.exe`.\n\nDo not create a gate merely to answer a worker's `ask`.\n\n\n\n# Placement and remote execution\n\nLoad this reference before creating a new worktree or placing work through SSH,\nWSL, or another connected Orca server.\n\n## Placement choices\n\nA fresh worker means a fresh agent terminal, not a new Git worktree. Use the\ncurrent or an exact existing workspace by default. Create a worktree only when\nthe user requested one or a concrete checkout or filesystem conflict makes\nsharing unsafe.\n\n```text\n# Current workspace; setup is not rerun.\nORCA orchestration worker-start --task --worktree current --agent codex --json\n\n# Stacked child worktree.\nORCA orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n\n# Independent top-level worktree.\nORCA orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nCurrent and exact existing workspaces create a fresh terminal unless\n`--terminal` is explicit. Folder workspaces are first-class; do not invoke Git\nor require worktree lineage when the selected workspace is a folder.\n\nRegister a folder workspace through project setup. `repo add --path `\nrequires a valid Git repository and rejects a plain directory:\n\n```text\nORCA project setup-existing-folder --project --host --path --kind folder --json\n```\n\nThen place work on the returned workspace with an exact selector. A worktree\nselector needs the full `::` value Orca returned, passed as\n`id:`; a bare repo id is not a worktree id. `new-child` and\n`new-top-level` are worktree creation and do not apply to a folder.\n\nNew worktrees use agent-first creation and run setup by default. Preserve the\nrepository's startup policy: `start-immediately` can report setup as `running`,\nwhile `wait-for-setup` gates prompt delivery on success. Orca lineage, Git base,\nfilesystem isolation, coordination parentage, UI grouping, and execution host\nare separate decisions.\n\n## Connected servers\n\nThe Run and Tasks remain authoritative on the current server. `--on` selects\nonly the worker's execution server and appears only on `worker-start`:\n\n```text\nORCA orchestration worker-start --task --on --worktree new-top-level --repo --name --agent codex --setup run --json\n```\n\nRemote `current` and `new-child` are invalid because they are ambiguous across\nservers. Use an exact discovered remote workspace, or `new-top-level` with an\nexact remote repository selector. After start, route every follow-up, read,\nstop, and cleanup by Dispatch ID; never repeat `--on` or substitute a remote\nterminal handle.\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\nORCA orchestration worker-list --run --include-remote --json\n```\n\n`worker-list` reads local fleet state only; enumerate remote workers with\n`--include-remote` or every one of them reads `unverifiable`. Scope every list\nwith `--run `: unscoped, it reports every Dispatch this runtime has\nrecorded, and the workers you are waiting on are lost in that history.\n\n## Execution-host and mixed-version floor\n\nThe execution host owns process, filesystem, transcript, stop, and cleanup\nfacts. Render only `live`, `unverifiable`, or `exited`. Connection loss, relay\nabsence, missing client inventory, or timeout yields `unverifiable`, never\nsynthetic exit and never a client-local substitute action.\n\nClients and servers update independently. Optional response fields may be\nabsent. Forward model/effort, transcript reads, cleanup, or another new remote\noperation only when the peer advertises the relevant capability; unknown stream\nopcodes can be silently dropped. A narrow unsupported response may degrade to a\ndocumented older path, but must not broaden the target or cross the execution\nboundary. Changing host-published content reaches old clients even without a\nwire-shape change, so preserve established semantics or negotiate the behavior.\n\nFor WSL, use the exact executable and arguments returned by Orca so the distro\nand packaged launcher remain bound. Do not translate a printed `orca-ide`\nrecovery command into a PATH-resolved local command.\n\n\n\n# Recovery and cleanup\n\nLoad this reference only after a failed/stopped/unknown attempt, explicit retry\ndecision, stop/abandon request, retention request, or uncertain release.\n\n| Proven state | Safe action |\n| ----------------------- | ------------------------------------------------------------------ |\n| `ready` or active | Keep waiting; optionally read bounded output |\n| `failed` or `stopped` | Start a replacement with `--retry-of`; repeat placement explicitly |\n| `outcome_unknown` | Inspect, then choose `worker-stop` or explicit `worker-abandon` |\n| Accepted `worker_done` | Reuse, retain, or release |\n| Remote contact lost | Preserve `unverifiable`; do not stop or retry from absence alone |\n| `unverifiable` liveness | Keep waiting or inspect; never stop, abandon, retry, or release |\n| Proven `exited` agent | Enumerate with `worker-list`; follow its `nextAction` |\n\n## Inspect before acting\n\n```text\nORCA orchestration worker-list --run --json\nORCA orchestration worker-list --run --include-remote --json\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\n```\n\n`worker-list` is the enumerating command and the authority on agent liveness:\neach row carries `projection.liveness`, `projection.attention.categories`,\n`projection.attention.requiresAction`, and a literal `projection.nextAction`\nargv to run. Always scope it with `--run `; an unscoped list reports\nevery Dispatch this runtime has ever recorded and buries the live ones.\n`worker-show`'s `observation.status` is PTY liveness only, so a `live` terminal\nwhose agent died at a trust prompt still reads `live` there.\n\nWhen the two disagree, the fleet verdict decides — unless the fleet row is\n`unverifiable` for a reason that names a gap on this client rather than a fact\nabout the worker. `missing_status`, `host_unavailable`, and\n`capability_unsupported` are such gaps: the first means this runtime holds no\nstatus row, the second that it could not ask the execution host at all, and the\nthird that a stale peer answered but lacks the fleet-snapshot capability.\nAgainst any of them, a `worker-show` verdict sourced from the execution host is\nthe better evidence and outranks the row. Only `host_unavailable` is contact\nloss; the other two mean the host was never asked or answered without the\ncapability.\n\nThis never promotes absence. `unverifiable` from either command still authorizes\nnothing — only a positive `live` or `exited` verdict does.\n\nA worker started with `--on ` reads `unverifiable` until you\nenumerate with `--include-remote`, which asks its execution host for the\nverdict. Past 100 rows the response pages, so follow `page.nextCursor` with\n`--cursor ` until `page.hasMore` is false.\n\n## Stall needs positive evidence\n\nLeave the wait only on positive proof the agent stopped: `exited` liveness, the\nworker's own observation of process exit, or a transcript whose final agent turn\nsent no `worker_done`. Only then choose `worker-stop` or `worker-abandon`.\n\n`unverifiable` is always absence — `missing_status`, `stale_status`,\n`restored_unconfirmed`, or a remote worker with no connection — and a null\n`agentWait` or an unchanged `worker-read` tail is that same absence seen again.\nAbsence never authorizes stop, abandon, retry, or release: keep waiting, or\ninspect until you hold one of the positive signals above. A `nextAction` that\nnames an inspecting command is asking for evidence, not for cleanup.\n\n`worker-read --source auto` uses a proven provider transcript when available and\notherwise returns bounded terminal output with a typed `fallbackReason`.\nContinue with its top-level cursor, which is pinned to that source. If Orca\nreports `source_changed`, restart without the old cursor. A bounded initial\ntranscript tail can return an EOF cursor that follows only newly appended records;\nread `contentComplete`, `clipping`, and `warnings` before assuming omitted older\nrecords are pageable. Never guess a provider session ID, transcript path, or\nremote terminal handle.\n\n## Was the mutation applied?\n\nWhen a mutation's response was lost and named no Dispatch, do not replay blind.\nEvery orchestration mutation accepts `--retry-request `, which reuses one\noperation identity so Orca can replay, join, or recover it instead of starting a\nduplicate. Ask what happened first:\n\n```text\nORCA orchestration request-show --request --json\n```\n\n`completed` means the mutation already took effect; read its recorded receipt\ninstead of rerunning. `pending` means the original mutation is still running or\nOrca restarted before recording its outcome; replay the original command with\n`--retry-request `. `absent` means this runtime holds no receipt\nunder your caller identity — that is not proof nothing happened, so inspect the\naffected Task, Dispatch, and terminal before deciding whether to retry.\n\nWhen a worker's terminal accepted input but the submit is unconfirmed, use\n`terminal send --wait-submit `: it observes the accepted prompt for that\nlong and, on timeout, returns the input-accepted receipt without resending.\n\n## Refused starts\n\n`dispatch` and `worker-start` refuse the following preflight cases with a stable\n`error.code`; read it before choosing a recovery, and treat `error.data.nextSteps`\nas the exact recovery text. Older hosts may omit `data`, so treat every field as\noptional.\n\n| Code | Meaning | Recovery |\n| -------------------- | --------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ |\n| `task_not_found` | No Task with that id, or not in the bound Run (`data.taskId`, `data.runId`) | Check `task-list --json`; create the Task with `task-create` if it does not exist |\n| `task_not_startable` | Task cannot start now: not `ready`, or invalid `--retry-of` (`data.status`, `data.unmetDependencies`, `data.retryOf`) | Wait for running dependencies with `check --wait`; retry or unblock failed ones; inspect `dispatch-show` if already dispatched |\n| `inject_rejected` | `--inject` refused because no recognized agent runs in the target (`data.terminal`, `data.reason`) | Start a recognized agent there or pick another terminal; or dispatch without `--inject` and use `terminal send` |\n| `runtime_error` | Any other failure, including a target terminal that already owns an active Dispatch | Read the message, inspect state, and do not retry unchanged |\n\n## Retry, stop, and abandon\n\nRetry only a positively proven failed or stopped attempt. Name the failed Task\nwith `--task`, since `--spec` creates a new one. Placement is never silently\ninherited:\n\n```text\nORCA orchestration worker-start --task --retry-of --worktree --agent --json\n```\n\nAfter three consecutive failures for one Task, its dispatch context\ncircuit-breaks and the Task is failed. Do not route around that boundary with a\nnew Run or an unrelated Dispatch.\n\nFor `outcome_unknown`, inspect first, then make an explicit choice:\n\n```text\nORCA orchestration worker-stop --dispatch --json\nORCA orchestration worker-abandon --dispatch --json\n```\n\n`worker-stop` closes only the exact proven supervised agent terminal. It never\ndeletes the worktree, setup terminal, configured tabs, or unrelated processes.\n`worker-abandon` fences orchestration while accepting that resources may remain\nlive; it performs no remote, process, or filesystem action.\n\n## Retain and release\n\n```text\nORCA orchestration worker-retain --dispatch --json\nORCA orchestration worker-release --dispatch --json\n```\n\nRetain only when the user explicitly wants the settled terminal kept live.\nRelease works after succeeded and failed reports, archives readable output, and\ncloses only the exact terminal owned by that settled Dispatch. Replays may call\nrelease again safely. Reused, pre-existing, setup, coordinator, active,\nuser-taken-over, and unproven terminals are retained.\n\nA `worker-start` that failed before its agent was ready still owns the terminal\nit created. Its receipt names `worker-release`, and `worker-list` reports that\nrow as `reclaimable`; release it there rather than closing the terminal by hand.\n\nNever release because of timeout, TUI idle, heartbeat, status, question,\nescalation, or stale/rejected completion. If the receipt says `release_pending`\nor `release_unknown`, follow its exact recovery action. Never substitute\n`terminal close`.\n\n`orchestration reset` is destructive recovery. Do not run it during active\ncoordination unless the user explicitly abandons that state.\n\n\n\n# Worker contract\n\nThe injected preamble is authoritative. Copy its command rather than\nreconstructing flags. In particular, preserve the exact executable, worker\nhandle, Dispatch capability, Task ID, and Dispatch ID.\n\n## Heartbeat\n\nSend heartbeats only at the cadence required by the live preamble. Skip them\nwhile blocked inside `ask` or `check --wait`; those calls are liveness signals.\n\n```text\nORCA orchestration send --from --dispatch-capability --type heartbeat --subject \"alive\" --task-id --dispatch-id --phase \"\"\n```\n\nUse typed lifecycle flags, not a hand-written JSON payload. A heartbeat proves\nliveness, never completion.\n\n## Ask and resume\n\nUse Orca `ask` whenever the coordinator must answer. Never open a local question\nTUI the coordinator cannot answer.\n\n```text\nORCA orchestration ask --from --dispatch-capability --question \"\" --options \",\" --timeout-ms 600000\n\nORCA orchestration ask --from --dispatch-capability --resume --timeout-ms 600000\n```\n\nA timeout or disconnect leaves the original question pending. Resume its\nmessage ID; do not create a duplicate question.\n\n## Reading coordinator follow-ups\n\nThe coordinator steers a running worker with `send --to dispatch:`. That\nenqueue is durable but does not interrupt you, so nothing arrives unless you\nlook:\n\n```text\nORCA orchestration check --terminal --json\n```\n\nRun it at each natural checkpoint — before starting a new file, after a test\nrun — and once more immediately before `worker_done`, so a redirect or a\ncancellation lands before the Task settles. `check` names its caller with\n`--terminal`, never `--from`. Stop checking after `worker_done`.\n\nIf `check` returns `consumer_fenced`, this process no longer owns its Dispatch:\nthe Attempt was re-attached to another worker or settled without you. Stop, do\nnot send `worker_done`, and do not retry the check. An empty `check` never means\nyou were replaced; `consumer_fenced` is the only way you learn that.\n\n## Escalation\n\nEscalate only before completion and only when the coordinator must intervene:\n\n```text\nORCA orchestration send --from --dispatch-capability --type escalation --subject \"Blocked: \" --body \"
\" --task-id --dispatch-id \n```\n\n## Completion\n\nSend exactly one terminal report. `--body` is three sentences: what changed,\nwhat was found, and what remains. Use `--outcome failed` when the requested work\nis not complete; never hide failure in prose or silently exit.\n\nAppend `--files-modified` or `--report-path` only when applicable, using actual\npaths. Do not send documentation placeholders as metadata.\n\n```text\nORCA orchestration send --from --dispatch-capability --type worker_done --subject \"\" --body \"\" --task-id --dispatch-id --outcome succeeded\n```\n\nAfter `worker_done`, end the dispatched turn and idle. Do not poll, close your\nown terminal, or begin unrelated work. A later direct user instruction is new\nuser-owned work and must not reuse settled lifecycle IDs; a supervised follow-up\narrives with a fresh preamble and Task block.\n" // oxfmt-ignore const ORCHESTRATION_COORDINATOR_LOOP_REFERENCE_MARKDOWN = "# Coordinator loop\n\nLoad this reference for expanded DAG waves, per-invocation launch preferences,\nsame-terminal reuse, or review ownership. The compact guide remains the source\nof truth for the loop order and completion boundary.\n\n## Ready waves\n\nCreate independent Tasks before the first wait. Encode only real dependencies,\nthen use the ready view as external memory:\n\n```text\nORCA orchestration task-create --spec \"\" --deps --json\nORCA orchestration task-list --ready --brief --json\n```\n\n`--brief` collapses whitespace and caps echoed specs at 160 characters;\n`spec_truncated` identifies shortened rows. Omit it when full specs are needed or\nwhen an older CLI rejects the flag. A nested worker must respect\n`nested_worker_depth_exceeded`; creating another Run does not reset depth.\n\n## Launch preferences\n\nFor a fresh Claude, Codex, or Cursor terminal, `--model` accepts an opaque\nprovider model ID. Pass it only when the user named a model; otherwise omit it\nso the worker inherits the user's configured agent default. Add `--effort` only\nwhen that model supports it:\n\n```text\nORCA orchestration worker-start --task --worktree current --agent claude --model opus --effort high --json\n```\n\n`--effort` requires `--model`; neither option combines with `--terminal`. A\nconnected worker server must advertise launch-preference support before Orca\nforwards either field. Compare `launch.requested` with `launch.effective`; never\nclaim a model or effort from requested arguments alone.\n\n## Reuse after settlement\n\nChoose the terminal's next owner before acknowledging the Delivery. When the\nsame exact agent has immediate follow-up work, recover the proven handle and\ntransfer cleanup ownership to the new Dispatch:\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-start --task --terminal --json\n```\n\nOtherwise explicitly retain or release the settled worker. Do not leave it live\nonly to inspect output; archived output remains available through `worker-read`.\n\n## Review ownership\n\nA review-only `worker_done` authorizes synthesis of findings, not coordinator\nfile edits. Dispatch or hand off fixes unless the user explicitly assigned them\nto the coordinator. If the user's plan names a next owner, post-review fixes and\nPR preparation remain with that owner; the coordinator routes and synthesizes.\n" @@ -81,7 +81,7 @@ const ORCHESTRATION_LEGACY_CONTRACT_MIGRATION_REFERENCE_MARKDOWN = "# Legacy con const ORCHESTRATION_LOW_LEVEL_TOPOLOGY_REFERENCE_MARKDOWN = "# Low-level topology\n\nLoad this reference only when `worker-start` cannot express required custom argv\nor terminal topology. It is not the normal supervised loop and is never a full\nhandoff recipe.\n\n```text\nORCA terminal create --worktree active --title --command \"\" --json\nORCA terminal wait --terminal --for tui-idle --timeout-ms 60000 --json\nORCA orchestration dispatch --task --to --inject --json\n```\n\nWait for readiness only when startup could lose injected input. Prefer\nagent-first `worker-start` whenever its argv and topology are sufficient.\n\n`dispatch --inject` creates authoritative Task/Dispatch context but deliberately\nkeeps an operator-created process unsupervised: it creates no supervised worker\nresource row. `worker-show`, `worker-read`, and `worker-list` report the lane as\n`unsupervised`; `worker-stop` and `worker-abandon` do not close that process, and\nsettled retain/release take no process action.\n\nUse `worker-start --terminal ` when lifecycle ownership of an existing\nagent terminal is required. Never imply that low-level dispatch retroactively\nowns a process, never use it to route around the nested-depth limit, and never\nuse it for an ownership handoff.\n" // oxfmt-ignore -const ORCHESTRATION_MESSAGING_AND_GATES_REFERENCE_MARKDOWN = "# Messaging and gates\n\nLoad this reference for inbox replay, attempt-specific guidance, group\naddresses, blocking questions, or coordinator-managed DAG decisions.\n\nA successful `send` proves durable enqueue. Wake and nudge are best-effort\nattention only: neither proves the recipient read the message, began a turn, or\naccepted steering.\n\n## Coordinator delivery loop\n\n`check` names its caller with `--terminal ` and is the only verb that\nrejects `--from`. Omit `--terminal` inside an Orca terminal, where Orca resolves\nthe caller; pass it explicitly from anywhere else, including a dispatched\nworker reading coordinator follow-ups.\n\nA consuming coordinator `check` returns the bound Run's oldest FIFO Delivery,\nup to 50 messages, and replays that exact batch until acknowledged. Process\nevery row and required terminal ownership decision before `--ack`. Type filters\ndecide when a waiter wakes; they do not authorize skipping older actionable\nmail. A Delivery therefore always carries the whole FIFO batch whatever its\ntypes, and a `check` without `--wait` hands that batch over unfiltered.\n`--peek` and `--all` are read-only inspection, not progress through the\ncoordinator inbox.\n\nAn empty wait or timeout is a checkpoint. Continue rolling waits until every\nexpected Dispatch settles. Heartbeat or visible activity means alive, not done.\n\n## Addresses\n\nUse a stable Dispatch address for attempt-specific coordinator guidance:\n\n```text\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nDo not substitute a remote terminal handle. Omit `--from` for ordinary\ncoordinator calls; a dispatched worker instead copies the exact `--from` and\ncapability arguments in its preamble. `check` is the exception: it identifies\nits caller with `--terminal`, never `--from`.\n\nGroup addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`,\n`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Use them only for\nintentional fan-out status or questions. `worker_done`, heartbeat, and other\nDispatch lifecycle messages never target groups.\n\n## Questions and gates\n\nA worker uses `ask`; its timeout leaves one durable question pending, which the\nworker resumes by message ID. The coordinator answers that message with `reply`.\n\nUse a gate only for a coordinator-owned Task-DAG decision:\n\n```text\nORCA orchestration gate-create --task --question \"\" --options --json\nORCA orchestration gate-resolve --id --resolution \"\" --json\nORCA orchestration gate-list --task --json\n```\n\nPass `json_array` using the quoting rules of the active shell; do not copy POSIX\nsingle-quote syntax into PowerShell or `cmd.exe`.\n\nDo not create a gate merely to answer a worker's `ask`.\n" +const ORCHESTRATION_MESSAGING_AND_GATES_REFERENCE_MARKDOWN = "# Messaging and gates\n\nLoad this reference for inbox replay, attempt-specific guidance, group\naddresses, blocking questions, or coordinator-managed DAG decisions.\n\nA successful `send` proves durable enqueue. Wake and nudge are best-effort\nattention only: neither proves the recipient read the message, began a turn, or\naccepted steering.\n\n## Coordinator delivery loop\n\n`check` names its caller with `--terminal ` and is the only verb that\nrejects `--from`. Omit `--terminal` inside an Orca terminal, where Orca resolves\nthe caller; pass it explicitly from anywhere else, including a dispatched\nworker reading coordinator follow-ups.\n\nA consuming coordinator `check` returns the bound Run's oldest FIFO Delivery,\nup to 50 messages, and replays that exact batch until acknowledged. Process\nevery row and required terminal ownership decision before `--ack`. Type filters\ndecide when a waiter wakes; they do not authorize skipping older actionable\nmail. A Delivery therefore always carries the whole FIFO batch whatever its\ntypes, and a `check` without `--wait` hands that batch over unfiltered.\n`--peek` and `--all` are read-only inspection, not progress through the\ncoordinator inbox.\n\nAn empty wait or timeout is a checkpoint. Continue rolling waits until every\nexpected Dispatch settles. Heartbeat or visible activity means alive, not done.\n\n## Addresses\n\nUse a stable Dispatch address for attempt-specific coordinator guidance:\n\n```text\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\n```\n\nDo not substitute a remote terminal handle. Omit `--from` for ordinary\ncoordinator calls; a dispatched worker instead copies the exact `--from` and\ncapability arguments in its preamble. `check` is the exception: it identifies\nits caller with `--terminal`, never `--from`.\n\nGroup addresses include `@all`, `@idle`, `@claude`, `@codex`, `@opencode`,\n`@gemini`, `@droid`, `@grok`, `@cursor`, and `@worktree:`. Every group but\n`@worktree:` means the live Dispatches of the sender's own Run. Mail goes\nto each `dispatch:` mailbox, except a worker coordinating a child Run\nreceives it in that `run:` mailbox. A sender bound to no Run is refused;\n`--run` must match the group audience and never grants membership.\nA Run group excludes its owning coordinator; a worker raising a blocker sends\nto `run:`. A worker that created its own Run addresses that Run's workers,\nnot its siblings. `@worktree:` reaches matching workspace terminals,\nincluding coordinators. Use groups only for intentional fan-out status or\nquestions. `worker_done`, heartbeat, and other\nDispatch lifecycle messages never target groups.\n\n## Questions and gates\n\nA worker uses `ask`; its timeout leaves one durable question pending, which the\nworker resumes by message ID. The coordinator answers that message with `reply`.\n\nUse a gate only for a coordinator-owned Task-DAG decision:\n\n```text\nORCA orchestration gate-create --task --question \"\" --options --json\nORCA orchestration gate-resolve --id --resolution \"\" --json\nORCA orchestration gate-list --task --json\n```\n\nPass `json_array` using the quoting rules of the active shell; do not copy POSIX\nsingle-quote syntax into PowerShell or `cmd.exe`.\n\nDo not create a gate merely to answer a worker's `ask`.\n" // oxfmt-ignore const ORCHESTRATION_PLACEMENT_AND_REMOTE_REFERENCE_MARKDOWN = "# Placement and remote execution\n\nLoad this reference before creating a new worktree or placing work through SSH,\nWSL, or another connected Orca server.\n\n## Placement choices\n\nA fresh worker means a fresh agent terminal, not a new Git worktree. Use the\ncurrent or an exact existing workspace by default. Create a worktree only when\nthe user requested one or a concrete checkout or filesystem conflict makes\nsharing unsafe.\n\n```text\n# Current workspace; setup is not rerun.\nORCA orchestration worker-start --task --worktree current --agent codex --json\n\n# Stacked child worktree.\nORCA orchestration worker-start --task --worktree new-child --name --agent codex --setup run --json\n\n# Independent top-level worktree.\nORCA orchestration worker-start --task --worktree new-top-level --name --agent codex --setup run --json\n```\n\nCurrent and exact existing workspaces create a fresh terminal unless\n`--terminal` is explicit. Folder workspaces are first-class; do not invoke Git\nor require worktree lineage when the selected workspace is a folder.\n\nRegister a folder workspace through project setup. `repo add --path `\nrequires a valid Git repository and rejects a plain directory:\n\n```text\nORCA project setup-existing-folder --project --host --path --kind folder --json\n```\n\nThen place work on the returned workspace with an exact selector. A worktree\nselector needs the full `::` value Orca returned, passed as\n`id:`; a bare repo id is not a worktree id. `new-child` and\n`new-top-level` are worktree creation and do not apply to a folder.\n\nNew worktrees use agent-first creation and run setup by default. Preserve the\nrepository's startup policy: `start-immediately` can report setup as `running`,\nwhile `wait-for-setup` gates prompt delivery on success. Orca lineage, Git base,\nfilesystem isolation, coordination parentage, UI grouping, and execution host\nare separate decisions.\n\n## Connected servers\n\nThe Run and Tasks remain authoritative on the current server. `--on` selects\nonly the worker's execution server and appears only on `worker-start`:\n\n```text\nORCA orchestration worker-start --task --on --worktree new-top-level --repo --name --agent codex --setup run --json\n```\n\nRemote `current` and `new-child` are invalid because they are ambiguous across\nservers. Use an exact discovered remote workspace, or `new-top-level` with an\nexact remote repository selector. After start, route every follow-up, read,\nstop, and cleanup by Dispatch ID; never repeat `--on` or substitute a remote\nterminal handle.\n\n```text\nORCA orchestration worker-show --dispatch --json\nORCA orchestration worker-read --dispatch --limit 50 --json\nORCA orchestration send --to dispatch: --subject \"Follow-up\" --body \"\" --json\nORCA orchestration worker-list --run --include-remote --json\n```\n\n`worker-list` reads local fleet state only; enumerate remote workers with\n`--include-remote` or every one of them reads `unverifiable`. Scope every list\nwith `--run `: unscoped, it reports every Dispatch this runtime has\nrecorded, and the workers you are waiting on are lost in that history.\n\n## Execution-host and mixed-version floor\n\nThe execution host owns process, filesystem, transcript, stop, and cleanup\nfacts. Render only `live`, `unverifiable`, or `exited`. Connection loss, relay\nabsence, missing client inventory, or timeout yields `unverifiable`, never\nsynthetic exit and never a client-local substitute action.\n\nClients and servers update independently. Optional response fields may be\nabsent. Forward model/effort, transcript reads, cleanup, or another new remote\noperation only when the peer advertises the relevant capability; unknown stream\nopcodes can be silently dropped. A narrow unsupported response may degrade to a\ndocumented older path, but must not broaden the target or cross the execution\nboundary. Changing host-published content reaches old clients even without a\nwire-shape change, so preserve established semantics or negotiate the behavior.\n\nFor WSL, use the exact executable and arguments returned by Orca so the distro\nand packaged launcher remain bound. Do not translate a printed `orca-ide`\nrecovery command into a PATH-resolved local command.\n" diff --git a/src/cli/runtime/websocket-transport.test.ts b/src/cli/runtime/websocket-transport.test.ts index 4c178b71fd8..e6fc34699ff 100644 --- a/src/cli/runtime/websocket-transport.test.ts +++ b/src/cli/runtime/websocket-transport.test.ts @@ -18,6 +18,8 @@ import { launchOrcaApp } from './launch' import { addEnvironmentFromPairingCode } from './environments' import { RuntimeClientError } from './types' import { + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY, AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY, @@ -72,6 +74,8 @@ describe('CLI remote WebSocket transport', () => { expect.objectContaining({ clientCapabilities: [ AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY, SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY, SESSION_TABS_AUTHORITATIVE_INVENTORY_RUNTIME_CAPABILITY, AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, diff --git a/src/cli/specs/orchestration.ts b/src/cli/specs/orchestration.ts index e62911b2c76..d75123ee6b7 100644 --- a/src/cli/specs/orchestration.ts +++ b/src/cli/specs/orchestration.ts @@ -70,6 +70,8 @@ export const ORCHESTRATION_COMMAND_SPECS: CommandSpec[] = [ notes: [ 'Valid --type values: status, dispatch, worker_done, merge_ready, escalation, handoff, decision_gate, question, heartbeat.', 'To answer a worker question, use orchestration reply --id --body with the same Orca CLI executable.', + 'Group addresses (@all, @idle, @codex, ...) reach the live Dispatches of your own Run; a sender in no Run must use run: or dispatch:. @worktree: names one workspace.', + 'Run groups exclude their owning coordinator; send to run: to raise something with yours. Nested coordinators receive group mail in their child Run mailbox; @worktree: includes workspace coordinators.', 'On Windows PowerShell, quote group addresses such as --to "@all" or --to "@worktree:".', "worker_done and heartbeat are exact-Dispatch signals and cannot target groups; omit --to to use the Dispatch's Run mailbox.", 'worker_done requires --outcome succeeded or --outcome failed.', diff --git a/src/cli/terminal-format.test.ts b/src/cli/terminal-format.test.ts index 42c036471eb..294fbc09cee 100644 --- a/src/cli/terminal-format.test.ts +++ b/src/cli/terminal-format.test.ts @@ -1,5 +1,16 @@ import { describe, expect, it } from 'vitest' -import { formatTerminalClose, formatTerminalFocus, formatTerminalSend } from './terminal-format' +import type { + RuntimeTerminalShow, + RuntimeTerminalWait, + RuntimeTerminalWaitBlockedReason +} from '../shared/runtime-terminal-contracts' +import { + formatTerminalClose, + formatTerminalFocus, + formatTerminalSend, + formatTerminalShow, + formatTerminalWait +} from './terminal-format' describe('formatTerminalFocus', () => { it('distinguishes superseded navigation from a winning focus', () => { @@ -171,3 +182,73 @@ describe('formatTerminalSend', () => { expect(output).toContain('--retry-request prompt-swallowed --wait-submit ') }) }) + +// Why: an older host still publishes the codex-* tokens for dialogs its matcher never proved were +// Codex's, so a Gemini/Cursor/Antigravity user reads a Codex label unless the CLI names the neutral one. +describe('blocked-reason rendering against a mixed-version host', () => { + function showResult(reason?: RuntimeTerminalWaitBlockedReason): { + terminal: RuntimeTerminalShow + } { + return { + terminal: { + handle: 'term_agy', + ptyId: 'pty-1', + paneRuntimeId: 1, + rendererGraphEpoch: 1, + worktreeId: 'worktree-1', + worktreePath: '/tmp/w', + branch: 'main', + tabId: 'tab-1', + leafId: 'leaf-1', + title: 'Antigravity', + connected: true, + writable: true, + lastOutputAt: null, + preview: 'Do you trust the files in this folder?', + agentWait: { source: 'prompt-text', reason } + } + } + } + + function waitResult(blockedReason: RuntimeTerminalWaitBlockedReason): { + wait: RuntimeTerminalWait + } { + return { + wait: { + handle: 'term_agy', + condition: 'tui-idle', + satisfied: false, + status: 'running', + exitCode: null, + blockedReason + } + } + } + + // Why one assertion over every reason: a test that only asserts the *absence* of an alias suffix + // passes when the aliasing code is deleted, so each case is paired with a legacy token that must + // gain one. + it.each([ + ['codex-trust-workspace', 'codex-trust-workspace (agent-trust-workspace)'], + ['codex-update-prompt', 'codex-update-prompt (agent-update-prompt)'], + ['codex-cwd-prompt', 'codex-cwd-prompt (agent-cwd-prompt)'], + ['codex-hooks-review-prompt', 'codex-hooks-review-prompt (agent-hooks-review-prompt)'], + ['codex-interactive-prompt', 'codex-interactive-prompt (agent-interactive-prompt)'], + // This build published these itself, so there is nothing to reinterpret. + ['agent-trust-workspace', 'agent-trust-workspace'], + ['codex-model-migration-prompt', 'codex-model-migration-prompt'] + ] as const)('renders %s as %s on both wait and show', (reason, rendered) => { + expect(formatTerminalWait(waitResult(reason)).split('\n').at(-1)).toBe( + `blockedReason: ${rendered}` + ) + expect(formatTerminalShow(showResult(reason))).toContain( + `agentWait: ${rendered} (via prompt-text)` + ) + }) + + it('still describes a wait with no reason at all', () => { + expect(formatTerminalShow(showResult(undefined))).toContain( + 'agentWait: interactive prompt (via prompt-text)' + ) + }) +}) diff --git a/src/cli/terminal-format.ts b/src/cli/terminal-format.ts index 46c26556889..06d897828fe 100644 --- a/src/cli/terminal-format.ts +++ b/src/cli/terminal-format.ts @@ -1,5 +1,6 @@ import { PTY_LIVE_NOTE, describeUnconfirmedStop } from '../shared/pty-liveness-verdict' import { structuredChatPtyWriteRefusalCopy } from '../shared/agent-session-pty-write-refusal-copy' +import { describeTerminalWaitBlockedReason } from '../shared/terminal-wait-blocked-reason-legacy-alias' import { formatListingHostScope, type WithAnnotatedHostScope } from './omitted-host-scope-selectors' import type { RuntimeTerminalClose, @@ -118,7 +119,10 @@ function formatAgentWait(agentWait: RuntimeTerminalShow['agentWait']): string { if (!agentWait) { return 'none' } - return `${agentWait.reason ?? 'interactive prompt'} (via ${agentWait.source})` + if (!agentWait.reason) { + return `interactive prompt (via ${agentWait.source})` + } + return `${describeTerminalWaitBlockedReason(agentWait.reason)} (via ${agentWait.source})` } export function formatTerminalRead(result: { terminal: RuntimeTerminalRead }): string { @@ -278,7 +282,7 @@ export function formatTerminalWait(result: { wait: RuntimeTerminalWait }): strin `exitCode: ${result.wait.exitCode ?? 'null'}` ] if (result.wait.blockedReason) { - lines.push(`blockedReason: ${result.wait.blockedReason}`) + lines.push(`blockedReason: ${describeTerminalWaitBlockedReason(result.wait.blockedReason)}`) } return lines.join('\n') } diff --git a/src/main/agent-hooks/first-work-branch-rename.test.ts b/src/main/agent-hooks/first-work-branch-rename.test.ts index 9cd4d544041..8ecd4997aa0 100644 --- a/src/main/agent-hooks/first-work-branch-rename.test.ts +++ b/src/main/agent-hooks/first-work-branch-rename.test.ts @@ -100,10 +100,14 @@ describe('maybeAutoRenameBranchOnFirstWork', () => { isPendingFirstAgentMessageRename: () => true }) const items: AgentJournalRenderItem[] = [] + // A real journal's sequence only ever advances, so the feed's projection + // cache must miss on every publish here: this test is about the rename. + let sequence = 0 const journal = { snapshot: () => ({ items }), lastActivityAt: () => 1, - isReadOnly: false + isReadOnly: false, + cursor: () => ({ epoch: 1, sequence: (sequence += 1) }) } as unknown as AgentSessionJournal const pending: Promise[] = [] const observe = vi.fn((summary, options) => { @@ -175,6 +179,7 @@ describe('maybeAutoRenameBranchOnFirstWork', () => { const journal = { isReadOnly: false, lastActivityAt: () => 1, + cursor: () => ({ epoch: 1, sequence: 1 }), snapshot: () => ({ items: [ { body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'Fix auth' }] } }, diff --git a/src/main/agent-hooks/server-ingest-structured-status.test.ts b/src/main/agent-hooks/server-ingest-structured-status.test.ts new file mode 100644 index 00000000000..5b43cc10bce --- /dev/null +++ b/src/main/agent-hooks/server-ingest-structured-status.test.ts @@ -0,0 +1,277 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { mkdtempSync, readFileSync, rmSync, writeFileSync, mkdirSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' +import { + structuredAgentSessionPaneKey, + structuredAgentSessionTabId +} from '../../shared/structured-agent-session-projection' +import { AgentHookServer, _internals } from './server' +import { PANE } from './server.test-fixtures' + +const { getCohortAtEmitMock, trackMock } = vi.hoisted(() => ({ + getCohortAtEmitMock: vi.fn(), + trackMock: vi.fn() +})) + +vi.mock('../telemetry/client', () => ({ + track: trackMock +})) + +vi.mock('../telemetry/cohort-classifier', () => ({ + getCohortAtEmit: getCohortAtEmitMock +})) + +const SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' +const TAB = structuredAgentSessionTabId(SESSION) +const STRUCTURED_PANE = structuredAgentSessionPaneKey(TAB, SESSION) +const OBSERVED_AT = 1_757_030_400_000 + +function summary(over: Partial = {}): AgentSessionStatusSummary { + return { + sessionId: SESSION, + workspaceId: 'repo-1::/workspace/app', + agent: 'codex', + status: 'working', + hostExecutionOwned: true, + latestPrompt: 'ship the thing', + model: 'gpt-6-astra', + toolName: 'shell', + toolInput: 'sleep 30', + lastAssistantMessage: 'on it', + updatedAt: OBSERVED_AT, + ...over + } +} + +beforeEach(() => { + _internals.resetCachesForTests() + trackMock.mockReset() + getCohortAtEmitMock.mockReset() + getCohortAtEmitMock.mockReturnValue({ nth_repo_added: 2 }) +}) + +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('AgentHookServer ingestStructuredStatus', () => { + it('stores the projection as a row under the pane key the renderer derives', () => { + const server = new AgentHookServer() + server.ingestStructuredStatus(summary()) + + expect(server.getStatusSnapshot()).toEqual([ + expect.objectContaining({ + paneKey: STRUCTURED_PANE, + tabId: TAB, + worktreeId: 'repo-1::/workspace/app', + connectionId: null, + state: 'working', + agentType: 'codex', + prompt: 'ship the thing', + model: 'gpt-6-astra', + toolName: 'shell', + toolInput: 'sleep 30', + lastAssistantMessage: 'on it', + structuredHost: 'owned', + // The journal clock, not the ingest clock: a restart's republish is not new evidence. + evidenceObservedAt: OBSERVED_AT, + stateStartedAt: OBSERVED_AT + }) + ]) + expect(server.getStatusSnapshot()[0]?.observation?.origin).toBe('structured') + }) + + // The same mapping the sidebar applies, so the two surfaces cannot disagree about one session. + it('maps attention to blocked and idle to done', () => { + const server = new AgentHookServer() + server.ingestStructuredStatus(summary({ status: 'attention' })) + expect(server.getStatusSnapshot()[0]?.state).toBe('blocked') + server.ingestStructuredStatus(summary({ status: 'idle', updatedAt: OBSERVED_AT + 1 })) + expect(server.getStatusSnapshot()[0]?.state).toBe('done') + }) + + it('marks a session whose provider child is gone as held, not owned', () => { + const server = new AgentHookServer() + server.ingestStructuredStatus(summary({ hostExecutionOwned: undefined })) + expect(server.getStatusSnapshot()[0]?.structuredHost).toBe('held') + }) + + it('keeps the state start while later evidence of the same state arrives', () => { + const server = new AgentHookServer() + server.ingestStructuredStatus(summary()) + server.ingestStructuredStatus(summary({ toolName: 'read', updatedAt: OBSERVED_AT + 5_000 })) + + expect(server.getStatusSnapshot()[0]).toMatchObject({ + toolName: 'read', + evidenceObservedAt: OBSERVED_AT + 5_000, + stateStartedAt: OBSERVED_AT + }) + }) + + // Null status means no turn has been persisted; the chat shows nothing, so neither does this. + it('holds no row for a session without a persisted turn, and drops one that regresses to none', () => { + const server = new AgentHookServer() + server.ingestStructuredStatus(summary({ status: null })) + expect(server.getStatusSnapshot()).toEqual([]) + + server.ingestStructuredStatus(summary()) + server.ingestStructuredStatus(summary({ status: null })) + expect(server.getStatusSnapshot()).toEqual([]) + }) + + it('drops the row when the host stops holding the session', () => { + const server = new AgentHookServer() + server.ingestStructuredStatus(summary()) + server.dropStructuredStatus(SESSION) + expect(server.getStatusSnapshot()).toEqual([]) + }) + + // The resume-identity remnant a dismissed PTY pane keeps exists so the agent can be resumed in + // that pane. A structured session has no pane, and the record store owns its resume identity — + // so a remnant here would be an unclearable row that every null-status publish re-minted. + it('leaves no resume-identity remnant behind, even carrying a provider session', () => { + const server = new AgentHookServer() + const withProviderSession = summary({ + providerSession: { key: 'session_id', id: 'codex-thread-1' } + }) + server.ingestStructuredStatus(withProviderSession) + expect(server.getStatusSnapshot()[0]?.providerSession).toEqual({ + key: 'session_id', + id: 'codex-thread-1' + }) + + server.dropStructuredStatus(SESSION) + expect(server.getStatusSnapshot()).toEqual([]) + }) + + // Structured rows are never serialized, so persisting one could only rewrite the file already + // on disk — once per debounce window for the whole of every streaming chat. + // "Exactly one writer per pane key" has to hold for deletes too: the renderer's feed bridge owns + // this pane, so a pane-status-clear would be main reaching into a row it does not write. + it('drops the row without sending the renderer a clear for a pane it does not write', () => { + const server = new AgentHookServer() + const cleared: unknown[] = [] + const dropped: string[] = [] + server.setPaneStatusClearListener((clear) => cleared.push(clear)) + server.subscribeStatusDrop((paneKey) => dropped.push(paneKey)) + + server.ingestStructuredStatus(summary()) + server.dropStructuredStatus(SESSION) + + expect(server.getStatusSnapshot()).toEqual([]) + expect(cleared).toEqual([]) + expect(dropped).toEqual([STRUCTURED_PANE]) + }) + + it('schedules no persist for a structured row, while a hook row still does', () => { + const server = new AgentHookServer() + const persists: number[] = [] + const scheduled = server as unknown as { scheduleStatusPersist: () => void } + const original = scheduled.scheduleStatusPersist.bind(server) + scheduled.scheduleStatusPersist = () => { + persists.push(1) + original() + } + + server.ingestStructuredStatus(summary()) + expect(persists).toHaveLength(0) + + server.ingestTerminalStatus({ + paneKey: PANE, + connectionId: null, + payload: { state: 'working', prompt: 'watch the build', agentType: 'claude' } + }) + expect(persists).toHaveLength(1) + }) + + it('leaves a hook-reported pane alone', () => { + const server = new AgentHookServer() + server.ingestTerminalStatus({ + paneKey: PANE, + connectionId: null, + payload: { state: 'working', prompt: 'watch the build', agentType: 'claude' } + }) + server.ingestStructuredStatus(summary()) + + const byPane = new Map(server.getStatusSnapshot().map((row) => [row.paneKey, row])) + expect(byPane.get(PANE)?.structuredHost).toBeUndefined() + expect(byPane.get(STRUCTURED_PANE)?.structuredHost).toBe('owned') + }) +}) + +describe('structured rows and last-status.json', () => { + let userDataPath: string + + beforeEach(() => { + userDataPath = mkdtempSync(join(tmpdir(), 'orca-structured-status-')) + }) + + afterEach(() => { + rmSync(userDataPath, { recursive: true, force: true }) + }) + + function lastStatusPath(): string { + return join(userDataPath, 'agent-hooks', 'last-status.json') + } + + // The journal is the durable truth and the host republishes on restore; a persisted copy would + // hydrate as unconfirmed and fight that republish. + it('are never written, while hook rows still are', async () => { + const server = new AgentHookServer() + await server.start({ env: 'production', userDataPath }) + try { + server.ingestTerminalStatus({ + paneKey: PANE, + connectionId: null, + payload: { state: 'working', prompt: 'watch the build', agentType: 'claude' } + }) + server.ingestStructuredStatus(summary()) + server.flushStatusPersistSync() + } finally { + server.stop() + } + + const file = JSON.parse(readFileSync(lastStatusPath(), 'utf8')) as { + entries: Record + } + expect(Object.keys(file.entries)).toEqual([PANE]) + + const restored = new AgentHookServer() + await restored.start({ env: 'production', userDataPath }) + try { + expect(restored.getStatusSnapshot().map((row) => row.paneKey)).toEqual([PANE]) + } finally { + restored.stop() + } + }) + + it('are dropped on hydrate if some other writer put one on disk', async () => { + mkdirSync(join(userDataPath, 'agent-hooks'), { recursive: true }) + writeFileSync( + lastStatusPath(), + JSON.stringify({ + version: 2, + entries: { + [STRUCTURED_PANE]: { + paneKey: STRUCTURED_PANE, + tabId: TAB, + connectionId: null, + receivedAt: Date.now(), + stateStartedAt: Date.now(), + structuredHost: 'owned', + payload: { state: 'working', prompt: 'ship the thing', agentType: 'codex' } + } + } + }) + ) + const server = new AgentHookServer() + await server.start({ env: 'production', userDataPath }) + try { + expect(server.getStatusSnapshot()).toEqual([]) + } finally { + server.stop() + } + }) +}) diff --git a/src/main/agent-hooks/server/server-cleanup.ts b/src/main/agent-hooks/server/server-cleanup.ts index 4fcd0b5e58e..d7669c7449f 100644 --- a/src/main/agent-hooks/server/server-cleanup.ts +++ b/src/main/agent-hooks/server/server-cleanup.ts @@ -22,12 +22,18 @@ export abstract class AgentHookServerCleanup extends AgentHookServerAuthorityFen } /** Drop only the status row (user dismissal); do NOT wipe prompt/tool caches since the pane's agent may still be alive. Use clearPaneState for PTY-teardown. */ - dropStatusEntry(paneKey: string): void { + dropStatusEntry( + paneKey: string, + /** Defaults to true: a dismissed pane can still be resumed in place. A structured session has + * no pane to resume into and its record store owns resume identity, so it passes false. */ + options?: { preserveResumeIdentity?: boolean } + ): void { const deleted = this.deleteStatusEntry(paneKey, { preserveAuthority: true }) if (!deleted) { return } - const retained = this.toRetainedProviderSessionRow(deleted) + const retained = + options?.preserveResumeIdentity === false ? null : this.toRetainedProviderSessionRow(deleted) if (retained) { this.state.lastStatusByPaneKey.set(deleted.paneKey, retained) } diff --git a/src/main/agent-hooks/server/server-ingest-remote.ts b/src/main/agent-hooks/server/server-ingest-remote.ts index 18b0a837c32..fae24c93b41 100644 --- a/src/main/agent-hooks/server/server-ingest-remote.ts +++ b/src/main/agent-hooks/server/server-ingest-remote.ts @@ -17,9 +17,9 @@ import { launchTokenHash } from '../../../shared/agent-hook-spool' import { parsePaneKey } from '../../../shared/stable-pane-id' import type { AgentHookEventPayload } from '../../../shared/agent-hook-listener/listener-event' import { isValidPiProviderSessionOnly } from './server-status-identity' -import { AgentHookServerIngestTerminal } from './server-ingest-terminal' +import { AgentHookServerIngestStructured } from './server-ingest-structured' -export abstract class AgentHookServerIngestRemote extends AgentHookServerIngestTerminal { +export abstract class AgentHookServerIngestRemote extends AgentHookServerIngestStructured { /** Ingest a payload from the relay JSON-RPC channel (not the local HTTP server); connectionId is stamped here. Main is still the SSH trust boundary, so re-run the canonical normalizer before caching. */ ingestRemote( envelope: { diff --git a/src/main/agent-hooks/server/server-ingest-structured.ts b/src/main/agent-hooks/server/server-ingest-structured.ts new file mode 100644 index 00000000000..45b0e5c0015 --- /dev/null +++ b/src/main/agent-hooks/server/server-ingest-structured.ts @@ -0,0 +1,66 @@ +import type { AgentSessionStatusSummary } from '../../../shared/agent-session-wire' +import type { ParsedAgentStatusPayload } from '../../../shared/agent-status-types' +import { + structuredAgentSessionPaneKey, + structuredAgentSessionStatusState, + structuredAgentSessionTabId +} from '../../../shared/structured-agent-session-projection' +import { AgentHookServerIngestTerminal } from './server-ingest-terminal' + +/** + * Structured (native chat) sessions have no PTY and no hook script, so nothing else reaches this + * store for them. The host projects each session's journal into a summary; this is where that + * summary becomes the same row every other agent has, keyed by the pane key the renderer derives. + */ +export abstract class AgentHookServerIngestStructured extends AgentHookServerIngestTerminal { + ingestStructuredStatus(summary: AgentSessionStatusSummary): void { + const paneKey = structuredStatusPaneKey(summary.sessionId) + // No persisted turn yet: the chat shows nothing, so neither does any status reader. + if (!summary.status) { + this.dropStructuredStatus(summary.sessionId) + return + } + if (this.getAgentStatusDisposition(paneKey) !== 'accept') { + return + } + const payload: ParsedAgentStatusPayload = { + state: structuredAgentSessionStatusState(summary.status), + prompt: summary.latestPrompt, + agentType: summary.agent, + ...(summary.model ? { model: summary.model } : {}), + ...(summary.toolName ? { toolName: summary.toolName } : {}), + ...(summary.toolInput ? { toolInput: summary.toolInput } : {}), + ...(summary.lastAssistantMessage + ? { lastAssistantMessage: summary.lastAssistantMessage } + : {}) + } + // The journal clock stamps the evidence so a restart's republish does not read as fresh work. + this.applyNormalizedStatus( + { + paneKey, + tabId: structuredAgentSessionTabId(summary.sessionId), + worktreeId: summary.workspaceId, + connectionId: null, + structuredHost: summary.hostExecutionOwned ? 'owned' : 'held', + ...(summary.providerSession ? { providerSession: summary.providerSession } : {}), + payload + }, + undefined, + 'structured', + summary.updatedAt + ) + } + + /** The host no longer holds the session; its last projection is history the journal keeps. + * `dropStatusEntry`, not `clearPaneState`: the renderer's own bridge still owns this pane key, + * so a pane-status-clear would make main a second writer for it. */ + dropStructuredStatus(sessionId: string): void { + this.dropStatusEntry(structuredStatusPaneKey(sessionId), { preserveResumeIdentity: false }) + } +} + +// The DERIVED pane key the renderer publishes, never the orchestration bearer handle or the minted +// worker pane key: both of those are credentials. +function structuredStatusPaneKey(sessionId: string): string { + return structuredAgentSessionPaneKey(structuredAgentSessionTabId(sessionId), sessionId) +} diff --git a/src/main/agent-hooks/server/server-persistence-validation.ts b/src/main/agent-hooks/server/server-persistence-validation.ts index 061646731ee..6c0136aaa51 100644 --- a/src/main/agent-hooks/server/server-persistence-validation.ts +++ b/src/main/agent-hooks/server/server-persistence-validation.ts @@ -43,6 +43,10 @@ export function sanitizeHydratedEntry( if (record.paneKey !== paneKey) { return null } + // Why: structured rows are never written; the host republishes the live projection on restore. + if (record.structuredHost !== undefined) { + return null + } const tabId = record.tabId if (tabId !== undefined && (typeof tabId !== 'string' || tabId.length === 0)) { return null diff --git a/src/main/agent-hooks/server/server-persistence.ts b/src/main/agent-hooks/server/server-persistence.ts index ecb33c44d37..f5b66222d6d 100644 --- a/src/main/agent-hooks/server/server-persistence.ts +++ b/src/main/agent-hooks/server/server-persistence.ts @@ -27,6 +27,11 @@ export abstract class AgentHookServerPersistence extends AgentHookServerHydratio continue } const enrichedPayload = payload as EnrichedAgentHookEventPayload + // Why: the session journal is the durable truth for a structured row and the host republishes + // it on restore; a persisted copy would hydrate unconfirmed and fight that republish. + if (enrichedPayload.structuredHost) { + continue + } const childOnlyBoundary = enrichedPayload.claudeLeadBoundaryChildOnly === true const { claudeRunningNonAgentTask: _claudeRunningNonAgentTask, diff --git a/src/main/agent-hooks/server/server-state.ts b/src/main/agent-hooks/server/server-state.ts index 7dc8125576e..b18677689d0 100644 --- a/src/main/agent-hooks/server/server-state.ts +++ b/src/main/agent-hooks/server/server-state.ts @@ -137,7 +137,8 @@ export abstract class AgentHookServerState { protected abstract markPaneClosedForAgentStatus(paneKey: string): void protected abstract attachStatusTiming( payload: AgentHookEventPayload, - now?: number + now?: number, + observedAt?: number ): EnrichedAgentHookEventPayload protected abstract hashPromptForTelemetryDedupe(prompt: string): string protected abstract maybeTrackAgentPromptSent( @@ -152,7 +153,8 @@ export abstract class AgentHookServerState { protected abstract applyNormalizedStatus( payload: AgentHookEventPayload, onAccepted?: () => void, - origin?: AgentStatusObservationOrigin + origin?: AgentStatusObservationOrigin, + observedAt?: number ): EnrichedAgentHookEventPayload protected abstract emitEnrichedStatus(enriched: EnrichedAgentHookEventPayload): void protected abstract clearAssistantMessageRetry(paneKey: string): void diff --git a/src/main/agent-hooks/server/server-status-application.ts b/src/main/agent-hooks/server/server-status-application.ts index f7e1126d11b..025c431dff6 100644 --- a/src/main/agent-hooks/server/server-status-application.ts +++ b/src/main/agent-hooks/server/server-status-application.ts @@ -17,9 +17,12 @@ import { AgentHookServerStatusDisposition } from './server-status-disposition' const MAX_REMEMBERED_EVIDENCE_OBSERVATIONS = 1024 export abstract class AgentHookServerStatusApplication extends AgentHookServerStatusDisposition { + /** `observedAt` is the producer's own clock for evidence that has one (a session journal); it + * stamps the evidence and state-start times while `receivedAt` keeps delivery order. */ protected attachStatusTiming( payload: AgentHookEventPayload, - now = Date.now() + now = Date.now(), + observedAt?: number ): EnrichedAgentHookEventPayload { const previous = this.state.lastStatusByPaneKey.get(payload.paneKey) as | EnrichedAgentHookEventPayload @@ -39,12 +42,12 @@ export abstract class AgentHookServerStatusApplication extends AgentHookServerSt const stateStartedAt = previous && previous.payload.state === payload.payload.state && !commandCodeNewTurn ? previous.stateStartedAt - : now + : (observedAt ?? now) // Why: `stateStartedAt` tracks the current state, while `receivedAt` tracks every arrival. return { ...payload, receivedAt: now, - evidenceObservedAt: this.resolveEvidenceObservedAt(payload, previous, now), + evidenceObservedAt: observedAt ?? this.resolveEvidenceObservedAt(payload, previous, now), stateStartedAt } } diff --git a/src/main/agent-hooks/server/server-status-identity.ts b/src/main/agent-hooks/server/server-status-identity.ts index a4ee28f1bb8..4f6e920d86e 100644 --- a/src/main/agent-hooks/server/server-status-identity.ts +++ b/src/main/agent-hooks/server/server-status-identity.ts @@ -68,6 +68,7 @@ export function toAgentStatusIpcPayload( ...(entry.promptInteractionKey ? { promptInteractionKey: entry.promptInteractionKey } : {}), ...(entry.restoredUnconfirmed ? { restoredUnconfirmed: true } : {}), ...(entry.observation ? { observation: entry.observation } : {}), + ...(entry.structuredHost ? { structuredHost: entry.structuredHost } : {}), ...entry.payload } } diff --git a/src/main/agent-hooks/server/server-status-update.ts b/src/main/agent-hooks/server/server-status-update.ts index 981da873e3d..1a3798efe47 100644 --- a/src/main/agent-hooks/server/server-status-update.ts +++ b/src/main/agent-hooks/server/server-status-update.ts @@ -23,7 +23,8 @@ export abstract class AgentHookServerStatusUpdate extends AgentHookServerStatusA protected applyNormalizedStatus( payload: AgentHookEventPayload, onAccepted?: () => void, - origin: AgentStatusObservationOrigin = 'hook' + origin: AgentStatusObservationOrigin = 'hook', + observedAt?: number ): EnrichedAgentHookEventPayload { if (payload.hookEventName === 'UserPromptSubmit') { // Why: the prompt boundary is authoritative even when text is unchanged; its next OSC working row must not inherit the prior cron/background turn stamp. @@ -179,8 +180,8 @@ export abstract class AgentHookServerStatusUpdate extends AgentHookServerStatusA this.maybeTrackAgentPromptSent(effectivePayload, previous) } const enriched = { - ...this.attachStatusTiming(boundaryAwarePayload, now), - observation: this.stampObservation(boundaryAwarePayload, origin, now) + ...this.attachStatusTiming(boundaryAwarePayload, now, observedAt), + observation: this.stampObservation(boundaryAwarePayload, origin, observedAt ?? now) } if ( typeof enriched.payload.turnCompletedAt === 'number' && @@ -198,7 +199,11 @@ export abstract class AgentHookServerStatusUpdate extends AgentHookServerStatusA this.runtimeObservedStatusPaneKeys.add(enriched.paneKey) } this.state.lastStatusByPaneKey.set(enriched.paneKey, enriched) - this.scheduleStatusPersist() + // Why skipped for structured rows: the serializer drops them, so the whole walk and stringify + // can only ever reproduce the last file — once per debounce window for a streaming chat. + if (!enriched.structuredHost) { + this.scheduleStatusPersist() + } this.notifyStatusChangeListeners() this.emitEnrichedStatus(enriched) return enriched diff --git a/src/main/ai-vault-search/session-search-clock.ts b/src/main/ai-vault-search/session-search-clock.ts new file mode 100644 index 00000000000..c9eaf609a34 --- /dev/null +++ b/src/main/ai-vault-search/session-search-clock.ts @@ -0,0 +1,24 @@ +// Why injected rather than the globals: every freshness guarantee this indexer +// makes is "within one reconcile interval", and a guarantee stated in wall time +// is only a claim until a test can advance the clock and watch it hold. + +/** Opaque to the indexer; a fake clock hands back whatever it likes. */ +export type SessionSearchTimerHandle = object | number + +export type SessionSearchClock = { + now(): number + setTimeout(callback: () => void, ms: number): SessionSearchTimerHandle + clearTimeout(handle: SessionSearchTimerHandle): void +} + +export const systemSessionSearchClock: SessionSearchClock = { + now: () => Date.now(), + setTimeout: (callback, ms) => { + const timer = setTimeout(callback, ms) + // Nothing here should hold the process open: the index is a cache, and a + // pending reconcile is never a reason to keep a CLI or a child alive. + timer.unref?.() + return timer + }, + clearTimeout: (handle) => clearTimeout(handle as NodeJS.Timeout) +} diff --git a/src/main/ai-vault-search/session-search-content-hash.test.ts b/src/main/ai-vault-search/session-search-content-hash.test.ts new file mode 100644 index 00000000000..1e7232fc855 --- /dev/null +++ b/src/main/ai-vault-search/session-search-content-hash.test.ts @@ -0,0 +1,48 @@ +import { expect, it } from 'vitest' +import { + EMPTY_CONTENT_HASH, + foldContentHash, + isCollapsibleContentHash +} from './session-search-content-hash' +import { userMessages } from './session-search-index-test-fixture' + +it('reaches the same digest whether the prefix arrives whole or in two appends', () => { + const messages = userMessages('turn', 5) + const whole = foldContentHash(EMPTY_CONTENT_HASH, messages) + const resumed = foldContentHash( + foldContentHash(EMPTY_CONTENT_HASH, messages.slice(0, 2)), + messages.slice(2) + ) + + expect(resumed).toEqual(whole) + expect(whole.count).toBe(5) +}) + +it('freezes once the prefix limit is reached so later appends cannot move it', () => { + // Found rather than imported: the limit is the module's business, and a test + // that reads it off the export cannot notice the fold ignoring it. + const capped = foldContentHash(EMPTY_CONTENT_HASH, userMessages('turn', 64)) + expect(capped.count).toBeLessThan(64) + expect(foldContentHash(capped, userMessages('later', 20))).toEqual(capped) +}) + +it('separates two conversations that share an opening prompt', () => { + const shared = userMessages('same opening', 1) + const first = foldContentHash(EMPTY_CONTENT_HASH, [ + ...shared, + { role: 'user', text: 'left', timestamp: null } + ]) + const second = foldContentHash(EMPTY_CONTENT_HASH, [ + ...shared, + { role: 'user', text: 'right', timestamp: null } + ]) + expect(first.hash).not.toBe(second.hash) +}) + +it('refuses to collapse on a prefix too short to mean anything', () => { + const one = foldContentHash(EMPTY_CONTENT_HASH, userMessages('only turn', 1)) + expect(isCollapsibleContentHash(one.hash, one.count)).toBe(false) + const two = foldContentHash(EMPTY_CONTENT_HASH, userMessages('two turns', 2)) + expect(isCollapsibleContentHash(two.hash, two.count)).toBe(true) + expect(isCollapsibleContentHash(null, 9)).toBe(false) +}) diff --git a/src/main/ai-vault-search/session-search-content-hash.ts b/src/main/ai-vault-search/session-search-content-hash.ts new file mode 100644 index 00000000000..a9d2ab229da --- /dev/null +++ b/src/main/ai-vault-search/session-search-content-hash.ts @@ -0,0 +1,45 @@ +import { createHash } from 'node:crypto' +import type { TranscriptMessage } from '../ai-vault/session-transcript-consumers' + +// Why: Claude `--resume` and Codex fork copy the parent transcript into a new +// file under a new session id, so one conversation lands N times in results. +// The shared opening prefix is what identifies the copy; the tail diverges. +const CONTENT_HASH_MESSAGE_LIMIT = 8 +// One shared opening prompt is not evidence of a fork; two turns is. +const CONTENT_HASH_MIN_MESSAGES = 2 + +export type SessionContentHash = { hash: string | null; count: number } + +export const EMPTY_CONTENT_HASH: SessionContentHash = { hash: null, count: 0 } + +/** + * Chained digest over the first `CONTENT_HASH_MESSAGE_LIMIT` messages. Chaining + * (rather than hashing one joined string) makes it resumable, so an `append` + * can finish a prefix a short `replace` started; once the limit is reached the + * value is frozen and later appends leave it untouched. + */ +export function foldContentHash( + previous: SessionContentHash, + messages: readonly TranscriptMessage[] +): SessionContentHash { + let { hash, count } = previous + for (const message of messages) { + if (count >= CONTENT_HASH_MESSAGE_LIMIT) { + break + } + hash = createHash('sha256') + .update(hash ?? '') + .update('\0') + .update(message.role) + .update('\0') + .update(message.text) + .digest('hex') + count += 1 + } + return { hash, count } +} + +/** Sessions collapse only on a hash that covers enough turns to mean anything. */ +export function isCollapsibleContentHash(hash: string | null, count: number): hash is string { + return hash !== null && count >= CONTENT_HASH_MIN_MESSAGES +} diff --git a/src/main/ai-vault-search/session-search-cwd-key.test.ts b/src/main/ai-vault-search/session-search-cwd-key.test.ts new file mode 100644 index 00000000000..24d915eedc9 --- /dev/null +++ b/src/main/ai-vault-search/session-search-cwd-key.test.ts @@ -0,0 +1,37 @@ +import { expect, it } from 'vitest' +import { folderGroupKey } from '../../shared/ai-vault-session-filters' +import { cwdKey } from './session-search-file-records' + +// The sidebar groups sessions by `folderGroupKey`, which is the shared +// normalizer under a `folder:` prefix. A hit's `cwd_key` has to be the same +// string, or joining an indexed hit to a sidebar group returns nothing. +const CASES: [name: string, cwd: string][] = [ + ['a POSIX path', '/repo/app'], + ['a trailing slash', '/repo/app/'], + ['a Windows drive', 'C:\\Users\\me\\repo'], + ['a WSL interop mount', '/mnt/c/Users/me/repo'], + ['a WSL UNC path', '\\\\wsl.localhost\\Ubuntu\\home\\me\\repo'], + ['the wsl$ alias for the same path', '//wsl$/Ubuntu/home/me/repo'], + ['a Linux path from inside WSL', '/home/me/repo'], + ['the filesystem root', '/'] +] + +it.each(CASES)('keys %s exactly as the sidebar does', (_name, cwd) => { + expect(`folder:${cwdKey(cwd)}`).toBe(folderGroupKey(cwd)) +}) + +it('keeps the root as a path rather than collapsing it to nothing', () => { + // An empty key is indistinguishable from "no cwd", and the scope filter builds + // its child prefix as `key + '/'`, which would be `//` for an empty key. + expect(cwdKey('/')).toBe('/') +}) + +it('has no key for a session whose cwd the transcript never recorded', () => { + expect(cwdKey(null)).toBeNull() +}) + +it('folds the two WSL UNC aliases onto one key', () => { + expect(cwdKey('\\\\wsl.localhost\\Ubuntu\\home\\me\\repo')).toBe( + cwdKey('//wsl$/ubuntu/home/me/repo') + ) +}) diff --git a/src/main/ai-vault-search/session-search-degraded-roots.ts b/src/main/ai-vault-search/session-search-degraded-roots.ts new file mode 100644 index 00000000000..0432cdbc99f --- /dev/null +++ b/src/main/ai-vault-search/session-search-degraded-roots.ts @@ -0,0 +1,88 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import type { SessionSearchDirectoryReader } from './session-search-directory-listings' + +/** A scan root this pass could not read through, and what stopped it. */ +export type SessionSearchDegradedRoot = { root: string; reason: string } + +/** + * Roots a pass could not read, derived from that pass alone. + * + * There is no root-health state machine any more and nothing is carried between + * passes: "degraded" now means one of two things this pass observed, both of + * which are readdir results. + * + * 1. Discovery recorded a scan issue against the root itself — a stalled WSL + * distro, a gate refusal, an unreadable tree. + * 2. The retirement walk could not prove a file the index holds under that root + * either present or gone, because a directory between the file and the root + * refused to list, or because the root itself is not there. + * 3. A root that yielded no transcripts refuses to list at all. The file walker + * swallows a readdir failure and returns, so without this an EACCES root and + * an agent that was never installed both arrive as "no files" — reporting + * the first as an empty index is the loss-of-contact-as-absence mistake + * docs/reference/ssh-execution-boundary.md forbids. + * + * The second is what reports a detached volume, and it needs no memory of + * previous passes: the evidence is the index's own rows plus this pass's + * readdir errors. A root the index holds nothing under and cannot list is + * reported by the third; a root that is simply missing is not reported at all, + * because that is what an agent nobody installed looks like. + */ +export function scanIssueDegradedRoots( + roots: readonly string[], + issues: readonly AiVaultScanIssue[] +): SessionSearchDegradedRoot[] { + const degraded = new Map() + for (const issue of issues) { + // 'notice' rows are scanner commentary; a per-file failure is not a root's. + if (issue.kind !== 'notice' && roots.includes(issue.path)) { + degraded.set(issue.path, issue.message) + } + } + return [...degraded].map(([root, reason]) => ({ root, reason })) +} + +/** One entry per root, first reason kept, so a pass reports each root once. */ +export function mergeDegradedRoots( + ...groups: readonly (readonly SessionSearchDegradedRoot[])[] +): SessionSearchDegradedRoot[] { + const merged = new Map() + for (const group of groups) { + for (const degraded of group) { + if (!merged.has(degraded.root)) { + merged.set(degraded.root, degraded.reason) + } + } + } + return [...merged].map(([root, reason]) => ({ root, reason })) +} + +// A missing root is not a broken one: an uninstalled agent's root answers +// exactly this, and the index holding rows under it is what the retirement +// walk reports instead. +const MISSING_ROOT = new Set(['ENOENT', 'ENOTDIR']) + +/** + * Roots that yielded no transcripts and cannot be listed either. + * + * Only roots a pass found empty are read: one that returned files is readable + * by construction. The read shares the pass's listing cache, so a root the + * retirement walk also has to ask about costs one readdir between them. + */ +export async function unreadableRoots( + roots: readonly string[], + listings: SessionSearchDirectoryReader, + signal?: AbortSignal +): Promise { + const degraded: SessionSearchDegradedRoot[] = [] + for (const root of roots) { + if (signal?.aborted) { + break + } + const listing = await listings.namesIn(root, signal) + if (!listing.listed && !(listing.code !== null && MISSING_ROOT.has(listing.code))) { + degraded.push({ root, reason: listing.message }) + } + } + return degraded +} diff --git a/src/main/ai-vault-search/session-search-deleted-sources.test.ts b/src/main/ai-vault-search/session-search-deleted-sources.test.ts new file mode 100644 index 00000000000..d1c367d632f --- /dev/null +++ b/src/main/ai-vault-search/session-search-deleted-sources.test.ts @@ -0,0 +1,356 @@ +import { chmod, mkdir, rm, writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { parserPublishesMessages } from '../ai-vault/session-scanner-agent-parser' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { retireDeletedSessionSearchSources } from './session-search-deleted-sources' +import { + SessionSearchDirectoryListings, + type SessionSearchDirectoryListing, + type SessionSearchDirectoryReader +} from './session-search-directory-listings' +import { + openSessionSearchIndexerHarness, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' +import { SessionSearchStore } from './session-search-store' + +// The invariants this file exists to pin are written at the top of +// session-search-deleted-sources.ts. Each one is named in the tests below. + +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 + +let harness: SessionSearchIndexerHarness +let store: SessionSearchStore +let removed: string[] + +beforeEach(async () => { + resetTranscriptConsumersForTests() + harness = await openSessionSearchIndexerHarness('ss-deleted-sources') + removed = [] + store = new SessionSearchStore(harness.databasePath) + // Only the removal matters here; the store's own removal path has its own tests. + store.removeFile = (path: string) => removed.push(path) +}) + +afterEach(async () => { + store.close() + await harness.cleanup() +}) + +/** A reader that answers with whatever a stalled mount would, per directory. */ +function readerAnswering( + answers: Record +): SessionSearchDirectoryReader { + return { + namesIn: (directory) => + Promise.resolve( + answers[directory] ?? { listed: false, code: 'ENOENT', message: 'no such directory' } + ) + } +} + +function retire( + paths: readonly string[], + options: { + roots?: readonly string[] + emptiedRoots?: ReadonlySet + enumeratedContainers?: ReadonlyMap> + listings?: SessionSearchDirectoryReader + directoryLimit?: number + } = {} +) { + return retireDeletedSessionSearchSources({ + store, + paths, + roots: options.roots ?? [harness.roots.claudeProjectsDir ?? ''], + emptiedRoots: options.emptiedRoots, + enumeratedContainers: options.enumeratedContainers, + listings: options.listings ?? new SessionSearchDirectoryListings(), + directoryLimit: options.directoryLimit + }) +} + +// I4: a file the user deleted retires on the first pass that proves it, with no +// waiting period, because its directory listed and it was not in the listing. +it('retires a deleted file the moment its own directory lists without it', async () => { + const kept = join(harness.claudeProjectDir, 'kept.jsonl') + await mkdir(harness.claudeProjectDir, { recursive: true }) + await writeFile(kept, '{}') + const deleted = join(harness.claudeProjectDir, 'deleted.jsonl') + + const result = await retire([kept, deleted]) + expect(result.retired).toEqual([deleted]) + expect(removed).toEqual([deleted]) + // A file that is still there is settled, not watched: it is neither retired + // nor carried into the next pass as unfinished business. + expect(result.unverifiable).toEqual([]) + expect(result.degradedRoots).toEqual([]) +}) + +// I4, the other shape: the directory itself is gone, so the question moves up +// one level and the root answers it. +it('retires a whole project directory the user deleted', async () => { + const sibling = join(harness.roots.claudeProjectsDir ?? '', 'other', 'kept.jsonl') + await mkdir(join(harness.roots.claudeProjectsDir ?? '', 'other'), { recursive: true }) + await writeFile(sibling, '{}') + const gone = join(harness.claudeProjectDir, 'inside-a-deleted-project.jsonl') + + const result = await retire([gone]) + expect(result.retired).toEqual([gone]) + expect(result.unverifiable).toEqual([]) +}) + +// I1 and I2: a root that is not there proves nothing. The walk stops at the +// configured root and never asks what is above it, so a home directory on an +// unmounted volume — the shape a detached drive or a dropped SSH mount takes — +// leaves every row exactly where it was. +it('keeps every row under a root that is not there', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + const held = [join(harness.claudeProjectDir, 'one.jsonl'), join(root, 'flat.jsonl')] + + const result = await retire(held) + expect(result.retired).toEqual([]) + expect(result.unverifiable).toEqual(held) + // The root is named, once, so a caller can say which tree is unreachable. + expect(result.degradedRoots).toEqual([{ root, reason: `${root} could not be listed.` }]) +}) + +// I3: the same answer with no memory at all. Nothing here is carried from a +// previous pass, which is what makes the first sweep after a restart — when a +// volume is most likely to be missing — behave like every other pass. +it('keeps a missing root on a pass that has seen nothing before it', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + const held = join(harness.claudeProjectDir, 'one.jsonl') + const first = await retire([held], { emptiedRoots: new Set() }) + const second = await retire([held], { emptiedRoots: new Set() }) + expect([first.retired, second.retired]).toEqual([[], []]) + expect(second.degradedRoots.map((one) => one.root)).toEqual([root]) +}) + +// I2: an unreadable directory is not an empty one. EACCES stops the walk where +// it is rather than being walked up like a missing component. +it.skipIf(!CAN_DENY_READ)('keeps rows under a directory that refuses to list', async () => { + const blocked = join(harness.roots.claudeProjectsDir ?? '', 'blocked') + await mkdir(blocked, { recursive: true }) + const hidden = join(blocked, 'hidden.jsonl') + await writeFile(hidden, '{}') + await chmod(blocked, 0o000) + try { + const result = await retire([hidden]) + expect(result.retired).toEqual([]) + expect(result.unverifiable).toEqual([hidden]) + expect(result.degradedRoots.map((one) => one.root)).toEqual([harness.roots.claudeProjectsDir]) + } finally { + await chmod(blocked, 0o755) + } +}) + +// I2, without needing a filesystem that can produce it: a stalled network mount +// answers EIO or a WSL gate refusal, and neither is ENOENT. This is the SSH and +// WSL case — loss of contact is never evidence of absence. +it('keeps rows when a directory answers with a transport failure', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + const held = join(harness.claudeProjectDir, 'one.jsonl') + for (const listing of [ + { listed: false as const, code: 'EIO', message: 'input/output error' }, + { listed: false as const, code: 'ETIMEDOUT', message: 'the mount stopped answering' }, + { listed: false as const, code: null, message: 'The distro stopped responding.' } + ]) { + const result = await retire([held], { + listings: readerAnswering({ [harness.claudeProjectDir]: listing }) + }) + expect(result.retired).toEqual([]) + expect(result.degradedRoots).toEqual([{ root, reason: listing.message }]) + } +}) + +// The one bit of memory, and the only thing it buys: a root that held +// transcripts on the previous pass and lists empty on this one gets one pass of +// grace, so a directory swapped out for a moment cannot retire a tree. +it('holds a root that went from holding transcripts to empty in one pass', async () => { + const root = harness.roots.claudeProjectsDir ?? '' + await mkdir(root, { recursive: true }) + const held = join(harness.claudeProjectDir, 'one.jsonl') + + const grace = await retire([held], { emptiedRoots: new Set([root]) }) + expect(grace.retired).toEqual([]) + expect(grace.unverifiable).toEqual([held]) + + // The next pass has no transition to point at, so the empty listing is what + // it says it is: the user emptied the root. + const after = await retire([held], { emptiedRoots: new Set() }) + expect(after.retired).toEqual([held]) +}) + +// A flat-layout agent, where the mountpoint IS the session directory, is the +// one shape the grace exists for: there is no intermediate directory whose +// absence could stop the walk. +it('holds a flat root that emptied in one pass, and retires it on the next', async () => { + const root = harness.roots.copilotSessionsDir ?? '' + await mkdir(root, { recursive: true }) + const held = join(root, 'session.jsonl') + + expect((await retire([held], { roots: [root], emptiedRoots: new Set([root]) })).retired).toEqual( + [] + ) + expect((await retire([held], { roots: [root] })).retired).toEqual([held]) +}) + +// OpenClaw's discovery merges two directories into one delimiter-joined label. +// Roots reach this function as the real directories behind that label, so one +// of them being unreachable never touches the other's rows. +it('judges each merged-root directory on its own', async () => { + const current = join(harness.roots.openclawStateDir ?? '', 'agents') + const legacy = join(harness.roots.openclawLegacyStateDir ?? '', 'agents') + const onMissing = join(current, 'main', 'sessions', 'mounted.jsonl') + const deleted = join(legacy, 'main', 'sessions', 'deleted.jsonl') + await mkdir(join(legacy, 'main', 'sessions'), { recursive: true }) + + const result = await retire([onMissing, deleted], { roots: [current, legacy] }) + expect(result.retired).toEqual([deleted]) + expect(result.unverifiable).toEqual([onMissing]) + expect(result.degradedRoots.map((one) => one.root)).toEqual([current]) +}) + +// A row under no configured root is judged by its own directory and nothing +// above it, so a moved profile is never retired on the strength of a root that +// no longer covers it. +it('judges a row under no configured root by its own directory', async () => { + const orphanDir = join(harness.root, 'moved-profile') + await mkdir(orphanDir, { recursive: true }) + const gone = join(orphanDir, 'gone.jsonl') + const present = join(orphanDir, 'present.jsonl') + await writeFile(present, '{}') + + const result = await retire([gone, present], { roots: [] }) + expect(result.retired).toEqual([gone]) + // No configured root owns it, so nothing is reported as degraded for it. + expect(result.degradedRoots).toEqual([]) +}) + +// I8. A synthetic row names a container and an entry inside it. Walking the +// row's own path would report every one of them gone, and walking only the +// container proves nothing about the entry: a session deleted inside a database +// that is still there would never be retired at all. +it('proves a synthetic row against its container, not against its own path', async () => { + const db = join(harness.root, 'opencode.db') + await writeFile(db, '') + const kept = `${db}#session-1` + const deleted = `${db}#session-2` + const enumeratedContainers = new Map([[db, new Set(['session-1'])]]) + + const result = await retire([kept, deleted], { roots: [], enumeratedContainers }) + expect(result.retired).toEqual([deleted]) + expect(result.unverifiable).toEqual([]) +}) + +it('keeps a synthetic row when this pass did not enumerate its container', async () => { + const db = join(harness.root, 'opencode.db') + await writeFile(db, '') + const row = `${db}#session-1` + + // A cycle asks for the newest N per agent, so a row it did not return may be + // the one after them. It enumerates nothing and therefore proves nothing. + await expect(retire([row], { roots: [] })).resolves.toMatchObject({ + retired: [], + unverifiable: [row] + }) + + // An enumeration that returned nothing at all is not evidence either: a + // database whose schema this scanner no longer recognises reads as empty + // with no error, and believing it would retire every session in one pass. + await expect( + retire([row], { roots: [], enumeratedContainers: new Map([[db, new Set()]]) }) + ).resolves.toMatchObject({ retired: [], unverifiable: [row] }) +}) + +it('retires a synthetic row when the container it came from is gone', async () => { + const db = join(harness.root, 'opencode.db') + await writeFile(db, '') + const row = `${db}#session-1` + const enumeratedContainers = new Map([[db, new Set(['session-1'])]]) + await expect(retire([row], { roots: [], enumeratedContainers })).resolves.toMatchObject({ + retired: [] + }) + + await rm(db) + await expect(retire([row], { roots: [], enumeratedContainers })).resolves.toMatchObject({ + retired: [row] + }) +}) + +// Nothing in this PR can hold a synthetic row: the index pass refuses a source +// whose parser decodes its messages where the message channel cannot reach +// them, and OpenCode's SQLite sessions are read on a worker thread. The rule +// above is the guard for the day that changes -- without it the walk would read +// `#` as a filename and retire every such row the moment it appeared. +it('does not index a source whose messages the channel cannot reach', () => { + const db = join(harness.root, 'opencode.db') + expect( + parserPublishesMessages({ + agent: 'opencode', + codexHome: null, + file: { path: `${db}#session-1`, mtimeMs: 1, modifiedAt: '', sizeBytes: 0 } + }) + ).toBe(false) +}) + +// Round 12, F1. The cap counts directories because that is what costs: rows +// sharing one are a single read and then map lookups. +it('caps the directories one pass reads, not the rows it answers', async () => { + const roots = [harness.claudeProjectDir] + const inside = (folder: string, name: string): string => + join(harness.claudeProjectDir, folder, name) + for (const folder of ['one', 'two', 'three']) { + await mkdir(join(harness.claudeProjectDir, folder), { recursive: true }) + } + // Four rows in each of three directories: three reads, twelve answers. + const paths = ['one', 'two', 'three'].flatMap((folder) => + ['a', 'b', 'c', 'd'].map((name) => inside(folder, name)) + ) + + const result = await retire(paths, { roots, directoryLimit: 2 }) + + // Two directories' worth answered, all eight of their rows, and the third + // directory's four left for the pass after this one. + expect(result.retired).toEqual(paths.slice(0, 8)) + expect(result.unchecked).toEqual(paths.slice(8)) +}) + +// The starvation this replaced: an unreadable directory answers `unverifiable` +// for every row under it and never becomes readable, so a cap on rows let one +// such directory hold the walk for as long as the permission stayed wrong. +it.skipIf(!CAN_DENY_READ)( + 'is not starved by many rows under one unreadable directory', + async () => { + const locked = join(harness.claudeProjectDir, 'locked') + await mkdir(locked, { recursive: true }) + const blocked = Array.from({ length: 520 }, (_unused, index) => + join(locked, `locked-${index}.jsonl`) + ) + const deleted = join(harness.claudeProjectDir, 'deleted.jsonl') + await chmod(locked, 0o000) + try { + const result = await retire([...blocked, deleted], { directoryLimit: 512 }) + + expect(result.retired).toEqual([deleted]) + expect(result.unverifiable).toHaveLength(blocked.length) + expect(result.unchecked).toEqual([]) + } finally { + await chmod(locked, 0o700) + } + } +) + +it('reads each directory once however many files it is asked about', async () => { + await mkdir(harness.claudeProjectDir, { recursive: true }) + const listings = new SessionSearchDirectoryListings() + await retire( + Array.from({ length: 50 }, (_unused, index) => + join(harness.claudeProjectDir, `gone-${index}.jsonl`) + ), + { listings } + ) + expect(listings.size).toBe(1) +}) diff --git a/src/main/ai-vault-search/session-search-deleted-sources.ts b/src/main/ai-vault-search/session-search-deleted-sources.ts new file mode 100644 index 00000000000..e600cd3c4b1 --- /dev/null +++ b/src/main/ai-vault-search/session-search-deleted-sources.ts @@ -0,0 +1,263 @@ +import { basename, dirname } from 'node:path' +import type { SessionSearchDegradedRoot } from './session-search-degraded-roots' +import type { SessionSearchDirectoryReader } from './session-search-directory-listings' +import { isUnderScanRoot } from './session-search-scan-roots' +import { splitSyntheticSessionSource } from './session-search-synthetic-sources' +import type { SessionSearchStore } from './session-search-store' + +/* + * Retirement invariants. Every one of these is a test; changing this file means + * changing the list, not working around it. + * + * I1. A row is retired only when its file is PROVEN gone: some directory + * between the file and its configured root lists successfully, and the next + * path component toward the file is absent from that listing. + * I2. If no directory from the file's parent up to the configured root can be + * listed, nothing is proven and no row is dropped. ENOENT/ENOTDIR is walked + * up (the directory itself is a missing component of some ancestor); + * EACCES, EIO, a WSL gate refusal, anything else, is unverifiable at once. + * I3. The rule is the same on the first pass after a process start and on every + * later pass. It needs no memory of what previous passes saw, because the + * walk is bounded at the configured root and never reasons about what is + * above it. + * I4. A file, or a project directory, the user really deleted retires on the + * first pass that proves it. There is no waiting period and no census. + * I8. A row whose path names an entry inside a container rather than a file of + * its own is proven the same way, one level up: the container must be + * present, and the pass must have enumerated it in full and successfully. + * A listing is a listing whether it comes from readdir or from a database. + * + * What I3 costs, stated rather than hidden: a volume mounted at exactly a + * configured root, unmounted so that the mountpoint stays present and lists + * empty, is indistinguishable from a root the user emptied. It retires. The + * realistic unmount shapes do not: a mount above the root leaves the root + * itself missing (the walk stops at the root boundary), and an unreadable root + * is an error, not a listing. One bit per root buys the remaining grace: a root + * that held transcripts on the previous pass and holds none on this one is + * unverifiable for that pass, so a single flap cannot retire a tree. + */ + +// Walked up rather than believed: a directory that ENOENTs is itself the +// missing component its parent has to be asked about. +const MISSING_DIRECTORY = new Set(['ENOENT', 'ENOTDIR']) + +export type SessionSearchRetirement = { + /** Paths proven gone and dropped from the index. */ + retired: string[] + /** Rows kept: this pass could prove the file neither present nor gone. */ + unverifiable: string[] + /** Paths the per-pass cap left for next time. */ + unchecked: string[] + /** Roots owning at least one unverifiable verdict, with the reason. */ + degradedRoots: SessionSearchDegradedRoot[] +} + +export type SessionSearchRetirementArgs = { + store: SessionSearchStore + /** Held paths this pass did not discover; everything else is still there. */ + paths: readonly string[] + /** The real directories this pass walked; the longest one containing a path bounds its walk. */ + roots: readonly string[] + /** Roots that listed transcripts on the previous pass and none on this one. */ + emptiedRoots?: ReadonlySet + /** + * Containers this pass enumerated in full, with the ids each holds. Only a + * census builds it; see session-search-synthetic-sources.ts for the bar a + * container has to meet before it appears here. + */ + enumeratedContainers?: ReadonlyMap> + /** One readdir per directory per pass, shared with the rest of the pass. */ + listings: SessionSearchDirectoryReader + /** + * Directories this walk may read before the pass moves on. + * + * Directories, not rows. A row whose walk finds its directory already read is + * answered from the pass's cache and costs nothing, so counting rows made an + * unreadable directory able to starve the whole walk: five hundred rows under + * one EACCES directory are one readdir and five hundred identical + * unverifiable verdicts, and a row for a file the user really deleted, sorted + * behind them, was never reached on any pass. + */ + directoryLimit?: number + signal?: AbortSignal +} + +type SessionSearchSourceVerdict = + | { verdict: 'gone' } + | { verdict: 'present' } + | { verdict: 'unverifiable'; reason: string } + +/** + * Retires index rows for sources that are provably gone. + * + * One function, called by both the sweep and the cycle, because either one + * alone deleting a user's history the first time a mount is missing is the bug + * this feature kept shipping. There is no separate root fence: the walk cannot + * reach a verdict of `gone` without a successful listing, so an unreadable or + * missing root produces `unverifiable` structurally rather than by a guard + * somebody has to remember to call (docs/reference/ssh-execution-boundary.md: + * loss of contact is never evidence of absence). + */ +export async function retireDeletedSessionSearchSources( + args: SessionSearchRetirementArgs +): Promise { + const { store, paths, signal } = args + const emptiedRoots = args.emptiedRoots ?? new Set() + const directoryLimit = args.directoryLimit ?? Number.POSITIVE_INFINITY + // Every directory this walk asked for, whether the pass had already read it + // or not. What it bounds is real work: a repeat of one already in here is a + // map lookup, and only a name that is new to it can cost a readdir. + const asked = new Set() + const listings: SessionSearchDirectoryReader = { + namesIn: (directory, signal) => { + asked.add(directory) + return args.listings.namesIn(directory, signal) + } + } + const retirement: SessionSearchRetirement = { + retired: [], + unverifiable: [], + unchecked: [], + degradedRoots: [] + } + const degraded = new Map() + for (const [index, path] of paths.entries()) { + // A synthetic row names a container and an entry inside it, never a file of + // its own; walking the row's own path would report every one of them gone. + const synthetic = splitSyntheticSessionSource(path) + const filePath = synthetic?.container ?? path + // Why capped at all: the sweep hands over every path it holds and did not + // discover, and under an unmount that is the whole index. What is left is + // simply still undiscovered next pass, so the walk finishes over the ones + // that follow rather than holding this one. + // + // Spent past the bound only by a row that starts somewhere new. One this + // walk has already read is answered from the map, so refusing it would buy + // nothing and would leave the budget hostage to whichever directory the + // rows happened to be sorted by. + if (signal?.aborted || (asked.size >= directoryLimit && !asked.has(dirname(filePath)))) { + retirement.unchecked.push(...paths.slice(index)) + break + } + const root = configuredRootFor(filePath, args.roots) + const containerProof = await proveSource(filePath, root ?? dirname(filePath), { + listings, + emptiedRoots, + signal + }) + const proof = synthetic + ? proveSyntheticSource(synthetic, containerProof, args.enumeratedContainers) + : containerProof + if (proof.verdict === 'gone') { + store.removeFile(path) + retirement.retired.push(path) + continue + } + if (proof.verdict === 'present') { + continue + } + retirement.unverifiable.push(path) + // Only a configured root is an alarm worth raising: a row under no root + // this scan walks is already reported on its own, as an orphan. + if (root !== null && !degraded.has(root)) { + degraded.set(root, proof.reason) + } + } + retirement.degradedRoots = [...degraded].map(([root, reason]) => ({ root, reason })) + return retirement +} + +/** + * Walks from the file toward its configured root, asking each directory whether + * the next component toward the file is there. The first directory that answers + * decides; a directory that is itself missing moves the question up one level. + * + * The loop cannot pass the configured root, which is what makes the whole thing + * memoryless: everything above the root — a home directory on an unmounted + * volume, a detached drive, an SSH mount that is not there — is out of scope by + * construction rather than by a state machine that has to remember it. + */ +async function proveSource( + path: string, + root: string, + context: { + listings: SessionSearchDirectoryReader + emptiedRoots: ReadonlySet + signal?: AbortSignal + } +): Promise { + let directory = dirname(path) + let child = basename(path) + while (directory === root || isUnderScanRoot(directory, root)) { + const listing = await context.listings.namesIn(directory, context.signal) + if (!listing.listed) { + if (listing.code !== null && MISSING_DIRECTORY.has(listing.code)) { + const parent = dirname(directory) + if (parent === directory) { + break + } + child = basename(directory) + directory = parent + continue + } + return { verdict: 'unverifiable', reason: listing.message } + } + if (listing.names.has(child)) { + return { verdict: 'present' } + } + if (directory === root && context.emptiedRoots.has(root)) { + // One pass of grace, so a root that blinks empty for a moment — a sync + // client mid-swap, a mount that has not settled — cannot retire a tree. + return { + verdict: 'unverifiable', + reason: 'Listed no transcripts where it listed some on the previous pass.' + } + } + return { verdict: 'gone' } + } + return { verdict: 'unverifiable', reason: `${root} could not be listed.` } +} + +/** + * A synthetic row is proven by its container's own enumeration, one level above + * where the filesystem walk stops. + * + * The container has to be present first: a database on a volume that is not + * there proves nothing about the sessions inside it, and a database that is + * gone takes its sessions with it. Only then does the enumeration decide, and + * only when this pass made one that was exhaustive and successful -- a cycle + * asks for the newest N per agent, so an id it did not return may just be the + * one after them. + */ +function proveSyntheticSource( + synthetic: { container: string; id: string }, + containerProof: SessionSearchSourceVerdict, + enumerated?: ReadonlyMap> +): SessionSearchSourceVerdict { + if (containerProof.verdict !== 'present') { + return containerProof + } + const ids = enumerated?.get(synthetic.container) + // An enumeration that returned nothing at all is not evidence that the + // container holds nothing: a source whose schema this scanner no longer + // recognises reads as empty with no error to see, and believing it would + // retire every entry in one pass. + if (!ids || ids.size === 0) { + return { + verdict: 'unverifiable', + reason: `${synthetic.container} was not enumerated in full this pass.` + } + } + return ids.has(synthetic.id) ? { verdict: 'present' } : { verdict: 'gone' } +} + +/** Longest configured root containing the path, or null for a row under none. */ +function configuredRootFor(path: string, roots: readonly string[]): string | null { + let owner: string | null = null + for (const root of roots) { + if (isUnderScanRoot(path, root) && (owner === null || root.length > owner.length)) { + owner = root + } + } + return owner +} diff --git a/src/main/ai-vault-search/session-search-directory-listings.test.ts b/src/main/ai-vault-search/session-search-directory-listings.test.ts new file mode 100644 index 00000000000..e0143cd7a27 --- /dev/null +++ b/src/main/ai-vault-search/session-search-directory-listings.test.ts @@ -0,0 +1,61 @@ +import { mkdir, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { beforeEach, expect, it, vi } from 'vitest' +import { SessionSearchDirectoryListings } from './session-search-directory-listings' + +const { readdir } = vi.hoisted(() => ({ readdir: vi.fn() })) + +vi.mock('../native-chat/wsl-transcript-fs-access', () => ({ + wslGatedReaddir: readdir +})) + +beforeEach(() => { + readdir.mockReset() +}) + +// A WSL root is a UNC path into the distro, and reading it with raw `fs` is +// what makes a stalled distro look like an empty directory. The gated primitive +// is the same one discovery walks with, so a refusal arrives as an error the +// walk treats as unverifiable rather than as "nothing here". +it('reads through the gated primitive, on the scan lane', async () => { + const unc = '\\\\wsl$\\Ubuntu\\home\\me\\.claude\\projects' + readdir.mockResolvedValueOnce([{ name: 'one.jsonl' }]) + const listings = new SessionSearchDirectoryListings() + + const listing = await listings.namesIn(unc) + + expect(readdir).toHaveBeenCalledWith(unc, 'scan', undefined) + expect(listing).toEqual({ listed: true, names: new Set(['one.jsonl']) }) +}) + +it('reports the code a failed read carried, so ENOENT and EACCES stay apart', async () => { + readdir.mockRejectedValueOnce(Object.assign(new Error('permission denied'), { code: 'EACCES' })) + const listings = new SessionSearchDirectoryListings() + expect(await listings.namesIn('/blocked')).toEqual({ + listed: false, + code: 'EACCES', + message: 'permission denied' + }) +}) + +it('reads a directory once per pass, error or not', async () => { + readdir.mockRejectedValue(Object.assign(new Error('gone'), { code: 'ENOENT' })) + const listings = new SessionSearchDirectoryListings() + await listings.namesIn('/gone') + await listings.namesIn('/gone') + expect(readdir).toHaveBeenCalledTimes(1) + expect(listings.size).toBe(1) +}) + +it('is a real directory read when nothing is mocked out from under it', async () => { + readdir.mockImplementation(async (path: string) => { + const { readdir: real } = await import('node:fs/promises') + return (await real(path, { withFileTypes: true })) as unknown + }) + const root = join(tmpdir(), `ss-listings-${process.pid}`) + await mkdir(root, { recursive: true }) + await writeFile(join(root, 'present.jsonl'), '{}') + const listing = await new SessionSearchDirectoryListings().namesIn(root) + expect(listing.listed && listing.names.has('present.jsonl')).toBe(true) +}) diff --git a/src/main/ai-vault-search/session-search-directory-listings.ts b/src/main/ai-vault-search/session-search-directory-listings.ts new file mode 100644 index 00000000000..f2cb13f9168 --- /dev/null +++ b/src/main/ai-vault-search/session-search-directory-listings.ts @@ -0,0 +1,70 @@ +import { wslGatedReaddir } from '../native-chat/wsl-transcript-fs-access' + +/** One directory read: the names it holds, or what stopped the read. */ +export type SessionSearchDirectoryListing = + | { listed: true; names: ReadonlySet } + | { listed: false; code: string | null; message: string } + +/** + * What the retirement walk needs of a directory: its names, or why not. + * + * An interface rather than the class, so a test can hand the walk an EIO or a + * gate refusal — the shapes a stalled network mount answers with, which no + * temporary directory can be made to produce. + */ +export type SessionSearchDirectoryReader = { + namesIn(directory: string, signal?: AbortSignal): Promise +} + +/** + * Every directory one pass had to read, read once. + * + * The retirement walk asks the same directories about many files — a project + * directory holds hundreds of transcripts — and under an unmount every path + * under a root walks up through the same ancestors. One readdir per directory + * per pass keeps that bounded, and it also makes the pass self-consistent: two + * files in one directory cannot get contradictory verdicts because the + * directory changed between them. + * + * Reads go through the same gated primitive discovery uses, so a WSL UNC path + * is routed to the distro's helper process rather than read with raw fs, and a + * gate refusal arrives as an error rather than as an empty directory. + */ +export class SessionSearchDirectoryListings implements SessionSearchDirectoryReader { + private readonly listings = new Map() + + async namesIn(directory: string, signal?: AbortSignal): Promise { + const cached = this.listings.get(directory) + if (cached) { + return cached + } + const listing = await readDirectory(directory, signal) + this.listings.set(directory, listing) + return listing + } + + /** Directories read this pass; only tests and cost accounting need it. */ + get size(): number { + return this.listings.size + } +} + +async function readDirectory( + directory: string, + signal?: AbortSignal +): Promise { + try { + const entries = await wslGatedReaddir(directory, 'scan', signal) + return { listed: true, names: new Set(entries.map((entry) => entry.name)) } + } catch (error) { + const code = + error && typeof error === 'object' && 'code' in error && typeof error.code === 'string' + ? error.code + : null + return { + listed: false, + code, + message: error instanceof Error ? error.message : String(error) + } + } +} diff --git a/src/main/ai-vault-search/session-search-file-cursor.ts b/src/main/ai-vault-search/session-search-file-cursor.ts new file mode 100644 index 00000000000..2ba978b00ae --- /dev/null +++ b/src/main/ai-vault-search/session-search-file-cursor.ts @@ -0,0 +1,41 @@ +import type { FileWithMtime } from '../ai-vault/session-scanner-types' + +// Why the index keeps its own cursor: the parse cache's cursor answers "what +// does the session list already show", which is a different question from "what +// bytes of this file are already rows". They diverge the moment either side +// declines a read, so neither may consult the other. + +/** Filesystem identity, when discovery could prove it. */ +export type SessionSearchFileIdentity = { dev: number; ino: number } | null + +/** + * What the index holds for one transcript. + * + * A null `byteOffset` is a file the index holds rows for and cannot continue: + * a chunked read committed a prefix, and the reader only hands out an offset + * when a read finishes. Null rather than a flag because every caller that does + * arithmetic on the offset then has to say what it means here, at compile time, + * instead of ignoring a boolean it did not know to read. + */ +export type SessionSearchIndexedFile = { + byteOffset: number | null + mtimeMs: number + sizeBytes: number | null +} + +/** + * Whether this file has to be read from the start, whatever its stat says. + * + * The mtime and size are the real ones, so a freshness check that compares only + * those would call a half-written file current and never re-read it. Every such + * check must start here. + */ +export function requiresWholeRead(indexed: SessionSearchIndexedFile | null): boolean { + return indexed !== null && indexed.byteOffset === null +} + +export function fileIdentity(file: FileWithMtime): SessionSearchFileIdentity { + return typeof file.dev === 'number' && typeof file.ino === 'number' + ? { dev: file.dev, ino: file.ino } + : null +} diff --git a/src/main/ai-vault-search/session-search-file-records.ts b/src/main/ai-vault-search/session-search-file-records.ts new file mode 100644 index 00000000000..2ee79cbdc13 --- /dev/null +++ b/src/main/ai-vault-search/session-search-file-records.ts @@ -0,0 +1,134 @@ +import { fileIdentity } from './session-search-file-cursor' +import type { AiVaultSession } from '../../shared/ai-vault-types' +import type { TranscriptSessionIdentity } from '../ai-vault/session-transcript-consumers' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type SyncDatabase from '../sqlite/sync-database' +import { EMPTY_CONTENT_HASH, type SessionContentHash } from './session-search-content-hash' +import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' + +/** + * The stored comparison key for a session's working directory. + * + * Why the shared normalizer verbatim: the sidebar already groups sessions by + * `folderGroupKey`, which is this function under a prefix. A second spelling + * here means any later join between an indexed hit and a sidebar group returns + * nothing. An earlier version qualified a WSL cwd with its distro so two + * distros could not collide at `/home/me/repo`; that is a real collision, but it + * is one every SSH host has too, neither key qualifies for SSH, and the fix for + * it is a column that names the execution host, not a path key that only some + * hosts spell differently. + */ +export function cwdKey(cwd: string | null): string | null { + return cwd ? normalizeRuntimePathForComparison(cwd) : null +} + +export class SessionSearchFileRecords { + constructor(private readonly db: SyncDatabase) {} + /** + * The row a read hangs its messages off, before the parser has said what the + * session is. The same transaction fills it in: from the decoded session when + * the read finished, and from `updateProvisionalSession` when this is a chunk + * of one that has not. + */ + createSessionRow(candidate: SessionFileCandidate): number { + return Number( + this.db + .prepare( + `INSERT INTO sessions(agent,session_id,file_path,title,resume_command) + VALUES (?,'',?,'','')` + ) + .run(candidate.agent, candidate.file.path).lastInsertRowid + ) + } + + /** + * Writes what the parser knows so far onto a session a chunk is committing. + * + * Rows a chunk commits answer searches the moment they land, so the session + * they hang off has to be nameable before the read producing it ends — and it + * may never end, because a crash between chunks leaves exactly this row. That + * is why the identity is required rather than optional: a read that has none + * does not chunk at all. The final commit overwrites all of it from the + * decoded session; until then the title in particular is provisional. + */ + updateProvisionalSession(rowId: number, identity: TranscriptSessionIdentity): void { + this.db + .prepare( + `UPDATE sessions SET session_id = ?, title = ?, cwd = ?, cwd_key = ?, + created_at = ?, updated_at = ? WHERE id = ?` + ) + .run( + identity.sessionId, + identity.title ?? '', + identity.cwd, + cwdKey(identity.cwd), + identity.createdAt, + identity.updatedAt, + rowId + ) + } + + contentHash(rowId: number): SessionContentHash { + const row = this.db + .prepare('SELECT content_hash, content_hash_count FROM sessions WHERE id = ?') + .get(rowId) as { content_hash: string | null; content_hash_count: number } | undefined + return row ? { hash: row.content_hash, count: row.content_hash_count } : EMPTY_CONTENT_HASH + } + + updateSession(session: AiVaultSession, rowId: number, contentHash: SessionContentHash): void { + const values = [ + session.agent, + session.sessionId, + session.filePath, + session.codexHome, + session.title, + session.cwd, + cwdKey(session.cwd), + session.branch, + session.createdAt, + session.updatedAt, + session.messageCount, + session.resumeCommand, + contentHash.hash, + contentHash.count + ] + this.db + .prepare( + `UPDATE sessions SET agent = ?, session_id = ?, file_path = ?, codex_home = ?, title = ?, + cwd = ?, cwd_key = ?, branch = ?, created_at = ?, updated_at = ?, message_count = ?, resume_command = ?, + content_hash = ?, content_hash_count = ? WHERE id = ?` + ) + .run(...values, rowId) + } + + upsertFile( + candidate: SessionFileCandidate, + byteOffset: number, + sessionRowId: number | null + ): void { + const { file } = candidate + const identity = fileIdentity(file) + this.db + .prepare( + `INSERT INTO files(path, dev, ino, byte_offset, mtime_ms, size_bytes, session_row_id) + VALUES (?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(path) DO UPDATE SET + -- Partial observations must never create a pair that no stat proved. + dev = CASE WHEN excluded.dev IS NOT NULL AND excluded.ino IS NOT NULL + THEN excluded.dev ELSE files.dev END, + ino = CASE WHEN excluded.dev IS NOT NULL AND excluded.ino IS NOT NULL + THEN excluded.ino ELSE files.ino END, + byte_offset = excluded.byte_offset, mtime_ms = excluded.mtime_ms, + size_bytes = excluded.size_bytes, session_row_id = excluded.session_row_id` + ) + .run( + file.path, + identity?.dev ?? null, + identity?.ino ?? null, + byteOffset, + file.mtimeMs, + file.sizeBytes ?? null, + sessionRowId + ) + } +} diff --git a/src/main/ai-vault-search/session-search-file-write.test.ts b/src/main/ai-vault-search/session-search-file-write.test.ts new file mode 100644 index 00000000000..2c0be88b026 --- /dev/null +++ b/src/main/ai-vault-search/session-search-file-write.test.ts @@ -0,0 +1,623 @@ +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import SyncDatabase from '../sqlite/sync-database' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { cwdKey } from './session-search-file-records' +import { requiresWholeRead } from './session-search-file-cursor' +import { SessionSearchIndexWriter } from './session-search-index-writer' +import { deleteExpiredSearchFiles } from './session-search-retention-delete' +import { + openSessionSearchIndexFile, + replayTranscriptRead, + syntheticCandidate, + syntheticSession, + SYNTHETIC_TRANSCRIPT, + userMessages, + type SessionSearchIndexFile +} from './session-search-index-test-fixture' +import { SessionSearchStore } from './session-search-store' + +// Every assertion here reads through `index.db`, a second connection to the same +// file. That is the whole consistency model: one transaction per file in WAL +// mode, so another handle sees the last committed state and never a session part +// way through being rewritten. + +let index: SessionSearchIndexFile +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + index = await openSessionSearchIndexFile('ss-file-write') + errors = [] + store = new SessionSearchStore(index.path, (error) => errors.push(error)) + registerSessionSearchIndexConsumer(store) +}) + +afterEach(async () => { + vi.restoreAllMocks() + resetTranscriptConsumersForTests() + store.close() + await index.close() +}) + +function matches(db: SyncDatabase, table: string, term: string): number { + return ( + db + .prepare( + `SELECT count(*) AS n FROM ${table} JOIN messages m ON m.id = ${table}.rowid + JOIN sessions s ON s.id = m.session_row_id WHERE ${table} MATCH ?` + ) + .get(term) as { n: number } + ).n +} + +/** Fails the nth statement matching `pick`, wherever the writer prepares it. */ +function failOnStatement(pick: (sql: string) => boolean, nth: number): void { + const prepare = SyncDatabase.prototype.prepare + let seen = 0 + vi.spyOn(SyncDatabase.prototype, 'prepare').mockImplementation(function ( + this: SyncDatabase, + sql: string + ) { + if (pick(sql) && ++seen === nth) { + throw new Error('index write crashed mid transaction') + } + return prepare.call(this, sql) + }) +} + +function counts(db: SyncDatabase): Record { + const one = (sql: string): number => (db.prepare(sql).get() as { n: number }).n + return { + sessions: one('SELECT count(*) AS n FROM sessions'), + messages: one('SELECT count(*) AS n FROM messages'), + files: one('SELECT count(*) AS n FROM files'), + full: one('SELECT count(*) AS n FROM messages_fts') + } +} + +it('writes a whole read in one transaction', () => { + replayTranscriptRead({ messages: userMessages('needle text', 300) }) + + const after = counts(index.db) + expect(after.sessions).toBe(1) + expect(after.messages).toBe(300) + expect(after.full).toBe(300) + expect(errors).toEqual([]) +}) + +it('files every row in one FTS table, under the column its role owns', () => { + replayTranscriptRead({ + messages: [ + { role: 'user', text: 'alpha question', timestamp: null }, + { role: 'assistant', text: 'beta answer', timestamp: null }, + { role: 'tool', text: 'gamma tool output', timestamp: null } + ] + }) + + // One table carries all three; the conversation scope is a column filter over + // it, which is what the second table used to be. + expect(counts(index.db).full).toBe(3) + expect(matches(index.db, 'messages_fts', 'gamma')).toBe(1) + expect(matches(index.db, 'messages_fts', '{user_text assistant_text}: gamma')).toBe(0) + expect(matches(index.db, 'messages_fts', '{user_text assistant_text}: beta')).toBe(1) +}) + +it('leaves the index exactly as it found it when a read never finishes', () => { + const write = store.beginWrite(syntheticCandidate(), 'replace', 0)! + for (const message of userMessages('neverfinished', 200)) { + write.add(message) + } + // The process dies here: the rows only ever existed in this buffer. + expect(counts(index.db)).toMatchObject({ + sessions: 0, + messages: 0, + files: 0 + }) +}) + +it('rolls a whole file back when a write throws part way through its transaction', () => { + replayTranscriptRead({ + messages: userMessages('firstgeneration', 3), + outcome: { byteOffset: 40 } + }) + const before = counts(index.db) + + failOnStatement((sql) => sql.startsWith('INSERT INTO messages('), 50) + replayTranscriptRead({ + messages: userMessages('crashedgeneration', 100), + outcome: { byteOffset: 900 } + }) + vi.restoreAllMocks() + + // Not one of the 49 rows that were already inserted survived, the previous + // generation is untouched, and the cursor still describes what is really here. + expect(counts(index.db)).toEqual(before) + expect(matches(index.db, 'messages_fts', 'crashedgeneration')).toBe(0) + expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(3) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(40) + expect(errors).toHaveLength(1) + // The row itself says the read failed, which is the only reason anything was + // lost and the only record that outlives this read. + expect( + index.db.prepare('SELECT state, fail_count FROM files WHERE path = ?').get(SYNTHETIC_TRANSCRIPT) + ).toMatchObject({ state: 'failed', fail_count: 1 }) + + // And the connection is usable again: a transaction left open by the failure + // would take down every write after it, not just the one that threw. + replayTranscriptRead({ + messages: userMessages('afterthecrash', 2), + outcome: { byteOffset: 900 } + }) + expect(matches(index.db, 'messages_fts', 'afterthecrash')).toBe(2) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(900) +}) + +it('takes the rows back when recording the cursor is what fails', () => { + replayTranscriptRead({ + messages: userMessages('firstgeneration', 3), + outcome: { byteOffset: 40 } + }) + + // The cursor is written last, so this is the crash point that would leave rows + // no cursor describes: a later append would continue from an offset those rows + // already cover, and index the same span twice. + failOnStatement((sql) => sql.startsWith('INSERT INTO files('), 1) + replayTranscriptRead({ + messages: userMessages('crashedgeneration', 5), + outcome: { byteOffset: 900 } + }) + vi.restoreAllMocks() + + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 3 }) + expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(3) + expect(matches(index.db, 'messages_fts', 'crashedgeneration')).toBe(0) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(40) +}) + +it('shows a reader on another handle one generation or the other, never a mixture', async () => { + replayTranscriptRead({ + messages: userMessages('firstgeneration', 3), + outcome: { byteOffset: 40 } + }) + expect(counts(index.db).messages).toBe(3) + + const write = store.beginWrite(syntheticCandidate(), 'replace', 0)! + for (const message of userMessages('secondgeneration', 7)) { + write.add(message) + // Every point at which the other handle could issue a query mid-read. + expect(counts(index.db).messages).toBe(3) + expect(matches(index.db, 'messages_fts', 'secondgeneration')).toBe(0) + } + expect( + write.commit({ + session: syntheticSession(), + byteOffset: 900, + incomplete: false + }) + ).toBe(true) + + expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(0) + expect(matches(index.db, 'messages_fts', 'secondgeneration')).toBe(7) + // The old three are cut loose, not deleted, so they are still on disk and + // already unreachable; the drain the store scheduled hands them back. + expect(counts(index.db).messages).toBe(10) + await vi.waitFor(() => { + expect(counts(index.db).messages).toBe(7) + }) +}) + +// Four of these fill the 400-char ceiling the two tests below construct. +const CHUNKED_MESSAGE = `chunkedneedle ${'filler '.repeat(12)}nd` + +const PROVISIONAL_IDENTITY = { + sessionId: 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee', + cwd: '/repo/app', + title: 'provisional title', + createdAt: '2026-05-01T10:00:00.000Z', + updatedAt: '2026-05-01T10:05:00.000Z' +} + +// Only a read that can name its session chunks at all, so every test below that +// wants a chunk has to supply one. +const named = (): typeof PROVISIONAL_IDENTITY => PROVISIONAL_IDENTITY + +it('leaves the session consistent after every chunk of a file too large for one transaction', () => { + expect(CHUNKED_MESSAGE.length).toBe(100) + const writer = new SessionSearchIndexWriter(index.db, 400) + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)! + for (const [position, message] of userMessages(CHUNKED_MESSAGE, 10).entries()) { + write.add(message) + const rows = counts(index.db).messages + // Four messages per chunk, and nothing else reaches the file between them. + expect(rows).toBe(Math.floor((position + 1) / 4) * 4) + // Whatever landed is a coherent prefix of this session and answers searches. + expect(matches(index.db, 'messages_fts', 'chunkedneedle')).toBe(rows) + if (rows > 0) { + // The cursor a chunk leaves refuses every append rather than inventing an + // offset the reader never gave it. + expect(requiresWholeRead(writer.indexedFile(SYNTHETIC_TRANSCRIPT, null))).toBe(true) + expect(writer.beginWrite(syntheticCandidate(), 'append', 0)).toBeNull() + } + } + expect(counts(index.db).messages).toBe(8) + + expect( + write.commit({ + session: syntheticSession(), + byteOffset: 4096, + incomplete: false + }) + ).toBe(true) + expect(counts(index.db)).toMatchObject({ + sessions: 1, + messages: 10, + full: 10 + }) + expect(writer.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(4096) +}) + +it('holds the ceiling against a single message larger than it', () => { + const writer = new SessionSearchIndexWriter(index.db, 8000) + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)! + const exec = SyncDatabase.prototype.exec + let opened = 0 + vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function ( + this: SyncDatabase, + sql: string + ) { + if (sql === 'BEGIN IMMEDIATE') { + opened += 1 + } + exec.call(this, sql) + }) + + // One conversation turn, three times the ceiling. Checked once per message, + // this commits all 24,000 characters in a single transaction — the ceiling + // bounds nothing that a message can exceed on its own. + write.add({ role: 'assistant', text: 'a'.repeat(24_000), timestamp: null }) + vi.restoreAllMocks() + + expect(opened).toBe(3) + expect(counts(index.db).messages).toBe(3) + expect( + write.commit({ + session: syntheticSession(), + byteOffset: 4096, + incomplete: false + }) + ).toBe(true) + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 3 }) +}) + +it('names a session on its first chunk, not only when the read ends', () => { + const writer = new SessionSearchIndexWriter(index.db, 400) + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)! + for (const message of userMessages(CHUNKED_MESSAGE, 10)) { + write.add(message) + } + + // The chunks that landed already answer searches, so the session they hang + // off has to be nameable on another handle before the read ends. This is also + // the whole record a crash between chunks leaves behind. + expect(counts(index.db).messages).toBe(8) + expect( + index.db.prepare('SELECT session_id, cwd, cwd_key, title, created_at FROM sessions').get() + ).toEqual({ + session_id: PROVISIONAL_IDENTITY.sessionId, + cwd: '/repo/app', + cwd_key: cwdKey('/repo/app'), + title: 'provisional title', + created_at: '2026-05-01T10:00:00.000Z' + }) + + // And the decoded session still wins at the end: the mid-read title is + // provisional, never a value the final commit has to defer to. + expect( + write.commit({ + session: syntheticSession({ title: 'the settled title' }), + byteOffset: 4096, + incomplete: false + }) + ).toBe(true) + expect(index.db.prepare('SELECT title FROM sessions').get()).toEqual({ + title: 'the settled title' + }) +}) + +it('commits a whole-file read over the ceiling in one transaction, never a chunk', () => { + // The whole-file readers (Grok, Cursor, Gemini, OpenCode) pass no identity: + // their formats are rewritten in place and have no resumable state to ask. + const writer = new SessionSearchIndexWriter(index.db, 400) + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0)! + const exec = SyncDatabase.prototype.exec + let opened = 0 + vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function ( + this: SyncDatabase, + sql: string + ) { + if (sql === 'BEGIN IMMEDIATE') { + opened += 1 + } + exec.call(this, sql) + }) + + for (const message of userMessages(CHUNKED_MESSAGE, 10)) { + write.add(message) + // Chunking here would publish rows under a session with an empty id, an + // empty title and a null cwd, and an interrupted read would leave that + // prefix answering searches for good. + expect(counts(index.db)).toMatchObject({ sessions: 0, messages: 0, files: 0 }) + } + expect(write.commit({ session: syntheticSession(), byteOffset: 4096, incomplete: false })).toBe( + true + ) + vi.restoreAllMocks() + + expect(opened).toBe(1) + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 10, full: 10 }) + // And a real cursor, not the partial sentinel a chunk would have left. + expect(writer.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(4096) +}) + +it('starts chunking only once the parser has an id to name the session with', () => { + const writer = new SessionSearchIndexWriter(index.db, 400) + let decoded: typeof PROVISIONAL_IDENTITY | null = null + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, () => decoded)! + for (const message of userMessages(CHUNKED_MESSAGE, 4)) { + write.add(message) + } + // Past the ceiling, but the parser has decoded nothing: the buffer keeps + // growing rather than naming a session it cannot name. + expect(counts(index.db).messages).toBe(0) + + decoded = PROVISIONAL_IDENTITY + write.add(userMessages(CHUNKED_MESSAGE, 1)[0]!) + + // Everything held goes with the first chunk that can say what it is. + expect(counts(index.db).messages).toBe(5) + expect(index.db.prepare('SELECT session_id, cwd FROM sessions').get()).toEqual({ + session_id: PROVISIONAL_IDENTITY.sessionId, + cwd: '/repo/app' + }) +}) + +it('reports a chunk-partial file as held, and as one that must be read whole', () => { + const writer = new SessionSearchIndexWriter(index.db, 400) + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)! + for (const message of userMessages(CHUNKED_MESSAGE, 10)) { + write.add(message) + } + + // Held, with no cursor to continue. Reporting nothing here reads as "never + // indexed", so a caller asks for whatever read the parse cache offers, the + // reader picks append, and only a decline heals it a cycle later. + const held = writer.indexedFile(SYNTHETIC_TRANSCRIPT, null) + expect(held).not.toBeNull() + expect(held?.byteOffset).toBeNull() + expect(requiresWholeRead(held)).toBe(true) + expect(held?.mtimeMs).toBe(syntheticCandidate().file.mtimeMs) + + // A file this index has never seen is still the other answer, so the two + // states a caller has to tell apart are distinguishable. + expect(writer.indexedFile('/never-seen.jsonl', null)).toBeNull() + expect(requiresWholeRead(null)).toBe(false) + + // And no offset continues it, including the one the chunk recorded. + for (const offset of [0, -1, 400, 1000]) { + expect(writer.beginWrite(syntheticCandidate(), 'append', offset)).toBeNull() + } +}) + +it('re-reads a chunked file whole when its writer died between chunks', async () => { + const writer = new SessionSearchIndexWriter(index.db, 400) + const abandoned = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)! + for (const message of userMessages(CHUNKED_MESSAGE, 10)) { + abandoned.add(message) + } + expect(counts(index.db).messages).toBe(8) + + // Nothing can continue that prefix, so the only way forward is a whole re-read, + // and that replaces every row the dead writer left. + expect(requiresWholeRead(writer.indexedFile(SYNTHETIC_TRANSCRIPT, null))).toBe(true) + const replacement = writer.beginWrite(syntheticCandidate(), 'replace', 0)! + replacement.add(userMessages('wholereread', 1)[0]!) + expect( + replacement.commit({ + session: syntheticSession(), + byteOffset: 4096, + incomplete: false + }) + ).toBe(true) + // The eight stranded rows stop answering the moment the replace commits, and + // the drain hands them back after it rather than inside it. + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 9 }) + expect(matches(index.db, 'messages_fts', 'chunkedneedle')).toBe(0) + await deleteExpiredSearchFiles(index.db, null, () => false) + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 1 }) +}) + +it('stops a chunked read whose file was removed between its chunks', () => { + const writer = new SessionSearchIndexWriter(index.db, 400) + const write = writer.beginWrite(syntheticCandidate(), 'replace', 0, named)! + const messages = userMessages(CHUNKED_MESSAGE, 10) + for (const message of messages.slice(0, 4)) { + write.add(message) + } + expect(counts(index.db).messages).toBe(4) + + writer.removeFile(SYNTHETIC_TRANSCRIPT) + const exec = SyncDatabase.prototype.exec + let opened = 0 + vi.spyOn(SyncDatabase.prototype, 'exec').mockImplementation(function ( + this: SyncDatabase, + sql: string + ) { + if (sql === 'BEGIN IMMEDIATE') { + opened += 1 + } + exec.call(this, sql) + }) + for (const message of messages.slice(4)) { + write.add(message) + } + expect(write.commit({ session: syntheticSession(), byteOffset: 4096, incomplete: false })).toBe( + false + ) + vi.restoreAllMocks() + + // Not one row of the removed source came back. The read stopped at the first + // refusal rather than reopening a transaction it already knows will roll back, + // once for every message left in a file that may be a hundred megabytes. + expect(opened).toBe(1) + expect(counts(index.db)).toMatchObject({ sessions: 0, messages: 0, files: 0, full: 0 }) +}) + +it('fences a first-ever read whose file was removed before it committed', () => { + const candidate = syntheticCandidate({ path: '/never-indexed.jsonl' }) + const write = store.beginWrite(candidate, 'replace', 0)! + for (const message of userMessages('removedbeforefirstcommit', 3)) { + write.add(message) + } + // The path was never indexed, so there is no cursor for the removal to move. + // PR 3's retirement sweep removes exactly these: paths the index deferred over + // budget and never wrote, while the registered consumer is fed concurrently. + store.removeFile('/never-indexed.jsonl') + + expect(write.commit({ session: syntheticSession(), byteOffset: 300, incomplete: false })).toBe( + false + ) + expect(counts(index.db)).toMatchObject({ sessions: 0, messages: 0, files: 0, full: 0 }) +}) + +it('replaces the previous generation without ever showing both', async () => { + replayTranscriptRead({ messages: userMessages('firstgeneration', 10) }) + replayTranscriptRead({ messages: userMessages('secondgeneration', 10) }) + + expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(0) + expect(matches(index.db, 'messages_fts', 'secondgeneration')).toBe(10) + await vi.waitFor(() => { + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 10, full: 10 }) + }) +}) + +it('replaces a generation by cutting the old one loose, not by deleting it inline', async () => { + const writer = new SessionSearchIndexWriter(index.db) + const first = writer.beginWrite(syntheticCandidate(), 'replace', 0)! + for (const message of userMessages('firstgeneration', 200)) { + first.add(message) + } + expect(first.commit({ session: syntheticSession(), byteOffset: 100, incomplete: false })).toBe( + true + ) + const before = index.db.prepare('SELECT id FROM sessions').get() as { id: number } + expect(counts(index.db).messages).toBe(200) + + const second = writer.beginWrite(syntheticCandidate(), 'replace', 0)! + for (const message of userMessages('secondgeneration', 3)) { + second.add(message) + } + expect(second.commit({ session: syntheticSession(), byteOffset: 200, incomplete: false })).toBe( + true + ) + + // The transaction inserted three rows and deleted one, rather than deleting + // two hundred: all 203 are still on disk, and the old 200 already answer + // nothing, because every retrieval joins `sessions`. + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 203, full: 203 }) + expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(0) + expect(matches(index.db, 'messages_fts', 'secondgeneration')).toBe(3) + + // A new session row, with `files` repointed at it in that same transaction. + // AUTOINCREMENT never hands the freed id back while orphans still name it. + const after = index.db.prepare('SELECT id FROM sessions').get() as { id: number } + expect(after.id).toBeGreaterThan(before.id) + expect(index.db.prepare('SELECT session_row_id FROM files').get()).toEqual({ + session_row_id: after.id + }) + + await deleteExpiredSearchFiles(index.db, null, () => false) + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 3, full: 3 }) +}) + +it('drains what a replace cut loose without being asked', async () => { + replayTranscriptRead({ messages: userMessages('firstgeneration', 200) }) + replayTranscriptRead({ messages: userMessages('secondgeneration', 3) }) + + // The store schedules the reclaim the way it schedules retention's. Hiding a + // generation and never reclaiming it would grow the file by every re-read. + expect(matches(index.db, 'messages_fts', 'firstgeneration')).toBe(0) + await vi.waitFor(() => { + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 3, full: 3 }) + }) + expect(errors).toEqual([]) +}) + +it('continues a session across an append rather than replaying it', () => { + replayTranscriptRead({ + messages: userMessages('openingturn', 3), + outcome: { byteOffset: 40 } + }) + replayTranscriptRead({ + messages: userMessages('laterturn', 2), + mode: 'append', + previousByteOffset: 40, + outcome: { byteOffset: 90 } + }) + + expect(counts(index.db)).toMatchObject({ sessions: 1, messages: 5 }) + expect(matches(index.db, 'messages_fts', 'openingturn')).toBe(3) + expect(matches(index.db, 'messages_fts', 'laterturn')).toBe(2) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(90) +}) + +it('stops answering for a removed file the moment it is removed', () => { + replayTranscriptRead({ messages: userMessages('removedneedle', 3) }) + store.removeFile(SYNTHETIC_TRANSCRIPT) + + expect(counts(index.db)).toMatchObject({ + sessions: 0, + messages: 0, + files: 0, + full: 0 + }) + expect(matches(index.db, 'messages_fts', 'removedneedle')).toBe(0) +}) + +it('writes nothing for an incomplete read and owes the file a whole re-read', () => { + replayTranscriptRead({ + messages: userMessages('incompleteread', 300), + outcome: { incomplete: true } + }) + + expect(counts(index.db)).toMatchObject({ + sessions: 0, + messages: 0, + full: 0 + }) + // One row, holding nothing but the failure: an incomplete read indexes no + // content, and the count of how often it has happened at this stat is the + // only thing that stops the file being read again on every pass. + expect(index.db.prepare('SELECT byte_offset, state, fail_count FROM files').get()).toMatchObject({ + byte_offset: 0, + state: 'failed', + fail_count: 1 + }) + expect(errors).toEqual([]) +}) + +it('exposes the handle a composed reader queries through', () => { + replayTranscriptRead({ messages: userMessages('composedreader', 3) }) + + // PR 4's engine reads through this rather than opening a second connection, + // so it sees a write the moment the transaction commits. + expect(store.connection.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ n: 3 }) +}) + +it('closes twice without turning the second call into an error', () => { + store.close() + // node:sqlite throws ERR_INVALID_STATE on a second close of one handle, and a + // store is closed both by whoever owns it and by a teardown that cannot know. + expect(() => store.close()).not.toThrow() + store = new SessionSearchStore(index.path, (error) => errors.push(error)) +}) diff --git a/src/main/ai-vault-search/session-search-identifier-split.test.ts b/src/main/ai-vault-search/session-search-identifier-split.test.ts new file mode 100644 index 00000000000..24f8a67d5c3 --- /dev/null +++ b/src/main/ai-vault-search/session-search-identifier-split.test.ts @@ -0,0 +1,29 @@ +import { expect, it } from 'vitest' +import { identifierShadowTerms, identifierShadowText } from './session-search-identifier-split' + +it('splits a camel-case symbol into its pieces and keeps the whole', () => { + expect(identifierShadowTerms('call resolveTerminalPath here')).toEqual([ + 'resolveterminalpath', + 'resolve', + 'terminal', + 'path' + ]) +}) + +it('splits a path into its segments and extension', () => { + // The whole path already tokenizes on its own; only the pieces need shadowing. + expect(identifierShadowText('src/main/foo-bar.ts')).toBe('src main foo bar ts') +}) + +it('leaves ordinary prose alone', () => { + expect(identifierShadowTerms('the quick brown fox')).toEqual([]) +}) + +it('shadows a screaming-case constant', () => { + expect(identifierShadowTerms('MAX_RETRIES')).toEqual(['max', 'retries']) +}) + +it('stops at the term limit rather than growing with the message', () => { + const text = Array.from({ length: 50 }, (_unused, index) => `alpha_beta${index}`).join(' ') + expect(identifierShadowTerms(text, 10)).toHaveLength(10) +}) diff --git a/src/main/ai-vault-search/session-search-identifier-split.ts b/src/main/ai-vault-search/session-search-identifier-split.ts new file mode 100644 index 00000000000..e2df1822cfa --- /dev/null +++ b/src/main/ai-vault-search/session-search-identifier-split.ts @@ -0,0 +1,54 @@ +// Identifier shadow terms: `resolveTerminalPath` → `resolve terminal path`, +// `src/main/foo-bar.ts` → `src main foo bar ts`. Stored in a separate FTS5 +// column so a partial identifier still matches; the largest single accuracy +// win measured in the retrieval shoot-out (MRR 0.50 → 0.55). + +const RAW_TOKEN = /[A-Za-z0-9_./-]+/g +const CAMEL_PIECE = /[A-Z]+(?![a-z])|[A-Z][a-z0-9]*|[a-z0-9]+/g +const SEPARATOR = /[_./-]+/ +// Worth shadowing: has a separator, a camel boundary, or is SCREAMING_CASE. +const INTERESTING = /[_./-]|[a-z0-9][A-Z]|^[A-Z]{2,}[0-9_]*$/ +const MIN_TOKEN = 3 +const MAX_TOKEN = 120 +const MIN_PIECE = 2 + +function hasMixedCase(piece: string): boolean { + return /[a-z]/.test(piece) && /[A-Z]/.test(piece) +} + +export function identifierShadowTerms(text: string, limit = 4000): string[] { + const out: string[] = [] + const seen = new Set() + for (const match of text.matchAll(RAW_TOKEN)) { + const token = match[0] + if (token.length < MIN_TOKEN || token.length > MAX_TOKEN || !INTERESTING.test(token)) { + continue + } + const parts: string[] = [] + for (const piece of token.split(SEPARATOR)) { + if (!piece) { + continue + } + parts.push(piece) + if (hasMixedCase(piece)) { + parts.push(...(piece.match(CAMEL_PIECE) ?? [])) + } + } + for (const part of parts) { + const lowered = part.toLowerCase() + if (lowered.length < MIN_PIECE || seen.has(lowered)) { + continue + } + seen.add(lowered) + out.push(lowered) + if (out.length >= limit) { + return out + } + } + } + return out +} + +export function identifierShadowText(text: string, limit?: number): string { + return identifierShadowTerms(text, limit).join(' ') +} diff --git a/src/main/ai-vault-search/session-search-index-consumer.test.ts b/src/main/ai-vault-search/session-search-index-consumer.test.ts new file mode 100644 index 00000000000..ee855409fb7 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-consumer.test.ts @@ -0,0 +1,342 @@ +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { + openSessionSearchIndexFile, + replayTranscriptRead, + syntheticCandidate, + syntheticSession, + SYNTHETIC_TRANSCRIPT, + userMessages, + type SessionSearchIndexFile +} from './session-search-index-test-fixture' +import { SessionSearchStore } from './session-search-store' + +let index: SessionSearchIndexFile +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + index = await openSessionSearchIndexFile('ss-index-consumer') + errors = [] + store = new SessionSearchStore(index.path, (error) => errors.push(error)) + registerSessionSearchIndexConsumer(store) +}) + +afterEach(async () => { + resetTranscriptConsumersForTests() + store.close() + await index.close() +}) + +function indexedMessages(): number { + return ( + index.db.prepare('SELECT count(*) AS n FROM messages').get() as { + n: number + } + ).n +} + +function cursor(): number | null | undefined { + return store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset +} + +/** What the row itself says it still owes, which is the only record there is. */ +function owed(): { state: string; fail_count: number } | undefined { + return index.db + .prepare('SELECT state, fail_count FROM files WHERE path = ?') + .get(SYNTHETIC_TRANSCRIPT) as { state: string; fail_count: number } | undefined +} + +it('appends onto its own cursor and carries the content hash forward', async () => { + replayTranscriptRead({ + messages: userMessages('first half', 3), + outcome: { byteOffset: 100 } + }) + const first = index.db + .prepare('SELECT content_hash AS hash, content_hash_count AS count FROM sessions') + .get() as { hash: string; count: number } + + replayTranscriptRead({ + mode: 'append', + previousByteOffset: 100, + messages: userMessages('second half', 2), + outcome: { byteOffset: 220 } + }) + + expect(indexedMessages()).toBe(5) + expect(cursor()).toBe(220) + const second = index.db + .prepare('SELECT content_hash AS hash, content_hash_count AS count FROM sessions') + .get() as { hash: string; count: number } + expect(second.count).toBe(first.count + 2) + expect(second.hash).not.toBe(first.hash) + expect(owed()).toMatchObject({ state: 'current', fail_count: 0 }) +}) + +it('appends onto a file it read through and decoded no session from', async () => { + // An excluded Codex worker transcript: read through, nothing to index, and + // still growing. Its cursor is sound, so a re-read of the whole file every + // pass buys nothing. + replayTranscriptRead({ + messages: userMessages('excluded span', 3), + outcome: { session: null, byteOffset: 100 } + }) + expect(cursor()).toBe(100) + expect(owed()).toMatchObject({ state: 'current', fail_count: 0 }) + + replayTranscriptRead({ + mode: 'append', + previousByteOffset: 100, + messages: userMessages('decoded at last', 2), + outcome: { byteOffset: 220 } + }) + + expect(indexedMessages()).toBe(2) + expect(cursor()).toBe(220) + expect(owed()).toMatchObject({ state: 'current', fail_count: 0 }) +}) + +it('declines an append that starts past its own cursor and records the file', async () => { + replayTranscriptRead({ + messages: userMessages('indexed span', 3), + outcome: { byteOffset: 100 } + }) + + // The session list read further than this index did, so the appended span + // continues from bytes the index never saw. + replayTranscriptRead({ + mode: 'append', + previousByteOffset: 900, + messages: userMessages('unseen span', 4), + outcome: { byteOffset: 1200 } + }) + + expect(indexedMessages()).toBe(3) + expect(cursor()).toBe(100) + expect(owed()).toMatchObject({ state: 'due' }) +}) + +it('declines a file whose identity changed under the same path', async () => { + const original = syntheticCandidate({ dev: 1, ino: 10 }) + replayTranscriptRead({ + candidate: original, + messages: userMessages('original file', 2), + outcome: { byteOffset: 100 } + }) + + replayTranscriptRead({ + candidate: syntheticCandidate({ dev: 1, ino: 77 }), + mode: 'append', + previousByteOffset: 100, + messages: userMessages('replacement file', 2), + outcome: { byteOffset: 200 } + }) + + expect(indexedMessages()).toBe(2) + expect(owed()?.state).not.toBe('current') +}) + +it('never advances the cursor for an incomplete read', async () => { + replayTranscriptRead({ + messages: userMessages('complete span', 3), + outcome: { byteOffset: 100 } + }) + + replayTranscriptRead({ + mode: 'append', + previousByteOffset: 100, + messages: userMessages('partial span', 5), + outcome: { byteOffset: 400, incomplete: true } + }) + + expect(indexedMessages()).toBe(3) + expect(cursor()).toBe(100) + expect( + ( + index.db.prepare('SELECT count(*) AS n FROM messages').get() as { + n: number + } + ).n + ).toBe(3) + expect(owed()?.state).not.toBe('current') +}) + +it('indexes nothing at all from a read that was incomplete from the start', async () => { + replayTranscriptRead({ + messages: userMessages('unreachable', 4), + outcome: { byteOffset: 0, incomplete: true } + }) + + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 0 + }) + expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ + n: 0 + }) + // No cursor, because nothing was read through. The row exists all the same: + // it is where the failure is counted, and a file that fails on its first read + // is exactly the one that has no row of its own to count on. + expect(cursor()).toBe(0) + expect(owed()).toMatchObject({ state: 'failed', fail_count: 1 }) +}) + +it('drops a file whose parser returned no session', async () => { + replayTranscriptRead({ + messages: userMessages('was indexed', 3), + outcome: { byteOffset: 100 } + }) + + replayTranscriptRead({ + messages: userMessages('now rejected', 2), + outcome: { session: null, byteOffset: 300 } + }) + + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 0 + }) + expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ + n: 0 + }) + // The file is still read through, so a later scan does not re-read it. + expect(cursor()).toBe(300) +}) + +it('writes nothing for a source whose parser cannot reach the channel', async () => { + // An OpenCode SQLite candidate decodes in a worker, so every read of it is + // incomplete, and no re-read would help. + const candidate = { + ...syntheticCandidate({ path: '/opencode/opencode.db#session-1' }), + agent: 'opencode' as const + } + replayTranscriptRead({ + candidate, + messages: [], + outcome: { byteOffset: 0, incomplete: true } + }) + + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 0 + }) + // No row at all, which is the record: the next pass reads a path the + // file table does not name. + expect(owed()).toBeUndefined() +}) + +it('ignores a candidate older than the retention cutoff', async () => { + store.setRetentionCutoffMs(Date.now()) + replayTranscriptRead({ messages: userMessages('too old', 3) }) + + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 0 + }) + // No row at all, which is the record: the next pass reads a path the + // file table does not name. + expect(owed()).toBeUndefined() +}) + +it('keeps the session list running when the index write fails', async () => { + replayTranscriptRead({ + messages: userMessages('healthy', 2), + outcome: { byteOffset: 100 } + }) + index.db.exec('DROP TABLE messages_fts') + + expect(() => + replayTranscriptRead({ + mode: 'append', + previousByteOffset: 100, + messages: userMessages('broken', 400), + outcome: { byteOffset: 500 } + }) + ).not.toThrow() + expect(errors.length).toBeGreaterThan(0) + expect(owed()?.state).not.toBe('current') +}) + +it('unregisters cleanly, leaving later reads unindexed', async () => { + resetTranscriptConsumersForTests() + replayTranscriptRead({ messages: userMessages('after unregister', 3) }) + + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 0 + }) +}) + +it('drops a removed source and keeps its cursor gone', async () => { + replayTranscriptRead({ + messages: userMessages('present', 3), + outcome: { byteOffset: 100 } + }) + store.removeFile(SYNTHETIC_TRANSCRIPT) + + expect(cursor()).toBeUndefined() + expect(index.db.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 0 + }) + expect(index.db.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ + n: 0 + }) +}) + +it('writes the session metadata the read decoded', async () => { + replayTranscriptRead({ + messages: userMessages('metadata', 1), + outcome: { + session: syntheticSession({ + sessionId: 'abc-123', + title: 'a titled session', + cwd: '/repo/app', + branch: 'main', + messageCount: 1, + resumeCommand: 'claude --resume abc-123' + }), + byteOffset: 42 + } + }) + + expect( + index.db + .prepare('SELECT session_id, title, cwd, cwd_key, branch, resume_command FROM sessions') + .get() + ).toEqual({ + session_id: 'abc-123', + title: 'a titled session', + cwd: '/repo/app', + cwd_key: '/repo/app', + branch: 'main', + resume_command: 'claude --resume abc-123' + }) +}) + +it('keeps a proven file identity when a later read cannot stat it', async () => { + const withIdentity = syntheticCandidate({ dev: 1, ino: 10 }) + replayTranscriptRead({ + candidate: withIdentity, + messages: userMessages('first', 2), + outcome: { byteOffset: 100 } + }) + + // A host that cannot prove identity re-reads the same file. + replayTranscriptRead({ + candidate: syntheticCandidate(), + mode: 'append', + previousByteOffset: 100, + messages: userMessages('second', 2), + outcome: { byteOffset: 200 } + }) + expect(indexedMessages()).toBe(4) + + // The stored identity survived, so a rename-replace is still detectable. + replayTranscriptRead({ + candidate: syntheticCandidate({ dev: 1, ino: 99 }), + mode: 'append', + previousByteOffset: 200, + messages: userMessages('replacement', 2), + outcome: { byteOffset: 300 } + }) + + expect(indexedMessages()).toBe(4) + expect(cursor()).toBe(200) + expect(owed()?.state).not.toBe('current') +}) diff --git a/src/main/ai-vault-search/session-search-index-consumer.ts b/src/main/ai-vault-search/session-search-index-consumer.ts new file mode 100644 index 00000000000..a457ca5e7c3 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-consumer.ts @@ -0,0 +1,146 @@ +import { parserPublishesMessages } from '../ai-vault/session-scanner-agent-parser' +import { + registerTranscriptConsumer, + type TranscriptConsumer, + type TranscriptMessage, + type TranscriptReadConsumer, + type TranscriptReadOutcome, + type TranscriptReadStart +} from '../ai-vault/session-transcript-consumers' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import { fileIdentity } from './session-search-file-cursor' +import type { SessionSearchFileWrite } from './session-search-index-writer' +import type { SessionSearchStore } from './session-search-store' + +/** + * The search index as a consumer of the transcript reader. + * + * It keeps its own cursor in the `files` table and never consults the parse + * cache: the two answer different questions and diverge the moment either + * declines a read. + * + * Every refusal leaves the cursor where it was and writes what the next pass + * needs on the row itself, because the row is the only thing that outlives this + * read. A declined append is `due`: the index is behind on a span no append + * reaches, so the file has to be read whole. A read that started and did not + * commit is `failed`, counted, and stamped with the stat it failed at, which is + * what stops an unreadable transcript being retried on every pass for ever. + */ +export class SessionSearchIndexConsumer implements TranscriptConsumer { + constructor(private readonly store: SessionSearchStore) {} + + beginRead(start: TranscriptReadStart): TranscriptReadConsumer | null { + const { candidate } = start + if (!parserPublishesMessages(candidate)) { + this.noteUnreachableParser(candidate) + return null + } + if (start.mode === 'append') { + const cursor = this.store.indexedFile(candidate.file.path, fileIdentity(candidate.file)) + if (!cursor || cursor.byteOffset !== start.previousByteOffset) { + // This index never saw the span before `previousByteOffset`; appending + // here would leave a hole no later read can fill. A null cursor is the + // file a chunked read left half written, which no offset continues. + // Either way the next pass has to read this file from the start. + this.store.setFileState(candidate.file.path, 'due') + return null + } + } + const write = this.store.beginWrite( + candidate, + start.mode, + start.previousByteOffset, + start.identity + ) + if (!write) { + // A closed store, a candidate outside the retention window, or a row that + // moved under this read. Only a row that exists has anything to record. + this.store.setFileState(candidate.file.path, 'due') + return null + } + return new SessionSearchReadConsumer(this.store, start, write) + } + + /** + * A source no read can ever index, recorded as one this index has seen. + * + * A parser that decodes where the message channel cannot reach it -- OpenCode's + * SQLite sessions today -- publishes nothing, so no read of it will ever + * commit a row. Leaving the file table silent about it is not free: the next + * pass sees a path the index holds nothing for, asks for a read, and asking + * over a warm cache drops the session list's own resume point. The sidebar's + * fold is thrown away and the whole database is decoded again, on every pass, + * for ever. + * + * The row written is the shape the store already has for a read that went + * through and decoded no session: cursor at the file's size, no session row. + * The decide step then skips it until its stat moves, and the retirement walk + * retires it like any other row when it goes. + */ + private noteUnreachableParser(candidate: SessionFileCandidate): void { + const write = this.store.beginWrite(candidate, 'replace', 0) + const committed = + write?.commit({ + session: null, + byteOffset: candidate.file.sizeBytes ?? 0, + incomplete: false + }) === true + if (committed) { + this.store.writeCommitted(candidate) + } + } +} + +class SessionSearchReadConsumer implements TranscriptReadConsumer { + private failed = false + + constructor( + private readonly store: SessionSearchStore, + private readonly start: TranscriptReadStart, + private readonly write: SessionSearchFileWrite + ) {} + + message(message: TranscriptMessage): void { + if (this.failed) { + return + } + try { + this.write.add(message) + } catch (error) { + // Never throws back into the reader: the channel would drop this consumer + // for the rest of the read and `finish` would never run. Failing here + // keeps the whole read on one path — the buffer is dropped and the file is + // re-read. + this.failed = true + this.store.reportWriteFailure(error) + } + } + + finish(outcome: TranscriptReadOutcome): void { + const { candidate } = this.start + let committed = false + try { + // An incomplete read's rows are not the whole span, so the cursor must not + // move past them; the file is re-read whole instead. + committed = !this.failed && !outcome.incomplete && this.write.commit(outcome) + } catch (error) { + this.store.reportWriteFailure(error) + } + if (committed) { + this.store.writeCommitted(candidate) + return + } + // Counted against the stat it failed at, not merely recorded: a transcript + // the reader cannot open fails identically on every pass, and only a change + // to this stat can mean the file itself changed. + this.store.setFileState(candidate.file.path, 'failed', candidate.file.mtimeMs) + } +} + +/** + * Registers the index with the reader and returns the unregister function. + * Nothing in production calls this yet: PR 3 owns when the index is live. + */ +export function registerSessionSearchIndexConsumer(store: SessionSearchStore): () => void { + return registerTranscriptConsumer(new SessionSearchIndexConsumer(store)) +} diff --git a/src/main/ai-vault-search/session-search-index-pass.test.ts b/src/main/ai-vault-search/session-search-index-pass.test.ts new file mode 100644 index 00000000000..c32eec59a96 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-pass.test.ts @@ -0,0 +1,194 @@ +import { appendFile, rm, stat, utimes } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { runSessionSearchIndexPass } from './session-search-index-pass' +import { parseTranscript } from './session-search-transcript-fixtures' +import { + claudeLines, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' +import { discoverSessionSearchCandidates } from './session-search-scan-roots' +import { SessionSearchStore } from './session-search-store' + +const FIRST = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +const SECOND = 'bbbbbbbb-cccc-4ddd-8eee-ffffffffffff' + +let harness: SessionSearchIndexerHarness +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + harness = await openSessionSearchIndexerHarness('ss-index-pass') + await writeClaudeTranscript(transcript(FIRST), ['the first transcript'], FIRST) + await writeClaudeTranscript(transcript(SECOND), ['the second transcript'], SECOND) + store = openStore() +}) + +afterEach(async () => { + resetTranscriptConsumersForTests() + store.close() + await harness.cleanup() +}) + +function transcript(sessionId: string): string { + return join(harness.claudeProjectDir, `${sessionId}.jsonl`) +} + +function openStore(): SessionSearchStore { + const opened = new SessionSearchStore(harness.databasePath, (error) => errors.push(error)) + registerSessionSearchIndexConsumer(opened) + return opened +} + +async function candidates() { + return ( + await discoverSessionSearchCandidates(harness.roots, { + limitPerAgent: Number.POSITIVE_INFINITY + }) + ).candidates +} + +/** What a pass hands the read loop: the store's rows, read once. */ +function rows() { + return new Map(store.files().map((row) => [row.path, row])) +} + +function pass(options: { overdue?: () => boolean } = {}) { + return runSessionSearchIndexPass(store, [], { rows: rows(), ...options }) +} + +async function passOverAll(options: { overdue?: () => boolean } = {}) { + return runSessionSearchIndexPass(store, await candidates(), { rows: rows(), ...options }) +} + +function states(): Record { + return Object.fromEntries(store.files().map((row) => [row.path, row.state])) +} + +it('re-reads nothing it already holds, even with a cold session-list cache', async () => { + const first = await passOverAll() + expect(first.stats.fullParses).toBe(2) + + // A restart: the parse cache is gone, the index's `files` table is not. + store.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store = openStore() + + const second = await passOverAll() + expect(second.stats).toMatchObject({ fullParses: 0, incremental: 0, reused: 0, bytesRead: 0 }) + expect(errors).toEqual([]) +}) + +it('resumes into a grown transcript instead of re-reading it whole', async () => { + await passOverAll() + await appendFile(transcript(FIRST), `${claudeLines(['a later turn'], FIRST, 10).join('\n')}\n`) + + const second = await passOverAll() + expect(second.stats).toMatchObject({ incremental: 1, fullParses: 0 }) +}) + +// Nothing is recorded about what a deadline cut off, because being owed is a +// fact about the row: the file is read on the next pass for the same reason it +// was owed on this one. +it('leaves what it ran out of time for owed, with nothing written down', async () => { + const all = await candidates() + const cut = await runSessionSearchIndexPass(store, all, { rows: rows(), overdue: () => true }) + + expect(cut.outOfTime).toBe(true) + expect(store.files()).toHaveLength(1) + const second = await passOverAll() + expect(second.stats.fullParses).toBe(1) + expect(store.files()).toHaveLength(2) +}) + +// The deadline is never applied before the pass has read anything, so a single +// transcript larger than one deadline is read alone rather than starved. +it('reads one file even when the deadline has already expired', async () => { + const only = (await candidates()).slice(0, 1) + const alone = await runSessionSearchIndexPass(store, only, { rows: rows(), overdue: () => true }) + + expect(alone.outOfTime).toBe(false) + expect(store.files()).toHaveLength(1) +}) + +it('skips a source the reader cannot even open without failing the pass', async () => { + const all = await candidates() + await rm(transcript(FIRST)) + await runSessionSearchIndexPass(store, all, { rows: rows() }) + + // One session indexed, and the missing one recorded as a failed read rather + // than as content the index holds. + expect(harness.read((db) => db.prepare('SELECT count(*) AS n FROM sessions').get())).toEqual({ + n: 1 + }) + expect(states()[transcript(FIRST)]).toBe('failed') +}) + +// Finding 6: mtime alone is not the freshness key. A transcript that grows +// while keeping its mtime (a same-second append, a restored timestamp) is a +// different file to the index, and reading only mtime would skip it forever. +it('re-reads a file that grew without its mtime moving', async () => { + const path = transcript(FIRST) + // A whole-millisecond stamp, so restoring it later reproduces it exactly. + const frozen = new Date(1_740_000_000_000) + await utimes(path, frozen, frozen) + await passOverAll() + + await appendFile(path, `${claudeLines(['a same-mtime append'], FIRST, 20).join('\n')}\n`) + await utimes(path, frozen, frozen) + expect((await stat(path)).mtimeMs).toBe(frozen.getTime()) + + const second = await passOverAll() + expect(second.stats.fullParses + second.stats.incremental).toBe(1) +}) + +// Finding 5: the decision reads the session list's cache and then changes it, +// so outside the per-path lane an overlapping list parse stores its entry in +// between and the forced read degrades into a reuse. +it('is not overtaken by a list parse racing the same path', async () => { + const path = transcript(FIRST) + const all = await candidates() + const only = all.filter((candidate) => candidate.file.path === path) + + // The list parses this path first, so its cursor covers the file, and again + // concurrently with the index's pass so the two interleave. + await parseTranscript(path) + await Promise.all([ + parseTranscript(path), + runSessionSearchIndexPass(store, only, { rows: rows() }) + ]) + + expect(harness.read((db) => db.prepare('SELECT count(*) AS n FROM sessions').get())).toEqual({ + n: 1 + }) +}) + +// Finding 4d: a declined read is a parse that returns normally and indexes +// nothing. It has to leave the row owing a read, not looking covered. +it('leaves a declined read owed rather than recorded as held', async () => { + const only = (await candidates()).slice(0, 1) + // What a store that refuses a write looks like from the consumer's side: the + // read runs, and nothing is written. + store.beginWrite = () => null + + const stats = await runSessionSearchIndexPass(store, only, { rows: rows() }) + + expect(stats.stats.fullParses).toBe(1) + expect(harness.read((db) => db.prepare('SELECT count(*) AS n FROM sessions').get())).toEqual({ + n: 0 + }) + expect(store.files()).toEqual([]) +}) + +it('reads nothing when there is nothing to read', async () => { + expect((await pass()).stats).toMatchObject({ fullParses: 0 }) +}) diff --git a/src/main/ai-vault-search/session-search-index-pass.ts b/src/main/ai-vault-search/session-search-index-pass.ts new file mode 100644 index 00000000000..05ccea1a314 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-pass.ts @@ -0,0 +1,84 @@ +import { throwIfAiVaultScanCancelled } from '../ai-vault/ai-vault-scan-cancellation' +import { + createSessionParseStats, + parseAgentSessionFileCached, + type SessionParseStats +} from '../ai-vault/session-scanner-parse-cache' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import { fileIdentity } from './session-search-file-cursor' +import { sessionSearchReadDecision } from './session-search-read-decision' +import type { SessionSearchFileRow, SessionSearchStore } from './session-search-store' + +export type SessionSearchIndexPassOptions = { + signal?: AbortSignal + /** The store's rows for this pass, read once. Absent means the index holds nothing. */ + rows: ReadonlyMap + /** + * True once the pass has spent its wall-clock deadline. The one bound on how + * long a pass reads for: files and bytes are proxies for time, and the thing + * worth capping is the share of the wall clock an unasked background index + * takes. Never applied before the pass has read anything, so an oversized + * transcript is read alone rather than deferred for ever. + */ + overdue?: () => boolean +} + +/** + * Reads whatever the decide step says is owed, until the deadline. + * + * Nothing is recorded about what it did not reach. A candidate the deadline cut + * off is still owed on the next pass for the same reason it was owed on this + * one — its row says so — so there is no queue to keep, nothing to bound, and + * nothing to drop. What the reads themselves leave behind is written by the + * index consumer onto the rows. + */ +export async function runSessionSearchIndexPass( + store: SessionSearchStore, + candidates: readonly SessionFileCandidate[], + options: SessionSearchIndexPassOptions +): Promise<{ stats: SessionParseStats; outOfTime: boolean }> { + const stats = createSessionParseStats() + const cutoffMs = store.retentionCutoff + let read = 0 + let outOfTime = false + for (const candidate of candidates) { + throwIfAiVaultScanCancelled(options.signal) + const path = candidate.file.path + const row = options.rows.get(path) + const decision = sessionSearchReadDecision({ + candidate, + row, + // Only asked for a path the index holds something for; for the rest the + // decision is already made and this would be a query per new file. + cursor: row ? store.indexedFile(path, fileIdentity(candidate.file)) : null, + cutoffMs + }) + if (decision === 'skip') { + continue + } + // The decide step is one cursor lookup, so it runs for the whole list even + // once the deadline has gone: knowing what is owed costs nothing, and the + // count of what a pass left is worth more than the microseconds. + outOfTime ||= read > 0 && options.overdue?.() === true + if (outOfTime) { + continue + } + // The clock the deadline reads is one the owner may close behind: the read + // below writes to the store, so stop here rather than on a shut handle. + throwIfAiVaultScanCancelled(options.signal) + read += 1 + try { + await parseAgentSessionFileCached(candidate, process.platform, stats, decision) + } catch (error) { + throwIfAiVaultScanCancelled(options.signal) + // The reader reports a read it could not finish to the consumer, which is + // what records the failure on the row; nothing is counted here. + console.warn( + '[ai-vault-search] indexing skipped', + candidate.agent, + error instanceof Error ? error.name : 'ParseError' + ) + } + } + return { stats, outOfTime } +} diff --git a/src/main/ai-vault-search/session-search-index-test-fixture.ts b/src/main/ai-vault-search/session-search-index-test-fixture.ts new file mode 100644 index 00000000000..baa3e2976fe --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-test-fixture.ts @@ -0,0 +1,124 @@ +import { mkdtemp } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { removeTree } from '../../shared/windows-transient-lock-removal' +import type { AiVaultSession } from '../../shared/ai-vault-types' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import { TranscriptMessageChannel } from '../ai-vault/session-transcript-channel' +import type { + TranscriptMessage, + TranscriptReadOutcome, + TranscriptReadStart +} from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { openSessionSearchDatabase } from './session-search-schema' + +export const SYNTHETIC_TRANSCRIPT = 'synthetic-transcript' + +export function syntheticCandidate( + overrides: Partial = {} +): SessionFileCandidate { + const at = new Date(1740000000000) + return { + agent: 'claude', + codexHome: null, + file: { + path: SYNTHETIC_TRANSCRIPT, + mtimeMs: at.getTime(), + modifiedAt: at.toISOString(), + sizeBytes: 4096, + ...overrides + } + } +} + +export function syntheticSession(overrides: Partial = {}): AiVaultSession { + const at = new Date(1740000000000).toISOString() + return { + id: 'fixture', + executionHostId: 'local', + agent: 'claude', + sessionId: 'fixture', + title: 'fixture session', + cwd: '/fixture', + branch: null, + model: null, + filePath: SYNTHETIC_TRANSCRIPT, + codexHome: null, + createdAt: at, + updatedAt: at, + modifiedAt: at, + messageCount: 0, + totalTokens: 0, + previewMessages: [], + queuedMessageCount: 0, + subagentTranscriptCount: 0, + resumeCommand: '', + subagent: null, + ...overrides + } +} + +export function userMessages(text: string, count: number): TranscriptMessage[] { + return Array.from({ length: count }, (_unused, index) => ({ + role: 'user' as const, + text, + timestamp: new Date(1740000000000 + index * 1000).toISOString() + })) +} + +/** + * Drives one read through the real fan-out channel, so a test exercises the + * registration path the transcript reader uses rather than the consumer alone. + */ +export function replayTranscriptRead(args: { + candidate?: SessionFileCandidate + mode?: TranscriptReadStart['mode'] + previousByteOffset?: number + messages: TranscriptMessage[] + outcome?: Partial +}): void { + const candidate = args.candidate ?? syntheticCandidate() + const mode = args.mode ?? 'replace' + const channel = new TranscriptMessageChannel() + channel.beginRead({ + candidate, + mode, + previousByteOffset: args.previousByteOffset ?? 0 + }) + for (const message of args.messages) { + channel.push(message) + } + channel.finishRead({ + session: syntheticSession(), + byteOffset: 4096, + incomplete: false, + ...args.outcome + }) +} + +export type SessionSearchIndexFile = { + path: string + /** The store keeps its own connection private, so row assertions need this one. */ + db: SyncDatabase + close: () => Promise +} + +/** An on-disk index: `:memory:` is per-connection, so a second reader needs a real file. */ +export async function openSessionSearchIndexFile(name: string): Promise { + const root = await mkdtemp(join(tmpdir(), `${name}-`)) + const path = join(root, 'index.sqlite') + const db = openSessionSearchDatabase(path) + let open = true + return { + path, + db, + close: async () => { + if (open) { + open = false + db.close() + } + await removeTree(root) + } + } +} diff --git a/src/main/ai-vault-search/session-search-index-writer.test.ts b/src/main/ai-vault-search/session-search-index-writer.test.ts new file mode 100644 index 00000000000..1be12e35bc7 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-writer.test.ts @@ -0,0 +1,226 @@ +import { afterEach, beforeEach, expect, it } from 'vitest' +import { SessionSearchIndexConsumer } from './session-search-index-consumer' +import { + openSessionSearchIndexFile, + syntheticCandidate, + syntheticSession, + SYNTHETIC_TRANSCRIPT, + userMessages, + type SessionSearchIndexFile +} from './session-search-index-test-fixture' +import { SessionSearchStore } from './session-search-store' + +// The store is driven directly here. Every guard below is also shadowed by the +// consumer's own check, so a test that goes through the consumer proves nothing +// about which of the two is holding. + +let index: SessionSearchIndexFile +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + index = await openSessionSearchIndexFile('ss-index-writer') + errors = [] + store = new SessionSearchStore(index.path, (error) => errors.push(error)) +}) + +afterEach(async () => { + store.close() + await index.close() +}) + +function count(table: string): number { + return ( + index.db.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { + n: number + } + ).n +} + +function indexRead(previousByteOffset: number, byteOffset: number, text: string): boolean { + const write = store.beginWrite( + syntheticCandidate(), + previousByteOffset === 0 ? 'replace' : 'append', + previousByteOffset + ) + if (!write) { + return false + } + for (const message of userMessages(text, 2)) { + write.add(message) + } + return write.commit({ + session: syntheticSession(), + byteOffset, + incomplete: false + }) +} + +it('refuses an append whose predecessor offset is not the committed cursor', () => { + expect(indexRead(0, 100, 'first')).toBe(true) + + expect(store.beginWrite(syntheticCandidate(), 'append', 900)).toBeNull() + expect(store.beginWrite(syntheticCandidate(), 'append', 99)).toBeNull() + // The one offset that does continue the committed span is accepted. + expect(store.beginWrite(syntheticCandidate(), 'append', 100)).not.toBeNull() +}) + +it('refuses to commit a write whose cursor moved underneath it', () => { + const stale = store.beginWrite(syntheticCandidate(), 'replace', 0)! + for (const message of userMessages('stalegeneration', 40)) { + stale.add(message) + } + // A second read of the same path finishes first. Without the parse file lane + // this is the overlap that would otherwise resurrect the stale rows. + expect(indexRead(0, 200, 'winninggeneration')).toBe(true) + + expect( + stale.commit({ + session: syntheticSession(), + byteOffset: 100, + incomplete: false + }) + ).toBe(false) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)?.byteOffset).toBe(200) + expect(count('sessions')).toBe(1) + expect(count('messages')).toBe(2) + expect(errors).toEqual([]) +}) + +it('refuses to commit a write whose file was removed mid-read', () => { + expect(indexRead(0, 100, 'firstgeneration')).toBe(true) + const write = store.beginWrite(syntheticCandidate(), 'append', 100)! + for (const message of userMessages('afterremoval', 10)) { + write.add(message) + } + store.removeFile(SYNTHETIC_TRANSCRIPT) + + // Committing here would put a source back that its owner proved was deleted. + expect( + write.commit({ + session: syntheticSession(), + byteOffset: 300, + incomplete: false + }) + ).toBe(false) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, null)).toBeNull() + expect(count('sessions')).toBe(0) + expect(count('messages')).toBe(0) + expect(count('files')).toBe(0) +}) + +it('declines a behind cursor in beginRead before it ever reaches the store', () => { + const attempted: number[] = [] + const stub = { + indexedFile: () => ({ byteOffset: 100, mtimeMs: 1, sizeBytes: 1 }), + beginWrite: (_candidate: unknown, _mode: unknown, previousByteOffset: number) => { + attempted.push(previousByteOffset) + return { add: () => undefined, commit: () => true } + }, + setFileState: () => undefined + } as unknown as SessionSearchStore + const consumer = new SessionSearchIndexConsumer(stub) + + expect( + consumer.beginRead({ + candidate: syntheticCandidate(), + mode: 'append', + previousByteOffset: 900 + }) + ).toBeNull() + // The store was never asked, so the writer's own guard cannot be what refused. + expect(attempted).toEqual([]) + expect( + consumer.beginRead({ + candidate: syntheticCandidate(), + mode: 'append', + previousByteOffset: 100 + }) + ).not.toBeNull() + expect(attempted).toEqual([100]) +}) + +it("hands the read's identity accessor to the store", () => { + const captured: unknown[] = [] + const stub = { + indexedFile: () => null, + beginWrite: ( + _candidate: unknown, + _mode: unknown, + _previousByteOffset: unknown, + identity: unknown + ) => { + captured.push(identity) + return { add: () => undefined, commit: () => true } + }, + setFileState: () => undefined + } as unknown as SessionSearchStore + const identity = (): null => null + + new SessionSearchIndexConsumer(stub).beginRead({ + candidate: syntheticCandidate(), + mode: 'replace', + previousByteOffset: 0, + identity + }) + + // Dropped here, a chunked read writes rows under a session with no id and no + // cwd for as long as the read lasts, and for ever if it crashes first. + expect(captured).toEqual([identity]) +}) + +it('treats half a recorded identity as no identity at all', () => { + // New partial observations are not stored as identities. + const partial = { + ...syntheticCandidate({ dev: 7 }), + agent: 'claude' as const + } + const write = store.beginWrite(partial, 'replace', 0)! + for (const message of userMessages('halfidentity', 2)) { + write.add(message) + } + write.commit({ + session: syntheticSession(), + byteOffset: 100, + incomplete: false + }) + expect(index.db.prepare('SELECT dev, ino FROM files').get()).toEqual({ + dev: null, + ino: null + }) + // Older indexes may still carry a half-pair. + index.db.exec('UPDATE files SET dev = 7') + + // One matching number is not proof of sameness, and one mismatching number is + // not proof of replacement. Neither compares, so neither declines. + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, { dev: 7, ino: 99 })?.byteOffset).toBe(100) + expect(store.indexedFile(SYNTHETIC_TRANSCRIPT, { dev: 8, ino: 99 })?.byteOffset).toBe(100) + expect(store.beginWrite(syntheticCandidate({ dev: 8, ino: 99 }), 'append', 100)).not.toBeNull() +}) + +it.each([ + [null, { dev: null, ino: null }], + [ + { dev: 7, ino: 11 }, + { dev: 7, ino: 11 } + ] +])('never combines partial stats with the previous identity %j', (initial, expected) => { + const observations = [initial ?? {}, { dev: 9 }, { ino: 13 }, { dev: 17, ino: 19 }] + for (const [position, identity] of observations.entries()) { + const write = store.beginWrite( + syntheticCandidate(identity), + position ? 'append' : 'replace', + position * 100 + )! + expect( + write.commit({ + session: syntheticSession(), + byteOffset: (position + 1) * 100, + incomplete: false + }) + ).toBe(true) + expect(index.db.prepare('SELECT dev, ino FROM files').get()).toEqual( + position === 3 ? { dev: 17, ino: 19 } : expected + ) + } +}) diff --git a/src/main/ai-vault-search/session-search-index-writer.ts b/src/main/ai-vault-search/session-search-index-writer.ts new file mode 100644 index 00000000000..a5c981b5bb2 --- /dev/null +++ b/src/main/ai-vault-search/session-search-index-writer.ts @@ -0,0 +1,359 @@ +import type SyncDatabase from '../sqlite/sync-database' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type { + TranscriptMessage, + TranscriptReadOutcome, + TranscriptSessionIdentity +} from '../ai-vault/session-transcript-consumers' +import { EMPTY_CONTENT_HASH, foldContentHash } from './session-search-content-hash' +import type { + SessionSearchFileIdentity, + SessionSearchIndexedFile +} from './session-search-file-cursor' +import { SessionSearchFileRecords } from './session-search-file-records' +import { + deleteSearchMessages, + insertSearchMessage, + searchMessageRows +} from './session-search-message-rows' + +/** + * How much decoded text one transaction may carry. + * + * A file's rows are buffered in memory and written in one transaction, so the + * whole read is either in the index or not. The ceiling is what keeps that + * promise affordable: at the measured 26 MB of transcript per second it caps a + * single commit near a second and the WAL it produces near 64 MB, and it is far + * above the largest real transcript (the 40-session benchmark corpus is 10.5 MB + * in total), so an ordinary file never reaches it. Above the ceiling the read is + * cut into chunks that each leave the index consistent — but only a read that + * can name its session chunks at all. See `add`. + */ +export const SESSION_SEARCH_COMMIT_CHARS = 32 * 1024 * 1024 + +/** + * The cursor of a file whose rows are a prefix, written by a chunk of a read + * that has not reached the end of the file. + * + * The reader hands out byte offsets only when a read finishes, so a chunk has + * no honest offset to record. This one is unusable on purpose: `indexedFile` + * reports no cursor for it, so an append is declined and the file is re-read + * whole. The rows are still a coherent prefix of that session and answer + * searches until the re-read replaces them. + */ +const PARTIAL_FILE_CURSOR = -1 + +type FileRow = { + dev: number | null + ino: number | null + byte_offset: number + mtime_ms: number + size_bytes: number | null + session_row_id: number | null +} + +type FileCursor = Pick + +export type SessionSearchFileWrite = { + /** + * Buffers one message, committing a chunk when the buffer reaches the ceiling + * — and only while this read can name the session it is writing. + * + * A chunk's rows answer searches the moment they land, so a read with no + * `identity` would publish them under a session with an empty id, an empty + * title and a null cwd, and an interrupted read would leave that prefix + * behind for good. The readers that supply no identity are the whole-file + * ones (Grok, Cursor, Gemini, OpenCode), whose formats are rewritten in place + * and have no resumable state to ask; they are also small — the largest on + * the author's machine is 5 MB — so buffering one to the end and committing + * it whole costs nothing. Chunking stays reserved for the readers that can + * say which session this is before the read ends. + */ + add(message: TranscriptMessage): void + /** + * Writes this file's rows, its session and its cursor in one transaction. + * False when the file's record changed under this read — it was removed, or + * another writer moved the cursor these rows continue from. A read that never + * calls this leaves the index exactly as it found it, unless it chunked. + */ + commit(outcome: TranscriptReadOutcome): boolean +} + +export class SessionSearchIndexWriter { + private readonly records: SessionSearchFileRecords + // Removals per path, so a write can prove its source was not dropped under it + // rather than infer it from the cursor. In memory is enough: one process owns + // the index, and a removal only has to fence writes this process opened. + private readonly removals = new Map() + + constructor( + private readonly db: SyncDatabase, + private readonly commitChars: number = SESSION_SEARCH_COMMIT_CHARS, + /** + * Called after a transaction that left a session's messages with no session + * row, so the owner can start the bounded drain that reclaims them. + * Synchronous work here would put the cost back where it was taken from. + */ + private readonly onOrphanedRows: () => void = () => undefined + ) { + this.records = new SessionSearchFileRecords(db) + } + + /** + * What the index holds for this file, or null when it holds nothing usable: + * an unknown path, or one whose recorded identity no longer matches. + * + * A file a chunked read left half written is reported, with a null cursor. + * Reporting nothing for it would read as "never indexed", so the caller would + * ask for whatever read the parse cache offers, the reader would pick append, + * and the decline would be the only thing that ever forced the whole read. + */ + indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null { + const row = this.db + .prepare( + 'SELECT dev, ino, byte_offset, mtime_ms, size_bytes, session_row_id FROM files WHERE path = ?' + ) + .get(path) as FileRow | undefined + if (!row) { + return null + } + // Older indexes can carry half-pairs; only a complete identity can prove replacement. + if (identity && row.dev !== null && row.ino !== null) { + if (row.dev !== identity.dev || row.ino !== identity.ino) { + return null + } + } + return { + byteOffset: row.byte_offset === PARTIAL_FILE_CURSOR ? null : row.byte_offset, + mtimeMs: row.mtime_ms, + sizeBytes: row.size_bytes + } + } + + /** + * Opens a buffered write for one read, or returns null when the read cannot + * extend what the index holds: an `append` whose predecessor byte offset is + * not this index's own cursor covers a span the index never saw. + */ + beginWrite( + candidate: SessionFileCandidate, + mode: 'replace' | 'append', + previousByteOffset: number, + identity?: () => TranscriptSessionIdentity | null + ): SessionSearchFileWrite | null { + const path = candidate.file.path + const cursor = this.cursor(path) + if (mode === 'append') { + // The partial sentinel is not a byte offset, so nothing continues it — + // including a caller that reads it back off the row and passes it in. + if (cursor === undefined || cursor.byte_offset === PARTIAL_FILE_CURSOR) { + return null + } + if (cursor.byte_offset !== previousByteOffset) { + return null + } + } + // A file the index read through and decoded no session from still has a + // cursor worth continuing: it has no session row to hang new rows off, so + // this read makes one. Declining instead would force a whole re-read of + // that file on every pass for as long as it grows. + return this.buffered(candidate, cursor, mode === 'append', identity) + } + + /** + * Drops a source: its session, its rows and its file record, in one + * transaction. Unbounded on purpose — the caller has proven this one file is + * gone and expects it out of results when the call returns, and a read of it + * that is still in flight is fenced by the cursor its commit re-reads. + */ + removeFile(path: string): void { + this.removals.set(path, (this.removals.get(path) ?? 0) + 1) + const cursor = this.cursor(path) + this.db.exec('BEGIN IMMEDIATE') + try { + this.dropSession(cursor?.session_row_id ?? null) + this.db.prepare('DELETE FROM files WHERE path = ?').run(path) + this.db.exec('COMMIT') + } catch (error) { + this.db.exec('ROLLBACK') + throw error + } + } + + private cursor(path: string): FileCursor | undefined { + return this.db + .prepare('SELECT session_row_id,byte_offset FROM files WHERE path = ?') + .get(path) as FileCursor | undefined + } + + private buffered( + candidate: SessionFileCandidate, + opened: FileCursor | undefined, + append: boolean, + identity?: () => TranscriptSessionIdentity | null + ): SessionSearchFileWrite { + const db = this.db + const path = candidate.file.path + const buffer: TranscriptMessage[] = [] + let bufferedChars = 0 + // What this write believes the file record holds. Re-read inside every + // transaction: a `removeFile` or another writer between two chunks means + // these rows no longer continue anything, and committing on top of that + // would resurrect a deleted source or duplicate a span. + let expected = opened + const removalsAtStart = this.removals.get(path) ?? 0 + // The session row is reused across re-reads of one file, so a `replace` + // swaps a session's rows rather than minting a second generation of it. + let session = opened?.session_row_id ?? null + let hash = append && session !== null ? this.records.contentHash(session) : EMPTY_CONTENT_HASH + // A replace owns the session's whole row set, so the old generation goes in + // the same transaction as the first of the new one. Chunk two onwards must + // not repeat it. + // + // It goes by being cut loose, not by being deleted. Deleting every old row + // inline sizes the transaction by the session being replaced rather than by + // the chunk being written: 1,286 ms against 720 ms fresh on the 100 MB + // corpus, and it grows with the history. Instead the first transaction + // mints a new session row, points `files` at it and deletes the one old + // `sessions` row. Every retrieval joins `sessions`, so the old generation + // stops answering the moment that commits, and its messages are reclaimed + // afterwards by the same bounded drain retention uses — which is where the + // old rows would have ended up had the process died here anyway. + // `sessions.id` is AUTOINCREMENT, so the freed id is never handed to + // another session while those rows still name it (round 8). + let replaced = append + // Set by the transaction that cut a generation loose; read once it commits. + let orphaned = false + // Set when the file record moved under this read. Nothing this write holds + // can land after that, so it stops buffering rather than reopening a + // transaction it already knows will roll back, once per remaining message. + let fenced = false + + // Why a counter and not the cursor alone: on a path this index never wrote, + // `expected` and the absent row are both undefined, so the cursor compare + // reads a removal as no change and the write recreates the source. + const current = (): boolean => { + if ((this.removals.get(path) ?? 0) !== removalsAtStart) { + return false + } + const row = this.cursor(path) + return ( + row?.session_row_id === expected?.session_row_id && + row?.byte_offset === expected?.byte_offset + ) + } + + /** + * `outcome` is null for a chunk of a read that has not reached the file's + * end, and `named` is what that chunk writes onto its session row. + */ + const write = ( + outcome: TranscriptReadOutcome | null, + named: TranscriptSessionIdentity | null + ): boolean => { + const decoded = outcome?.session ?? null + db.exec('BEGIN IMMEDIATE') + try { + if (!current()) { + db.exec('ROLLBACK') + return false + } + if (outcome && !decoded) { + // Read through, but nothing to search: the cursor advances so the file + // is not re-read whole on every pass, and whatever generation was here + // — including this read's own committed chunks — goes with it. + this.dropSession(session) + session = null + this.records.upsertFile(candidate, outcome.byteOffset, null) + } else { + if (replaced) { + session ??= this.records.createSessionRow(candidate) + } else { + const previous = session + session = this.records.createSessionRow(candidate) + if (previous !== null) { + db.prepare('DELETE FROM sessions WHERE id = ?').run(previous) + orphaned = true + } + replaced = true + } + for (const row of buffer) { + insertSearchMessage(db, session, row) + } + if (decoded) { + this.records.updateSession(decoded, session, hash) + } else if (named) { + // A chunk's rows answer searches as soon as they land, so the + // session they hang off is written with whatever the parser has + // decoded rather than left empty until a read that may never end. + // `add` refuses to chunk without this, so it is never absent here. + this.records.updateProvisionalSession(session, named) + } + this.records.upsertFile( + candidate, + outcome ? outcome.byteOffset : PARTIAL_FILE_CURSOR, + session + ) + } + db.exec('COMMIT') + } catch (error) { + db.exec('ROLLBACK') + throw error + } + // After the transaction that cut them loose is durable, never before: a + // rollback leaves the old session row standing and nothing to reclaim. + if (orphaned) { + orphaned = false + this.onOrphanedRows() + } + expected = { + session_row_id: session, + byte_offset: outcome ? outcome.byteOffset : PARTIAL_FILE_CURSOR + } + buffer.length = 0 + bufferedChars = 0 + return true + } + + return { + add: (message) => { + if (fenced) { + return + } + hash = foldContentHash(hash, [message]) + // The ceiling is checked per row, not per message: one message is a whole + // conversation turn and may be megabytes, so checking it after the whole + // message had been buffered let a single one carry a transaction as far + // past the ceiling as it was large. + for (const row of searchMessageRows([message])) { + buffer.push(row) + bufferedChars += row.text.length + if (bufferedChars < this.commitChars) { + continue + } + // Publishing a chunk under a session nothing can identify is worse + // than holding the buffer: the rows answer searches at once, and an + // interrupted read leaves that prefix for good. A read with nothing + // to name it keeps buffering and commits whole at `finish`. + const named = identity?.() ?? null + if (named && !write(null, named)) { + fenced = true + buffer.length = 0 + bufferedChars = 0 + return + } + } + }, + commit: (outcome) => !fenced && write(outcome, null) + } + } + + /** Caller's transaction: drops a session and every row that hangs off it. */ + private dropSession(sessionRowId: number | null): void { + if (sessionRowId === null) { + return + } + deleteSearchMessages(this.db, sessionRowId) + this.db.prepare('DELETE FROM sessions WHERE id = ?').run(sessionRowId) + } +} diff --git a/src/main/ai-vault-search/session-search-indexer-options.ts b/src/main/ai-vault-search/session-search-indexer-options.ts new file mode 100644 index 00000000000..08eb4aeb2af --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer-options.ts @@ -0,0 +1,49 @@ +import type { SessionSearchClock } from './session-search-clock' +import type { SessionSearchScanRoots } from './session-search-scan-roots' + +/** Default cycle. Long enough that a machine with thousands of transcripts is + * not re-statting continuously, short enough that a live conversation shows up + * while the user is still in it. */ +export const DEFAULT_SESSION_SEARCH_RECONCILE_INTERVAL_MS = 20_000 +/** Newest-N per agent root: the same recency rule the session sidebar applies. */ +export const DEFAULT_SESSION_SEARCH_RECENT_PER_AGENT = 12 +/** + * A quarter of the interval: the only bound on how long one pass reads for. + * + * The timer re-arms after a pass settles, so a pass that spends its whole + * deadline is followed by a full interval of quiet — five seconds of reading in + * every twenty-five, a fifth of the wall clock, and the stated ceiling is a + * quarter. Files the deadline cut off go back on the queue at full speed rather + * than being read slowly, which is what a load-average back-off did instead. + */ +export const DEFAULT_SESSION_SEARCH_PASS_DEADLINE_FRACTION = 4 +/** + * Cycles between whole-machine sweeps: five minutes at the default interval. + * + * A sweep is the only pass that sees a file nothing has told the indexer about + * — an old transcript deleted, a root that came back, a tree restored from a + * backup — so the cadence is what replaces every re-arm-on-recovery rule. A + * warm sweep is stats and readdirs, not reads, because the pass skips anything + * the index already covers at its current stat. + */ +export const DEFAULT_SESSION_SEARCH_FULL_SWEEP_EVERY_CYCLES = 15 + +/** + * Everything an indexer is. Immutable after construction: a settings change is + * `close()` and a new instance, which is also how the index is thrown away + * (`close()`, `removeSessionSearchDatabase(databasePath)`, construct again). + */ +export type SessionSearchIndexerOptions = { + databasePath: string + roots: SessionSearchScanRoots + /** null = all history; otherwise only transcripts modified within this many days. */ + historyDays: number | null + clock?: SessionSearchClock + reconcileIntervalMs?: number + recentPerAgent?: number + /** Wall time one pass may read for; the rest goes back on the queue. */ + passDeadlineMs?: number + /** Cycles between whole-machine sweeps. */ + fullSweepEveryCycles?: number + onError?: (error: unknown) => void +} diff --git a/src/main/ai-vault-search/session-search-indexer-test-fixture.ts b/src/main/ai-vault-search/session-search-indexer-test-fixture.ts new file mode 100644 index 00000000000..f8510510807 --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer-test-fixture.ts @@ -0,0 +1,168 @@ +import { mkdir, mkdtemp, rename, rm, stat, utimes, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import SyncDatabase from '../sqlite/sync-database' +import { isolatedScanRoots } from '../ai-vault/session-scanner-test-fixtures' +import type { SessionSearchClock, SessionSearchTimerHandle } from './session-search-clock' +import type { SessionSearchScanRoots } from './session-search-scan-roots' +import { assistantRecord, userRecord } from './session-search-transcript-fixtures' + +const CLOCK_EPOCH_MS = 1_740_000_000_000 + +/** Wall time the indexer's guarantee is stated in, under the test's control. */ +export class FakeSessionSearchClock implements SessionSearchClock { + private time = CLOCK_EPOCH_MS + private nextId = 1 + private nowCalls = 0 + private readonly timers = new Map void }>() + + /** + * What each `now()` reading costs. A pass reads the clock once per file it is + * about to read, so this is how a test spends a pass's deadline without + * waiting: it is the wall time the reads themselves take. + */ + costPerNowMs = 0 + + /** + * Runs on every `now()`, with the call number. The only synchronous seam into + * a running pass: the deadline check is what a pass consults between files. + */ + onNow: ((call: number) => void) | null = null + + now(): number { + const at = this.time + this.time += this.costPerNowMs + this.onNow?.(++this.nowCalls) + return at + } + + setTimeout(callback: () => void, ms: number): SessionSearchTimerHandle { + const id = this.nextId++ + this.timers.set(id, { at: this.time + ms, callback }) + return id + } + + clearTimeout(handle: SessionSearchTimerHandle): void { + this.timers.delete(handle as number) + } + + /** Moves time forward and fires every timer that came due, in order. */ + advance(ms: number): void { + this.time += ms + for (const [id, timer] of [...this.timers].sort((left, right) => left[1].at - right[1].at)) { + if (timer.at <= this.time) { + this.timers.delete(id) + timer.callback() + } + } + } + + get pendingTimers(): number { + return this.timers.size + } +} + +export type SessionSearchIndexerHarness = { + root: string + databasePath: string + roots: SessionSearchScanRoots + claudeProjectDir: string + /** A second connection: the store keeps its own private. */ + read: (query: (db: SyncDatabase) => T) => T + /** Plants what a killed writer would have left; nothing in the app writes here. */ + write: (query: (db: SyncDatabase) => T) => T + cleanup: () => Promise +} + +export async function openSessionSearchIndexerHarness( + name: string +): Promise { + const root = await mkdtemp(join(tmpdir(), `${name}-`)) + const roots = isolatedScanRoots(root) + const databasePath = join(root, 'index', 'index.sqlite') + return { + root, + databasePath, + roots, + claudeProjectDir: join(roots.claudeProjectsDir, 'project'), + read: (query) => withConnection(databasePath, true, query), + write: (query) => withConnection(databasePath, false, query), + cleanup: () => rm(root, { recursive: true, force: true }) + } +} + +function withConnection( + path: string, + readonlyConnection: boolean, + query: (db: SyncDatabase) => T +): T { + const db = new SyncDatabase(path, { readonly: readonlyConnection }) + try { + return query(db) + } finally { + db.close() + } +} + +/** A native-chat-shaped Claude transcript: the same records the app itself writes. */ +export async function writeClaudeTranscript( + path: string, + turns: readonly string[], + sessionId: string +): Promise { + await mkdir(dirname(path), { recursive: true }) + await writeFile(path, `${claudeLines(turns, sessionId, 0).join('\n')}\n`) +} + +export function claudeLines( + turns: readonly string[], + sessionId: string, + startIndex: number +): string[] { + return turns.flatMap((turn, offset) => [ + userRecord(startIndex + offset * 2, turn, sessionId), + assistantRecord(startIndex + offset * 2 + 1, `noted: ${turn}`, sessionId) + ]) +} + +/** + * Replaces a transcript the way an editor or a sync client does: a new inode + * renamed over the old name. Same byte length on purpose, so the only thing + * that can tell the two files apart is their filesystem identity. + */ +export async function renameReplaceTranscript( + path: string, + turns: readonly string[], + sessionId: string +): Promise { + const before = await stat(path) + const replacement = `${path}.replacement` + await writeClaudeTranscript(replacement, turns, sessionId) + await rename(replacement, path) + const later = new Date(before.mtimeMs + 5_000) + await utimes(path, later, later) +} + +/** + * A message-graph transcript, the shape OpenClaw, Pi, OMP and Prime Agent + * write. The session id comes from the file name, so callers name the file. + */ +export async function writeMessageGraphTranscript( + path: string, + turns: readonly string[] +): Promise { + await mkdir(dirname(path), { recursive: true }) + const lines = turns.flatMap((turn, index) => [ + JSON.stringify({ + type: 'message', + timestamp: new Date(CLOCK_EPOCH_MS + index * 120_000).toISOString(), + message: { role: 'user', content: turn } + }), + JSON.stringify({ + type: 'message', + timestamp: new Date(CLOCK_EPOCH_MS + index * 120_000 + 60_000).toISOString(), + message: { role: 'assistant', content: `noted: ${turn}` } + }) + ]) + await writeFile(path, `${lines.join('\n')}\n`) +} diff --git a/src/main/ai-vault-search/session-search-indexer.test.ts b/src/main/ai-vault-search/session-search-indexer.test.ts new file mode 100644 index 00000000000..5985628ba2d --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer.test.ts @@ -0,0 +1,1090 @@ +import { existsSync, mkdirSync, rmSync, utimesSync, writeFileSync } from 'node:fs' +import { appendFile, chmod, mkdir, rm, stat, utimes } from 'node:fs/promises' +import { dirname, join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { removeSessionSearchDatabase } from './session-search-schema' +import { parseTranscript } from './session-search-transcript-fixtures' +import { + claudeLines, + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + renameReplaceTranscript, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +const INTERVAL_MS = 20_000 +// chmod cannot deny root, and Windows ignores the mode bits entirely, so the +// two refusal tests would assert on an unreached branch there. +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const SESSION_ID = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +const OTHER_SESSION_ID = 'bbbbbbbb-cccc-4ddd-8eee-ffffffffffff' +const SETTLED_SESSION_ID = 'dddddddd-cccc-4ddd-8eee-ffffffffffff' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-indexer') + indexer = null +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function newIndexer( + overrides: Partial[0]> = {} +) { + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS, + onError: (error) => errors.push(error), + ...overrides + }) + return indexer +} + +/** Sessions a published-view read returns for one term, the only legal shape. */ +function sessionsMatching(term: string): string[] { + return harness.read((db: SyncDatabase) => + ( + db + .prepare( + `SELECT DISTINCT s.session_id AS id FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY s.session_id` + ) + .all(term) as { id: string }[] + ).map((row) => row.id) + ) +} + +function indexedSessionCount(): number { + return harness.read( + (db: SyncDatabase) => + (db.prepare('SELECT count(*) AS n FROM sessions').get() as { n: number }).n + ) +} + +/** The row the store holds for a path, which is the indexer's whole memory of it. */ +function rowFor(path: string) { + return harness.read((db: SyncDatabase) => + db.prepare('SELECT state, fail_count AS failCount FROM files WHERE path = ?').get(path) + ) as { state: string; failCount: number } | undefined +} + +function fileState(path: string): string | undefined { + return rowFor(path)?.state +} + +/** The byte offset the index recorded; PR 2 stores -1 for a half-written file. */ +function indexedByteOffset(path: string): number | undefined { + return harness.read( + (db: SyncDatabase) => + ( + db.prepare('SELECT byte_offset AS offset FROM files WHERE path = ?').get(path) as + | { offset: number } + | undefined + )?.offset + ) +} + +/** What a chunk of a read that never finished leaves on the file row. */ +function plantPartialCursor(path: string): void { + harness.write((db: SyncDatabase) => + db.prepare('UPDATE files SET byte_offset = -1 WHERE path = ?').run(path) + ) +} + +function indexedCursor(path: string): { mtime_ms: number; size_bytes: number } | undefined { + return harness.read( + (db: SyncDatabase) => + db.prepare('SELECT mtime_ms, size_bytes FROM files WHERE path = ?').get(path) as + | { mtime_ms: number; size_bytes: number } + | undefined + ) +} + +function transcriptPath(name = SESSION_ID): string { + return join(harness.claudeProjectDir, `${name}.jsonl`) +} + +/** + * Starts the indexer over a root that already holds one indexed transcript, so + * the opening sweep is behind us and `reconcile()` runs a cycle. It is dated + * ahead of everything the caller writes afterwards, so it stays inside any + * recency window and is skipped rather than read. + */ +async function startAfterASweep( + overrides: Partial[0]> = {} +): Promise { + const settled = transcriptPath(SETTLED_SESSION_ID) + await writeClaudeTranscript(settled, ['a conversation from before'], SETTLED_SESSION_ID) + // Wall time, not the fake clock: recency is decided by real file mtimes. + const ahead = new Date(Date.now() + 3_600_000) + await utimes(settled, ahead, ahead) + await newIndexer(overrides).start() +} + +/** + * Makes every pass stop after `files` reads: the pass consults the clock once + * per file it is about to read, and each reading costs a quarter of the + * deadline it is measured against. + */ +function readsPerPass(files: number): { passDeadlineMs: number } { + clock.costPerNowMs = 1_000 + return { passDeadlineMs: files * 1_000 } +} + +/** Advances one reconcile interval and waits for the cycle it fires. */ +async function nextCycle(): Promise { + clock.advance(INTERVAL_MS) + await indexer?.settled() +} + +it('reflects a grown transcript within one reconcile interval', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['find the flaky terminal reattach'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('reattach')).toEqual([SESSION_ID]) + expect(sessionsMatching('quarantine')).toEqual([]) + + await appendFile( + path, + `${claudeLines(['quarantine the leaking pty'], SESSION_ID, 10).join('\n')}\n` + ) + await nextCycle() + + expect(sessionsMatching('quarantine')).toEqual([SESSION_ID]) + expect(errors).toEqual([]) +}) + +it('reflects a rename-replaced transcript within one reconcile interval', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['original content aaaa'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('original')).toEqual([SESSION_ID]) + const original = await stat(path) + + await renameReplaceTranscript(path, ['swapped content bbbbb'], SESSION_ID) + // Same length, different inode: only the identity check can tell them apart. + expect((await stat(path)).size).toBe(original.size) + await nextCycle() + + expect(sessionsMatching('swapped')).toEqual([SESSION_ID]) + expect(sessionsMatching('original')).toEqual([]) + expect(errors).toEqual([]) +}) + +it('retires a deleted transcript within one reconcile interval', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['a session about to be deleted'], SESSION_ID) + await writeClaudeTranscript( + transcriptPath(OTHER_SESSION_ID), + ['a surviving session'], + OTHER_SESSION_ID + ) + await newIndexer().start() + await nextCycle() + expect(sessionsMatching('deleted')).toEqual([SESSION_ID]) + + await rm(path) + await nextCycle() + + expect(sessionsMatching('deleted')).toEqual([]) + expect(sessionsMatching('surviving')).toEqual([OTHER_SESSION_ID]) +}) + +it.skipIf(!CAN_DENY_READ)( + 'keeps rows for a source it cannot stat, because loss of contact is not deletion', + async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['an unverifiable session'], SESSION_ID) + await newIndexer().start() + await nextCycle() + + // The tree is gone from discovery's point of view, but the transcript itself + // was never proven absent: an unreadable parent is not a deleted file. + await chmod(harness.claudeProjectDir, 0o000) + try { + await nextCycle() + expect(sessionsMatching('unverifiable')).toEqual([SESSION_ID]) + } finally { + await chmod(harness.claudeProjectDir, 0o755) + } + } +) + +it('resumes after close and reopen without re-reading what it already indexed', async () => { + await writeClaudeTranscript(transcriptPath(), ['first indexed session'], SESSION_ID) + await writeClaudeTranscript( + transcriptPath(OTHER_SESSION_ID), + ['second indexed session'], + OTHER_SESSION_ID + ) + await newIndexer().start() + const indexedRows = harness.read((db: SyncDatabase) => + db.prepare('SELECT count(*) AS n FROM messages').get() + ) + indexer?.close() + + // A restart is a cold parse cache over a warm index; only the `files` table + // can say what has already been read. + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + const reopened = newIndexer() + await reopened.start() + + // `filesIndexed` is the count of rows the index holds at their current stat, + // so it stays 2. That nothing was opened again is the read loop's own test. + expect(reopened.status()).toMatchObject({ filesIndexed: 2, filesDue: 0 }) + expect( + harness.read((db: SyncDatabase) => db.prepare('SELECT count(*) AS n FROM messages').get()) + ).toEqual(indexedRows) + expect(sessionsMatching('indexed')).toEqual([SESSION_ID, OTHER_SESSION_ID].sort()) +}) + +// F12, as the immutable design states it: the history window is a construction +// argument, so widening it is a new instance whose opening sweep admits the +// older files, and narrowing it is the purge that opens every full sweep. +it('widens history by constructing a new instance and narrows by purging on its first sweep', async () => { + const fresh = transcriptPath() + const old = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(fresh, ['a recent conversation'], SESSION_ID) + await writeClaudeTranscript(old, ['an ancient conversation'], OTHER_SESSION_ID) + const longAgo = new Date(clock.now() - 120 * 86_400_000) + await utimes(old, longAgo, longAgo) + + // Newest-one per root, so the widened-in transcript is outside the recency + // window a cycle re-stats: only a full sweep can reach it. + await newIndexer({ historyDays: 30, recentPerAgent: 1 }).start() + expect(sessionsMatching('recent')).toEqual([SESSION_ID]) + expect(sessionsMatching('ancient')).toEqual([]) + + // Widening cannot be served from the index: those files were never read. + indexer?.close() + await newIndexer({ historyDays: null, recentPerAgent: 1 }).start() + expect(sessionsMatching('ancient')).toEqual([OTHER_SESSION_ID]) + + indexer?.close() + await newIndexer({ historyDays: 30, recentPerAgent: 1 }).start() + expect(sessionsMatching('ancient')).toEqual([]) + expect(sessionsMatching('recent')).toEqual([SESSION_ID]) +}) + +it.skipIf(!CAN_DENY_READ)( + 'names an unreadable root as degraded and keeps indexing the others', + async () => { + const blocked = join(harness.roots.codexSessionsDir ?? '', 'blocked') + await mkdir(blocked, { recursive: true }) + await writeClaudeTranscript(transcriptPath(), ['a readable claude session'], SESSION_ID) + await chmod(harness.roots.codexSessionsDir ?? '', 0o000) + try { + await newIndexer().start() + const status = indexer?.status() + expect(status?.phase).toBe('degraded') + expect(status?.degradedRoots.map((root) => root.root)).toContain( + harness.roots.codexSessionsDir + ) + expect(status?.degradedRoots[0]?.reason).toBeTruthy() + // A degraded root is not a degraded index: everything else still lands. + expect(sessionsMatching('readable')).toEqual([SESSION_ID]) + } finally { + await chmod(harness.roots.codexSessionsDir ?? '', 0o755) + } + } +) + +// The one bound on a pass. What it does not reach is owed on the next pass for +// the same reason it was owed on this one -- its row says so, or it has no row +// -- so nothing is written down and nothing can be lost. +it('reads what one pass has time for and finishes the rest on the next', async () => { + // The sweep is behind us, so this is the reconciler fitting four new files + // into a deadline that stops it after two. + await startAfterASweep(readsPerPass(2)) + for (let index = 0; index < 4; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript( + transcriptPath(session), + [`deadlined session number ${index}`], + session + ) + } + await indexer?.reconcile() + expect(indexer?.status()).toMatchObject({ filesIndexed: 3, filesDue: 0 }) + + await indexer?.reconcile() + expect(sessionsMatching('deadlined')).toHaveLength(4) + expect(indexer?.status().filesIndexed).toBe(5) + + // Settled, and it stays settled: nothing changed, so the cycle after this + // one opens none of them. + clock.costPerNowMs = 0 + await nextCycle() + expect(indexer?.status()).toMatchObject({ filesIndexed: 5, phase: 'current' }) +}) + +// First enablement inside a running app is the normal case, not an edge: the +// session list has been scanning since launch, so every transcript already has +// a cursor sitting at its current stat and the index has nothing at all. +it('fills an empty index over a warm session-list cache on the first reconcile', async () => { + await startAfterASweep() + const path = transcriptPath() + await writeClaudeTranscript(path, ['scanned before the index existed'], SESSION_ID) + // An ordinary parse now reuses its cached fold and opens no file, so no + // consumer is asked and there is nothing for a decline to record. + await parseTranscript(path) + + await indexer?.reconcile() + + expect(sessionsMatching('scanned')).toEqual([SESSION_ID]) +}) + +it('fills an empty index over a warm session-list cache on the first sweep', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['scanned before the index existed'], SESSION_ID) + await parseTranscript(path) + + await newIndexer().start() + + expect(sessionsMatching('scanned')).toEqual([SESSION_ID]) +}) + +// Finding 1: a sweep cut short used to be abandoned part way through. A pass +// that hands reads back is not an unfinished sweep -- its discovery and its +// retirement both completed -- so it must not re-arm one, and the queue is what +// carries the reads it did not reach until the whole machine is covered. +it('covers the whole machine over the passes that follow a truncated sweep', async () => { + const sessions = Array.from( + { length: 20 }, + (_unused, index) => `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + ) + for (const session of sessions) { + await writeClaudeTranscript(transcriptPath(session), [`sweepwide session ${session}`], session) + } + + // One transcript a pass, so the opening sweep reaches a twentieth of them. + await newIndexer(readsPerPass(1)).start() + expect(indexedSessionCount()).toBeGreaterThan(0) + expect(indexedSessionCount()).toBeLessThan(sessions.length) + + for (let cycle = 0; cycle < sessions.length; cycle++) { + await nextCycle() + } + + expect(indexedSessionCount()).toBe(sessions.length) + expect(indexer?.status().phase).toBe('current') +}) + +// Finding 2: the store's cutoff was set once at construction while purges used +// a fresh one, so a sweep deleted the row and the accept check re-indexed it. +it('moves the retention window with the clock instead of freezing it at construction', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['an entry that ages out'], SESSION_ID) + // Dated on the same clock the retention window is measured against. + const now = new Date(clock.now()) + await utimes(path, now, now) + await newIndexer({ historyDays: 1 }).start() + expect(sessionsMatching('ages')).toEqual([SESSION_ID]) + + clock.advance(3 * 86_400_000) + await indexer?.reconcile({ full: true }) + + expect(sessionsMatching('ages')).toEqual([]) + await nextCycle() + expect(sessionsMatching('ages')).toEqual([]) +}) + +// Round 10, H1. A cycle proves a deletion by comparing what the previous pass +// watched against what it discovers. A sweep used to watch only what it could +// not settle, which is nothing on a healthy machine, so the cycle after a sweep +// had no candidates at all and the cycle after that no longer remembered the +// file: a transcript deleted in that interval survived until the next sweep, +// up to `fullSweepEveryCycles` later. +it('retires a transcript deleted between a sweep and the cycle after it', async () => { + const going = transcriptPath() + const staying = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(going, ['a session deleted right after the sweep'], SESSION_ID) + await writeClaudeTranscript(staying, ['a surviving session'], OTHER_SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('deleted')).toEqual([SESSION_ID]) + + // No cycle in between: the sweep is the only pass that has seen this file. + await rm(going) + await nextCycle() + + expect(sessionsMatching('deleted')).toEqual([]) + expect(sessionsMatching('surviving')).toEqual([OTHER_SESSION_ID]) +}) + +// Round 10, M2. A sweep that throws part way learned nothing, and the flag that +// says one is owed was taken on entry. Losing it there leaves nothing armed to +// try again, so the machine outside the recency window goes unread until +// something else happens to ask for a sweep. +it('keeps a sweep due when the one that was running threw', async () => { + const older = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(older, ['an older conversation'], OTHER_SESSION_ID) + const yesterday = new Date(Date.now() - 86_400_000) + await utimes(older, yesterday, yesterday) + await writeClaudeTranscript(transcriptPath(), ['the newest conversation'], SESSION_ID) + + // Newest-one per root, so only a sweep can reach the older file. The clock is + // read inside the pass, which is where a failure part way through lands. + newIndexer({ recentPerAgent: 1 }) + let thrown = false + clock.onNow = () => { + if (thrown || indexedSessionCount() === 0) { + return + } + thrown = true + throw new Error('the sweep fell over') + } + await indexer?.start() + await indexer?.settled() + clock.onNow = null + + expect(errors.map((error) => (error as Error).message)).toEqual(['the sweep fell over']) + expect(sessionsMatching('older')).toEqual([]) + + // The pass after it is a sweep, not a cycle: a cycle reads one file per root. + await nextCycle() + expect(sessionsMatching('older')).toEqual([OTHER_SESSION_ID]) +}) + +// Round 10, M1. A transcript the reader cannot open is recorded stale by the +// consumer on every attempt, so it was re-read every cycle for ever: pending +// stuck at one, a failure count climbing without bound, and a phase that never +// left `indexing`. One file with the wrong mode bits read as a real backlog. +it.skipIf(!CAN_DENY_READ)('stops re-reading a transcript it cannot read', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['a session behind the wrong mode bits'], SESSION_ID) + await chmod(path, 0o000) + try { + await newIndexer().start() + for (let cycle = 0; cycle < 4; cycle++) { + await nextCycle() + } + + // Held out by its own row: three failures at one unchanged stat, counted on + // the row itself, and a phase that says the index knows it is not covering + // something rather than one that describes work it will never do. + expect(indexer?.status()).toMatchObject({ filesDue: 0, filesFailed: 1, phase: 'degraded' }) + expect(rowFor(path)?.failCount).toBeGreaterThanOrEqual(3) + + // And the hold is released by the only thing that can mean the file + // changed: its stat. + await chmod(path, 0o644) + const later = new Date(Date.now() + 60_000) + await utimes(path, later, later) + await nextCycle() + + expect(sessionsMatching('mode')).toEqual([SESSION_ID]) + expect(indexer?.status()).toMatchObject({ filesFailed: 0, phase: 'current' }) + } finally { + await chmod(path, 0o644) + } +}) + +// Round 10, M2. `close()` mid-pass left the pass reading a shut handle: three +// `database is not open` errors reached the owner, for a close they asked for. +it('reports nothing to its owner when it is closed part way through a pass', async () => { + await writeClaudeTranscript(transcriptPath(), ['one'], SESSION_ID) + const other = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(other, ['two'], OTHER_SESSION_ID) + const later = new Date(Date.now() + 60_000) + await utimes(other, later, later) + newIndexer() + + // Between two files: the pass reads the clock once per file it is about to + // read, and closing there is what a quit during a sweep looks like. + let closed = false + clock.onNow = () => { + if (closed || indexedSessionCount() === 0) { + return + } + closed = true + indexer?.close() + } + await indexer?.start() + await indexer?.settled() + clock.onNow = null + + expect(errors).toEqual([]) +}) + +// Round 10, M2, the other half: `status()` on a closed indexer opened a shut +// database, reported the failure, and answered zero files. +it('reports what it last knew after it is closed, without reading the database', async () => { + await writeClaudeTranscript(transcriptPath(), ['indexed before the close'], SESSION_ID) + await newIndexer().start() + expect(indexer?.status().filesIndexed).toBe(1) + + indexer?.close() + + expect(indexer?.status()).toMatchObject({ phase: 'closed', filesIndexed: 1 }) + expect(errors).toEqual([]) +}) + +// Round 10, L1. Two indexers on one database both register with the reader, so +// every transcript is read and written twice and the second write is fenced by +// the first at random. The recipe for every configuration change is +// close-then-construct, so the ordering that causes this is the one the recipe +// rules out; this is what says so rather than letting it corrupt quietly. +// Round 12, F2. The claim was staked before the store opened, so an open that +// threw left the path owned by an object that does not exist and every later +// construction was refused -- including the one that fixes whatever broke it. +it('releases the database path when the open itself throws', () => { + // A directory where the database file goes: the open fails, nothing is owned. + mkdirSync(harness.databasePath, { recursive: true }) + expect(() => newIndexer()).toThrow() + + rmSync(harness.databasePath, { recursive: true, force: true }) + expect(() => newIndexer()).not.toThrow() +}) + +it('refuses a second indexer on a database one already owns', () => { + newIndexer() + expect(() => newIndexer()).toThrow(/already has a live indexer/) +}) + +// PR 2 records a cursor no append continues for a file a chunked read left half +// written, and reports it as a null offset. The mtime and size on that row are +// the whole file's, so a freshness check comparing only those calls a prefix +// current and leaves it in the index for good. +it('re-reads a file a chunked read left half written, and settles it in one pass', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['the committed half'], SESSION_ID) + await newIndexer().start() + const whole = (await stat(path)).size + expect(indexedByteOffset(path)).toBe(whole) + indexer?.close() + + plantPartialCursor(path) + await newIndexer().start() + + // Nothing about the file changed, and it was read anyway: the whole of it, + // because there is no cursor to continue from. + expect(indexedByteOffset(path)).toBe(whole) + expect(fileState(path)).toBe('current') + expect(indexer?.status().phase).toBe('current') + indexer?.close() + + // A half-written file that also grew is repaired by one pass rather than two. + // The session list's resume point would have the reader offer an append here, + // and an append onto a partial cursor is a read the consumer declines. + plantPartialCursor(path) + await appendFile(path, `${claudeLines(['the lost half'], SESSION_ID, 10).join('\n')}\n`) + await newIndexer().start() + + expect(sessionsMatching('lost')).toEqual([SESSION_ID]) + expect(indexer?.status()).toMatchObject({ filesDue: 0, phase: 'current' }) +}) + +it('reports closed once it is closed, whatever it was doing before', async () => { + await writeClaudeTranscript(transcriptPath(), ['before the close'], SESSION_ID) + await newIndexer().start() + expect(indexer?.status().phase).toBe('current') + indexer?.close() + expect(indexer?.status().phase).toBe('closed') +}) + +// Finding 6: a queued entry carries the stat it was recorded with. Reading at +// that stat writes a cursor describing a file that no longer looks like this, +// so the next cycle distrusts it and re-reads it, forever. +it('reads a deferred file at its current stat, not the one the pass first saw', async () => { + // One file a pass, so the older one is left for the pass after this. + await startAfterASweep(readsPerPass(1)) + const older = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(older, ['the deferred conversation'], OTHER_SESSION_ID) + await writeClaudeTranscript(transcriptPath(), ['the newer conversation'], SESSION_ID) + const ahead = new Date((await stat(transcriptPath())).mtimeMs + 60_000) + await utimes(transcriptPath(), ahead, ahead) + + await indexer?.reconcile() + // No row for it at all, which is exactly why the next pass reads it. + expect(rowFor(older)).toBeUndefined() + + await appendFile( + older, + `${claudeLines(['appended while deferred'], OTHER_SESSION_ID, 10).join('\n')}\n` + ) + await indexer?.reconcile() + + expect(sessionsMatching('appended')).toEqual([OTHER_SESSION_ID]) + // The cursor has to describe the file as it is now; recorded against the + // stat the earlier pass saw it would be re-read on every cycle from here on. + const cursor = indexedCursor(older) + const current = await stat(older) + expect(cursor).toEqual({ mtime_ms: current.mtimeMs, size_bytes: current.size }) +}) + +// A declined read records the stat it was declined at. By the time the store +// hands it back the file has usually moved on again, and reading at the +// recorded stat writes a cursor the next cycle immediately distrusts. +it('reads a declined file at its current stat, not the one it was recorded with', async () => { + const path = transcriptPath() + await writeClaudeTranscript(path, ['the recorded conversation'], SESSION_ID) + await newIndexer().start() + + // A warm session-list cache over an empty index: the reader offers an append + // continuing an offset this index has never seen, so the consumer declines it + // and records the stat it declined at. + indexer?.close() + removeSessionSearchDatabase(harness.databasePath) + newIndexer() + await appendFile(path, `${claudeLines(['declined turn'], SESSION_ID, 10).join('\n')}\n`) + await parseTranscript(path) + // The index holds nothing for it, which is the record: a path the file table + // does not name is read from the start by the next pass. + expect(indexer?.status().filesIndexed).toBe(0) + + await appendFile(path, `${claudeLines(['later turn'], SESSION_ID, 20).join('\n')}\n`) + // The sweep is declined too -- the list's cursor is still ahead of the index + // -- so it is the pass after it that reads the file whole. + await indexer?.start() + await nextCycle() + + expect(sessionsMatching('later')).toEqual([SESSION_ID]) + const current = await stat(path) + expect(indexedCursor(path)).toEqual({ mtime_ms: current.mtimeMs, size_bytes: current.size }) +}) + +// Round 2, item 1: the sweep kept the rows and a cycle twenty seconds later +// deleted them, because the degraded-root fence was on the sweep path only. +it.skipIf(!CAN_DENY_READ)( + 'keeps an unlistable root through the cycles that follow the sweep', + async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + + await chmod(harness.roots.claudeProjectsDir ?? '', 0o000) + try { + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + + await nextCycle() + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + expect(indexer?.status().phase).toBe('degraded') + } finally { + await chmod(harness.roots.claudeProjectsDir ?? '', 0o755) + } + } +) + +// A root that cannot be listed is never believed to be empty, however many +// times it is asked: an error is not a listing, and only a listing is proof. +it.skipIf(!CAN_DENY_READ)('keeps an unlistable root degraded across repeated sweeps', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await newIndexer().start() + + await chmod(harness.roots.claudeProjectsDir ?? '', 0o000) + try { + for (let sweep = 0; sweep < 5; sweep++) { + await indexer?.reconcile({ full: true }) + } + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + expect(indexer?.status().phase).toBe('degraded') + } finally { + await chmod(harness.roots.claudeProjectsDir ?? '', 0o755) + } +}) + +// The first sweep of every process is exactly when a volume is most likely to +// be detached, and it is the pass with nothing behind it to compare against. +it('keeps a root that is gone at the first sweep after a restart', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await newIndexer().start() + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + // The volume is not there when the process comes back. + await rm(harness.roots.claudeProjectsDir ?? '', { recursive: true, force: true }) + await newIndexer().start() + + const status = indexer?.status() + expect(status?.phase).toBe('degraded') + expect(status?.degradedRoots.map((root) => root.root)).toContain(harness.roots.claudeProjectsDir) + expect(sessionsMatching('removable')).toEqual([SESSION_ID]) + + // And it clears once the volume is back. + await writeClaudeTranscript(transcriptPath(), ['a session on a removable volume'], SESSION_ID) + await indexer?.reconcile({ full: true }) + expect(indexer?.status()).toMatchObject({ phase: 'current', degradedRoots: [] }) +}) + +// Round 7: what the stateless walk costs, stated rather than hidden. A volume +// mounted at EXACTLY a configured root, unmounted so the mountpoint stays +// present and lists empty, is indistinguishable from a root the user emptied: +// there is no directory left whose absence could stop the walk. Inside one +// process the transition buys a pass of grace; across a restart there is no +// transition to see and the rows retire. The unmounts that actually happen are +// above the root, and the next test is the one that covers them. +it('retires an emptied configured root, one pass after it emptied', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on the mounted volume'], SESSION_ID) + await newIndexer().start() + + // The transcripts go; the root itself stays there and stays readable. + await rm(harness.claudeProjectDir, { recursive: true, force: true }) + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('mounted')).toEqual([SESSION_ID]) + expect(indexer?.status().phase).toBe('degraded') + + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('mounted')).toEqual([]) + expect(indexer?.status()).toMatchObject({ phase: 'current', degradedRoots: [] }) +}) + +// The same root, with no previous pass to compare against: nothing carries the +// transition across a restart, and the empty listing is proof on its own. +it('retires an emptied configured root at once on the first pass of a process', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session on the mounted volume'], SESSION_ID) + await newIndexer().start() + expect(sessionsMatching('mounted')).toEqual([SESSION_ID]) + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + await rm(harness.claudeProjectDir, { recursive: true, force: true }) + await newIndexer().start() + expect(sessionsMatching('mounted')).toEqual([]) +}) + +// The shape a real unmount takes: on Linux, WSL and sshfs the mountpoint is +// above the agent's root, so the root itself is missing. The walk stops at the +// root boundary and never asks the empty parent anything, which is what makes +// this hold with no memory on the first pass of a process. +it('proves nothing from an empty directory above the configured root', async () => { + for (let index = 0; index < 3; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`mounted session ${index}`], session) + } + await newIndexer().start() + expect(sessionsMatching('mounted')).toHaveLength(3) + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + // The volume that carried the agent's root is gone; what it was mounted + // under is still there, still listable, and empty of it. + await rm(harness.roots.claudeProjectsDir ?? '', { recursive: true, force: true }) + await newIndexer().start() + + const status = indexer?.status() + expect(status?.phase).toBe('degraded') + expect(status?.degradedRoots.map((root) => root.root)).toContain(harness.roots.claudeProjectsDir) + expect(sessionsMatching('mounted')).toHaveLength(3) +}) + +// A cycle only reads the newest N per agent, so a remounted volume would give +// up its newest transcript and keep the rest unreachable. Nothing watches for a +// recovery any more: the sweep cadence is what reaches it. +it('reads a root that came back on the next periodic sweep', async () => { + // Detached before anything was ever indexed, so the sweep correctly finds + // nothing and reports no alarm. + await newIndexer({ recentPerAgent: 1, fullSweepEveryCycles: 2 }).start() + expect(indexer?.status()).toMatchObject({ degradedRoots: [], filesIndexed: 0 }) + + for (let index = 0; index < 3; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`remounted session ${index}`], session) + } + + // Two cycles reach the newest one each; the sweep they are counting down to + // reads the rest. + await nextCycle() + await nextCycle() + expect(sessionsMatching('remounted')).toHaveLength(1) + + await nextCycle() + expect(sessionsMatching('remounted')).toHaveLength(3) +}) + +// Round 4, item 3: rows under no configured root. The walk judges each row on +// its own directory and proves nothing about one it cannot reach, so a profile +// that moved keeps its history rather than losing it. +it('keeps rows under no configured root, and retires them only when gone', async () => { + const moved = transcriptPath() + await writeClaudeTranscript(moved, ['a session in the old profile'], SESSION_ID) + await newIndexer().start() + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + + // The profile moves: same index, a root that no longer covers those rows. + const elsewhere = join(harness.root, 'moved-profile') + newIndexer({ roots: { ...harness.roots, claudeProjectsDir: elsewhere } }) + await indexer?.start() + // Still on disk, so the rows stay: this is a configuration problem, not a + // licence to delete a user's history. + expect(sessionsMatching('profile')).toEqual([SESSION_ID]) + + await rm(moved) + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('profile')).toEqual([]) +}) + +// Round 7 replaced "only a census may conclude" with "whoever can prove it". +// A cycle walks the same directories and reaches the same verdict, so a project +// directory the user deleted does not wait for the next sweep. +it('lets a cycle retire a project directory the user deleted', async () => { + await writeClaudeTranscript(transcriptPath(), ['a session about to vanish'], SESSION_ID) + await newIndexer().start() + + await rm(harness.claudeProjectDir, { recursive: true, force: true }) + // The pass that sees the root go from holding transcripts to holding none + // gives it one pass of grace, whether it is a sweep or a cycle. + await indexer?.reconcile({ full: true }) + expect(sessionsMatching('vanish')).toEqual([SESSION_ID]) + + await nextCycle() + expect(sessionsMatching('vanish')).toEqual([]) + expect(indexer?.status()).toMatchObject({ phase: 'current', degradedRoots: [] }) +}) + +// C1: `close()` disarmed the timer and aborted the task in flight, but left the +// queue running, so a task queued a moment earlier still reopened a store and +// registered a consumer behind an indexer whose caller had finished with it. +it('stops everything on close, including work already queued', async () => { + await writeClaudeTranscript(transcriptPath(), ['indexed before the close'], SESSION_ID) + await newIndexer().start() + + const queued = indexer?.reconcile({ full: true }) + indexer?.close() + await queued + + // The queued pass never ran: had it run, it would have reached for a store + // this close had already shut, and reported the failure. + expect(errors).toEqual([]) + // And the timer is gone with it, so no later tick can queue another. + expect(clock.pendingTimers).toBe(0) + clock.advance(5 * INTERVAL_MS) + await indexer?.settled() + expect(errors).toEqual([]) + + // No store and no consumer: a scan after the close writes nothing. + const after = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(after, ['written after the close'], OTHER_SESSION_ID) + await parseTranscript(after) + expect(sessionsMatching('written')).toEqual([]) + expect(sessionsMatching('indexed')).toEqual([SESSION_ID]) +}) + +// What replaced `clear()`, exactly as the PR body documents it. The recipe is +// three statements because the indexer owns one store for one lifetime; the +// method it replaces owned a second one and had to keep the two in step. +it('throws the index away and rebuilds it by constructing a new instance', async () => { + await writeClaudeTranscript(transcriptPath(), ['indexed before the clear'], SESSION_ID) + await newIndexer().start() + expect(existsSync(harness.databasePath)).toBe(true) + + indexer?.close() + removeSessionSearchDatabase(harness.databasePath) + expect(existsSync(harness.databasePath)).toBe(false) + + // The session list's cache is warm, which is what a clear inside a running + // app leaves behind; the sweep reads whole rather than trusting it. + await newIndexer().start() + expect(sessionsMatching('indexed')).toEqual([SESSION_ID]) +}) + +it('refuses a reconcile before it is started and after it is closed', async () => { + newIndexer() + expect(() => indexer?.reconcile()).toThrow(/start\(\) first/) + + await indexer?.start() + await indexer?.reconcile() + indexer?.close() + expect(() => indexer?.reconcile()).toThrow(/closed/) +}) + +// I7: the sweep reads transcript bytes, so it stops at the same deadline every +// other pass does. It plans the whole machine and hands back what it had no +// time for; the passes that follow drain the plan without re-discovering. +it('stops the opening sweep at its deadline and drains the rest over the passes that follow', async () => { + for (let index = 0; index < 5; index++) { + const session = `0000000${index}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`backlogged session ${index}`], session) + } + await newIndexer(readsPerPass(2)).start() + expect(indexer?.status().filesIndexed).toBe(2) + + await nextCycle() + expect(indexer?.status().filesIndexed).toBe(4) + + await nextCycle() + expect(sessionsMatching('backlogged')).toHaveLength(5) + expect(indexer?.status()).toMatchObject({ filesIndexed: 5, filesDue: 0 }) +}) + +// The sweep cadence, with nobody asking for it: a file outside the recency +// window that appears after the opening sweep is unreachable until the next +// periodic one, and the count of cycles is the whole rule. +it('sweeps on its cadence without anyone asking', async () => { + await writeClaudeTranscript(transcriptPath(), ['the newest conversation'], SESSION_ID) + await newIndexer({ recentPerAgent: 1, fullSweepEveryCycles: 2 }).start() + + const older = transcriptPath(OTHER_SESSION_ID) + await writeClaudeTranscript(older, ['an older conversation'], OTHER_SESSION_ID) + const yesterday = new Date(Date.now() - 86_400_000) + await utimes(older, yesterday, yesterday) + + await nextCycle() + await nextCycle() + expect(sessionsMatching('older')).toEqual([]) + + await nextCycle() + expect(sessionsMatching('older')).toEqual([OTHER_SESSION_ID]) +}) + +// A cycle lists the newest N per agent, so every older row it holds is +// undiscovered and would be walked every twenty seconds. It proves the newest +// slice of them instead, capped: a transcript recent enough for the window is +// recent enough to be in the slice, and the rest are the next sweep's to reach. +// Round 12, F1. A directory that cannot be listed answers `unverifiable` for +// every row under it, on every pass, for as long as the permission stays wrong. +// With the walk capped at rows rather than at directories, five hundred such +// rows spent the whole budget on one readdir's worth of verdicts and a row for +// a file the user really deleted, sorted behind them, was never reached: six +// full sweeps and it was still held. +it.skipIf(!CAN_DENY_READ)('retires a deleted file behind a block of unreadable rows', async () => { + // A healthy project directory, so the root never looks emptied. + await writeClaudeTranscript(transcriptPath(), ['a live conversation'], SESSION_ID) + const locked = join(harness.roots.claudeProjectsDir ?? '', 'locked') + await mkdir(locked, { recursive: true }) + newIndexer() + + // What an unreadable tree leaves behind: rows the walk can never settle, + // planted ahead of the deleted one in the order the table returns them. + harness.write((db: SyncDatabase) => { + const insert = db.prepare( + `INSERT INTO files(path, byte_offset, mtime_ms, size_bytes, state) + VALUES (?, 0, ?, 10, 'current')` + ) + for (let index = 0; index < 520; index++) { + insert.run(join(locked, `locked-${index}.jsonl`), 1_700_000_000_000 + index) + } + return insert.run(join(harness.claudeProjectDir, 'deleted.jsonl'), 1_700_000_999_000) + }) + const deleted = join(harness.claudeProjectDir, 'deleted.jsonl') + const holdsDeleted = (): boolean => rowFor(deleted) !== undefined + + await chmod(locked, 0o000) + try { + await indexer?.start() + + expect(holdsDeleted()).toBe(false) + // And the block itself is neither retired nor forgotten: unreadable is not + // deleted, and the root is named as degraded rather than emptied. + expect(indexer?.status().filesIndexed).toBe(521) + expect(indexer?.status().phase).toBe('degraded') + } finally { + await chmod(locked, 0o700) + } +}) + +it('proves deletions for the newest rows it holds, and leaves the tail to a sweep', async () => { + const total = 530 + const oldest = transcriptPath('00000000-bbbb-4ccc-8ddd-eeeeeeeeeeee') + await writeClaudeTranscript( + oldest, + ['the oldest session'], + '00000000-bbbb-4ccc-8ddd-eeeeeeeeeeee' + ) + const longAgo = new Date(Date.now() - total * 60_000) + await utimes(oldest, longAgo, longAgo) + // Indexed on its own first, so it is the earliest row in the table as well as + // the oldest file. A slice that trusted the table's own order rather than the + // mtime would take it, and take it first. + await newIndexer().start() + + for (let index = 1; index < total; index++) { + const session = `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + const path = transcriptPath(session) + await writeClaudeTranscript(path, [`capped session ${index}`], session) + const at = new Date(Date.now() - (total - index) * 60_000) + await utimes(path, at, at) + } + await indexer?.reconcile({ full: true }) + expect(indexedSessionCount()).toBe(total) + + // Older than the cap reaches: 530 rows, twelve of them rediscovered by the + // cycle, leaves 518 undiscovered against a cap of 512. + await rm(oldest) + await nextCycle() + expect(indexedSessionCount()).toBe(total) + + await indexer?.reconcile({ full: true }) + expect(indexedSessionCount()).toBe(total - 1) +}) + +// F1: `fullSweepDue` stayed set across the sweep's await and was cleared on the +// way out, so a request raised while a sweep was running was erased by the +// sweep it arrived during. The pass takes the flag on entry now, and an +// unfinished sweep is what puts it back. +it('runs another sweep when one is asked for during a sweep', async () => { + for (let index = 0; index < 20; index++) { + const session = `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`recent session ${index}`], session) + } + const late = transcriptPath(OTHER_SESSION_ID) + + // Newest-one per root, so nothing but a second sweep can reach a file that + // appears after this sweep's discovery has already run. The clock is the one + // synchronous seam into a pass: it is read between files. + newIndexer({ recentPerAgent: 1 }) + let armed = false + // Once a row has landed the pass is provably inside its read loop, which is + // after it took the sweep flag and before it hands its verdicts back. + clock.onNow = () => { + if (armed || indexedSessionCount() === 0) { + return + } + armed = true + mkdirSync(dirname(late), { recursive: true }) + writeFileSync(late, `${claudeLines(['a late conversation'], OTHER_SESSION_ID, 0).join('\n')}\n`) + const backdated = new Date(Date.now() - 86_400_000) + utimesSync(late, backdated, backdated) + void indexer?.reconcile({ full: true }) + } + await indexer?.start() + await indexer?.settled() + + expect(sessionsMatching('late')).toEqual([OTHER_SESSION_ID]) +}) + +// The duty cycle, as a test: a pass reads for at most its deadline and hands +// the rest back, and the timer only re-arms once the pass has settled, so the +// share of the wall clock the index takes is bounded by construction. +it('hands the rest of a pass back when it runs out of wall time', async () => { + for (let index = 0; index < 20; index++) { + const session = `0000${String(index).padStart(4, '0')}-bbbb-4ccc-8ddd-eeeeeeeeeeee` + await writeClaudeTranscript(transcriptPath(session), [`deadlined session ${index}`], session) + } + await newIndexer(readsPerPass(16)).start() + + expect(indexer?.status().filesIndexed).toBe(16) + + // And the pass after it picks up exactly the four it did not reach. + await nextCycle() + expect(indexer?.status()).toMatchObject({ filesIndexed: 20, filesDue: 0 }) +}) diff --git a/src/main/ai-vault-search/session-search-indexer.ts b/src/main/ai-vault-search/session-search-indexer.ts new file mode 100644 index 00000000000..26213242c49 --- /dev/null +++ b/src/main/ai-vault-search/session-search-indexer.ts @@ -0,0 +1,325 @@ +import { systemSessionSearchClock, type SessionSearchClock } from './session-search-clock' +import { + DEFAULT_SESSION_SEARCH_FULL_SWEEP_EVERY_CYCLES, + DEFAULT_SESSION_SEARCH_PASS_DEADLINE_FRACTION, + DEFAULT_SESSION_SEARCH_RECENT_PER_AGENT, + DEFAULT_SESSION_SEARCH_RECONCILE_INTERVAL_MS, + type SessionSearchIndexerOptions +} from './session-search-indexer-options' +import { SessionSearchDirectoryListings } from './session-search-directory-listings' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { runSessionSearchPass } from './session-search-pass' +import { sessionSearchHistoryCutoffMs } from './session-search-retention-policy' +import { SessionSearchStore, type SessionSearchStateCounts } from './session-search-store' +import type { SessionSearchDegradedRoot } from './session-search-degraded-roots' +import { SessionSearchWorkLoop } from './session-search-work-loop' + +/** + * Database paths a live indexer already owns. + * + * One process, one writer, one consumer registration per index. Two indexers on + * one path both register with the reader, so every transcript is read and + * written twice and the second write is fenced by the first at random. The + * recipe for every configuration change is close-then-construct, so the + * ordering that causes this is the one the recipe already rules out; this is + * what says so rather than letting it corrupt quietly. + */ +const liveIndexerPaths = new Set() + +export type SessionSearchIndexPhase = 'idle' | 'indexing' | 'current' | 'degraded' | 'closed' + +export type SessionSearchIndexStatus = { + phase: SessionSearchIndexPhase + /** Rows whose content matches the file at the stat the row records. */ + filesIndexed: number + /** Rows owed a whole read: a declined append, or a window that widened. */ + filesDue: number + /** Rows whose last read did not commit. */ + filesFailed: number + degradedRoots: SessionSearchDegradedRoot[] + lastReconcileAt: number | null + /** When a whole-machine sweep last finished; null until one has. */ + lastSweepCompletedAt: number | null +} + +/** + * Owns freshness for the index store: a whole-machine sweep, then a timer that + * keeps the newest N transcripts per agent reconciled and sweeps again every + * `fullSweepEveryCycles`. + * + * A library, not a service. It knows nothing about Electron, the app lifecycle, + * settings storage, IPC or the panel, and nothing here reads a setting or + * registers itself anywhere. Whoever constructs it decides all of that. + * + * **The store is the only memory.** Every question a pass asks between passes — + * what is owed a read, what has failed and how often, what the index holds and + * therefore what may have been deleted, what to report — is answered by a row + * in the `files` table. There is no queue, no watch set, no hold-out map and no + * counter with a reset rule. + * + * What is left here, and why none of it can be a row: + * - `previousRootsWithFiles`, the one bit per root the retirement walk's grace + * needs. Deliberately not durable: see the mountpoint trade in + * `session-search-deleted-sources.ts`. + * - `cyclesSinceSweep` and `sweepNext`, which are about the timer rather than + * about any file, and mean nothing to a second process. + * - `degradedRoots`, `lastReconcileAt` and `lastSweepCompletedAt`: what the last + * pass observed, held so `status()` can answer between passes. + * - `lastCounts`, the one cached query result, read only after `close()` so that + * describing what happened does not reopen a handle the owner has finished + * with. While the indexer is open every call re-queries. + * + * **Immutable after construction.** There is no `pause`, `resume`, `clear` or + * `setHistoryDays`. A configuration change is `close()` and a new instance; + * throwing the index away is + * `close(); removeSessionSearchDatabase(databasePath);` and a new instance. + * Widening retention is a new instance whose opening sweep admits the older + * files; narrowing is the purge that opens every full sweep. + * + * The guarantee it makes: while started, a transcript among the newest N per + * agent that grows, is replaced or is deleted is reflected in the index within + * one reconcile interval. Everything else is reached by the periodic sweep. + */ +export class SessionSearchIndexer { + private readonly ownershipPath: string + private readonly clock: SessionSearchClock + private readonly intervalMs: number + private readonly passDeadlineMs: number + private readonly recentPerAgent: number + private readonly fullSweepEveryCycles: number + private readonly onError: (error: unknown) => void + + private readonly loop: SessionSearchWorkLoop + private readonly store: SessionSearchStore + private readonly unregister: () => void + /** Null until a pass has recorded one; an empty set is a real observation. */ + private previousRootsWithFiles: ReadonlySet | null = null + private degradedRoots: SessionSearchDegradedRoot[] = [] + private lastReconcileAt: number | null = null + private lastSweepCompletedAt: number | null = null + private lastCounts: SessionSearchStateCounts | null = null + private cyclesSinceSweep = 0 + private sweepNext = false + private started = false + private closed = false + + constructor(private readonly options: SessionSearchIndexerOptions) { + this.ownershipPath = resolve(options.databasePath) + this.clock = options.clock ?? systemSessionSearchClock + this.intervalMs = options.reconcileIntervalMs ?? DEFAULT_SESSION_SEARCH_RECONCILE_INTERVAL_MS + this.passDeadlineMs = + options.passDeadlineMs ?? + Math.max(1, Math.floor(this.intervalMs / DEFAULT_SESSION_SEARCH_PASS_DEADLINE_FRACTION)) + this.recentPerAgent = options.recentPerAgent ?? DEFAULT_SESSION_SEARCH_RECENT_PER_AGENT + this.fullSweepEveryCycles = Math.max( + 1, + options.fullSweepEveryCycles ?? DEFAULT_SESSION_SEARCH_FULL_SWEEP_EVERY_CYCLES + ) + const onError = options.onError ?? ((error) => console.warn('[ai-vault-search]', error)) + this.onError = onError + this.loop = new SessionSearchWorkLoop({ + clock: this.clock, + intervalMs: this.intervalMs, + onFailure: onError + }) + if (liveIndexerPaths.has(this.ownershipPath)) { + throw new Error( + `SessionSearchIndexer: ${options.databasePath} already has a live indexer; close it first` + ) + } + // Store, registration and indexer share one lifetime, which is what makes + // the object immutable: there is no second open to get out of step with. + // Claimed only once the store is open, because a construction that throws + // has no `close()` to release the claim: registering first would leave the + // path owned by an object that does not exist, and every later attempt at + // it -- including the one that fixes whatever broke the open -- would be + // refused for the life of the process. + this.store = new SessionSearchStore(options.databasePath, onError) + liveIndexerPaths.add(this.ownershipPath) + this.store.setRetentionCutoffMs(this.cutoffMs()) + this.unregister = registerSessionSearchIndexConsumer(this.store) + } + + /** Runs a full sweep, then reconciles on the interval until closed. */ + start(): Promise { + if (this.closed || this.started) { + return this.loop.settled + } + this.started = true + this.sweepNext = true + return this.tick() + } + + /** + * Runs one pass now, off the timer. A full pass sweeps every root. + * + * Refused before `start()` and after `close()`: a pass against an indexer + * nobody started writes the index once and leaves it to go stale with no + * timer armed to notice the next change, and a pass against a closed one has + * no store to write to. Both are caller bugs, so both throw rather than + * resolving as though a pass had run. + */ + reconcile(options: { full?: boolean } = {}): Promise { + if (this.closed) { + throw new Error('SessionSearchIndexer.reconcile: the indexer is closed') + } + if (!this.started) { + throw new Error('SessionSearchIndexer.reconcile: start() first') + } + this.sweepNext ||= options.full === true + return this.tick() + } + + /** + * What the index holds, read from the rows rather than tallied. + * + * A second connection can compute every number here with one `GROUP BY`, + * which is the point: nothing is counted as it happens, so nothing can drift + * from what the database actually holds or need a rule about when to reset. + */ + status(): SessionSearchIndexStatus { + // A closed indexer reports what it last knew: opening a shut handle to + // answer a call whose whole job is to describe what happened is how a close + // came to report a database error to the owner who asked for it. + const settled = (this.closed ? this.lastCounts : this.readCounts()) ?? { + current: 0, + due: 0, + failed: 0 + } + return { + phase: this.phase(settled), + filesIndexed: settled.current, + filesDue: settled.due, + filesFailed: settled.failed, + degradedRoots: this.degradedRoots.map((root) => ({ ...root })), + lastReconcileAt: this.lastReconcileAt, + lastSweepCompletedAt: this.lastSweepCompletedAt + } + } + + /** Stops everything. Nothing queued before this call may run afterwards. */ + close(): void { + if (this.closed) { + return + } + // Read before the handle goes, so a status call afterwards reports what the + // index last held rather than opening a database its owner has finished with. + this.lastCounts = this.readCounts() ?? this.lastCounts + this.closed = true + // The loop, not just its timer: a task queued before this call would + // otherwise still run against a store this line is about to close. + this.loop.close() + this.unregister() + this.store.close() + liveIndexerPaths.delete(this.ownershipPath) + } + + /** Tests only: everything else drives this through the timer. */ + settled(): Promise { + return this.loop.settled + } + + private readCounts(): SessionSearchStateCounts | null { + try { + const counts = this.store.stateCounts() + this.lastCounts = counts + return counts + } catch (error) { + this.onError(error) + return this.lastCounts + } + } + + /** + * `current` is a claim, so it takes all three: no row owed a read, no row + * whose last read failed, and a whole sweep that finished. `idle` is the + * other end of it — an indexer nobody started has not promised to index + * anything, and calling that `current` would claim an index nobody built is + * up to date. + */ + private phase(counts: SessionSearchStateCounts): SessionSearchIndexPhase { + if (this.closed) { + return 'closed' + } + if (!this.started) { + return 'idle' + } + // A root the pass could not read, or a file it could not read: both are gaps + // the index knows about and cannot close on its own. + if (this.degradedRoots.length > 0 || counts.failed > 0) { + return 'degraded' + } + return counts.due === 0 && this.lastSweepCompletedAt !== null ? 'current' : 'indexing' + } + + private tick(): Promise { + return this.loop.queue( + (signal) => this.pass(signal), + () => void this.tick() + ) + } + + private async pass(signal: AbortSignal): Promise { + // The window moves with the clock, and the decide step reads it from the + // store. Setting it once at construction leaves a sweep purging rows that + // the very next candidate check happily re-indexes. + this.store.setRetentionCutoffMs(this.cutoffMs()) + // The one bound on a pass: wall time. What it does not reach is still owed, + // because a row says so and nothing had to be written down. + const startedAt = this.clock.now() + const full = this.sweepNext + // Taken on entry, not cleared on the way out: a `reconcile({ full: true })` + // raised while this pass is running sets it again, and clearing it at the + // end would erase that request along with this pass's own. + this.sweepNext = false + try { + const result = await runSessionSearchPass({ + store: this.store, + roots: this.options.roots, + full, + recentPerAgent: this.recentPerAgent, + previousRootsWithFiles: this.previousRootsWithFiles ?? undefined, + overdue: () => this.clock.now() - startedAt >= this.passDeadlineMs, + // One readdir per directory for the whole pass, shared by every step. + listings: new SessionSearchDirectoryListings(), + signal + }) + if (!result.completed) { + // A pass cut short learned nothing about root health, and publishing its + // empty findings would clear a live alarm. A sweep stays owed. + this.sweepNext ||= full + return + } + this.degradedRoots = result.degradedRoots + this.previousRootsWithFiles = result.rootsWithFiles + this.lastReconcileAt = this.clock.now() + // A backlog outside the recency window is only visible to a sweep, so a + // pass that ran out of time asks for one. It is self-limiting: the first + // pass that finishes its reads hands the interval back to cycles. + this.sweepNext ||= result.outOfTime + if (full) { + this.lastSweepCompletedAt = this.lastReconcileAt + this.cyclesSinceSweep = 0 + return + } + // A root that came back, a tree restored from a backup, an old transcript + // deleted: only a sweep sees any of it, and the count of cycles is the + // whole rule for when one is owed. + this.cyclesSinceSweep += 1 + if (this.cyclesSinceSweep >= this.fullSweepEveryCycles) { + this.sweepNext = true + } + } catch (error) { + // The flag is this method's to hold, so it is this method's to give back: + // a pass that threw part way learned nothing, and losing it here would + // leave nothing armed to try again. + this.sweepNext ||= full + throw error + } + } + + private cutoffMs(): number | null { + return sessionSearchHistoryCutoffMs(this.options.historyDays, this.clock.now()) + } +} +import { resolve } from 'node:path' diff --git a/src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts b/src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts new file mode 100644 index 00000000000..79cbe70eb7f --- /dev/null +++ b/src/main/ai-vault-search/session-search-lifecycle-matrix.test.ts @@ -0,0 +1,378 @@ +import { chmod, mkdir, rename, rm } from 'node:fs/promises' +import { delimiter, dirname, join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import type { SessionSearchIndexerOptions } from './session-search-indexer-options' +import { removeSessionSearchDatabase } from './session-search-schema' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + writeMessageGraphTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +/* + * The lifecycle matrix: every operation a caller can perform, against every + * shape an unreachable root takes, against both ways discovery reports a root. + * + * The indexer is immutable, so "every operation" is a shorter list than it was: + * `pause`, `resume`, `clear`, `setHistoryDays` and `invalidate` are gone, and + * the two of them a caller still needs — a settings change and throwing the + * index away — are here as what replaced them, a new instance over the same + * path. In their place are the two passes the immutable design added: the + * periodic sweep, and a pass whose wall-clock deadline expires on its first file. + * + * What each cell asserts: + * A. No row is retired for a file that still exists. Throwing the index away + * is the one exception, and it is stated per operation rather than excused. + * B. The unreachable root is named in `degradedRoots`, by a real directory + * path — never the delimiter-joined label a merged discovery reports. + * C. The phase is never `current` while a root is degraded. + * D. Once the root is reachable again, a sweep indexes everything under it. + * + * Round 6 ran this as a throwaway harness on the previous design; it lives in + * the repository now. Two of its shapes changed with the stateless walk. The + * "present but empty mountpoint" shape is gone, because a readable root that + * lists nothing is no longer treated as unreachable — that is a root the user + * emptied, and `session-search-deleted-sources.ts` states the trade. In its + * place is a root whose transcripts sit behind an unreadable subdirectory, + * which is the partial-tree case the old shape never covered. + */ + +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const INTERVAL_MS = 20_000 +const SESSIONS = ['aaaaaaaa', 'bbbbbbbb', 'cccccccc'] + +type RootShape = { + name: string + /** Where the unreachable root's transcripts live, and where its files go. */ + detachedRoot: (harness: SessionSearchIndexerHarness) => string + detachedFile: (harness: SessionSearchIndexerHarness, session: string) => string + writeDetached: (path: string, session: string) => Promise + healthyFile: (harness: SessionSearchIndexerHarness, session: string) => string + writeHealthy: (path: string, session: string) => Promise +} + +const OPENCLAW_SESSION_DIR = join('agents', 'main', 'sessions') + +const ROOT_SHAPES: RootShape[] = [ + { + name: 'roots discovery reports one per directory', + detachedRoot: (harness) => harness.roots.claudeProjectsDir ?? '', + detachedFile: (harness, session) => join(harness.claudeProjectDir, `${session}.jsonl`), + writeDetached: (path, session) => + writeClaudeTranscript(path, [`detached ${session}`], fullSessionId(session)), + healthyFile: (harness, session) => join(harness.roots.piSessionsDir ?? '', `${session}.jsonl`), + writeHealthy: (path, session) => writeMessageGraphTranscript(path, [`healthy ${session}`]) + }, + { + name: 'roots a merged discovery joins into one label', + detachedRoot: (harness) => join(harness.roots.openclawStateDir ?? '', 'agents'), + detachedFile: (harness, session) => + join(harness.roots.openclawStateDir ?? '', OPENCLAW_SESSION_DIR, `${session}.jsonl`), + writeDetached: (path, session) => writeMessageGraphTranscript(path, [`detached ${session}`]), + healthyFile: (harness, session) => + join(harness.roots.openclawLegacyStateDir ?? '', OPENCLAW_SESSION_DIR, `${session}.jsonl`), + writeHealthy: (path, session) => writeMessageGraphTranscript(path, [`healthy ${session}`]) + } +] + +type UnreachableShape = { + name: string + needsDeniedRead: boolean + /** + * Whether an empty index can see this at all. Reading the root itself is the + * one probe a pass makes with no rows to go on: a root that answers ENOENT is + * what an uninstalled agent answers too, and a readable root with an + * unreadable subdirectory is swallowed by the file walker, which returns + * rather than reporting. Both are invisible until the index holds a row under + * the root, which is the evidence the retirement walk runs on. + */ + visibleWithNoRows: boolean + detach: (root: string, transcriptDir: string, parked: string) => Promise + attach: (root: string, transcriptDir: string, parked: string) => Promise +} + +const UNREACHABLE_SHAPES: UnreachableShape[] = [ + { + name: 'the root itself is not there', + needsDeniedRead: false, + visibleWithNoRows: false, + detach: (root, _transcriptDir, parked) => rename(root, parked), + attach: (root, _transcriptDir, parked) => rename(parked, root) + }, + { + name: 'the root refuses to list', + needsDeniedRead: true, + visibleWithNoRows: true, + detach: (root) => chmod(root, 0o000), + attach: (root) => chmod(root, 0o755) + }, + { + name: 'the transcripts sit behind a directory that refuses to list', + needsDeniedRead: true, + visibleWithNoRows: false, + detach: (_root, transcriptDir) => chmod(transcriptDir, 0o000), + attach: (_root, transcriptDir) => chmod(transcriptDir, 0o755) + } +] + +type Operation = { + name: string + /** True when the operation throws the index away, so no row survives it. */ + clearsIndex?: boolean + /** Healthy-root sessions the operation deletes from disk. */ + deletes?: readonly string[] + /** Construction options for every indexer this cell opens. */ + options?: Partial + run: (context: MatrixContext) => Promise +} + +const OPERATIONS: Operation[] = [ + { name: 'one cycle', run: (context) => context.cycle() }, + { + name: 'two cycles', + run: async (context) => { + await context.cycle() + await context.cycle() + } + }, + { + name: 'close and restart', + run: (context) => context.reopen() + }, + { + name: 'two full reconciles', + run: async (context) => { + await context.indexer().reconcile({ full: true }) + await context.indexer().reconcile({ full: true }) + } + }, + { + name: 'one healthy transcript deleted', + deletes: SESSIONS.slice(0, 1), + run: (context) => context.cycle() + }, + { + name: 'every healthy transcript deleted', + deletes: SESSIONS, + run: async (context) => { + // Twice: a root that goes from holding transcripts to holding none in one + // pass is unverifiable for that pass, so the second is the proving one. + await context.indexer().reconcile({ full: true }) + await context.indexer().reconcile({ full: true }) + } + }, + { + // The cadence that replaced every re-arm-on-recovery rule: no caller asks + // for this sweep, so the cell drives it off the timer alone. + name: 'the periodic sweep comes round', + options: { fullSweepEveryCycles: 2 }, + run: async (context) => { + await context.cycle() + await context.cycle() + await context.cycle() + } + }, + { + // Every pass is out of wall time from its first file, so each one hands + // almost all of its work back. A pass that read almost nothing must still + // not conclude anything about what it did not reach. + name: 'every pass out of time at its first file', + options: { passDeadlineMs: 0 }, + run: async (context) => { + await context.cycle() + await context.cycle() + } + }, + { + // What replaced `setHistoryDays`: a new instance over the same database. + // Every transcript here was written just now, so a 30-day window holds all + // of them and no row may be purged. + name: 'reconstructed for a narrower history window', + run: (context) => context.reopen({ historyDays: 30 }) + }, + { + // What replaced `clear()`, exactly as the PR body documents it. + name: 'the index thrown away and rebuilt', + clearsIndex: true, + run: (context) => context.reopen({ removeDatabase: true }) + } +] + +type MatrixContext = { + indexer: () => SessionSearchIndexer + /** Closes and constructs again over the same path: the immutable design's one edit. */ + reopen: (args?: { historyDays?: number | null; removeDatabase?: boolean }) => Promise + cycle: () => Promise + detachedRoot: string + detachedPaths: string[] +} + +function fullSessionId(prefix: string): string { + return `${prefix}-bbbb-4ccc-8ddd-eeeeeeeeeeee` +} + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-lifecycle') + indexer = null +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function open(overrides: Partial = {}): SessionSearchIndexer { + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS, + ...overrides + }) + return indexer +} + +/** + * Runs cycles until the index stops growing. Every operation but the + * out-of-time one settles on the first call; that one reads a transcript a pass. + */ +async function driveUntilIndexed(maxCycles: number): Promise { + let held = indexedSessions().length + for (let cycle = 0; cycle < maxCycles; cycle++) { + clock.advance(INTERVAL_MS) + await indexer?.settled() + const now = indexedSessions().length + if (now === held) { + return + } + held = now + } +} + +/** Session ids the index answers for, whichever agent wrote them. */ +function indexedSessions(): string[] { + return harness + .read( + (db: SyncDatabase) => + db.prepare('SELECT session_id AS id FROM sessions').all() as { id: string }[] + ) + .map((row) => row.id) + .sort() +} + +for (const roots of ROOT_SHAPES) { + for (const unreachable of UNREACHABLE_SHAPES) { + describe.skipIf(unreachable.needsDeniedRead && !CAN_DENY_READ)( + `${roots.name}, ${unreachable.name}`, + () => { + for (const operation of OPERATIONS) { + it(operation.name, async () => { + const detachedRoot = roots.detachedRoot(harness) + const detachedPaths = SESSIONS.map((session) => roots.detachedFile(harness, session)) + const healthyPaths = SESSIONS.map((session) => roots.healthyFile(harness, session)) + for (const [index, session] of SESSIONS.entries()) { + await roots.writeDetached(detachedPaths[index] ?? '', session) + await roots.writeHealthy(healthyPaths[index] ?? '', session) + } + const transcriptDir = dirname(detachedPaths[0] ?? '') + const parked = join(harness.root, 'parked-root') + + await open(operation.options).start() + // A deadline that expires on the first file reads one transcript a + // pass, so the setup drives passes until the index has caught up. + await driveUntilIndexed(SESSIONS.length * 2) + const detachedIds = detachedPaths.map((_path, index) => + roots === ROOT_SHAPES[0] + ? fullSessionId(SESSIONS[index] ?? '') + : (SESSIONS[index] ?? '') + ) + const healthyIds = SESSIONS.map((session) => session) + expect(indexedSessions()).toEqual([...detachedIds, ...healthyIds].sort()) + // One cycle so the watch set holds the recency window, which is the + // state a running indexer is in when a volume goes away. + clock.advance(INTERVAL_MS) + await indexer?.settled() + + await unreachable.detach(detachedRoot, transcriptDir, parked) + try { + const kept = SESSIONS.filter((session) => !operation.deletes?.includes(session)) + for (const session of operation.deletes ?? []) { + await rm(healthyPaths[SESSIONS.indexOf(session)] ?? '') + } + await operation.run({ + indexer: () => indexer as SessionSearchIndexer, + reopen: async (args = {}) => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + if (args.removeDatabase === true) { + removeSessionSearchDatabase(harness.databasePath) + } + const overrides = { ...operation.options } + if ('historyDays' in args) { + overrides.historyDays = args.historyDays + } + await open(overrides).start() + await driveUntilIndexed(SESSIONS.length * 2) + }, + cycle: async () => { + clock.advance(INTERVAL_MS) + await indexer?.settled() + }, + detachedRoot, + detachedPaths + }) + + // A: nothing that still exists lost its rows. + const survivingDetached = operation.clearsIndex ? [] : detachedIds + expect(indexedSessions()).toEqual([...survivingDetached, ...kept].sort()) + + const status = indexer?.status() + const degraded = status?.degradedRoots.map((root) => root.root) ?? [] + // With no rows under it, the only thing a pass can go on is + // whether the root itself refuses to list. + if (operation.clearsIndex && !unreachable.visibleWithNoRows) { + expect(degraded).not.toContain(detachedRoot) + } else { + // B: named, by a real directory rather than a joined label. + expect(degraded).toContain(detachedRoot) + expect(degraded.every((root) => !root.includes(delimiter))).toBe(true) + // C: not current while a root is degraded. + expect(status?.phase).not.toBe('current') + } + } finally { + await unreachable.attach(detachedRoot, transcriptDir, parked) + } + + // D: reachable again, a sweep reads the whole tree back. + await mkdir(dirname(healthyPaths[0] ?? ''), { recursive: true }) + await indexer?.reconcile({ full: true }) + await driveUntilIndexed(SESSIONS.length * 2) + expect(indexedSessions()).toEqual( + [ + ...detachedIds, + ...SESSIONS.filter((session) => !operation.deletes?.includes(session)) + ].sort() + ) + }) + } + } + ) + } +} diff --git a/src/main/ai-vault-search/session-search-live-transcript.test.ts b/src/main/ai-vault-search/session-search-live-transcript.test.ts new file mode 100644 index 00000000000..1ea58ca5283 --- /dev/null +++ b/src/main/ai-vault-search/session-search-live-transcript.test.ts @@ -0,0 +1,208 @@ +import { mkdtemp, rm, writeFile, appendFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { + registerTranscriptConsumer, + resetTranscriptConsumersForTests, + type TranscriptSessionIdentity +} from '../ai-vault/session-transcript-consumers' +import { requestWholeTranscriptRead } from '../ai-vault/session-transcript-reader' +import SyncDatabase from '../sqlite/sync-database' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { SessionSearchStore } from './session-search-store' +import { + assistantRecord, + CLAUDE_SESSION_ID as SESSION_ID, + CODEX_ROLLOUT_FILE, + CODEX_SESSION_ID, + codexRolloutLines, + parseTranscript, + userRecord +} from './session-search-transcript-fixtures' + +let tempRoots: string[] = [] +let store: SessionSearchStore +// The store keeps its connection private, so row assertions need a second one. +let reader: SyncDatabase +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + const path = join(await makeTempDir(), 'index.sqlite') + store = new SessionSearchStore(path, (error) => errors.push(error)) + registerSessionSearchIndexConsumer(store) + reader = new SyncDatabase(path, { readonly: true }) +}) + +afterEach(async () => { + resetTranscriptConsumersForTests() + reader.close() + store.close() + await Promise.all(tempRoots.map((root) => rm(root, { recursive: true, force: true }))) + tempRoots = [] +}) + +async function makeTempDir(): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-session-search-live-')) + tempRoots.push(root) + return root +} + +/** Sessions a query would return for one FTS term, read on a second handle. */ +function sessionsMatching(term: string): string[] { + return ( + reader + .prepare( + `SELECT DISTINCT s.session_id AS id FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY s.session_id` + ) + .all(term) as { id: string }[] + ).map((row) => row.id) +} + +it('indexes a Claude transcript through the reader and resumes on append', async () => { + const root = await makeTempDir() + const path = join(root, `${SESSION_ID}.jsonl`) + await writeFile( + path, + `${[ + userRecord(0, 'find the flaky terminal reattach'), + assistantRecord(1, 'look at resolveTerminalPath first') + ].join('\n')}\n` + ) + await parseTranscript(path) + expect(errors).toEqual([]) + expect(sessionsMatching('reattach')).toEqual([SESSION_ID]) + // The identifier column shadows a camel-case symbol into its pieces. + expect(sessionsMatching('terminal')).toEqual([SESSION_ID]) + + await appendFile(path, `${assistantRecord(2, 'the zygomorphic follow-up landed')}\n`) + const resumed = await parseTranscript(path) + // The reader resumed, so the index saw an `append`, not a whole re-read. + expect(resumed.stats).toMatchObject({ incremental: 1, fullParses: 0 }) + expect(errors).toEqual([]) + expect(sessionsMatching('zygomorphic')).toEqual([SESSION_ID]) + // An append extends one session rather than creating a second. + expect(reader.prepare('SELECT count(*) AS n FROM sessions').get()).toEqual({ + n: 1 + }) +}) + +it('keeps a tool result searchable but out of the conversation half', async () => { + const root = await makeTempDir() + const codexHome = await makeTempDir() + const path = join(root, CODEX_ROLLOUT_FILE) + await writeFile( + path, + `${codexRolloutLines( + ['rg', 'pericardium'], + `outputonly ${'padding '.repeat(600)}tailonly`, + 'promptonly search for the module' + ).join('\n')}\n` + ) + await parseTranscript(path, 'codex', codexHome) + expect(errors).toEqual([]) + + expect(sessionsMatching('pericardium')).toHaveLength(1) + // The prompt is conversation; the command output is not, and the column + // filter is what tells them apart. + expect(sessionsMatching('outputonly')).toHaveLength(1) + expect(sessionsMatching('tailonly')).toHaveLength(0) + expect(sessionsMatching('rg')).toHaveLength(1) + expect(sessionsMatching('{user_text assistant_text}: promptonly')).toHaveLength(1) + expect(sessionsMatching('{user_text assistant_text}: outputonly')).toHaveLength(0) + expect(sessionsMatching('{user_text assistant_text}: rg')).toHaveLength(0) +}) + +/** What `start.identity()` returns at each message of one read. */ +function recordIdentityPerMessage(): (TranscriptSessionIdentity | null)[] { + const seen: (TranscriptSessionIdentity | null)[] = [] + registerTranscriptConsumer({ + beginRead: (start) => ({ + message: () => { + seen.push(start.identity?.() ?? null) + }, + finish: () => undefined + }) + }) + return seen +} + +it('names the session mid-read, before the reader has finished the file', async () => { + const root = await makeTempDir() + const path = join(root, `${SESSION_ID}.jsonl`) + await writeFile( + path, + `${[ + userRecord(0, 'find the flaky terminal reattach'), + assistantRecord(1, 'look at resolveTerminalPath first') + ].join('\n')}\n` + ) + const seen = recordIdentityPerMessage() + await parseTranscript(path) + + // A chunked read commits partway through a file this size or larger, so what + // it can name the session with is exactly this. + expect(seen.length).toBeGreaterThan(0) + expect(seen[0]).toMatchObject({ + sessionId: SESSION_ID, + cwd: '/repo/app', + createdAt: expect.any(String) + }) +}) + +it('names a Codex session mid-read from its own opening record', async () => { + const root = await makeTempDir() + const codexHome = await makeTempDir() + const path = join(root, CODEX_ROLLOUT_FILE) + await writeFile( + path, + `${codexRolloutLines(['rg', 'pericardium'], 'src/main/pericardium.ts:12: match', 'search for the pericardium module').join('\n')}\n` + ) + const seen = recordIdentityPerMessage() + await parseTranscript(path, 'codex', codexHome) + + // Codex builds its own resumable state rather than the shared accumulator + // fold, so it is the other half of the surface a chunked commit depends on. + expect(seen[0]).toMatchObject({ + sessionId: CODEX_SESSION_ID, + cwd: '/repo/app' + }) +}) + +it('indexes a file the session list already read past, once a whole read is asked for', async () => { + const root = await makeTempDir() + const path = join(root, `${SESSION_ID}.jsonl`) + await writeFile(path, `${userRecord(0, 'the opening prompt')}\n`) + + // The state on first enablement inside a running app: the session list has + // read this file, so the parse cache is warm, while the index is empty. + resetTranscriptConsumersForTests() + await parseTranscript(path) + registerSessionSearchIndexConsumer(store) + + await appendFile(path, `${assistantRecord(1, 'a zygomorphic reply')}\n`) + const appended = await parseTranscript(path) + expect(appended.stats).toMatchObject({ incremental: 1, fullParses: 0 }) + // The append continued from a byte offset the index never saw, so it declined. + expect(sessionsMatching('zygomorphic')).toEqual([]) + + // The index holds no row for this file at all, and that is the record: a + // path the file table does not name is read from the start by the next pass, + // which is what asks the reader to drop the session list's resume point. + expect(store.files()).toEqual([]) + requestWholeTranscriptRead(path) + + const reread = await parseTranscript(path) + expect(reread.stats).toMatchObject({ incremental: 0, fullParses: 1 }) + expect(errors).toEqual([]) + expect(sessionsMatching('zygomorphic')).toEqual([SESSION_ID]) + expect(sessionsMatching('opening')).toEqual([SESSION_ID]) + expect(store.files().map((row) => row.state)).toEqual(['current']) +}) diff --git a/src/main/ai-vault-search/session-search-merged-roots.test.ts b/src/main/ai-vault-search/session-search-merged-roots.test.ts new file mode 100644 index 00000000000..8d41b047cae --- /dev/null +++ b/src/main/ai-vault-search/session-search-merged-roots.test.ts @@ -0,0 +1,135 @@ +import { chmod, rm } from 'node:fs/promises' +import { delimiter, join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeMessageGraphTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +// OpenClaw is the one agent whose roots are alternates for a single install, so +// discovery reports them as ONE discovery whose rootDir is every path joined by +// the platform's path delimiter. That string is not a directory: readdir on it +// answers ENOENT, containment never matches a real file, and a scan issue +// recorded against a real root never compares equal to it. Everything that +// judges a root works on the constituent directories, taken from the same +// source table discovery reads, never by splitting the label -- a directory may +// legally contain the delimiter. + +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const INTERVAL_MS = 20_000 + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-merged-roots') +}) + +afterEach(async () => { + indexer.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +/** OpenClaw reads `/agents/**` and keeps only paths through `sessions`. */ +function openclawTranscript(stateDir: string, name: string): string { + return join(stateDir, 'agents', 'main', 'sessions', `${name}.jsonl`) +} + +function sessionsMatching(term: string): string[] { + return harness.read((db: SyncDatabase) => + ( + db + .prepare( + `SELECT DISTINCT s.session_id AS id FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY s.session_id` + ) + .all(term) as { id: string }[] + ).map((row) => row.id) + ) +} + +it.skipIf(!CAN_DENY_READ)('fences one merged root without taking its partner down', async () => { + const current = harness.roots.openclawStateDir ?? '' + const legacy = harness.roots.openclawLegacyStateDir ?? '' + const mounted = openclawTranscript(current, 'mounted-session') + const local = openclawTranscript(legacy, 'local-session') + await writeMessageGraphTranscript(mounted, ['a conversation on the mounted volume']) + await writeMessageGraphTranscript(local, ['a conversation on local disk']) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS + }) + await indexer.start() + expect(sessionsMatching('conversation').sort()).toEqual(['local-session', 'mounted-session']) + + // One of the two roots goes away; the other is untouched. + await chmod(join(current, 'agents'), 0o000) + try { + await indexer.reconcile({ full: true }) + + const status = indexer.status() + const degraded = status.degradedRoots.map((root) => root.root) + // A real directory, not the joined string discovery reports. + expect(degraded).toContain(join(current, 'agents')) + expect(degraded.every((root) => !root.includes(delimiter))).toBe(true) + // Unprovable, so the unreadable root keeps its rows. + expect(sessionsMatching('mounted')).toEqual(['mounted-session']) + } finally { + await chmod(join(current, 'agents'), 0o755) + } +}) + +it('retires from one merged root while its partner is healthy', async () => { + const current = harness.roots.openclawStateDir ?? '' + const legacy = harness.roots.openclawLegacyStateDir ?? '' + const going = openclawTranscript(current, 'going-session') + await writeMessageGraphTranscript(going, ['a conversation about to be deleted']) + // A sibling in the same root, so deleting one leaves the root listing files + // and therefore healthy: this is a deletion, not an unmount. + await writeMessageGraphTranscript(openclawTranscript(current, 'sibling-session'), [ + 'a conversation beside it' + ]) + await writeMessageGraphTranscript(openclawTranscript(legacy, 'staying-session'), [ + 'a conversation that stays' + ]) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS + }) + await indexer.start() + expect(sessionsMatching('conversation').sort()).toEqual([ + 'going-session', + 'sibling-session', + 'staying-session' + ]) + + // A genuine deletion inside a healthy root still retires normally. + await rm(going) + await indexer.reconcile({ full: true }) + + expect(sessionsMatching('deleted')).toEqual([]) + expect(indexer.status().degradedRoots).toEqual([]) + expect(sessionsMatching('conversation').sort()).toEqual(['sibling-session', 'staying-session']) +}) diff --git a/src/main/ai-vault-search/session-search-message-rows.test.ts b/src/main/ai-vault-search/session-search-message-rows.test.ts new file mode 100644 index 00000000000..d2cf2fbca61 --- /dev/null +++ b/src/main/ai-vault-search/session-search-message-rows.test.ts @@ -0,0 +1,249 @@ +import { expect, it } from 'vitest' +import type { TranscriptMessage } from '../ai-vault/session-transcript-consumers' +import { insertSearchMessage, searchMessageRows } from './session-search-message-rows' +import { + openSessionSearchIndexFile, + type SessionSearchIndexFile +} from './session-search-index-test-fixture' + +/** Every column of the FTS table, so an assertion cannot miss the shadow terms. */ +async function indexedColumns( + index: SessionSearchIndexFile, + message: TranscriptMessage +): Promise { + for (const row of searchMessageRows([message])) { + insertSearchMessage(index.db, 1, row) + } + const full = index.db + .prepare('SELECT user_text, assistant_text, tool_text, identifiers FROM messages_fts') + .all() as Record[] + return full.flatMap((row) => Object.values(row)) +} + +it('splits an oversized message on a line boundary and keeps every character', () => { + const line = `${'padding '.repeat(11)}word\n` + const text = line.repeat(400) + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])].map( + (row) => row.text + ) + + expect(chunks.length).toBeGreaterThan(1) + expect(chunks.join('')).toBe(text) + for (const chunk of chunks) { + expect(chunk.length).toBeLessThanOrEqual(8000) + expect(chunk.endsWith('\n')).toBe(true) + } +}) + +it('cuts at whitespace rather than through the word on the boundary', async () => { + const index = await openSessionSearchIndexFile('ss-rows-whitespace') + try { + // The 8,000th character lands inside `pericardium`. Cutting at the target + // would file `per` under one row and `icardium` under another, and the word + // the user types would match neither. + const text = `${' '.repeat(7997)}pericardium` + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])] + expect(chunks.map((row) => row.text).join('')).toBe(text) + for (const row of chunks) { + insertSearchMessage(index.db, 1, row) + } + + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get('pericardium') + ).toEqual({ n: 1 }) + } finally { + await index.close() + } +}) + +it.each(['/repo/pericardium.ts', 'PROJ-12345', 'C++', 'cafe\u0301ine'])( + 'preserves the exact FTS token %s at a chunk boundary', + async (token) => { + const index = await openSessionSearchIndexFile('ss-rows-tokenchars') + try { + const text = ' '.repeat(7998) + token + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])] + expect(chunks.map((row) => row.text).join('')).toBe(text) + for (const row of chunks) { + insertSearchMessage(index.db, 1, row) + } + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get(`"${token}"`) + ).toEqual({ n: 1 }) + } finally { + await index.close() + } + } +) + +it.each(['\u0305', '\u030d', '\u0332'])( + 'cuts at a combining mark unicode61 treats as a separator: %s', + async (mark) => { + const index = await openSessionSearchIndexFile('ss-rows-unicode-separator') + try { + const text = `${'x'.repeat(7997)}${mark}pericardium` + for (const row of searchMessageRows([{ role: 'user', text, timestamp: null }])) { + insertSearchMessage(index.db, 1, row) + } + expect( + index.db + .prepare("SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH 'pericardium'") + .get() + ).toEqual({ n: 1 }) + } finally { + await index.close() + } + } +) + +it('backs up to any whitespace, not only a newline', () => { + // An ideographic space separates words in a CJK transcript exactly as a + // space does here, and a newline-only backoff tears the token after it. + const text = `${'\u4e00'.repeat(7000)}\u3000${'\u4e8c'.repeat(2000)}` + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])].map( + (row) => row.text + ) + + expect(chunks[0]).toBe(`${'\u4e00'.repeat(7000)}\u3000`) + expect(chunks.join('')).toBe(text) +}) + +it('cuts at punctuation when the window holds no whitespace at all', async () => { + const index = await openSessionSearchIndexFile('ss-rows-minified') + try { + // Valid minified JSON, the shape a tool result carries: 8,000 characters + // without a single space. The 8,000th lands inside `pericardium`, and a + // whitespace-only backoff has nothing in the window to back up to, so it + // files `perica` under one row and `rdium` under the next. + const text = `{"pad":"${'x'.repeat(7976)}","note":"pericardium"}` + expect(JSON.parse(text)).toEqual({ pad: 'x'.repeat(7976), note: 'pericardium' }) + expect(text.slice(7994, 8005)).toBe('pericardium') + expect(/\s/.test(text)).toBe(false) + + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])] + expect(chunks.map((row) => row.text).join('')).toBe(text) + for (const row of chunks) { + insertSearchMessage(index.db, 1, row) + } + + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get('pericardium') + ).toEqual({ n: 1 }) + } finally { + await index.close() + } +}) + +it('keeps a 9,000-character identifier whole rather than cutting at its underscores', () => { + // `_` sits inside a token for this tokenizer, so it is not a boundary. A + // snake_case name that long holds none at all, and the target itself is the + // honest cut — backing up to every `_` would file the name in pieces. + const text = 'ab_'.repeat(3000) + expect(text.length).toBe(9000) + const chunks = [...searchMessageRows([{ role: 'user', text, timestamp: null }])].map( + (row) => row.text + ) + + expect(chunks.map((chunk) => chunk.length)).toEqual([8000, 1000]) + expect(chunks.join('')).toBe(text) +}) + +it('still chunks a message that holds no whitespace at all', () => { + // A 20,000-character token is not a word, so the target itself is the cut and + // the message is still bounded. + const chunks = [ + ...searchMessageRows([{ role: 'user', text: 'a'.repeat(20_000), timestamp: null }]) + ] + expect(chunks.map((row) => row.text.length)).toEqual([8000, 8000, 4000]) +}) + +it('leaves a message that fits as a single row', () => { + const rows = [...searchMessageRows([{ role: 'user', text: 'short enough', timestamp: null }])] + expect(rows.map((row) => row.text)).toEqual(['short enough']) +}) + +it('caps a tool row at its head and never caps the conversation', async () => { + const index = await openSessionSearchIndexFile('ss-rows-tool-cap') + try { + // The reader hands over untruncated text (its own bound is 256 KB per + // message and a consumer may be handed more); the cap is this module's. + const output = `pericardium ${'padding '.repeat(140_000)}` + expect(output.length).toBeGreaterThan(1024 * 1024) + + const toolRows = [...searchMessageRows([{ role: 'tool', text: output, timestamp: null }])] + expect(toolRows).toHaveLength(1) + expect(toolRows[0]!.text.length).toBe(3072) + // The head is what identifies what ran, so it is what survives. + expect(toolRows[0]!.text.startsWith('pericardium ')).toBe(true) + + // The same text as an assistant turn is conversation, and keeps every byte. + const assistantRows = [ + ...searchMessageRows([{ role: 'assistant', text: output, timestamp: null }]) + ] + expect(assistantRows.map((row) => row.text).join('')).toBe(output) + expect(assistantRows.length).toBeGreaterThan(100) + + for (const row of toolRows) { + insertSearchMessage(index.db, 1, row) + } + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get('pericardium') + ).toEqual({ n: 1 }) + } finally { + await index.close() + } +}) + +it('files a tool row under the tool column alone', async () => { + const index = await openSessionSearchIndexFile('ss-message-rows-tool') + try { + for (const row of searchMessageRows([ + { role: 'tool', text: 'rg pericardium', timestamp: null } + ])) { + insertSearchMessage(index.db, 1, row) + } + expect(index.db.prepare('SELECT count(*) AS n FROM messages_fts').get()).toEqual({ n: 1 }) + // What makes a conversation-scoped search exclude it: the column filter, not + // a second table. + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get('{user_text assistant_text}: pericardium') + ).toEqual({ n: 0 }) + expect( + index.db + .prepare('SELECT count(*) AS n FROM messages_fts WHERE messages_fts MATCH ?') + .get('{tool_text}: pericardium') + ).toEqual({ n: 1 }) + } finally { + await index.close() + } +}) + +it('stores a chunk exactly as the transcript wrote it', async () => { + const index = await openSessionSearchIndexFile('ss-rows-verbatim') + try { + const text = 'deploy with AKIAIOSFODNN7EXAMPLE and the resolveTerminalPath fix' + const stored = await indexedColumns(index, { + role: 'assistant', + text, + timestamp: null + }) + + // The index is a second copy of content the user already holds in plaintext, + // so it neither rewrites nor drops any of it. + expect(stored).toContain(text) + // Identifier shadow terms come off that same raw chunk. + expect(stored.some((column) => column.includes('resolve terminal path'))).toBe(true) + } finally { + await index.close() + } +}) diff --git a/src/main/ai-vault-search/session-search-message-rows.ts b/src/main/ai-vault-search/session-search-message-rows.ts new file mode 100644 index 00000000000..21254c183aa --- /dev/null +++ b/src/main/ai-vault-search/session-search-message-rows.ts @@ -0,0 +1,135 @@ +import type SyncDatabase from '../sqlite/sync-database' +import type { TranscriptMessage } from '../ai-vault/session-transcript-consumers' +import { sliceAtCodeUnitLimit } from '../ai-vault/session-scanner-text-normalization' +import { identifierShadowText } from './session-search-identifier-split' + +const CHUNK_TARGET_CHARS = 8000 + +/** + * How much of one tool output is indexed. Its head: a command, its arguments and + * the first lines of what it printed are what a user searches for, while the + * tail is the padding that makes these messages large in the first place. + * + * Tool output is 80-97 % of a transcript's bytes, and a single one can be a + * quarter of a megabyte (the reader's own per-message bound). Without this the + * index, the in-memory buffer a read holds and the transaction it commits are + * all sized by how much a tool printed rather than by how much is worth + * searching. 3 KB was the accuracy/size sweet spot in the original design + * measurement. User and assistant text is never capped: it is the conversation, + * and it is small. + */ +const TOOL_ROW_CHARS = 3072 + +// Keep unicode61's tokenchars intact, including before an available space. +// SQLite ext/fts5/fts5_unicode2.c: sqlite3Fts5UnicodeIsdiacritic, with remove_diacritics=1. +const FOLDED_DIACRITIC = + /[\u0300-\u0304\u0306-\u030c\u030f\u0311\u031b\u0323-\u0328\u032d-\u032e\u0330-\u0331]/ +const TOKEN_BOUNDARY = /[^\p{L}\p{N}\p{Co}_.\-/+\uD800-\uDFFF]/u + +/** + * Index just past the last token boundary in `[floor, end)`, or -1 when the + * window holds none. Not only a newline: a wrapped paragraph, a CJK transcript + * separated by ideographic spaces and a minified log all chunk on a boundary a + * tokenizer would have picked anyway. + */ +function lastTokenBoundaryEnd(text: string, floor: number, end: number): number { + for (let at = end - 1; at >= floor; at--) { + if (TOKEN_BOUNDARY.test(text[at]!) && !FOLDED_DIACRITIC.test(text[at]!)) { + return at + 1 + } + } + return -1 +} + +/** + * Splits an oversized message into rows of at most `CHUNK_TARGET_CHARS`, cutting + * on a token boundary so no token is torn in half and every word stays + * searchable. A phrase that straddles two chunks is not matched: chunks are + * separate FTS rows and FTS5 cannot span them. + */ +function* textChunks(text: string): Generator { + if (text.length <= CHUNK_TARGET_CHARS) { + yield text + return + } + let start = 0 + while (start < text.length) { + let end = Math.min(text.length, start + CHUNK_TARGET_CHARS) + if (end < text.length) { + // Only the second half of the window: backing up further would trade a + // torn token for chunks half the size. No boundary at all in 4,000 + // characters is not a word, so the target itself is the honest cut. + const split = lastTokenBoundaryEnd(text, start + CHUNK_TARGET_CHARS / 2, end) + if (split > start) { + end = split + } + } + yield text.slice(start, end) + start = end + } +} + +/** + * The row policy for one message: a `tool` message becomes one capped row, and + * anything else becomes N chunks, because FTS5 ranks a short row far better + * than a huge one. + */ +export function* searchMessageRows( + messages: Iterable +): Generator { + for (const message of messages) { + if (message.role === 'tool') { + yield { + ...message, + text: sliceAtCodeUnitLimit(message.text, TOOL_ROW_CHARS) + } + continue + } + for (const text of textChunks(message.text)) { + yield { ...message, text } + } + } +} + +/** + * Writes one row into `messages` and `messages_fts` in the caller's + * transaction, so a message is never present in one and absent from the other. + * A conversation-scoped query filters the columns rather than reading a second + * table (see the schema). + */ +export function insertSearchMessage( + db: SyncDatabase, + sessionId: number, + message: TranscriptMessage +): void { + const text = message.text + const id = db + .prepare('INSERT INTO messages(session_row_id, role, ts) VALUES (?, ?, ?)') + .run(sessionId, message.role, message.timestamp).lastInsertRowid + const user = message.role === 'user' ? text : '' + const assistant = message.role === 'assistant' ? text : '' + const tool = message.role === 'tool' ? text : '' + db.prepare( + 'INSERT INTO messages_fts(rowid,user_text,assistant_text,tool_text,identifiers) VALUES (?,?,?,?,?)' + ).run(id, user, assistant, tool, identifierShadowText(text)) +} + +/** + * Deletes up to `limit` of a session's rows from `messages` and `messages_fts`, + * in the caller's transaction, and reports how many went. Bounded + * because a retention sweep must not hold one transaction over a whole + * session; a replace passes no limit, since its rows and their replacements + * have to land together. + */ +export function deleteSearchMessages(db: SyncDatabase, sessionId: number, limit = -1): number { + const ids = db + .prepare('SELECT id FROM messages WHERE session_row_id = ? LIMIT ?') + .all(sessionId, limit) as { id: number }[] + const full = db.prepare('DELETE FROM messages_fts WHERE rowid = ?') + const message = db.prepare('DELETE FROM messages WHERE id = ?') + for (const { id } of ids) { + full.run(id) + message.run(id) + } + return ids.length +} diff --git a/src/main/ai-vault-search/session-search-native-chat-indexing.test.ts b/src/main/ai-vault-search/session-search-native-chat-indexing.test.ts new file mode 100644 index 00000000000..deab4e8b785 --- /dev/null +++ b/src/main/ai-vault-search/session-search-native-chat-indexing.test.ts @@ -0,0 +1,113 @@ +import { appendFile, mkdir, writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +// Reviewer F4, and the plan's fourth open decision: a conversation held in +// Orca's own chat is the same file in the same place as one held in the +// terminal, so it must be searchable through the same path with no panel +// mounted, no scanner service running, and nobody calling refresh. Everything +// below is the library and the filesystem. + +const INTERVAL_MS = 20_000 +const SESSION_ID = 'cccccccc-dddd-4eee-8fff-000000000000' +const CWD = '/repo/orca' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-native-chat') +}) + +afterEach(async () => { + indexer.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +/** The rows Orca's native chat writes: uuid, block content, cwd on the first turn. */ +function nativeChatTurn(uuid: string, role: 'user' | 'assistant', text: string): string { + const timestamp = new Date(1_740_000_000_000 + Number(uuid.slice(-2)) * 60_000).toISOString() + return JSON.stringify({ + type: role, + uuid, + sessionId: SESSION_ID, + timestamp, + cwd: CWD, + gitBranch: 'main', + message: { + role, + ...(role === 'assistant' ? { model: 'claude-fable-5' } : {}), + content: [{ type: 'text', text }] + } + }) +} + +function messageTexts(term: string): { role: string; session: string }[] { + return harness.read( + (db: SyncDatabase) => + db + .prepare( + `SELECT m.role AS role, s.session_id AS session FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH ? ORDER BY m.id` + ) + .all(term) as { role: string; session: string }[] + ) +} + +it('indexes a native-chat conversation and its later turns with no panel and no service', async () => { + const path = join(harness.claudeProjectDir, `${SESSION_ID}.jsonl`) + await mkdir(harness.claudeProjectDir, { recursive: true }) + await writeFile( + path, + `${[ + nativeChatTurn('turn-01', 'user', 'why does the relay drop the lease at 105 seconds'), + nativeChatTurn('turn-02', 'assistant', 'that is the client silence watchdog, not a cliff') + ].join('\n')}\n` + ) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS + }) + await indexer.start() + + expect(messageTexts('watchdog')).toEqual([{ role: 'assistant', session: SESSION_ID }]) + expect(harness.read((db: SyncDatabase) => db.prepare('SELECT cwd FROM sessions').get())).toEqual({ + cwd: CWD + }) + + // The conversation continues in the panel; nothing tells the index about it. + await appendFile( + path, + `${[ + nativeChatTurn('turn-03', 'user', 'and the fleetwide 4408 bursts'), + nativeChatTurn('turn-04', 'assistant', 'those are desktop lease rotations, cohort waves') + ].join('\n')}\n` + ) + clock.advance(INTERVAL_MS) + await indexer.settled() + + expect(messageTexts('cohort')).toEqual([{ role: 'assistant', session: SESSION_ID }]) + expect(messageTexts('4408')).toEqual([{ role: 'user', session: SESSION_ID }]) + expect(indexer.status().phase).toBe('current') +}) diff --git a/src/main/ai-vault-search/session-search-opencode-decline.test.ts b/src/main/ai-vault-search/session-search-opencode-decline.test.ts new file mode 100644 index 00000000000..1ed3a201727 --- /dev/null +++ b/src/main/ai-vault-search/session-search-opencode-decline.test.ts @@ -0,0 +1,177 @@ +import { mkdirSync } from 'node:fs' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' + +// Only the thread hop is replaced: both implementations below are the repo's +// own in-process readers, which the worker entry calls on the other side. +export const openCodeParseCalls: string[] = [] +vi.mock('../ai-vault/session-scanner-opencode-sqlite-worker-spawn', async () => { + const list = await import('../ai-vault/session-scanner-opencode-sqlite-list') + const parse = await import('../ai-vault/session-scanner-opencode-sqlite') + const own = await import('./session-search-opencode-decline.test') + return { + resolveOpenCodeSqliteWorkerEntryPath: () => null, + listOpenCodeSqliteSessionsViaWorker: ( + args: Parameters[0] + ) => list.listOpenCodeSqliteSessions(args), + parseOpenCodeSqliteSessionViaWorker: ( + args: Parameters[0] + ) => { + own.openCodeParseCalls.push(args.sessionId) + return parse.parseOpenCodeSqliteSession(args) + } + } +}) +import Database from '../sqlite/sync-database' +import { getSessionParseCacheEntry } from '../ai-vault/session-parse-cache-store' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { buildOpenCodeSqliteCandidatePath } from '../ai-vault/session-scanner-opencode-sqlite-paths' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +/* + * Round 12, F3. An OpenCode SQLite session decodes where the message channel + * cannot reach it, so no read of one will ever commit a row. The consumer + * declined it and wrote nothing, which left the file table silent about a + * source discovery returns on every pass: the decide step saw a path the index + * held nothing for, asked for a read, and asking for one over a warm cache + * drops the session list's own resume point. Every OpenCode session was fully + * decoded on every pass and the sidebar's fold was thrown away with it, which + * is the cache STA-1278 and STA-1417 added. + */ + +const SESSION = 'ses_r12' +const CLAUDE_SESSION = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null = null + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-opencode-decline') + indexer = null + openCodeParseCalls.length = 0 +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function writeOpenCodeDb(path: string, sessionId: string): void { + const db = new Database(path) + db.exec(` + CREATE TABLE session ( + id TEXT PRIMARY KEY, project_id TEXT NOT NULL, parent_id TEXT, slug TEXT NOT NULL, + directory TEXT NOT NULL, title TEXT NOT NULL, version TEXT NOT NULL, share_url TEXT, + summary_additions INTEGER, summary_deletions INTEGER, summary_files INTEGER, + summary_diffs TEXT, revert TEXT, permission TEXT, + time_created INTEGER NOT NULL, time_updated INTEGER NOT NULL, time_compacting INTEGER, + time_archived INTEGER, workspace_id TEXT, path TEXT, agent TEXT, model TEXT, + cost REAL DEFAULT 0 NOT NULL, tokens_input INTEGER DEFAULT 0 NOT NULL, + tokens_output INTEGER DEFAULT 0 NOT NULL, tokens_reasoning INTEGER DEFAULT 0 NOT NULL, + tokens_cache_read INTEGER DEFAULT 0 NOT NULL, tokens_cache_write INTEGER DEFAULT 0 NOT NULL, + metadata TEXT + ); + CREATE TABLE message ( + id TEXT PRIMARY KEY, session_id TEXT NOT NULL, time_created INTEGER NOT NULL, + time_updated INTEGER NOT NULL, data TEXT NOT NULL + ); + CREATE TABLE project ( + id TEXT PRIMARY KEY, worktree TEXT NOT NULL, vcs TEXT, name TEXT, icon_url TEXT, + icon_color TEXT, time_created INTEGER NOT NULL, time_updated INTEGER NOT NULL, + time_initialized INTEGER, sandboxes TEXT NOT NULL, commands TEXT, icon_url_override TEXT + ); + CREATE TABLE part ( + id TEXT PRIMARY KEY, message_id TEXT NOT NULL, session_id TEXT NOT NULL, + time_created INTEGER NOT NULL, time_updated INTEGER NOT NULL, data TEXT NOT NULL + ); + `) + db.prepare( + `INSERT INTO session (id, project_id, parent_id, slug, directory, title, version, + time_created, time_updated, agent, model, cost, tokens_input, tokens_output, + tokens_reasoning, tokens_cache_read, tokens_cache_write) + VALUES (?, 'proj-1', NULL, 'slug-1', '/tmp/opencode', 'OpenCode title', '1.0.0', + ?, ?, 'build', '{"id":"glm"}', 0, 1, 1, 0, 0, 0)` + ).run(sessionId, 1_740_000_000_000, 1_740_000_100_000) + db.prepare( + `INSERT INTO message (id, session_id, time_created, time_updated, data) VALUES (?, ?, ?, ?, ?)` + ).run( + 'msg-1', + sessionId, + 1_740_000_000_000, + 1_740_000_000_000, + JSON.stringify({ role: 'user', time: { created: 1_740_000_000_000 } }) + ) + db.prepare( + `INSERT INTO part (id, message_id, session_id, time_created, time_updated, data) VALUES (?, ?, ?, ?, ?, ?)` + ).run( + 'part-1', + 'msg-1', + sessionId, + 1_740_000_000_000, + 1_740_000_000_000, + JSON.stringify({ type: 'text', text: 'hello opencode' }) + ) + db.prepare( + `INSERT INTO project (id, worktree, name, time_created, time_updated, sandboxes) + VALUES ('proj-1', '/tmp/opencode', 'proj', ?, ?, '[]')` + ).run(1_740_000_000_000, 1_740_000_000_000) + db.close() +} + +it('reads an OpenCode session once, not on every pass', async () => { + const dbPath = join(harness.root, 'opencode-db', 'opencode.db') + mkdirSync(join(harness.root, 'opencode-db'), { recursive: true }) + writeOpenCodeDb(dbPath, SESSION) + const claudePath = join(harness.claudeProjectDir, 'control.jsonl') + await writeClaudeTranscript(claudePath, ['control turn'], CLAUDE_SESSION) + + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: { ...harness.roots, opencodeDbPaths: [dbPath] }, + historyDays: null, + clock, + reconcileIntervalMs: 20_000, + onError: () => undefined + }) + await indexer.start() + + const syntheticPath = buildOpenCodeSqliteCandidatePath(dbPath, SESSION) + const openCodeAfterFirst = getSessionParseCacheEntry(syntheticPath) + const claudeAfterFirst = getSessionParseCacheEntry(claudePath) + + await indexer.reconcile() + await indexer.reconcile() + + // One decode across three passes, and the session list's cached fold for it + // is the same object it was after the first: nothing invalidated it. + expect(openCodeParseCalls).toHaveLength(1) + expect(getSessionParseCacheEntry(syntheticPath)).toBe(openCodeAfterFirst) + // The control, which the index really does hold, is untouched either way. + expect(getSessionParseCacheEntry(claudePath)).toBe(claudeAfterFirst) + + // What makes it skippable: a row saying the index has seen this source and + // holds no session for it, which is the shape a read-through-with-no-session + // already leaves. + const rows = harness.read((db) => + db.prepare('SELECT path, state, session_row_id FROM files ORDER BY path').all() + ) as { path: string; state: string; session_row_id: number | null }[] + expect(rows).toHaveLength(2) + expect(rows.find((row) => row.path === syntheticPath)).toMatchObject({ + state: 'current', + session_row_id: null + }) + expect(indexer.status()).toMatchObject({ filesDue: 0, filesFailed: 0, phase: 'current' }) +}) diff --git a/src/main/ai-vault-search/session-search-pass.ts b/src/main/ai-vault-search/session-search-pass.ts new file mode 100644 index 00000000000..d4f36015226 --- /dev/null +++ b/src/main/ai-vault-search/session-search-pass.ts @@ -0,0 +1,216 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import { ensureSessionParseCacheLoaded } from '../ai-vault/session-parse-cache-persistence' +import { + cursorChatMetaRefusals, + withCursorChatMetaScan +} from '../ai-vault/session-scanner-cursor-chat-meta' +import { recordSessionScanIssue } from '../ai-vault/session-scan-issues' +import { + mergeDegradedRoots, + scanIssueDegradedRoots, + unreadableRoots, + type SessionSearchDegradedRoot +} from './session-search-degraded-roots' +import { retireDeletedSessionSearchSources } from './session-search-deleted-sources' +import type { SessionSearchDirectoryReader } from './session-search-directory-listings' +import { runSessionSearchIndexPass } from './session-search-index-pass' +import { + discoverSessionSearchCandidates, + isUnderScanRoot, + sessionSearchEmptiedRoots, + sessionSearchRootListings, + type SessionSearchScanRoots +} from './session-search-scan-roots' +import type { SessionSearchFileRow, SessionSearchStore } from './session-search-store' +import { sessionSearchEnumeratedContainers } from './session-search-synthetic-sources' + +/** + * Rows a cycle proves present or gone, newest first. + * + * Why bounded and why newest first: a cycle lists the newest N per agent, so + * every older row it holds is undiscovered and would otherwise be walked every + * twenty seconds. Newest first is what makes the guarantee hold — a transcript + * recent enough for the window to cover is recent enough to be in this slice, + * so its deletion is proven on the very next cycle whenever it happened. + */ +const RETIREMENT_ROWS_PER_CYCLE = 512 + +/** + * Directories either pass may read proving deletions. + * + * The bound on the walk is readdirs, not rows: rows sharing a directory are one + * read and then map lookups, and a directory that answers an error answers it + * once for every row under it. Counting rows instead let one unreadable + * directory hold the whole walk for as long as it stayed unreadable. + */ +const RETIREMENT_DIRECTORIES_PER_PASS = 512 + +export type SessionSearchPassArgs = { + store: SessionSearchStore + roots: SessionSearchScanRoots + /** A sweep lists every root; a cycle lists the newest N per agent. */ + full: boolean + recentPerAgent: number + /** Real roots that listed transcripts on the previous pass; undefined before the first. */ + previousRootsWithFiles?: ReadonlySet + /** True once the pass is out of wall time; reads stop, everything else finishes. */ + overdue?: () => boolean + /** One readdir per directory for the whole pass, shared by every step. */ + listings: SessionSearchDirectoryReader + signal?: AbortSignal +} + +export type SessionSearchPassResult = { + /** Real roots this pass listed transcripts under, for the next pass to compare against. */ + rootsWithFiles: Set + degradedRoots: SessionSearchDegradedRoot[] + /** False when the pass was cut short; its conclusions are not to be recorded. */ + completed: boolean + /** + * True when the deadline stopped the reads with candidates still owed. + * + * The caller's one use for it: a cycle lists the newest N per agent, so a + * backlog outside that window is only *visible* to a sweep. Without this a + * first run would index the recency window in its opening pass and then crawl, + * making progress only on the periodic sweep every five minutes. + */ + outOfTime: boolean +} + +/** + * One pass. Four steps, the same four whether it sweeps or cycles. + * + * 1. **Discover.** The only filesystem walk: every root on a sweep, the newest + * N per agent on a cycle. Everything below is decided from what it returns. + * 2. **Decide and read.** Per candidate, its stat against its row. Reads stop + * at the deadline and nothing is recorded about what was left, because being + * owed is a fact about the row and not an entry in a queue. + * 3. **Retire.** Candidates are the rows this pass's discovery did not return, + * inside the scope that discovery covered. The stateless walk proves each + * one gone, present or unverifiable; only `gone` deletes. + * 4. **Report.** Root health for this pass. The counts are a query, made by the + * caller against the same rows, so nothing here is tallied. + * + * The pass keeps nothing. Everything it learns is either on a row or in the + * result the caller compares against the next pass. + */ +export async function runSessionSearchPass( + args: SessionSearchPassArgs +): Promise { + const { store, signal } = args + if (args.full) { + // Every sweep opens with the purge, so a window narrower than the last + // instance held is applied by the first sweep of this one. + await store.purgeOlderThan(store.retentionCutoff, signal) + } + await ensureSessionParseCacheLoaded() + return withCursorChatMetaScan(async () => { + const swept = await discoverSessionSearchCandidates(args.roots, { + limitPerAgent: args.full ? Number.POSITIVE_INFINITY : args.recentPerAgent, + signal + }) + const issues: AiVaultScanIssue[] = [...swept.issues] + + let completed = true + let outOfTime = false + const rows = new Map(store.files().map((row) => [row.path, row])) + try { + const read = await runSessionSearchIndexPass(store, swept.candidates, { + signal, + rows, + overdue: args.overdue + }) + outOfTime = read.outOfTime + } catch (error) { + if (!signal?.aborted) { + throw error + } + completed = false + } + + const listings = sessionSearchRootListings(args.roots, swept.discoveries) + const roots = listings.map((listing) => listing.root) + const rootsWithFiles = new Set( + listings.filter((listing) => listing.files > 0).map((listing) => listing.root) + ) + // Undefined, not empty, before any pass has recorded one: an empty set is a + // real observation and this is the absence of one. + const previousRootsWithFiles = args.previousRootsWithFiles + // A pass cut short saw part of the machine, so its silence about a path is + // not evidence; it retires nothing and publishes no verdicts. + const retirement = completed + ? await retireDeletedSessionSearchSources({ + store, + paths: retirementCandidates(rows, swept, roots, args.full), + roots, + // Only a sweep enumerates without a per-agent limit, so only a sweep + // may prove a synthetic row's container holds it no longer. + enumeratedContainers: args.full + ? sessionSearchEnumeratedContainers(swept.candidates, issues) + : undefined, + emptiedRoots: previousRootsWithFiles + ? sessionSearchEmptiedRoots(previousRootsWithFiles, rootsWithFiles) + : new Set(), + listings: args.listings, + directoryLimit: RETIREMENT_DIRECTORIES_PER_PASS, + signal + }) + : { retired: [], unverifiable: [], unchecked: [], degradedRoots: [] } + + for (const refusal of cursorChatMetaRefusals()) { + // One issue per refused chats root, not one per Cursor transcript. + recordSessionScanIssue(issues, { + agent: 'cursor', + path: refusal.chatsRoot, + message: refusal.message + }) + } + // Roots that listed no transcripts and cannot be listed either: the walker + // swallows a readdir failure, so this is the only place it surfaces. + const unlistable = completed + ? await unreadableRoots( + roots.filter((root) => !rootsWithFiles.has(root)), + args.listings, + signal + ) + : [] + + return { + rootsWithFiles, + degradedRoots: mergeDegradedRoots( + scanIssueDegradedRoots(roots, issues), + retirement.degradedRoots, + unlistable + ), + completed, + outOfTime + } + }) +} + +/** + * Rows this pass's discovery did not return, inside the scope it covered. + * + * A sweep covers everything, so every undiscovered row is a candidate. A cycle + * covers the newest N per agent, so it may only judge rows under a root it + * actually listed, and it takes the newest of those: an older row is not + * evidence of anything a cycle looked for, and the next sweep is what reaches + * it. This is the whole of what used to be a watch set carried between passes. + */ +function retirementCandidates( + rows: ReadonlyMap, + swept: { candidates: readonly { file: { path: string } }[] }, + roots: readonly string[], + full: boolean +): string[] { + const discovered = new Set(swept.candidates.map((candidate) => candidate.file.path)) + const undiscovered = [...rows.values()].filter((row) => !discovered.has(row.path)) + if (full) { + return undiscovered.map((row) => row.path) + } + return undiscovered + .filter((row) => roots.some((root) => isUnderScanRoot(row.path, root))) + .sort((left, right) => right.mtimeMs - left.mtimeMs) + .slice(0, RETIREMENT_ROWS_PER_CYCLE) + .map((row) => row.path) +} diff --git a/src/main/ai-vault-search/session-search-read-decision.test.ts b/src/main/ai-vault-search/session-search-read-decision.test.ts new file mode 100644 index 00000000000..64eb8866d02 --- /dev/null +++ b/src/main/ai-vault-search/session-search-read-decision.test.ts @@ -0,0 +1,117 @@ +import { expect, it } from 'vitest' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type { SessionSearchIndexedFile } from './session-search-file-cursor' +import { + SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT, + sessionSearchReadDecision +} from './session-search-read-decision' +import type { SessionSearchFileRow } from './session-search-store' + +const PATH = '/transcripts/one.jsonl' +const MTIME = 1_740_000_000_000 + +function candidate(overrides: Partial = {}): SessionFileCandidate { + return { + agent: 'claude', + codexHome: null, + file: { + path: PATH, + mtimeMs: MTIME, + modifiedAt: new Date(MTIME).toISOString(), + sizeBytes: 100, + ...overrides + } + } +} + +function row(overrides: Partial = {}): SessionSearchFileRow { + return { + path: PATH, + identity: null, + mtimeMs: MTIME, + sizeBytes: 100, + state: 'current', + failCount: 0, + failedMtimeMs: null, + ...overrides + } +} + +const cursor: SessionSearchIndexedFile = { byteOffset: 100, mtimeMs: MTIME, sizeBytes: 100 } + +function decide(args: { + file?: Partial + row?: SessionSearchFileRow | undefined + cursor?: SessionSearchIndexedFile | null + cutoffMs?: number | null +}) { + return sessionSearchReadDecision({ + candidate: candidate(args.file), + row: 'row' in args ? args.row : row(), + cursor: 'cursor' in args ? (args.cursor ?? null) : cursor, + cutoffMs: args.cutoffMs ?? null + }) +} + +it('reads a path the index holds nothing for, and lets the reader continue where it can', () => { + // Not `whole`: there is no span this index has to reach past, and the first + // enablement inside a running app has a warm list cursor to make use of. + expect(decide({ row: undefined })).toBe('any') +}) + +it('skips a file the index already covers at this stat', () => { + expect(decide({})).toBe('skip') +}) + +it('reads a file whose stat moved, however it moved', () => { + expect(decide({ file: { mtimeMs: MTIME + 1 } })).toBe('any') + // Grown without its mtime moving: a same-second append, or a restored stamp. + expect(decide({ file: { sizeBytes: 200 } })).toBe('any') +}) + +it('reads a file outside the retention window not at all', () => { + expect(decide({ row: undefined, cutoffMs: MTIME + 1 })).toBe('skip') + // And retention wins over everything else that would have asked for a read. + expect(decide({ row: row({ state: 'due' }), cutoffMs: MTIME + 1 })).toBe('skip') +}) + +it('reads a row owed a whole read from the start', () => { + expect(decide({ row: row({ state: 'due' }) })).toBe('whole') +}) + +it('reads whole rather than appending onto a cursor that continues nothing', () => { + // A different file at the same name: the identity check hands back no cursor. + expect(decide({ cursor: null })).toBe('whole') + // A chunked read that committed a prefix and no offset any append continues. + expect(decide({ cursor: { byteOffset: null, mtimeMs: MTIME, sizeBytes: 100 } })).toBe('whole') + // Shorter than the index read to, so this is not that file any more. + expect(decide({ file: { sizeBytes: 40 }, cursor })).toBe('whole') +}) + +it('retries a failed read until it has failed enough times at one stat', () => { + for (let failures = 1; failures < SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT; failures++) { + expect( + decide({ row: row({ state: 'failed', failCount: failures, failedMtimeMs: MTIME }) }) + ).toBe('any') + } + expect( + decide({ + row: row({ + state: 'failed', + failCount: SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT, + failedMtimeMs: MTIME + }) + }) + ).toBe('skip') +}) + +it('starts trying again the moment a held-out file changes', () => { + // The stat is the whole release condition, so nothing has to remember when + // the failures happened or schedule a retry. + expect( + decide({ + file: { mtimeMs: MTIME + 1 }, + row: row({ state: 'failed', failCount: 9, failedMtimeMs: MTIME }) + }) + ).toBe('any') +}) diff --git a/src/main/ai-vault-search/session-search-read-decision.ts b/src/main/ai-vault-search/session-search-read-decision.ts new file mode 100644 index 00000000000..cb25ad27ccb --- /dev/null +++ b/src/main/ai-vault-search/session-search-read-decision.ts @@ -0,0 +1,100 @@ +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type { SessionParseReadRequirement } from '../ai-vault/session-scanner-parse-cache' +import { requiresWholeRead, type SessionSearchIndexedFile } from './session-search-file-cursor' +import type { SessionSearchFileRow } from './session-search-store' + +/** + * Failures at one unchanged stat before a file is left alone. + * + * Three rather than one, because a single failure is often a transcript being + * rewritten under the read; three at the same mtime is not. The retry policy is + * the stat itself: an edit, a restore, or a `touch` after a `chmod` all move it, + * and nothing else does, so no timer is needed and none is kept. + */ +export const SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT = 3 + +/** + * What a pass owes one candidate: nothing, a read, or a read from the start. + * + * `any` and `whole` are the reader's own lanes. `whole` drops the session + * list's resume point, which is the only way to reach a span this index never + * saw; `any` asks for some bytes and lets the reader continue where it can, + * which is what the first enablement inside a running app needs — a warm list + * cursor sitting at the file's current stat would otherwise open nothing. + */ +export type SessionSearchReadDecision = 'skip' | SessionParseReadRequirement + +/** + * The whole of the indexer's decide step, as a function of the candidate's stat + * and the row the store holds for it. No pass state, no queue, no memory: the + * same inputs give the same answer on the first pass after a restart as on the + * hundredth of a long-running process, which is what lets a deadline cut a pass + * short with nothing to record. What did not get read is still owed, because + * being owed is a fact about the row. + */ +export function sessionSearchReadDecision(args: { + candidate: SessionFileCandidate + /** The file table's row, or undefined when the index holds nothing for it. */ + row: SessionSearchFileRow | undefined + /** The cursor for this candidate's identity; null when it is not continuable. */ + cursor: SessionSearchIndexedFile | null + /** Oldest transcript mtime worth holding rows for, or null for all history. */ + cutoffMs: number | null +}): SessionSearchReadDecision { + const { candidate, row, cursor, cutoffMs } = args + const file = candidate.file + // Retention first: a file outside the window is not worth reading whatever + // else is true of it, and the purge is what removes any row it still has. + if (cutoffMs !== null && file.mtimeMs < cutoffMs) { + return 'skip' + } + if (!row) { + // Nothing held for this path. Not `whole`, because the reader can continue + // from wherever it likes: there is no span this index has to reach past. + return 'any' + } + if (heldOut(row, file.mtimeMs)) { + return 'skip' + } + if (row.state === 'due') { + // The index is behind on a span no append reaches: a declined append, or a + // window that widened to admit this file. + return 'whole' + } + if (cursor === null || requiresWholeRead(cursor)) { + // A different file at the same name, or a chunked read that left a prefix + // and no cursor. Appending onto either would splice two spans together. + return 'whole' + } + const size = file.sizeBytes + if (typeof size === 'number' && cursor.byteOffset !== null && cursor.byteOffset > size) { + // Shorter than the index read to: this is not the file that cursor came from. + return 'whole' + } + if (row.state === 'failed') { + // Still within its retries, or the stat moved since it last failed. + return 'any' + } + return statMatches(row, file) ? 'skip' : 'any' +} + +/** + * True when this file has failed enough times at exactly this stat to stop + * trying. The stat is the whole release condition, so a file nobody touches is + * never read again and one that changes is read on the next pass that sees it. + */ +function heldOut(row: SessionSearchFileRow, mtimeMs: number): boolean { + return ( + row.state === 'failed' && + row.failCount >= SESSION_SEARCH_FAILURES_BEFORE_HELD_OUT && + row.failedMtimeMs === mtimeMs + ) +} + +/** The row already describes the file as it is now. */ +function statMatches(row: SessionSearchFileRow, file: SessionFileCandidate['file']): boolean { + return ( + row.mtimeMs === file.mtimeMs && + (row.sizeBytes === null || file.sizeBytes === undefined || row.sizeBytes === file.sizeBytes) + ) +} diff --git a/src/main/ai-vault-search/session-search-retention-delete.test.ts b/src/main/ai-vault-search/session-search-retention-delete.test.ts new file mode 100644 index 00000000000..c21c8d7fe2b --- /dev/null +++ b/src/main/ai-vault-search/session-search-retention-delete.test.ts @@ -0,0 +1,188 @@ +import { expect, it } from 'vitest' +import type SyncDatabase from '../sqlite/sync-database' +import { + deleteExpiredSearchFiles, + RETENTION_DELETE_ROWS_PER_STEP +} from './session-search-retention-delete' +import { openSessionSearchIndexFile } from './session-search-index-test-fixture' +import { SessionSearchStore } from './session-search-store' + +function seed(db: SyncDatabase, id: number, rows: number, mtime: number): void { + db.prepare( + `INSERT INTO sessions(id,agent,session_id,file_path,title,cwd,cwd_key,resume_command) + VALUES (?, 'claude', ?, ?, 'synthetic retention', '/fixture', '/fixture', '')` + ).run(id, String(id), String(id)) + db.prepare('INSERT INTO files(path,byte_offset,mtime_ms,session_row_id) VALUES (?,1,?,?)').run( + String(id), + mtime, + id + ) + db.exec('BEGIN') + for (let i = 0; i < rows; i++) { + const row = db + .prepare("INSERT INTO messages(session_row_id,role) VALUES (?,'user')") + .run(id).lastInsertRowid + db.prepare('INSERT INTO messages_fts(rowid,user_text) VALUES (?,?)').run(row, 'retentionneedle') + } + db.exec('COMMIT') +} + +/** + * Sessions a search would still return. Every retrieval joins a message to its + * session, which is what makes cutting the session loose enough to hide the + * whole thing while its rows are still being reclaimed. + */ +function visibleSessionIds(db: SyncDatabase): string[] { + return ( + db + .prepare( + `SELECT DISTINCT s.session_id AS id FROM messages_fts + JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id + WHERE messages_fts MATCH 'retentionneedle' ORDER BY s.session_id` + ) + .all() as { id: string }[] + ).map((row) => row.id) +} + +function count(db: SyncDatabase, table: string): number { + return (db.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { n: number }).n +} + +it('seeks the expiring end of the file list instead of scanning it', async () => { + const index = await openSessionSearchIndexFile('ss-retention-plan') + try { + seed(index.db, 1, 1, 1) + const plan = ( + index.db + .prepare('EXPLAIN QUERY PLAN SELECT path FROM files WHERE mtime_ms < ? ORDER BY mtime_ms') + .all(100) as { detail: string }[] + ) + .map((row) => row.detail) + .join(' ') + // Without files_mtime this is "SCAN files" plus a "USE TEMP B-TREE FOR ORDER BY". + expect(plan).toContain('files_mtime') + expect(plan).not.toContain('TEMP B-TREE') + } finally { + await index.close() + } +}) + +it('hides an expiring session at once, then reclaims its rows in bounded steps', async () => { + const index = await openSessionSearchIndexFile('ss-retention-yield') + seed(index.db, 1, 1025, 1) + seed(index.db, 2, 1, 200) + let previous = 1025 + const steps: number[] = [] + try { + await deleteExpiredSearchFiles( + index.db, + 100, + () => false, + async () => { + const left = count(index.db, 'messages WHERE session_row_id=1') + steps.push(previous - left) + previous = left + // Cut loose in the very first transaction, so no query ever sees it with + // some of its messages already gone. + expect(visibleSessionIds(index.db)).toEqual(['2']) + } + ) + // The file transaction, then one bounded batch per step until the rows are gone. + expect(steps).toEqual([0, RETENTION_DELETE_ROWS_PER_STEP, 256, 256, 256, 1]) + expect(count(index.db, 'messages_fts')).toBe(1) + expect(count(index.db, 'sessions')).toBe(1) + } finally { + await index.close() + } +}) + +it('finishes an interrupted deletion after reopening', async () => { + const index = await openSessionSearchIndexFile('ss-retention-resume') + let store = new SessionSearchStore(index.path) + let closed = false + let steps = 0 + try { + seed(index.db, 1, 513, 1) + await deleteExpiredSearchFiles( + index.db, + 100, + () => closed, + async () => { + if (++steps === 2) { + store.close() + closed = true + } + } + ) + // Some rows went, the rest did not, and nothing recorded that anywhere. + const stranded = count(index.db, 'messages') + expect(stranded).toBeGreaterThan(0) + expect(stranded).toBeLessThan(513) + expect(visibleSessionIds(index.db)).toEqual([]) + + store = new SessionSearchStore(index.path) + closed = false + // Rows nothing points at are the whole record of unfinished work, so the + // rest goes even with retention now unlimited. + await store.purgeOlderThan(null) + expect(count(index.db, 'messages')).toBe(0) + expect(count(index.db, 'messages_fts')).toBe(0) + } finally { + if (!closed) { + store.close() + } + await index.close() + } +}) + +it('cancels retention between batches and resumes without exposing a partial session', async () => { + const index = await openSessionSearchIndexFile('ss-retention-cancel') + const store = new SessionSearchStore(index.path) + try { + seed(index.db, 1, 1025, 1) + const controller = new AbortController() + const purge = store.purgeOlderThan(100, controller.signal) + setImmediate(() => controller.abort()) + await purge + const remaining = count(index.db, 'messages') + expect(remaining).toBeGreaterThan(0) + expect(remaining).toBeLessThan(1025) + expect(visibleSessionIds(index.db)).toEqual([]) + await store.purgeOlderThan(null) + expect(count(index.db, 'messages')).toBe(0) + } finally { + store.close() + await index.close() + } +}) + +it('keeps a file a read refreshed after the expiry list was taken', async () => { + const index = await openSessionSearchIndexFile('ss-retention-refreshed') + try { + seed(index.db, 1, 2, 1) + seed(index.db, 2, 2, 2) + let refreshed = false + // The scan of `files` happens once, up front. A read of the second transcript + // lands while the first is being deleted, which makes it new enough to keep. + await deleteExpiredSearchFiles( + index.db, + 100, + () => false, + async () => { + if (!refreshed) { + refreshed = true + index.db.prepare('UPDATE files SET mtime_ms = 500 WHERE path = ?').run('2') + } + } + ) + + // Only the per-file transaction re-reading the mtime it is about to act on + // keeps that session; the list it came from says both should go. + expect(count(index.db, 'files')).toBe(1) + expect(visibleSessionIds(index.db)).toEqual(['2']) + expect(count(index.db, 'messages')).toBe(2) + } finally { + await index.close() + } +}) diff --git a/src/main/ai-vault-search/session-search-retention-delete.ts b/src/main/ai-vault-search/session-search-retention-delete.ts new file mode 100644 index 00000000000..e8d407f8be9 --- /dev/null +++ b/src/main/ai-vault-search/session-search-retention-delete.ts @@ -0,0 +1,102 @@ +import { setImmediate as yieldToEventLoop } from 'node:timers/promises' +import type SyncDatabase from '../sqlite/sync-database' +import { deleteSearchMessages } from './session-search-message-rows' + +export const RETENTION_DELETE_ROWS_PER_STEP = 256 +// Why in step with the deletes rather than one sweep at the end: `auto_vacuum = +// INCREMENTAL` holds every freed page until something asks for it back, and +// asking for a whole purge's worth at once is one long stall (40 ms per 22 MB +// freed, measured) instead of many short ones. +const RECLAIM_PAGES_PER_STEP = 2000 + +/** + * Drops every file older than the cutoff, then hands its rows back in bounded + * steps. + * + * The two halves are separate on purpose. Cutting a session loose from its file + * is one small transaction, and it is what makes the session stop answering + * searches — every read joins `sessions`, so a row whose session is gone is + * already unreachable. Reclaiming those rows is the expensive half, and it can + * be paused, interrupted or resumed at any point without a reader ever seeing a + * session that is half deleted. A crash in the middle leaves rows nothing + * points at, and `drainOrphanedMessages` finds them on the next pass. + */ +export async function deleteExpiredSearchFiles( + db: SyncDatabase, + cutoffMs: number | null, + closed: () => boolean, + yieldStep: () => Promise = yieldToEventLoop +): Promise { + if (cutoffMs !== null) { + const expired = db + .prepare('SELECT path FROM files WHERE mtime_ms < ? ORDER BY mtime_ms') + .all(cutoffMs) as { path: string }[] + for (const { path } of expired) { + if (closed()) { + return + } + db.exec('BEGIN IMMEDIATE') + try { + // Re-read under the lock: a read of this file may have landed since the + // list was taken, which makes it new enough to keep. + const file = db + .prepare('SELECT session_row_id FROM files WHERE path = ? AND mtime_ms < ?') + .get(path, cutoffMs) as { session_row_id: number | null } | undefined + if (file) { + db.prepare('DELETE FROM sessions WHERE id = ?').run(file.session_row_id) + db.prepare('DELETE FROM files WHERE path = ?').run(path) + } + db.exec('COMMIT') + } catch (error) { + db.exec('ROLLBACK') + throw error + } + await yieldStep() + } + } + await drainOrphanedMessages(db, closed, yieldStep) +} + +/** + * Deletes rows whose session no longer exists, a bounded batch per transaction. + * + * That set is exactly what retention, a replace that cut its old generation + * loose, a removed source and an interrupted earlier drain leave behind, so the + * index needs no record of unfinished work beyond the rows themselves. + * + * Exported for the store, which runs it after a replace commits for the same + * reason retention runs it after its own small transaction: cutting a session + * loose is what hides it, and reclaiming its rows is the half that must not + * hold one transaction. + */ +export async function drainOrphanedMessages( + db: SyncDatabase, + closed: () => boolean, + yieldStep: () => Promise = yieldToEventLoop +): Promise { + // Ordered by session so one call to this walks a session's rows to the end + // before paying for the scan that finds the next one. + const nextOrphan = db.prepare( + `SELECT session_row_id FROM messages + WHERE session_row_id NOT IN (SELECT id FROM sessions) LIMIT 1` + ) + let orphan = (nextOrphan.get() as { session_row_id: number } | undefined)?.session_row_id + while (orphan !== undefined && !closed()) { + db.exec('BEGIN IMMEDIATE') + let deleted = 0 + try { + deleted = deleteSearchMessages(db, orphan, RETENTION_DELETE_ROWS_PER_STEP) + db.exec('COMMIT') + } catch (error) { + db.exec('ROLLBACK') + throw error + } + db.pragma(`incremental_vacuum(${RECLAIM_PAGES_PER_STEP})`) + if (deleted < RETENTION_DELETE_ROWS_PER_STEP) { + orphan = (nextOrphan.get() as { session_row_id: number } | undefined)?.session_row_id + } + await yieldStep() + } + // A `removeFile` frees its pages outside this loop and may leave none to drain. + db.pragma(`incremental_vacuum(${RECLAIM_PAGES_PER_STEP})`) +} diff --git a/src/main/ai-vault-search/session-search-retention-policy.test.ts b/src/main/ai-vault-search/session-search-retention-policy.test.ts new file mode 100644 index 00000000000..71c6cdb0862 --- /dev/null +++ b/src/main/ai-vault-search/session-search-retention-policy.test.ts @@ -0,0 +1,26 @@ +import { expect, it } from 'vitest' +import { sessionSearchHistoryCutoffMs } from './session-search-retention-policy' + +const NOW = 1_740_000_000_000 + +it('treats a fractional or non-positive day count as no bound at all', () => { + // A day count that floors to zero would read as "all history" in one place + // and "cutoff is now" in the other; both sides answer null. + expect(sessionSearchHistoryCutoffMs(0.4, NOW)).toBeNull() + expect(sessionSearchHistoryCutoffMs(0, NOW)).toBeNull() + expect(sessionSearchHistoryCutoffMs(-30, NOW)).toBeNull() + expect(sessionSearchHistoryCutoffMs(30, NOW)).toBe(NOW - 30 * 86_400_000) + // Clamped rather than unbounded: a caller asking for three thousand years of + // history gets the ceiling, not an mtime before the epoch. + expect(sessionSearchHistoryCutoffMs(999_999, NOW)).toBe(NOW - 3_650 * 86_400_000) +}) + +// The cutoff is read from the clock on every pass, not frozen at construction: +// a purge and the accept check that follows it must not disagree about where +// the window is, or the sweep deletes rows the next candidate re-indexes. +it('moves the cutoff with the clock', () => { + const later = NOW + 86_400_000 + expect(sessionSearchHistoryCutoffMs(30, later)).toBe( + (sessionSearchHistoryCutoffMs(30, NOW) ?? 0) + 86_400_000 + ) +}) diff --git a/src/main/ai-vault-search/session-search-retention-policy.ts b/src/main/ai-vault-search/session-search-retention-policy.ts new file mode 100644 index 00000000000..c7fa8a0b6d1 --- /dev/null +++ b/src/main/ai-vault-search/session-search-retention-policy.ts @@ -0,0 +1,25 @@ +const DAY_MS = 86_400_000 +const HISTORY_DAYS_MAX = 3_650 + +/** + * The retention window, as the indexer's callers state it and as the store + * consumes it. Settings storage is PR 3b's problem; this is the arithmetic. + */ +function normalizeSessionSearchHistoryDays(value: number | null): number | null { + if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) { + return null + } + // Why floor then re-check: a fractional day floors to 0, which reads as "all + // history" on one side and "now" on the other; make the two agree. + const days = Math.floor(value) + return days <= 0 ? null : Math.min(HISTORY_DAYS_MAX, days) +} + +/** The oldest transcript mtime worth indexing; null means no bound. */ +export function sessionSearchHistoryCutoffMs( + historyDays: number | null, + nowMs: number +): number | null { + const days = normalizeSessionSearchHistoryDays(historyDays) + return days === null ? null : nowMs - days * DAY_MS +} diff --git a/src/main/ai-vault-search/session-search-row-identity.test.ts b/src/main/ai-vault-search/session-search-row-identity.test.ts new file mode 100644 index 00000000000..4adc655b584 --- /dev/null +++ b/src/main/ai-vault-search/session-search-row-identity.test.ts @@ -0,0 +1,116 @@ +import { afterEach, beforeEach, expect, it } from 'vitest' +import type SyncDatabase from '../sqlite/sync-database' +import { deleteExpiredSearchFiles } from './session-search-retention-delete' +import { + openSessionSearchIndexFile, + syntheticCandidate, + syntheticSession, + userMessages, + type SessionSearchIndexFile +} from './session-search-index-test-fixture' +import { SessionSearchStore } from './session-search-store' + +// A session row id outlives the row: it names the rows in `messages` until a +// retention drain has walked all of them, which takes many transactions. These +// tests are about what may be handed that id in the meantime. + +let index: SessionSearchIndexFile +let store: SessionSearchStore +let errors: unknown[] + +beforeEach(async () => { + index = await openSessionSearchIndexFile('ss-row-identity') + errors = [] + store = new SessionSearchStore(index.path, (error) => errors.push(error)) +}) + +afterEach(async () => { + store.close() + await index.close() +}) + +const OLD_MTIME = 1_000 +const LIVE_MTIME = 1_000_000 +const LIVE_PATH = '/live.jsonl' + +function count(db: SyncDatabase, table: string): number { + return (db.prepare(`SELECT count(*) AS n FROM ${table}`).get() as { n: number }).n +} + +/** Rows a search would return for a term: the join every retrieval makes. */ +function matches(db: SyncDatabase, term: string): number { + return ( + db + .prepare( + `SELECT count(*) AS n FROM messages_fts JOIN messages m ON m.id = messages_fts.rowid + JOIN sessions s ON s.id = m.session_row_id WHERE messages_fts MATCH ?` + ) + .get(term) as { n: number } + ).n +} + +function indexFile(path: string, mtimeMs: number, text: string, rows: number): void { + const write = store.beginWrite(syntheticCandidate({ path, mtimeMs }), 'replace', 0)! + for (const message of userMessages(text, rows)) { + write.add(message) + } + expect(write.commit({ session: syntheticSession(), byteOffset: 50, incomplete: false })).toBe( + true + ) +} + +it('never hands a live session the rows of a purged one', async () => { + // Two expiring transcripts, each large enough that reclaiming their rows takes + // several transactions, and one live transcript the parser decoded no session + // from — so it holds a cursor and no session row of its own. + indexFile('/old-a.jsonl', OLD_MTIME, 'purgedneedle', 400) + indexFile('/old-b.jsonl', OLD_MTIME, 'purgedneedle', 400) + const live = syntheticCandidate({ path: LIVE_PATH, mtimeMs: LIVE_MTIME }) + const opening = store.beginWrite(live, 'replace', 0)! + opening.add(userMessages('excluded', 1)[0]!) + expect(opening.commit({ session: null, byteOffset: 50, incomplete: false })).toBe(true) + + let appended = false + await deleteExpiredSearchFiles( + index.db, + LIVE_MTIME, + () => false, + async () => { + // The window: both expiring sessions are cut loose, most of their rows are + // still on disk, and the live transcript grows. The append is legitimate — + // it continues this index's own cursor — and it needs a session row. + if (appended || count(index.db, 'sessions') > 0) { + return + } + appended = true + const write = store.beginWrite(live, 'append', 50)! + for (const message of userMessages('liveneedle', 2)) { + write.add(message) + } + expect( + write.commit({ session: syntheticSession(), byteOffset: 120, incomplete: false }) + ).toBe(true) + } + ) + + expect(appended).toBe(true) + // Reusing a freed id would adopt whatever of that session's rows the drain had + // not reached, and put them behind a live session no purge will visit again. + expect(matches(index.db, 'purgedneedle')).toBe(0) + expect(matches(index.db, 'liveneedle')).toBe(2) + expect(count(index.db, 'messages')).toBe(2) + expect(errors).toEqual([]) +}) + +it('never reissues a session row id a delete freed', () => { + for (const path of ['/a.jsonl', '/b.jsonl', '/c.jsonl']) { + indexFile(path, OLD_MTIME, 'seeded', 1) + } + const before = (index.db.prepare('SELECT max(id) AS id FROM sessions').get() as { id: number }).id + index.db.exec('DELETE FROM sessions') + + indexFile('/d.jsonl', OLD_MTIME, 'seeded', 1) + expect((index.db.prepare('SELECT id FROM sessions').get() as { id: number }).id).toBeGreaterThan( + before + ) +}) diff --git a/src/main/ai-vault-search/session-search-scan-roots.test.ts b/src/main/ai-vault-search/session-search-scan-roots.test.ts new file mode 100644 index 00000000000..51b7381c78b --- /dev/null +++ b/src/main/ai-vault-search/session-search-scan-roots.test.ts @@ -0,0 +1,58 @@ +import { expect, it } from 'vitest' +import { delimiter, join } from 'node:path' +import type { SessionFileDiscovery } from '../ai-vault/session-scanner-types' +import { sessionSearchRootListings } from './session-search-scan-roots' + +const STATE = '/tmp/ss-roots/openclaw-state' +const LEGACY = '/tmp/ss-roots/openclaw-legacy' + +function file(path: string): SessionFileDiscovery['files'][number] { + return { path, mtimeMs: 0, modifiedAt: new Date(0).toISOString() } +} + +it('splits a merged discovery into the real directories behind it', () => { + const current = join(STATE, 'agents') + const legacy = join(LEGACY, 'agents') + const listings = sessionSearchRootListings( + { openclawStateDir: STATE, openclawLegacyStateDir: LEGACY }, + [ + { + agent: 'openclaw', + // What discovery reports for an agent whose roots are alternates. + rootDir: [current, legacy].join(delimiter), + files: [ + file(join(current, 'a', 'sessions', 'one.jsonl')), + file(join(current, 'a', 'sessions', 'two.jsonl')), + file(join(legacy, 'b', 'sessions', 'three.jsonl')) + ] + } + ] + ) + + const byRoot = Object.fromEntries(listings.map((one) => [one.root, one.files])) + expect(byRoot[current]).toBe(2) + expect(byRoot[legacy]).toBe(1) + // The joined string is never reported as a directory. + expect(listings.every((one) => !one.root.includes(delimiter))).toBe(true) +}) + +it('attributes a file by path segment, not by string prefix', () => { + const agents = join(STATE, 'agents') + const legacy = join(LEGACY, 'agents') + const listings = sessionSearchRootListings( + { openclawStateDir: STATE, openclawLegacyStateDir: LEGACY }, + [ + { + agent: 'openclaw', + rootDir: [agents, legacy].join(delimiter), + // A sibling directory whose name merely starts with a root's name. It + // is under no root, so it belongs to none of them. + files: [file(join(`${agents}-old`, 'b', 'sessions', 'two.jsonl'))] + } + ] + ) + + const byRoot = Object.fromEntries(listings.map((one) => [one.root, one.files])) + expect(byRoot[agents]).toBe(0) + expect(byRoot[legacy]).toBe(0) +}) diff --git a/src/main/ai-vault-search/session-search-scan-roots.ts b/src/main/ai-vault-search/session-search-scan-roots.ts new file mode 100644 index 00000000000..8df3510199e --- /dev/null +++ b/src/main/ai-vault-search/session-search-scan-roots.ts @@ -0,0 +1,135 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import { AI_VAULT_AGENT_SOURCES } from '../ai-vault/session-scanner-agent-sources' +import { normalizedWslHomeDirs } from '../ai-vault/session-scanner-roots' +import { sessionCandidatesFromDiscoveries } from '../ai-vault/session-scanner-candidates' +import { discoverAiVaultSessionSources } from '../ai-vault/session-scanner-source-discovery' +import type { + AiVaultScanOptions, + SessionFileCandidate, + SessionFileDiscovery +} from '../ai-vault/session-scanner-types' + +/** One real directory a scan walked, and what it listed there. */ +export type SessionSearchRootListing = { root: string; files: number } + +/** + * Where the indexer looks. The caller resolves these so the index enumerates + * exactly the trees the session list does; the indexer owns the bounds + * (`limit`, `limitPerAgent`, `unlimited`) and its own cancellation, so those + * are not the caller's to set. + */ +export type SessionSearchScanRoots = Omit< + AiVaultScanOptions, + 'signal' | 'limit' | 'unlimited' | 'limitPerAgent' | 'scopePaths' +> + +export type SessionSearchDiscovery = { + /** Newest first, Codex hardlink aliases collapsed, exactly as a list scan sees them. */ + candidates: SessionFileCandidate[] + discoveries: SessionFileDiscovery[] + issues: AiVaultScanIssue[] +} + +/** + * The discovery half of a list scan, without the parse. `limitPerAgent` is the + * sidebar's own recency rule (`SessionNewestFiles` keeps the newest N per root); + * passing Infinity is what makes a sweep whole. + */ +export async function discoverSessionSearchCandidates( + roots: SessionSearchScanRoots, + args: { limitPerAgent: number; signal?: AbortSignal } +): Promise { + const issues: AiVaultScanIssue[] = [] + const options: AiVaultScanOptions = { ...roots, signal: args.signal } + const discoveries = await discoverAiVaultSessionSources({ + options, + limitPerAgent: args.limitPerAgent, + issues + }) + const candidates = await sessionCandidatesFromDiscoveries(discoveries, options) + return { candidates, discoveries, issues } +} + +/** + * Containment on path segments, not on string prefix, and on both separators: + * discovery joins with the platform's, a configured root can arrive spelled + * with the other, and `/a/agents-old` is not inside `/a/agents`. + */ +export function isUnderScanRoot(path: string, root: string): boolean { + return root.length > 0 && (path.startsWith(`${root}/`) || path.startsWith(`${root}\\`)) +} + +/** + * The real directories behind a scan's discoveries, with their file counts. + * + * Why this exists: an agent whose roots are alternates for one install reports + * them as a single discovery whose `rootDir` is every path joined by the + * platform's path delimiter. That string is not a directory. Health probes + * readdir it and get ENOENT, a containment check never matches a file under it, + * and a scan issue recorded against a real root never equals it — so the fence + * meant to protect an unmounted tree is inert for exactly the agent most likely + * to have one. Splitting the joined string back apart would be worse: a + * directory may legally contain the delimiter. The constituent paths come from + * the same source table discovery read. + */ +export function sessionSearchRootListings( + roots: SessionSearchScanRoots, + discoveries: readonly SessionFileDiscovery[] +): SessionSearchRootListing[] { + const wslHomeDirs = normalizedWslHomeDirs(roots.wslHomeDirs) + const counts = new Map() + for (const discovery of discoveries) { + const constituents = constituentRoots(roots, wslHomeDirs, discovery) + for (const root of constituents) { + counts.set(root, counts.get(root) ?? 0) + } + for (const file of discovery.files) { + const owner = owningRoot(constituents, file.path) + if (owner !== null) { + counts.set(owner, (counts.get(owner) ?? 0) + 1) + } + } + } + return [...counts].map(([root, files]) => ({ root, files })) +} + +function constituentRoots( + roots: SessionSearchScanRoots, + wslHomeDirs: readonly string[], + discovery: SessionFileDiscovery +): string[] { + const declared = AI_VAULT_AGENT_SOURCES[discovery.agent]?.rootDirs(roots, wslHomeDirs) ?? [] + if (declared.includes(discovery.rootDir)) { + return [discovery.rootDir] + } + // Either a merged discovery, whose rootDir is the joined string, or a source + // that builds its own discoveries (OpenCode, Antigravity) and reports a real + // directory that this table does not list. + return declared.length > 0 ? declared : [discovery.rootDir] +} + +function owningRoot(constituents: readonly string[], path: string): string | null { + let owner: string | null = null + for (const root of constituents) { + if (isUnderScanRoot(path, root) && (owner === null || root.length > owner.length)) { + owner = root + } + } + return owner +} + +/** + * Roots that listed transcripts on the previous pass and list none on this one. + * + * The one bit of memory the retirement walk gets, and what it buys: a root that + * blinks empty for a single pass is unverifiable rather than proven gone, so a + * sync client swapping a directory out cannot retire a tree. It is deliberately + * not evidence that survives the process — see the invariant block in + * `session-search-deleted-sources.ts` for what that costs and why. + */ +export function sessionSearchEmptiedRoots( + previous: ReadonlySet, + current: ReadonlySet +): Set { + return new Set([...previous].filter((root) => !current.has(root))) +} diff --git a/src/main/ai-vault-search/session-search-schema.test.ts b/src/main/ai-vault-search/session-search-schema.test.ts new file mode 100644 index 00000000000..b23f36cde55 --- /dev/null +++ b/src/main/ai-vault-search/session-search-schema.test.ts @@ -0,0 +1,350 @@ +import type * as NodeFs from 'node:fs' +import { mkdtemp, readFile, stat, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + removeTree, + WINDOWS_RM_MAX_RETRIES, + WINDOWS_RM_RETRY_DELAY_MS +} from '../../shared/windows-transient-lock-removal' +import SyncDatabase from '../sqlite/sync-database' +import { + SESSION_SEARCH_SCHEMA_VERSION, + openSessionSearchDatabase, + removeSessionSearchDatabase +} from './session-search-schema' + +const recordedRmSync = vi.hoisted(() => vi.fn()) +vi.mock('node:fs', async () => { + const actual = await vi.importActual('node:fs') + return { + ...actual, + rmSync: (...args: Parameters) => { + recordedRmSync(...args) + return actual.rmSync(...args) + } + } +}) + +let roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.map((root) => removeTree(root))) + roots = [] +}) + +async function tempDatabasePath(): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-session-search-schema-')) + roots.push(root) + return join(root, 'index.sqlite') +} + +function schemaVersion(db: SyncDatabase): string | undefined { + return ( + db.prepare("SELECT value FROM meta WHERE key = 'schema_version'").get() as + | { value: string } + | undefined + )?.value +} + +describe('openSessionSearchDatabase', () => { + it('keeps a current-version index and its rows', async () => { + const path = await tempDatabasePath() + const first = openSessionSearchDatabase(path) + first.prepare("INSERT INTO files(path,byte_offset,mtime_ms) VALUES ('a',1,1)").run() + first.close() + + const second = openSessionSearchDatabase(path) + expect(schemaVersion(second)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + expect(second.prepare('SELECT COUNT(*) AS c FROM files').get()).toEqual({ + c: 1 + }) + second.close() + }) + + it('carries one FTS table and throws away an index that carries two', async () => { + const path = await tempDatabasePath() + const fresh = openSessionSearchDatabase(path) + const tables = (): string[] => + ( + fresh + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name LIKE '%_fts'") + .all() as { name: string }[] + ).map((row) => row.name) + expect(tables()).toEqual(['messages_fts']) + + // What an index written before this bump looks like: the second table, and + // rows in it. `CREATE TABLE IF NOT EXISTS` would leave both in place, so + // only the version bump makes that file go. + fresh.exec('CREATE VIRTUAL TABLE conversation_fts USING fts5(user_text, assistant_text)') + fresh.prepare("INSERT INTO files(path,byte_offset,mtime_ms) VALUES ('a',1,1)").run() + fresh.prepare("UPDATE meta SET value = '3' WHERE key = 'schema_version'").run() + fresh.close() + + const rebuilt = openSessionSearchDatabase(path) + expect(schemaVersion(rebuilt)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + expect( + rebuilt + .prepare("SELECT count(*) AS n FROM sqlite_master WHERE name = 'conversation_fts'") + .get() + ).toEqual({ n: 0 }) + expect(rebuilt.prepare('SELECT COUNT(*) AS c FROM files').get()).toEqual({ c: 0 }) + rebuilt.close() + }) + + it('replaces the file on a version mismatch instead of dropping tables in place', async () => { + const path = await tempDatabasePath() + const stale = openSessionSearchDatabase(path) + stale.prepare("INSERT INTO files(path,byte_offset,mtime_ms) VALUES ('a',1,1)").run() + stale + .prepare("UPDATE meta SET value = ? WHERE key = 'schema_version'") + .run(String(SESSION_SEARCH_SCHEMA_VERSION + 1)) + stale.close() + // Why: a stale sidecar must go with the main file, or SQLite replays it into the new one. + await writeFile(`${path}-wal`, 'stale wal bytes') + const before = await stat(path) + + const fresh = openSessionSearchDatabase(path) + expect(schemaVersion(fresh)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + expect(fresh.prepare('SELECT COUNT(*) AS c FROM files').get()).toEqual({ + c: 0 + }) + fresh.close() + // Why not inode: ext4 hands a freed inode straight back to the next create. + // The planted sidecar is gone (a fresh WAL is checkpointed away on close). + await expect(stat(`${path}-wal`)).rejects.toMatchObject({ code: 'ENOENT' }) + expect((await stat(path)).mtimeMs).toBeGreaterThanOrEqual(before.mtimeMs) + }) + + it('removes the database with every sidecar', async () => { + const path = await tempDatabasePath() + openSessionSearchDatabase(path).close() + await writeFile(`${path}-shm`, '') + removeSessionSearchDatabase(path) + for (const suffix of ['', '-wal', '-shm']) { + await expect(stat(`${path}${suffix}`)).rejects.toMatchObject({ + code: 'ENOENT' + }) + } + }) +}) + +it('rebuilds a file too corrupt to open instead of refusing forever', async () => { + const path = await tempDatabasePath() + const healthy = openSessionSearchDatabase(path) + healthy.prepare("INSERT INTO files(path,byte_offset,mtime_ms) VALUES ('a',1,1)").run() + healthy.close() + // A torn page, not a truncation: SQLite opens the header and fails on the read. + const bytes = await readFile(path) + bytes.fill(0x7f, 4096, Math.min(bytes.length, 12_288)) + await writeFile(path, bytes) + + const rebuilt = openSessionSearchDatabase(path) + try { + expect(schemaVersion(rebuilt)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + expect(rebuilt.prepare('SELECT COUNT(*) AS c FROM files').get()).toEqual({ + c: 0 + }) + } finally { + rebuilt.close() + } +}) + +it('rebuilds a file that is not a database at all', async () => { + const path = await tempDatabasePath() + await writeFile(path, 'not a SQLite database') + + const rebuilt = openSessionSearchDatabase(path) + try { + expect(schemaVersion(rebuilt)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + } finally { + rebuilt.close() + } +}) + +it('gives up rather than looping when a fresh file still cannot be opened', async () => { + const path = await tempDatabasePath() + await writeFile(path, 'not a SQLite database') + // Every open of this path fails, so the one permitted retry is exhausted. + const open = vi.spyOn(SyncDatabase.prototype, 'pragma').mockImplementation(() => { + throw Object.assign(new Error('database disk image is malformed'), { + code: 'SQLITE_CORRUPT' + }) + }) + try { + expect(() => openSessionSearchDatabase(path)).toThrow(/malformed/) + } finally { + open.mockRestore() + } +}) + +it('surfaces the unlink failure itself when a stale index cannot be removed', async () => { + const path = await tempDatabasePath() + const stale = openSessionSearchDatabase(path) + stale + .prepare("UPDATE meta SET value = ? WHERE key = 'schema_version'") + .run(String(SESSION_SEARCH_SCHEMA_VERSION + 1)) + stale.close() + recordedRmSync.mockReset() + recordedRmSync.mockImplementation(() => { + throw Object.assign(new Error('EPERM: operation not permitted, unlink'), { + code: 'EPERM' + }) + }) + try { + // The stale handle is closed before the unlink, so the failure path must not + // close it again: ERR_INVALID_STATE would bury the cause and would not be + // classified as worth a rebuild. + expect(() => openSessionSearchDatabase(path)).toThrow(/EPERM/) + expect(() => openSessionSearchDatabase(path)).not.toThrow(/not open/) + } finally { + recordedRmSync.mockReset() + } +}) + +it('creates the directory the index lives in', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-session-search-mkdir-')) + roots.push(root) + // The real layout: `/ai-vault-search/index.sqlite`, where nothing + // has made that folder yet. SQLite would fail with `unable to open database + // file`, which is correctly not treated as corruption, so it never retries. + const db = openSessionSearchDatabase(join(root, 'ai-vault-search', 'index.sqlite')) + try { + expect(schemaVersion(db)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + } finally { + db.close() + } +}) + +it('rebuilds a newer index rather than reading a schema it does not know', async () => { + const path = await tempDatabasePath() + const newer = openSessionSearchDatabase(path) + newer.prepare("INSERT INTO files(path,byte_offset,mtime_ms) VALUES ('a',1,1)").run() + newer + .prepare("UPDATE meta SET value = ? WHERE key = 'schema_version'") + .run(String(SESSION_SEARCH_SCHEMA_VERSION + 1)) + newer.close() + + const rebuilt = openSessionSearchDatabase(path) + try { + expect(schemaVersion(rebuilt)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + expect(rebuilt.prepare('SELECT COUNT(*) AS c FROM files').get()).toEqual({ + c: 0 + }) + } finally { + rebuilt.close() + } +}) + +it('rebuilds when meta exists but its version row is gone', async () => { + const path = await tempDatabasePath() + const damaged = openSessionSearchDatabase(path) + damaged.prepare("INSERT INTO files(path,byte_offset,mtime_ms) VALUES ('a',1,1)").run() + // A meta table with no version is a damaged index, never a fresh one: seeding + // the current version over it would keep whatever the old schema left behind. + damaged.prepare("DELETE FROM meta WHERE key = 'schema_version'").run() + damaged.close() + + const rebuilt = openSessionSearchDatabase(path) + try { + expect(schemaVersion(rebuilt)).toBe(String(SESSION_SEARCH_SCHEMA_VERSION)) + expect(rebuilt.prepare('SELECT COUNT(*) AS c FROM files').get()).toEqual({ + c: 0 + }) + } finally { + rebuilt.close() + } +}) + +it('opens with the pragmas the write path depends on', async () => { + const db = openSessionSearchDatabase(await tempDatabasePath()) + try { + // auto_vacuum=2 is INCREMENTAL, and only takes on an empty file: without it + // a purge cannot hand pages back in bounded steps. + expect(Number(db.pragma('auto_vacuum', { simple: true }))).toBe(2) + expect(String(db.pragma('journal_mode', { simple: true })).toLowerCase()).toBe('wal') + expect(Number(db.pragma('synchronous', { simple: true }))).toBe(1) + // A WAL with no size limit never hands its space back after a large write. + expect(Number(db.pragma('journal_size_limit', { simple: true }))).toBe(8388608) + // Zero here turns every contended write into an immediate SQLITE_BUSY. + expect(Number(db.pragma('busy_timeout', { simple: true }))).toBe(5000) + } finally { + db.close() + } +}) + +it("walks a session's rows through an index rather than scanning the table", async () => { + const db = openSessionSearchDatabase(await tempDatabasePath()) + try { + // The replace delete and the orphan drain both take this path, once per file. + const plan = ( + db + .prepare('EXPLAIN QUERY PLAN SELECT id FROM messages WHERE session_row_id = ? LIMIT ?') + .all(1, 1) as { detail: string }[] + ) + .map((row) => row.detail) + .join(' ') + expect(plan).toContain('messages_session') + } finally { + db.close() + } +}) + +it('keeps only the session indexes a retrieval query can seek', async () => { + const db = openSessionSearchDatabase(await tempDatabasePath()) + try { + const names = ( + db + .prepare("SELECT name FROM sqlite_master WHERE type='index' AND tbl_name='sessions'") + .all() as { name: string }[] + ) + .map((row) => row.name) + .sort() + // One per shape PR 4's retrieval seeks: the agent filter, the newest-first + // order and date window, and the folder-prefix range scan. Fork folding reads + // `content_hash` off rows it already holds, so that column is not indexed. + expect(names).toEqual(['sessions_agent', 'sessions_cwd_key', 'sessions_updated_at']) + } finally { + db.close() + } +}) + +it("retries a Windows lock that outlives rmSync's own retries", async () => { + const path = await tempDatabasePath() + openSessionSearchDatabase(path).close() + vi.spyOn(process, 'platform', 'get').mockReturnValue('win32') + recordedRmSync.mockReset() + const locked = Object.assign(new Error('EPERM: operation not permitted'), { + code: 'EPERM' + }) + recordedRmSync.mockImplementationOnce(() => { + throw locked + }) + try { + expect(() => removeSessionSearchDatabase(path)).not.toThrow() + expect(recordedRmSync.mock.calls.length).toBe(5) + await expect(stat(path)).rejects.toMatchObject({ code: 'ENOENT' }) + } finally { + recordedRmSync.mockReset() + vi.restoreAllMocks() + } +}) + +it('gives Windows the shared retry options for a late handle release', async () => { + const path = await tempDatabasePath() + vi.spyOn(process, 'platform', 'get').mockReturnValue('win32') + recordedRmSync.mockClear() + try { + removeSessionSearchDatabase(path) + expect(recordedRmSync).toHaveBeenCalled() + for (const [, options] of recordedRmSync.mock.calls) { + expect(options).toMatchObject({ + maxRetries: WINDOWS_RM_MAX_RETRIES, + retryDelay: WINDOWS_RM_RETRY_DELAY_MS + }) + } + } finally { + vi.restoreAllMocks() + } +}) diff --git a/src/main/ai-vault-search/session-search-schema.ts b/src/main/ai-vault-search/session-search-schema.ts new file mode 100644 index 00000000000..ca525c47c0e --- /dev/null +++ b/src/main/ai-vault-search/session-search-schema.ts @@ -0,0 +1,209 @@ +import { mkdirSync } from 'node:fs' +import { dirname } from 'node:path' +import SyncDatabase from '../sqlite/sync-database' +import { removeTreeSync } from '../../shared/windows-transient-lock-removal' + +// The index stores transcript content as written, with no redaction. A secret in +// a transcript is already plaintext under the user's home directory and is +// treated as compromised; this is a second copy of content the user already +// holds. What a snippet may carry once it leaves this machine is a transport +// policy, decided where the wire is. + +// Bump to drop and rebuild: the index is a cache over the transcripts, never a source. +export const SESSION_SEARCH_SCHEMA_VERSION = 5 + +// unicode61 keeps `_ . - /` inside tokens so paths and identifiers match exactly; +// the `identifiers` column carries the split form (see session-search-identifier-split). +// Why: `+` keeps `C++` a token of its own instead of the letter `c`; `#` is +// left out so `#123` still answers a search for `123`. +const TOKENIZER = `tokenize="unicode61 tokenchars '_.-/+'"` + +const SCHEMA_SQL = ` +CREATE TABLE IF NOT EXISTS meta(key TEXT PRIMARY KEY, value TEXT NOT NULL); +CREATE TABLE IF NOT EXISTS sessions( + -- AUTOINCREMENT, because this id names rows in the messages table for longer + -- than the row itself lives: retention cuts a session loose in one + -- transaction and reclaims its messages over many. A plain rowid is reissued + -- as max+1, so a session created inside that window would be handed a freed + -- id and adopt whatever of the purged conversation the drain had not reached, + -- behind a live session no later purge visits. + id INTEGER PRIMARY KEY AUTOINCREMENT, + agent TEXT NOT NULL, + session_id TEXT NOT NULL, + -- The transcript this session was decoded from. Not unique: OpenCode's SQLite + -- sessions all report the store's own path here, while files.path holds the + -- synthetic db#sessionId candidate that really is one per session. + file_path TEXT NOT NULL, + codex_home TEXT, + title TEXT NOT NULL, + cwd TEXT, + cwd_key TEXT, + branch TEXT, + created_at TEXT, + updated_at TEXT, + message_count INTEGER NOT NULL DEFAULT 0, + resume_command TEXT NOT NULL, + -- Chained digest of the first N messages; forks of one conversation share it. + content_hash TEXT, + content_hash_count INTEGER NOT NULL DEFAULT 0 +); +CREATE INDEX IF NOT EXISTS sessions_agent ON sessions(agent); +CREATE INDEX IF NOT EXISTS sessions_updated_at ON sessions(updated_at); +CREATE INDEX IF NOT EXISTS sessions_cwd_key ON sessions(cwd_key); +CREATE TABLE IF NOT EXISTS files( + path TEXT PRIMARY KEY, + dev INTEGER, + ino INTEGER, + byte_offset INTEGER NOT NULL, + mtime_ms REAL NOT NULL, + size_bytes INTEGER, + session_row_id INTEGER, + -- What this row still owes a reader, so that nothing has to be remembered + -- between passes. 'current': the rows match the file at the stat recorded + -- here. 'due': the index is behind on content it cannot reach by appending, + -- so the next pass reads the file whole. 'failed': the last read did not + -- commit, and the two columns below are what stop it being retried for ever. + state TEXT NOT NULL DEFAULT 'current', + fail_count INTEGER NOT NULL DEFAULT 0, + -- The mtime the failures were observed at. A file that fails at one stat is + -- left alone once it has failed enough times, and only a change to this stat + -- can mean the file itself changed, so it is the whole retry policy. + failed_mtime_ms REAL +); +-- Retention walks the expiring end of this column; without it that is a full scan and a sort. +CREATE INDEX IF NOT EXISTS files_mtime ON files(mtime_ms); +CREATE TABLE IF NOT EXISTS messages( + id INTEGER PRIMARY KEY, + session_row_id INTEGER NOT NULL, + role TEXT NOT NULL, + ts TEXT +); +-- Both the replace delete and the orphan drain walk a session's rows through this. +CREATE INDEX IF NOT EXISTS messages_session ON messages(session_row_id); +-- One FTS table, not two. A conversation-scoped search is a column filter on +-- this one — 'MATCH {user_text assistant_text}: q' with bm25 weights that zero +-- the other two — and PR 4 measured that at 1.16-1.36x the p95 of a dedicated +-- second table on a 105 MB corpus, under the 2x bar the decision was set at. +CREATE VIRTUAL TABLE IF NOT EXISTS messages_fts USING fts5( + user_text, assistant_text, tool_text, identifiers, ${TOKENIZER}, detail=full +); +` + +/** + * Opens the index, rebuilding it whenever what is on disk cannot be trusted: + * a different schema version, a version SQLite cannot report, or a file torn + * badly enough that opening or recovery fails. The index is a cache over the + * transcripts, so throwing away a bad one costs a re-scan and nothing else; + * refusing to open would strand the feature until a human deleted the file. + */ +export function openSessionSearchDatabase(path: string): SyncDatabase { + // SQLite will not create the directory, and its failure is `unable to open + // database file`, which is correctly not corruption — so without this the + // feature strands on a profile that has never held an index. + if (path !== ':memory:') { + mkdirSync(dirname(path), { recursive: true }) + } + try { + return openExisting(path) + } catch (error) { + if (!isUnusableDatabaseError(error)) { + throw error + } + // One retry only: a second failure on a file we just created is not corruption. + removeSessionSearchDatabase(path) + return openExisting(path) + } +} + +function openExisting(path: string): SyncDatabase { + // Nulled while no handle is open, because closing an already-closed handle + // throws ERR_INVALID_STATE, which would replace whatever really failed — + // an unlink refused by a virus scanner or a second Orca holding the file — + // with an error nothing classifies as worth rebuilding for. + let db: SyncDatabase | null = openWithPragmas(path) + try { + if (isStaleSchema(db)) { + // Why: DROP TABLE on a multi-GB FTS index takes minutes and runs inside the + // scanner service's init, past its ready timeout; unlinking is instant. + db.close() + db = null + removeSessionSearchDatabase(path) + db = openWithPragmas(path) + } + db.exec(SCHEMA_SQL) + db.prepare('INSERT OR REPLACE INTO meta(key, value) VALUES (?, ?)').run( + 'schema_version', + String(SESSION_SEARCH_SCHEMA_VERSION) + ) + return db + } catch (error) { + db?.close() + throw error + } +} + +// SQLite reports a torn file at the first statement that has to read a page, so +// this has to match on the message as well as the code. +const UNUSABLE_DATABASE = + /SQLITE_CORRUPT|SQLITE_NOTADB|file is not a database|database disk image is malformed/i + +function isUnusableDatabaseError(error: unknown): boolean { + if (!(error instanceof Error)) { + return false + } + const code = (error as { code?: unknown }).code + return ( + (typeof code === 'string' && UNUSABLE_DATABASE.test(code)) || + UNUSABLE_DATABASE.test(error.message) + ) +} + +function openWithPragmas(path: string): SyncDatabase { + const db = new SyncDatabase(path) + try { + // Why: only takes effect on an empty file; it is what lets a purge hand pages + // back in bounded steps instead of a full VACUUM. Set before any table exists. + db.pragma('auto_vacuum = INCREMENTAL') + // The whole consistency model: a file's rows and its cursor land in one + // transaction, and a reader on another handle sees the last committed state + // of the index rather than a session half way through being rewritten. + db.pragma('journal_mode = WAL') + db.pragma('synchronous = NORMAL') + db.pragma('journal_size_limit = 8388608') + db.pragma('busy_timeout = 5000') + return db + } catch (error) { + db?.close() + throw error + } +} + +export function removeSessionSearchDatabase(path: string): void { + if (path === ':memory:') { + return + } + for (const suffix of ['', '-wal', '-shm', '-journal']) { + removeTreeSync(`${path}${suffix}`) + } +} + +/** + * Whether what is on disk has to be thrown away. No `meta` table at all is a + * file with nothing in it to throw away, and removing it would make the first + * open of every new profile a create-remove-create. A meta table whose version + * row is missing or unparseable is a damaged index rather than a new one: + * seeding the current version over it would keep whatever rows the old schema + * left. + */ +function isStaleSchema(db: SyncDatabase): boolean { + const table = db + .prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'meta'") + .get() + if (!table) { + return false + } + const row = db.prepare("SELECT value FROM meta WHERE key = 'schema_version'").get() as + | { value: string } + | undefined + return (row ? Number(row.value) : Number.NaN) !== SESSION_SEARCH_SCHEMA_VERSION +} diff --git a/src/main/ai-vault-search/session-search-store-is-memory.test.ts b/src/main/ai-vault-search/session-search-store-is-memory.test.ts new file mode 100644 index 00000000000..2e90a75862f --- /dev/null +++ b/src/main/ai-vault-search/session-search-store-is-memory.test.ts @@ -0,0 +1,265 @@ +import { chmod, rm, utimes } from 'node:fs/promises' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import type SyncDatabase from '../sqlite/sync-database' +import { SessionSearchIndexer } from './session-search-indexer' +import { + FakeSessionSearchClock, + openSessionSearchIndexerHarness, + writeClaudeTranscript, + type SessionSearchIndexerHarness +} from './session-search-indexer-test-fixture' + +/* + * S1-S5: the store is the only memory. + * + * Every question the indexer answers between passes -- what is owed a read, + * what has failed and how often, what it holds and therefore what may have been + * deleted, what to report -- is a row in the `files` table. These tests check + * that from outside the object: a second connection, hand-written SQL, and the + * clock. Two things outlive a pass and are not rows, and both are named here: + * the timer, and one bit per root for the retirement walk's grace. + */ + +const INTERVAL_MS = 20_000 +const CAN_DENY_READ = process.platform !== 'win32' && process.getuid?.() !== 0 +const FIRST = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +const SECOND = 'bbbbbbbb-cccc-4ddd-8eee-ffffffffffff' +const THIRD = 'cccccccc-dddd-4eee-8fff-000000000000' + +let harness: SessionSearchIndexerHarness +let clock: FakeSessionSearchClock +let indexer: SessionSearchIndexer | null +let errors: unknown[] + +beforeEach(async () => { + resetSessionParseCacheForTests() + resetTranscriptConsumersForTests() + errors = [] + clock = new FakeSessionSearchClock() + harness = await openSessionSearchIndexerHarness('ss-memory') + indexer = null +}) + +afterEach(async () => { + indexer?.close() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await harness.cleanup() +}) + +function newIndexer( + overrides: Partial[0]> = {} +): SessionSearchIndexer { + indexer = new SessionSearchIndexer({ + databasePath: harness.databasePath, + roots: harness.roots, + historyDays: null, + clock, + reconcileIntervalMs: INTERVAL_MS, + onError: (error) => errors.push(error), + ...overrides + }) + return indexer +} + +function transcriptPath(name: string): string { + return join(harness.claudeProjectDir, `${name}.jsonl`) +} + +async function nextCycle(): Promise { + clock.advance(INTERVAL_MS) + await indexer?.settled() +} + +/** The whole `files` table as a second connection sees it, ordered for comparison. */ +function fileTable(): unknown[] { + return harness.read((db: SyncDatabase) => + db + .prepare( + `SELECT path, dev, ino, byte_offset, mtime_ms, size_bytes, session_row_id, + state, fail_count, failed_mtime_ms + FROM files ORDER BY path` + ) + .all() + ) +} + +// S1. The status is a query. A counter kept beside the rows is what needs a +// rule about when to reset, and every such rule this feature grew was wrong. +it('S1: reports exactly what a hand-written query over the rows reports', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['one'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['two'], SECOND) + await newIndexer().start() + + const bySql = (): Record => + Object.fromEntries( + ( + harness.read((db: SyncDatabase) => + db.prepare('SELECT state, count(*) AS n FROM files GROUP BY state').all() + ) as { state: string; n: number }[] + ).map((row) => [row.state, Number(row.n)]) + ) + + const reported = indexer?.status() + const counted = bySql() + expect(reported?.filesIndexed).toBe(counted.current ?? 0) + expect(reported?.filesDue).toBe(counted.due ?? 0) + expect(reported?.filesFailed).toBe(counted.failed ?? 0) + expect(reported?.filesIndexed).toBe(2) + + // And it stays a query: delete a row behind the indexer's back and the very + // next call reports the table, not a number it remembered. + harness.write((db: SyncDatabase) => + db.prepare('DELETE FROM files WHERE path = ?').run(transcriptPath(FIRST)) + ) + expect(indexer?.status().filesIndexed).toBe(1) +}) + +// S2. A deletion is proven by comparing the rows against what discovery +// returned, so the moment it happened does not matter. Every boundary a pass +// has is a moment a file can go. +it('S2: retires a file deleted right after the opening sweep', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await newIndexer().start() + + await rm(transcriptPath(FIRST)) + await nextCycle() + + expect(fileTable()).toHaveLength(1) +}) + +it('S2: retires a file deleted right after a cycle', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await newIndexer().start() + await nextCycle() + + await rm(transcriptPath(FIRST)) + await nextCycle() + + expect(fileTable()).toHaveLength(1) +}) + +it('S2: retires a file deleted right after a periodic sweep', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await newIndexer({ fullSweepEveryCycles: 2 }).start() + await nextCycle() + await nextCycle() + // The third pass is the periodic sweep; the file goes the moment it ends. + await nextCycle() + + await rm(transcriptPath(FIRST)) + await nextCycle() + + expect(fileTable()).toHaveLength(1) +}) + +it('S2: retires a file deleted while a pass was out of time', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['going'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['staying'], SECOND) + await writeClaudeTranscript(transcriptPath(THIRD), ['also staying'], THIRD) + // One transcript a pass: the opening sweep leaves two of the three unread. + clock.costPerNowMs = 1_000 + await newIndexer({ passDeadlineMs: 1_000 }).start() + expect(fileTable()).toHaveLength(1) + + await rm(transcriptPath(FIRST)) + await nextCycle() + await nextCycle() + + // Read what it could, and proved the deletion in the same pass it was still + // catching up in: retirement is not what the deadline bounds. + expect((fileTable() as { path: string }[]).map((row) => row.path)).not.toContain( + transcriptPath(FIRST) + ) +}) + +// S3. The stat is the whole retry policy: a file that fails at one stat stops +// being read, and only a change to that stat starts it again. +it.skipIf(!CAN_DENY_READ)( + 'S3: stops reading a file that fails three times at one stat', + async () => { + const path = transcriptPath(FIRST) + await writeClaudeTranscript(path, ['behind the wrong mode bits'], FIRST) + await chmod(path, 0o000) + try { + await newIndexer().start() + for (let cycle = 0; cycle < 4; cycle++) { + await nextCycle() + } + + const row = harness.read((db: SyncDatabase) => + db.prepare('SELECT state, fail_count AS failCount FROM files WHERE path = ?').get(path) + ) as { state: string; failCount: number } + // Three, not four and not seven: the pass after the third costs nothing. + expect(row).toEqual({ state: 'failed', failCount: 3 }) + expect(indexer?.status()).toMatchObject({ filesFailed: 1, phase: 'degraded' }) + + // Only the stat releases it. + await chmod(path, 0o644) + const later = new Date(Date.now() + 60_000) + await utimes(path, later, later) + await nextCycle() + + expect(indexer?.status()).toMatchObject({ filesIndexed: 1, filesFailed: 0 }) + } finally { + await chmod(path, 0o644) + } + } +) + +// S4. Two passes over an unchanged filesystem leave the table byte for byte as +// they found it. Anything that drifted would be state the rows do not hold. +it('S4: leaves the file table identical across passes with no change on disk', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['one'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['two'], SECOND) + await newIndexer({ fullSweepEveryCycles: 2 }).start() + + const afterSweep = fileTable() + await nextCycle() + expect(fileTable()).toEqual(afterSweep) + await nextCycle() + expect(fileTable()).toEqual(afterSweep) + // Including across the periodic sweep, which reads the same rows again. + await nextCycle() + expect(fileTable()).toEqual(afterSweep) + expect(errors).toEqual([]) +}) + +// S5. Nothing a close interrupts needs repairing: the next instance reads the +// rows as they stand and decides from them alone. +it('S5: leaves the store consistent when a close interrupts a pass', async () => { + await writeClaudeTranscript(transcriptPath(FIRST), ['one'], FIRST) + await writeClaudeTranscript(transcriptPath(SECOND), ['two'], SECOND) + await writeClaudeTranscript(transcriptPath(THIRD), ['three'], THIRD) + newIndexer() + let closed = false + clock.onNow = () => { + if (closed || fileTable().length === 0) { + return + } + closed = true + indexer?.close() + } + await indexer?.start() + await indexer?.settled() + clock.onNow = null + + const interrupted = fileTable() + expect(interrupted.length).toBeGreaterThan(0) + expect(interrupted.length).toBeLessThan(3) + expect(errors).toEqual([]) + + // A new instance over the same database: no repair pass, no recovery, just + // the rows and what they say is owed. + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + await newIndexer().start() + + expect(indexer?.status()).toMatchObject({ filesIndexed: 3, filesDue: 0, filesFailed: 0 }) +}) diff --git a/src/main/ai-vault-search/session-search-store.ts b/src/main/ai-vault-search/session-search-store.ts new file mode 100644 index 00000000000..daa9267561b --- /dev/null +++ b/src/main/ai-vault-search/session-search-store.ts @@ -0,0 +1,314 @@ +import type SyncDatabase from '../sqlite/sync-database' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' +import type { TranscriptSessionIdentity } from '../ai-vault/session-transcript-consumers' +import type { + SessionSearchFileIdentity, + SessionSearchIndexedFile +} from './session-search-file-cursor' +import { + SESSION_SEARCH_COMMIT_CHARS, + SessionSearchIndexWriter, + type SessionSearchFileWrite +} from './session-search-index-writer' +import { deleteExpiredSearchFiles, drainOrphanedMessages } from './session-search-retention-delete' +import { openSessionSearchDatabase } from './session-search-schema' + +/** + * What a row still owes a reader. + * + * `current`: the rows match the file at the stat this row records. + * `due`: the index is behind on a span it cannot reach by appending, so the + * next pass must read the file whole. + * `failed`: the last read did not commit; `failCount` and `failedMtimeMs` are + * what stop it being retried for ever. + */ +export type SessionSearchFileState = 'current' | 'due' | 'failed' + +/** + * One row of the index's own file table. + * + * This is the indexer's whole memory between passes: what it holds, at what + * stat, and what each row still owes. Nothing it decides is answered from + * anywhere else, which is why a second connection can check its status. + */ +export type SessionSearchFileRow = { + path: string + identity: SessionSearchFileIdentity + mtimeMs: number + sizeBytes: number | null + state: SessionSearchFileState + failCount: number + failedMtimeMs: number | null +} + +/** How many rows are in each state; the whole of the indexer's progress report. */ +export type SessionSearchStateCounts = { current: number; due: number; failed: number } + +/** + * Owns the index database. PR 2 scope: the write half only — the transcript + * consumer writes through it and nothing reads from it yet. Lifecycle (who + * indexes, when, and how the re-read set is drained) belongs to the service. + */ +export class SessionSearchStore { + private readonly db: SyncDatabase + private readonly writer: SessionSearchIndexWriter + private closed = false + private retentionCutoffMs: number | null = null + // One drain at a time. A replace that commits while one is running asks for + // another pass rather than starting a second walk of the same rows. + private draining = false + private drainRequested = false + + constructor( + path: string, + private readonly onError: (error: unknown) => void = (error) => + console.warn( + '[ai-vault-search] index write failed:', + error instanceof Error ? error.name : 'IndexError' + ) + ) { + this.db = openSessionSearchDatabase(path) + this.writer = new SessionSearchIndexWriter(this.db, SESSION_SEARCH_COMMIT_CHARS, () => + this.scheduleOrphanDrain() + ) + } + + /** + * Reclaims the rows a replace cut loose, once its transaction has committed. + * + * The same split retention makes, for the same reason: deleting the old + * session row is what stops it answering, because every retrieval joins + * `sessions`, and handing its messages back is the expensive half that must + * not hold one transaction. Nothing records the work: rows whose session row + * is gone are the whole record, so a crash before or during a drain is found + * by the next one. + */ + private scheduleOrphanDrain(): void { + this.drainRequested = true + if (this.draining || this.closed) { + return + } + this.draining = true + // Off the committing stack. An async function runs synchronously up to its + // first `await`, so calling the drain here would put its first batch back + // inside the call that committed the replace — the cost this took out. + void Promise.resolve().then(() => this.runOrphanDrain()) + } + + private async runOrphanDrain(): Promise { + try { + while (this.drainRequested && !this.closed) { + this.drainRequested = false + await drainOrphanedMessages(this.db, () => this.closed) + } + } catch (error) { + if (!this.closed) { + this.onError(error) + } + } finally { + this.draining = false + } + } + + /** + * The index handle, for a reader composed over this store (PR 4's engine). + * + * Two rules come with it, both measured in this PR. **Never hold a read + * transaction across an `await`**: a checkpoint cannot pass an open read + * snapshot, so a paginated read that opened `BEGIN` and yielded between pages + * takes the WAL from 10 MB to 266 MB and it does not come back. And **no + * `.iterate()` that outlives its statement**, which is the same pin by + * another name. Every retrieval a single synchronous statement is the whole + * contract. + */ + get connection(): SyncDatabase { + return this.db + } + + /** The oldest transcript mtime worth indexing; PR 3 derives it from the retention setting. */ + setRetentionCutoffMs(cutoffMs: number | null): void { + this.retentionCutoffMs = cutoffMs + } + + /** The cutoff a caller's own decide step compares a candidate's mtime against. */ + get retentionCutoff(): number | null { + return this.retentionCutoffMs + } + + /** + * Whether this candidate is new enough to hold rows for. + * + * Enforced here as well as in the indexer's decide step, and not only there: + * the consumer observes every read the session list makes, not only the ones + * the index asked for, so a sidebar scan of a transcript outside the window + * would otherwise index rows the next purge deletes again. + */ + private withinRetention(candidate: SessionFileCandidate): boolean { + return this.retentionCutoffMs === null || candidate.file.mtimeMs >= this.retentionCutoffMs + } + + indexedFile(path: string, identity: SessionSearchFileIdentity): SessionSearchIndexedFile | null { + try { + return this.writer.indexedFile(path, identity) + } catch (error) { + this.onError(error) + return null + } + } + + /** Null when this read cannot extend the index, or when the store refuses writes. */ + beginWrite( + candidate: SessionFileCandidate, + mode: 'replace' | 'append', + previousByteOffset: number, + identity?: () => TranscriptSessionIdentity | null + ): SessionSearchFileWrite | null { + if (this.closed || !this.withinRetention(candidate)) { + return null + } + try { + return this.writer.beginWrite(candidate, mode, previousByteOffset, identity) + } catch (error) { + this.reportWriteFailure(error) + return null + } + } + + /** + * A read that landed. Written after the commit rather than inside it: the + * transaction owns the rows and the cursor, and a crash between the two + * leaves a row that says `failed` over content that is in fact current, which + * the next pass fixes by reading a file it did not have to. + */ + writeCommitted(candidate: SessionFileCandidate): void { + this.setFileState(candidate.file.path, 'current') + } + + reportWriteFailure(error: unknown): void { + this.onError(error) + } + + /** + * Every row this index holds. The candidate list for retirement and the whole + * of the status, read in one query so that no pass has to carry either. + * + * The cursor is deliberately not here: whether a row can be continued is + * `indexedFile`'s question, and one spelling of the half-written sentinel is + * enough. + */ + files(): SessionSearchFileRow[] { + return ( + this.db + .prepare( + `SELECT path, dev, ino, mtime_ms AS mtimeMs, size_bytes AS sizeBytes, + state, fail_count AS failCount, failed_mtime_ms AS failedMtimeMs + FROM files` + ) + .all() as (Omit & { + dev: number | null + ino: number | null + })[] + ).map((row) => ({ + path: row.path, + identity: + typeof row.dev === 'number' && typeof row.ino === 'number' + ? { dev: row.dev, ino: row.ino } + : null, + mtimeMs: row.mtimeMs, + sizeBytes: row.sizeBytes, + state: row.state, + failCount: row.failCount, + failedMtimeMs: row.failedMtimeMs + })) + } + + /** + * Moves a row's read state. + * + * `failed` also counts the failure and records the stat it happened at, which + * is what lets the next pass tell "this file has never worked" from "this + * file has changed since it last failed". A path with no row is a no-op: the + * next pass reads it because the index holds nothing for it. + */ + setFileState(path: string, state: SessionSearchFileState, atMtimeMs?: number): void { + try { + if (state === 'failed') { + // Inserted when there is no row, because the common unreadable file is + // one the index never managed to hold: a transcript behind the wrong + // mode bits fails on its very first read, and with nowhere to write the + // count it would be read again on every pass for the life of the + // process. The cursor is zero and there is no session, which is what + // "the index holds nothing for this file" already looks like. + this.db + .prepare( + `INSERT INTO files(path, byte_offset, mtime_ms, state, fail_count, failed_mtime_ms) + VALUES (?, 0, ?, 'failed', 1, ?) + ON CONFLICT(path) DO UPDATE SET + state = 'failed', + fail_count = files.fail_count + 1, + failed_mtime_ms = excluded.failed_mtime_ms` + ) + .run(path, atMtimeMs ?? 0, atMtimeMs ?? null) + return + } + this.db + .prepare( + 'UPDATE files SET state = ?, fail_count = 0, failed_mtime_ms = NULL WHERE path = ?' + ) + .run(state, path) + } catch (error) { + this.onError(error) + } + } + + /** Rows per state. The status is this query and the pass's own degraded roots. */ + stateCounts(): SessionSearchStateCounts { + const rows = this.db.prepare('SELECT state, count(*) AS n FROM files GROUP BY state').all() as { + state: SessionSearchFileState + n: number + }[] + const counts: SessionSearchStateCounts = { current: 0, due: 0, failed: 0 } + for (const row of rows) { + counts[row.state] = Number(row.n) + } + return counts + } + + /** + * Drops a source's rows. Only a proven deletion may call this: an unreadable + * source is `unverifiable`, not `missing`, and keeps its rows + * (docs/reference/ssh-execution-boundary.md). + */ + removeFile(path: string): void { + try { + this.writer.removeFile(path) + } catch (error) { + this.onError(error) + } + } + + /** Cuts expired sessions loose at once, then reclaims their rows in resumable batches. */ + async purgeOlderThan(cutoffMs: number | null, signal?: AbortSignal): Promise { + try { + await deleteExpiredSearchFiles( + this.db, + cutoffMs, + () => this.closed || signal?.aborted === true + ) + } catch (error) { + if (!this.closed) { + this.onError(error) + } + } + } + + close(): void { + // node:sqlite throws ERR_INVALID_STATE on a second close, and a store is + // closed both by its owner and by a test's teardown. + if (this.closed) { + return + } + this.closed = true + this.db.close() + } +} diff --git a/src/main/ai-vault-search/session-search-synthetic-corpus.test.ts b/src/main/ai-vault-search/session-search-synthetic-corpus.test.ts new file mode 100644 index 00000000000..9b6a48beb4e --- /dev/null +++ b/src/main/ai-vault-search/session-search-synthetic-corpus.test.ts @@ -0,0 +1,44 @@ +import { rm } from 'node:fs/promises' +import { join } from 'node:path' +import { expect, it } from 'vitest' +import { resetSessionParseCacheForTests } from '../ai-vault/session-scanner-parse-cache' +import { resetTranscriptConsumersForTests } from '../ai-vault/session-transcript-consumers' +import { registerSessionSearchIndexConsumer } from './session-search-index-consumer' +import { SessionSearchStore } from './session-search-store' +import { writeSyntheticTranscriptCorpus } from './session-search-synthetic-corpus' +import { parseTranscript } from './session-search-transcript-fixtures' + +it.each([Infinity, -Infinity, Number.NaN, -1, 1.5])( + 'rejects invalid corpus loop bounds: %s', + async (value) => { + for (const field of ['sessions', 'turnsPerSession', 'toolResultWords']) { + await expect(writeSyntheticTranscriptCorpus({ [field]: value })).rejects.toThrow(RangeError) + } + } +) + +it.each([0, 200, 2000])( + 'counts the indexed messages with %s tool words', + async (toolResultWords) => { + const corpus = await writeSyntheticTranscriptCorpus({ + sessions: 1, + turnsPerSession: 1, + toolResultWords + }) + const store = new SessionSearchStore(join(corpus.root, 'index.sqlite')) + const unregister = registerSessionSearchIndexConsumer(store) + try { + await parseTranscript(corpus.files[0]!) + expect(corpus.messageCount).toBe(toolResultWords === 0 ? 3 : 4) + expect(store.connection.prepare('SELECT count(*) AS n FROM messages').get()).toEqual({ + n: corpus.messageCount + }) + } finally { + unregister() + resetTranscriptConsumersForTests() + resetSessionParseCacheForTests() + store.close() + await rm(corpus.root, { recursive: true, force: true }) + } + } +) diff --git a/src/main/ai-vault-search/session-search-synthetic-corpus.ts b/src/main/ai-vault-search/session-search-synthetic-corpus.ts new file mode 100644 index 00000000000..14e8a27fe5f --- /dev/null +++ b/src/main/ai-vault-search/session-search-synthetic-corpus.ts @@ -0,0 +1,154 @@ +import { mkdtemp, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' + +// Why synthetic and in-repo: the cost model has to be reproducible on any host +// and must never read a real transcript. The shapes here mirror what a Claude +// JSONL transcript actually holds — prose turns, a pasted diff, tool calls and +// their output — because the index's disk cost tracks the mix, not the size. + +const WORDS = [ + 'terminal', + 'reattach', + 'worktree', + 'resolveTerminalPath', + 'src/main/ai-vault/session-transcript-reader.ts', + 'the', + 'index', + 'cursor', + 'byteOffset', + 'publish', + 'staged', + 'transaction', + 'MAX_RETRIES', + 'relay', + 'daemon', + 'pty', + 'snapshot', + 'because' +] + +/** Deterministic: the same seed gives the same corpus on every host and run. */ +function mulberry32(seed: number): () => number { + let state = seed >>> 0 + return () => { + state = (state + 0x6d2b79f5) >>> 0 + let t = Math.imul(state ^ (state >>> 15), 1 | state) + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } +} + +function words(random: () => number, count: number): string { + const out: string[] = [] + for (let index = 0; index < count; index++) { + out.push(WORDS[Math.floor(random() * WORDS.length)]) + } + return out.join(' ') +} + +export type SyntheticCorpus = { + root: string + files: string[] + /** Total bytes of transcript written, the denominator of write amplification. */ + transcriptBytes: number + messageCount: number +} + +export type SyntheticCorpusOptions = { + sessions?: number + turnsPerSession?: number + seed?: number + /** + * Words per tool result. The default keeps tool output at about half the + * message text; the real distribution is 80-97 %, which is what prices the + * tool-row cap, so the benchmark runs a second arm well above the default. + */ + toolResultWords?: number +} + +/** Writes a corpus of Claude JSONL transcripts and reports what it cost on disk. */ +export async function writeSyntheticTranscriptCorpus( + options: SyntheticCorpusOptions = {} +): Promise { + const sessions = options.sessions ?? 40 + const turns = options.turnsPerSession ?? 60 + const toolWords = options.toolResultWords ?? 200 + for (const [name, value] of Object.entries({ + sessions, + turnsPerSession: turns, + toolResultWords: toolWords + })) { + if (!Number.isSafeInteger(value) || value < 0) { + throw new RangeError(`${name} must be a finite non-negative safe integer`) + } + } + const random = mulberry32(options.seed ?? 1) + const root = await mkdtemp(join(tmpdir(), 'orca-search-corpus-')) + const files: string[] = [] + let transcriptBytes = 0 + let messageCount = 0 + + for (let session = 0; session < sessions; session++) { + const sessionId = `00000000-0000-4000-8000-${String(session).padStart(12, '0')}` + const lines: string[] = [] + for (let turn = 0; turn < turns; turn++) { + const at = new Date(1740000000000 + turn * 60_000).toISOString() + lines.push( + JSON.stringify({ + type: 'user', + sessionId, + timestamp: at, + cwd: `/repo/app-${session % 7}`, + gitBranch: 'main', + message: { role: 'user', content: words(random, 40) } + }) + ) + lines.push( + JSON.stringify({ + type: 'assistant', + sessionId, + timestamp: at, + message: { + role: 'assistant', + model: 'claude-fable-5', + content: [ + { type: 'text', text: words(random, 120) }, + { + type: 'tool_use', + name: 'Bash', + input: { command: `rg ${words(random, 3)}` } + } + ] + } + }) + ) + lines.push( + JSON.stringify({ + type: 'user', + sessionId, + timestamp: at, + message: { + role: 'user', + content: [ + { + type: 'tool_result', + tool_use_id: 'toolu_1', + content: words(random, toolWords) + } + ] + } + }) + ) + // Empty tool results emit no searchable message. + messageCount += toolWords === 0 ? 3 : 4 + } + const path = join(root, `${sessionId}.jsonl`) + const body = `${lines.join('\n')}\n` + await writeFile(path, body) + transcriptBytes += Buffer.byteLength(body) + files.push(path) + } + + return { root, files, transcriptBytes, messageCount } +} diff --git a/src/main/ai-vault-search/session-search-synthetic-sources.ts b/src/main/ai-vault-search/session-search-synthetic-sources.ts new file mode 100644 index 00000000000..86222b9c5fe --- /dev/null +++ b/src/main/ai-vault-search/session-search-synthetic-sources.ts @@ -0,0 +1,56 @@ +import type { AiVaultScanIssue } from '../../shared/ai-vault-types' +import { splitOpenCodeSqliteCandidate } from '../ai-vault/session-scanner-opencode-sqlite-paths' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' + +/** + * A row whose path names a container and an entry inside it rather than a file + * of its own. OpenCode's SQLite sessions are the one shape today + * (`#`), which is why this reads through that source's + * own splitter rather than reinventing the encoding. + */ +export type SessionSearchSyntheticSource = { container: string; id: string } + +export function splitSyntheticSessionSource(path: string): SessionSearchSyntheticSource | null { + const openCode = splitOpenCodeSqliteCandidate(path) + return openCode ? { container: openCode.dbPath, id: openCode.sessionId } : null +} + +/** + * Which containers a pass enumerated in full, and every id each of them held. + * + * This is the synthetic equivalent of a directory listing, and it has to meet + * the same bar before the retirement walk may prove anything from it: + * + * - **Exhaustive.** Only a sweep enumerates without a per-agent limit. A cycle + * asks for the newest N, so an id it did not return may simply be the N+1th. + * Callers that are not a census do not build this at all. + * - **Successful.** A container a scan issue names could not be read, and a + * read that failed returns no ids rather than an error the walk can see. A + * named container is left out, so its rows stay unverifiable. + * - **Non-empty.** A container that returned nothing is not evidence that it + * holds nothing: a database whose schema this scanner no longer recognises + * returns an empty list with no error at all, and believing it would retire + * every session in one pass. The cost is one stale row per container whose + * last entry the user deletes, until the container gains an entry or goes. + */ +export function sessionSearchEnumeratedContainers( + candidates: readonly SessionFileCandidate[], + issues: readonly AiVaultScanIssue[] +): Map> { + const containers = new Map>() + for (const candidate of candidates) { + const synthetic = splitSyntheticSessionSource(candidate.file.path) + if (!synthetic) { + continue + } + const ids = containers.get(synthetic.container) ?? new Set() + ids.add(synthetic.id) + containers.set(synthetic.container, ids) + } + for (const issue of issues) { + if (issue.kind !== 'notice') { + containers.delete(issue.path) + } + } + return containers +} diff --git a/src/main/ai-vault-search/session-search-transcript-fixtures.ts b/src/main/ai-vault-search/session-search-transcript-fixtures.ts new file mode 100644 index 00000000000..bc7eda9a8ff --- /dev/null +++ b/src/main/ai-vault-search/session-search-transcript-fixtures.ts @@ -0,0 +1,118 @@ +import { stat } from 'node:fs/promises' +import { + createSessionParseStats, + parseAgentSessionFileCached, + type SessionParseStats +} from '../ai-vault/session-scanner-parse-cache' +import type { SessionFileCandidate } from '../ai-vault/session-scanner-types' + +// Transcript builders shared by the session-search store tests; each file owns +// its temp directories, this module only shapes records and drives the parser. + +export const CLAUDE_SESSION_ID = 'aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee' +export const CODEX_SESSION_ID = '019f0000-1111-7222-8333-444444444444' +export const CODEX_ROLLOUT_FILE = `rollout-2026-05-01T10-00-00-${CODEX_SESSION_ID}.jsonl` + +const RECORD_EPOCH_MS = 1740000000000 + +export function recordTimestamp(index: number): string { + return new Date(RECORD_EPOCH_MS + index * 60_000).toISOString() +} + +export function userRecord( + index: number, + content: unknown, + sessionId = CLAUDE_SESSION_ID, + cwd = '/repo/app' +): string { + return JSON.stringify({ + type: 'user', + sessionId, + timestamp: recordTimestamp(index), + cwd, + gitBranch: 'main', + message: { role: 'user', content } + }) +} + +export function assistantRecord( + index: number, + content: unknown, + sessionId = CLAUDE_SESSION_ID +): string { + return JSON.stringify({ + type: 'assistant', + sessionId, + timestamp: recordTimestamp(index), + message: { role: 'assistant', model: 'claude-fable-5', content } + }) +} + +export async function sessionCandidate( + agent: SessionFileCandidate['agent'], + path: string, + codexHome: string | null = null +): Promise { + const fileStat = await stat(path) + return { + agent, + codexHome, + file: { + path, + mtimeMs: fileStat.mtimeMs, + modifiedAt: fileStat.mtime.toISOString(), + sizeBytes: fileStat.size, + dev: fileStat.dev, + ino: fileStat.ino + } + } +} + +export async function parseTranscript( + path: string, + agent: SessionFileCandidate['agent'] = 'claude', + codexHome: string | null = null +): Promise<{ stats: SessionParseStats }> { + const stats = createSessionParseStats() + await parseAgentSessionFileCached( + await sessionCandidate(agent, path, codexHome), + process.platform, + stats + ) + return { stats } +} + +function codexLine(record: Record): string { + return JSON.stringify(record) +} + +/** Minimal Codex rollout: meta, one user message, one completed shell command. */ +export function codexRolloutLines(command: string[], output: string, prompt: string): string[] { + return [ + codexLine({ + timestamp: recordTimestamp(0), + type: 'session_meta', + payload: { id: CODEX_SESSION_ID, cwd: '/repo/app', git: { branch: 'main' } } + }), + codexLine({ + timestamp: recordTimestamp(1), + type: 'response_item', + payload: { type: 'message', role: 'user', content: prompt } + }), + codexLine({ + timestamp: recordTimestamp(2), + type: 'response_item', + payload: { + type: 'function_call', + call_id: 'call-1', + name: 'shell', + arguments: JSON.stringify({ command }) + } + }), + codexLine({ + timestamp: recordTimestamp(3), + type: 'response_item', + payload: { type: 'function_call_output', call_id: 'call-1', output } + }) + ] +} diff --git a/src/main/ai-vault-search/session-search-work-loop.ts b/src/main/ai-vault-search/session-search-work-loop.ts new file mode 100644 index 00000000000..41a0ad98497 --- /dev/null +++ b/src/main/ai-vault-search/session-search-work-loop.ts @@ -0,0 +1,87 @@ +import type { SessionSearchClock, SessionSearchTimerHandle } from './session-search-clock' + +export type SessionSearchWorkLoopOptions = { + clock: SessionSearchClock + intervalMs: number + /** A task that threw for a reason other than its own abort. */ + onFailure: (error: unknown) => void +} + +/** + * Runs the indexer's passes one at a time, on an interval, until it is closed. + * + * Separate from the indexer because it is the part with no opinion about + * transcripts: a task chain that never overlaps itself, a timer that only ever + * has one pending tick, and a close that cancels both. Arming inside the chain + * rather than beside it is what makes `settled` mean "everything queued so far + * has finished, including the re-arm", which is what a fake-clock test needs. + */ +export class SessionSearchWorkLoop { + private timer: SessionSearchTimerHandle | null = null + private controller: AbortController | null = null + private chain: Promise = Promise.resolve() + private closed = false + + constructor(private readonly options: SessionSearchWorkLoopOptions) {} + + /** Everything queued so far. Never rejects: a task's failure is reported, not thrown. */ + get settled(): Promise { + return this.chain + } + + /** Queues `work` behind whatever is running, then re-arms the interval. */ + queue(work: (signal: AbortSignal) => Promise, tick: () => void): Promise { + const chained = this.chain + .then( + () => this.run(work), + () => this.run(work) + ) + .then(() => this.arm(tick)) + this.chain = chained + return chained + } + + /** + * Stops the timer, the task in flight and everything queued behind it. Nothing + * queued before this call may run afterwards: that is what lets the indexer + * close its store here and know no pass will reach for it. + */ + close(): void { + this.closed = true + if (this.timer !== null) { + this.options.clock.clearTimeout(this.timer) + this.timer = null + } + this.controller?.abort() + } + + private arm(tick: () => void): void { + if (this.closed || this.timer !== null) { + return + } + this.timer = this.options.clock.setTimeout(() => { + this.timer = null + tick() + }, this.options.intervalMs) + } + + private async run(work: (signal: AbortSignal) => Promise): Promise { + if (this.closed) { + return + } + const controller = new AbortController() + this.controller = controller + try { + await work(controller.signal) + } catch (error) { + // An aborted task is a close, never a failure. + if (!controller.signal.aborted) { + this.options.onFailure(error) + } + } finally { + if (this.controller === controller) { + this.controller = null + } + } + } +} diff --git a/src/main/ai-vault/session-scanner-accumulator.ts b/src/main/ai-vault/session-scanner-accumulator.ts index 88e09627eb7..18f273d6be3 100644 --- a/src/main/ai-vault/session-scanner-accumulator.ts +++ b/src/main/ai-vault/session-scanner-accumulator.ts @@ -23,7 +23,11 @@ import { normalizePreviewText, timestampMs } from './session-scanner-values' -import { NO_TRANSCRIPT_MESSAGES, type TranscriptMessageSink } from './session-transcript-consumers' +import { + NO_TRANSCRIPT_MESSAGES, + type TranscriptMessageSink, + type TranscriptSessionIdentity +} from './session-transcript-consumers' import { boundedText, transcriptMessageRole, @@ -64,6 +68,28 @@ export function createAccumulator(args: { } } +/** + * The session identity a fold holds right now. Null until it has an id, which + * every supported format writes in the opening lines of the transcript. + */ +export function accumulatorSessionIdentity( + accumulator: SessionAccumulator +): TranscriptSessionIdentity | null { + const sessionId = accumulator.sessionId.trim() + if (!sessionId) { + return null + } + return { + sessionId, + cwd: accumulator.cwd, + // The generated fallback is `finalizeSession`'s, not this one's: a title + // that is still absent mid-read is better said to be absent. + title: accumulator.title ?? accumulator.fallbackTitle, + createdAt: accumulator.createdAt, + updatedAt: accumulator.updatedAt + } +} + export function cloneSessionAccumulator(accumulator: SessionAccumulator): SessionAccumulator { return { ...accumulator, previewMessages: [...accumulator.previewMessages] } } @@ -77,6 +103,7 @@ export function accumulatorFoldResumeState( ): ResumableSessionParseState { return { consumeLine: (line) => consumeRecordLine(accumulator, line), + identity: () => accumulatorSessionIdentity(accumulator), clone: () => accumulatorFoldResumeState(cloneSessionAccumulator(accumulator), consumeRecordLine), touchFile: (file) => { diff --git a/src/main/ai-vault/session-scanner-codex-message-records.ts b/src/main/ai-vault/session-scanner-codex-message-records.ts index 5aa739275ff..a8a5c3c9d13 100644 --- a/src/main/ai-vault/session-scanner-codex-message-records.ts +++ b/src/main/ai-vault/session-scanner-codex-message-records.ts @@ -1,3 +1,7 @@ +import { + publishCodexResponseTool, + publishCodexCompletedTool +} from './session-scanner-codex-tool-records' import { normalizePromptField } from '../../shared/agent-status-field-normalization' import { addPreviewContent } from './session-scanner-accumulator' import type { SessionAccumulator } from './session-scanner-types' @@ -8,6 +12,10 @@ export function consumeCodexResponseMessage( payload: Record, timestamp: unknown ): boolean { + publishCodexResponseTool(accumulator, payload, timestamp) + if (payload.type !== 'message') { + return false + } accumulator.messageCount++ const role = payload.role === 'assistant' ? 'assistant' : payload.role === 'user' ? 'user' : 'unknown' @@ -24,6 +32,7 @@ export function consumeCodexCompletedMessage( payload: Record, timestamp: unknown ): boolean { + publishCodexCompletedTool(accumulator, payload, timestamp) const item = asRecord(payload.item) if (!item) { return false diff --git a/src/main/ai-vault/session-scanner-codex-parser.ts b/src/main/ai-vault/session-scanner-codex-parser.ts index 02a385400de..b3571d4a337 100644 --- a/src/main/ai-vault/session-scanner-codex-parser.ts +++ b/src/main/ai-vault/session-scanner-codex-parser.ts @@ -4,6 +4,7 @@ import type { AiVaultSession } from '../../shared/ai-vault-types' import { readCodexSessionIndexTitle } from './session-scanner-codex-title-index' import type { ExecutionHostId } from '../../shared/execution-host' import { + accumulatorSessionIdentity, cloneSessionAccumulator, createAccumulator, finalizeSession, @@ -153,19 +154,13 @@ function consumeCodexRecordLine(state: CodexSessionParseState, line: string): vo accumulator.title = metadataTitle state.titleSource = 'meta' } - const cwd = extractString(payload.cwd) - if (cwd) { - accumulator.cwd = cwd - } + accumulator.cwd = extractString(payload.cwd) ?? accumulator.cwd accumulator.branch = extractGitBranch(payload.git) ?? accumulator.branch return } if (record.type === 'turn_context' && payload) { - const cwd = extractString(payload.cwd) - if (cwd) { - accumulator.cwd = cwd - } + accumulator.cwd = extractString(payload.cwd) ?? accumulator.cwd const model = extractModel(payload) if (model) { accumulator.model = model @@ -177,7 +172,7 @@ function consumeCodexRecordLine(state: CodexSessionParseState, line: string): vo return } - if (record.type === 'response_item' && payload.type === 'message') { + if (record.type === 'response_item') { if (state.historyMode === 'paginated') { return } @@ -285,7 +280,10 @@ function codexResumeStateFromParseState( return { consumeLine: (line) => consumeCodexRecordLine(state, line), consumeLineBytes: (line) => { - const timelineOnlyRecord = readCodexTimelineOnlyRecord(line) + const timelineOnlyRecord = readCodexTimelineOnlyRecord( + line, + state.accumulator.messages.active && state.historyMode !== 'paginated' + ) if (timelineOnlyRecord) { updateTimeline(state.accumulator, timelineOnlyRecord.timestamp) } else { @@ -293,6 +291,7 @@ function codexResumeStateFromParseState( } }, shouldStop: () => state.rejectedWorkerSession, + identity: () => accumulatorSessionIdentity(state.accumulator), clone: () => codexResumeStateFromParseState(cloneCodexParseState(state), codexHome, titleReader), touchFile: (file) => { diff --git a/src/main/ai-vault/session-scanner-codex-record-fast-path.ts b/src/main/ai-vault/session-scanner-codex-record-fast-path.ts index 1c4322f89f6..852ec174e95 100644 --- a/src/main/ai-vault/session-scanner-codex-record-fast-path.ts +++ b/src/main/ai-vault/session-scanner-codex-record-fast-path.ts @@ -1,3 +1,5 @@ +import { CODEX_TOOL_RESPONSE_TYPES } from './session-scanner-codex-tool-records' + // Records below this size are decoded and parsed exactly: JSON.parse on a // kilobyte costs less than the risk of a prefix heuristic, and the scan cost // this path exists to remove is entirely in megabyte-scale records. @@ -22,7 +24,10 @@ const PARSED_EVENT_TYPES = new Set([ ]) /** Returns the timestamp only when the record cannot affect other visible session fields. */ -export function readCodexTimelineOnlyRecord(line: Buffer): { timestamp: string } | null { +export function readCodexTimelineOnlyRecord( + line: Buffer, + includeTools = false +): { timestamp: string } | null { if (line.length <= CODEX_RECORD_PREFIX_LIMIT) { return null } @@ -41,6 +46,13 @@ export function readCodexTimelineOnlyRecord(line: Buffer): { timestamp: string } if (!payloadType) { return null } + if ( + includeTools && + recordType === 'response_item' && + CODEX_TOOL_RESPONSE_TYPES.has(payloadType) + ) { + return null + } const parsedPayloadTypes = recordType === 'response_item' ? PARSED_RESPONSE_ITEM_TYPES : PARSED_EVENT_TYPES return parsedPayloadTypes.has(payloadType) ? null : { timestamp } diff --git a/src/main/ai-vault/session-scanner-codex-tool-records.test.ts b/src/main/ai-vault/session-scanner-codex-tool-records.test.ts new file mode 100644 index 00000000000..a5aa909bfa1 --- /dev/null +++ b/src/main/ai-vault/session-scanner-codex-tool-records.test.ts @@ -0,0 +1,125 @@ +import { expect, it } from 'vitest' +import { createCodexSessionResumeState } from './session-scanner-codex-parser' +import type { TranscriptMessage } from './session-transcript-consumers' +import { readCodexTimelineOnlyRecord } from './session-scanner-codex-record-fast-path' + +const timestamp = '2026-05-01T10:00:00.000Z' +const file = { + path: '/fixture/rollout.jsonl', + mtimeMs: Date.parse(timestamp), + modifiedAt: timestamp +} +const record = (type: string, payload: Record): Buffer => + Buffer.from(JSON.stringify({ timestamp, type, payload })) + +it.each(['function_call_output', 'custom_tool_call_output'])( + 'reads large %s records only when a consumer needs them', + (type) => { + const line = record('response_item', { type, output: 'outputonly '.repeat(300) }) + expect(readCodexTimelineOnlyRecord(line)).toEqual({ timestamp }) + expect(readCodexTimelineOnlyRecord(line, true)).toBeNull() + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!(line) + expect(messages).toEqual([{ role: 'tool', text: 'outputonly '.repeat(300), timestamp }]) + } +) + +it.each([false, true])( + 'uses one tool representation across append when paginated=%s', + async (paginated) => { + const messages: TranscriptMessage[] = [] + let state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + const consume = (type: string, payload: Record) => + state.consumeLineBytes!(record(type, payload)) + consume('session_meta', { id: 'session-1', history_mode: paginated ? 'paginated' : 'full' }) + consume('response_item', { type: 'message', role: 'user', content: 'promptonly' }) + consume('event_msg', { + type: 'item_completed', + item: { type: 'UserMessage', content: [{ type: 'text', text: 'promptonly' }] } + }) + consume('response_item', { + type: 'function_call', + name: 'shell', + arguments: '{"command":"commandonly"}' + }) + // The next scan resumes between the call and its output. + state = state.clone() + consume('response_item', { type: 'function_call_output', output: 'outputonly' }) + consume('event_msg', { + type: 'item_completed', + item: { type: 'CommandExecution', command: ['commandonly'], aggregated_output: 'outputonly' } + }) + expect(messages.filter((message) => message.text.includes('commandonly'))).toHaveLength(1) + expect(messages.filter((message) => message.text === 'outputonly')).toEqual([ + { role: 'tool', text: 'outputonly', timestamp } + ]) + expect(messages.filter((message) => message.role === 'user')).toHaveLength(1) + expect(await state.finalize(process.platform)).toMatchObject({ messageCount: 1 }) + } +) + +it.each([ + { type: 'add', content: '+ addedneedle' }, + { type: 'delete', content: '+ addedneedle' }, + { type: 'update', unified_diff: '+ addedneedle', move_path: null } +])('publishes paginated $type file changes', (change) => { + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!(record('session_meta', { id: 'session-1', history_mode: 'paginated' })) + state.consumeLineBytes!( + record('event_msg', { + type: 'item_completed', + item: { type: 'FileChange', changes: { 'src/changed.ts': change } } + }) + ) + expect(messages.map((message) => [message.role, message.text])).toEqual([ + ['tool', 'apply_patch: src/changed.ts'], + ['tool', '+ addedneedle'] + ]) +}) + +it('normalizes custom calls and structured results through the existing content reader', () => { + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!( + record('response_item', { type: 'custom_tool_call', name: 'apply_patch', input: 'patchneedle' }) + ) + state.consumeLineBytes!( + record('response_item', { + type: 'custom_tool_call_output', + output: { content: [{ type: 'text', text: 'resultneedle' }] } + }) + ) + expect(messages.map((message) => message.text)).toEqual([ + 'apply_patch: patchneedle', + 'resultneedle' + ]) +}) + +it('keeps local shell argv searchable', () => { + const messages: TranscriptMessage[] = [] + const state = createCodexSessionResumeState(file, null, { + active: true, + push: (message) => messages.push(message) + }) + state.consumeLineBytes!( + record('response_item', { + type: 'local_shell_call', + action: { type: 'exec', command: ['rg', 'argvneedle'] } + }) + ) + expect(messages.map((message) => message.text)).toEqual(['tool: rg argvneedle']) +}) diff --git a/src/main/ai-vault/session-scanner-codex-tool-records.ts b/src/main/ai-vault/session-scanner-codex-tool-records.ts new file mode 100644 index 00000000000..976d5eff36d --- /dev/null +++ b/src/main/ai-vault/session-scanner-codex-tool-records.ts @@ -0,0 +1,95 @@ +import { timestampIso } from './session-scanner-accumulator' +import { asRecord } from './session-scanner-record-value' +import type { SessionAccumulator } from './session-scanner-types' +import { transcriptMessagesFromContent } from './session-transcript-message-content' + +export const CODEX_TOOL_RESPONSE_TYPES = new Set([ + 'function_call', + 'local_shell_call', + 'custom_tool_call', + 'function_call_output', + 'custom_tool_call_output' +]) + +function publishToolContent( + accumulator: SessionAccumulator, + content: unknown, + timestamp: unknown +): void { + for (const message of transcriptMessagesFromContent('tool', content, timestampIso(timestamp))) { + accumulator.messages.push(message) + } +} + +export function publishCodexResponseTool( + accumulator: SessionAccumulator, + payload: Record, + timestamp: unknown +): void { + if (!accumulator.messages.active || !CODEX_TOOL_RESPONSE_TYPES.has(String(payload.type))) { + return + } + if (payload.type === 'function_call_output' || payload.type === 'custom_tool_call_output') { + const output = asRecord(payload.output) + publishToolContent( + accumulator, + [{ type: 'tool_result', content: output?.content ?? output?.output ?? payload.output }], + timestamp + ) + return + } + const input = payload.arguments ?? payload.input ?? payload.action + const action = asRecord(input) + const normalizedInput = + action && Array.isArray(action.command) + ? { ...action, command: action.command.filter((part) => typeof part === 'string').join(' ') } + : input + publishToolContent( + accumulator, + [ + { + type: 'tool_use', + name: payload.name ?? 'tool', + input: normalizedInput + } + ], + timestamp + ) +} + +export function publishCodexCompletedTool( + accumulator: SessionAccumulator, + payload: Record, + timestamp: unknown +): void { + if (!accumulator.messages.active) { + return + } + const item = asRecord(payload.item) + if (item?.type === 'CommandExecution' || item?.type === 'command_execution') { + const command = Array.isArray(item.command) + ? item.command.filter((part) => typeof part === 'string').join(' ') + : item.command + publishToolContent( + accumulator, + [ + { type: 'tool_use', name: 'shell', input: command }, + { type: 'tool_result', content: item.aggregated_output ?? item.aggregatedOutput } + ], + timestamp + ) + } else if (item?.type === 'FileChange' || item?.type === 'file_change') { + const changes = asRecord(item.changes) ?? {} + for (const [path, value] of Object.entries(changes)) { + const change = asRecord(value) + publishToolContent( + accumulator, + [ + { type: 'tool_use', name: 'apply_patch', input: { path } }, + { type: 'tool_result', content: change?.unified_diff ?? change?.content } + ], + timestamp + ) + } + } +} diff --git a/src/main/ai-vault/session-scanner-omp-subagent-transcripts.ts b/src/main/ai-vault/session-scanner-omp-subagent-transcripts.ts index 57cadebeee0..07f3646b537 100644 --- a/src/main/ai-vault/session-scanner-omp-subagent-transcripts.ts +++ b/src/main/ai-vault/session-scanner-omp-subagent-transcripts.ts @@ -101,6 +101,7 @@ export function withOmpSubagentTranscriptCount( ): ResumableSessionParseState { return { consumeLine: (line) => state.consumeLine(line), + identity: () => state.identity?.() ?? null, clone: () => withOmpSubagentTranscriptCount(state.clone(), transcriptFilePath), touchFile: (file) => state.touchFile(file), finalize: async (platform, options) => { diff --git a/src/main/ai-vault/session-scanner-parse-cache.ts b/src/main/ai-vault/session-scanner-parse-cache.ts index 46fb4754a32..6ebb1a76a2b 100644 --- a/src/main/ai-vault/session-scanner-parse-cache.ts +++ b/src/main/ai-vault/session-scanner-parse-cache.ts @@ -26,6 +26,7 @@ import { import { readResumableTranscript, readWholeTranscript, + requestWholeTranscriptRead, type TranscriptReadStats } from './session-transcript-reader' @@ -104,29 +105,67 @@ export function createSessionParseStats(): SessionParseStats { export async function parseAgentSessionFileCached( candidate: SessionFileCandidate, platform: NodeJS.Platform, - stats?: SessionParseStats + stats?: SessionParseStats, + requireRead?: SessionParseReadRequirement ): Promise { // The whole lookup-read-store sequence runs in the lane: a concurrent parse of // the same path shares this entry's resume point and its message channel. return inSessionParseFileLane(candidate.file.path, () => - parseCachedInLane(candidate, platform, stats) + parseCachedInLane(candidate, platform, stats, requireRead) + ) +} + +/** + * What a caller other than the session list needs out of this parse. + * + * `any`: some bytes must be read. A cursor already at the file's current stat + * is dropped so the reader opens it; one that is merely behind is left alone, + * because an append is a read. + * + * `whole`: the file must be re-read from zero, for a consumer whose own cursor + * covers a span this one does not. + * + * Why it is a parameter and not two calls around this one: the decision reads + * cache state and then changes it, so outside the per-path lane an overlapping + * list parse can store its entry in between and the forced read silently + * degrades to a reuse. + */ +export type SessionParseReadRequirement = 'any' | 'whole' + +/** + * True when this cursor already sits at the transcript's current stat, so a + * parse would reuse the cached fold and read no bytes at all. + */ +function sessionParseCacheCoversTranscript( + candidate: SessionFileCandidate, + platform: NodeJS.Platform +): boolean { + const { file } = candidate + const entry = getSessionParseCacheEntry(file.path) + return ( + entry !== undefined && + entry.platform === platform && + entry.mtimeMs === file.mtimeMs && + (entry.sizeBytes === null || file.sizeBytes === undefined || entry.sizeBytes === file.sizeBytes) ) } async function parseCachedInLane( candidate: SessionFileCandidate, platform: NodeJS.Platform, - stats?: SessionParseStats + stats?: SessionParseStats, + requireRead?: SessionParseReadRequirement ): Promise { const { file } = candidate + if ( + requireRead === 'whole' || + (requireRead === 'any' && sessionParseCacheCoversTranscript(candidate, platform)) + ) { + requestWholeTranscriptRead(file.path) + } const entry = getSessionParseCacheEntry(file.path) - const transcriptUnchanged = - entry !== undefined && - entry.platform === platform && - entry.mtimeMs === file.mtimeMs && - (entry.sizeBytes === null || file.sizeBytes === undefined || entry.sizeBytes === file.sizeBytes) - if (transcriptUnchanged) { + if (entry !== undefined && sessionParseCacheCoversTranscript(candidate, platform)) { if (sidecarUnchanged(entry.sidecar, file.sidecar)) { return reuseCachedSession(candidate, entry, stats) } diff --git a/src/main/ai-vault/session-scanner-primary-parsers.ts b/src/main/ai-vault/session-scanner-primary-parsers.ts index 9783f8a0520..62149920b5a 100644 --- a/src/main/ai-vault/session-scanner-primary-parsers.ts +++ b/src/main/ai-vault/session-scanner-primary-parsers.ts @@ -12,6 +12,7 @@ import type { } from './session-scanner-types' import type { TranscriptMessageSink } from './session-transcript-consumers' import { + accumulatorSessionIdentity, addPreviewContent, createAccumulator, finalizeSession, @@ -205,6 +206,7 @@ function claudeResumeStateFromParseState( ): ResumableSessionParseState { return { consumeLine: (line) => consumeClaudeSessionLine(state, line), + identity: () => accumulatorSessionIdentity(state.accumulator), clone: () => claudeResumeStateFromParseState(cloneClaudeSessionParseState(state)), touchFile: (file) => { state.accumulator.modifiedAt = file.modifiedAt diff --git a/src/main/ai-vault/session-scanner-text-normalization.ts b/src/main/ai-vault/session-scanner-text-normalization.ts index 99f73a93d02..50fc54ab7aa 100644 --- a/src/main/ai-vault/session-scanner-text-normalization.ts +++ b/src/main/ai-vault/session-scanner-text-normalization.ts @@ -1,3 +1,7 @@ +import { sliceAtCodeUnitLimit } from '../../shared/surrogate-safe-text-slice' + +export { sliceAtCodeUnitLimit } + const SESSION_TITLE_TEXT_LIMIT = 96 const SESSION_PREVIEW_TEXT_LIMIT = 220 const ELLIPSIS = '...' @@ -42,15 +46,6 @@ export function normalizePreviewText(value: string): string | null { return finalizeNormalizedText(normalizeStringText(value, SESSION_PREVIEW_TEXT_LIMIT)) } -/** Cut to `limit` UTF-16 code units without splitting a trailing surrogate pair. */ -export function sliceAtCodeUnitLimit(value: string, limit: number): string { - if (value.length <= limit) { - return value - } - const end = limit > 0 && isHighSurrogate(value.charCodeAt(limit - 1)) ? limit - 1 : limit - return value.slice(0, end) -} - function normalizeContentText(value: unknown, limit: number): string | null { if (typeof value === 'string') { return finalizeNormalizedText(normalizeStringText(value, limit)) diff --git a/src/main/ai-vault/session-scanner-types.ts b/src/main/ai-vault/session-scanner-types.ts index b1d480aa944..f4b60270544 100644 --- a/src/main/ai-vault/session-scanner-types.ts +++ b/src/main/ai-vault/session-scanner-types.ts @@ -5,7 +5,10 @@ import type { AiVaultSessionPreviewMessage } from '../../shared/ai-vault-types' import type { ExecutionHostId } from '../../shared/execution-host' -import type { TranscriptMessageSink } from './session-transcript-consumers' +import type { + TranscriptMessageSink, + TranscriptSessionIdentity +} from './session-transcript-consumers' import type { SessionSidecarObservation } from './session-sidecar-stat' export type AiVaultScanOptions = { @@ -103,6 +106,9 @@ export type ResumableSessionParseState = { consumeLineBytes?(line: Buffer): void // Lets a parser terminate an excluded transcript without draining the file. shouldStop?(): boolean + // What the fold knows about the session right now, for a consumer that has to + // commit before the read ends (see TranscriptSessionIdentity). + identity?(): TranscriptSessionIdentity | null clone(): ResumableSessionParseState // Refresh per-scan file metadata (mtime display string) without re-parsing. touchFile(file: FileWithMtime): void diff --git a/src/main/ai-vault/session-transcript-consumers.ts b/src/main/ai-vault/session-transcript-consumers.ts index 6707b298586..7aff8dcf87b 100644 --- a/src/main/ai-vault/session-transcript-consumers.ts +++ b/src/main/ai-vault/session-transcript-consumers.ts @@ -27,12 +27,34 @@ export const NO_TRANSCRIPT_MESSAGES: TranscriptMessageSink = { push: () => undefined } +/** + * What a parser has decoded about the session so far, mid-read. + * + * Provisional by construction: it is read before the file ends, so a title can + * still change and a timestamp can still move. Every field the transcript + * formats put in their opening lines, which is what a consumer that has to + * commit before the read finishes needs to name what it is holding. + */ +export type TranscriptSessionIdentity = { + sessionId: string + cwd: string | null + title: string | null + createdAt: string | null + updatedAt: string | null +} + export type TranscriptReadStart = { candidate: SessionFileCandidate /** `replace`: the whole file is being re-read; `append`: a resumed read. */ mode: 'replace' | 'append' /** Byte offset the messages of this read continue from. */ previousByteOffset: number + /** + * The session identity decoded so far, or null before the parser has an id. + * Called during the read, never here: nothing is decoded yet when a read + * begins. Absent when the read has no resumable parse state to ask. + */ + identity?: () => TranscriptSessionIdentity | null } export type TranscriptReadOutcome = { diff --git a/src/main/ai-vault/session-transcript-reader.ts b/src/main/ai-vault/session-transcript-reader.ts index 228b0c832a3..3fc0ddcc146 100644 --- a/src/main/ai-vault/session-transcript-reader.ts +++ b/src/main/ai-vault/session-transcript-reader.ts @@ -3,7 +3,10 @@ import type { AiVaultSession } from '../../shared/ai-vault-types' import { parseAgentSessionFile, parserPublishesMessages } from './session-scanner-agent-parser' import { consumeCompleteJsonlLines } from './session-scanner-jsonl-reader' import type { ResumableSessionParseState, SessionFileCandidate } from './session-scanner-types' -import type { SessionParseResumePoint } from './session-parse-cache-store' +import { + invalidateSessionParseCacheEntry, + type SessionParseResumePoint +} from './session-parse-cache-store' import { TranscriptMessageChannel } from './session-transcript-channel' const NEWLINE_BYTE = 0x0a @@ -29,6 +32,24 @@ export type ResumableTranscriptRead = { resume: SessionParseResumePoint } +/** + * Ask for the next read of `path` to be a whole-file `replace`. + * + * Why this lives here: a consumer never chooses its own mode. The reader picks + * `append` or `replace` from the resume point the session list left behind, so a + * consumer that declined an append has no way to get the span it missed — with + * an empty index and a warm parse cache, every read arrives as `append`, every + * one is declined, and nothing is ever indexed. Dropping the resume point is the + * one lever that changes the next read's mode, and only the reader's own cache + * owns it. + * + * The cost is a re-parse for the session list too. That is the honest price of a + * second consumer being behind, and it is paid once per file rather than per scan. + */ +export function requestWholeTranscriptRead(path: string): void { + invalidateSessionParseCacheEntry(path) +} + /** * Read an append-only transcript, resuming from `resume` when the file only * grew and the recorded offset still sits on a line boundary. Anything else @@ -70,7 +91,10 @@ export async function readResumableTranscript(args: { channel.beginRead({ candidate: args.candidate, mode: canResume ? 'append' : 'replace', - previousByteOffset: startOffset + previousByteOffset: startOffset, + // Read by a consumer during the read, not here: the fold has decoded + // nothing yet at this point of a whole-file read. + identity: () => state.identity?.() ?? null }) try { const readResult = await consumeCompleteJsonlLines({ diff --git a/src/main/browser/browser-manager-auth-user-agent.test.ts b/src/main/browser/browser-manager-auth-user-agent.test.ts index eb0bd75b354..71e128a9c72 100644 --- a/src/main/browser/browser-manager-auth-user-agent.test.ts +++ b/src/main/browser/browser-manager-auth-user-agent.test.ts @@ -51,7 +51,7 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_ELECTRON_UA + GUEST_CLEAN_UA } from './browser-manager-viewport-test-fixtures' const { @@ -197,9 +197,8 @@ describe('browserManager', () => { // Why: popup child windows get attachGuestPolicies but are never entered into tabIdByWebContentsId, // so a direct lookup of the UA mode misses the native opt-out. That is worse than doing nothing — - // native sessions never install the header-level Firefox switch, so the popup would send the - // Electron UA on the wire while navigator.userAgent claimed Firefox. Google sign-in popups are a - // first-class surface. + // native sessions skip setupGoogleAuthUserAgentOverride, so the popup would send the raw Electron UA on the + // wire while navigator.userAgent claimed Firefox. Google sign-in popups are a first-class surface. it('leaves the UA untouched on auth hosts for a popup owned by a native-UA profile', () => { const ownerGuest = { id: 415, @@ -544,7 +543,7 @@ describe('browserManager', () => { ) expect(uaWrites.length).toBeGreaterThan(0) for (const [, params] of uaWrites) { - expect((params as { userAgent: string }).userAgent).toBe(GUEST_ELECTRON_UA) + expect((params as { userAgent: string }).userAgent).toBe(GUEST_CLEAN_UA) } }) }) diff --git a/src/main/browser/browser-manager-navigation.ts b/src/main/browser/browser-manager-navigation.ts index 5cb46ccb682..c11626fd516 100644 --- a/src/main/browser/browser-manager-navigation.ts +++ b/src/main/browser/browser-manager-navigation.ts @@ -1,4 +1,5 @@ import { openPopupWithOriginBar, type PopupChildWindowOptions } from './popup-origin-bar-window' +import { cleanElectronUserAgent } from './browser-session-ua' import { getBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { googleAuthUserAgent, isGoogleAuthUrl } from './browser-google-auth-ua' import { buildViewportUserAgentOverride } from './browser-viewport-user-agent' @@ -14,7 +15,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // not the request header, so the header-level Firefox switch in setupGoogleAuthUserAgentOverride // must be matched here per navigation or the two layers disagree — itself a bot tell. // Restores the session's base identity off the auth hosts. Native-UA profiles opt out - // of the Firefox switch, so they keep their untouched identity everywhere. + // of the whole clean-UA path, so they keep their untouched identity everywhere. protected applyGoogleAuthUserAgent( guest: Electron.WebContents, url: string, @@ -23,8 +24,8 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility const browserPageId = this.tabIdByWebContentsId.get(guest.id) // Why: popup child windows get these policies but are never in tabIdByWebContentsId, so a direct // lookup misses the native-UA opt-out and would hand a native profile's popup the Firefox UA. - // That is worse than doing nothing: native sessions never install the header-level Firefox - // switch, so the popup would send the Electron UA on the wire while navigator.userAgent claims Firefox. + // That is worse than doing nothing: native sessions skip setupGoogleAuthUserAgentOverride, so + // the popup would send the raw Electron UA on the wire while navigator.userAgent claims Firefox. const ownerTabId = this.resolveBrowserTabIdForGuestWebContentsId(guest.id) // Session state is authoritative before renderer registration and after a native profile imports a source UA. const mode = @@ -61,8 +62,9 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility if (this.canOverrideUserAgentOverCdp(guest)) { authOverrideIssuedOverCdp = true // Why: go through the viewport builder rather than writing nextUa raw, so both CDP writers - // resolve one identity for this URL — Firefox on auth hosts, the session's base identity - // off them, any mobile preset preserved. + // resolve one identity for this URL — Firefox on auth hosts, the profile's clean base off + // them, any mobile preset preserved. Writing the session UA directly would put the + // unlaundered Electron token back on the wire. void this.applyAuthUserAgentOverrideOverCdp( guest, (browserPageId ? this.viewportUaOverrideMobileByTabId.get(browserPageId) : undefined) ?? @@ -186,7 +188,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: Emulation.setUserAgentOverride is set once and stands across every later navigation, // outranking setUserAgent for navigator.userAgent. A viewport preset applied before reaching an - // auth host would otherwise pin navigator.userAgent to the session's preset UA while the + // auth host would otherwise pin navigator.userAgent to the Chrome-shaped preset UA while the // request header says Firefox — the two-layer disagreement this scope exists to remove. protected reapplyViewportUserAgentOverride( guest: Electron.WebContents, @@ -220,7 +222,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: the session UA is the profile's stable base identity. guest.getUserAgent() is not: // applyGoogleAuthUserAgent leaves it pinned to the Firefox auth UA once a guest switches to // the CDP override, so reading it back here would republish that identity on ordinary hosts. - baseUserAgent: baseUserAgent ?? guest.session.getUserAgent() + baseUserAgent: cleanElectronUserAgent(baseUserAgent ?? guest.session.getUserAgent()) }) ) } diff --git a/src/main/browser/browser-manager-viewport-override.test.ts b/src/main/browser/browser-manager-viewport-override.test.ts index 0ffc3c2a6e1..0228d8f9ed8 100644 --- a/src/main/browser/browser-manager-viewport-override.test.ts +++ b/src/main/browser/browser-manager-viewport-override.test.ts @@ -49,6 +49,7 @@ import { import { createViewportGuestFactory, flushViewportOps, + GUEST_CLEAN_UA, GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' @@ -206,7 +207,7 @@ describe('browserManager', () => { mobile: false }) expect(debuggerSendCommand).toHaveBeenLastCalledWith('Emulation.setUserAgentOverride', { - userAgent: GUEST_ELECTRON_UA + userAgent: GUEST_CLEAN_UA }) // Navigating to the auth host must move the standing override to the Firefox identity. @@ -217,11 +218,11 @@ describe('browserManager', () => { userAgent: googleAuthUserAgent() }) - // Leaving the auth host restores the session's own preset UA. + // Leaving the auth host restores the clean Chrome-shaped preset UA. debuggerSendCommand.mockClear() willRedirect({ preventDefault: vi.fn() }, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) }) // Why: not an ordering race — debugger.sendCommand dispatches in call order over one channel, so @@ -240,9 +241,9 @@ describe('browserManager', () => { } // Why mobile: on the desktop branch the break is masked by coincidence — applyGoogleAuthUserAgent - // has already switched the WebContents UA to Firefox, so the stale-URL desktop path happens to - // emit Firefox anyway. The mobile branch derives a Chrome-shaped iPhone UA from the session base - // and exposes the real defect. + // has already switched the WebContents UA to Firefox, and cleanElectronUserAgent passes a Firefox + // UA through untouched, so the stale-URL desktop path happens to emit Firefox anyway. The mobile + // branch derives a Chrome-shaped iPhone UA from that same base and exposes the real defect. it('does not leave the Chrome preset UA standing when a mobile preset lands mid-navigation onto an auth host', async () => { const { guest, debuggerSendCommand } = makeGuest(4251, 'https://example.com/') // Hold the preset's first CDP command open so the navigation lands inside its await window. @@ -331,7 +332,7 @@ describe('browserManager', () => { // Without the fix the resuming preset re-reads getURL() as the auth host and clobbers the // navigation's correct write, stranding the Firefox UA on a non-auth page. - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) }) it('falls back to the committed URL once a navigation commits or fails', async () => { @@ -377,7 +378,7 @@ describe('browserManager', () => { await flushViewportOps() expect(guest.setUserAgent).toHaveBeenLastCalledWith(GUEST_ELECTRON_UA) - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) // A later preset must also resolve the committed, non-auth URL. debuggerSendCommand.mockClear() @@ -456,7 +457,7 @@ describe('browserManager', () => { expect(guest.setUserAgent).not.toHaveBeenCalled() expect(debuggerSendCommand).not.toHaveBeenCalledWith( 'Emulation.setUserAgentOverride', - expect.objectContaining({ userAgent: GUEST_ELECTRON_UA }) + expect.objectContaining({ userAgent: GUEST_CLEAN_UA }) ) }) @@ -516,7 +517,7 @@ describe('browserManager', () => { didFailLoad(null, -3, 'Aborted', 'https://accounts.google.com/redirected', true) await flushViewportOps() expect(guest.setUserAgent).not.toHaveBeenCalled() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) }) it('preserves the auth identity when a viewport preset is cleared after a redirect', async () => { @@ -591,7 +592,7 @@ describe('browserManager', () => { didStartNavigation(null, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) }) it('reapplies a preset when navigation starts during its final UA write', async () => { diff --git a/src/main/browser/browser-manager-viewport-test-fixtures.ts b/src/main/browser/browser-manager-viewport-test-fixtures.ts index 076524ce5d0..b440a768b71 100644 --- a/src/main/browser/browser-manager-viewport-test-fixtures.ts +++ b/src/main/browser/browser-manager-viewport-test-fixtures.ts @@ -3,6 +3,8 @@ import type { BrowserManagerMocks } from './browser-manager-test-harness' export const GUEST_ELECTRON_UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/134.0.0.0 Electron/30.0.0 Safari/537.36' +export const GUEST_CLEAN_UA = + 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/134.0.0.0 Safari/537.36' // Why: viewport UA writes are queued on the per-tab chain, so draining it takes more than one // microtask hop; loop until the chain is empty rather than guessing a tick count. diff --git a/src/main/browser/browser-session-partition-policies.test.ts b/src/main/browser/browser-session-partition-policies.test.ts index 952c199536c..b092cc8344b 100644 --- a/src/main/browser/browser-session-partition-policies.test.ts +++ b/src/main/browser/browser-session-partition-policies.test.ts @@ -82,6 +82,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: async () => false })) vi.mock('./browser-session-ua', () => ({ + cleanElectronUserAgent: (userAgent: string) => userAgent, setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ diff --git a/src/main/browser/browser-session-partition-policies.ts b/src/main/browser/browser-session-partition-policies.ts index ec3c68fb45e..b7022185174 100644 --- a/src/main/browser/browser-session-partition-policies.ts +++ b/src/main/browser/browser-session-partition-policies.ts @@ -9,7 +9,7 @@ import { } from './browser-session-proxy' import { hasSystemMediaAccess, requestSystemMediaAccess } from './browser-media-access' import { isAutoGrantedBrowserSessionPermission } from './browser-session-permission-policy' -import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' +import { cleanElectronUserAgent, setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { allowsBrowserWebAuthnPermission, @@ -92,7 +92,9 @@ export function installBrowserSessionPartitionPolicies( } browserManager.installCertificateRequestGuard(sess) - if (profile.userAgentMode !== 'native') { + if (profile.userAgentMode !== 'native' && typeof sess.getUserAgent === 'function') { + const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) + sess.setUserAgent(cleanUA) setupGoogleAuthUserAgentOverride(sess) } if (options?.permissions === 'deny') { @@ -189,6 +191,10 @@ export function applyBrowserSessionUserAgentModes(profiles: BrowserSessionProfil if (profile.userAgentMode === 'native') { continue } + + // Why: imported sessions need the same Chrome-shaped identity after app restart. + const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) + sess.setUserAgent(cleanUA) setupGoogleAuthUserAgentOverride(sess) } catch { /* session not available yet (e.g. unit tests or pre-ready) */ diff --git a/src/main/browser/browser-session-partition-proxy-install.test.ts b/src/main/browser/browser-session-partition-proxy-install.test.ts index 4beeaab04c9..d0eee031ea0 100644 --- a/src/main/browser/browser-session-partition-proxy-install.test.ts +++ b/src/main/browser/browser-session-partition-proxy-install.test.ts @@ -44,6 +44,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: vi.fn(async () => false) })) vi.mock('./browser-session-ua', () => ({ + cleanElectronUserAgent: vi.fn((ua: string) => ua), setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ diff --git a/src/main/browser/browser-session-registry.persistence.test.ts b/src/main/browser/browser-session-registry.persistence.test.ts index 67653c0c76c..b6495d3d1e1 100644 --- a/src/main/browser/browser-session-registry.persistence.test.ts +++ b/src/main/browser/browser-session-registry.persistence.test.ts @@ -2,6 +2,10 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' const USER_DATA = '/user-data' const META_PATH = `${USER_DATA}/browser-session-meta.json` +const RAW_ELECTRON_USER_AGENT = + 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Orca/1.4.198 Chrome/150.0.7871.224 Electron/43.4.1 Safari/537.36' +const CLEAN_USER_AGENT = + 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/150.0.7871.224 Safari/537.36' type FsState = { files: Map @@ -27,6 +31,7 @@ function installModuleMocks( copyFailures = new Set() ): { sessionFromPartitionMock: ReturnType + cleanElectronUserAgentMock: ReturnType setupGoogleAuthUserAgentOverrideMock: ReturnType browserManagerHandleGuestWillDownloadMock: ReturnType browserManagerNotifyPermissionDeniedMock: ReturnType @@ -35,8 +40,7 @@ function installModuleMocks( const sessionFromPartitionMock = vi.fn((partition: string) => ({ partition, setUserAgent: vi.fn(), - getUserAgent: vi.fn(() => 'Mozilla/5.0 Electron/31 Orca'), - webRequest: { onBeforeSendHeaders: vi.fn() }, + getUserAgent: vi.fn(() => RAW_ELECTRON_USER_AGENT), setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -46,6 +50,7 @@ function installModuleMocks( clearStorageData: vi.fn().mockResolvedValue(undefined), clearCache: vi.fn().mockResolvedValue(undefined) })) + const cleanElectronUserAgentMock = vi.fn(() => CLEAN_USER_AGENT) const setupGoogleAuthUserAgentOverrideMock = vi.fn() const browserManagerHandleGuestWillDownloadMock = vi.fn() const browserManagerNotifyPermissionDeniedMock = vi.fn() @@ -120,6 +125,7 @@ function installModuleMocks( requestSystemMediaAccess: requestSystemMediaAccessMock })) vi.doMock('./browser-session-ua', () => ({ + cleanElectronUserAgent: cleanElectronUserAgentMock, setupGoogleAuthUserAgentOverride: setupGoogleAuthUserAgentOverrideMock })) // This suite models replay with an in-memory filesystem. The real file-backed SQLite merge has @@ -149,6 +155,7 @@ function installModuleMocks( return { sessionFromPartitionMock, + cleanElectronUserAgentMock, setupGoogleAuthUserAgentOverrideMock, browserManagerHandleGuestWillDownloadMock, browserManagerNotifyPermissionDeniedMock, @@ -234,24 +241,30 @@ describe('BrowserSessionRegistry persistence', () => { }) }) - // Why: the stock Electron UA is what clears Cloudflare; only the Google auth switch installs. - it('keeps the stock UA and installs the Google auth switch for profiles without an override', async () => { + it('keeps UA cleaning as the fallback for profiles without an override', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = - installModuleMocks(fsState) + const { + sessionFromPartitionMock, + cleanElectronUserAgentMock, + setupGoogleAuthUserAgentOverrideMock + } = installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Default identity') const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value - expect(profileSession.setUserAgent).not.toHaveBeenCalled() + expect(cleanElectronUserAgentMock).toHaveBeenCalledWith(RAW_ELECTRON_USER_AGENT) + expect(profileSession.setUserAgent).toHaveBeenCalledWith(CLEAN_USER_AGENT) expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalledWith(profileSession) }) it('leaves UA and client hints untouched for native-mode profiles', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = - installModuleMocks(fsState) + const { + sessionFromPartitionMock, + cleanElectronUserAgentMock, + setupGoogleAuthUserAgentOverrideMock + } = installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Google', { userAgentMode: 'native' }) @@ -259,6 +272,7 @@ describe('BrowserSessionRegistry persistence', () => { const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value const { getBrowserSessionUserAgentMode } = await import('./browser-session-user-agent-mode') expect(profileSession.setUserAgent).not.toHaveBeenCalled() + expect(cleanElectronUserAgentMock).not.toHaveBeenCalled() expect(setupGoogleAuthUserAgentOverrideMock).not.toHaveBeenCalled() expect(getBrowserSessionUserAgentMode(profileSession as never)).toBe('native') }) @@ -382,7 +396,7 @@ describe('BrowserSessionRegistry persistence', () => { // Why: imports before Aug 2026 persisted a synthesized source-browser UA // (fork imports as a broken Chrome/1.x, Chrome imports as a valid version). // Neither may ever be applied again — the engine-derived UA is the only one. - it('ignores legacy persisted UAs, valid or broken, and keeps the engine UA', async () => { + it('ignores legacy persisted UAs, valid or broken, and applies the engine UA', async () => { const importedPartition = 'persist:orca-browser-session-11111111-1111-4111-8111-111111111111' const brokenUa = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/1.158.1 Safari/537.36' @@ -408,8 +422,11 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = - installModuleMocks(fsState) + const { + sessionFromPartitionMock, + cleanElectronUserAgentMock, + setupGoogleAuthUserAgentOverrideMock + } = installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -417,8 +434,15 @@ describe('BrowserSessionRegistry persistence', () => { const appliedUas = sessionFromPartitionMock.mock.results.flatMap((r) => r.value.setUserAgent.mock.calls.map((c: unknown[]) => c[0]) ) - // Why: no persisted UA is ever written back; every profile keeps the engine's stock UA. - expect(appliedUas).toEqual([]) + expect(appliedUas).not.toContain(brokenUa) + expect(appliedUas).not.toContain(validUa) + // Why: every non-native profile falls to Orca's own cleaned engine UA. + expect(appliedUas.length).toBeGreaterThan(0) + expect(appliedUas.every((ua) => ua === CLEAN_USER_AGENT)).toBe(true) + expect(cleanElectronUserAgentMock).toHaveBeenCalled() + expect( + cleanElectronUserAgentMock.mock.calls.every(([ua]) => ua === RAW_ELECTRON_USER_AGENT) + ).toBe(true) expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalled() }) diff --git a/src/main/browser/browser-session-registry.test.ts b/src/main/browser/browser-session-registry.test.ts index ae81ffda4e2..5c628723777 100644 --- a/src/main/browser/browser-session-registry.test.ts +++ b/src/main/browser/browser-session-registry.test.ts @@ -54,7 +54,6 @@ describe('BrowserSessionRegistry', () => { askForMediaAccessMock.mockResolvedValue(true) getMediaAccessStatusMock.mockReturnValue('granted') sessionFromPartitionMock.mockReturnValue({ - webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -530,9 +529,6 @@ describe('BrowserSessionRegistry', () => { }) describe('setupGoogleAuthUserAgentOverride', () => { - const STOCK_UA = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/147.0.6890.3 Electron/43.0.0 Safari/537.36' - function install(): (details: unknown, callback: ReturnType) => void { const onBeforeSendHeaders = vi.fn() setupGoogleAuthUserAgentOverride({ webRequest: { onBeforeSendHeaders } } as never) @@ -543,53 +539,54 @@ describe('BrowserSessionRegistry', () => { return onBeforeSendHeaders.mock.calls[0][1] } - // Why: the Electron token is what clears Cloudflare Turnstile; a Chrome-shaped UA with no - // client hints is what it rejects, so ordinary hosts must see the session's UA untouched. - it('leaves the stock Electron UA and its client hints alone off the auth hosts', () => { - const listener = install() + it('leaves ordinary-host identity headers untouched', () => { const callback = vi.fn() - listener( + install()( { - url: 'https://example.com/api', - requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', Cookie: 'abc=123' } + url: 'https://example.com/', + requestHeaders: { + 'User-Agent': 'Mozilla/5.0 Chrome/150.0.0.0 Safari/537.36', + 'sec-ch-ua': 'browser-owned', + Cookie: 'abc=123' + } }, callback ) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toBe(STOCK_UA) - expect(modified['sec-ch-ua']).toBe('old') - expect(modified.Cookie).toBe('abc=123') + + expect(callback.mock.calls[0][0].requestHeaders).toEqual({ + 'User-Agent': 'Mozilla/5.0 Chrome/150.0.0.0 Safari/537.36', + 'sec-ch-ua': 'browser-owned', + Cookie: 'abc=123' + }) }) it('presents a Firefox UA and strips client hints on Google auth hosts', () => { - const listener = install() const callback = vi.fn() - listener( + install()( { url: 'https://accounts.google.com/v3/signin/identifier', requestHeaders: { - 'User-Agent': STOCK_UA, + 'User-Agent': 'Chrome/147', 'sec-ch-ua': 'old', - 'sec-ch-ua-full-version-list': 'old', - 'sec-ch-ua-platform': '"macOS"' + 'SEC-CH-UA-Full-Version-List': 'old', + 'sec-ch-ua-platform': '"macOS"', + Accept: 'text/html' } }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toBe(googleAuthUserAgent()) + expect(modified['User-Agent']).toMatch(/Firefox\/\d/) expect(modified['User-Agent']).not.toContain('Chrome') - expect(modified['sec-ch-ua']).toBeUndefined() - expect(modified['sec-ch-ua-full-version-list']).toBeUndefined() - expect(modified['sec-ch-ua-platform']).toBeUndefined() + expect(Object.keys(modified).some((key) => key.toLowerCase().startsWith('sec-ch-ua'))).toBe( + false + ) + expect(modified.Accept).toBe('text/html') }) it('strips client hints on a cross-host request that carries the Firefox auth UA', () => { - const listener = install() const callback = vi.fn() - // Subresource/XHR to a non-auth Google host while the auth document is on - // screen: the WebContents Firefox UA leaks onto the request header. - listener( + install()( { url: 'https://play.google.com/log', requestHeaders: { @@ -611,19 +608,19 @@ describe('BrowserSessionRegistry', () => { expect(modified['sec-ch-ua-mobile']).toBeUndefined() }) - it('keeps the session identity on Google app subdomains (not auth hosts)', () => { - const listener = install() + it('keeps the session identity on Google app subdomains', () => { const callback = vi.fn() - listener( + install()( { url: 'https://myaccount.google.com/', - requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old' } + requestHeaders: { 'User-Agent': 'Chrome/150', 'sec-ch-ua': 'browser-owned' } }, callback ) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toBe(STOCK_UA) - expect(modified['sec-ch-ua']).toBe('old') + expect(callback.mock.calls[0][0].requestHeaders).toEqual({ + 'User-Agent': 'Chrome/150', + 'sec-ch-ua': 'browser-owned' + }) }) }) }) diff --git a/src/main/browser/browser-session-ua-wire-identity.electron.test.ts b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts index 4e0719b2745..9fc1c760643 100644 --- a/src/main/browser/browser-session-ua-wire-identity.electron.test.ts +++ b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts @@ -5,12 +5,19 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterAll, describe, expect, it } from 'vitest' import { build as buildVite } from 'vite' +import { + LOCAL_HTTPS_TEST_CERTIFICATE, + LOCAL_HTTPS_TEST_PRIVATE_KEY +} from './browser-local-https-test-certificate' -// Why this runs a real Electron: Cloudflare Turnstile rejects a Chrome-shaped UA that ships no -// client hints (error 600010) and clears a declared Electron client. The header layer is the -// only place that identity can be proven, and the vm-based unit tests cannot see Chromium's -// header emission at all. Every partition must therefore keep the stock Electron UA on the wire -// for ordinary hosts and present the Firefox identity on Google's sign-in hosts only. +// Why this runs a real Electron: sites that hold a transplanted session re-check the browser +// identity that minted it, and an `Orca/x.y.z … Electron/x.y.z` UA is not one any browser sends — +// LinkedIn and x.com revoked live sessions over it (STA-7147). The header layer is the only place +// that identity can be proven, and the vm-based unit tests cannot see Chromium's header emission +// at all. Every clean-mode partition must therefore strip the Electron and app tokens on the +// wire for ordinary hosts and present the Firefox identity on Google's sign-in hosts only. This +// focused revocation fix does not claim full Chrome fingerprint parity; native mode remains the +// fallback for sites that reject the cleaned identity, including some Turnstile deployments. const electronBinary = createRequire(import.meta.url)('electron') as string const fixtureRoots: string[] = [] @@ -27,12 +34,24 @@ const FIXTURE_LAUNCH_ATTEMPTS = 2 type CapturedRequest = { url: string userAgent: string | null - clientHints: string[] + clientHints: Record +} + +type UserAgentBrand = { + brand: string + version: string +} + +type NavigatorUserAgentData = { + brands: UserAgentBrand[] + highEntropy: { fullVersionList?: UserAgentBrand[] } } type FixtureResult = { + rawUserAgent: string sessionUserAgent: string navigatorUserAgent: string + navigatorUserAgentData: NavigatorUserAgentData | null requests: CapturedRequest[] } @@ -47,9 +66,14 @@ function neverReachedElectronReady(fixtureResult: string): boolean { function buildFixtureMain(modulePath: string, resultPath: string): string { return ` const { app, BrowserWindow, session } = require('electron') +const { createServer } = require('node:https') const { writeFileSync } = require('node:fs') -const { setupGoogleAuthUserAgentOverride } = require(${JSON.stringify(modulePath)}) +const { cleanElectronUserAgent, setupGoogleAuthUserAgentOverride } = require(${JSON.stringify(modulePath)}) const resultPath = ${JSON.stringify(resultPath)} +// Why: production's UA carries an app token ("Orca/1.4.198") between the engine comment and +// Chrome/, and an unnamed fixture emits none — which would leave half of cleanElectronUserAgent +// unexercised while the test still passed. +app.setName('OrcaWireIdentityFixture') let currentStep = 'starting' const mark = (step) => { currentStep = step @@ -65,38 +89,73 @@ async function run() { mark('ready') const partition = 'persist:wire-identity-test' const sess = session.fromPartition(partition) + // Mirrors installBrowserSessionPartitionPolicies for a non-native profile. + const rawUserAgent = sess.getUserAgent() + const cleanUa = cleanElectronUserAgent(rawUserAgent) + sess.setUserAgent(cleanUa) setupGoogleAuthUserAgentOverride(sess) - mark('auth switch installed') + mark('clean identity installed') - // Why: onSendHeaders reports the headers exactly as they leave the network stack, after the - // product's onBeforeSendHeaders listener has rewritten them. The requests must actually be - // dispatched for it to fire, so the session is pointed at a proxy that refuses every - // connection: nothing reaches the real hosts and every load fails fast. - await sess.setProxy({ proxyRules: 'http://127.0.0.1:9', proxyBypassRules: '<-loopback>' }) + sess.setCertificateVerifyProc((_request, callback) => callback(0)) const requests = [] sess.webRequest.onSendHeaders({ urls: ['https://*/*'] }, (details) => { const headers = details.requestHeaders || {} const uaKey = Object.keys(headers).find((key) => key.toLowerCase() === 'user-agent') + const clientHints = {} + for (const [key, value] of Object.entries(headers)) { + if (key.toLowerCase().startsWith('sec-ch-ua')) { + clientHints[key.toLowerCase()] = value + } + } requests.push({ url: details.url, userAgent: uaKey ? headers[uaKey] : null, - clientHints: Object.keys(headers) - .filter((key) => key.toLowerCase().startsWith('sec-ch-ua')) - .sort() + clientHints }) }) + const server = createServer( + { + cert: ${JSON.stringify(LOCAL_HTTPS_TEST_CERTIFICATE)}, + key: ${JSON.stringify(LOCAL_HTTPS_TEST_PRIVATE_KEY)} + }, + (_request, response) => { + response.setHeader('Accept-CH', 'Sec-CH-UA-Full-Version-List') + response.end('identity') + } + ) + await new Promise((resolve, reject) => { + server.once('error', reject) + server.listen(0, '127.0.0.1', resolve) + }) + const origin = 'https://127.0.0.1:' + server.address().port const window = new BrowserWindow({ show: false, webPreferences: { partition } }) mark('window created') - for (const url of ['https://example.com/', 'https://accounts.google.com/v3/signin/identifier']) { - await window.loadURL(url).catch(() => {}) + let navigatorUserAgent + let navigatorUserAgentData + try { + await window.loadURL(origin + '/') + navigatorUserAgent = await window.webContents.executeJavaScript('navigator.userAgent') + navigatorUserAgentData = await window.webContents.executeJavaScript( + "(async () => { const data = navigator.userAgentData; return data ? { brands: data.brands, highEntropy: await data.getHighEntropyValues(['fullVersionList']) } : null })()" + ) + await window.webContents.executeJavaScript( + 'fetch("/hints").then((response) => response.text())' + ) + } finally { + await new Promise((resolve) => server.close(resolve)) } + + // Dispatch a real auth-host request without allowing it to reach the Internet. + await sess.setProxy({ proxyRules: 'http://127.0.0.1:9', proxyBypassRules: '<-loopback>' }) + await window.loadURL('https://accounts.google.com/v3/signin/identifier').catch(() => {}) mark('navigations attempted') - const navigatorUserAgent = await window.webContents.executeJavaScript('navigator.userAgent') clearTimeout(timeout) writeFileSync(resultPath, JSON.stringify({ + rawUserAgent, sessionUserAgent: sess.getUserAgent(), navigatorUserAgent, + navigatorUserAgentData, requests })) window.destroy() @@ -156,18 +215,47 @@ async function runFixture(): Promise { } } +function parseClientHintBrands(value: string): UserAgentBrand[] { + return [...value.matchAll(/"([^"]+)";v="([^"]+)"/g)].map((match) => ({ + brand: match[1], + version: match[2] + })) +} + describe('browser session wire identity under Electron', () => { - it('sends the stock Electron UA to ordinary hosts and Firefox to Google auth hosts', async () => { + it('strips the Electron and app tokens for ordinary hosts and sends Firefox to Google auth hosts', async () => { const result = await runFixture() - // Presence precondition: the stock identity still carries the Electron token that the old - // Chrome-shaped rewrite stripped, so an identity check below cannot pass on an empty UA. - expect(result.sessionUserAgent).toMatch(/ Electron\/\d/) + // Presence precondition: the raw identity really does carry the tokens, so the absence + // assertions below cannot pass vacuously on an empty or already-clean UA. + expect(result.rawUserAgent).toMatch(/ Electron\/\d/) + expect(result.rawUserAgent).toMatch(/\(KHTML, like Gecko\) \S+ Chrome\//) - const ordinary = result.requests.find((request) => request.url === 'https://example.com/') + // The whole point of STA-7147: nothing between the engine comment and Chrome/, and no + // Electron token anywhere — the shape a real Chrome sends. + expect(result.sessionUserAgent).not.toContain('Electron/') + expect(result.sessionUserAgent).toMatch(/\(KHTML, like Gecko\) Chrome\/[\d.]+ Safari\/537\.36$/) + + const ordinary = result.requests.find((request) => request.url.endsWith('/hints')) expect(ordinary, JSON.stringify(result.requests)).toBeDefined() expect(ordinary?.userAgent).toBe(result.sessionUserAgent) expect(result.navigatorUserAgent).toBe(result.sessionUserAgent) + expect(result.navigatorUserAgentData).not.toBeNull() + + // Chromium owns both client-hint surfaces. Rewriting only the request headers would make this + // comparison fail while leaving the legacy UA assertions above green. + const wireBrands = parseClientHintBrands(ordinary?.clientHints['sec-ch-ua'] ?? '') + expect(wireBrands).toEqual(result.navigatorUserAgentData?.brands) + expect(wireBrands.some(({ brand }) => /Electron|Orca/i.test(brand))).toBe(false) + const chromeMajor = result.sessionUserAgent.match(/Chrome\/(\d+)/)?.[1] + expect(wireBrands.find(({ brand }) => brand === 'Chromium')?.version).toBe(chromeMajor) + + const fullVersionList = ordinary?.clientHints['sec-ch-ua-full-version-list'] + if (fullVersionList) { + expect(parseClientHintBrands(fullVersionList)).toEqual( + result.navigatorUserAgentData?.highEntropy.fullVersionList + ) + } const auth = result.requests.find((request) => request.url.startsWith('https://accounts.google.com/') @@ -175,6 +263,6 @@ describe('browser session wire identity under Electron', () => { expect(auth, JSON.stringify(result.requests)).toBeDefined() expect(auth?.userAgent).toMatch(/Firefox\/\d/) expect(auth?.userAgent).not.toContain('Chrome') - expect(auth?.clientHints).toEqual([]) + expect(auth?.clientHints).toEqual({}) }) }) diff --git a/src/main/browser/browser-session-ua.ts b/src/main/browser/browser-session-ua.ts index 1375ebc66f5..b7fd26449f5 100644 --- a/src/main/browser/browser-session-ua.ts +++ b/src/main/browser/browser-session-ua.ts @@ -8,11 +8,23 @@ import { stripClientHints } from './browser-google-auth-ua' -// Why: the session keeps Electron's stock UA. Stripping the Electron/app tokens to look like -// plain Chrome is what Cloudflare Turnstile rejects (error 600010): a Chrome UA that ships no -// client hints reads as a spoof, while a declared Electron client clears the same challenge. -// This handler only owns the Google auth-host Firefox switch, which is a proven, host-scoped -// exception that must stay consistent across the header and every cross-host subresource. +// Why: Electron's default UA includes "Electron/X.X.X" and the app name +// (e.g. "orca/1.2.3"), an impossible identity for sessions imported from Chrome. +// This focused revocation fix strips only those tokens; it does not attempt full Chrome +// impersonation, and Chromium's client-hint identity remains browser-owned. +export function cleanElectronUserAgent(ua: string): string { + return ( + ua + .replace(/\s+Electron\/\S+/, '') + // Why: \S+ matches any non-whitespace token (e.g. "orca/1.3.8-rc.0") + // including pre-release semver strings that [\d.]+ would miss. + .replace(/(\)\s+)\S+\s+(Chrome\/)/, '$1$2') + ) +} + +// Why: Chromium already publishes one internally consistent client-hint identity through both +// request headers and navigator.userAgentData. This handler only owns the host-scoped Firefox +// exception; synthesizing Chrome brands here would make those two browser-owned surfaces disagree. export function setupGoogleAuthUserAgentOverride(sess: Session): void { const firefoxUa = googleAuthUserAgent() @@ -24,11 +36,20 @@ export function setupGoogleAuthUserAgentOverride(sess: Session): void { // sec-ch-ua* because real Firefox sends none. setUserAgentHeader(headers, firefoxUa) stripClientHints(headers) - } else if (currentUserAgent(headers) === firefoxUa) { - // Why: while the auth document is on screen the WebContents UA is Firefox, so its - // cross-host subresource/XHR requests carry the Firefox UA yet still bear Chromium - // client hints — a sharper cross-host identity tell than either alone. + callback({ requestHeaders: headers }) + return + } + if (currentUserAgent(headers) === firefoxUa) { + // Why: while the auth document is on screen the WebContents UA is Firefox, + // so its cross-host subresource/XHR requests (gstatic, play.google.com, the + // sign-in challenge endpoints) reach here carrying the Firefox UA yet still + // bearing Chromium client hints. Rewriting those to Chrome pairs a Firefox + // UA with Chrome hints — a sharper cross-host identity tell than either + // alone, which can stall Google's password-submit challenge. Real Firefox + // sends no client hints, so strip them to keep one identity for the flow. stripClientHints(headers) + callback({ requestHeaders: headers }) + return } callback({ requestHeaders: headers }) }) diff --git a/src/main/browser/browser-viewport-user-agent.ts b/src/main/browser/browser-viewport-user-agent.ts index b8159a44c2a..691ed0b88a8 100644 --- a/src/main/browser/browser-viewport-user-agent.ts +++ b/src/main/browser/browser-viewport-user-agent.ts @@ -44,7 +44,8 @@ export function buildViewportUserAgentOverride(args: { return { userAgent: googleAuthUserAgent() } } if (!args.mobile) { - // Why: desktop presets republish the session's own identity unchanged. + // Why: desktop presets republish the session's clean identity, or a preset would put the + // Electron/app tokens back on the wire and a transplanted session gets revoked (STA-7147). return { userAgent: args.baseUserAgent } } const chromeMajor = extractChromeMajor(args.baseUserAgent) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.test.ts b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts index 62fbd8cf203..2270b0627fd 100644 --- a/src/main/claude/claude-agent-sdk-user-message-queue.test.ts +++ b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts @@ -1,6 +1,9 @@ import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' import { describe, expect, it } from 'vitest' -import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' +import { + claudeUserMessageWasProvablyUnwritten, + createClaudeUserMessageQueue +} from './claude-agent-sdk-user-message-queue' /** * The SDK's input pump is `for await (const frame of prompt) { await transport.write(frame) }`. @@ -24,7 +27,7 @@ const settled = (promise: Promise): Promise<'settled' | 'pending'> => ]) describe('claude user message queue', () => { - it('rejects the frame the SDK pulled but abandoned without writing', async () => { + it('treats a frame the SDK pulled and abandoned as write-outcome unknown', async () => { const queue = createClaudeUserMessageQueue() const pump = queue.messages[Symbol.asyncIterator]() const sent = queue.push(frame('hello')) @@ -33,9 +36,11 @@ describe('claude user message queue', () => { await pump.return?.(undefined) await expect(settled(sent)).resolves.toBe('settled') - await expect(sent).rejects.toThrow( - 'claude stream-json input ended before the frame was written' - ) + const error = await sent.catch((caught: unknown) => caught) + expect(error).toMatchObject({ + message: 'claude stream-json input ended before confirming the frame write' + }) + expect(claudeUserMessageWasProvablyUnwritten(error)).toBe(false) }) it('rejects an in-flight frame from fail() when the SDK never resumes the pump', async () => { @@ -47,7 +52,22 @@ describe('claude user message queue', () => { queue.fail(new Error('claude stream-json exited: child died')) await expect(settled(sent)).resolves.toBe('settled') - await expect(sent).rejects.toThrow('claude stream-json exited: child died') + const error = await sent.catch((caught: unknown) => caught) + expect(error).toMatchObject({ message: 'claude stream-json exited: child died' }) + expect(claudeUserMessageWasProvablyUnwritten(error)).toBe(false) + }) + + it('marks only frames still queued in Orca as provably unwritten', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const inFlight = queue.push(frame('first')).catch((caught: unknown) => caught) + await pump.next() + const queued = queue.push(frame('second')).catch((caught: unknown) => caught) + + queue.fail(new Error('claude stream-json exited: child died')) + + expect(claudeUserMessageWasProvablyUnwritten(await inFlight)).toBe(false) + expect(claudeUserMessageWasProvablyUnwritten(await queued)).toBe(true) }) it('still settles a written frame only once the pump asks for the next one', async () => { diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.ts b/src/main/claude/claude-agent-sdk-user-message-queue.ts index 87fa6660159..3cd0ad7a6f9 100644 --- a/src/main/claude/claude-agent-sdk-user-message-queue.ts +++ b/src/main/claude/claude-agent-sdk-user-message-queue.ts @@ -6,18 +6,42 @@ type QueuedMessage = { reject: (error: Error) => void } +type ClaudeUserMessageFailureDisposition = 'unwritten' | 'write-outcome-unknown' + +class ClaudeUserMessageFailure extends Error { + readonly disposition: ClaudeUserMessageFailureDisposition + + constructor(disposition: ClaudeUserMessageFailureDisposition, cause: Error) { + super(cause.message, { cause }) + this.name = 'ClaudeUserMessageFailure' + this.disposition = disposition + } +} + +export function claudeUnwrittenUserMessageError(cause: Error): Error { + return new ClaudeUserMessageFailure('unwritten', cause) +} + +export function claudeUserMessageWasProvablyUnwritten(error: unknown): boolean { + return error instanceof ClaudeUserMessageFailure && error.disposition === 'unwritten' +} + +function claudeAmbiguousUserMessageError(cause: Error): Error { + return new ClaudeUserMessageFailure('write-outcome-unknown', cause) +} + export type ClaudeUserMessageQueue = { /** The SDK's streaming-input prompt; it stays open until `end`. */ messages: AsyncIterable /** Resolves once the SDK has finished writing the frame to the child. */ push: (message: SDKUserMessage) => Promise - /** Reject every unwritten frame, in-flight included; a caller waiting on a send must not hang past the exit. */ + /** Reject every unsettled frame; an in-flight frame carries an ambiguous write outcome. */ fail: (error: Error) => void end: () => void } /** The rejection an abandoned frame carries when nothing else has named a cause yet. */ -const UNWRITTEN_FRAME_MESSAGE = 'claude stream-json input ended before the frame was written' +const UNCONFIRMED_FRAME_MESSAGE = 'claude stream-json input ended before confirming the frame write' export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { const queued: QueuedMessage[] = [] @@ -57,7 +81,9 @@ export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { // is the same "the frame reached the child" proof the hand-rolled write gave. next.resolve() } else { - rejectInFlight(failure ?? new Error(UNWRITTEN_FRAME_MESSAGE)) + rejectInFlight( + claudeAmbiguousUserMessageError(failure ?? new Error(UNCONFIRMED_FRAME_MESSAGE)) + ) } } continue @@ -76,7 +102,7 @@ export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { push: (message) => new Promise((resolve, reject) => { if (failure) { - reject(failure) + reject(claudeUnwrittenUserMessageError(failure)) return } queued.push({ message, resolve, reject }) @@ -85,11 +111,11 @@ export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { fail: (error) => { failure ??= error for (const entry of queued.splice(0)) { - entry.reject(error) + entry.reject(claudeUnwrittenUserMessageError(error)) } // A pump that never resumes cannot run the generator's cleanup, so the // exit path has to reach the in-flight frame itself. - rejectInFlight(error) + rejectInFlight(claudeAmbiguousUserMessageError(error)) notify() }, end: () => { diff --git a/src/main/claude/claude-background-task-frames.ts b/src/main/claude/claude-background-task-frames.ts new file mode 100644 index 00000000000..d954128d91b --- /dev/null +++ b/src/main/claude/claude-background-task-frames.ts @@ -0,0 +1,106 @@ +// Field readers for the Claude SDK's background-task lifecycle frames +// (task_started / task_updated / task_notification / background_tasks_changed). +// Pure and bounded: every reader rejects absent, non-string, or oversized +// values so a malformed frame degrades to "field unknown", never to a throw. + +import type { + AgentSessionBackgroundTask, + AgentSessionBackgroundTaskRunState +} from '../../shared/agent-session-wire' + +const MAX_TASK_ID_LENGTH = 512 +const MAX_TASK_TEXT_LENGTH = 512 + +export type ClaudeBackgroundTaskKind = AgentSessionBackgroundTask['kind'] + +export function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null ? (value as Record) : null +} + +/** The bound every task id shares, wherever it enters. An id the roster stores + * becomes a durable entry key, so a provisional one takes the same bound the + * announced path applies — an over-long id is rejected, never truncated. */ +export function isBoundedClaudeTaskId(value: string): boolean { + return value.length > 0 && value.length <= MAX_TASK_ID_LENGTH +} + +export function taskId(message: Record): string | null { + const value = message.task_id + return typeof value === 'string' && isBoundedClaudeTaskId(value) ? value : null +} + +function boundedTaskText(value: unknown): string | undefined { + if (typeof value !== 'string') { + return undefined + } + const trimmed = value.trim().replace(/\s+/g, ' ') + return trimmed.length > 0 ? trimmed.slice(0, MAX_TASK_TEXT_LENGTH) : undefined +} + +export function taskDescription(value: unknown): string | undefined { + return boundedTaskText(value) +} + +/** The provider-reported identity for a task. Subagent frames have carried the + * type under both `agent_type` and `subagent_type` across SDK versions. */ +export function taskName(frame: Record): string | undefined { + return ( + boundedTaskText(frame.name) ?? + boundedTaskText(frame.agent_type) ?? + boundedTaskText(frame.subagent_type) + ) +} + +export function classifyClaudeBackgroundTaskKind(taskType: unknown): ClaudeBackgroundTaskKind { + switch (taskType) { + case 'local_agent': + return 'agent' + case 'local_workflow': + return 'workflow' + case 'local_bash': + return 'command' + case 'monitor': + return 'monitor' + default: + return 'unknown' + } +} + +/** Cumulative token usage from a task_progress / task_notification frame. */ +export function taskUsageTotalTokens(frame: Record): number | undefined { + const usage = record(frame.usage) + const total = usage?.total_tokens + return typeof total === 'number' && Number.isFinite(total) && total >= 0 + ? Math.floor(total) + : undefined +} + +/** Settled state for a terminal status. Null for anything else — an unreadable + * status never settles a task by itself. */ +export function terminalClaudeTaskRunState( + status: unknown +): AgentSessionBackgroundTaskRunState | null { + switch (status) { + case 'completed': + return 'done' + case 'failed': + return 'blocked' + case 'killed': + case 'stopped': + return 'idle' + default: + return null + } +} + +/** Live state for a non-terminal status. Null leaves the tracked state alone. */ +export function liveClaudeTaskRunState(status: unknown): AgentSessionBackgroundTaskRunState | null { + switch (status) { + case 'pending': + case 'running': + case 'paused': + return 'working' + default: + return null + } +} diff --git a/src/main/claude/claude-background-task-resume.test.ts b/src/main/claude/claude-background-task-resume.test.ts new file mode 100644 index 00000000000..c043d485304 --- /dev/null +++ b/src/main/claude/claude-background-task-resume.test.ts @@ -0,0 +1,92 @@ +import { describe, expect, it } from 'vitest' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' + +const agent = { + task_id: 'a962f88aa82feb1c1', + task_type: 'local_agent', + description: 'Long proof writer' +} +const shell = { task_id: 'bcl6x3ixf', task_type: 'local_bash', description: 'sleep 150' } +const sibling = { task_id: 'sibling', task_type: 'local_agent' } + +function system(subtype: string, fields: Record) { + return { type: 'system', subtype, ...fields } +} + +describe('Claude background task pause/resume ownership', () => { + it('moves a retained child back to live ownership across eviction, outcome, and auto-resume', () => { + let now = 100 + const tracker = new ClaudeBackgroundTaskTracker(() => now) + const roster = (tasks: unknown[]) => + tracker.observe(system('background_tasks_changed', { tasks })) + roster([agent, sibling, shell]) + tracker.observe( + system('task_progress', { task_id: agent.task_id, usage: { total_tokens: 18000 } }) + ) + tracker.observe({ type: 'result' }) + expect(tracker.state?.tasks).toHaveLength(3) + roster([sibling, shell]) + expect(tracker.state?.settledTasks).toBeUndefined() + tracker.observe( + system('task_updated', { task_id: agent.task_id, patch: { status: 'completed' } }) + ) + tracker.observe( + system('task_notification', { + task_id: agent.task_id, + status: 'completed', + usage: { total_tokens: 19003 } + }) + ) + expect(tracker.state?.settledTasks).toEqual([ + expect.objectContaining({ + id: agent.task_id, + state: 'done', + startedAt: 100, + totalTokens: 19003 + }) + ]) + now = 150000 + roster([agent, sibling, shell]) + expect(tracker.state?.settledTasks).toBeUndefined() + expect(tracker.state?.tasks).toEqual([ + expect.objectContaining({ + id: agent.task_id, + state: 'working', + startedAt: 100, + totalTokens: 19003 + }), + expect.objectContaining({ id: sibling.task_id }), + expect.objectContaining({ id: shell.task_id }) + ]) + expect(tracker.stoppableTaskIds).toEqual([agent.task_id, sibling.task_id, shell.task_id]) + roster([sibling, shell]) + tracker.observe( + system('task_notification', { + task_id: agent.task_id, + status: 'completed', + usage: { total_tokens: 21000 } + }) + ) + expect(tracker.state?.settledTasks).toEqual([ + expect.objectContaining({ id: agent.task_id, startedAt: 100, totalTokens: 21000 }) + ]) + roster([]) + expect(tracker.state).toBeNull() + }) + + it('reconciles an edge-only resume without keeping its earlier settled copy', () => { + const tracker = new ClaudeBackgroundTaskTracker(() => 100) + for (const task of [agent, sibling]) { + tracker.observe(system('task_started', { ...task, is_backgrounded: true })) + } + tracker.observe(system('task_notification', { task_id: agent.task_id, status: 'completed' })) + tracker.observe( + system('task_updated', { + task_id: agent.task_id, + patch: { status: 'running', is_backgrounded: true } + }) + ) + expect(tracker.state?.tasks).toHaveLength(2) + expect(tracker.state?.settledTasks).toBeUndefined() + }) +}) diff --git a/src/main/claude/claude-background-task-tracker.test.ts b/src/main/claude/claude-background-task-tracker.test.ts index d8f316d7dcd..43912bfa560 100644 --- a/src/main/claude/claude-background-task-tracker.test.ts +++ b/src/main/claude/claude-background-task-tracker.test.ts @@ -16,6 +16,29 @@ function aggregate(tasks: unknown[]): Record { return system('background_tasks_changed', { tasks }) } +function trackerAt(times: number[]): ClaudeBackgroundTaskTracker { + let index = 0 + return new ClaudeBackgroundTaskTracker(() => times[Math.min(index++, times.length - 1)]) +} + +/** The identity and stoppability of each published row, which is what the + * foreground cases below are about; `startedAt` and `state` have their own + * tests and would only make these brittle. */ +function rows(tracker: ClaudeBackgroundTaskTracker): { id: string; stoppable?: boolean }[] { + return (tracker.state?.tasks ?? []).map((task) => ({ + id: task.id, + ...(task.stoppable === undefined ? {} : { stoppable: task.stoppable }) + })) +} + +function started(id: string, backgrounded: boolean): Record { + return system('task_started', { + task_id: id, + task_type: 'local_agent', + is_backgrounded: backgrounded + }) +} + describe('ClaudeBackgroundTaskTracker', () => { it('classifies SDK task types without inferring them from descriptions', () => { expect(classifyClaudeBackgroundTaskKind('local_agent')).toBe('agent') @@ -25,27 +48,30 @@ describe('ClaudeBackgroundTaskTracker', () => { expect(classifyClaudeBackgroundTaskKind('future_task')).toBe('unknown') }) - it('waits for the foreground turn to settle before monitoring a background task', () => { - const tracker = new ClaudeBackgroundTaskTracker() + it('publishes a backgrounded task while the foreground turn is still running', () => { + const tracker = trackerAt([100]) tracker.observe({ type: 'user' }, true) - tracker.observe( - system('task_started', { - task_id: 'task-1', - task_type: 'local_agent', - is_backgrounded: true - }) - ) - expect(tracker.state).toBeNull() - - expect(tracker.observe(result())).toBe(true) + expect( + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + ).toBe(true) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: 'task-1', kind: 'agent' }] + tasks: [{ id: 'task-1', kind: 'agent', state: 'working', startedAt: 100 }] }) + + // The turn settling changes nothing the strip renders. + expect(tracker.observe(result())).toBe(false) + expect(tracker.state?.tasks).toHaveLength(1) }) it('uses an explicit background update for a foreground task and ignores progress alone', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe({ type: 'user' }, true) tracker.observe( system('task_started', { @@ -63,12 +89,12 @@ describe('ClaudeBackgroundTaskTracker', () => { tracker.observe(system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: true } })) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: 'task-1', kind: 'command' }] + tasks: [{ id: 'task-1', kind: 'command', state: 'working', startedAt: 100 }] }) }) it('publishes bounded display details when a running task description changes', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) expect( tracker.observe( system('task_started', { @@ -81,7 +107,15 @@ describe('ClaudeBackgroundTaskTracker', () => { ).toBe(true) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + tasks: [ + { + id: 'task-1', + kind: 'command', + description: 'run the build', + state: 'working', + startedAt: 100 + } + ] }) expect( @@ -103,8 +137,193 @@ describe('ClaudeBackgroundTaskTracker', () => { ).toBe(false) }) + it('carries provider-reported names and re-derives classification per transition', () => { + const tracker = trackerAt([100]) + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'future_task', + is_backgrounded: true + }) + ) + expect(tracker.state?.tasks?.[0]).toMatchObject({ kind: 'unknown' }) + + expect( + tracker.observe( + system('task_updated', { + task_id: 'task-1', + patch: { task_type: 'local_agent', agent_type: 'deep_review' } + }) + ) + ).toBe(true) + expect(tracker.state?.tasks?.[0]).toMatchObject({ + kind: 'agent', + name: 'deep_review', + state: 'working' + }) + }) + + it('retains settled siblings beside live work and exits with the last live task', () => { + const tracker = trackerAt([100, 200]) + tracker.observe( + system('task_started', { task_id: 'task-a', task_type: 'local_agent', is_backgrounded: true }) + ) + tracker.observe( + system('task_started', { task_id: 'task-b', task_type: 'local_agent', is_backgrounded: true }) + ) + + expect( + tracker.observe(system('task_updated', { task_id: 'task-a', patch: { status: 'completed' } })) + ).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-b', kind: 'agent', state: 'working', startedAt: 200 }], + settledTasks: [{ id: 'task-a', kind: 'agent', state: 'done', startedAt: 100 }] + }) + expect(tracker.stoppableTaskIds).toEqual(['task-b']) + + expect( + tracker.observe(system('task_updated', { task_id: 'task-b', patch: { status: 'killed' } })) + ).toBe(true) + expect(tracker.state).toBeNull() + }) + + it('settles a sibling from the captured producer order: aggregate eviction, then the outcome', () => { + // Verbatim sequence from a real SDK capture (2026-09-07): the aggregate + // roster arrives FIRST, already missing the finished task, and the + // terminal edges trail in the same tick. + const tracker = trackerAt([100, 200]) + tracker.observe( + system('task_started', { + task_id: 'bh4zn8der', + tool_use_id: 'toolu_01M', + description: 'Sleep for 5 seconds', + is_backgrounded: true, + task_type: 'local_bash' + }) + ) + tracker.observe( + aggregate([ + { task_id: 'bh4zn8der', task_type: 'local_bash', description: 'Sleep for 5 seconds' }, + { task_id: 'bprosaiim', task_type: 'local_bash', description: 'Sleep for 25 seconds' } + ]) + ) + + // The settling child is evicted by the aggregate before any outcome frame. + tracker.observe( + aggregate([ + { task_id: 'bprosaiim', task_type: 'local_bash', description: 'Sleep for 25 seconds' } + ]) + ) + tracker.observe( + system('task_updated', { + task_id: 'bh4zn8der', + patch: { status: 'completed', end_time: 1788804376515 } + }) + ) + expect( + tracker.observe( + system('task_notification', { + task_id: 'bh4zn8der', + tool_use_id: 'toolu_01M', + status: 'completed', + summary: 'Background command "Sleep for 5 seconds" completed (exit code 0)', + usage: { total_tokens: 18130, tool_uses: 1, duration_ms: 10772 } + }) + ) + ).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [ + { + id: 'bprosaiim', + kind: 'command', + description: 'Sleep for 25 seconds', + state: 'working', + startedAt: 200 + } + ], + settledTasks: [ + { + id: 'bh4zn8der', + kind: 'command', + description: 'Sleep for 5 seconds', + state: 'done', + startedAt: 100, + totalTokens: 18130 + } + ] + }) + + // Last task killed, same captured order: the strip exits. + tracker.observe(aggregate([])) + tracker.observe(system('task_updated', { task_id: 'bprosaiim', patch: { status: 'killed' } })) + tracker.observe(system('task_notification', { task_id: 'bprosaiim', status: 'stopped' })) + expect(tracker.state).toBeNull() + }) + + it('carries task_progress usage into a live row without clobbering its name', () => { + const tracker = trackerAt([100]) + tracker.observe( + system('task_started', { + task_id: 'agent-1', + task_type: 'local_agent', + subagent_type: 'general-purpose', + description: 'Sleep 6 seconds test', + is_backgrounded: true + }) + ) + expect( + tracker.observe( + system('task_progress', { + task_id: 'agent-1', + description: 'Running Sleep for 6 seconds', + subagent_type: 'general-purpose', + usage: { total_tokens: 14866, tool_uses: 1, duration_ms: 2818 }, + last_tool_name: 'Bash' + }) + ) + ).toBe(true) + expect(tracker.state?.tasks?.[0]).toEqual({ + id: 'agent-1', + kind: 'agent', + // Progress descriptions are transient activity, never the task's name. + description: 'Sleep 6 seconds test', + name: 'general-purpose', + state: 'working', + startedAt: 100, + totalTokens: 14866 + }) + }) + + it('maps terminal statuses onto settled states', () => { + const tracker = trackerAt([100, 200]) + tracker.observe( + system('task_started', { task_id: 'live', task_type: 'local_agent', is_backgrounded: true }) + ) + tracker.observe( + system('task_started', { task_id: 'failed', task_type: 'local_agent', is_backgrounded: true }) + ) + tracker.observe(system('task_notification', { task_id: 'failed', status: 'failed' })) + expect(tracker.state?.settledTasks).toEqual([ + { id: 'failed', kind: 'agent', state: 'blocked', startedAt: 200 } + ]) + }) + + it('leaves a task open when a patch cannot be read', () => { + const tracker = trackerAt([100]) + tracker.observe( + system('task_started', { task_id: 'task-1', task_type: 'local_agent', is_backgrounded: true }) + ) + expect(tracker.observe(system('task_updated', { task_id: 'task-1', patch: 'garbage' }))).toBe( + false + ) + expect(tracker.state?.tasks).toHaveLength(1) + expect(tracker.state?.settledTasks).toBeUndefined() + }) + it('replaces its roster from aggregate lifecycle frames and preserves stoppable provider ids', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) expect( tracker.observe( aggregate([ @@ -117,8 +336,8 @@ describe('ClaudeBackgroundTaskTracker', () => { expect(tracker.state).toEqual({ state: 'monitoring', tasks: [ - { id: 'task-agent', kind: 'agent', description: 'agent' }, - { id: 'task-bash', kind: 'command', description: 'bash' } + { id: 'task-agent', kind: 'agent', description: 'agent', state: 'working', startedAt: 100 }, + { id: 'task-bash', kind: 'command', description: 'bash', state: 'working', startedAt: 100 } ] }) @@ -128,18 +347,31 @@ describe('ClaudeBackgroundTaskTracker', () => { ) ).toBe(true) expect(tracker.stoppableTaskIds).toEqual(['task-next']) - expect(tracker.state).toEqual({ - state: 'monitoring', - tasks: [{ id: 'task-next', kind: 'workflow', description: 'workflow' }] - }) expect(tracker.observe(aggregate([]))).toBe(true) expect(tracker.stoppableTaskIds).toEqual([]) expect(tracker.state).toBeNull() }) + it('preserves first-seen timestamps across aggregate roster replacement', () => { + const tracker = trackerAt([100, 200]) + tracker.observe( + system('task_started', { task_id: 'task-1', task_type: 'local_agent', is_backgrounded: true }) + ) + tracker.observe( + aggregate([ + { task_id: 'task-1', task_type: 'local_agent' }, + { task_id: 'task-2', task_type: 'local_bash' } + ]) + ) + expect(tracker.state?.tasks).toEqual([ + { id: 'task-1', kind: 'agent', state: 'working', startedAt: 100 }, + { id: 'task-2', kind: 'command', state: 'working', startedAt: 200 } + ]) + }) + it('excludes ambient aggregate tasks', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe( aggregate([ { task_id: 'ambient', task_type: 'monitor', description: 'watcher', ambient: true }, @@ -151,7 +383,7 @@ describe('ClaudeBackgroundTaskTracker', () => { }) it('does not let late edge frames revive tasks cleared by an aggregate roster', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe( aggregate([{ task_id: 'task-late', task_type: 'local_agent', description: 'agent' }]) ) @@ -173,7 +405,7 @@ describe('ClaudeBackgroundTaskTracker', () => { }) it('lets an authoritative aggregate roster replace earlier terminal-edge evidence', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe(system('task_notification', { task_id: 'task-live', status: 'completed' })) tracker.observe( @@ -183,12 +415,28 @@ describe('ClaudeBackgroundTaskTracker', () => { expect(tracker.stoppableTaskIds).toEqual(['task-live']) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: 'task-live', kind: 'agent', description: 'agent' }] + tasks: [ + { id: 'task-live', kind: 'agent', description: 'agent', state: 'working', startedAt: 100 } + ] }) }) + it('retracts a settled copy when an authoritative roster reports the task live again', () => { + const tracker = trackerAt([100, 200, 300]) + const tasks = [ + { task_id: 'agent', task_type: 'local_agent', description: 'Review sample' }, + { task_id: 'shell', task_type: 'local_bash' } + ] + tracker.observe(aggregate(tasks)) + tracker.observe(system('task_notification', { task_id: 'agent', status: 'completed' })) + expect(tracker.state?.settledTasks).toHaveLength(1) + tracker.observe(aggregate(tasks)) + expect(tracker.state?.tasks?.map((task) => task.id)).toEqual(['agent', 'shell']) + expect(tracker.state?.settledTasks).toBeUndefined() + }) + it('keeps terminal edges authoritative on either side of aggregate replacement', () => { - const terminalFirst = new ClaudeBackgroundTaskTracker() + const terminalFirst = trackerAt([100]) terminalFirst.observe( system('task_notification', { task_id: 'task-first', status: 'completed' }) ) @@ -202,7 +450,7 @@ describe('ClaudeBackgroundTaskTracker', () => { ) expect(terminalFirst.state).toBeNull() - const terminalLast = new ClaudeBackgroundTaskTracker() + const terminalLast = trackerAt([100]) terminalLast.observe( aggregate([{ task_id: 'task-last', task_type: 'local_agent', description: 'agent' }]) ) @@ -218,7 +466,7 @@ describe('ClaudeBackgroundTaskTracker', () => { }) it('keeps terminal evidence authoritative across duplicates and out-of-order starts', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) const terminal = system('task_notification', { task_id: 'task-late', status: 'completed' }) tracker.observe(terminal) tracker.observe(terminal) @@ -239,7 +487,7 @@ describe('ClaudeBackgroundTaskTracker', () => { ) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: 'task-live', kind: 'monitor' }] + tasks: [{ id: 'task-live', kind: 'monitor', state: 'monitoring', startedAt: 100 }] }) expect( tracker.observe(system('task_updated', { task_id: 'task-live', patch: { status: 'killed' } })) @@ -249,17 +497,24 @@ describe('ClaudeBackgroundTaskTracker', () => { it('recognizes task types that are registered only as background work', () => { for (const taskType of ['local_workflow', 'monitor']) { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe(system('task_started', { task_id: taskType, task_type: taskType })) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: taskType, kind: taskType === 'local_workflow' ? 'workflow' : 'monitor' }] + tasks: [ + { + id: taskType, + kind: taskType === 'local_workflow' ? 'workflow' : 'monitor', + state: taskType === 'local_workflow' ? 'working' : 'monitoring', + startedAt: 100 + } + ] }) } }) it('admits unknown background updates conservatively and bounds edge-only fallback ids', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe( system('task_updated', { task_id: 'unknown', patch: { is_backgrounded: true } }) ) @@ -278,7 +533,7 @@ describe('ClaudeBackgroundTaskTracker', () => { }) it('bounds aggregate rosters and resets to the edge-only fallback on clear', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe( aggregate( Array.from({ length: 400 }, (_, index) => ({ @@ -301,23 +556,30 @@ describe('ClaudeBackgroundTaskTracker', () => { expect(tracker.stoppableTaskIds).toEqual(['edge-after-reset']) }) - it('gates aggregate monitoring behind foreground turn completion', () => { - const tracker = new ClaudeBackgroundTaskTracker() + it('publishes an aggregate roster observed mid-turn', () => { + const tracker = trackerAt([100]) tracker.observe({ type: 'user' }, true) - tracker.observe( - aggregate([{ task_id: 'task-live', task_type: 'local_bash', description: 'command' }]) - ) - expect(tracker.state).toBeNull() - - expect(tracker.observe(result())).toBe(true) + expect( + tracker.observe( + aggregate([{ task_id: 'task-live', task_type: 'local_bash', description: 'command' }]) + ) + ).toBe(true) expect(tracker.state).toEqual({ state: 'monitoring', - tasks: [{ id: 'task-live', kind: 'command', description: 'command' }] + tasks: [ + { + id: 'task-live', + kind: 'command', + description: 'command', + state: 'working', + startedAt: 100 + } + ] }) }) it('ignores ambient SDK tasks and clears all liveness when the session ends', () => { - const tracker = new ClaudeBackgroundTaskTracker() + const tracker = trackerAt([100]) tracker.observe( system('task_started', { task_id: 'ambient', @@ -337,4 +599,137 @@ describe('ClaudeBackgroundTaskTracker', () => { expect(tracker.clear()).toBe(true) expect(tracker.state).toBeNull() }) + it('marks a foreground row not stoppable and leaves a backgrounded row alone', () => { + // `stopTask` has no foreground target, so the row must not offer a Stop that + // resolves to an empty list and silently reports nothing cancelled. A + // backgrounded row stays untouched on the wire: absent means stoppable. + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe(started('fore-1', false)) + tracker.observe(started('back-1', true)) + + expect(rows(tracker)).toEqual([{ id: 'fore-1', stoppable: false }, { id: 'back-1' }]) + expect(tracker.stoppableTaskIds).toEqual(['back-1']) + }) + + it('keeps live foreground work across an aggregate roster that never lists it', () => { + // `background_tasks_changed` enumerates BACKGROUNDED work only, so it is + // authoritative over that class alone. Treating it as the whole world wiped + // every in-flight foreground row and then dropped every later start. + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe(started('fore-1', false)) + tracker.observe( + aggregate([{ task_id: 'back-1', task_type: 'local_bash', description: 'bash' }]) + ) + + // A retained row also keeps the place the user is already reading it in. + expect(rows(tracker)).toEqual([{ id: 'fore-1', stoppable: false }, { id: 'back-1' }]) + + // A foreground start after the roster is new work, not a stale echo. + tracker.observe(started('fore-2', false)) + expect(rows(tracker)).toEqual([ + { id: 'fore-1', stoppable: false }, + { id: 'back-1' }, + { id: 'fore-2', stoppable: false } + ]) + + // Turn end still retires the foreground rows and only those. + tracker.observe(result()) + expect(rows(tracker)).toEqual([{ id: 'back-1' }]) + }) + + it('drops a backgrounded start the roster no longer lists but bounds what it retains', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + for (let index = 0; index < 300; index += 1) { + tracker.observe(started(`fore-${index}`, false)) + } + tracker.observe( + aggregate([{ task_id: 'back-1', task_type: 'local_bash', description: 'bash' }]) + ) + + const ids = rows(tracker).map((row) => row.id) + // 255 retained foreground rows plus the roster's own entry: retention is + // real and still counts against the cap. + expect(ids).toHaveLength(256) + // When the cap bites, the STALEST retained row goes, not the newest. + expect(ids).toContain('fore-299') + expect(ids).not.toContain('fore-44') + expect(ids).toContain('back-1') + + // Aggregate authority over its OWN class is unchanged. + tracker.observe(started('stale', true)) + expect(tracker.stoppableTaskIds).toEqual(['back-1']) + }) + + it('keeps a finished foreground id dead across a roster that never listed it', () => { + // The start guard only convicts BACKGROUNDED starts now, so terminal + // evidence is the only thing left defending a finished foreground id — and + // the roster carries no evidence about one, so it must not wipe it. + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe(started('fore-1', false)) + tracker.observe(system('task_notification', { task_id: 'fore-1', status: 'completed' })) + expect(tracker.state).toBeNull() + + tracker.observe( + aggregate([{ task_id: 'back-1', task_type: 'local_bash', description: 'bash' }]) + ) + tracker.observe(started('fore-1', false)) + + expect(rows(tracker)).toEqual([{ id: 'back-1' }]) + }) + + it('retires a phantom foreground row when the next turn starts', () => { + // A foreground `task_started` with no turn open has no `result` coming to + // retire it, so it would sit in the strip — with no stop of its own — and + // refuse a conversation command. Turn start is the same evidence `result` + // is, and settling on it is cleanup only: nothing gates visibility on it. + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe(started('phantom', false)) + expect(rows(tracker)).toEqual([{ id: 'phantom', stoppable: false }]) + + tracker.observe({ type: 'user' }, true) + expect(tracker.state).toBeNull() + }) + + it('settles a previous turn the way the subagent roster settles it', () => { + // On this same frame the roster's `settleTurn` moves a still-working + // FOREGROUND child to `unverifiable` and leaves a backgrounded one alone. + // The strip has no `unverifiable` row, so keeping one would assert `live` + // for work Orca has already stopped vouching for. + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe(started('fore', false)) + tracker.observe(started('back', true)) + + // No `result` for that turn; the next one starting is its only end. + tracker.observe({ type: 'user' }, true) + expect(rows(tracker)).toEqual([{ id: 'back' }]) + }) + + it('empties only between one task retiring and the next starting', () => { + // The strip's mid-turn unmount in a sequential fan-out is TRUTHFUL: A leaves + // on the provider's own terminal frame, B does not exist yet, and nothing + // sweeps A early. Foreground work is not retained as a settled row either, + // so an empty roster means no task is running. + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe(started('A', false)) + expect(rows(tracker)).toEqual([{ id: 'A', stoppable: false }]) + tracker.observe(system('task_notification', { task_id: 'A', status: 'completed' })) + expect(tracker.state).toBeNull() + tracker.observe(started('B', false)) + expect(rows(tracker)).toEqual([{ id: 'B', stoppable: false }]) + + // Backgrounded work spanning the same gap holds the roster open, so an + // empty one is never work the strip is hiding. + const spanned = new ClaudeBackgroundTaskTracker() + spanned.observe({ type: 'user' }, true) + spanned.observe(started('bg', true)) + spanned.observe(started('A', false)) + spanned.observe(system('task_notification', { task_id: 'A', status: 'completed' })) + expect(rows(spanned)).toEqual([{ id: 'bg' }]) + }) }) diff --git a/src/main/claude/claude-background-task-tracker.ts b/src/main/claude/claude-background-task-tracker.ts index de24b9a4fba..14f4271b7d8 100644 --- a/src/main/claude/claude-background-task-tracker.ts +++ b/src/main/claude/claude-background-task-tracker.ts @@ -1,78 +1,54 @@ import type { AgentSessionBackgroundTask, + AgentSessionBackgroundTaskRunState, AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import { + classifyClaudeBackgroundTaskKind, + liveClaudeTaskRunState, + record, + taskDescription, + taskId, + taskName, + taskUsageTotalTokens, + terminalClaudeTaskRunState +} from './claude-background-task-frames' +import { + ClaudeSettledBackgroundTasks, + claudeBackgroundTaskDetail, + type TrackedClaudeBackgroundTask +} from './claude-settled-background-tasks' + +// `claude-subagent-*` reads this channel through these names; the readers themselves +// live in the frames module so both consumers share one definition. +export { + classifyClaudeBackgroundTaskKind, + isBoundedClaudeTaskId, + taskDescription as claudeTaskDescription, + taskId as claudeTaskId +} from './claude-background-task-frames' +export type { ClaudeBackgroundTaskKind } from './claude-background-task-frames' const MAX_TRACKED_TASKS = 256 -const MAX_TASK_ID_LENGTH = 512 -const MAX_TASK_DESCRIPTION_LENGTH = 512 -const TERMINAL_TASK_STATES = new Set(['completed', 'failed', 'killed', 'stopped']) - -export type ClaudeBackgroundTaskKind = AgentSessionBackgroundTask['kind'] - -type TrackedTask = { - backgrounded: boolean - kind: ClaudeBackgroundTaskKind - description?: string -} - -function record(value: unknown): Record | null { - return typeof value === 'object' && value !== null ? (value as Record) : null -} - -/** The bound every task id shares, wherever it enters. An id the roster stores - * becomes a durable entry key, so a provisional one takes the same bound the - * announced path applies — an over-long id is rejected, never truncated. */ -export function isBoundedClaudeTaskId(value: string): boolean { - return value.length > 0 && value.length <= MAX_TASK_ID_LENGTH -} - -/** The task's canonical, resume-stable id. Shared with the subagent roster so - * both readers of this channel agree on what identifies a task. */ -export function claudeTaskId(message: Record): string | null { - const value = message.task_id - return typeof value === 'string' && isBoundedClaudeTaskId(value) ? value : null -} - -/** A task's human label, collapsed and bounded. */ -export function claudeTaskDescription(value: unknown): string | undefined { - if (typeof value !== 'string') { - return undefined - } - const trimmed = value.trim().replace(/\s+/g, ' ') - return trimmed.length > 0 ? trimmed.slice(0, MAX_TASK_DESCRIPTION_LENGTH) : undefined -} - -export function classifyClaudeBackgroundTaskKind(taskType: unknown): ClaudeBackgroundTaskKind { - switch (taskType) { - case 'local_agent': - return 'agent' - case 'local_workflow': - return 'workflow' - case 'local_bash': - return 'command' - case 'monitor': - return 'monitor' - default: - return 'unknown' - } -} export class ClaudeBackgroundTaskTracker { - private readonly tasks = new Map() + private readonly tasks = new Map() + private readonly retention = new ClaudeSettledBackgroundTasks() private readonly terminalTaskIds = new Set() private aggregateRosterObserved = false - private foregroundTurnActive = false private monitoring = false private publishedTasksFingerprint = '' + constructor(private readonly now: () => number = () => Date.now()) {} + get state(): AgentSessionBackgroundTaskState | null { if (!this.monitoring) { return null } return { state: 'monitoring', - tasks: this.backgroundTaskDetails() + tasks: this.backgroundTaskDetails(), + ...(this.retention.hasSettled ? { settledTasks: this.retention.settledDetails() } : {}) } } @@ -87,16 +63,24 @@ export class ClaudeBackgroundTaskTracker { } observe(message: Record, startsTurn = false): boolean { - if (startsTurn) { - this.foregroundTurnActive = true + // Background work publishes through a foreground turn: the strip stays + // honest mid-fan-out and the client alone decides when the idle-only + // monitoring label may speak. + // + // A new turn is the same evidence `result` is: nothing the previous turn + // left foreground is still that turn's work. CLEANUP ONLY — a row's + // visibility never consults `startsTurn`, which is Orca's own + // dispatch-correlation bookkeeping and false by design for undispatched + // turns, so a missed one degrades to the old behaviour and can never hide + // live work. + if (startsTurn || message.type === 'result') { + this.settleForegroundTasks() } - if (message.type === 'result') { - this.foregroundTurnActive = false - } else if (message.type === 'system') { + if (message.type === 'system') { if (!this.observeSystemFrame(message) && !startsTurn) { return false } - } else if (!startsTurn) { + } else if (!startsTurn && message.type !== 'result') { return false } return this.refreshMonitoring() @@ -104,47 +88,62 @@ export class ClaudeBackgroundTaskTracker { clear(): boolean { this.tasks.clear() + this.retention.clear() this.terminalTaskIds.clear() this.aggregateRosterObserved = false - this.foregroundTurnActive = false return this.refreshMonitoring() } + /** `result` is the outcome of every task the provider marked foreground, so + * they stop being live work. Backgrounded tasks outlive the turn and are + * never swept here — only their own terminal frame retires them. */ + private settleForegroundTasks(): void { + for (const task of this.tasks.values()) { + if (!task.backgrounded) { + task.liveInTurn = false + } + } + } + + private settle( + id: string, + state: AgentSessionBackgroundTaskRunState, + outcome: { totalTokens?: number } = {} + ): void { + this.retention.settle(id, state, outcome, this.tasks.get(id)) + this.finish(id) + } + private observeSystemFrame(message: Record): boolean { if (message.subtype === 'background_tasks_changed') { this.replaceAggregateRoster(message.tasks) return true } - const id = claudeTaskId(message) + const id = taskId(message) if (!id) { return false } if (message.subtype === 'task_notification') { - this.finish(id) + // The notification is affirmative terminal evidence even when its status + // field is unreadable — matching the liveness semantics this edge always had. + this.settle(id, terminalClaudeTaskRunState(message.status) ?? 'done', { + totalTokens: taskUsageTotalTokens(message) + }) + return true + } + if (message.subtype === 'task_progress') { + // Progress `description` is the current activity ("Running "), not + // the task's name — only usage (and a missing identity) may update. + const existing = this.tasks.get(id) + const totalTokens = taskUsageTotalTokens(message) + if (!existing?.backgrounded || totalTokens === undefined) { + return false + } + this.tasks.set(id, { ...existing, totalTokens, name: existing.name ?? taskName(message) }) return true } if (message.subtype === 'task_updated') { - const patch = record(message.patch) - if (!patch) { - return false - } - if (TERMINAL_TASK_STATES.has(String(patch.status))) { - this.finish(id) - return true - } - const existing = this.tasks.get(id) - if ( - (patch.is_backgrounded === true || claudeTaskDescription(patch.description)) && - (!this.aggregateRosterObserved || existing) - ) { - this.upsert(id, { - backgrounded: patch.is_backgrounded === true || existing?.backgrounded === true, - kind: existing?.kind ?? 'unknown', - description: claudeTaskDescription(patch.description) ?? existing?.description - }) - return true - } - return false + return this.observeTaskUpdated(id, message) } if (message.subtype !== 'task_started' || this.terminalTaskIds.has(id)) { return false @@ -153,56 +152,131 @@ export class ClaudeBackgroundTaskTracker { this.finish(id) return true } - if (this.aggregateRosterObserved && !this.tasks.has(id)) { + const kind = classifyClaudeBackgroundTaskKind(message.task_type) + const backgrounded = + message.is_backgrounded === true || kind === 'workflow' || kind === 'monitor' + // The aggregate roster enumerates BACKGROUND work only, so it is authoritative + // over that class alone. A foreground start it could never have listed is not + // stale evidence, and dropping it here silently killed foreground rows. + if (this.aggregateRosterObserved && backgrounded && !this.tasks.has(id)) { return false } - const kind = classifyClaudeBackgroundTaskKind(message.task_type) this.upsert(id, { - backgrounded: message.is_backgrounded === true || kind === 'workflow' || kind === 'monitor', + backgrounded, kind, - description: claudeTaskDescription(message.description) + description: taskDescription(message.description), + name: taskName(message), + state: liveClaudeTaskRunState(message.status) ?? undefined, + startedAt: this.now() }) return true } + private observeTaskUpdated(id: string, message: Record): boolean { + const patch = record(message.patch) + if (!patch) { + return false + } + const settledState = terminalClaudeTaskRunState(patch.status) + if (settledState) { + this.settle(id, settledState) + return true + } + const existing = this.tasks.get(id) + // Classification is re-derived per transition: a later frame that reveals a + // real type moves the task between buckets instead of pinning first-seen. + const patchKind = + 'task_type' in patch ? classifyClaudeBackgroundTaskKind(patch.task_type) : undefined + const liveState = liveClaudeTaskRunState(patch.status) + const hasContent = + patch.is_backgrounded === true || + taskDescription(patch.description) !== undefined || + taskName(patch) !== undefined || + liveState !== null || + (patchKind !== undefined && patchKind !== 'unknown') + if (hasContent && (!this.aggregateRosterObserved || existing)) { + this.upsert(id, { + backgrounded: patch.is_backgrounded === true || existing?.backgrounded === true, + kind: patchKind ?? existing?.kind ?? 'unknown', + description: taskDescription(patch.description), + name: taskName(patch), + state: liveState ?? undefined, + startedAt: this.now() + }) + return true + } + return false + } + private replaceAggregateRoster(value: unknown): void { if (!Array.isArray(value)) { return } + const prior = new Map(this.tasks) this.aggregateRosterObserved = true this.tasks.clear() - this.terminalTaskIds.clear() + const roster = new Map() for (const valueTask of value) { - if (this.tasks.size >= MAX_TRACKED_TASKS) { + if (roster.size >= MAX_TRACKED_TASKS) { break } const task = record(valueTask) if (!task || task.ambient === true) { continue } - const id = claudeTaskId(task) + const id = taskId(task) if (!id) { continue } - this.tasks.set(id, { + // An authoritative live roster supersedes an earlier terminal edge — for + // the ids it actually lists. Wiping the whole set left a finished + // FOREGROUND id undefended, since the start guard now convicts only + // backgrounded starts. + this.terminalTaskIds.delete(id) + const existing = prior.get(id) ?? this.retention.resume(id) + const kind = classifyClaudeBackgroundTaskKind(task.task_type) + roster.set(id, { backgrounded: true, - kind: classifyClaudeBackgroundTaskKind(task.task_type), - description: claudeTaskDescription(task.description) + liveInTurn: true, + kind: kind !== 'unknown' ? kind : (existing?.kind ?? 'unknown'), + description: taskDescription(task.description) ?? existing?.description, + name: taskName(task) ?? existing?.name, + state: liveClaudeTaskRunState(task.status) ?? existing?.state, + startedAt: existing?.startedAt ?? this.now(), + totalTokens: existing?.totalTokens }) } + // Live foreground work is not in a BACKGROUND roster and is not superseded + // by one. Budget counted up front so eviction drops the STALEST retained + // rows rather than the newest, and roster entries are never starved. + const retainable = [...prior].filter( + ([id, task]) => !task.backgrounded && task.liveInTurn && !roster.has(id) + ) + let evict = Math.max(0, roster.size + retainable.length - MAX_TRACKED_TASKS) + // Retained rows keep their own relative order and stay ahead of the roster, + // so a live row the user is reading does not drop below it when a roster + // frame lands. Within the roster the PROVIDER's order wins — including for + // a task it reports live again, which belongs where the provider lists it + // rather than appended after the rows that outlived it. + for (const [id, task] of retainable) { + if (evict > 0) { + evict -= 1 + continue + } + this.tasks.set(id, task) + } + for (const [id, task] of roster) { + this.tasks.set(id, task) + } + for (const [id, task] of prior) { + if (task.backgrounded && !this.tasks.has(id)) { + this.retention.rememberRemoved(id, task) + } + } } - private upsert(id: string, task: TrackedTask): void { - const existing = this.tasks.get(id) - if (existing) { - this.tasks.set(id, { - backgrounded: existing.backgrounded || task.backgrounded, - kind: existing.kind === 'unknown' ? task.kind : existing.kind, - description: task.description ?? existing.description - }) - return - } - if (this.tasks.size >= MAX_TRACKED_TASKS) { + private upsert(id: string, task: Omit): void { + if (!this.tasks.has(id) && this.tasks.size >= MAX_TRACKED_TASKS) { let foregroundId: string | undefined for (const [candidateId, candidate] of this.tasks) { if (!candidate.backgrounded) { @@ -215,7 +289,23 @@ export class ClaudeBackgroundTaskTracker { } this.tasks.delete(foregroundId) } - this.tasks.set(id, task) + const existing = this.tasks.get(id) ?? this.retention.resume(id) + this.terminalTaskIds.delete(id) + if (existing) { + this.tasks.set(id, { + backgrounded: existing.backgrounded || task.backgrounded, + // A settled foreground task is not revived by a late edge frame. + liveInTurn: existing.liveInTurn, + kind: task.kind !== 'unknown' ? task.kind : existing.kind, + description: task.description ?? existing.description, + name: task.name ?? existing.name, + state: task.state ?? existing.state, + startedAt: existing.startedAt, + totalTokens: existing.totalTokens + }) + return + } + this.tasks.set(id, { ...task, liveInTurn: true }) } private finish(id: string): void { @@ -231,9 +321,12 @@ export class ClaudeBackgroundTaskTracker { } private refreshMonitoring(): boolean { - const details = this.foregroundTurnActive ? [] : this.backgroundTaskDetails() + const details = this.backgroundTaskDetails() + if (details.length === 0 && this.retention.hasSettled) { + this.retention.flushSettled() + } const next = details.length > 0 - const fingerprint = next ? JSON.stringify(details) : '' + const fingerprint = next ? JSON.stringify([details, this.retention.settledDetails()]) : '' if (next === this.monitoring && fingerprint === this.publishedTasksFingerprint) { return false } @@ -245,14 +338,10 @@ export class ClaudeBackgroundTaskTracker { private backgroundTaskDetails(): AgentSessionBackgroundTask[] { const details: AgentSessionBackgroundTask[] = [] for (const [id, task] of this.tasks) { - if (!task.backgrounded) { + if (!task.backgrounded && !task.liveInTurn) { continue } - details.push({ - id, - kind: task.kind, - ...(task.description ? { description: task.description } : {}) - }) + details.push(claudeBackgroundTaskDetail(id, task)) } return details } diff --git a/src/main/claude/claude-settled-background-tasks.ts b/src/main/claude/claude-settled-background-tasks.ts new file mode 100644 index 00000000000..73064becec4 --- /dev/null +++ b/src/main/claude/claude-settled-background-tasks.ts @@ -0,0 +1,141 @@ +// Retention state for background tasks that have reached a terminal edge. +// +// The real producer settles a task in two steps inside one tick: +// `background_tasks_changed` arrives FIRST with the task already absent, then +// `task_updated` / `task_notification` carry the outcome. So the terminal edge +// must be able to settle a task the live roster no longer holds — that is what +// `rememberRemoved` preserves. A removal whose outcome frame never arrives +// simply vanishes: removed tasks are never rendered and never guessed into a +// finished state. + +import type { + AgentSessionBackgroundTask, + AgentSessionBackgroundTaskRunState +} from '../../shared/agent-session-wire' + +const MAX_RETAINED_TASKS = 256 + +export type TrackedClaudeBackgroundTask = { + backgrounded: boolean + /** Foreground work is turn-scoped: the provider's `result` (or the next turn + * starting) is its outcome, so it stays visible only until that frame. + * Backgrounded work ignores this and is retired only by its own edge. */ + liveInTurn: boolean + kind: AgentSessionBackgroundTask['kind'] + description?: string + name?: string + state?: AgentSessionBackgroundTaskRunState + /** First-observed epoch ms; preserved across updates and roster replacement + * so clients can render elapsed and keep a stable first-seen sort. */ + startedAt: number + totalTokens?: number +} + +export function claudeBackgroundTaskDetail( + id: string, + task: TrackedClaudeBackgroundTask +): AgentSessionBackgroundTask { + return { + id, + kind: task.kind, + ...(task.description ? { description: task.description } : {}), + ...(task.name ? { name: task.name } : {}), + state: task.state ?? (task.kind === 'monitor' ? 'monitoring' : 'working'), + startedAt: task.startedAt, + ...(task.totalTokens !== undefined ? { totalTokens: task.totalTokens } : {}), + // Only a backgrounded row has a stop the host can target; absent means yes. + ...(task.backgrounded ? {} : { stoppable: false }) + } +} + +function setBounded(map: Map, key: K, value: V): void { + map.delete(key) + map.set(key, value) + if (map.size > MAX_RETAINED_TASKS) { + const oldest = map.keys().next() + if (!oldest.done) { + map.delete(oldest.value) + } + } +} + +export class ClaudeSettledBackgroundTasks { + private readonly settled = new Map() + private readonly recentlyRemoved = new Map() + + /** An aggregate roster evicted a still-live backgrounded task; hold its + * details so the outcome frame trailing in the same tick can settle it. */ + rememberRemoved(id: string, task: TrackedClaudeBackgroundTask): void { + setBounded(this.recentlyRemoved, id, task) + } + + /** Terminal edge for `id`. `liveSource` is the live roster's entry when it + * still has one; otherwise the recently-removed copy is consumed. A second + * edge (updated, then notification) re-derives the settled state and can + * add the final usage the first edge lacked. */ + settle( + id: string, + state: AgentSessionBackgroundTaskRunState, + outcome: { totalTokens?: number }, + liveSource: TrackedClaudeBackgroundTask | undefined + ): void { + const source = liveSource ?? this.recentlyRemoved.get(id) + const already = this.settled.get(id) + if (source?.backgrounded) { + setBounded(this.settled, id, { + ...claudeBackgroundTaskDetail(id, { + ...source, + totalTokens: outcome.totalTokens ?? source.totalTokens + }), + state + }) + } else if (already) { + this.settled.set(id, { + ...already, + state, + ...(outcome.totalTokens !== undefined ? { totalTokens: outcome.totalTokens } : {}) + }) + } + this.recentlyRemoved.delete(id) + } + + /** Positive live evidence transfers identity back to the tracker, never the old outcome. */ + resume(id: string): TrackedClaudeBackgroundTask | undefined { + const settled = this.settled.get(id) + const removed = this.recentlyRemoved.get(id) + this.settled.delete(id) + this.recentlyRemoved.delete(id) + const source = settled ?? removed + if (!source || source.startedAt === undefined) { + return undefined + } + // Positive live evidence, so it re-enters live in this turn too; a resumed + // task is always backgrounded, which is what actually gates its visibility. + return { + ...source, + backgrounded: true, + liveInTurn: true, + state: undefined, + startedAt: source.startedAt + } + } + + get hasSettled(): boolean { + return this.settled.size > 0 + } + + settledDetails(): AgentSessionBackgroundTask[] { + return [...this.settled.values()] + } + + /** Settled context only makes sense beside live work; the strip exits at the + * same instant it always has — when the last live task ends. */ + flushSettled(): void { + this.settled.clear() + } + + clear(): void { + this.settled.clear() + this.recentlyRemoved.clear() + } +} diff --git a/src/main/claude/claude-slash-command-catalog.test.ts b/src/main/claude/claude-slash-command-catalog.test.ts index 20d79f9f53d..f4374e52e4f 100644 --- a/src/main/claude/claude-slash-command-catalog.test.ts +++ b/src/main/claude/claude-slash-command-catalog.test.ts @@ -86,8 +86,8 @@ it('accepts descriptor reloads, removing old skills while retaining terminal fil } expect(catalog.observe(reload)).toBe(true) expect(catalog.commands).toEqual([ - { name: 'clear', kind: 'command' }, - { name: 'new-skill', kind: 'skill' } + { name: 'clear', kind: 'command', description: 'Clear' }, + { name: 'new-skill', kind: 'skill', description: 'New' } ]) expect(catalog.observe(reload)).toBe(false) expect(catalog.observe({ ...reload, commands: [] })).toBe(true) @@ -130,3 +130,146 @@ it('publishes classification becoming authoritative even when the name and kind expect(catalog.observe(init({ slash_commands: ['clear'], skills: [] }))).toBe(true) expect(catalog.commands).toEqual([{ name: 'clear', kind: 'command' }]) }) + +it('keeps the description and argument hint a descriptor report authored', () => { + const catalog = new ClaudeSlashCommandCatalog(undefined, { + commands: [ + { name: 'goal', description: 'Set or view the goal', argumentHint: '' }, + { name: 'quiet', description: '', argumentHint: '' } + ] + }) + expect(catalog.commands).toEqual([ + { + name: 'goal', + kind: 'command', + kindUnspecified: true, + description: 'Set or view the goal', + argumentHint: '' + }, + { name: 'quiet', kind: 'command', kindUnspecified: true } + ]) +}) + +it('bounds the row text a provider can put in the picker', () => { + const catalog = new ClaudeSlashCommandCatalog(undefined, { + commands: [ + { name: 'long', description: 'x'.repeat(201), argumentHint: 'y'.repeat(101) }, + { name: 'wrong-type', description: 42, argumentHint: { text: 'no' } }, + { name: 'blank', description: ' ' }, + { name: 'long-whitespace', description: `Visible${' '.repeat(201)}` }, + { name: 'wrapped', description: 'first line\n second line' } + ] + }) + expect(catalog.commands).toEqual([ + { name: 'long', kind: 'command', kindUnspecified: true }, + { name: 'wrong-type', kind: 'command', kindUnspecified: true }, + { name: 'blank', kind: 'command', kindUnspecified: true }, + { name: 'long-whitespace', kind: 'command', kindUnspecified: true }, + { + name: 'wrapped', + kind: 'command', + kindUnspecified: true, + description: 'first line second line' + } + ]) +}) + +it('does not let malformed descriptor names consume the command detail budget', () => { + const catalog = new ClaudeSlashCommandCatalog(undefined, { + commands: [ + ...Array.from({ length: 512 }, (_, index) => ({ + name: `invalid name ${index}`, + description: 'Rejected with its name' + })), + { name: 'goal', description: 'Set or view the goal', argumentHint: '' } + ] + }) + expect(catalog.commands).toEqual([ + { + name: 'goal', + kind: 'command', + kindUnspecified: true, + description: 'Set or view the goal', + argumentHint: '' + } + ]) +}) + +it('combines non-empty fields from duplicate descriptors without discarding earlier text', () => { + const catalog = new ClaudeSlashCommandCatalog(undefined, { + commands: [ + { name: 'goal', description: 'Set or view the goal' }, + { name: 'goal', argumentHint: '' } + ] + }) + expect(catalog.commands).toEqual([ + { + name: 'goal', + kind: 'command', + kindUnspecified: true, + description: 'Set or view the goal', + argumentHint: '' + } + ]) +}) + +it('carries descriptor text across the name-only stream init that classifies it', () => { + const catalog = new ClaudeSlashCommandCatalog(undefined, { + commands: [ + { name: 'clear', description: 'Clear conversation' }, + { name: 'ref-oss', description: 'A skill' } + ] + }) + expect(catalog.observe(init({ slash_commands: ['clear', 'ref-oss'], skills: ['ref-oss'] }))).toBe( + true + ) + expect(catalog.commands).toEqual([ + { name: 'clear', kind: 'command', description: 'Clear conversation' }, + { name: 'ref-oss', kind: 'skill', description: 'A skill' } + ]) +}) + +it('reports a description-only change and lets a later report drop the text', () => { + const catalog = new ClaudeSlashCommandCatalog(init({ slash_commands: ['clear'], skills: [] })) + expect(catalog.commands).toEqual([{ name: 'clear', kind: 'command' }]) + const changed = { + type: 'system', + subtype: 'commands_changed', + commands: [{ name: 'clear', description: 'Clear conversation history' }] + } + expect(catalog.observe(changed)).toBe(true) + expect(catalog.commands).toEqual([ + { name: 'clear', kind: 'command', description: 'Clear conversation history' } + ]) + expect(catalog.observe(changed)).toBe(false) + expect(catalog.observe({ ...changed, commands: [{ name: 'clear' }] })).toBe(true) + expect(catalog.commands).toEqual([{ name: 'clear', kind: 'command' }]) +}) + +it('describes nothing when the session reported names only', () => { + expect(readClaudeSlashCommands(init())).toEqual([ + { name: 'clear', kind: 'command' }, + { name: 'ref-oss', kind: 'skill' }, + { name: 'opsx:apply', kind: 'command' } + ]) + expect(new ClaudeSlashCommandCatalog(init()).commands).toEqual([ + { name: 'clear', kind: 'command' }, + { name: 'ref-oss', kind: 'skill' }, + { name: 'opsx:apply', kind: 'command' } + ]) +}) + +it('still hides terminal-only names however well the provider describes them', () => { + const catalog = new ClaudeSlashCommandCatalog(init()) + expect( + catalog.observe({ + type: 'system', + subtype: 'commands_changed', + commands: [ + { name: 'doctor', description: 'Diagnose the CLI install' }, + { name: 'ref-oss', description: 'A skill' } + ] + }) + ).toBe(true) + expect(catalog.commands).toEqual([{ name: 'ref-oss', kind: 'skill', description: 'A skill' }]) +}) diff --git a/src/main/claude/claude-slash-command-catalog.ts b/src/main/claude/claude-slash-command-catalog.ts index b1f65d93d50..2a4d91260f0 100644 --- a/src/main/claude/claude-slash-command-catalog.ts +++ b/src/main/claude/claude-slash-command-catalog.ts @@ -3,6 +3,16 @@ import type { AgentSessionSlashCommand } from '../../shared/agent-session-wire' // Stream init carries name arrays; control initialization and reloads carry descriptors. const MAX_COMMANDS = 512 const MAX_NAME_LENGTH = 200 +const MAX_DESCRIPTION_LENGTH = 200 +const MAX_ARGUMENT_HINT_LENGTH = 100 + +/** The provider's own row text for one command, absent when it reported none. */ +type CommandDetail = Pick + +function commandName(value: unknown): string | undefined { + const name = typeof value === 'string' ? value.trim() : '' + return name.length > 0 && name.length <= MAX_NAME_LENGTH && !/\s/u.test(name) ? name : undefined +} function names(value: unknown): string[] { if (!Array.isArray(value)) { @@ -13,20 +23,61 @@ function names(value: unknown): string[] { if (seen.size >= MAX_COMMANDS) { break } - const name = typeof entry === 'string' ? entry.trim() : '' - if (name.length > 0 && name.length <= MAX_NAME_LENGTH && !/\s/u.test(name)) { + const name = commandName(entry) + if (name !== undefined) { seen.add(name) } } return [...seen] } -function descriptorNames(value: unknown): string[] { - return names( - Array.isArray(value) - ? value.map((entry) => (entry !== null && typeof entry === 'object' ? entry.name : undefined)) - : [] - ) +/** A single picker row's worth of provider text: unusable values are dropped, not truncated. */ +function rowText(value: unknown, maxLength: number): string | undefined { + if (typeof value !== 'string' || value.length > maxLength) { + return undefined + } + const collapsed = value.replace(/\s+/gu, ' ').trim() + return collapsed.length > 0 && collapsed.length <= maxLength ? collapsed : undefined +} + +function descriptorCatalog(value: unknown): { + names: string[] + details: Map +} { + const names: string[] = [] + const seen = new Set() + const details = new Map() + if (!Array.isArray(value)) { + return { names, details } + } + for (const entry of value) { + if (seen.size >= MAX_COMMANDS) { + break + } + if (entry === null || typeof entry !== 'object') { + continue + } + const name = commandName(entry.name) + if (name === undefined) { + continue + } + if (!seen.has(name)) { + seen.add(name) + names.push(name) + } + const previous = details.get(name) + const description = previous?.description ?? rowText(entry.description, MAX_DESCRIPTION_LENGTH) + const argumentHint = + previous?.argumentHint ?? rowText(entry.argumentHint, MAX_ARGUMENT_HINT_LENGTH) + if (description === undefined && argumentHint === undefined) { + continue + } + details.set(name, { + ...(description === undefined ? {} : { description }), + ...(argumentHint === undefined ? {} : { argumentHint }) + }) + } + return { names, details } } function carriesCommandCatalog(message: Record): boolean { @@ -56,6 +107,7 @@ export class ClaudeSlashCommandCatalog { private hasSkillClassification = false private hidden = new Set() private commandNames = new Set() + private details = new Map() constructor(initMessage?: Record, initialization?: unknown) { // SessionStart can prove acquisition before the first stream init exists. @@ -65,11 +117,15 @@ export class ClaudeSlashCommandCatalog { 'commands' in initialization && Array.isArray(initialization.commands) ) { - this.entries = descriptorNames(initialization.commands).map((name) => ({ - name, - kind: 'command', - kindUnspecified: true - })) + const catalog = descriptorCatalog(initialization.commands) + this.details = catalog.details + this.entries = this.describe( + catalog.names.map((name) => ({ + name, + kind: 'command', + kindUnspecified: true + })) + ) } if (initMessage) { this.observe(initMessage) @@ -80,13 +136,18 @@ export class ClaudeSlashCommandCatalog { return this.entries } + /** Provider row text, carried across the name-only frames that never restate it. */ + private describe(entries: AgentSessionSlashCommand[]): AgentSessionSlashCommand[] { + return entries.map((entry) => ({ ...entry, ...this.details.get(entry.name) })) + } + /** True when this frame replaced the catalog with a different one. */ observe(message: Record): boolean { let next: AgentSessionSlashCommand[] if (carriesCommandCatalog(message)) { this.hasSkillClassification = true this.hidden = new Set(names(message.terminal_slash_commands)) - next = readClaudeSlashCommands(message) + next = this.describe(readClaudeSlashCommands(message)) this.commandNames = new Set( next.filter((entry) => entry.kind === 'command').map((entry) => entry.name) ) @@ -95,13 +156,17 @@ export class ClaudeSlashCommandCatalog { message.subtype === 'commands_changed' && Array.isArray(message.commands) ) { - next = descriptorNames(message.commands) - .filter((name) => !this.hidden.has(name)) - .map((name) => - this.hasSkillClassification - ? { name, kind: this.commandNames.has(name) ? 'command' : 'skill' } - : { name, kind: 'command', kindUnspecified: true } - ) + const catalog = descriptorCatalog(message.commands) + this.details = catalog.details + next = this.describe( + catalog.names + .filter((name) => !this.hidden.has(name)) + .map((name) => + this.hasSkillClassification + ? { name, kind: this.commandNames.has(name) ? 'command' : 'skill' } + : { name, kind: 'command', kindUnspecified: true } + ) + ) } else { return false } @@ -112,7 +177,9 @@ export class ClaudeSlashCommandCatalog { (entry, index) => entry.name === this.entries?.[index]?.name && entry.kind === this.entries?.[index]?.kind && - entry.kindUnspecified === this.entries?.[index]?.kindUnspecified + entry.kindUnspecified === this.entries?.[index]?.kindUnspecified && + entry.description === this.entries?.[index]?.description && + entry.argumentHint === this.entries?.[index]?.argumentHint ) ) { return false diff --git a/src/main/claude/claude-stream-json-connection.ts b/src/main/claude/claude-stream-json-connection.ts index dd6bbc8a5eb..1a99bb92064 100644 --- a/src/main/claude/claude-stream-json-connection.ts +++ b/src/main/claude/claude-stream-json-connection.ts @@ -15,7 +15,10 @@ import { import { createClaudeChildTreeReaper, proveClaudeChildExit } from './claude-agent-sdk-exit-proof' import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' -import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' +import { + claudeUnwrittenUserMessageError, + createClaudeUserMessageQueue +} from './claude-agent-sdk-user-message-queue' import type { ClaudeStructuredSdkOptions } from './claude-structured-launch-resolution' export { ClaudeControlRequestError } @@ -236,7 +239,11 @@ export async function openClaudeStreamJsonConnection( const send = (message: Record): Promise => { if (closing || exited || terminalError || child.stdin.destroyed || !child.stdin.writable) { - return Promise.reject(terminalError ?? new Error('claude stream-json connection is closed')) + return Promise.reject( + claudeUnwrittenUserMessageError( + terminalError ?? new Error('claude stream-json connection is closed') + ) + ) } return inbox.push(message as unknown as SDKUserMessage) } diff --git a/src/main/claude/claude-structured-compaction.test.ts b/src/main/claude/claude-structured-compaction.test.ts index 46dac01f768..5255ce37ea1 100644 --- a/src/main/claude/claude-structured-compaction.test.ts +++ b/src/main/claude/claude-structured-compaction.test.ts @@ -1,6 +1,12 @@ -import { describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' import { StructuredSessionCompaction } from '../native-chat/agent-session-wire/structured-session-compaction' -import { isClaudeCompactionContent } from './claude-structured-compaction' +import { claudeUnwrittenUserMessageError } from './claude-agent-sdk-user-message-queue' +import { compactClaudeSession, isClaudeCompactionContent } from './claude-structured-compaction' +import { sessionFor } from './claude-structured-dispatch-test-support' + +afterEach(() => { + vi.useRealTimers() +}) describe('Claude compaction transcript content', () => { it('keeps generated summaries and command echoes out of the transcript only during explicit compaction', async () => { @@ -26,4 +32,35 @@ describe('Claude compaction transcript content', () => { await completion expect(isClaudeCompactionContent(tracker, event)).toBe(false) }) + + it('fails a provably unwritten command without waiting for the completion deadline', async () => { + vi.useFakeTimers() + const session = sessionFor( + vi.fn().mockRejectedValue(claudeUnwrittenUserMessageError(new Error('input closed'))) + ) + const pending = compactClaudeSession(session, new StructuredSessionCompaction(60_000), { + sessionId: 'orca-session', + fence: 1, + turnId: 'compact-1' + }) + + await vi.advanceTimersByTimeAsync(1) + + await expect(pending).resolves.toEqual({ error: 'provider_write_failed: input closed' }) + }) + + it('keeps waiting when the command write outcome is ambiguous', async () => { + vi.useFakeTimers() + const session = sessionFor(vi.fn().mockRejectedValue(new Error('input pump stopped'))) + const pending = compactClaudeSession(session, new StructuredSessionCompaction(10), { + sessionId: 'orca-session', + fence: 1, + turnId: 'compact-1' + }) + const rejection = expect(pending).rejects.toThrow('Compaction completion is unconfirmed.') + + await vi.advanceTimersByTimeAsync(10) + + await rejection + }) }) diff --git a/src/main/claude/claude-structured-compaction.ts b/src/main/claude/claude-structured-compaction.ts index a5bd3aa96e3..cda4652b6ca 100644 --- a/src/main/claude/claude-structured-compaction.ts +++ b/src/main/claude/claude-structured-compaction.ts @@ -2,25 +2,26 @@ import type { ClaudeSession, ClaudeStructuredSessionEvent } from './claude-struc import type { StructuredSessionCompaction } from '../native-chat/agent-session-wire/structured-session-compaction' import { dispatchClaudeTurn } from './claude-structured-dispatch' import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { dispatchDoubtProvesUndelivered } from '../native-chat/agent-session-journal/journal-dispatch-doubt-reasons' +/** Compaction needs no ack deadline of its own: `compactions.run` keeps its own + * 180s completion window and settles on Claude's terminal `result` frame, so + * the dispatch here only has to report a refusal to send. */ export function compactClaudeSession( session: ClaudeSession, compactions: StructuredSessionCompaction, - input: Parameters>[0], - timeoutMs: number + input: Parameters>[0] ): Promise<{ error?: string }> { return compactions.run( input.sessionId, session.providerSessionId, async () => { - const result = await dispatchClaudeTurn( - session, - { - clientMessageId: `compact-${input.fence}`, - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: '/compact' }] } - }, - timeoutMs - ) - if (result.state === 'rejected') { + const result = await dispatchClaudeTurn(session, { + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: '/compact' }] } + }) + if ( + result.state === 'rejected' || + (result.state === 'unknown' && dispatchDoubtProvesUndelivered(result.reason)) + ) { return { error: result.reason } } return undefined diff --git a/src/main/claude/claude-structured-dispatch-admission.test.ts b/src/main/claude/claude-structured-dispatch-admission.test.ts new file mode 100644 index 00000000000..6ed8f91f072 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch-admission.test.ts @@ -0,0 +1,123 @@ +// The contract the admission fix exists for: dispatch settles when the write +// completes, and nothing about elapsed time ever puts a message in doubt. + +import { describe, expect, it, vi } from 'vitest' +import { dispatchClaudeTurn, resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { + childExited, + sessionFor, + userMessage, + userReplayFrame +} from './claude-structured-dispatch-test-support' + +describe('Claude structured dispatch admission', () => { + it('settles a send queued behind a running turn when that turn starts, with no doubt in between', async () => { + vi.useFakeTimers() + try { + const session = sessionFor() + const settled = vi.fn() + const running = await dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) + const runningUuid = session.dispatchWaiters[0]!.sentUuid + expect(resolveClaudeReplayWaiter(session, userReplayFrame(runningUuid, 'one'), settled)).toBe( + true + ) + + // Queued while turn one is still running: Claude cannot echo it until that + // turn ends, so nothing about the wait is evidence of a delivery problem. + const queued = await dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'two' }]) + }) + const queuedUuid = session.dispatchWaiters[0]!.sentUuid + expect(running).toEqual({ state: 'admitted' }) + expect(queued).toEqual({ state: 'admitted' }) + + await vi.advanceTimersByTimeAsync(10 * 60_000) + expect(session.dispatchWaiters).toHaveLength(1) + expect(session.retiredDispatchWaiters).toHaveLength(0) + expect(settled).toHaveBeenCalledTimes(1) + + // Turn one ends and turn two starts: the echo lands and settles the send. + expect(resolveClaudeReplayWaiter(session, userReplayFrame(queuedUuid, 'two'), settled)).toBe( + true + ) + expect(settled).toHaveBeenLastCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: queuedUuid } + }) + expect(session.activeTurnId).toBe(queuedUuid) + } finally { + vi.useRealTimers() + } + }) + + it('returns as soon as the write completes, without awaiting the echo', async () => { + const session = sessionFor() + await expect( + dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) + ).resolves.toEqual({ state: 'admitted' }) + expect(session.connection.send).toHaveBeenCalledTimes(1) + // Still unacknowledged, and deliberately so: the waiter outlives the call. + expect(session.dispatchWaiters).toHaveLength(1) + expect(session.dispatchWaiters[0]!.settledUuid).toBeUndefined() + }) + + it('resolves every live waiter and retires it when the child exits', async () => { + const session = sessionFor() + await dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) + await dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'two' }]) + }) + expect(session.dispatchWaiters).toHaveLength(2) + + childExited(session) + + expect(session.dispatchWaiters).toHaveLength(0) + expect(session.retiredDispatchWaiters).toHaveLength(2) + expect(session.retiredDispatchWaiters.every((waiter) => waiter.retired === true)).toBe(true) + }) + + it('bounds pending replay identities instead of retaining an unbounded queue', async () => { + const session = sessionFor() + for (let index = 0; index < 64; index += 1) { + await expect( + dispatchClaudeTurn(session, { + clientMessageId: `client-${index}`, + body: userMessage([{ type: 'text', text: String(index) }]) + }) + ).resolves.toEqual({ state: 'admitted' }) + } + + await expect( + dispatchClaudeTurn(session, { + clientMessageId: 'client-over-capacity', + body: userMessage([{ type: 'text', text: 'one too many' }]) + }) + ).resolves.toEqual({ state: 'rejected', reason: 'claude structured dispatch queue is full' }) + expect(session.dispatchWaiters).toHaveLength(64) + expect(session.connection.send).toHaveBeenCalledTimes(64) + }) + + it('does not publish a journal settlement for a provider-control turn', async () => { + const session = sessionFor() + const settled = vi.fn() + await dispatchClaudeTurn(session, { + body: userMessage([{ type: 'text', text: '/compact' }]) + }) + const uuid = session.dispatchWaiters[0]!.sentUuid + + resolveClaudeReplayWaiter(session, userReplayFrame(uuid, '/compact'), settled) + + expect(settled).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-structured-dispatch-test-support.ts b/src/main/claude/claude-structured-dispatch-test-support.ts new file mode 100644 index 00000000000..83971a38d13 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch-test-support.ts @@ -0,0 +1,52 @@ +import { vi, type Mock } from 'vitest' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import { retireClaudeDispatchWaiters } from './claude-structured-dispatch' +import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' +import { ClaudeSlashCommandCatalog } from './claude-slash-command-catalog' + +export function sessionFor(send: Mock = vi.fn().mockResolvedValue(undefined)): ClaudeSession { + return { + connection: { send } as unknown as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), + commands: new ClaudeSlashCommandCatalog(), + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +export function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem { + return { kind: 'message', role: 'user', blocks } +} + +/** The child died. Nothing else retires a live waiter now that no deadline does. */ +export function childExited(session: ClaudeSession): void { + retireClaudeDispatchWaiters(session) +} + +export function userReplayFrame(uuid: string, text: string): Record { + return { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid, + message: { role: 'user', content: [{ type: 'text', text }] } + } +} diff --git a/src/main/claude/claude-structured-dispatch.test.ts b/src/main/claude/claude-structured-dispatch.test.ts index 933bd77774b..6ea5832b25d 100644 --- a/src/main/claude/claude-structured-dispatch.test.ts +++ b/src/main/claude/claude-structured-dispatch.test.ts @@ -2,104 +2,86 @@ import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { describe, expect, it, vi } from 'vitest' -import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' import { dispatchClaudeTurn, resolveClaudeReplayWaiter } from './claude-structured-dispatch' import { readClaudeImage } from './claude-structured-dispatch-content' +import { claudeUnwrittenUserMessageError } from './claude-agent-sdk-user-message-queue' import type { ClaudeSession } from './claude-structured-session-state' -import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' -import { ClaudeSlashCommandCatalog } from './claude-slash-command-catalog' - -function sessionFor(send = vi.fn().mockResolvedValue(undefined)): ClaudeSession { - return { - connection: { send } as unknown as ClaudeSession['connection'], - providerSessionId: 'provider-session', - claudeConfigDir: '/accounts/claude', - leafUuid: null, - fence: 1, - acquisitionGeneration: 'generation-1', - prompts: {} as ClaudeSession['prompts'], - dispatchWaiters: [], - retiredDispatchWaiters: [], - replayContentFallbackBlocked: false, - backgroundTasks: new ClaudeBackgroundTaskTracker(), - commands: new ClaudeSlashCommandCatalog(), - dispatchSequence: 0, - optionMutationSequence: 0, - options: new Map(), - reportedOptions: {}, - reportedModelMutation: 0, - confirmedOptions: new Set(), - restoreSkippedOptions: new Set(), - capabilities: [], - events: undefined, - translator: null - } -} - -function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem { - return { kind: 'message', role: 'user', blocks } -} - -function userReplayFrame(uuid: string, text: string): Record { - return { - type: 'user', - parent_tool_use_id: null, - session_id: 'provider-session', - uuid, - message: { role: 'user', content: [{ type: 'text', text }] } - } -} +import { + childExited, + sessionFor, + userMessage, + userReplayFrame +} from './claude-structured-dispatch-test-support' describe('Claude structured dispatch image limits', () => { it.each(['isMeta', 'isSynthetic', 'isCompactSummary'])( 'does not acknowledge a dispatch with %s context even when the client uuid matches', async (flag) => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/example' }]) }, - 1000 - ) + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: '/example' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const sentUuid = session.dispatchWaiters[0]!.sentUuid const replay = userReplayFrame(sentUuid, '/example') - expect(resolveClaudeReplayWaiter(session, { ...replay, [flag]: true })).toBe(false) + expect(resolveClaudeReplayWaiter(session, { ...replay, [flag]: true }, settled)).toBe(false) expect(session.dispatchWaiters).toHaveLength(1) - expect(resolveClaudeReplayWaiter(session, replay)).toBe(true) - await expect(dispatched).resolves.toMatchObject({ - state: 'accepted', - providerIdentity: { uuid: sentUuid } + expect(settled).not.toHaveBeenCalled() + expect(resolveClaudeReplayWaiter(session, replay, settled)).toBe(true) + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-1', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: sentUuid } }) } ) - it('recovers the active identity when a timed-out replay arrives late', async () => { + it('takes the active turn identity from a replay that lands after dispatch returned', async () => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, - 500 - ) + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid - await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(session.activeTurnId).toBeUndefined() expect(resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid!, 'one'))).toBe(true) expect(session.activeTurnId).toBe(sentUuid) expect(session.activeTurnSequence).toBe(session.dispatchSequence) }) - it('settles the send a timed-out replay proves was delivered', async () => { + it('recovers the active identity when a replay lands after the child died', async () => { const session = sessionFor() - const settled = vi.fn() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, - 500 - ) + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid - await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + childExited(session) + expect(session.dispatchWaiters).toHaveLength(0) + expect(session.retiredDispatchWaiters).toHaveLength(1) + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid!, 'one'))).toBe(true) + expect(session.activeTurnId).toBe(sentUuid) + expect(session.activeTurnSequence).toBe(session.dispatchSequence) + }) + + it('settles the send the replay proves was delivered, whenever it arrives', async () => { + const session = sessionFor() + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid!, 'one'), settled) expect(settled).toHaveBeenCalledWith({ @@ -111,20 +93,19 @@ describe('Claude structured dispatch image limits', () => { it('settles a superseded dispatch even though it no longer owns the turn identity', async () => { const session = sessionFor() const settled = vi.fn() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, - 500 - ) + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const firstUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, - 100 - ) + const second = dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'two' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const secondUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid @@ -138,51 +119,62 @@ describe('Claude structured dispatch image limits', () => { clientMessageId: 'client-1', providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: firstUuid } }) - resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'), settled) - await expect(second).resolves.toMatchObject({ state: 'accepted' }) - expect(settled).toHaveBeenCalledTimes(1) + expect(resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'), settled)).toBe( + true + ) + await expect(second).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenLastCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: secondUuid } + }) }) it('never lets a late replay for dispatch A resolve dispatch B', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, - 500 - ) + const settled = vi.fn() + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const firstUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, - 100 - ) - await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect( + dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'two' }]) + }) + ).resolves.toEqual({ state: 'admitted' }) const secondUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid expect(resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one'))).toBe(false) expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) - expect(resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'))).toBe(true) - await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + expect(resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'), settled)).toBe( + true + ) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: secondUuid } + }) }) it('does not let an identical late replay for dispatch A resolve active dispatch B', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, - 500 - ) + const settled = vi.fn() + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, - 100 - ) + const second = dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const secondUuid = session.dispatchWaiters[0]!.sentUuid @@ -191,34 +183,35 @@ describe('Claude structured dispatch image limits', () => { ) expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) - resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) - await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt'), settled) + await expect(second).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: secondUuid } + }) }) it('does not let a fresh-UUID replay for an evicted dispatch resolve active dispatch B', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, - 100 - ) + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid const fillerDispatches = await Promise.all( Array.from({ length: 64 }, (_, index) => - dispatchClaudeTurn( - session, - { - clientMessageId: `filler-${index}`, - body: userMessage([{ type: 'text', text: 'same prompt' }]) - }, - 5 - ) + dispatchClaudeTurn(session, { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }) ) ) - expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(fillerDispatches.every((outcome) => outcome.state === 'admitted')).toBe(true) + childExited(session) expect(session.retiredDispatchWaiters).toHaveLength(64) expect(session.replayContentFallbackBlocked).toBe(true) expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( @@ -231,47 +224,48 @@ describe('Claude structured dispatch image limits', () => { } expect(session.retiredDispatchWaiters).toHaveLength(0) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, - 100 - ) + const second = dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const secondUuid = session.dispatchWaiters[0]!.sentUuid + const settled = vi.fn() expect( resolveClaudeReplayWaiter(session, userReplayFrame('provider-a-late', 'same prompt')) ).toBe(false) expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) - resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) - await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt'), settled) + await expect(second).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: secondUuid } + }) }) it('does not let a fresh-UUID result for an evicted slash dispatch resolve active dispatch B', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 100 - ) + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid const fillerDispatches = await Promise.all( Array.from({ length: 64 }, (_, index) => - dispatchClaudeTurn( - session, - { - clientMessageId: `filler-${index}`, - body: userMessage([{ type: 'text', text: '/permissions' }]) - }, - 5 - ) + dispatchClaudeTurn(session, { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) ) ) - expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(fillerDispatches.every((outcome) => outcome.state === 'admitted')).toBe(true) + childExited(session) expect(session.retiredDispatchWaiters).toHaveLength(64) expect(session.replayContentFallbackBlocked).toBe(true) expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( @@ -292,13 +286,13 @@ describe('Claude structured dispatch image limits', () => { } expect(session.retiredDispatchWaiters).toHaveLength(0) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 100 - ) + const second = dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const secondUuid = session.dispatchWaiters[0]!.sentUuid + const settled = vi.fn() expect( resolveClaudeReplayWaiter(session, { @@ -311,70 +305,127 @@ describe('Claude structured dispatch image limits', () => { expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) expect( - resolveClaudeReplayWaiter(session, { - type: 'result', - subtype: 'success', - session_id: 'provider-session', - uuid: 'result-b', - user_message_uuid: secondUuid - }) + resolveClaudeReplayWaiter( + session, + { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }, + settled + ) ).toBe(false) - await expect(second).resolves.toMatchObject({ - providerIdentity: { uuid: 'result-b' } + await expect(second).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: 'result-b' } }) }) it('does not let a legacy result for timed-out ordinary dispatch A resolve slash dispatch B', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'ordinary' }]) }, - 100 - ) + const settled = vi.fn() + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'ordinary' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 100 - ) + const second = dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) expect( - resolveClaudeReplayWaiter(session, { - type: 'result', - subtype: 'success', - session_id: 'provider-session', - uuid: 'legacy-result-a' - }) + resolveClaudeReplayWaiter( + session, + { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'legacy-result-a' + }, + settled + ) ).toBe(false) - await expect(second).resolves.toMatchObject({ state: 'unknown' }) + await expect(second).resolves.toEqual({ state: 'admitted' }) + // Ambiguous, so it settles nothing: the slash waiter is still waiting. + expect(session.dispatchWaiters).toHaveLength(1) + expect(settled).not.toHaveBeenCalled() }) it('removes only its own waiter when a later send fails', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, - 100 - ) + const settled = vi.fn() + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const firstWaiter = session.dispatchWaiters[0] - session.connection.send = vi.fn().mockRejectedValue(new Error('broken pipe')) + session.connection.send = vi + .fn() + .mockRejectedValue(claudeUnwrittenUserMessageError(new Error('broken pipe'))) + // A refused write is a transport fact, and the only thing besides child exit + // that puts one message's delivery in doubt. await expect( - dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, - 100 - ) - ).resolves.toMatchObject({ state: 'unknown', reason: 'broken pipe' }) + dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: 'two' }]) + }) + ).resolves.toEqual({ state: 'unknown', reason: 'provider_write_failed: broken pipe' }) expect(session.dispatchWaiters).toEqual([firstWaiter]) const firstUuid = (firstWaiter as { sentUuid?: string }).sentUuid - resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one')) - await expect(first).resolves.toMatchObject({ providerIdentity: { uuid: firstUuid } }) + resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one'), settled) + await expect(first).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-1', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: firstUuid } + }) + }) + + it('does not let a provably unwritten attempt block retry correlation', async () => { + const send = vi + .fn() + .mockRejectedValueOnce(claudeUnwrittenUserMessageError(new Error('broken pipe'))) + .mockResolvedValue(undefined) + const session = sessionFor(send) + const body = userMessage([{ type: 'text', text: 'retry me' }]) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }) + ).resolves.toEqual({ state: 'unknown', reason: 'provider_write_failed: broken pipe' }) + expect(session.dispatchWaiters).toHaveLength(0) + expect(session.retiredDispatchWaiters).toHaveLength(0) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }) + ).resolves.toEqual({ state: 'admitted' }) + expect(resolveClaudeReplayWaiter(session, userReplayFrame('fresh-replay', 'retry me'))).toBe( + true + ) + expect(session.activeTurnId).toBe('fresh-replay') + }) + + it('does not claim an SDK-pulled frame was unwritten when its write outcome is ambiguous', async () => { + const session = sessionFor(vi.fn().mockRejectedValue(new Error('input pump stopped'))) + + await expect( + dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) + ).resolves.toEqual({ + state: 'unknown', + reason: 'provider_write_outcome_unknown: input pump stopped' + }) }) it('keeps a replay accepted before its send reports failure', async () => { @@ -386,35 +437,39 @@ describe('Claude structured dispatch image limits', () => { session = sessionFor(send) await expect( - dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, - 100 - ) + dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'one' }]) + }) ).resolves.toMatchObject({ state: 'accepted', providerIdentity: { uuid: 'turn-race' } }) expect(session.dispatchWaiters).toHaveLength(0) }) it('accepts a slash command from its result receipt when Claude omits the user replay', async () => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 100 - ) + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) expect( - resolveClaudeReplayWaiter(session, { - type: 'result', - subtype: 'success', - session_id: 'provider-session', - uuid: 'command-result-uuid' - }) + resolveClaudeReplayWaiter( + session, + { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'command-result-uuid' + }, + settled + ) ).toBe(false) - await expect(dispatched).resolves.toEqual({ - state: 'accepted', + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-1', providerIdentity: { provider: 'claude', sessionId: 'provider-session', @@ -425,32 +480,38 @@ describe('Claude structured dispatch image limits', () => { it('accepts a slash command sent with an attachment from its result receipt', async () => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { - clientMessageId: 'client-1', - body: userMessage([ - { type: 'text', text: '/permissions' }, - { type: 'image-ref', url: 'https://example.test/a.png' } - ]) - }, - 100 - ) + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([ + { type: 'text', text: '/permissions' }, + { type: 'image-ref', url: 'https://example.test/a.png' } + ]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) // The mapper moves the image ahead of the prompt, so Claude runs the command and replies // with a result receipt instead of a user replay. expect( - resolveClaudeReplayWaiter(session, { - type: 'result', - subtype: 'success', - session_id: 'provider-session', - uuid: 'command-result-uuid' - }) + resolveClaudeReplayWaiter( + session, + { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'command-result-uuid' + }, + settled + ) ).toBe(false) - await expect(dispatched).resolves.toMatchObject({ - state: 'accepted', - providerIdentity: { uuid: 'command-result-uuid' } + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-1', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'command-result-uuid' + } }) // The sent order is the fix: the waiter's verdict alone was already what it is today. expect(session.connection.send).toHaveBeenCalledWith( @@ -468,68 +529,76 @@ describe('Claude structured dispatch image limits', () => { it('does not take a result receipt for leading whitespace Claude never reads as a command', async () => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { - clientMessageId: 'client-1', - body: userMessage([{ type: 'text', text: ' /permissions' }]) - }, - 100 - ) + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: ' /permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) expect( - resolveClaudeReplayWaiter(session, { - type: 'result', - subtype: 'success', - session_id: 'provider-session', - uuid: 'unrelated-result-uuid' - }) + resolveClaudeReplayWaiter( + session, + { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'unrelated-result-uuid' + }, + settled + ) ).toBe(false) - await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(session.dispatchWaiters).toHaveLength(1) + expect(settled).not.toHaveBeenCalled() }) - it('correlates a later slash-command result by user_message_uuid despite a timed-out slash waiter', async () => { + it('correlates a later slash-command result by user_message_uuid despite a retired slash waiter', async () => { const session = sessionFor() - const first = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 500 - ) + const settled = vi.fn() + const first = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) - await expect(first).resolves.toMatchObject({ state: 'unknown' }) + await expect(first).resolves.toEqual({ state: 'admitted' }) + childExited(session) - const second = dispatchClaudeTurn( - session, - { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 500 - ) + const second = dispatchClaudeTurn(session, { + clientMessageId: 'client-2', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const secondUuid = session.dispatchWaiters[0]!.sentUuid expect( - resolveClaudeReplayWaiter(session, { - type: 'result', - subtype: 'success', - session_id: 'provider-session', - uuid: 'result-b', - user_message_uuid: secondUuid - }) + resolveClaudeReplayWaiter( + session, + { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }, + settled + ) ).toBe(false) - await expect(second).resolves.toMatchObject({ - state: 'accepted', - providerIdentity: { uuid: 'result-b' } + await expect(second).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-2', + providerIdentity: { provider: 'claude', sessionId: 'provider-session', uuid: 'result-b' } }) }) it('does not mistake a normal turn result for its missing user replay', async () => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'hello' }]) }, - 100 - ) + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: 'hello' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) expect( @@ -541,31 +610,40 @@ describe('Claude structured dispatch image limits', () => { ).toBe(false) expect(session.dispatchWaiters).toHaveLength(1) expect( - resolveClaudeReplayWaiter(session, { - type: 'user', - parent_tool_use_id: null, - session_id: 'provider-session', - uuid: 'user-replay-uuid', - message: { - role: 'user', - content: [{ type: 'text', text: 'hello' }] - } - }) + resolveClaudeReplayWaiter( + session, + { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: 'hello' }] + } + }, + settled + ) ).toBe(true) - await expect(dispatched).resolves.toMatchObject({ - state: 'accepted', - providerIdentity: { uuid: 'user-replay-uuid' } + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-1', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'user-replay-uuid' + } }) }) it('ignores a top-level tool-result user frame while waiting for a slash command replay', async () => { const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, - 100 - ) + const settled = vi.fn() + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'text', text: '/permissions' }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) resolveClaudeReplayWaiter(session, { @@ -580,19 +658,24 @@ describe('Claude structured dispatch image limits', () => { }) expect(session.dispatchWaiters).toHaveLength(1) - resolveClaudeReplayWaiter(session, { - type: 'user', - parent_tool_use_id: null, - session_id: 'provider-session', - uuid: 'user-replay-uuid', - message: { - role: 'user', - content: [{ type: 'text', text: '/permissions' }] - } - }) + resolveClaudeReplayWaiter( + session, + { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: '/permissions' }] + } + }, + settled + ) - await expect(dispatched).resolves.toEqual({ - state: 'accepted', + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + clientMessageId: 'client-1', providerIdentity: { provider: 'claude', sessionId: 'provider-session', @@ -611,7 +694,7 @@ describe('Claude structured dispatch image limits', () => { ) await expect( - dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }) ).resolves.toEqual({ state: 'rejected', reason: 'Claude messages support at most 20 images' }) expect(session.connection.send).not.toHaveBeenCalled() }) @@ -630,7 +713,7 @@ describe('Claude structured dispatch image limits', () => { const body = userMessage(paths.map((path) => ({ type: 'image-ref' as const, path }))) await expect( - dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }) ).resolves.toEqual({ state: 'rejected', reason: `Claude images must total no more than ${20 * 1024 * 1024} bytes` @@ -650,7 +733,7 @@ describe('Claude structured dispatch image limits', () => { const body = userMessage([{ type: 'image-ref', path }]) await expect( - dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }) ).resolves.toEqual({ state: 'rejected', reason: `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` @@ -668,11 +751,10 @@ describe('Claude structured dispatch image limits', () => { const path = join(directory, 'small.png') await writeFile(path, Buffer.alloc(64)) const session = sessionFor() - const dispatched = dispatchClaudeTurn( - session, - { clientMessageId: 'client-1', body: userMessage([{ type: 'image-ref', path }]) }, - 100 - ) + const dispatched = dispatchClaudeTurn(session, { + clientMessageId: 'client-1', + body: userMessage([{ type: 'image-ref', path }]) + }) await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid resolveClaudeReplayWaiter(session, { @@ -684,7 +766,7 @@ describe('Claude structured dispatch image limits', () => { ] } }) - await expect(dispatched).resolves.toMatchObject({ state: 'accepted' }) + await expect(dispatched).resolves.toEqual({ state: 'admitted' }) expect(allocUnsafe).toHaveBeenCalled() expect(allocUnsafe.mock.calls.some(([size]) => size === 64 + 1)).toBe(true) expect(allocUnsafe.mock.calls.some(([size]) => size >= 5 * 1024 * 1024)).toBe(false) @@ -694,7 +776,7 @@ describe('Claude structured dispatch image limits', () => { } }) - it('bounds retained waiter identity bytes when image dispatches time out', async () => { + it('bounds retained waiter identity bytes when image dispatches are retired', async () => { const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) try { const path = join(directory, 'large.png') @@ -703,9 +785,10 @@ describe('Claude structured dispatch image limits', () => { const body = userMessage([{ type: 'image-ref', path }]) await Promise.all( Array.from({ length: 64 }, (_, index) => - dispatchClaudeTurn(session, { clientMessageId: `client-${index}`, body }, 1) + dispatchClaudeTurn(session, { clientMessageId: `client-${index}`, body }) ) ) + childExited(session) expect(session.retiredDispatchWaiters).toHaveLength(64) const retainedKeyBytes = session.retiredDispatchWaiters.reduce( diff --git a/src/main/claude/claude-structured-dispatch.ts b/src/main/claude/claude-structured-dispatch.ts index 3080b5479e4..7533102c7aa 100644 --- a/src/main/claude/claude-structured-dispatch.ts +++ b/src/main/claude/claude-structured-dispatch.ts @@ -15,10 +15,16 @@ import { claudeDispatchInvokesSlashCommand, claudeDispatchMessageContent } from './claude-structured-dispatch-content' +import { + dispatchWriteFailureReason, + dispatchWriteOutcomeUnknownReason +} from '../native-chat/agent-session-journal/journal-dispatch-doubt-reasons' +import { claudeUserMessageWasProvablyUnwritten } from './claude-agent-sdk-user-message-queue' const MAX_RETIRED_DISPATCH_WAITERS = 64 +const MAX_ACTIVE_DISPATCH_WAITERS = 64 -/** A dispatch whose ack window expired, proven delivered by this replay. */ +/** Directly settles provider-proven delivery; the durable replay row independently reconciles it. */ export type ClaudeLateDispatchSettlement = (input: { clientMessageId: string providerIdentity: AgentJournalItemIdentity @@ -55,7 +61,7 @@ export function resolveClaudeReplayWaiter( (candidate) => candidate.sentUuid === userMessageUuid ) if (exact) { - settleWaiter(session, exact, uuid) + settleWaiter(session, exact, uuid, onSettledLate) return isUserReplay && exact.dispatchSequence === session.dispatchSequence } const retired = session.retiredDispatchWaiters.find( @@ -70,7 +76,7 @@ export function resolveClaudeReplayWaiter( const exact = session.dispatchWaiters.find((candidate) => candidate.sentUuid === uuid) if (exact) { - settleWaiter(session, exact, uuid) + settleWaiter(session, exact, uuid, onSettledLate) return isUserReplay && exact.dispatchSequence === session.dispatchSequence } const retired = session.retiredDispatchWaiters.find((candidate) => candidate.sentUuid === uuid) @@ -90,7 +96,7 @@ export function resolveClaudeReplayWaiter( (candidate) => candidate.replayContentKey === replayContentKey ) if (compatible.length === 1) { - settleWaiter(session, compatible[0]!, uuid) + settleWaiter(session, compatible[0]!, uuid, onSettledLate) return compatible[0]!.dispatchSequence === session.dispatchSequence } } else if (!session.replayContentFallbackBlocked && session.dispatchWaiters.length === 0) { @@ -120,22 +126,36 @@ export function resolveClaudeReplayWaiter( } const waiter = uuid ? session.dispatchWaiters.shift() : undefined if (waiter && uuid) { - clearTimeout(waiter.timer) - waiter.settledUuid = uuid - waiter.resolve(uuid) + settleWaiter(session, waiter, uuid, onSettledLate) return isUserReplay } return false } -function settleWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter, uuid: string): void { +function settleWaiter( + session: ClaudeSession, + waiter: ClaudeDispatchWaiter, + uuid: string, + onSettledLate?: ClaudeLateDispatchSettlement +): void { const index = session.dispatchWaiters.indexOf(waiter) if (index !== -1) { session.dispatchWaiters.splice(index, 1) } - clearTimeout(waiter.timer) waiter.settledUuid = uuid waiter.resolve(uuid) + // Dispatch returned on admission. Settle delivery unfenced while the sequence + // still fences which turn owns the identity; see `recoverLateIdentity`. + if (waiter.clientMessageId) { + onSettledLate?.({ + clientMessageId: waiter.clientMessageId, + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + }) + } + if (waiter.dispatchSequence === session.dispatchSequence) { + session.activeTurnId = uuid + session.activeTurnSequence = waiter.dispatchSequence + } } function forgetRetiredWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { @@ -158,10 +178,12 @@ function recoverLateIdentity( // The provider acted on this dispatch, so the send it came from is delivered. // Unfenced on purpose: the dispatch-sequence check below only decides which // turn owns the identity, while delivery is settled for good either way. - onSettledLate?.({ - clientMessageId: waiter.clientMessageId, - providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } - }) + if (waiter.clientMessageId) { + onSettledLate?.({ + clientMessageId: waiter.clientMessageId, + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + }) + } if (waiter.dispatchSequence === session.dispatchSequence) { session.activeTurnId = uuid session.activeTurnSequence = waiter.dispatchSequence @@ -169,13 +191,19 @@ function recoverLateIdentity( return isUserReplay && waiter.dispatchSequence === session.dispatchSequence } +/** + * A waiter with no deadline. The echo Claude sends is emitted when the provider + * STARTS the turn, so a message queued behind a running turn cannot be echoed + * until that turn ends — an interval bounded only by the previous turn. Elapsed + * time is therefore not evidence about delivery, and nothing here expires. + * Waiters are retired by process facts instead: a failed write, or child exit. + */ function waitForReplay( session: ClaudeSession, - timeoutMs: number, acceptsResult: boolean, sentUuid: string, replayContentKey: string, - clientMessageId: string + clientMessageId: string | null ): { waiter: ClaudeDispatchWaiter; promise: Promise } { let waiter!: ClaudeDispatchWaiter const promise = new Promise((resolve) => { @@ -185,28 +213,22 @@ function waitForReplay( sentUuid, dispatchSequence: session.dispatchSequence, replayContentKey, - resolve, - timer: setTimeout(() => { - const index = session.dispatchWaiters.indexOf(waiter) - if (index !== -1) { - session.dispatchWaiters.splice(index, 1) - } - retireWaiter(session, waiter) - resolve(null) - }, timeoutMs) + resolve } - waiter.timer.unref?.() session.dispatchWaiters.push(waiter) }) return { waiter, promise } } -function retireWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { +function forgetWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { const index = session.dispatchWaiters.indexOf(waiter) if (index !== -1) { session.dispatchWaiters.splice(index, 1) } - clearTimeout(waiter.timer) +} + +function retireWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + forgetWaiter(session, waiter) if (!waiter.retired) { waiter.retired = true session.retiredDispatchWaiters.push(waiter) @@ -220,10 +242,19 @@ function retireWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): voi } } +/** Nothing expires a waiter, so the child's death is what ends every live one. + * Retired rather than dropped: their identities stay joinable, bounded by + * `MAX_RETIRED_DISPATCH_WAITERS`. */ +export function retireClaudeDispatchWaiters(session: ClaudeSession): void { + for (const waiter of session.dispatchWaiters.splice(0)) { + retireWaiter(session, waiter) + waiter.resolve(null) + } +} + export async function dispatchClaudeTurn( session: ClaudeSession, - input: { clientMessageId: string; body: AgentJournalMessageItem }, - timeoutMs: number + input: { clientMessageId?: string; body: AgentJournalMessageItem } ): Promise { let content: unknown[] try { @@ -231,6 +262,9 @@ export async function dispatchClaudeTurn( } catch (error) { return { state: 'rejected', reason: (error as Error).message } } + if (session.dispatchWaiters.length >= MAX_ACTIVE_DISPATCH_WAITERS) { + return { state: 'rejected', reason: 'claude structured dispatch queue is full' } + } const dispatchSequence = ++session.dispatchSequence // Read the sent content, not the journal blocks: only the mapped trailing prompt decides // whether Claude runs a command, so the two cannot disagree about which frame settles this. @@ -238,11 +272,10 @@ export async function dispatchClaudeTurn( const sentUuid = randomUUID() const replay = waitForReplay( session, - timeoutMs, acceptsResult, sentUuid, claudeDispatchContentKey(content), - input.clientMessageId + input.clientMessageId ?? null ) const replayed = replay.promise try { @@ -266,21 +299,24 @@ export async function dispatchClaudeTurn( } } } - if (!waiter.retired) { + const provablyUnwritten = claudeUserMessageWasProvablyUnwritten(error) + if (provablyUnwritten) { + forgetWaiter(session, waiter) + forgetRetiredWaiter(session, waiter) + waiter.resolve(null) + } else if (!waiter.retired) { retireWaiter(session, waiter) waiter.resolve(null) } - return { state: 'unknown', reason: (error as Error).message } + return { + state: 'unknown', + reason: provablyUnwritten + ? dispatchWriteFailureReason(error) + : dispatchWriteOutcomeUnknownReason(error) + } } - const uuid = await replayed - if (uuid) { - session.activeTurnId = uuid - session.activeTurnSequence = dispatchSequence - } - return uuid - ? { - state: 'accepted', - providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } - } - : { state: 'unknown', reason: 'claude accepted a message but did not replay its uuid in time' } + // The write is the admission signal. Awaiting the echo here would block on the + // turn already running, which is why the deadline this replaces kept declaring + // doubt about messages that were delivered. `settleWaiter` finishes the job. + return { state: 'admitted' } } diff --git a/src/main/claude/claude-structured-item-translation.ts b/src/main/claude/claude-structured-item-translation.ts index 1c1673b59cd..0fdb231f6ef 100644 --- a/src/main/claude/claude-structured-item-translation.ts +++ b/src/main/claude/claude-structured-item-translation.ts @@ -179,6 +179,7 @@ export function claudeToolBody(input: { kind: 'tool-call', name: input.tool.name, input: input.tool.input, + callId: input.tool.id, state: input.result ? (input.result.failed ? 'failed' : 'completed') : 'running', ...(input.result ? { output: boundInlineText(input.result.output, DEFAULT_JOURNAL_PAYLOAD_LIMITS).bounded } diff --git a/src/main/claude/claude-structured-journal-translation-turn-timing.test.ts b/src/main/claude/claude-structured-journal-translation-turn-timing.test.ts new file mode 100644 index 00000000000..12d9619ab3d --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation-turn-timing.test.ts @@ -0,0 +1,260 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { + StructuredAgentSessionAppendOptions, + StructuredAgentSessionEventSink +} from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { readAgentJournalTurn } from '../../shared/agent-session-turn-record' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' +import { acquired, fakeClaude } from './claude-structured-session-test-support' + +type Append = { + identity: AgentJournalItemIdentity + body: AgentJournalItemBody + options: StructuredAgentSessionAppendOptions | undefined +} + +function sinkState() { + const items: Append[] = [] + const tombstones: AgentJournalItemIdentity[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body, options) => items.push({ identity, body, options }), + appendTombstone: (identity) => tombstones.push(identity), + publish: vi.fn() + } + const lifecycle = () => + items.flatMap((item) => { + const turn = readAgentJournalTurn(item.body) + return turn ? [{ ...turn, options: item.options }] : [] + }) + return { sink, items, tombstones, lifecycle } +} + +function userTurn(uuid: string, observedAt?: number): ClaudeStructuredSessionEvent { + return { + type: 'message', + sessionId: 'orca-session', + startsTurn: true, + ...(observedAt === undefined ? {} : { observedAt }), + message: { + type: 'user', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { role: 'user', content: [{ type: 'text', text: 'go' }] } + } + } +} + +function result( + observedAt?: number, + fields: Record = {} +): ClaudeStructuredSessionEvent { + return { + type: 'message', + sessionId: 'orca-session', + ...(observedAt === undefined ? {} : { observedAt }), + message: { + type: 'result', + subtype: 'success', + uuid: 'result-1', + session_id: 'claude-session', + is_error: false, + result: 'done', + ...fields + } + } +} + +const USER_1_KEY = 'claude:claude-session:user-1' + +describe('Claude structured turn timing', () => { + afterEach(() => { + vi.useRealTimers() + }) + + it('stamps the running row with the host start time as both startedAt and row ts', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + + expect(state.lifecycle()).toEqual([ + { + turnId: 'user-1', + state: 'running', + startedAt: 1_000, + userItemId: USER_1_KEY, + options: { observedAt: 1_000 } + } + ]) + }) + + it('keys the running row to the user echo that opened the turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + + // The echo itself is never a row; its provider key is what the submission adopted. + expect(state.items.some((item) => item.body.kind === 'message')).toBe(false) + expect(state.lifecycle().at(-1)?.userItemId).toBe(USER_1_KEY) + }) + + it('revises the running row to completed with the result receipt time', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + translator.handle(result(4_500)) + + expect(state.tombstones).toEqual([]) + expect(state.lifecycle().at(-1)).toEqual({ + turnId: 'user-1', + state: 'completed', + startedAt: 1_000, + completedAt: 4_500, + userItemId: USER_1_KEY, + options: {} + }) + expect(state.items.at(-1)?.identity).toEqual(state.items[0]?.identity) + }) + + it('carries the provider-measured duration onto the completed row', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + translator.handle(result(4_500, { duration_ms: 3_210 })) + + expect(state.lifecycle().at(-1)).toMatchObject({ + state: 'completed', + durationMs: 3_210, + userItemId: USER_1_KEY + }) + }) + + it('omits durationMs when the result reports none', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + translator.handle(result(4_500)) + + expect(state.lifecycle().at(-1)).not.toHaveProperty('durationMs') + }) + + it('revises an open turn to interrupted when the session ends without a result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + translator.handle({ + type: 'ended', + sessionId: 'orca-session', + reason: 'child exited', + observedAt: 2_250 + }) + + expect(state.tombstones).toEqual([]) + expect(state.lifecycle().at(-1)).toMatchObject({ + turnId: 'user-1', + state: 'interrupted', + startedAt: 1_000, + completedAt: 2_250 + }) + }) + + it('interrupts the open turn at the receipt time of the turn that replaces it', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 1_000)) + translator.handle(userTurn('user-2', 3_000)) + + expect(state.lifecycle().map(({ options: _options, ...row }) => row)).toEqual([ + { turnId: 'user-1', state: 'running', startedAt: 1_000, userItemId: USER_1_KEY }, + { + turnId: 'user-1', + state: 'interrupted', + startedAt: 1_000, + completedAt: 3_000, + userItemId: USER_1_KEY + }, + { + turnId: 'user-2', + state: 'running', + startedAt: 3_000, + userItemId: 'claude:claude-session:user-2' + } + ]) + }) + + it('falls back to the host clock when an event carries no receipt time', () => { + vi.useFakeTimers({ now: 50_000 }) + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1')) + vi.setSystemTime(56_000) + translator.handle(result()) + + expect(state.lifecycle().at(-1)).toMatchObject({ + state: 'completed', + startedAt: 50_000, + completedAt: 56_000 + }) + }) + + it('acquisition stamps turn boundaries from the host clock, never the frame timestamp', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const dispatch = adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'ship it' }] }, + fence: 7 + }) + await Promise.resolve() + const connection = claude.connections[0]! + connection.handlers.onMessage?.({ + ...connection.sent[0], + uuid: 'turn-1', + timestamp: '2001-01-01T00:00:00.000Z' + }) + await dispatch + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'assistant-1', + session_id: connection.sent[0]?.session_id, + timestamp: '2001-01-01T00:00:01.000Z', + message: { role: 'assistant', content: [{ type: 'text', text: 'ok' }] } + }) + connection.handlers.onMessage?.({ + type: 'result', + subtype: 'success', + uuid: 'result-1', + session_id: connection.sent[0]?.session_id, + timestamp: '2001-01-01T00:00:02.000Z', + is_error: false, + result: 'ok' + }) + + const messages = events.filter( + (event) => event.type === 'message' && event.message.type !== 'system' + ) + expect(messages).toEqual([ + expect.objectContaining({ startsTurn: true, observedAt: 1_700_000_000_500 }), + expect.not.objectContaining({ observedAt: expect.anything() }), + expect.objectContaining({ + message: expect.objectContaining({ type: 'result' }), + observedAt: 1_700_000_000_500 + }) + ]) + }) +}) diff --git a/src/main/claude/claude-structured-journal-translation.test.ts b/src/main/claude/claude-structured-journal-translation.test.ts index 951872f0c69..7a8ecf54344 100644 --- a/src/main/claude/claude-structured-journal-translation.test.ts +++ b/src/main/claude/claude-structured-journal-translation.test.ts @@ -20,6 +20,7 @@ import { DEFAULT_JOURNAL_PAYLOAD_LIMITS } from '../native-chat/agent-session-journal/journal-payload-bounds' import type { ClaudePendingPrompt } from './claude-structured-prompt-replies' +import { readAgentJournalTurn } from '../../shared/agent-session-turn-record' import { createClaudeJournalTranslator } from './claude-structured-journal-translation' function sinkState() { @@ -33,6 +34,17 @@ function sinkState() { return { sink, items, tombstones } } +/** `[recordId, state]` of every lifecycle append, in journal order. */ +function lifecycleAppends( + items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] +) { + return items.flatMap((item) => { + const identity = item.identity + const turn = identity.provider === 'legacy' ? readAgentJournalTurn(item.body) : null + return turn && identity.provider === 'legacy' ? [[identity.recordId, turn.state]] : [] + }) +} + function message( type: 'assistant' | 'user', uuid: string, @@ -343,14 +355,14 @@ describe('Claude structured journal translation', () => { item.body.kind === 'message' && item.body.role === 'user' ? [item.body.blocks] : [] ) ).toEqual([]) - expect( - state.items.some((item) => item.body.kind === 'status' && !item.body.turnLifecycle) - ).toBe(false) - expect( - state.tombstones.flatMap((identity) => - identity.provider === 'legacy' ? [identity.recordId] : [] - ) - ).toEqual(['turn-lifecycle:user-replay-1', 'turn-lifecycle:user-interrupt']) + expect(state.items.some((item) => item.body.kind === 'status')).toBe(false) + expect(state.tombstones).toEqual([]) + expect(lifecycleAppends(state.items)).toEqual([ + ['turn-lifecycle:user-replay-1', 'running'], + ['turn-lifecycle:user-replay-1', 'completed'], + ['turn-lifecycle:user-interrupt', 'running'], + ['turn-lifecycle:user-interrupt', 'interrupted'] + ]) }) it('does not reopen a completed turn when the SDK replays its user row after restart', () => { @@ -371,11 +383,9 @@ describe('Claude structured journal translation', () => { liveTranslator.handle({ ...replay, startsTurn: true }) liveTranslator.handle(resultFrame('success', { is_error: false, result: '' })) - expect(live.tombstones).toContainEqual({ - provider: 'legacy', - agent: 'claude', - sessionId: 'claude-session', - recordId: 'turn-lifecycle:picker-command-1' + expect(live.items.at(-1)).toMatchObject({ + identity: { provider: 'legacy', recordId: 'turn-lifecycle:picker-command-1' }, + body: { kind: 'turn', turnId: 'picker-command-1', state: 'completed' } }) liveTranslator.dispose() @@ -419,11 +429,7 @@ describe('Claude structured journal translation', () => { text: 'API Error: 529 upstream overloaded' }) // The turn still settles: the error is an extra row, not a stuck lifecycle. - expect( - state.tombstones.flatMap((identity) => - identity.provider === 'legacy' ? [identity.recordId] : [] - ) - ).toEqual(['turn-lifecycle:user-1']) + expect(lifecycleAppends(state.items).at(-1)).toEqual(['turn-lifecycle:user-1', 'completed']) }) it('drops the stream state of turns that ended without their final frame', () => { @@ -522,14 +528,13 @@ describe('Claude structured journal translation', () => { expect(keyed.get('orca:claude-tool%3Aclaude-session%3Atool-1')).toMatchObject({ kind: 'tool-call', name: 'Bash', + callId: 'tool-1', state: 'completed', output: { head: 'a.ts\nb.ts', truncated: false } }) - expect( - state.items.some( - (item) => item.body.kind === 'status' && item.body.turnLifecycle?.turnId === 'user-1' - ) - ).toBe(true) + expect(state.items.some((item) => readAgentJournalTurn(item.body)?.turnId === 'user-1')).toBe( + true + ) translator.handle( message( @@ -542,6 +547,7 @@ describe('Claude structured journal translation', () => { expect(state.items.at(-1)?.body).toMatchObject({ kind: 'tool-call', name: 'tool', + callId: 'tool-1', input: null, output: { head: 'done again' } }) @@ -551,11 +557,8 @@ describe('Claude structured journal translation', () => { sessionId: 'orca-session', message: { type: 'result', session_id: 'claude-session', uuid: 'result-1' } }) - expect(state.tombstones.at(-1)).toMatchObject({ - provider: 'legacy', - agent: 'claude', - recordId: 'turn-lifecycle:user-1' - }) + expect(state.tombstones).toEqual([]) + expect(lifecycleAppends(state.items).at(-1)).toEqual(['turn-lifecycle:user-1', 'completed']) }) it('bounds persisted thinking text to the shared journal payload limit', () => { @@ -566,8 +569,11 @@ describe('Claude structured journal translation', () => { translator.handle(message('assistant', 'assistant-thinking', [{ type: 'thinking', thinking }])) expect(state.items.at(-1)?.body).toEqual({ - kind: 'status', - text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + kind: 'message', + role: 'reasoning', + blocks: [ + { type: 'text', text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text } + ] }) }) @@ -582,9 +588,11 @@ describe('Claude structured journal translation', () => { ) expect(state.items.at(-1)?.body).toEqual({ - kind: 'status', - text: 'Claude is working…', - turnLifecycle: { turnId: 'user-image', state: 'running' } + kind: 'turn', + turnId: 'user-image', + state: 'running', + startedAt: expect.any(Number), + userItemId: 'claude:claude-session:user-image' }) }) @@ -603,14 +611,11 @@ describe('Claude structured journal translation', () => { ]) expect(state.items[0]?.body).toMatchObject({ kind: 'tool-call', + callId: 'tool-1', state: 'completed', output: { head: 'done' } }) - expect( - state.items.some( - (item) => item.body.kind === 'status' && item.body.turnLifecycle !== undefined - ) - ).toBe(false) + expect(state.items.some((item) => readAgentJournalTurn(item.body) !== null)).toBe(false) }) it('paints nothing for a user frame that carries no content', () => { diff --git a/src/main/claude/claude-structured-journal-translation.ts b/src/main/claude/claude-structured-journal-translation.ts index 759c2926763..8b71149cba2 100644 --- a/src/main/claude/claude-structured-journal-translation.ts +++ b/src/main/claude/claude-structured-journal-translation.ts @@ -39,6 +39,12 @@ import { import { ClaudeSubagentRoster } from './claude-subagent-roster' import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' +import { + claudeTurnEndForResult, + claudeTurnLifecycleItem, + type ClaudeCurrentTurn, + type ClaudeTurnEnd +} from './claude-turn-lifecycle-item' export type ClaudeJournalTranslatorDeps = { sink: StructuredAgentSessionEventSink @@ -71,23 +77,14 @@ export function createClaudeSessionJournalTranslator( : null } -function lifecycleIdentity(sessionId: string, turnId: string): AgentJournalItemIdentity { - return { - provider: 'legacy', - agent: 'claude', - sessionId, - recordId: `turn-lifecycle:${turnId}` - } -} - export function createClaudeJournalTranslator( deps: ClaudeJournalTranslatorDeps ): ClaudeJournalTranslator { const tools = new Map() const promptItems = new Map() const streamedBlocks = createClaudeStreamedBlockRegistry() - let currentTurn: { sessionId: string; turnId: string } | null = null - const groupKeyOf = (turn: { sessionId: string; turnId: string } | null): string | null => + let currentTurn: ClaudeCurrentTurn | null = null + const groupKeyOf = (turn: ClaudeCurrentTurn | null): string | null => turn ? `${turn.sessionId}:${turn.turnId}` : null const providerFallback = createClaudeProviderFrameFallback( deps.sink, @@ -106,21 +103,11 @@ export function createClaudeJournalTranslator( } }) - const publishLifecycle = (sessionId: string, turnId: string, running: boolean): void => { - const identity = lifecycleIdentity(sessionId, turnId) - if (running) { - deps.sink.appendItem(identity, { - kind: 'status', - text: 'Claude is working…', - turnLifecycle: { turnId, state: 'running' } - }) - } else { - deps.sink.appendTombstone(identity) - } + const publishLifecycle = (turn: ClaudeCurrentTurn, end?: ClaudeTurnEnd): void => { + const item = claudeTurnLifecycleItem(turn, end) + deps.sink.appendItem(item.identity, item.body, item.options) // Preserve first-work evidence when completion arrives before the journal drains. - deps.sink.publish({ - coalescingKey: running ? `turn-start:${sessionId}:${turnId}` : 'publish' - }) + deps.sink.publish({ coalescingKey: item.publishCoalescingKey }) } const publishActivity = (kind: string, payload: unknown): void => { @@ -142,7 +129,11 @@ export function createClaudeJournalTranslator( return true } - const handleMessage = (message: Record, startsTurn: boolean): boolean => { + const handleMessage = ( + message: Record, + startsTurn: boolean, + observedAt: number + ): boolean => { const envelope = readClaudeMessageEnvelope(message) if (!envelope) { return false @@ -189,8 +180,11 @@ export function createClaudeJournalTranslator( const thinking = claudeThinkingText(outputEnvelope) if (thinking) { deps.sink.appendItem(claudeThinkingIdentity(envelope.sessionId, envelope.uuid), { - kind: 'status', - text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + kind: 'message', + role: 'reasoning', + blocks: [ + { type: 'text', text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text } + ] }) changed = true } @@ -205,10 +199,16 @@ export function createClaudeJournalTranslator( // A new turn starting is the only end the previous one gets when its // result never arrives; settling it later would sweep THIS turn. subagents.settleTurn(groupKeyOf(currentTurn)) - publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + publishLifecycle(currentTurn, { state: 'interrupted', completedAt: observedAt }) } - currentTurn = { sessionId: envelope.sessionId, turnId: envelope.uuid } - publishLifecycle(envelope.sessionId, envelope.uuid, true) + currentTurn = { + sessionId: envelope.sessionId, + turnId: envelope.uuid, + startedAt: observedAt, + // A user echo lands on its own message identity, so this is the user row's key. + userItemId: agentJournalItemKey(identity) + } + publishLifecycle(currentTurn) deps.sink.setActivity?.(null) } if (changed) { @@ -248,7 +248,11 @@ export function createClaudeJournalTranslator( // No event will ever settle a child once the provider is gone. subagents.settleSession() if (currentTurn) { - publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + // The host saw the child end, so the turn's end is observed, not lost. + publishLifecycle(currentTurn, { + state: 'interrupted', + completedAt: event.observedAt ?? Date.now() + }) currentTurn = null } deps.sink.setActivity?.(null) @@ -271,7 +275,10 @@ export function createClaudeJournalTranslator( // reported as working will never be settled by an event. subagents.settleTurn(groupKeyOf(currentTurn)) if (currentTurn) { - publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + publishLifecycle( + currentTurn, + claudeTurnEndForResult(event.message, event.observedAt ?? Date.now()) + ) currentTurn = null } deps.sink.setActivity?.(null) @@ -291,7 +298,9 @@ export function createClaudeJournalTranslator( // fallback below still drops the raw frame instead of printing an opcode. subagents.observeSystemFrame(event.message) const kind = claudeProviderFrameKind(event.message) - if (!handleMessage(event.message, event.startsTurn === true)) { + if ( + !handleMessage(event.message, event.startsTurn === true, event.observedAt ?? Date.now()) + ) { providerFallback.append(kind, event.message) } publishActivity(kind, event.message) diff --git a/src/main/claude/claude-structured-session-acquisition.ts b/src/main/claude/claude-structured-session-acquisition.ts index 20ddde819b9..57bad1f1873 100644 --- a/src/main/claude/claude-structured-session-acquisition.ts +++ b/src/main/claude/claude-structured-session-acquisition.ts @@ -121,12 +121,16 @@ export async function acquireClaudeSession({ deps.onDispatchSettledLate?.({ sessionId, ...settlement }) ) : false + // Turn endpoints are stamped on the host clock, never the frame's own timestamp. + const observedAt = + startsTurn || message.type === 'result' ? { observedAt: deps.now?.() ?? Date.now() } : {} callbacks.deliver(attempt, sessionId, () => callbacks.emit(liveSession, input.events, { type: 'message', sessionId, message, - ...(startsTurn ? { startsTurn: true } : {}) + ...(startsTurn ? { startsTurn: true } : {}), + ...observedAt }) ) } diff --git a/src/main/claude/claude-structured-session-adapter-turns.test.ts b/src/main/claude/claude-structured-session-adapter-turns.test.ts new file mode 100644 index 00000000000..fc5eaa6b66e --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter-turns.test.ts @@ -0,0 +1,261 @@ +// What the adapter reports for one turn: how a dispatch is admitted and named, +// and which turn a cancellation is allowed to interrupt. + +import { describe, expect, it, vi } from 'vitest' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { claudeUnwrittenUserMessageError } from './claude-agent-sdk-user-message-queue' +import { + acquired, + fakeClaude, + PROVIDER_SESSION_ID, + USER_MESSAGE +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter turns and controls', () => { + it("admits a dispatch on the write and names it from Claude's replay", async () => { + const claude = fakeClaude({ replayUuid: 'user-provider-uuid' }) + const settled = vi.fn() + const adapter = await acquired(claude, {}, [], settled) + + const result = await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + + expect(result).toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + sessionId: 'session-1', + clientMessageId: 'client-1', + providerIdentity: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + uuid: 'user-provider-uuid' + } + }) + expect(claude.connections[0].sent[0]).toMatchObject({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'ship it' }] }, + session_id: PROVIDER_SESSION_ID + }) + }) + + it('does not put delivery in doubt while no replay uuid has arrived', async () => { + const settled = vi.fn() + const adapter = await acquired(fakeClaude({ replayUuid: null }), {}, [], settled) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toEqual({ state: 'admitted' }) + expect(settled).not.toHaveBeenCalled() + }) + + it('puts delivery in doubt only when the write itself fails', async () => { + const claude = fakeClaude({ replayUuid: null }) + const adapter = await acquired(claude) + claude.connections[0]!.send = async () => { + throw claudeUnwrittenUserMessageError(new Error('broken pipe')) + } + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toEqual({ state: 'unknown', reason: 'provider_write_failed: broken pipe' }) + }) + + it('requires an acknowledged interrupt and supports controlled options', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'sonnet', fence: 7 }) + ).resolves.toEqual({ model: 'sonnet' }) + expect(claude.connections[0].calls.slice(-2)).toEqual([ + { subtype: 'interrupt', params: {} }, + { subtype: 'set_model', params: { model: 'sonnet' } } + ]) + + claude.routes.interrupt = () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-2', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + + claude.routes.interrupt = () => { + throw new Error('claude interrupt request timed out') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-3', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('does not let a delayed cancellation for an earlier turn interrupt the later turn', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', 'turn-U'] }) + const adapter = await acquired(claude) + + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 6 }) + ).resolves.toEqual({ cancelled: false }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 1 + ) + }) + + it('does not cancel an acknowledged turn after a later dispatch is still unacknowledged', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', null] }) + const settled = vi.fn() + const adapter = await acquired(claude, {}, [], settled) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toEqual({ state: 'admitted' }) + expect(settled).toHaveBeenCalledWith({ + sessionId: 'session-1', + clientMessageId: 'client-T', + providerIdentity: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, uuid: 'turn-T' } + }) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toEqual({ state: 'admitted' }) + expect(claude.connections[0].sent).toHaveLength(2) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + }) + + it('classifies provider-declined options without treating timeouts as settled', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new ClaudeControlRequestError('set_model', 'model unavailable') + } + } + }) + const adapter = await acquired(claude) + + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'fable', fence: 7 }) + ).rejects.toMatchObject({ name: 'AgentSessionOptionRejectedError' }) + claude.routes.set_model = () => { + throw new Error('claude set_model request timed out') + } + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'opus', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('hydrates live model choices and maps the resolved current model to its CLI id', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { + list_models: () => [ + { value: 'default', resolvedModel: 'claude-opus-5', displayName: 'Default' }, + { + value: 'opus', + resolvedModel: 'claude-opus-5', + displayName: 'Opus', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet' + } + ] + } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toEqual({ + models: [ + { + id: 'opus', + label: 'Opus', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }, + { id: 'sonnet', label: 'Sonnet', isDefault: false, efforts: [] } + ], + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + }) + + it('keeps the shared Claude seed when live model discovery is unavailable', async () => { + const claude = fakeClaude({ + initModel: 'custom-model', + routes: { + list_models: () => { + throw new Error('unsupported') + } + } + }) + const adapter = await acquired(claude) + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + + expect(result.models.map((model) => model.id)).toEqual([ + 'fable', + 'opus', + 'sonnet', + 'haiku', + 'custom-model' + ]) + expect(result.current).toEqual({ + model: 'custom-model', + effort: 'high', + confirmed: ['model', 'effort'] + }) + }) +}) diff --git a/src/main/claude/claude-structured-session-adapter.test.ts b/src/main/claude/claude-structured-session-adapter.test.ts index 859ba9a8ed8..44a76f8402a 100644 --- a/src/main/claude/claude-structured-session-adapter.test.ts +++ b/src/main/claude/claude-structured-session-adapter.test.ts @@ -144,10 +144,11 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { expect(claude.connections[0]?.closeCount).toBe(1) }) - it('recovers a cancellable lifecycle when a timed-out replay arrives late', async () => { + it('recovers a cancellable lifecycle when the replay arrives after dispatch returned', async () => { const claude = fakeClaude({ replayUuid: null }) const events: ClaudeStructuredSessionEvent[] = [] - const adapter = await acquired(claude, {}, events) + const settled = vi.fn() + const adapter = await acquired(claude, {}, events, settled) await expect( adapter.dispatch({ @@ -156,7 +157,7 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { body: USER_MESSAGE, fence: 7 }) - ).resolves.toMatchObject({ state: 'unknown' }) + ).resolves.toEqual({ state: 'admitted' }) const sent = claude.connections[0]!.sent[0]! claude.connections[0]!.handlers.onMessage?.({ ...sent, @@ -170,6 +171,15 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { message: expect.objectContaining({ uuid: 'late-turn-1' }) }) ) + expect(settled).toHaveBeenCalledWith({ + sessionId: 'session-1', + clientMessageId: 'client-1', + providerIdentity: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + uuid: 'late-turn-1' + } + }) await expect( adapter.cancelTurn({ sessionId: 'session-1', turnId: 'late-turn-1', fence: 7 }) ).resolves.toEqual({ cancelled: true }) @@ -178,7 +188,8 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { it('quarantines SDK frames without the acquired session identity', async () => { const claude = fakeClaude({ replayUuid: null }) const events: ClaudeStructuredSessionEvent[] = [] - const adapter = await acquired(claude, {}, events) + const settled = vi.fn() + const adapter = await acquired(claude, {}, events, settled) const connection = claude.connections[0]! connection.handlers.onMessage?.({ @@ -193,13 +204,14 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } }) - const dispatch = adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-1', - body: USER_MESSAGE, - fence: 7 - }) - await Promise.resolve() + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toEqual({ state: 'admitted' }) expect(connection.sent).toHaveLength(1) connection.handlers.onMessage?.({ ...connection.sent[0], @@ -208,14 +220,20 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { }) await Promise.resolve() expect(events.filter((event) => event.type === 'message')).toHaveLength(1) + expect(settled).not.toHaveBeenCalled() connection.handlers.onMessage?.({ ...connection.sent[0], session_id: PROVIDER_SESSION_ID }) - await expect(dispatch).resolves.toMatchObject({ - state: 'accepted', - providerIdentity: { uuid: connection.sent[0]!.uuid } + expect(settled).toHaveBeenCalledWith({ + sessionId: 'session-1', + clientMessageId: 'client-1', + providerIdentity: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + uuid: connection.sent[0]!.uuid + } }) }) @@ -382,231 +400,6 @@ describe('ClaudeStructuredSessionAdapter.acquire', () => { }) }) -describe('ClaudeStructuredSessionAdapter turns and controls', () => { - it('accepts a dispatch only after Claude replays its provider uuid', async () => { - const claude = fakeClaude({ replayUuid: 'user-provider-uuid' }) - const adapter = await acquired(claude) - - const result = await adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-1', - body: USER_MESSAGE, - fence: 7 - }) - - expect(result).toEqual({ - state: 'accepted', - providerIdentity: { - provider: 'claude', - sessionId: PROVIDER_SESSION_ID, - uuid: 'user-provider-uuid' - } - }) - expect(claude.connections[0].sent[0]).toMatchObject({ - type: 'user', - message: { role: 'user', content: [{ type: 'text', text: 'ship it' }] }, - session_id: PROVIDER_SESSION_ID - }) - }) - - it('leaves delivery unconfirmed when no replay uuid arrives', async () => { - const adapter = await acquired(fakeClaude({ replayUuid: null })) - await expect( - adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-1', - body: USER_MESSAGE, - fence: 7 - }) - ).resolves.toMatchObject({ state: 'unknown' }) - }) - - it('requires an acknowledged interrupt and supports controlled options', async () => { - const claude = fakeClaude() - const adapter = await acquired(claude) - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-1', fence: 7 }) - ).resolves.toEqual({ cancelled: true }) - await expect( - adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'sonnet', fence: 7 }) - ).resolves.toEqual({ model: 'sonnet' }) - expect(claude.connections[0].calls.slice(-2)).toEqual([ - { subtype: 'interrupt', params: {} }, - { subtype: 'set_model', params: { model: 'sonnet' } } - ]) - - claude.routes.interrupt = () => { - throw new ClaudeControlRequestError('interrupt', 'not running') - } - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-2', fence: 7 }) - ).resolves.toEqual({ cancelled: false }) - - claude.routes.interrupt = () => { - throw new Error('claude interrupt request timed out') - } - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-3', fence: 7 }) - ).rejects.toThrow('timed out') - }) - - it('does not let a delayed cancellation for an earlier turn interrupt the later turn', async () => { - const claude = fakeClaude({ replayUuids: ['turn-T', 'turn-U'] }) - const adapter = await acquired(claude) - - await adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-T', - body: USER_MESSAGE, - fence: 7 - }) - await adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-U', - body: USER_MESSAGE, - fence: 7 - }) - - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) - ).resolves.toEqual({ cancelled: false }) - expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( - 0 - ) - - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 6 }) - ).resolves.toEqual({ cancelled: false }) - - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 7 }) - ).resolves.toEqual({ cancelled: true }) - expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( - 1 - ) - }) - - it('does not cancel an acknowledged turn after a later dispatch returns unknown', async () => { - const claude = fakeClaude({ replayUuids: ['turn-T', null] }) - const adapter = await acquired(claude) - - await expect( - adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-T', - body: USER_MESSAGE, - fence: 7 - }) - ).resolves.toMatchObject({ - state: 'accepted', - providerIdentity: { uuid: 'turn-T' } - }) - await expect( - adapter.dispatch({ - sessionId: 'session-1', - clientMessageId: 'client-U', - body: USER_MESSAGE, - fence: 7 - }) - ).resolves.toMatchObject({ state: 'unknown' }) - expect(claude.connections[0].sent).toHaveLength(2) - - await expect( - adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) - ).resolves.toEqual({ cancelled: false }) - expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( - 0 - ) - }) - - it('classifies provider-declined options without treating timeouts as settled', async () => { - const claude = fakeClaude({ - routes: { - set_model: () => { - throw new ClaudeControlRequestError('set_model', 'model unavailable') - } - } - }) - const adapter = await acquired(claude) - - await expect( - adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'fable', fence: 7 }) - ).rejects.toMatchObject({ name: 'AgentSessionOptionRejectedError' }) - claude.routes.set_model = () => { - throw new Error('claude set_model request timed out') - } - await expect( - adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'opus', fence: 7 }) - ).rejects.toThrow('timed out') - }) - - it('hydrates live model choices and maps the resolved current model to its CLI id', async () => { - const claude = fakeClaude({ - initModel: 'claude-sonnet-5', - routes: { - list_models: () => [ - { value: 'default', resolvedModel: 'claude-opus-5', displayName: 'Default' }, - { - value: 'opus', - resolvedModel: 'claude-opus-5', - displayName: 'Opus', - supportsEffort: true, - supportedEffortLevels: ['low', 'high'] - }, - { - value: 'sonnet', - resolvedModel: 'claude-sonnet-5', - displayName: 'Sonnet' - } - ] - } - }) - const adapter = await acquired(claude) - - await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toEqual({ - models: [ - { - id: 'opus', - label: 'Opus', - isDefault: true, - efforts: [ - { value: 'low', label: 'Low' }, - { value: 'high', label: 'High' } - ] - }, - { id: 'sonnet', label: 'Sonnet', isDefault: false, efforts: [] } - ], - current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } - }) - }) - - it('keeps the shared Claude seed when live model discovery is unavailable', async () => { - const claude = fakeClaude({ - initModel: 'custom-model', - routes: { - list_models: () => { - throw new Error('unsupported') - } - } - }) - const adapter = await acquired(claude) - const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) - - expect(result.models.map((model) => model.id)).toEqual([ - 'fable', - 'opus', - 'sonnet', - 'haiku', - 'custom-model' - ]) - expect(result.current).toEqual({ - model: 'custom-model', - effort: 'high', - confirmed: ['model', 'effort'] - }) - }) -}) - describe('ClaudeStructuredSessionAdapter acquisition cleanup', () => { /** A start that fails after the child self-exited, with its close verdict scripted. */ function failedStart( diff --git a/src/main/claude/claude-structured-session-adapter.ts b/src/main/claude/claude-structured-session-adapter.ts index bcae132f695..d613a175917 100644 --- a/src/main/claude/claude-structured-session-adapter.ts +++ b/src/main/claude/claude-structured-session-adapter.ts @@ -40,8 +40,6 @@ export type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' -const DISPATCH_ACK_TIMEOUT_MS = 10_000 - function backgroundTaskState(session: ClaudeSession): AgentSessionBackgroundTaskState | null { const state = session.backgroundTasks.state return state ? { ...state, supportsTaskStop: true } : null @@ -154,7 +152,8 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda reason: exit.error.message, cause: 'unexpected-exit', fence: exit.session.fence, - acquisitionGeneration: exit.session.acquisitionGeneration + acquisitionGeneration: exit.session.acquisitionGeneration, + observedAt: this.deps.now?.() ?? Date.now() } try { this.emit(exit.session, ended) @@ -219,19 +218,10 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda } dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => - dispatchClaudeTurn( - this.session(input.sessionId), - input, - this.deps.dispatchAckTimeoutMs ?? DISPATCH_ACK_TIMEOUT_MS - ) + dispatchClaudeTurn(this.session(input.sessionId), input) compact: NonNullable = (input) => - compactClaudeSession( - this.session(input.sessionId), - this.compactions, - input, - this.deps.dispatchAckTimeoutMs ?? DISPATCH_ACK_TIMEOUT_MS - ) + compactClaudeSession(this.session(input.sessionId), this.compactions, input) cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => { const session = this.session(input.sessionId) diff --git a/src/main/claude/claude-structured-session-close.test.ts b/src/main/claude/claude-structured-session-close.test.ts index 670c0daaf8c..0de5049a71c 100644 --- a/src/main/claude/claude-structured-session-close.test.ts +++ b/src/main/claude/claude-structured-session-close.test.ts @@ -55,7 +55,9 @@ describe('Claude published session close lifecycle', () => { expect(backgroundStates).toEqual([ { state: 'monitoring', - tasks: [{ id: 'background-1', kind: 'agent' }], + tasks: [ + { id: 'background-1', kind: 'agent', state: 'working', startedAt: expect.any(Number) } + ], supportsTaskStop: true } ]) @@ -74,7 +76,9 @@ describe('Claude published session close lifecycle', () => { expect(backgroundStates).toEqual([ { state: 'monitoring', - tasks: [{ id: 'background-1', kind: 'agent' }], + tasks: [ + { id: 'background-1', kind: 'agent', state: 'working', startedAt: expect.any(Number) } + ], supportsTaskStop: true }, null diff --git a/src/main/claude/claude-structured-session-close.ts b/src/main/claude/claude-structured-session-close.ts index de07439919d..18948ff6819 100644 --- a/src/main/claude/claude-structured-session-close.ts +++ b/src/main/claude/claude-structured-session-close.ts @@ -13,6 +13,7 @@ import { import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' import { closeProcessRegistry } from '../../shared/child-process/close-process-registry' +import { retireClaudeDispatchWaiters } from './claude-structured-dispatch' import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' export function claudeAcquisitionCleanupError( @@ -28,15 +29,10 @@ export function claudeAcquisitionCleanupError( : new AgentSessionAcquisitionExitUnprovenError(cause) } -export function settleClaudeDispatchWaiters(session: ClaudeSession): void { - for (const waiter of session.dispatchWaiters.splice(0)) { - clearTimeout(waiter.timer) - waiter.resolve(null) - } -} - export function settleClaudeExitedSession(session: ClaudeSession): void { - settleClaudeDispatchWaiters(session) + // The child is gone, so no replay can start these turns. Nothing else ends a + // waiter's life now that no deadline does. + retireClaudeDispatchWaiters(session) for (const prompt of session.prompts.clear()) { prompt.settle(null) } @@ -68,7 +64,7 @@ async function finalizeClaudePublishedSession( input: CloseClaudePublishedSessionInput, session: ClaudeSession ): Promise { - settleClaudeDispatchWaiters(session) + retireClaudeDispatchWaiters(session) // Settle every in-flight permission callback so closing leaves no dangling promise; `null` // writes no response, and the SDK ignores any post-cleanup answer regardless. for (const prompt of session.prompts.clear()) { @@ -109,7 +105,8 @@ async function finalizeClaudePublishedSession( const ended = { type: 'ended', sessionId: input.sessionId, - reason: 'claude session closed' + reason: 'claude session closed', + observedAt: Date.now() } as const let callbackError: unknown let callbackThrew = false diff --git a/src/main/claude/claude-structured-session-commands.test.ts b/src/main/claude/claude-structured-session-commands.test.ts index 1b2fb6bde31..29f2bcf4aa7 100644 --- a/src/main/claude/claude-structured-session-commands.test.ts +++ b/src/main/claude/claude-structured-session-commands.test.ts @@ -55,8 +55,14 @@ it.each([ const adapter = adapterFor(claude) try { await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const described = commands[0]?.description expect(adapter.readCommands('session-1')).toEqual( - commands.map(({ name }) => ({ name, kind: 'command', kindUnspecified: true })) + commands.map(({ name, description }) => ({ + name, + kind: 'command', + kindUnspecified: true, + ...(description ? { description } : {}) + })) ) expect(claude.connections[0].sent).toEqual([]) expect(claude.connections[0].calls.map(({ subtype }) => subtype)).toEqual([ @@ -70,7 +76,10 @@ it.each([ slash_commands: ['project:check'], skills: ['project:check'] }) - expect(adapter.readCommands('session-1')).toEqual([{ name: 'project:check', kind: 'skill' }]) + // The stream init classifies the name; the control seed's text survives it. + expect(adapter.readCommands('session-1')).toEqual([ + { name: 'project:check', kind: 'skill', ...(described ? { description: described } : {}) } + ]) } finally { await adapter.closeSession('session-1') } diff --git a/src/main/claude/claude-structured-session-recovery.test.ts b/src/main/claude/claude-structured-session-recovery.test.ts index 5bfe56cf156..bd8f8cb40b2 100644 --- a/src/main/claude/claude-structured-session-recovery.test.ts +++ b/src/main/claude/claude-structured-session-recovery.test.ts @@ -598,7 +598,8 @@ describe('ClaudeStructuredSessionAdapter transcript-derived recovery', () => { reason: 'crashed before replacement', cause: 'unexpected-exit', fence: 7, - acquisitionGeneration: firstAcquisition.acquisitionGeneration + acquisitionGeneration: firstAcquisition.acquisitionGeneration, + observedAt: 1_700_000_000_500 } ]) expect(replacement.link).toMatchObject({ diff --git a/src/main/claude/claude-structured-session-state.ts b/src/main/claude/claude-structured-session-state.ts index ea539065920..7c3dc5e050e 100644 --- a/src/main/claude/claude-structured-session-state.ts +++ b/src/main/claude/claude-structured-session-state.ts @@ -31,6 +31,8 @@ export type ClaudeStructuredSessionEvent = message: Record /** Present only when this replay acknowledged Orca's in-flight dispatch. */ startsTurn?: true + /** Host clock at receipt; stamped on turn boundaries only. */ + observedAt?: number } | { type: 'provider-frame'; sessionId: string; kind: string; payload: unknown } | { type: 'prompt'; sessionId: string; prompt: ClaudePendingPrompt } @@ -53,6 +55,8 @@ export type ClaudeStructuredSessionEvent = fence?: number acquisitionGeneration?: string settlementRetryRequired?: boolean + /** Host clock when the end was observed. */ + observedAt?: number } export type ClaudeStructuredSessionAdapterDeps = { @@ -60,7 +64,7 @@ export type ClaudeStructuredSessionAdapterDeps = { identity: AgentSessionJournalIdentity }) => Promise onEvent?: (event: ClaudeStructuredSessionEvent) => void - /** A dispatch whose ack timed out, proven delivered by a later provider replay. */ + /** Direct settlement path for a provider replay; its durable item row also reconciles delivery. */ onDispatchSettledLate?: (input: { sessionId: string clientMessageId: string @@ -77,7 +81,6 @@ export type ClaudeStructuredSessionAdapterDeps = { now?: () => number requestTimeoutMs?: number initTimeoutMs?: number - dispatchAckTimeoutMs?: number persistHandle?: (input: { sessionId: string providerSessionId: string @@ -96,18 +99,16 @@ export type ClaudeStructuredSessionAdapterDeps = { export type ClaudeDispatchWaiter = { resolve: (uuid: string | null) => void - timer: ReturnType acceptsResult: boolean - /** Carried so a replay that lands after the ack window can settle the journal - * submission this dispatch came from, not just the in-memory turn identity. */ - clientMessageId: string + /** Submission settled by the replay, or null for provider-control turns. */ + clientMessageId: string | null /** Client uuid echoed by Claude so a replay is tied to its own dispatch. */ sentUuid: string /** Sequence used to fence a late identity from a newer dispatch. */ dispatchSequence: number /** Set when the provider replay settled this waiter before send returned. */ settledUuid?: string - /** The waiter timed out or its write failed, but its replay may still arrive. */ + /** The write failed or the child died, but a replay may still name it. */ retired?: boolean /** Bounded digest/summary for compatibility CLIs that mint UUIDs. */ replayContentKey: string @@ -123,7 +124,7 @@ export type ClaudeSession = { acquisitionGeneration: string prompts: ClaudePromptRegistry dispatchWaiters: ClaudeDispatchWaiter[] - /** Bounded identities for dispatches whose ack was unknown when they returned. */ + /** Bounded identities for dispatches whose child died or whose write failed. */ retiredDispatchWaiters: ClaudeDispatchWaiter[] /** Once a retired waiter is evicted, legacy content-only replay matching is unsafe. */ replayContentFallbackBlocked: boolean diff --git a/src/main/claude/claude-structured-session-test-support.ts b/src/main/claude/claude-structured-session-test-support.ts index 15a9fcbbb8f..e728263d058 100644 --- a/src/main/claude/claude-structured-session-test-support.ts +++ b/src/main/claude/claude-structured-session-test-support.ts @@ -198,7 +198,8 @@ export function adapterFor( initTimeoutMs?: number, readTranscriptLeaf?: ClaudeStructuredSessionAdapterDeps['readTranscriptLeaf'], persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'], - onBackgroundTasksChanged?: ClaudeStructuredSessionAdapterDeps['onBackgroundTasksChanged'] + onBackgroundTasksChanged?: ClaudeStructuredSessionAdapterDeps['onBackgroundTasksChanged'], + onDispatchSettledLate?: ClaudeStructuredSessionAdapterDeps['onDispatchSettledLate'] ): ClaudeStructuredSessionAdapter { return new ClaudeStructuredSessionAdapter({ resolveLaunch: async () => ({ @@ -216,13 +217,13 @@ export function adapterFor( readProcessStartTime: async () => 1_700_000_000_000, now: () => 1_700_000_000_500, ...(initTimeoutMs === undefined ? {} : { initTimeoutMs }), - dispatchAckTimeoutMs: 10, persistHandle: persistHandle ?? (async (handle) => { persistedHandles.push(handle) }), ...(onBackgroundTasksChanged ? { onBackgroundTasksChanged } : {}), + ...(onDispatchSettledLate ? { onDispatchSettledLate } : {}), ...(readTranscriptLeaf ? { readTranscriptLeaf } : {}) }) } @@ -230,9 +231,20 @@ export function adapterFor( export async function acquired( claude: ReturnType, launch: Partial = {}, - events: ClaudeStructuredSessionEvent[] = [] + events: ClaudeStructuredSessionEvent[] = [], + onDispatchSettledLate?: ClaudeStructuredSessionAdapterDeps['onDispatchSettledLate'] ): Promise { - const adapter = adapterFor(claude, launch, events) + const adapter = adapterFor( + claude, + launch, + events, + undefined, + undefined, + undefined, + undefined, + undefined, + onDispatchSettledLate + ) await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) return adapter } diff --git a/src/main/claude/claude-tui-resume-real-binary.integration.test.ts b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts index 9ba3daf2285..f8822505c5a 100644 --- a/src/main/claude/claude-tui-resume-real-binary.integration.test.ts +++ b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts @@ -176,6 +176,7 @@ describe.skipIf(!claudeAuthenticated)('real Claude TUI resume proof', () => { const providerSessionId = randomUUID() const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') const events: ClaudeStructuredSessionEvent[] = [] + const settlements: { clientMessageId: string }[] = [] const adapter = new ClaudeStructuredSessionAdapter({ resolveLaunch: async () => ({ pathToClaudeCodeExecutable: command, @@ -191,6 +192,7 @@ describe.skipIf(!claudeAuthenticated)('real Claude TUI resume proof', () => { resumed: false }), onEvent: (event) => events.push(event), + onDispatchSettledLate: (settlement) => settlements.push(settlement), readProcessStartTime: async () => 1 }) let resumed: RunningTui | null = null @@ -211,8 +213,12 @@ describe.skipIf(!claudeAuthenticated)('real Claude TUI resume proof', () => { blocks: [{ type: 'text', text: 'Reply only with ORCA_RESUME_READY.' }] } }) - ).resolves.toMatchObject({ state: 'accepted' }) + ).resolves.toEqual({ state: 'admitted' }) await waitForStructuredResult(events) + // The real CLI's replay is what settles the send; dispatch only admitted it. + expect(settlements.map((settlement) => settlement.clientMessageId)).toContain( + 'real-product-turn' + ) const started = await waitForHook(eventsPath, 'startup') const transcriptPath = String(started.transcript_path) transcripts.push(transcriptPath) diff --git a/src/main/claude/claude-turn-lifecycle-item.ts b/src/main/claude/claude-turn-lifecycle-item.ts new file mode 100644 index 00000000000..00d7c4dd65e --- /dev/null +++ b/src/main/claude/claude-turn-lifecycle-item.ts @@ -0,0 +1,83 @@ +import type { + AgentJournalItemIdentity, + AgentJournalTurnItem +} from '../../shared/agent-session-journal-types' +import { agentJournalTurnBody } from '../../shared/agent-session-turn-record' +import type { StructuredAgentSessionAppendOptions } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { claudeText } from './claude-structured-item-translation' + +export type ClaudeCurrentTurn = { + sessionId: string + turnId: string + startedAt: number + /** Provider key of the user echo that opened the turn. */ + userItemId: string +} + +export type ClaudeTurnEnd = { + state: 'completed' | 'interrupted' + completedAt: number + /** The SDK's own measured turn duration; only a result frame carries one. */ + durationMs?: number +} + +/** A result the SDK reports as aborted is the user's stop, not the model's end. */ +export function claudeTurnEndForResult( + message: Record, + completedAt: number +): ClaudeTurnEnd { + const reason = message.is_error === true ? claudeText(message.terminal_reason) : null + const durationMs = message.duration_ms + return { + state: + reason === 'aborted_streaming' || reason === 'aborted_tools' ? 'interrupted' : 'completed', + completedAt, + ...(typeof durationMs === 'number' && Number.isFinite(durationMs) && durationMs >= 0 + ? { durationMs } + : {}) + } +} + +export function claudeTurnLifecycleIdentity( + sessionId: string, + turnId: string +): AgentJournalItemIdentity { + return { + provider: 'legacy', + agent: 'claude', + sessionId, + recordId: `turn-lifecycle:${turnId}` + } +} + +/** The lifecycle row is revised to its terminal state, never tombstoned, so the + * turn's host-clock endpoints outlive the turn. */ +export function claudeTurnLifecycleItem( + turn: ClaudeCurrentTurn, + end?: ClaudeTurnEnd +): { + identity: AgentJournalItemIdentity + body: AgentJournalTurnItem + options: StructuredAgentSessionAppendOptions + publishCoalescingKey: string +} { + const { sessionId, turnId, startedAt, userItemId } = turn + return { + identity: claudeTurnLifecycleIdentity(sessionId, turnId), + body: agentJournalTurnBody( + end + ? { + turnId, + state: end.state, + startedAt, + completedAt: end.completedAt, + userItemId, + ...(end.durationMs === undefined ? {} : { durationMs: end.durationMs }) + } + : { turnId, state: 'running', startedAt, userItemId } + ), + // The running row's ts is the turn start itself, so clients read no append lag. + options: end ? {} : { observedAt: startedAt }, + publishCoalescingKey: end ? 'publish' : `turn-start:${sessionId}:${turnId}` + } +} diff --git a/src/main/codex/codex-background-command-tracker.ts b/src/main/codex/codex-background-command-tracker.ts index becbb18a67c..844134881ad 100644 --- a/src/main/codex/codex-background-command-tracker.ts +++ b/src/main/codex/codex-background-command-tracker.ts @@ -10,9 +10,33 @@ const MAX_DESCRIPTION_CHARS = 512 type Command = { threadId: string; task: AgentSessionBackgroundTask; bytes: number } -/** Stays within the retained bound, so read-time qualification cannot outgrow admission. */ +/** The label's reserved share of the description. Reserved, not merely capped: + * a label free to spend the whole budget clips away the command it qualifies, + * leaving a command row naming an agent and no command — the failure this + * qualification exists to remove, in the other direction. `bytes` is counted + * before qualification, so this share is also what a published row may exceed + * the admitted count by. */ +const MAX_LABEL_CHARS = 96 + +/** Every cut in this file goes through here, clipped the way `boundSubagentField` + * clips the same provider string on the agent row: never mid surrogate pair, + * since a lone surrogate is lossy through any non-JSON UTF-8 hop. A composed + * row is cut a SECOND time, so a clip that is safe only where the label is + * bounded is not safe. No ordinal, because a row's identity is its `id`. */ +function boundText(value: string, max: number): string { + if (value.length <= max) { + return value + } + const keep = max - 1 + const last = value.charCodeAt(keep - 1) + const end = last >= 0xd800 && last <= 0xdbff ? keep - 1 : keep + return `${value.slice(0, end)}…` +} + +/** Resolved on read, and capped at the bound the admitted description already respects. */ function qualifiedDescription(label: string, description: string | undefined): string { - return (description ? `${label} — ${description}` : label).slice(0, MAX_DESCRIPTION_CHARS) + const name = boundText(label, MAX_LABEL_CHARS) + return boundText(description ? `${name} — ${description}` : name, MAX_DESCRIPTION_CHARS) } export class CodexBackgroundCommandTracker { @@ -122,8 +146,7 @@ export class CodexBackgroundCommandTracker { } const key = JSON.stringify([event.threadId, item.id]) const completed = event.method === 'item/completed' || item.status !== 'inProgress' - const description = readString(item, 'command') - ?.slice(0, MAX_DESCRIPTION_CHARS) + const description = boundText(readString(item, 'command') ?? '', MAX_DESCRIPTION_CHARS) .replace(/\s+/g, ' ') .trim() const value = { diff --git a/src/main/codex/codex-background-task-tracker.test.ts b/src/main/codex/codex-background-task-tracker.test.ts index 47987fe3fc0..0a70442a0b3 100644 --- a/src/main/codex/codex-background-task-tracker.test.ts +++ b/src/main/codex/codex-background-task-tracker.test.ts @@ -22,7 +22,8 @@ function turn( function activity( kind = 'started', parentTurn = PARENT_TURN, - child = CHILD + child = CHILD, + name = 'count_a' ): CodexBackgroundTaskEvent { return { method: 'item/started', @@ -35,7 +36,7 @@ function activity( id: `activity-${kind}`, kind, agentThreadId: child, - agentPath: '/root/count_a' + agentPath: `/root/${name}` } } } @@ -49,7 +50,11 @@ function runningChild(): CodexBackgroundTaskTracker { return tracker } -function command(threadId = PRIMARY, method = 'item/started'): CodexBackgroundTaskEvent { +function command( + threadId = PRIMARY, + method = 'item/started', + commandText = 'sleep 90' +): CodexBackgroundTaskEvent { return { method, threadId, @@ -61,7 +66,7 @@ function command(threadId = PRIMARY, method = 'item/started'): CodexBackgroundTa id: 'exec-1', processId: '71831', source: 'unifiedExecStartup', - command: 'sleep 90', + command: commandText, status: method === 'item/started' ? 'inProgress' : 'completed' } } @@ -103,15 +108,19 @@ describe('CodexBackgroundTaskTracker child execution ownership', () => { expect(tracker.state).toBeNull() }) - it('reports an executing child only after the foreground turn ends', () => { + it('reports an executing child while the spawning turn is still open', () => { const tracker = runningChild() - expect(tracker.state).toBeNull() - expect(tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))).toBe(true) - expect(tracker.state).toEqual({ + const running = { state: 'monitoring', supportsStopAll: false, tasks: [{ id: `codex-agent:${CHILD}`, kind: 'agent', description: 'count_a' }] - }) + } + // The strip is a live view: a fan-out is reported while it runs, not once + // the parent turn happens to end. + expect(tracker.state).toEqual(running) + // Turn end reveals children, it never settles them; the child is unchanged. + expect(tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN))).toBe(false) + expect(tracker.state).toEqual(running) }) it('never settles a child when a primary turn ends', () => { @@ -220,15 +229,16 @@ describe('CodexBackgroundTaskTracker child execution ownership', () => { }) describe('CodexBackgroundTaskTracker command integration', () => { - it('keeps a primary shell visible after the turn until its own completion', () => { + it('keeps a primary shell visible from launch until its own completion', () => { const tracker = new CodexBackgroundTaskTracker(PRIMARY) + const shell = [{ id: 'codex-command:primary:exec-1', kind: 'command', description: 'sleep 90' }] tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) tracker.observe(command()) - expect(tracker.state).toBeNull() + // Visible while the turn that launched it is still running. + expect(tracker.state?.tasks).toEqual(shell) tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) - expect(tracker.state?.tasks).toEqual([ - { id: 'codex-command:primary:exec-1', kind: 'command', description: 'sleep 90' } - ]) + expect(tracker.state?.tasks).toEqual(shell) + // Only the shell's own completion retires the row. tracker.observe(command(PRIMARY, 'item/completed')) expect(tracker.state).toBeNull() }) @@ -261,6 +271,63 @@ describe('CodexBackgroundTaskTracker command integration', () => { }) }) + it('keeps the command visible under a label that would otherwise fill the row', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(activity('started', PARENT_TURN, CHILD, 'L'.repeat(600))) + tracker.observe(command(CHILD)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + const description = tracker.state?.tasks?.[0]?.description + expect(description).toContain('sleep 90') + expect(description).toBe(`${'L'.repeat(95)}… — sleep 90`) + }) + + it('never cuts a label mid surrogate pair', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(activity('started', PARENT_TURN, CHILD, `${'L'.repeat(94)}\u{1F600}bad`)) + tracker.observe(command(CHILD)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + const description = tracker.state?.tasks?.[0]?.description ?? '' + expect(description.isWellFormed()).toBe(true) + expect(description).toBe(`${'L'.repeat(94)}… — sleep 90`) + }) + + it('never cuts a qualified command mid surrogate pair', () => { + // The label is bounded, then the COMPOSED row is bounded again. That second + // cut lands inside the description, so clipping only the label side leaves a + // lone surrogate — lossy through any non-JSON UTF-8 hop. + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/started', CHILD, CHILD_TURN)) + tracker.observe(activity('started', PARENT_TURN, CHILD, 'L'.repeat(96))) + // Places the pair exactly where a raw slice of the composed row splits it. + tracker.observe(command(CHILD, 'item/started', `${'C'.repeat(412)}\u{1F600}${'D'.repeat(200)}`)) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + tracker.observe(turn('turn/completed', CHILD, CHILD_TURN)) + const description = tracker.state?.tasks?.[0]?.description ?? '' + expect(description.length).toBeLessThanOrEqual(512) + expect(description.startsWith(`${'L'.repeat(96)} — `)).toBe(true) + expect(description.isWellFormed()).toBe(true) + }) + + it('never cuts an unqualified primary command mid surrogate pair', () => { + const tracker = new CodexBackgroundTaskTracker(PRIMARY) + tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) + // The pair straddles the raw description bound itself. + tracker.observe( + command(PRIMARY, 'item/started', `${'C'.repeat(511)}\u{1F600}${'D'.repeat(50)}`) + ) + tracker.observe(turn('turn/completed', PRIMARY, PARENT_TURN)) + const description = tracker.state?.tasks?.[0]?.description ?? '' + expect(description.length).toBeLessThanOrEqual(512) + expect(description.isWellFormed()).toBe(true) + }) + it('names a child shell whose label only arrives after the command', () => { const tracker = new CodexBackgroundTaskTracker(PRIMARY) tracker.observe(turn('turn/started', PRIMARY, PARENT_TURN)) diff --git a/src/main/codex/codex-background-task-tracker.ts b/src/main/codex/codex-background-task-tracker.ts index 2918f087809..a972b7bb4c1 100644 --- a/src/main/codex/codex-background-task-tracker.ts +++ b/src/main/codex/codex-background-task-tracker.ts @@ -12,7 +12,6 @@ import { boundSubagentField } from './codex-subagent-group-body' /** Projects the same child execution facts the durable roster consumes. */ export class CodexBackgroundTaskTracker { - private primaryTurnId: string | null = null private publishedFingerprint = '[]' private publishedState: AgentSessionBackgroundTaskState | null = null private readonly commands: CodexBackgroundCommandTracker @@ -44,29 +43,22 @@ export class CodexBackgroundTaskTracker { } if (frame.kind === 'subagent') { this.executions.register(frame.agentThreadId, frame.label, frame.parentTurnId) - } else if (frame.threadId === this.primaryThreadId) { - if (frame.state === 'working') { - this.primaryTurnId = frame.turnId - } else if (frame.turnId === this.primaryTurnId) { - this.primaryTurnId = null - } - } else { + } else if (frame.threadId !== this.primaryThreadId) { this.executions.observeTurn(frame.threadId, frame.turnId, frame.state) } + // A primary-turn frame only prompts a republish: turn end reveals children, + // it never settles them. Codex `spawn_agent` children keep reporting well + // past their parent turn, so nothing here may sweep the roster. return this.refresh() } clear(): boolean { this.executions.clear() this.commands.clear() - this.primaryTurnId = null return this.refresh() } private tasks(): AgentSessionBackgroundTask[] { - if (this.primaryTurnId !== null) { - return [] - } const children = this.executions.workingChildren() const agents: AgentSessionBackgroundTask[] = children.map((child, index) => ({ id: `codex-agent:${child.agentThreadId}`, diff --git a/src/main/codex/codex-notice-item-translation.test.ts b/src/main/codex/codex-notice-item-translation.test.ts index 15f638c4015..14a87cb8cc5 100644 --- a/src/main/codex/codex-notice-item-translation.test.ts +++ b/src/main/codex/codex-notice-item-translation.test.ts @@ -22,11 +22,16 @@ describe('plan document translation', () => { expect( codexItemBody({ id: 'r', type: 'reasoning', summary: ['Thinking through the problem.'] }) ).toEqual({ - kind: 'status', - text: 'Thinking through the problem.' + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Thinking through the problem.' }] }) expect(codexStreamingJournalItem({ id: 'r', type: 'reasoning' }, 'Thinking…')).toEqual({ - body: { kind: 'status', text: 'Thinking…' }, + body: { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'Thinking…' }] + }, handled: true }) }) diff --git a/src/main/codex/codex-persistent-command-retention.test.ts b/src/main/codex/codex-persistent-command-retention.test.ts index 2acb357cd71..949e3156ee2 100644 --- a/src/main/codex/codex-persistent-command-retention.test.ts +++ b/src/main/codex/codex-persistent-command-retention.test.ts @@ -113,6 +113,7 @@ describe('persistent command retention', () => { sessionId: 'session', threadId: `thread-${thread}`, turnId: 'turn', + turnLifecycle: null, sink, streams: items.streams, activeItems: items.activeItems @@ -207,6 +208,7 @@ describe('persistent command retention', () => { sessionId: 'session', threadId: 'root', turnId: 'turn', + turnLifecycle: null, sink, streams: items.streams, activeItems: items.activeItems diff --git a/src/main/codex/codex-requested-close-turn-timing.test.ts b/src/main/codex/codex-requested-close-turn-timing.test.ts new file mode 100644 index 00000000000..29039ffca3c --- /dev/null +++ b/src/main/codex/codex-requested-close-turn-timing.test.ts @@ -0,0 +1,97 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalItemBody } from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { CodexBackgroundTaskTracker } from './codex-background-task-tracker' +import { createCodexJournalTranslator } from './codex-structured-journal-translation' +import { closeCodexPublishedSession } from './codex-structured-session-close' +import type { CodexSession } from './codex-structured-session-state' + +afterEach(() => vi.useRealTimers()) + +describe('requested-close durable turn timing', () => { + it.each([true, false])( + 'keeps the first exit receipt when retry requestedClose=%s', + async (requestedClose) => { + vi.useFakeTimers() + vi.setSystemTime(1_000) + const terminalBodies: AgentJournalItemBody[] = [] + let refuseSettlement = true + const sink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {}, + tryAppendLifecycleBatch: (_id, mutations) => { + if (refuseSettlement) { + return { accepted: false, reason: 'backpressure' } + } + for (const mutation of mutations) { + if (mutation.kind === 'item') { + terminalBodies.push(mutation.body) + } + } + return { accepted: true } + } + } + const translator = createCodexJournalTranslator({ + sink, + sessionId: 'session-1', + primaryThreadId: () => 'thread-1', + now: () => Date.now() + }) + expect( + translator.handle({ + type: 'notification', + sessionId: 'session-1', + threadId: 'thread-1', + method: 'turn/started', + params: { turn: { id: 'turn-1' } }, + observedAt: 1_000 + }) + ).toEqual({ accepted: true }) + const session = { + connection: { close: vi.fn(async () => true) }, + backgroundTasks: new CodexBackgroundTaskTracker('thread-1'), + ended: false, + requestedClose: false, + fence: 7, + acquisitionGeneration: 'generation-1', + threadId: 'thread-1', + prompts: { clear: vi.fn() }, + translator + } as unknown as CodexSession + const sessions = new Map([['session-1', session]]) + const onEvent = vi.fn() + + vi.setSystemTime(2_000) + await expect(closeCodexPublishedSession(sessions, 'session-1')).resolves.toBe(false) + expect(sessions.get('session-1')).toBe(session) + expect(session.ended).toBe(false) + + refuseSettlement = false + vi.setSystemTime(60_000) + await expect( + closeCodexPublishedSession(sessions, 'session-1', onEvent, { + expectedAcquisitionGeneration: 'replacement-generation' + }) + ).resolves.toBe(false) + expect(onEvent).not.toHaveBeenCalled() + await expect( + closeCodexPublishedSession(sessions, 'session-1', onEvent, { requestedClose }) + ).resolves.toBe(true) + expect(sessions.has('session-1')).toBe(false) + expect(onEvent).toHaveBeenCalledWith( + expect.objectContaining({ + cause: requestedClose ? 'requested-close' : 'unexpected-exit', + acquisitionGeneration: 'generation-1', + observedAt: 2_000 + }) + ) + expect(terminalBodies.find((body) => body.kind === 'turn')).toMatchObject({ + kind: 'turn', + state: 'interrupted', + startedAt: 1_000, + completedAt: 2_000 + }) + } + ) +}) diff --git a/src/main/codex/codex-structured-item-translation.test.ts b/src/main/codex/codex-structured-item-translation.test.ts index 201df58bc90..53cf94e265c 100644 --- a/src/main/codex/codex-structured-item-translation.test.ts +++ b/src/main/codex/codex-structured-item-translation.test.ts @@ -10,6 +10,7 @@ import { codexItemIdentity, codexJournalItem, codexMessageBlocks, + codexStreamingJournalItem, CodexTurnOrdinals, MAX_CODEX_TURN_ORDINAL_BYTES, MAX_CODEX_TURN_ORDINAL_ENTRIES, @@ -201,6 +202,7 @@ describe('codex item bodies', () => { expect(codexItemBody(LIVE_TURN[2] as CodexThreadItem)).toEqual({ kind: 'tool-call', name: 'shell', + callId: 'item-2', input: { command: 'ls', cwd: '/tmp' }, exitCode: 0, state: 'completed', @@ -229,6 +231,7 @@ describe('codex item bodies', () => { expect(body).toEqual({ kind: 'tool-call', name: 'read', + callId: 'item-read', // `name` is the target's basename, which `path` already carries and no // label ever reads, so it stays out of the bounded journal payload. input: { command: "sed -n '1,200p' notes.txt", cwd: '/repo', path: '/repo/notes.txt' }, @@ -257,6 +260,7 @@ describe('codex item bodies', () => { ).toEqual({ kind: 'tool-call', name: 'search', + callId: 'item-search', input: { command: 'rg -n --no-heading beta .', cwd: '/repo', query: 'beta', directory: '.' }, state: 'running' }) @@ -276,6 +280,7 @@ describe('codex item bodies', () => { ).toEqual({ kind: 'tool-call', name: 'search', + callId: 'item-search-bare', input: { command: 'rg beta', cwd: '/repo' }, exitCode: 0, state: 'completed' @@ -296,6 +301,7 @@ describe('codex item bodies', () => { expect(body).toEqual({ kind: 'tool-call', name: 'list', + callId: 'item-list', input: { command: 'ls', cwd: '/repo' }, exitCode: 0, state: 'completed' @@ -326,6 +332,7 @@ describe('codex item bodies', () => { ).toEqual({ kind: 'tool-call', name: 'shell', + callId: 'item-mixed', input: { command: 'cat a.txt && ls src', cwd: '/repo' }, exitCode: 0, state: 'completed' @@ -349,6 +356,7 @@ describe('codex item bodies', () => { ).toEqual({ kind: 'tool-call', name: 'read', + callId: 'item-two-reads', input: { command: 'cat a.ts && cat b.ts', cwd: '/repo' }, exitCode: 0, state: 'completed' @@ -424,6 +432,7 @@ describe('codex item bodies', () => { ).toEqual({ kind: 'tool-call', name: 'read', + callId: 'item-read-null', input: { command: 'cat', cwd: '/repo' }, exitCode: 0, state: 'completed' @@ -451,6 +460,7 @@ describe('codex item bodies', () => { const shellRow = { kind: 'tool-call', name: 'shell', + callId: 'item-fallback', input: { command: 'ls', cwd: '/tmp' }, exitCode: 0, state: 'completed' @@ -592,12 +602,18 @@ describe('codex item bodies', () => { body: { kind: 'status', text, presentation: 'plan-document' }, handled: true }) + // A plan is a durable artifact, so it must never read as the model reasoning now. + expect(codexItemBody({ type: 'plan', id: 'plan-document', text })).not.toMatchObject({ + kind: 'message', + role: 'reasoning' + }) }) - it('renders reasoning as status and exposes an unknown item as a provider frame', () => { + it('renders reasoning as a typed message and exposes an unknown item as a provider frame', () => { expect(codexItemBody({ type: 'reasoning', id: 'r', text: 'thinking' })).toEqual({ - kind: 'status', - text: 'thinking' + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'thinking' }] }) expect(codexItemBody({ type: 'reasoning', id: 'r' })).toBeNull() expect(codexItemBody({ type: 'agentMessage', id: 'm', text: '' })).toBeNull() @@ -608,6 +624,15 @@ describe('codex item bodies', () => { }) }) + it('keeps non-reasoning item streams as status activity', () => { + expect( + codexStreamingJournalItem({ type: 'somethingCodexAddedLater', id: 'x' }, 'still working') + ).toEqual({ + body: { kind: 'status', text: 'still working' }, + handled: true + }) + }) + it('gives an mcp tool call a typed body with its own arguments as input', () => { expect( codexItemBody({ @@ -624,6 +649,7 @@ describe('codex item bodies', () => { // Server-qualified, and the arguments stay top level so the row label can // read `query`/`command`/`file_path` out of them. name: 'weather/get_forecast', + callId: 'mcp-1', mcpIdentity: { server: 'weather', tool: 'get_forecast' }, input: { city: 'Oslo' }, state: 'completed', @@ -672,6 +698,7 @@ describe('codex item bodies', () => { expect(codexItemBody({ type: 'mcpToolCall', id: 'm', tool: 't', arguments: {} })).toEqual({ kind: 'tool-call', name: 't', + callId: 'm', input: null, state: 'running' }) @@ -719,6 +746,7 @@ describe('codex item bodies', () => { expect(codexItemBody({ type: 'webSearch', id: 'w', query: '', action: null })).toEqual({ kind: 'tool-call', name: 'web_search', + callId: 'w', input: null, state: 'running' }) @@ -733,6 +761,7 @@ describe('codex item bodies', () => { ).toEqual({ kind: 'tool-call', name: 'web_search', + callId: 'w', input: { query: 'orca release notes', description: 'search', @@ -830,7 +859,11 @@ describe('codex item bodies', () => { summary: ['first', 'second'], content: [{ text: 'fallback' }] }) - ).toEqual({ kind: 'status', text: 'first\nsecond' }) + ).toEqual({ + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'first\nsecond' }] + }) }) it('refuses a value that is not a thread item at all', () => { diff --git a/src/main/codex/codex-structured-item-translation.ts b/src/main/codex/codex-structured-item-translation.ts index 576f3fb19ec..34f07eefda7 100644 --- a/src/main/codex/codex-structured-item-translation.ts +++ b/src/main/codex/codex-structured-item-translation.ts @@ -85,6 +85,10 @@ export type CodexJournalItem = { handled: boolean } +function reasoningMessageBody(text: string): AgentJournalItemBody { + return { kind: 'message', role: 'reasoning', blocks: [{ type: 'text', text }] } +} + function commandItem(item: CodexThreadItem): CodexJournalItem { const output = readFirstString(item, ['aggregatedOutput', 'aggregated_output']) const bounded = output === null ? null : boundInlineText(output, DEFAULT_JOURNAL_PAYLOAD_LIMITS) @@ -93,6 +97,7 @@ function commandItem(item: CodexThreadItem): CodexJournalItem { body: { kind: 'tool-call', name: parsed?.name ?? 'shell', + callId: item.id, // Raw command and cwd stay so the expanded view still shows what ran. input: boundToolInput( { command: item.command ?? null, cwd: item.cwd ?? null, ...parsed?.fields }, @@ -120,6 +125,7 @@ function fileChangeItem(item: CodexThreadItem): CodexJournalItem { body: { kind: 'tool-call', name: 'apply_patch', + callId: item.id, input: boundToolInput({ changes: item.changes ?? null }, DEFAULT_JOURNAL_PAYLOAD_LIMITS), state: commandState(item) }, @@ -171,6 +177,7 @@ function mcpToolCallItem(item: CodexThreadItem): CodexJournalItem { body: { kind: 'tool-call', name: mcpToolCallName(item), + callId: item.id, ...(server && tool ? { mcpIdentity: { server, tool } } : {}), input: boundToolInput(mcpToolArguments(item.arguments), DEFAULT_JOURNAL_PAYLOAD_LIMITS), state: failure === null ? commandState(item) : 'failed', @@ -213,6 +220,7 @@ function webSearchItem(item: CodexThreadItem): CodexJournalItem { body: { kind: 'tool-call', name: 'web_search', + callId: item.id, ...(results.length > 0 ? { webSearchResults: results } : {}), input: boundToolInput(webSearchInput(item), DEFAULT_JOURNAL_PAYLOAD_LIMITS), state: item.action === null || item.action === undefined ? 'running' : 'completed', @@ -268,7 +276,7 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { handled: true } } - if (item.type === 'reasoning' || item.type === 'plan') { + if (item.type === 'reasoning') { const text = readTextContent(item, 'text') ?? readTextContent(item, 'summary') ?? @@ -277,7 +285,7 @@ export function codexJournalItem(item: CodexThreadItem): CodexJournalItem { body: text === null ? null - : { kind: 'status', text: boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text }, + : reasoningMessageBody(boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text), handled: true } } @@ -327,5 +335,11 @@ export function codexStreamingJournalItem(item: CodexThreadItem, text: string): } } const bounded = boundInlineText(text, DEFAULT_JOURNAL_PAYLOAD_LIMITS) - return { body: { kind: 'status', text: bounded.text }, handled: true } + return { + body: + item.type === 'reasoning' + ? reasoningMessageBody(bounded.text) + : { kind: 'status', text: bounded.text }, + handled: true + } } diff --git a/src/main/codex/codex-structured-journal-contracts.ts b/src/main/codex/codex-structured-journal-contracts.ts index 56a543fe001..d7be380446f 100644 --- a/src/main/codex/codex-structured-journal-contracts.ts +++ b/src/main/codex/codex-structured-journal-contracts.ts @@ -5,6 +5,9 @@ import type { CodexSubagentExecutions } from './codex-subagent-executions' export type CodexJournalTranslatorDeps = { sink: StructuredAgentSessionEventSink + /** Keys restored lifecycle rows to the live identity; without it history restore skips them. */ + sessionId?: string + now?: () => number bindPromptItemId?: (journalItemId: string, threadId: string, promptKey: string) => void primaryThreadId?: () => string | null subagentExecutions?: CodexSubagentExecutions diff --git a/src/main/codex/codex-structured-journal-settlement.ts b/src/main/codex/codex-structured-journal-settlement.ts index 8c8bf091839..5785b273e92 100644 --- a/src/main/codex/codex-structured-journal-settlement.ts +++ b/src/main/codex/codex-structured-journal-settlement.ts @@ -1,6 +1,7 @@ import type { AgentJournalItemBody, - AgentJournalItemIdentity + AgentJournalItemIdentity, + AgentJournalTurnLifecycle } from '../../shared/agent-session-journal-types' import { partitionJournalLifecycleMutations } from '../native-chat/agent-session-journal/journal-lifecycle-batch-partition' import type { JournalLifecycleMutationInput } from '../native-chat/agent-session-journal/journal-row-builders' @@ -21,6 +22,10 @@ import { import type { CodexStructuredItemStreams } from './codex-structured-item-streams' import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' import { codexCommandOutlivesTurn } from './codex-command-lifecycle' +import { + codexTurnLifecycleBody, + codexTurnLifecycleIdentity +} from './codex-structured-journal-translation-turns' export type CodexActiveJournalItem = { threadId: string @@ -45,6 +50,8 @@ export function settleCodexJournalSession(input: { currentTurnIds: ReadonlyMap> primaryThreadId: string | null ordinals: CodexTurnOrdinals + /** Terminal lifecycle for a turn the provider left running when it ended. */ + settledTurnLifecycle: (threadId: string, turnId: string) => AgentJournalTurnLifecycle }): StructuredAgentSessionSinkAdmission { const mutations: JournalLifecycleMutationInput[] = [] const turnOrdinalsToForget: { threadId: string; turnId: string }[] = [] @@ -84,13 +91,9 @@ export function settleCodexJournalSession(input: { } for (const turnId of turnIds) { mutations.push({ - kind: 'tombstone', - identity: { - provider: 'legacy', - agent: 'codex', - sessionId: input.event.sessionId, - recordId: `turn-lifecycle:${turnId}` - } + kind: 'item', + identity: codexTurnLifecycleIdentity(input.event.sessionId, turnId), + body: codexTurnLifecycleBody(input.settledTurnLifecycle(threadId, turnId)) }) turnOrdinalsToForget.push({ threadId, turnId }) } @@ -109,6 +112,8 @@ export function settleCodexJournalTurn(input: { sessionId: string threadId: string turnId: string + /** Null off the primary thread: only the primary turn owns a lifecycle row. */ + turnLifecycle: AgentJournalTurnLifecycle | null sink: StructuredAgentSessionEventSink streams: CodexStructuredItemStreams activeItems: Map @@ -132,15 +137,17 @@ export function settleCodexJournalTurn(input: { } activeItemsToForget.push({ key, threadId: active.threadId, itemId: active.item.id }) } - mutations.push({ - kind: 'tombstone', - identity: { - provider: 'legacy', - agent: 'codex', - sessionId: input.sessionId, - recordId: `turn-lifecycle:${input.turnId}` - } - }) + // Revised, never tombstoned: the terminal row keeps the turn's duration durable. + if (input.turnLifecycle) { + mutations.push({ + kind: 'item', + identity: codexTurnLifecycleIdentity(input.sessionId, input.turnId), + body: codexTurnLifecycleBody(input.turnLifecycle) + }) + } + if (mutations.length === 0) { + return ADMITTED + } const admission = appendLifecycleMutations( input.sink, `turn-completed:${input.sessionId}:${input.threadId}:${input.turnId}`, diff --git a/src/main/codex/codex-structured-journal-translation-restore.ts b/src/main/codex/codex-structured-journal-translation-restore.ts index 155cba0c6a7..5631b8a132c 100644 --- a/src/main/codex/codex-structured-journal-translation-restore.ts +++ b/src/main/codex/codex-structured-journal-translation-restore.ts @@ -1,9 +1,15 @@ +import type { AgentJournalTurnLifecycle } from '../../shared/agent-session-journal-types' import type { CodexTurnOrdinals } from './codex-structured-item-translation' import { readCodexJournalRecord, readCodexJournalString } from './codex-structured-journal-translation-values' import type { CodexJournalTranslationAdmission } from './codex-structured-journal-translation' +import { + codexTurnLifecycleState, + codexTurnUserItemId +} from './codex-structured-journal-translation-turns' +import { readCodexTurnDurationMs, readCodexTurnStatus } from './codex-structured-thread-facts' /** Old providers may return the complete thread from resume. Keep that fallback * bounded before admitting any rows to the asynchronous sink. */ @@ -20,6 +26,10 @@ export function restoreCodexJournalThread(input: { method: string params: unknown }) => CodexJournalTranslationAdmission + /** Absent when the caller has no session identity to key lifecycle rows by. */ + restoreTurnLifecycle?: ( + turnLifecycle: AgentJournalTurnLifecycle + ) => CodexJournalTranslationAdmission flush: () => void }): CodexJournalTranslationAdmission { const turns = Array.isArray(input.thread.turns) ? input.thread.turns : [] @@ -30,8 +40,16 @@ export function restoreCodexJournalThread(input: { ? (Array.isArray(turn.items) ? turn.items : []).map((item) => ({ turnId, item })) : [] }) + const lifecycles = input.restoreTurnLifecycle + ? turns.flatMap( + (rawTurn) => historicalTurnLifecycle(input.threadId, readCodexJournalRecord(rawTurn)) ?? [] + ) + : [] const encodedBytes = Buffer.byteLength(JSON.stringify(items), 'utf8') - if (items.length > CODEX_RESTORE_MAX_OPERATIONS || encodedBytes > CODEX_RESTORE_MAX_BYTES) { + if ( + items.length + lifecycles.length > CODEX_RESTORE_MAX_OPERATIONS || + encodedBytes > CODEX_RESTORE_MAX_BYTES + ) { return { accepted: false, reason: 'backpressure' } } for (const rawTurn of turns) { @@ -53,7 +71,44 @@ export function restoreCodexJournalThread(input: { } input.currentTurnIds.delete(input.threadId) input.ordinals.forgetTurn(input.threadId, turnId) + const lifecycle = input.restoreTurnLifecycle + ? historicalTurnLifecycle(input.threadId, turn) + : null + if (lifecycle) { + const admission = input.restoreTurnLifecycle?.(lifecycle) ?? { accepted: true } + if (!admission.accepted) { + return admission + } + } } input.flush() return { accepted: true } } + +/** Codex reports both endpoints in unix seconds; a turn missing either has no durable duration. */ +function historicalTurnLifecycle( + threadId: string, + turn: Record +): AgentJournalTurnLifecycle | null { + const turnId = readCodexJournalString(turn, 'id') + const startedAt = turn.startedAt + const completedAt = turn.completedAt + if ( + !turnId || + typeof startedAt !== 'number' || + typeof completedAt !== 'number' || + !Number.isFinite(startedAt) || + !Number.isFinite(completedAt) + ) { + return null + } + const durationMs = readCodexTurnDurationMs(turn) + return { + turnId, + state: codexTurnLifecycleState(readCodexTurnStatus(turn)), + userItemId: codexTurnUserItemId(threadId, turnId), + startedAt: startedAt * 1000, + completedAt: completedAt * 1000, + ...(durationMs !== null ? { durationMs } : {}) + } +} diff --git a/src/main/codex/codex-structured-journal-translation-settlement.test.ts b/src/main/codex/codex-structured-journal-translation-settlement.test.ts index 1f2bc480a22..b5e60699ce1 100644 --- a/src/main/codex/codex-structured-journal-translation-settlement.test.ts +++ b/src/main/codex/codex-structured-journal-translation-settlement.test.ts @@ -218,12 +218,10 @@ describe('codex journal translation', () => { await expect(deferred.lifecycleBarrier()).resolves.toEqual({ ok: true }) expect(bodies).toEqual([ - expect.objectContaining({ - kind: 'status', - turnLifecycle: { turnId: TURN_ID, state: 'running' } - }), + expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'running' }), expect.objectContaining({ kind: 'tool-call', state: 'running' }), - expect.objectContaining({ kind: 'tool-call', state: 'failed' }) + expect.objectContaining({ kind: 'tool-call', state: 'failed' }), + expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'completed' }) ]) expect(publishes).toHaveLength(2) }) @@ -263,10 +261,7 @@ describe('codex journal translation', () => { await expect(deferred.lifecycleBarrier()).resolves.toEqual({ ok: true }) expect(bodies).toEqual([ - expect.objectContaining({ - kind: 'status', - turnLifecycle: { turnId: TURN_ID, state: 'running' } - }), + expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'running' }), expect.objectContaining({ kind: 'approval', resolution: expect.objectContaining({ state: 'pending' }) @@ -275,7 +270,8 @@ describe('codex journal translation', () => { kind: 'approval', resolution: expect.objectContaining({ state: 'cancelled' }) }), - { kind: 'status', text: 'Provider exited: lost child' } + { kind: 'status', text: 'Provider exited: lost child' }, + expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'interrupted' }) ]) expect(publishes).toHaveLength(2) }) @@ -349,7 +345,10 @@ describe('codex journal translation', () => { expect.objectContaining({ body: { kind: 'status', text: 'Provider exited: lost child' } }), - expect.objectContaining({ kind: 'tombstone' }) + expect.objectContaining({ + kind: 'item', + body: expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'interrupted' }) + }) ]) ) }) @@ -422,7 +421,10 @@ describe('codex journal translation', () => { kind: 'item', body: { kind: 'status', text: 'Provider exited: lost child' } }) - expect(flattened.at(-1)).toMatchObject({ kind: 'tombstone' }) + expect(flattened.at(-1)).toMatchObject({ + kind: 'item', + body: { kind: 'turn', state: 'interrupted' } + }) expectLifecycleBatchBounds(batches) }) @@ -542,15 +544,23 @@ describe('codex journal translation', () => { output: expect.objectContaining({ head: 'partial' }) }) }), - expect.objectContaining({ - kind: 'tombstone', + { + kind: 'item', identity: { provider: 'legacy', agent: 'codex', sessionId: SESSION_ID, recordId: `turn-lifecycle:${TURN_ID}` + }, + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'completed', + userItemId: `codex:${THREAD_ID}:${TURN_ID}:0`, + startedAt: expect.any(Number), + completedAt: expect.any(Number) } - }) + } ] } ]) @@ -789,8 +799,9 @@ describe('codex journal translation', () => { const reduced = new Map(tap.rows.map((row) => [row.key, row.body])) expect(reduced.get('orca:codex-item%3Athread-abc%3Ar-1')).toEqual({ - kind: 'status', - text: 'thinking' + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: 'thinking' }] }) expect(reduced.get('orca:codex-item%3Athread-abc%3Apatch-1')).toMatchObject({ kind: 'diff', diff --git a/src/main/codex/codex-structured-journal-translation-streams.test.ts b/src/main/codex/codex-structured-journal-translation-streams.test.ts index ea01f88e4c9..8a924a9d78a 100644 --- a/src/main/codex/codex-structured-journal-translation-streams.test.ts +++ b/src/main/codex/codex-structured-journal-translation-streams.test.ts @@ -486,7 +486,8 @@ describe('codex journal translation', () => { expect(translator.handle(notification('turn/completed', { turn: { id: TURN_ID } }))).toEqual({ accepted: true }) - expect(tap.tombstones).toContain('legacy:codex:session-1:turn-lifecycle%3Aturn-1') + // Lifecycle rows are revised in place, never tombstoned. + expect(tap.tombstones).toEqual([]) // The two maps share one bounded bucket budget; this assertion documents // the contract for future changes even though the maps are private. expect(MAX_CODEX_GENERIC_BOOKKEEPING_ENTRIES).toBeGreaterThanOrEqual( diff --git a/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts b/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts new file mode 100644 index 00000000000..a4aa3200aca --- /dev/null +++ b/src/main/codex/codex-structured-journal-translation-turn-boundaries.ts @@ -0,0 +1,128 @@ +import type { AgentJournalTurnLifecycle } from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + CODEX_JOURNAL_ADMITTED, + type CodexJournalTranslationAdmission +} from './codex-structured-journal-contracts' +import type { CodexJournalItems } from './codex-structured-journal-items' +import { settleCodexJournalTurn } from './codex-structured-journal-settlement' +import type { CodexJournalActiveTurns } from './codex-structured-journal-translation-turn-state' +import { + codexTurnLifecycleState, + codexTurnUserItemId, + publishCodexTurnLifecycle +} from './codex-structured-journal-translation-turns' +import { + readCodexTurnDurationMs, + readCodexTurnId, + readCodexTurnStatus +} from './codex-structured-thread-facts' + +type TurnBoundaryEvent = { + sessionId: string + threadId: string + params: unknown + observedAt?: number +} + +/** Opens and settles the durable lifecycle row for each primary-thread turn. */ +export class CodexJournalTurnBoundaries { + constructor( + private readonly deps: { + sink: StructuredAgentSessionEventSink + primaryThreadId: () => string | null + activeTurns: CodexJournalActiveTurns + items: Pick + flushSuppression: () => CodexJournalTranslationAdmission + resetActivity: (threadId: string) => void + now?: () => number + } + ) {} + + start(event: TurnBoundaryEvent): CodexJournalTranslationAdmission { + const turnId = readCodexTurnId(event.params) + if (!turnId) { + return CODEX_JOURNAL_ADMITTED + } + if (!this.deps.activeTurns.canRemember(event.threadId, turnId)) { + return { accepted: false, reason: 'backpressure' } + } + const startedAt = this.receiptTime(event) + const admission = publishCodexTurnLifecycle({ + sink: this.deps.sink, + primaryThreadId: this.deps.primaryThreadId(), + sessionId: event.sessionId, + threadId: event.threadId, + turnId, + state: 'running', + startedAt + }) + if (admission.accepted) { + this.deps.activeTurns.remember(event.threadId, turnId, startedAt) + this.deps.resetActivity(event.threadId) + } + return admission + } + + complete(event: TurnBoundaryEvent): CodexJournalTranslationAdmission { + const suppressionAdmission = this.deps.flushSuppression() + if (!suppressionAdmission.accepted) { + return suppressionAdmission + } + const turnId = readCodexTurnId(event.params) ?? this.deps.activeTurns.current(event.threadId) + if (!turnId) { + return CODEX_JOURNAL_ADMITTED + } + // The roster is deliberately NOT swept here. `spawn_agent` children outlive + // the turn that spawned them and go on reporting into the same group, so a + // turn boundary is no evidence contact was lost. Only `settleSession` may + // write `unverifiable`. + const admission = settleCodexJournalTurn({ + sink: this.deps.sink, + sessionId: event.sessionId, + threadId: event.threadId, + turnId, + turnLifecycle: + event.threadId === this.deps.primaryThreadId() + ? this.settled( + event.threadId, + turnId, + codexTurnLifecycleState(readCodexTurnStatus(event.params)), + this.receiptTime(event), + readCodexTurnDurationMs(event.params) + ) + : null, + streams: this.deps.items.streams, + activeItems: this.deps.items.activeItems + }) + if (admission.accepted) { + this.deps.items.ordinals.forgetTurn(event.threadId, turnId) + this.deps.activeTurns.forget(event.threadId, turnId) + this.deps.resetActivity(event.threadId) + } + return admission + } + + /** Terminal lifecycle for a remembered turn; `startedAt` is absent when the start was never seen. */ + settled( + threadId: string, + turnId: string, + state: 'completed' | 'interrupted', + completedAt: number, + durationMs: number | null = null + ): AgentJournalTurnLifecycle { + const startedAt = this.deps.activeTurns.startedAt(threadId, turnId) + return { + turnId, + state, + userItemId: codexTurnUserItemId(threadId, turnId), + ...(startedAt !== undefined ? { startedAt } : {}), + completedAt, + ...(durationMs !== null ? { durationMs } : {}) + } + } + + private receiptTime(event: TurnBoundaryEvent): number { + return event.observedAt ?? this.deps.now?.() ?? Date.now() + } +} diff --git a/src/main/codex/codex-structured-journal-translation-turn-lifecycle.test.ts b/src/main/codex/codex-structured-journal-translation-turn-lifecycle.test.ts new file mode 100644 index 00000000000..b57670851b5 --- /dev/null +++ b/src/main/codex/codex-structured-journal-translation-turn-lifecycle.test.ts @@ -0,0 +1,324 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' +import { + createDeferredStructuredAgentSessionEventSink, + type StructuredAgentSessionEventSink +} from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { CodexAppServerConnection } from './codex-app-server-connection' +import { createCodexJournalTranslator } from './codex-structured-journal-translation' +import { createCodexStructuredNotificationRetry } from './codex-structured-notification-retry' +import type { CodexStructuredSessionEvent } from './codex-structured-session-adapter' +import type { CodexSession } from './codex-structured-session-state' + +const SESSION_ID = 'session-1' +const THREAD_ID = 'thread-abc' +const TURN_ID = 'turn-1' +const LIFECYCLE_KEY = 'legacy:codex:session-1:turn-lifecycle%3Aturn-1' +const USER_ITEM_ID = 'codex:thread-abc:turn-1:0' + +type Row = { key: string; body: AgentJournalItemBody } + +function recorder() { + const rows: Row[] = [] + const tombstones: string[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity: AgentJournalItemIdentity, body) => + rows.push({ key: agentJournalItemKey(identity), body }), + appendTombstone: (identity) => tombstones.push(agentJournalItemKey(identity)), + publish: () => {} + } + return { sink, rows, tombstones } +} + +/** Latest body per identity, in first-seen order: what the journal reducer keeps. */ +function reduced(rows: readonly Row[]): Row[] { + const latest = new Map() + for (const row of rows) { + latest.set(row.key, row) + } + return [...latest.values()] +} + +function notification( + method: string, + params: unknown, + observedAt?: number +): CodexStructuredSessionEvent { + return { + type: 'notification', + sessionId: SESSION_ID, + threadId: THREAD_ID, + method, + params, + ...(observedAt !== undefined ? { observedAt } : {}) + } +} + +function translatorFor(tap: ReturnType, now?: () => number) { + return createCodexJournalTranslator({ + sink: tap.sink, + sessionId: SESSION_ID, + primaryThreadId: () => THREAD_ID, + ...(now ? { now } : {}) + }) +} + +const journals = createTrackedJournalOpener() +let root: string + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-codex-turn-lifecycle-')) +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) + vi.useRealTimers() +}) + +describe('codex turn lifecycle rows', () => { + it('opens the running row with the host receipt time and pins the row time to it', async () => { + const journal = await journals.open({ + identity: { + sessionId: SESSION_ID, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD_ID } + }, + now: () => 9_000, + journalDir: join(root, SESSION_ID) + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + const translator = createCodexJournalTranslator({ + sink: deferred.sink, + sessionId: SESSION_ID, + primaryThreadId: () => THREAD_ID + }) + deferred.bind({ journal, fence: 1, publish: () => {} }) + const before = journal.cursor() + + translator.handle(notification('turn/started', { turn: { id: TURN_ID } }, 1_000)) + await expect(deferred.drained()).resolves.toEqual({ ok: true }) + + const appended = journal.readSince(before) + expect(appended.ok && appended.rows).toEqual([ + expect.objectContaining({ + kind: 'item', + v: 3, + ts: 1_000, + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'running', + userItemId: USER_ITEM_ID, + startedAt: 1_000 + } + }) + ]) + expect(journal.snapshot().items).toEqual([ + expect.objectContaining({ + observedAt: 1_000, + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'running', + userItemId: USER_ITEM_ID, + startedAt: 1_000 + } + }) + ]) + deferred.close() + }) + + it('carries the provider duration and the same user item onto the terminal row', () => { + const tap = recorder() + const translator = translatorFor(tap) + + translator.handle(notification('turn/started', { turn: { id: TURN_ID } }, 1_000)) + translator.handle( + notification( + 'turn/completed', + { turn: { id: TURN_ID, status: 'completed', durationMs: 3_250 } }, + 4_500 + ) + ) + + expect(tap.rows.at(-1)).toEqual({ + key: LIFECYCLE_KEY, + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'completed', + userItemId: USER_ITEM_ID, + startedAt: 1_000, + completedAt: 4_500, + durationMs: 3_250 + } + }) + }) + + it.each(['interrupted', 'failed', 'cancelled'])( + 'maps a %s turn status to an interrupted lifecycle', + (status) => { + const tap = recorder() + const translator = translatorFor(tap) + + translator.handle(notification('turn/started', { turn: { id: TURN_ID } }, 1_000)) + translator.handle(notification('turn/completed', { turn: { id: TURN_ID, status } }, 2_000)) + + expect(tap.tombstones).toEqual([]) + expect(reduced(tap.rows)).toEqual([ + { + key: LIFECYCLE_KEY, + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'interrupted', + userItemId: USER_ITEM_ID, + startedAt: 1_000, + completedAt: 2_000 + } + } + ]) + } + ) + + it('stamps the host clock when a boundary arrives without a receipt time', () => { + const tap = recorder() + let clock = 10_000 + const translator = translatorFor(tap, () => (clock += 250)) + + translator.handle(notification('turn/started', { turn: { id: TURN_ID } })) + translator.handle(notification('turn/completed', { turn: { id: TURN_ID } })) + + expect(tap.rows.map((row) => row.body)).toMatchObject([ + { kind: 'turn', state: 'running', startedAt: 10_250 }, + { kind: 'turn', state: 'completed', startedAt: 10_250, completedAt: 10_500 } + ]) + }) + + it('writes only the end time, and no duration, when Codex reports neither', () => { + const tap = recorder() + const translator = translatorFor(tap) + + translator.handle(notification('turn/completed', { turn: { id: TURN_ID } }, 3_000)) + + expect(tap.rows).toEqual([ + { + key: LIFECYCLE_KEY, + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'completed', + userItemId: USER_ITEM_ID, + completedAt: 3_000 + } + } + ]) + }) + + it('replays a backpressured turn boundary with its original receipt time', async () => { + vi.useFakeTimers() + const connection = { + pauseReading: vi.fn(), + resumeReading: vi.fn() + } as unknown as CodexAppServerConnection + const translate = vi + .fn[0]['translate']>() + .mockReturnValueOnce({ accepted: false, reason: 'backpressure' }) + .mockReturnValue({ accepted: true }) + const retries = createCodexStructuredNotificationRetry({ + sessionFor: () => ({ connection, ended: false }) as CodexSession, + translate + }) + + expect(retries.handle(SESSION_ID, 'turn/started', { turn: { id: TURN_ID } }, 1_000)).toEqual({ + accepted: false, + reason: 'backpressure' + }) + await vi.advanceTimersByTimeAsync(50) + + expect(translate.mock.calls.map((call) => call[4])).toEqual([1_000, 1_000]) + expect(connection.resumeReading).not.toHaveBeenCalled() + }) + + it('restores terminal rows for historical turns with both endpoints, in milliseconds', () => { + const tap = recorder() + const translator = translatorFor(tap) + + expect( + translator.restoreThread(THREAD_ID, { + turns: [ + { + id: 'turn-done', + status: 'completed', + startedAt: 1_700_000_000, + completedAt: 1_700_000_042, + durationMs: 41_900, + items: [{ type: 'agentMessage', id: 'agent-done', text: 'done' }] + }, + { + id: 'turn-cut', + status: 'interrupted', + startedAt: 1_700_000_100, + completedAt: 1_700_000_101, + items: [] + }, + { id: 'turn-open', status: 'inProgress', startedAt: 1_700_000_200, items: [] }, + { id: 'turn-untimed', status: 'completed', items: [] } + ] + }) + ).toEqual({ accepted: true }) + + expect(tap.rows).toEqual([ + expect.objectContaining({ body: expect.objectContaining({ kind: 'message' }) }), + { + key: 'legacy:codex:session-1:turn-lifecycle%3Aturn-done', + body: { + kind: 'turn', + turnId: 'turn-done', + state: 'completed', + userItemId: 'codex:thread-abc:turn-done:0', + startedAt: 1_700_000_000_000, + completedAt: 1_700_000_042_000, + durationMs: 41_900 + } + }, + { + key: 'legacy:codex:session-1:turn-lifecycle%3Aturn-cut', + body: { + kind: 'turn', + turnId: 'turn-cut', + state: 'interrupted', + userItemId: 'codex:thread-abc:turn-cut:0', + startedAt: 1_700_000_100_000, + completedAt: 1_700_000_101_000 + } + } + ]) + expect(tap.tombstones).toEqual([]) + }) + + it('restores no lifecycle rows without a session identity to key them by', () => { + const tap = recorder() + const translator = createCodexJournalTranslator({ + sink: tap.sink, + primaryThreadId: () => THREAD_ID + }) + + translator.restoreThread(THREAD_ID, { + turns: [{ id: 'turn-done', status: 'completed', startedAt: 1, completedAt: 2, items: [] }] + }) + + expect(tap.rows).toEqual([]) + }) +}) diff --git a/src/main/codex/codex-structured-journal-translation-turn-state.test.ts b/src/main/codex/codex-structured-journal-translation-turn-state.test.ts index 303c0426f49..b1c11433e55 100644 --- a/src/main/codex/codex-structured-journal-translation-turn-state.test.ts +++ b/src/main/codex/codex-structured-journal-translation-turn-state.test.ts @@ -40,4 +40,16 @@ describe('CodexJournalActiveTurns', () => { expect(active.size).toBe(0) expect(active.bytes).toBe(0) }) + + it('remembers each turn start time until the turn is forgotten', () => { + const active = new CodexJournalActiveTurns() + + expect(active.remember('thread', 'turn-1', 1_000)).toBe(true) + expect(active.remember('thread', 'turn-1', 2_000)).toBe(true) + expect(active.startedAt('thread', 'turn-1')).toBe(1_000) + expect(active.startedAt('thread', 'turn-missing')).toBeUndefined() + + active.forget('thread', 'turn-1') + expect(active.startedAt('thread', 'turn-1')).toBeUndefined() + }) }) diff --git a/src/main/codex/codex-structured-journal-translation-turn-state.ts b/src/main/codex/codex-structured-journal-translation-turn-state.ts index 9a05b96fdd5..feca169ae99 100644 --- a/src/main/codex/codex-structured-journal-translation-turn-state.ts +++ b/src/main/codex/codex-structured-journal-translation-turn-state.ts @@ -5,6 +5,8 @@ export class CodexJournalActiveTurns { /** Bounds active turn keys retained across provider threads. */ static readonly MAX_ENTRIES = MAX_CODEX_ACTIVE_TURNS readonly byThread = new Map>() + /** Host turn-start receipt per remembered turn; the terminal row carries it forward. */ + private readonly startedAtByTurn = new Map() private activeCount = 0 private retainedBytes = 0 @@ -16,6 +18,10 @@ export class CodexJournalActiveTurns { return this.retainedBytes } + private turnKey(threadId: string, turnId: string): string { + return `${encodeURIComponent(threadId)}:${encodeURIComponent(turnId)}` + } + private entryBytes(threadId: string, turnId: string): number { return Buffer.byteLength(threadId, 'utf8') + Buffer.byteLength(turnId, 'utf8') } @@ -33,7 +39,11 @@ export class CodexJournalActiveTurns { return [...(this.byThread.get(threadId) ?? [])].at(-1) ?? null } - remember(threadId: string, turnId: string): boolean { + startedAt(threadId: string, turnId: string): number | undefined { + return this.startedAtByTurn.get(this.turnKey(threadId, turnId)) + } + + remember(threadId: string, turnId: string, startedAt?: number): boolean { const active = this.byThread.get(threadId) if (active?.has(turnId)) { return true @@ -41,6 +51,9 @@ export class CodexJournalActiveTurns { if (!this.canRemember(threadId, turnId)) { return false } + if (startedAt !== undefined) { + this.startedAtByTurn.set(this.turnKey(threadId, turnId), startedAt) + } if (active) { active.add(turnId) } else { @@ -52,6 +65,7 @@ export class CodexJournalActiveTurns { } forget(threadId: string, turnId: string): void { + this.startedAtByTurn.delete(this.turnKey(threadId, turnId)) const active = this.byThread.get(threadId) if (active?.delete(turnId)) { this.activeCount -= 1 @@ -64,6 +78,7 @@ export class CodexJournalActiveTurns { clear(): void { this.byThread.clear() + this.startedAtByTurn.clear() this.activeCount = 0 this.retainedBytes = 0 } diff --git a/src/main/codex/codex-structured-journal-translation-turns.ts b/src/main/codex/codex-structured-journal-translation-turns.ts index 7313946a53e..d1cd7ca2884 100644 --- a/src/main/codex/codex-structured-journal-translation-turns.ts +++ b/src/main/codex/codex-structured-journal-translation-turns.ts @@ -1,3 +1,12 @@ +import type { + AgentJournalItemIdentity, + AgentJournalTurnItem, + AgentJournalTurnLifecycle, + AgentJournalTurnLifecycleState +} from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { agentJournalTurnBody } from '../../shared/agent-session-turn-record' +import { CODEX_USER_MESSAGE_ORDINAL } from './codex-structured-turn-start' import type { StructuredAgentSessionEventSink, StructuredAgentSessionSinkAdmission @@ -5,56 +14,78 @@ import type { const ADMITTED: StructuredAgentSessionSinkAdmission = { accepted: true } +export function codexTurnLifecycleIdentity( + sessionId: string, + turnId: string +): AgentJournalItemIdentity { + return { + provider: 'legacy', + agent: 'codex', + sessionId, + recordId: `turn-lifecycle:${turnId}` + } +} + +/** Provider key of the user message that opened the turn; deterministic, so never remembered. */ +export function codexTurnUserItemId(threadId: string, turnId: string): string { + return agentJournalItemKey({ + provider: 'codex', + threadId, + turnId, + ordinal: CODEX_USER_MESSAGE_ORDINAL + }) +} + +export function codexTurnLifecycleBody( + turnLifecycle: AgentJournalTurnLifecycle +): AgentJournalTurnItem { + return agentJournalTurnBody(turnLifecycle) +} + +/** `turn/completed` is Codex's only turn-end notification; a missing status is a clean finish. */ +export function codexTurnLifecycleState( + status: string | null +): Extract { + return status === null || status === 'completed' ? 'completed' : 'interrupted' +} + export function publishCodexTurnLifecycle(input: { sink: StructuredAgentSessionEventSink primaryThreadId: string | null sessionId: string threadId: string turnId: string - state: 'running' | 'completed' + state: AgentJournalTurnLifecycleState + startedAt?: number + completedAt?: number + durationMs?: number }): StructuredAgentSessionSinkAdmission { if (input.primaryThreadId !== input.threadId) { return ADMITTED } - const identity = { - provider: 'legacy' as const, - agent: 'codex' as const, - sessionId: input.sessionId, - recordId: `turn-lifecycle:${input.turnId}` + const identity = codexTurnLifecycleIdentity(input.sessionId, input.turnId) + const body = codexTurnLifecycleBody({ + turnId: input.turnId, + state: input.state, + userItemId: codexTurnUserItemId(input.threadId, input.turnId), + ...(input.startedAt !== undefined ? { startedAt: input.startedAt } : {}), + ...(input.completedAt !== undefined ? { completedAt: input.completedAt } : {}), + ...(input.durationMs !== undefined ? { durationMs: input.durationMs } : {}) + }) + // The running row's `ts` is the host's turn-start receipt so clients can anchor a live counter. + const appendOptions = { + lifecycle: true, + ...(input.state === 'running' && input.startedAt !== undefined + ? { observedAt: input.startedAt } + : {}) } - if (input.state === 'completed') { - if (input.sink.tryAppendTombstone) { - const admission = input.sink.tryAppendTombstone(identity, { lifecycle: true }) - if (!admission.accepted) { - return admission - } - } else { - input.sink.appendTombstone(identity, { lifecycle: true }) - } - } else { - const admission = input.sink.tryAppendItem - ? input.sink.tryAppendItem( - identity, - { - kind: 'status', - text: 'Codex is working…', - turnLifecycle: { turnId: input.turnId, state: input.state } - }, - { lifecycle: true } - ) - : (input.sink.appendItem( - identity, - { - kind: 'status', - text: 'Codex is working…', - turnLifecycle: { turnId: input.turnId, state: input.state } - }, - { lifecycle: true } - ), - ADMITTED) + if (input.sink.tryAppendItem) { + const admission = input.sink.tryAppendItem(identity, body, appendOptions) if (!admission.accepted) { return admission } + } else { + input.sink.appendItem(identity, body, appendOptions) } // Preserve first-work evidence when completion arrives before the journal drains. const publishOptions = { diff --git a/src/main/codex/codex-structured-journal-translation.test.ts b/src/main/codex/codex-structured-journal-translation.test.ts index 0a791afa77e..7e2b2f45bea 100644 --- a/src/main/codex/codex-structured-journal-translation.test.ts +++ b/src/main/codex/codex-structured-journal-translation.test.ts @@ -49,6 +49,15 @@ function recorder() { } } +/** Latest body per identity, in first-seen order: what the journal reducer keeps. */ +function reduced(rows: readonly Row[]): Row[] { + const latest = new Map() + for (const row of rows) { + latest.set(row.key, row) + } + return [...latest.values()] +} + /** Fires the coalescing window on demand instead of on wall time. */ function manualWindow() { const pending: (() => void)[] = [] @@ -145,9 +154,7 @@ describe('codex journal translation', () => { expect( translator.handle(notification('turn/started', { turn: { id: 'turn-overflow' } })) ).toEqual({ accepted: false, reason: 'backpressure' }) - expect(tap.rows.filter((row) => row.body.kind === 'status')).toHaveLength( - MAX_CODEX_ACTIVE_TURNS - ) + expect(tap.rows.filter((row) => row.body.kind === 'turn')).toHaveLength(MAX_CODEX_ACTIVE_TURNS) expect(translator.handle(notification('turn/completed', { turn: { id: 'turn-0' } }))).toEqual({ accepted: true @@ -224,13 +231,26 @@ describe('codex journal translation', () => { { key: 'legacy:codex:session-1:turn-lifecycle%3Aturn-1', body: { - kind: 'status', - text: 'Codex is working…', - turnLifecycle: { turnId: TURN_ID, state: 'running' } + kind: 'turn', + turnId: TURN_ID, + state: 'running', + userItemId: `codex:${THREAD_ID}:${TURN_ID}:0`, + startedAt: expect.any(Number) + } + }, + { + key: 'legacy:codex:session-1:turn-lifecycle%3Aturn-1', + body: { + kind: 'turn', + turnId: TURN_ID, + state: 'completed', + userItemId: `codex:${THREAD_ID}:${TURN_ID}:0`, + startedAt: expect.any(Number), + completedAt: expect.any(Number) } } ]) - expect(tap.tombstones).toEqual(['legacy:codex:session-1:turn-lifecycle%3Aturn-1']) + expect(tap.tombstones).toEqual([]) }) it('closes every active turn when the provider session ends after a later turn starts', () => { @@ -244,29 +264,26 @@ describe('codex journal translation', () => { translator.handle(notification('turn/started', { turn: { id: 'turn-later' } })) translator.handle({ type: 'ended', sessionId: SESSION_ID, reason: 'app-server exited' }) - expect(tap.rows.filter((row) => row.body.kind === 'status')).toHaveLength(3) + expect(tap.rows.filter((row) => row.body.kind === 'turn')).toHaveLength(4) expect(tap.rows.map((row) => row.body)).toEqual([ - expect.objectContaining({ turnLifecycle: { turnId: 'turn-stale', state: 'running' } }), - expect.objectContaining({ turnLifecycle: { turnId: 'turn-later', state: 'running' } }), - expect.objectContaining({ text: 'Provider exited: app-server exited' }) + expect.objectContaining({ kind: 'turn', turnId: 'turn-stale', state: 'running' }), + expect.objectContaining({ kind: 'turn', turnId: 'turn-later', state: 'running' }), + expect.objectContaining({ text: 'Provider exited: app-server exited' }), + expect.objectContaining({ kind: 'turn', turnId: 'turn-stale', state: 'interrupted' }), + expect.objectContaining({ kind: 'turn', turnId: 'turn-later', state: 'interrupted' }) ]) - expect(tap.tombstones).toEqual([ - 'legacy:codex:session-1:turn-lifecycle%3Aturn-stale', - 'legacy:codex:session-1:turn-lifecycle%3Aturn-later' - ]) - // The tombstones remove both running rows from the reduced journal; no - // lifecycle identity remains live after a session end. + expect(tap.tombstones).toEqual([]) + // Both running rows are revised to interrupted, so no lifecycle identity + // remains live after a session end. expect( projectStructuredAgentSessionStatus( - tap.rows - .filter((row) => !tap.tombstones.includes(row.key)) - .map((row, sequence) => ({ - itemId: row.key, - revision: 1, - sequence: sequence + 1, - observedAt: sequence + 1, - body: row.body - })) + reduced(tap.rows).map((row, sequence) => ({ + itemId: row.key, + revision: 1, + sequence: sequence + 1, + observedAt: sequence + 1, + body: row.body + })) ) ).toBe('idle') }) @@ -283,21 +300,20 @@ describe('codex journal translation', () => { translator.handle(notification('turn/completed', { turn: { id: 'turn-stale' } })) translator.handle(notification('turn/completed', { turn: { id: 'turn-later' } })) - expect(tap.tombstones).toEqual([ - 'legacy:codex:session-1:turn-lifecycle%3Aturn-stale', - 'legacy:codex:session-1:turn-lifecycle%3Aturn-later' + expect(tap.tombstones).toEqual([]) + expect(reduced(tap.rows).map((row) => row.body)).toEqual([ + expect.objectContaining({ kind: 'turn', turnId: 'turn-stale', state: 'completed' }), + expect.objectContaining({ kind: 'turn', turnId: 'turn-later', state: 'completed' }) ]) expect( projectStructuredAgentSessionStatus( - tap.rows - .filter((row) => !tap.tombstones.includes(row.key)) - .map((row, sequence) => ({ - itemId: row.key, - revision: 1, - sequence: sequence + 1, - observedAt: sequence + 1, - body: row.body - })) + reduced(tap.rows).map((row, sequence) => ({ + itemId: row.key, + revision: 1, + sequence: sequence + 1, + observedAt: sequence + 1, + body: row.body + })) ) ).toBe('idle') }) @@ -431,7 +447,7 @@ describe('codex journal translation', () => { expect(window.idle()).toBe(true) }) - it('settles tools, prompts, exit status, and turn tombstone in one ordered batch', () => { + it('settles tools, prompts, exit status, and turn lifecycle in one ordered batch', () => { const tap = recorder() const batches: { settlementId: string; mutations: unknown[] }[] = [] tap.sink.appendLifecycleBatch = (settlementId, mutations) => { @@ -488,7 +504,10 @@ describe('codex journal translation', () => { kind: 'item', body: { kind: 'status', text: 'Provider exited: lost child' } }), - expect.objectContaining({ kind: 'tombstone' }) + expect.objectContaining({ + kind: 'item', + body: expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'interrupted' }) + }) ]) }) @@ -681,10 +700,7 @@ describe('codex journal translation', () => { await expect(deferred.lifecycleBarrier()).resolves.toEqual({ ok: true }) expect(bodies).toEqual([ - expect.objectContaining({ - kind: 'status', - turnLifecycle: { turnId: TURN_ID, state: 'running' } - }) + expect.objectContaining({ kind: 'turn', turnId: TURN_ID, state: 'running' }) ]) expect(publishes).toHaveLength(1) }) diff --git a/src/main/codex/codex-structured-journal-translation.ts b/src/main/codex/codex-structured-journal-translation.ts index cea129e5cdf..2907b34bc70 100644 --- a/src/main/codex/codex-structured-journal-translation.ts +++ b/src/main/codex/codex-structured-journal-translation.ts @@ -15,12 +15,10 @@ import { type CodexJournalTranslator, type CodexJournalTranslatorDeps } from './codex-structured-journal-contracts' -import { - settleCodexJournalSession, - settleCodexJournalTurn -} from './codex-structured-journal-settlement' +import { settleCodexJournalSession } from './codex-structured-journal-settlement' import { createCodexOversizedNotificationSettler } from './codex-structured-journal-translation-frames' import { restoreCodexJournalThread } from './codex-structured-journal-translation-restore' +import { CodexJournalTurnBoundaries } from './codex-structured-journal-translation-turn-boundaries' import { CodexJournalActiveTurns } from './codex-structured-journal-translation-turn-state' import { publishCodexTurnLifecycle } from './codex-structured-journal-translation-turns' import { readCodexTurnId } from './codex-structured-thread-facts' @@ -71,6 +69,21 @@ export function createCodexJournalTranslator( const flushStreams = (): CodexJournalTranslationAdmission => items.streams.flush() ? CODEX_JOURNAL_ADMITTED : { accepted: false, reason: 'backpressure' } let readActivity = createCodexProviderActivityReader() + const resetActivity = (threadId: string): void => { + if (threadId === (deps.primaryThreadId?.() ?? null)) { + readActivity = createCodexProviderActivityReader() + deps.sink.setActivity?.(null) + } + } + const turnBoundaries = new CodexJournalTurnBoundaries({ + sink: deps.sink, + primaryThreadId: () => deps.primaryThreadId?.() ?? null, + activeTurns, + items, + flushSuppression: () => genericFrames.flush(), + resetActivity, + ...(deps.now ? { now: deps.now } : {}) + }) const publishActivity = ( event: Extract, admission: CodexJournalTranslationAdmission @@ -109,6 +122,18 @@ export function createCodexJournalTranslator( ? translated.admission : { accepted: false, reason: 'untranslated' } }, + ...(deps.sessionId !== undefined + ? { + restoreTurnLifecycle: (turnLifecycle) => + publishCodexTurnLifecycle({ + sink: deps.sink, + primaryThreadId: deps.primaryThreadId?.() ?? null, + sessionId: deps.sessionId as string, + threadId, + ...turnLifecycle + }) + } + : {}), flush: items.streams.flush }) }, @@ -130,7 +155,14 @@ export function createCodexJournalTranslator( pendingPrompts: prompts.pending, currentTurnIds: activeTurns.byThread, primaryThreadId: deps.primaryThreadId?.() ?? null, - ordinals: items.ordinals + ordinals: items.ordinals, + settledTurnLifecycle: (threadId, turnId) => + turnBoundaries.settled( + threadId, + turnId, + 'interrupted', + event.observedAt ?? deps.now?.() ?? Date.now() + ) }) if (!admission.accepted) { return admission @@ -181,7 +213,9 @@ export function createCodexJournalTranslator( if (!childAdmission.accepted) { return childAdmission } - return event.method === 'turn/started' ? startTurn(event) : completeTurn(event) + return event.method === 'turn/started' + ? turnBoundaries.start(event) + : turnBoundaries.complete(event) } const compaction = compactions.handle(event) if (compaction) { @@ -242,70 +276,4 @@ export function createCodexJournalTranslator( compactions.clear() } } - - function startTurn( - event: Extract - ): CodexJournalTranslationAdmission { - const turnId = readCodexTurnId(event.params) - if (!turnId) { - return CODEX_JOURNAL_ADMITTED - } - if (!activeTurns.canRemember(event.threadId, turnId)) { - return { accepted: false, reason: 'backpressure' } - } - const admission = publishCodexTurnLifecycle({ - sink: deps.sink, - primaryThreadId: deps.primaryThreadId?.() ?? null, - sessionId: event.sessionId, - threadId: event.threadId, - turnId, - state: 'running' - }) - if (admission.accepted) { - activeTurns.remember(event.threadId, turnId) - if (event.threadId === (deps.primaryThreadId?.() ?? null)) { - readActivity = createCodexProviderActivityReader() - deps.sink.setActivity?.(null) - } - } - return admission - } - - function completeTurn(event: { - sessionId: string - threadId: string - params: unknown - }): CodexJournalTranslationAdmission { - const suppressionAdmission = genericFrames.flush() - if (!suppressionAdmission.accepted) { - return suppressionAdmission - } - const turnId = readCodexTurnId(event.params) ?? activeTurns.current(event.threadId) - if (!turnId) { - return CODEX_JOURNAL_ADMITTED - } - // The roster is deliberately NOT swept here. `spawn_agent` children outlive - // the turn that spawned them and go on reporting into the same group, so a - // turn boundary is no evidence contact was lost — and `turn/completed` is - // the only turn-end notification Codex sends, so an abort cannot be told - // apart from a clean finish either. Only `settleSession` may write - // `unverifiable`. - const admission = settleCodexJournalTurn({ - sink: deps.sink, - sessionId: event.sessionId, - threadId: event.threadId, - turnId, - streams: items.streams, - activeItems: items.activeItems - }) - if (admission.accepted) { - items.ordinals.forgetTurn(event.threadId, turnId) - activeTurns.forget(event.threadId, turnId) - if (event.threadId === (deps.primaryThreadId?.() ?? null)) { - readActivity = createCodexProviderActivityReader() - deps.sink.setActivity?.(null) - } - } - return admission - } } diff --git a/src/main/codex/codex-structured-notification-retry.ts b/src/main/codex/codex-structured-notification-retry.ts index ae4fccbe6d0..53b95c51403 100644 --- a/src/main/codex/codex-structured-notification-retry.ts +++ b/src/main/codex/codex-structured-notification-retry.ts @@ -6,7 +6,7 @@ const MAX_RETRY_EVENTS = 256 const MAX_RETRY_BYTES = 8 * 1024 * 1024 const RETRY_DELAY_MS = 25 -type PendingNotification = { method: string; params: unknown; bytes: number } +type PendingNotification = { method: string; params: unknown; bytes: number; observedAt?: number } type RetryState = { connection: CodexAppServerConnection events: PendingNotification[] @@ -22,7 +22,8 @@ export function createCodexStructuredNotificationRetry(deps: { sessionId: string, session: CodexSession, method: string, - params: unknown + params: unknown, + observedAt?: number ) => CodexJournalTranslationAdmission }) { const states = new Map() @@ -48,7 +49,13 @@ export function createCodexStructuredNotificationRetry(deps: { fail(sessionId, state, 'notification retry owner is no longer live') break } - const admission = deps.translate(sessionId, session, pending.method, pending.params) + const admission = deps.translate( + sessionId, + session, + pending.method, + pending.params, + pending.observedAt + ) if (!admission.accepted) { if (admission.reason === 'backpressure') { state.timer = setTimeout(() => { @@ -97,7 +104,8 @@ export function createCodexStructuredNotificationRetry(deps: { sessionId: string, connection: CodexAppServerConnection, method: string, - params: unknown + params: unknown, + observedAt: number | undefined ): void => { const bytes = Buffer.byteLength(JSON.stringify({ method, params }), 'utf8') let state = states.get(sessionId) @@ -111,7 +119,12 @@ export function createCodexStructuredNotificationRetry(deps: { fail(sessionId, state, 'notification retry queue overflow') return } - state.events.push({ method, params, bytes }) + state.events.push({ + method, + params, + bytes, + ...(observedAt !== undefined ? { observedAt } : {}) + }) state.bytes += bytes connection.pauseReading?.() } @@ -120,7 +133,8 @@ export function createCodexStructuredNotificationRetry(deps: { handle: ( sessionId: string, method: string, - params: unknown + params: unknown, + observedAt?: number ): CodexJournalTranslationAdmission => { const session = deps.sessionFor(sessionId) if (!session) { @@ -128,13 +142,13 @@ export function createCodexStructuredNotificationRetry(deps: { } const state = states.get(sessionId) if (state && state.events.length > 0) { - enqueue(sessionId, state.connection, method, params) + enqueue(sessionId, state.connection, method, params, observedAt) retry(sessionId, state.connection) return { accepted: false, reason: 'backpressure' } } - const admission = deps.translate(sessionId, session, method, params) + const admission = deps.translate(sessionId, session, method, params, observedAt) if (!admission.accepted) { - enqueue(sessionId, session.connection, method, params) + enqueue(sessionId, session.connection, method, params, observedAt) retry(sessionId, session.connection) } return admission diff --git a/src/main/codex/codex-structured-provider-events.ts b/src/main/codex/codex-structured-provider-events.ts index 0cc793249a9..989232ff1b2 100644 --- a/src/main/codex/codex-structured-provider-events.ts +++ b/src/main/codex/codex-structured-provider-events.ts @@ -1,20 +1,41 @@ import type { CodexAppServerServerRequest } from './codex-app-server-connection' import { disposeCodexServerRequest } from './codex-server-request-disposition' import type { CodexJournalTranslationAdmission } from './codex-structured-journal-translation' +import * as codexRewind from './codex-structured-rewind' import type { CodexSession, CodexStructuredSessionEvent } from './codex-structured-session-state' import { readCodexThreadId, readCodexTurnId } from './codex-structured-thread-facts' +import type { CodexStructuredTurnCancellation } from './codex-structured-turn-cancellation' type EmitCodexEvent = ( session: CodexSession, event: CodexStructuredSessionEvent ) => CodexJournalTranslationAdmission +/** One live notification's journal entry: rewind bookkeeping, cancellation deferral, delivery. */ +export function translateCodexNotification(input: { + sessionId: string + session: CodexSession + method: string + params: unknown + observedAt?: number + turnCancellation: Pick + emit: EmitCodexEvent +}): CodexJournalTranslationAdmission { + const { sessionId, session, method, params, observedAt } = input + codexRewind.observeCodexRewindActivity(session, method, params) + if (input.turnCancellation.handleNotification(sessionId, session, method, params, observedAt)) { + return { accepted: true } + } + return deliverCodexNotification(sessionId, session, method, params, input.emit, observedAt) +} + export function deliverCodexNotification( sessionId: string, session: CodexSession | undefined, method: string, params: unknown, - emit: EmitCodexEvent + emit: EmitCodexEvent, + observedAt?: number ): CodexJournalTranslationAdmission { if (!session) { return { accepted: true } @@ -23,7 +44,14 @@ export function deliverCodexNotification( const turnId = method === 'turn/started' && threadId === session.threadId ? readCodexTurnId(params) : null const turnWaiter = turnId ? session.turnIdWaiters[0] : undefined - const admission = emit(session, { type: 'notification', sessionId, threadId, method, params }) + const admission = emit(session, { + type: 'notification', + sessionId, + threadId, + method, + params, + ...(observedAt !== undefined ? { observedAt } : {}) + }) if (method === 'turn/started' && threadId === session.threadId) { if (admission.accepted && turnId && session.turnIdWaiters[0] === turnWaiter) { session.turnIdWaiters.shift() diff --git a/src/main/codex/codex-structured-session-acquire.ts b/src/main/codex/codex-structured-session-acquire.ts index 3f7f130e173..c9a306128b6 100644 --- a/src/main/codex/codex-structured-session-acquire.ts +++ b/src/main/codex/codex-structured-session-acquire.ts @@ -80,6 +80,8 @@ export async function acquireCodexStructuredSession(input: { const translator = acquireInput.events ? createCodexJournalTranslator({ sink: acquireInput.events, + sessionId, + ...(deps.now ? { now: deps.now } : {}), primaryThreadId: () => primaryThreadId, subagentExecutions, bindPromptItemId: (journalItemId, threadId, promptKey) => @@ -113,13 +115,16 @@ export async function acquireCodexStructuredSession(input: { env: buildCodexStructuredChildEnvironment(launch, acquireInput.spawnToken, sessionId) }, { - onNotification: (method, params) => + onNotification: (method, params) => { + // Stamped at receipt, ahead of any pre-publication buffering or retry. + const observedAt = isCodexTurnBoundary(method) ? (deps.now?.() ?? Date.now()) : undefined input.deliver( acquisition, sessionId, - () => notificationRetries.handle(sessionId, method, params), + () => notificationRetries.handle(sessionId, method, params, observedAt), Buffer.byteLength(JSON.stringify(params ?? null), 'utf8') - ), + ) + }, onServerRequest: (request) => input.deliver( acquisition, @@ -239,3 +244,7 @@ export async function acquireCodexStructuredSession(input: { attempt.finish() } } + +function isCodexTurnBoundary(method: string): boolean { + return method === 'turn/started' || method === 'turn/completed' +} diff --git a/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts b/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts index 3f1cd923e13..b2c579770fd 100644 --- a/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts +++ b/src/main/codex/codex-structured-session-adapter-lifecycle.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from 'vitest' +import { readAgentJournalTurn } from '../../shared/agent-session-turn-record' import type { AgentJournalMessageItem, AgentSessionJournalIdentity @@ -181,7 +182,8 @@ describe('CodexStructuredSessionAdapter lifecycle', () => { reason: 'codex app-server connection ended', cause: 'unexpected-exit', fence: 7, - acquisitionGeneration: 'generation-1' + acquisitionGeneration: 'generation-1', + observedAt: expect.any(Number) }) await expect( adapter.dispatch({ @@ -232,11 +234,14 @@ describe('CodexStructuredSessionAdapter lifecycle', () => { it('flushes the final coalesced text before a graceful close', async () => { const codex = fakeCodex() const bodies: AgentJournalMessageItem[] = [] + const lifecycles: unknown[] = [] const tombstones: unknown[] = [] const sink: StructuredAgentSessionEventSink = { - appendItem: (_identity, body) => { + appendItem: (identity, body) => { if (body.kind === 'message') { bodies.push(body) + } else if (readAgentJournalTurn(body)) { + lifecycles.push({ identity, turnLifecycle: readAgentJournalTurn(body) }) } }, appendTombstone: (identity) => { @@ -266,11 +271,22 @@ describe('CodexStructuredSessionAdapter lifecycle', () => { await adapter.closeSession('session-1') expect(bodies.at(-1)?.blocks).toEqual([{ type: 'text', text: 'last words' }]) - expect(tombstones).toContainEqual({ - provider: 'legacy', - agent: 'codex', - sessionId: 'session-1', - recordId: 'turn-lifecycle:turn-1' + // A requested close interrupts the open turn; the row is revised, not removed. + expect(tombstones).toEqual([]) + expect(lifecycles.at(-1)).toEqual({ + identity: { + provider: 'legacy', + agent: 'codex', + sessionId: 'session-1', + recordId: 'turn-lifecycle:turn-1' + }, + turnLifecycle: { + turnId: 'turn-1', + state: 'interrupted', + userItemId: `codex:${THREAD_ID}:turn-1:0`, + startedAt: expect.any(Number), + completedAt: expect.any(Number) + } }) }) }) diff --git a/src/main/codex/codex-structured-session-adapter.ts b/src/main/codex/codex-structured-session-adapter.ts index 8b0e649b177..061626f9724 100644 --- a/src/main/codex/codex-structured-session-adapter.ts +++ b/src/main/codex/codex-structured-session-adapter.ts @@ -30,9 +30,9 @@ import { type CodexStructuredSessionEvent } from './codex-structured-session-state' import { - deliverCodexNotification, deliverCodexServerRequest, - deliverCodexUnhandledFrame + deliverCodexUnhandledFrame, + translateCodexNotification } from './codex-structured-provider-events' import { CodexStructuredTurnCancellation } from './codex-structured-turn-cancellation' import { createCodexStructuredNotificationRetry } from './codex-structured-notification-retry' @@ -55,8 +55,16 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap constructor(private readonly deps: CodexStructuredSessionAdapterDeps) { this.notificationRetries = createCodexStructuredNotificationRetry({ sessionFor: (sessionId) => this.sessions.get(sessionId), - translate: (sessionId, session, method, params) => - this.translateNotification(sessionId, session, method, params) + translate: (sessionId, session, method, params, observedAt) => + translateCodexNotification({ + sessionId, + session, + method, + params, + observedAt, + turnCancellation: this.turnCancellation, + emit: (current, event) => this.emit(current, event) + }) }) this.teardown = new CodexStructuredSessionTeardown({ sessions: this.sessions, @@ -74,7 +82,8 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap emit: (session, event) => { const admission = this.emit(session, event) if (!admission.accepted && event.type === 'notification') { - this.notificationRetries.handle(event.sessionId, event.method, event.params) + const { sessionId, method, params, observedAt } = event + this.notificationRetries.handle(sessionId, method, params, observedAt) } return admission } @@ -120,21 +129,6 @@ export class CodexStructuredSessionAdapter implements StructuredAgentSessionAdap } } - private translateNotification( - sessionId: string, - session: CodexSession, - method: string, - params: unknown - ): CodexJournalTranslationAdmission { - codexRewind.observeCodexRewindActivity(session, method, params) - if (this.turnCancellation.handleNotification(sessionId, session, method, params)) { - return { accepted: true } - } - return deliverCodexNotification(sessionId, session, method, params, (current, event) => - this.emit(current, event) - ) - } - /** Journal first so observers never see an event ahead of its durable row. */ private emit( session: CodexSession, diff --git a/src/main/codex/codex-structured-session-background-tasks.test.ts b/src/main/codex/codex-structured-session-background-tasks.test.ts index aa6156027e8..54282e3d107 100644 --- a/src/main/codex/codex-structured-session-background-tasks.test.ts +++ b/src/main/codex/codex-structured-session-background-tasks.test.ts @@ -196,8 +196,10 @@ describe('codex background tasks reach the strip', () => { } }) await vi.waitFor(() => expect(adapter.backgroundTaskState('session-1')).toBeUndefined()) + // The open turn's lifecycle row is revised to interrupted, never tombstoned. expect(appendItem.mock.calls.map((call) => call[1])).toEqual([ - { kind: 'status', text: 'Provider exited: notification admission failed (failed)' } + { kind: 'status', text: 'Provider exited: notification admission failed (failed)' }, + expect.objectContaining({ kind: 'turn', state: 'interrupted' }) ]) expect(observed).toEqual([ expect.objectContaining({ @@ -215,29 +217,27 @@ describe('codex background tasks reach the strip', () => { } }) - it('publishes the orphaned fan-out once the spawning turn completes', async () => { + it('publishes the fan-out while it runs and keeps it past the spawning turn', async () => { const published: { sessionId: string; state: AgentSessionBackgroundTaskState | null }[] = [] const { adapter, codex } = await adapterWithSession(published) + const running = { + state: 'monitoring', + supportsStopAll: false, + tasks: [{ id: `codex-agent:${CHILD_ID}`, kind: 'agent', description: 'count_a' }] + } const spawn = subagentNotification('started') codex.handlers().onNotification?.(spawn.method, spawn.params) - // The child is still inside the turn, so the strip stays silent. - expect(published).toEqual([]) - expect(adapter.backgroundTaskState('session-1')).toBeNull() + // Mid-turn: the child is running, so the strip reports it now. + expect(published).toEqual([{ sessionId: 'session-1', state: running }]) + expect(adapter.backgroundTaskState('session-1')).toEqual(running) + published.length = 0 codex.handlers().onNotification?.(TURN_COMPLETED.method, TURN_COMPLETED.params) - expect(published).toEqual([ - { - sessionId: 'session-1', - state: { - state: 'monitoring', - supportsStopAll: false, - tasks: [{ id: `codex-agent:${CHILD_ID}`, kind: 'agent', description: 'count_a' }] - } - } - ]) - expect(adapter.backgroundTaskState('session-1')).toEqual(published[0].state) + // Turn end is not the child's outcome: no republish and no settle. + expect(published).toEqual([]) + expect(adapter.backgroundTaskState('session-1')).toEqual(running) }) it('clears the strip when the session closes', async () => { diff --git a/src/main/codex/codex-structured-session-cancel.test.ts b/src/main/codex/codex-structured-session-cancel.test.ts index aed0722b79b..2ea81d44786 100644 --- a/src/main/codex/codex-structured-session-cancel.test.ts +++ b/src/main/codex/codex-structured-session-cancel.test.ts @@ -80,7 +80,10 @@ async function acquired( codex: ReturnType, events: CodexStructuredSessionEvent[] = [], processControl: Partial< - Pick + Pick< + CodexStructuredSessionAdapterDeps, + 'captureTurnProcesses' | 'terminateTurnProcesses' | 'now' + > > = {} ): Promise { const adapter = new CodexStructuredSessionAdapter({ @@ -260,6 +263,32 @@ describe('CodexStructuredSessionAdapter.cancelTurn', () => { }) }) + it('keeps the receipt time of a completion deferred behind physical termination', async () => { + const events: CodexStructuredSessionEvent[] = [] + let clock = 5_000 + let finishTermination!: (terminated: boolean) => void + const termination = new Promise((resolve) => { + finishTermination = resolve + }) + const codex = fakeCodex() + codex.routes['turn/interrupt'] = () => { + completeTurn(codex) + return {} + } + const adapter = await acquired(codex, events, { + terminateTurnProcesses: async () => termination, + now: () => clock + }) + + const pending = adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-1', fence: 7 }) + await vi.waitFor(() => expect(codex.connections[0].calls.at(-1)?.method).toBe('turn/interrupt')) + clock = 9_000 + finishTermination(true) + await expect(pending).resolves.toEqual({ cancelled: true }) + + expect(events.at(-1)).toMatchObject({ method: 'turn/completed', observedAt: 5_000 }) + }) + it('does not strand a deferred completion when the interrupt receipt fails', async () => { const events: CodexStructuredSessionEvent[] = [] const codex = fakeCodex() diff --git a/src/main/codex/codex-structured-session-close.ts b/src/main/codex/codex-structured-session-close.ts index 5b29dc6c056..af814c51d9b 100644 --- a/src/main/codex/codex-structured-session-close.ts +++ b/src/main/codex/codex-structured-session-close.ts @@ -24,13 +24,15 @@ export function handleCodexSessionExit(input: { input.prompts?.clear() return false } + session.exitObservedAt ??= Date.now() const event: StructuredAgentSessionLifecycleEvent = { type: 'ended', sessionId: input.sessionId, reason: input.error.message, cause: session.requestedClose ? 'requested-close' : 'unexpected-exit', fence: session.fence, - acquisitionGeneration: session.acquisitionGeneration + acquisitionGeneration: session.acquisitionGeneration, + observedAt: session.exitObservedAt } as const // A synchronous sink rejection (usually backpressure) is handed to host // recovery, which appends the bounded fallback before reacquisition. diff --git a/src/main/codex/codex-structured-session-state.ts b/src/main/codex/codex-structured-session-state.ts index d004b0006e6..b341862d218 100644 --- a/src/main/codex/codex-structured-session-state.ts +++ b/src/main/codex/codex-structured-session-state.ts @@ -23,7 +23,15 @@ export type CodexStructuredLaunch = { } export type CodexStructuredSessionEvent = - | { type: 'notification'; sessionId: string; threadId: string; method: string; params: unknown } + | { + type: 'notification' + sessionId: string + threadId: string + method: string + params: unknown + /** Host receipt time of a turn boundary; survives retry and deferral so a replay is not re-stamped. */ + observedAt?: number + } | { type: 'server-request'; sessionId: string; threadId: string; method: string; params: unknown } | { type: 'provider-frame'; sessionId: string; threadId: string; kind: string; payload: unknown } | { @@ -37,7 +45,7 @@ export type CodexStructuredSessionEvent = } | StructuredAgentSessionLifecycleEvent /** Translator-only compatibility for callers that do not participate in host recovery. */ - | { type: 'ended'; sessionId: string; reason: string } + | { type: 'ended'; sessionId: string; reason: string; observedAt?: number } export type CodexStructuredSessionAdapterDeps = { resolveLaunch: (input: { @@ -66,6 +74,8 @@ export type CodexStructuredSessionAdapterDeps = { export type CodexSession = { connection: CodexAppServerConnection ended: boolean + /** First observed child exit survives rejected settlement admission. */ + exitObservedAt?: number requestedClose: boolean fence: number acquisitionGeneration: string diff --git a/src/main/codex/codex-structured-thread-facts.ts b/src/main/codex/codex-structured-thread-facts.ts index 349246ecf29..9b8213a57fc 100644 --- a/src/main/codex/codex-structured-thread-facts.ts +++ b/src/main/codex/codex-structured-thread-facts.ts @@ -37,3 +37,22 @@ export function readCodexTurnId(payload: unknown): string | null { } return nonEmptyString(record(root.turn)?.id) ?? nonEmptyString(root.turnId) } + +/** `turn/completed` carries `turn.status`; thread history puts `status` on the turn record itself. */ +export function readCodexTurnStatus(payload: unknown): string | null { + const root = record(payload) + if (!root) { + return null + } + return nonEmptyString(record(root.turn)?.status) ?? nonEmptyString(root.status) +} + +/** Codex's own turn duration, already in milliseconds; absent or malformed reads as null. */ +export function readCodexTurnDurationMs(payload: unknown): number | null { + const root = record(payload) + if (!root) { + return null + } + const value = record(root.turn)?.durationMs ?? root.durationMs + return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : null +} diff --git a/src/main/codex/codex-structured-turn-cancellation.ts b/src/main/codex/codex-structured-turn-cancellation.ts index 1f31197e8d5..97257d54fa4 100644 --- a/src/main/codex/codex-structured-turn-cancellation.ts +++ b/src/main/codex/codex-structured-turn-cancellation.ts @@ -50,7 +50,8 @@ export class CodexStructuredTurnCancellation { sessionId: string, session: CodexSession, method: string, - params: unknown + params: unknown, + observedAt?: number ): boolean { const threadId = readCodexThreadId(params) ?? session.threadId if (method !== 'turn/completed' || threadId !== session.threadId) { @@ -66,7 +67,8 @@ export class CodexStructuredTurnCancellation { sessionId, threadId, method, - params + params, + ...(observedAt !== undefined ? { observedAt } : {}) } state.deferredCompletions.set(turnId, event) return true diff --git a/src/main/codex/codex-structured-turn-start.ts b/src/main/codex/codex-structured-turn-start.ts index ffe4c850d5c..a24c5b6d5b8 100644 --- a/src/main/codex/codex-structured-turn-start.ts +++ b/src/main/codex/codex-structured-turn-start.ts @@ -7,6 +7,7 @@ import { } from './codex-app-server-connection' import { isCodexAppServerUnsupportedError } from './codex-app-server-session' import { readCodexTurnId } from './codex-structured-thread-facts' +import { DISPATCH_DOUBT_CODEX_TURN_UNNAMED } from '../native-chat/agent-session-journal/journal-dispatch-doubt-reasons' // Starting a Codex turn and learning its id, which are not the same event: // `turn/start` returns the id on newer builds and acks before it exists on @@ -115,7 +116,7 @@ export async function dispatchCodexTurn( throw error } return turnId === null - ? { state: 'unknown', reason: 'codex app-server started a turn it did not name in time' } + ? { state: 'unknown', reason: DISPATCH_DOUBT_CODEX_TURN_UNNAMED } : { state: 'accepted', providerIdentity: { diff --git a/src/main/daemon/daemon-launch-paths.ts b/src/main/daemon/daemon-launch-paths.ts index 9b5b2c3dca8..049800daff4 100644 --- a/src/main/daemon/daemon-launch-paths.ts +++ b/src/main/daemon/daemon-launch-paths.ts @@ -1,7 +1,9 @@ -import { existsSync, mkdirSync } from 'node:fs' +import { existsSync } from 'node:fs' import { connect } from 'node:net' import { join } from 'node:path' import { getAppEnvironment } from '../../shared/app-environment' +import { ensurePrivateDir } from './daemon-private-file-modes' +import { scheduleTerminalHistoryPermissionRepair } from './terminal-history-permission-repair' import { getDaemonLogFilePath } from '../observability/logs-directory' import { DaemonClient } from './client' import { daemonRecoveryProbeTimeoutMs } from './daemon-recovery-budget' @@ -10,13 +12,17 @@ import { PROTOCOL_VERSION, type ListSessionsResult } from './types' export function getDaemonRuntimeDir(): string { const dir = join(getAppEnvironment().getPath('userData'), 'daemon') - mkdirSync(dir, { recursive: true }) + ensurePrivateDir(dir) return dir } export function getDaemonHistoryDir(): string { const dir = join(getAppEnvironment().getPath('userData'), 'terminal-history') - mkdirSync(dir, { recursive: true }) + ensurePrivateDir(dir) + // Why here: the one accessor every history producer goes through, so the backlog sweep is hooked + // once per host that owns the files — native, WSL, or a remote SSH server's own main process. + // The scheduler defers and de-duplicates, so the several startup calls cost one late sweep. + void scheduleTerminalHistoryPermissionRepair(dir) return dir } diff --git a/src/main/daemon/daemon-private-file-modes.ts b/src/main/daemon/daemon-private-file-modes.ts new file mode 100644 index 00000000000..c0c96b9021d --- /dev/null +++ b/src/main/daemon/daemon-private-file-modes.ts @@ -0,0 +1,33 @@ +// Mode primitives for daemon-owned on-disk state. Terminal history persists verbatim screen and +// scrollback (checkpoint.json holds snapshotAnsi + scrollbackAnsi) and the runtime dir holds the +// daemon's auth token, so neither may be left at whatever umask applies to other local users. + +import { chmodSync, existsSync, mkdirSync } from 'node:fs' + +export const PRIVATE_DIR_MODE = 0o700 +export const PRIVATE_FILE_MODE = 0o600 + +/** Windows ignores POSIX mode bits and can reject chmod outright; hardening must never break a write. */ +export function supportsPosixFileModes(): boolean { + return process.platform !== 'win32' +} + +/** Best-effort repair for a path created before modes were pinned (or by an older daemon). */ +export function tightenPathMode(path: string, mode: number): void { + if (!supportsPosixFileModes()) { + return + } + try { + if (existsSync(path)) { + chmodSync(path, mode) + } + } catch { + // Read-only volumes, foreign ownership, exotic filesystems: leave the mode as found. + } +} + +/** mkdir with the private mode, plus a chmod repair for a directory that already existed. */ +export function ensurePrivateDir(dir: string): void { + mkdirSync(dir, { recursive: true, mode: PRIVATE_DIR_MODE }) + tightenPathMode(dir, PRIVATE_DIR_MODE) +} diff --git a/src/main/daemon/headless-osc-link-ranges.test.ts b/src/main/daemon/headless-osc-link-ranges.test.ts new file mode 100644 index 00000000000..cf8f9f757e2 --- /dev/null +++ b/src/main/daemon/headless-osc-link-ranges.test.ts @@ -0,0 +1,63 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { HeadlessEmulator } from './headless-emulator' + +// Why this suite: collectHeadlessOscLinkRanges skips its per-cell scan when +// xterm holds no OSC 8 registration. That skip is only safe if it can never +// fire while a link is reachable, so each case below pins one way it could. +let emulator: HeadlessEmulator | undefined + +const link = (uri: string, text: string): string => `\x1b]8;;${uri}\x1b\\${text}\x1b]8;;\x1b\\` + +afterEach(() => { + emulator?.dispose() + emulator = undefined +}) + +describe('headless OSC link ranges', () => { + it('finds a link written into the buffer', async () => { + emulator = new HeadlessEmulator({ cols: 80, rows: 24 }) + await emulator.write(`before ${link('https://example.com/a', 'CLICK')} after`) + + const ranges = emulator.getSnapshot().oscLinks ?? [] + expect(ranges).toHaveLength(1) + expect(ranges[0]).toMatchObject({ row: 0, uri: 'https://example.com/a' }) + }) + + it('returns nothing for a buffer that never emitted a link', async () => { + emulator = new HeadlessEmulator({ cols: 80, rows: 24 }) + await emulator.write('plain output with no hyperlink\r\n'.repeat(50)) + + expect(emulator.getSnapshot().oscLinks).toEqual([]) + }) + + // The dangerous case: restored ranges are seeded without xterm registering + // anything, so an early-out keyed only on the registry would drop them. + it('still maps restored ranges when the buffer itself has no link', async () => { + emulator = new HeadlessEmulator({ cols: 80, rows: 24 }) + await emulator.write('restored row') + const restored = { row: 0, startCol: 0, endCol: 4, uri: 'https://example.com/restored' } + emulator.setRestoredOscLinks([restored]) + + expect(emulator.getSnapshot().oscLinks).toEqual([restored]) + }) + + it('finds links far down a long scrollback, not just the visible screen', async () => { + emulator = new HeadlessEmulator({ cols: 80, rows: 24, scrollback: 5_000 }) + await emulator.write(`${link('https://example.com/top', 'TOP')}\r\n`) + await emulator.write('filler\r\n'.repeat(2_000)) + + const ranges = emulator.getSnapshot({ scrollbackRows: 5_000 }).oscLinks ?? [] + expect(ranges.map((range) => range.uri)).toContain('https://example.com/top') + }) + + it('keeps every distinct link when several are present', async () => { + emulator = new HeadlessEmulator({ cols: 80, rows: 24 }) + await emulator.write( + `${link('https://example.com/1', 'ONE')} ${link('https://example.com/2', 'TWO')}` + ) + + const uris = (emulator.getSnapshot().oscLinks ?? []).map((range) => range.uri) + expect(uris).toContain('https://example.com/1') + expect(uris).toContain('https://example.com/2') + }) +}) diff --git a/src/main/daemon/headless-osc-link-ranges.ts b/src/main/daemon/headless-osc-link-ranges.ts index 418a0c65166..ea017a7b928 100644 --- a/src/main/daemon/headless-osc-link-ranges.ts +++ b/src/main/daemon/headless-osc-link-ranges.ts @@ -1,10 +1,14 @@ -import type { Terminal } from '@xterm/headless' +import type { IBufferCell, IBufferLine, Terminal } from '@xterm/headless' import type { TerminalOscLinkRange } from '../../shared/terminal-osc-link-ranges' type TerminalWithOscLinks = Terminal & { _core?: { _oscLinkService?: { getLinkData: (linkId: number) => { uri?: string } | undefined + // Why read it: xterm registers every OSC 8 id here, so an empty registry + // proves the buffer holds no hyperlink and the per-cell scan can be skipped. + // Optional because it is private — an xterm that renames it just scans. + _dataByLinkId?: { size?: number } } } } @@ -14,6 +18,11 @@ type CellWithOscLink = { hasExtendedAttrs?: () => boolean } +/** True when xterm holds no OSC 8 registration at all, so no cell can carry one. */ +function hasNoRegisteredOscLinks(service: { _dataByLinkId?: { size?: number } }): boolean { + return service._dataByLinkId?.size === 0 +} + export function collectHeadlessOscLinkRanges( terminal: Terminal, scrollbackRows: number | undefined, @@ -26,9 +35,19 @@ export function collectHeadlessOscLinkRanges( return [] } const buffer = terminal.buffer.active + // Why before the scan: the walk below reads every cell of every row, and a + // session that never emitted a hyperlink — the overwhelming majority — would + // pay that for a guaranteed-empty result. `restoredLinks` still needs mapping. + if (hasNoRegisteredOscLinks(service) && restoredLinks.length === 0) { + return [] + } const startRow = scrollbackRows === undefined ? 0 : Math.max(0, buffer.length - terminal.rows - scrollbackRows) const ranges: TerminalOscLinkRange[] = [] + // Why one cell for the whole walk: xterm's getCell allocates a fresh CellData + // per call unless handed a target, which is a per-cell allocation across the + // entire scrollback. See the IBufferLine.getCell docs. + const scratchCell = buffer.getNullCell() for (let row = startRow; row < buffer.length; row += 1) { const line = buffer.getLine(row) if (!line) { @@ -38,7 +57,7 @@ export function collectHeadlessOscLinkRanges( let currentUrlId = 0 let currentStart = -1 for (let col = 0; col <= lineLength; col += 1) { - const urlId = col < lineLength ? getOscLinkIdAtCell(line, col) : 0 + const urlId = col < lineLength ? getOscLinkIdAtCell(line, col, scratchCell) : 0 if (urlId === currentUrlId) { continue } @@ -83,8 +102,8 @@ function dedupeOscLinkRanges(ranges: TerminalOscLinkRange[]): TerminalOscLinkRan }) } -function getOscLinkIdAtCell(line: { getCell: (col: number) => unknown }, col: number): number { - const cell = line.getCell(col) as CellWithOscLink | undefined +function getOscLinkIdAtCell(line: IBufferLine, col: number, scratchCell: IBufferCell): number { + const cell = line.getCell(col, scratchCell) as (IBufferCell & CellWithOscLink) | undefined // Why: OSC link IDs live in extended cell attrs; missing attrs means no link. return cell?.hasExtendedAttrs?.() && cell.extended?.urlId ? cell.extended.urlId : 0 } diff --git a/src/main/daemon/history-manager.ts b/src/main/daemon/history-manager.ts index 48517bb6653..200609ed703 100644 --- a/src/main/daemon/history-manager.ts +++ b/src/main/daemon/history-manager.ts @@ -1,7 +1,9 @@ import { join } from 'node:path' import { randomUUID } from 'node:crypto' -import { mkdirSync, writeFileSync, existsSync, unlinkSync } from 'node:fs' +import { existsSync } from 'node:fs' import { getHistorySessionDirName } from './history-paths' +import { ensurePrivateDir } from './daemon-private-file-modes' +import { clearReplayableTerminalHistorySessionFiles } from './terminal-history-session-files' import { fingerprintTerminalHistorySession, hasTerminalHistoryRecoveryProtection, @@ -9,6 +11,7 @@ import { type ActiveHistoryRecoveryFreeze, type HistoryRecoveryFreeze } from './terminal-history-recovery-quarantine' +import { TerminalHistoryRecoveryFreezes } from './terminal-history-recovery-freezes' import { removeTerminalHistorySessionTrees, schedulePendingSessionTreeRemovals @@ -17,6 +20,7 @@ import { TerminalHistorySessionWriter } from './terminal-history-session-writer' import { readTerminalHistoryMetaFromDir, updateTerminalHistoryMeta, + writeTerminalHistoryMeta, type SessionMeta } from './terminal-history-metadata' import type { PendingOutputRecord, TerminalSnapshot } from './types' @@ -36,7 +40,7 @@ export class HistoryManager { private writers = new Map() private disabledSessions = new Set() private mutations = new TerminalHistoryMutationTracker() - private recoveryFreezes = new Map() + private readonly recoveryFreezes: TerminalHistoryRecoveryFreezes private onWriteError?: (sessionId: string, error: Error) => void private checkpointMaxBytes: number @@ -46,6 +50,7 @@ export class HistoryManager { ) { this.onWriteError = opts?.onWriteError this.checkpointMaxBytes = opts?.checkpointMaxBytes ?? TERMINAL_HISTORY_CHECKPOINT_MAX_BYTES + this.recoveryFreezes = new TerminalHistoryRecoveryFreezes(basePath) // Why: a quit between tombstone and reclaim leaves the tree on disk; nothing else rescans the queue. schedulePendingSessionTreeRemovals(this.basePath) } @@ -54,7 +59,7 @@ export class HistoryManager { let recoveryFreeze = opts.recoveryFreeze try { this.disabledSessions.delete(sessionId) - const dir = join(this.basePath, getHistorySessionDirName(sessionId)) + const dir = this.sessionDir(sessionId) recoveryFreeze ??= await this.freezeForRecovery(sessionId) const activeFreeze = this.requireRecoveryFreeze(sessionId, recoveryFreeze) @@ -67,8 +72,8 @@ export class HistoryManager { ) { throw new Error('terminal_history_recovery_generation_changed') } - this.recoveryFreezes.delete(sessionId) - mkdirSync(dir, { recursive: true }) + this.recoveryFreezes.release(sessionId) + ensurePrivateDir(dir) const meta: SessionMeta = { cwd: opts.cwd, @@ -78,21 +83,10 @@ export class HistoryManager { endedAt: null, exitCode: null } - writeFileSync(join(dir, 'meta.json'), JSON.stringify(meta, null, 2)) + writeTerminalHistoryMeta(dir, meta) if (!opts.quarantineUnreadableRecovery) { - // Why: a crash before the first checkpoint must not replay a cleanly ended prior session. - for (const staleFile of [ - join(dir, 'checkpoint.json'), - join(dir, 'scrollback.bin'), - join(dir, 'output.log') - ]) { - try { - unlinkSync(staleFile) - } catch { - // ENOENT is expected for new sessions - } - } + clearReplayableTerminalHistorySessionFiles(dir) } this.writers.set( @@ -118,14 +112,14 @@ export class HistoryManager { token: randomUUID() } const activeFreeze: ActiveHistoryRecoveryFreeze = { handle } - this.recoveryFreezes.set(sessionId, activeFreeze) + this.recoveryFreezes.hold(sessionId, activeFreeze) try { await this.mutations.wait(sessionId) activeFreeze.fingerprint = fingerprintTerminalHistorySession(this.basePath, sessionId) return handle } catch (err) { if (this.recoveryFreezes.get(sessionId) === activeFreeze) { - this.recoveryFreezes.delete(sessionId) + this.recoveryFreezes.release(sessionId) } throw err } @@ -134,7 +128,7 @@ export class HistoryManager { abandonRecoveryFreeze(freeze?: HistoryRecoveryFreeze): void { const activeFreeze = freeze ? this.recoveryFreezes.get(freeze.sessionId) : undefined if (activeFreeze && activeFreeze.handle === freeze) { - this.recoveryFreezes.delete(activeFreeze.handle.sessionId) + this.recoveryFreezes.release(activeFreeze.handle.sessionId) } } @@ -155,7 +149,7 @@ export class HistoryManager { ) { throw new Error('terminal_history_recovery_generation_changed') } - this.recoveryFreezes.delete(sessionId) + this.recoveryFreezes.release(sessionId) } catch (err) { this.abandonRecoveryFreeze(recoveryFreeze) this.handleWriteError(sessionId, err) @@ -164,7 +158,7 @@ export class HistoryManager { } else if (this.recoveryFreezes.has(sessionId)) { return } - const dir = join(this.basePath, getHistorySessionDirName(sessionId)) + const dir = this.sessionDir(sessionId) this.writers.set( sessionId, new TerminalHistorySessionWriter(dir, false, this.checkpointMaxBytes) @@ -280,7 +274,7 @@ export class HistoryManager { async removeSession(sessionId: string): Promise { this.writers.delete(sessionId) this.disabledSessions.delete(sessionId) - this.recoveryFreezes.delete(sessionId) + this.recoveryFreezes.release(sessionId) await this.mutations.wait(sessionId) // Why tombstoned: writer handles are closed by here, so the trees only have to become unreachable — // they reach hundreds of MB and every terminal a worktree delete tears down awaits this. @@ -300,12 +294,11 @@ export class HistoryManager { } hasHistory(sessionId: string): boolean { - return existsSync(join(this.basePath, getHistorySessionDirName(sessionId), 'meta.json')) + return existsSync(join(this.sessionDir(sessionId), 'meta.json')) } readMeta(sessionId: string): SessionMeta | null { - const dir = join(this.basePath, getHistorySessionDirName(sessionId)) - return readTerminalHistoryMetaFromDir(dir) + return readTerminalHistoryMetaFromDir(this.sessionDir(sessionId)) } async dispose(): Promise { @@ -321,6 +314,7 @@ export class HistoryManager { } } this.writers.clear() + this.recoveryFreezes.releaseAll() } // Why: history is best-effort; callers fire-and-forget so a throw would be an unhandled rejection — disable instead. @@ -329,6 +323,10 @@ export class HistoryManager { this.onWriteError?.(sessionId, err as Error) } + private sessionDir(sessionId: string): string { + return join(this.basePath, getHistorySessionDirName(sessionId)) + } + private requireRecoveryFreeze( sessionId: string, recoveryFreeze: HistoryRecoveryFreeze diff --git a/src/main/daemon/terminal-history-metadata.ts b/src/main/daemon/terminal-history-metadata.ts index b93fd9a4a78..fd473a99f07 100644 --- a/src/main/daemon/terminal-history-metadata.ts +++ b/src/main/daemon/terminal-history-metadata.ts @@ -4,6 +4,7 @@ import { getHistorySessionDirName } from './history-paths' import { isValidTerminalHistorySize } from './terminal-history-dimensions' import { readTerminalHistoryJson } from './terminal-history-file-reader' import { TERMINAL_HISTORY_META_MAX_BYTES } from './terminal-history-file-limits' +import { PRIVATE_FILE_MODE, tightenPathMode } from './daemon-private-file-modes' export type SessionMeta = { cwd: string @@ -44,13 +45,21 @@ export function readTerminalHistoryMetaFromDir(dir: string): SessionMeta | null } } +/** meta.json records the session's cwd, so it is private like the rest of the tree. */ +export function writeTerminalHistoryMeta(dir: string, meta: SessionMeta): void { + const metaPath = join(dir, 'meta.json') + writeFileSync(metaPath, JSON.stringify(meta, null, 2), { mode: PRIVATE_FILE_MODE }) + // `mode` applies only at creation, so a rewrite of an older daemon's file needs the chmod. + tightenPathMode(metaPath, PRIVATE_FILE_MODE) +} + export function updateTerminalHistoryMeta(dir: string, updates: Partial): void { const meta = readTerminalHistoryMetaFromDir(dir) if (!meta) { return } Object.assign(meta, updates) - writeFileSync(join(dir, 'meta.json'), JSON.stringify(meta, null, 2)) + writeTerminalHistoryMeta(dir, meta) } function isSessionMeta(value: unknown): value is SessionMeta { diff --git a/src/main/daemon/terminal-history-permission-repair.ts b/src/main/daemon/terminal-history-permission-repair.ts new file mode 100644 index 00000000000..d2449efebfb --- /dev/null +++ b/src/main/daemon/terminal-history-permission-repair.ts @@ -0,0 +1,118 @@ +// History trees written before owner-only modes were pinned landed at whatever umask applied, which on +// a default umask leaves every checkpoint.json world-readable. This is the backlog repair: one bounded +// sweep of the base dir, marker-guarded so every later launch costs one existsSync rather than a walk +// over 10k session trees. Live trees are tightened per-session in terminal-history-session-files. +// +// The marker is a regular file, so `history-reader`'s directory-only session scan already skips it. + +import { existsSync, type Dirent } from 'node:fs' +import { chmod, readdir, writeFile } from 'node:fs/promises' +import { join, resolve } from 'node:path' +import { + PRIVATE_DIR_MODE, + PRIVATE_FILE_MODE, + supportsPosixFileModes +} from './daemon-private-file-modes' +import { isTerminalHistorySessionDirRecoveryProtected } from './terminal-history-recovery-quarantine' + +const REPAIR_MARKER_NAME = '.permissions-repaired-v1' +// Bounds the one-time walk: retention keeps 10k session trees, each a handful of files. +const MAX_REPAIR_ENTRIES = 200_000 +// base → session/quarantine owner → quarantined generation → files. +const MAX_REPAIR_DEPTH = 3 +// Same 10s the sibling history GC waits before walking this very tree, and for the same reason: +// stay off startup-critical I/O (see scheduleHistoryGc in src/main/terminal-history-gc.ts). +const REPAIR_START_DELAY_MS = 10_000 + +// Per-process, keyed by base path: getDaemonHistoryDir() is the accessor every history producer +// goes through, and a single startup calls it more than once. Never cleared, so a sweep that throws +// cannot wedge a retry loop — the on-disk marker is what carries the decision across launches. +const scheduledBasePaths = new Set() + +async function chmodQuietly(path: string, mode: number): Promise { + try { + await chmod(path, mode) + } catch { + // A path that cannot be tightened must not abort the rest of the sweep. + } +} + +async function tightenTree(root: string): Promise { + const queue: { dir: string; depth: number }[] = [{ dir: root, depth: 0 }] + let budget = MAX_REPAIR_ENTRIES + while (queue.length > 0 && budget > 0) { + const current = queue.shift() + if (!current) { + return + } + // Why skip: chmod moves the mode/ctime that the recovery fingerprint hashes, so sweeping a tree + // mid-freeze fails the re-check and silently stops that pane persisting for the rest of the run. + // Nothing is left loose — the session's own writer tightens its tree when it attaches. + if (current.depth > 0 && isTerminalHistorySessionDirRecoveryProtected(current.dir)) { + continue + } + await chmodQuietly(current.dir, PRIVATE_DIR_MODE) + let entries: Dirent[] + try { + entries = await readdir(current.dir, { withFileTypes: true }) + } catch { + continue + } + for (const entry of entries) { + budget -= 1 + if (budget <= 0) { + return + } + // Dirent types come from lstat, so symlinks match neither branch and are never chased. + const child = join(current.dir, entry.name) + if (entry.isDirectory()) { + if (current.depth < MAX_REPAIR_DEPTH) { + queue.push({ dir: child, depth: current.depth + 1 }) + } + } else if (entry.isFile()) { + // Re-checked per file: a freeze can open while this directory is being walked. + if (current.depth > 0 && isTerminalHistorySessionDirRecoveryProtected(current.dir)) { + break + } + await chmodQuietly(child, PRIVATE_FILE_MODE) + } + } + } +} + +/** Resolves `true` when the sweep ran. The marker is written even if some paths resisted chmod, so a + * permanently unfixable file cannot make every launch re-walk the tree. */ +export async function repairTerminalHistoryPermissions(basePath: string): Promise { + if (!supportsPosixFileModes() || !existsSync(basePath)) { + return false + } + const markerPath = join(basePath, REPAIR_MARKER_NAME) + if (existsSync(markerPath)) { + return false + } + await tightenTree(basePath) + try { + await writeFile(markerPath, '', { mode: PRIVATE_FILE_MODE }) + } catch { + // Marker write failed: the next launch repeats a bounded, idempotent sweep. + } + return true +} + +/** Deferred and once per base path per process, so daemon init neither waits on permission hardening + * nor runs two sweeps over one tree. Resolves with the sweep's outcome, or `null` when already + * scheduled; callers on the startup path ignore it. */ +export function scheduleTerminalHistoryPermissionRepair(basePath: string): Promise | null { + const key = resolve(basePath) + if (scheduledBasePaths.has(key)) { + return null + } + scheduledBasePaths.add(key) + const { promise, resolve: settle } = Promise.withResolvers() + const timer = setTimeout(() => { + repairTerminalHistoryPermissions(key).then(settle, () => settle(false)) + }, REPAIR_START_DELAY_MS) + // Why: a pending sweep must never be the reason the process (or a test worker) stays alive. + timer.unref() + return promise +} diff --git a/src/main/daemon/terminal-history-permissions.test.ts b/src/main/daemon/terminal-history-permissions.test.ts new file mode 100644 index 00000000000..91ea830b4b2 --- /dev/null +++ b/src/main/daemon/terminal-history-permissions.test.ts @@ -0,0 +1,334 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + chmodSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + rmSync, + statSync, + writeFileSync +} from 'node:fs' +import type * as NodeFs from 'node:fs' +import type * as NodeFsPromises from 'node:fs/promises' +import { HistoryManager } from './history-manager' +import { HistoryReader } from './history-reader' +import { getHistorySessionDirName } from './history-paths' +import { flushPendingSessionTreeRemovals } from './terminal-history-session-tombstone' +import { + repairTerminalHistoryPermissions, + scheduleTerminalHistoryPermissionRepair +} from './terminal-history-permission-repair' +import { tightenTerminalHistorySessionDirMode } from './terminal-history-session-files' +import type { TerminalModes, TerminalSnapshot } from './types' + +const onPosix = it.skipIf(process.platform === 'win32') +const REPAIR_MARKER_NAME = '.permissions-repaired-v1' + +const defaultModes: TerminalModes = { + bracketedPaste: false, + mouseTracking: false, + applicationCursor: false, + alternateScreen: false +} + +function makeSnapshot(overrides: Partial = {}): TerminalSnapshot { + return { + snapshotAnsi: 'secret scrollback\r\n', + scrollbackAnsi: '', + rehydrateSequences: '', + cwd: '/tmp', + modes: defaultModes, + cols: 80, + rows: 24, + scrollbackLines: 0, + ...overrides + } +} + +function modeOf(path: string): number { + return statSync(path).mode & 0o777 +} + +function sessionPath(baseDir: string, sessionId: string, file: string): string { + return join(baseDir, getHistorySessionDirName(sessionId), file) +} + +/** `process.platform` is read at call time, so the Windows branch is reachable from a POSIX runner. */ +function stubPlatform(platform: NodeJS.Platform): () => void { + const original = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { value: platform, configurable: true }) + return () => { + if (original) { + Object.defineProperty(process, 'platform', original) + } + } +} + +describe('terminal history file permissions', () => { + const createdDirs: string[] = [] + + /** Repair tests need a base dir with no HistoryManager racing its own startup sweep against them. */ + function isolatedDir(): string { + const created = mkdtempSync(join(tmpdir(), 'history-perms-test-')) + createdDirs.push(created) + return created + } + + afterEach(async () => { + await flushPendingSessionTreeRemovals() + for (const created of createdDirs.splice(0)) { + rmSync(created, { recursive: true, force: true }) + } + }) + + describe('newly written history', () => { + let dir: string + let mgr: HistoryManager + + beforeEach(() => { + dir = isolatedDir() + mgr = new HistoryManager(dir) + }) + + afterEach(async () => { + await mgr.dispose() + }) + + onPosix('pins 0o700 on the session directory and 0o600 on meta.json', async () => { + await mgr.openSession('sess-1', { cwd: '/home/user', cols: 80, rows: 24 }) + + expect(modeOf(join(dir, getHistorySessionDirName('sess-1')))).toBe(0o700) + expect(modeOf(sessionPath(dir, 'sess-1', 'meta.json'))).toBe(0o600) + }) + + onPosix('pins 0o600 on checkpoint.json, which holds verbatim scrollback', async () => { + await mgr.openSession('sess-1', { cwd: '/tmp', cols: 80, rows: 24 }) + await mgr.checkpoint('sess-1', makeSnapshot()) + + const checkpointPath = sessionPath(dir, 'sess-1', 'checkpoint.json') + expect(readFileSync(checkpointPath, 'utf-8')).toContain('secret scrollback') + expect(modeOf(checkpointPath)).toBe(0o600) + }) + + onPosix('pins 0o600 on output.log', async () => { + await mgr.openSession('sess-1', { cwd: '/tmp', cols: 80, rows: 24 }) + await mgr.appendIncrements('sess-1', 1, [{ kind: 'output', data: 'secret increment' }]) + + expect(modeOf(sessionPath(dir, 'sess-1', 'output.log'))).toBe(0o600) + }) + + onPosix( + 'tightens a checkpoint tmp left behind by an older daemon before renaming it', + async () => { + await mgr.openSession('sess-1', { cwd: '/tmp', cols: 80, rows: 24 }) + const tmpPath = `${sessionPath(dir, 'sess-1', 'checkpoint.json')}.tmp` + writeFileSync(tmpPath, 'stale', { mode: 0o644 }) + + await mgr.checkpoint('sess-1', makeSnapshot()) + + expect(modeOf(sessionPath(dir, 'sess-1', 'checkpoint.json'))).toBe(0o600) + } + ) + + onPosix('keeps the sweep marker out of the restorable-session listing', async () => { + await mgr.openSession('sess-1', { cwd: '/tmp', cols: 80, rows: 24 }) + await repairTerminalHistoryPermissions(dir) + + expect(existsSync(join(dir, REPAIR_MARKER_NAME))).toBe(true) + expect(new HistoryReader(dir).listRestorable()).toEqual(['sess-1']) + }) + }) + + describe('repairing history written before modes were pinned', () => { + /** A base dir shaped like one written under a default umask: world-readable throughout. */ + function seedLegacyTree(): { base: string; sessionDir: string; checkpointPath: string } { + const base = isolatedDir() + const sessionDir = join(base, getHistorySessionDirName('legacy')) + mkdirSync(sessionDir, { recursive: true }) + chmodSync(base, 0o755) + chmodSync(sessionDir, 0o755) + const checkpointPath = join(sessionDir, 'checkpoint.json') + writeFileSync(checkpointPath, '{"scrollbackAnsi":"secret"}') + chmodSync(checkpointPath, 0o644) + return { base, sessionDir, checkpointPath } + } + + onPosix('tightens a pre-existing 0o644 session tree when its writer attaches', () => { + const { sessionDir, checkpointPath } = seedLegacyTree() + + tightenTerminalHistorySessionDirMode(sessionDir) + + expect(modeOf(sessionDir)).toBe(0o700) + expect(modeOf(checkpointPath)).toBe(0o600) + }) + + onPosix('sweeps the whole base dir once and then short-circuits', async () => { + const { base, sessionDir, checkpointPath } = seedLegacyTree() + + await expect(repairTerminalHistoryPermissions(base)).resolves.toBe(true) + expect(modeOf(base)).toBe(0o700) + expect(modeOf(sessionDir)).toBe(0o700) + expect(modeOf(checkpointPath)).toBe(0o600) + + // Marker-guarded: a later launch must not re-walk 10k session trees. + chmodSync(checkpointPath, 0o644) + await expect(repairTerminalHistoryPermissions(base)).resolves.toBe(false) + expect(modeOf(checkpointPath)).toBe(0o644) + }) + + onPosix('leaves a session under an open recovery freeze alone', async () => { + const { base, sessionDir: legacyDir, checkpointPath } = seedLegacyTree() + const writeErrors: Error[] = [] + const mgr = new HistoryManager(base, { + onWriteError: (_sessionId, error) => writeErrors.push(error) + }) + try { + await mgr.openSession('frozen', { cwd: '/tmp', cols: 80, rows: 24 }) + await mgr.checkpoint('frozen', makeSnapshot()) + + // The production ordering: freeze fingerprints, the sweep runs, then the writer re-registers. + const freeze = await mgr.freezeForRecovery('frozen') + await expect(repairTerminalHistoryPermissions(base)).resolves.toBe(true) + mgr.registerWriter('frozen', freeze) + + expect(writeErrors.map((error) => error.message)).toEqual([]) + expect(mgr.isSessionDisabled('frozen')).toBe(false) + // Persistence, not just the absence of an error: the pane must still reach disk. + await mgr.checkpoint('frozen', makeSnapshot({ snapshotAnsi: 'after the sweep\r\n' })) + expect(readFileSync(sessionPath(base, 'frozen', 'checkpoint.json'), 'utf-8')).toContain( + 'after the sweep' + ) + } finally { + await mgr.dispose() + } + + // Narrow skip: every session that is not frozen is still tightened by the same sweep. + expect(modeOf(legacyDir)).toBe(0o700) + expect(modeOf(checkpointPath)).toBe(0o600) + }) + + onPosix('sweeps a session tree once its recovery freeze is released', async () => { + const base = isolatedDir() + const mgr = new HistoryManager(base) + try { + await mgr.openSession('thawed', { cwd: '/tmp', cols: 80, rows: 24 }) + const freeze = await mgr.freezeForRecovery('thawed') + mgr.abandonRecoveryFreeze(freeze) + } finally { + await mgr.dispose() + } + const sessionDir = join(base, getHistorySessionDirName('thawed')) + chmodSync(sessionDir, 0o755) + + await expect(repairTerminalHistoryPermissions(base)).resolves.toBe(true) + + expect(modeOf(sessionDir)).toBe(0o700) + }) + + onPosix('defers the sweep off the daemon-init critical path and runs it once', async () => { + const { base, checkpointPath } = seedLegacyTree() + vi.useFakeTimers() + try { + const first = scheduleTerminalHistoryPermissionRepair(base) + // Both startup accessors ask for the same tree; only the first arms a sweep. + expect(scheduleTerminalHistoryPermissionRepair(base)).toBeNull() + expect(vi.getTimerCount()).toBe(1) + + // Still armed, and the tree still untouched, well past daemon init — the sibling + // history GC waits the same 10s over this directory for the same reason. + await vi.advanceTimersByTimeAsync(9_999) + expect(vi.getTimerCount()).toBe(1) + expect(existsSync(join(base, REPAIR_MARKER_NAME))).toBe(false) + expect(modeOf(checkpointPath)).toBe(0o644) + + await vi.advanceTimersByTimeAsync(1) + await expect(first).resolves.toBe(true) + } finally { + vi.useRealTimers() + } + expect(modeOf(checkpointPath)).toBe(0o600) + }) + + onPosix('finishes and marks the sweep done even when every chmod is rejected', async () => { + const { base } = seedLegacyTree() + vi.resetModules() + vi.doMock('node:fs/promises', async () => { + const actual = await vi.importActual('node:fs/promises') + return { + ...actual, + default: actual, + chmod: () => Promise.reject(Object.assign(new Error('EPERM'), { code: 'EPERM' })) + } + }) + try { + const { repairTerminalHistoryPermissions: patchedRepair } = + await import('./terminal-history-permission-repair') + await expect(patchedRepair(base)).resolves.toBe(true) + } finally { + vi.doUnmock('node:fs/promises') + vi.resetModules() + } + + expect(existsSync(join(base, REPAIR_MARKER_NAME))).toBe(true) + }) + }) + + describe('hosts where POSIX modes do not apply', () => { + it('skips the repair sweep on win32 rather than touching the tree', async () => { + const base = isolatedDir() + const restore = stubPlatform('win32') + try { + await expect(repairTerminalHistoryPermissions(base)).resolves.toBe(false) + } finally { + restore() + } + expect(existsSync(join(base, REPAIR_MARKER_NAME))).toBe(false) + }) + + it('still writes history when the platform reports win32', async () => { + const base = isolatedDir() + const restore = stubPlatform('win32') + const mgr = new HistoryManager(base) + try { + await mgr.openSession('win-sess', { cwd: 'C:\\tmp', cols: 80, rows: 24 }) + await mgr.checkpoint('win-sess', makeSnapshot()) + } finally { + await mgr.dispose() + restore() + } + + expect(readFileSync(sessionPath(base, 'win-sess', 'checkpoint.json'), 'utf-8')).toContain( + 'secret scrollback' + ) + }) + + onPosix('still writes history when chmod itself throws', async () => { + const base = isolatedDir() + vi.resetModules() + vi.doMock('node:fs', async () => { + const actual = await vi.importActual('node:fs') + const chmodSyncThrows = (): never => { + throw Object.assign(new Error('EPERM: operation not permitted'), { code: 'EPERM' }) + } + return { ...actual, default: actual, chmodSync: chmodSyncThrows } + }) + try { + const { HistoryManager: PatchedHistoryManager } = await import('./history-manager') + const mgr = new PatchedHistoryManager(base) + await mgr.openSession('chmodless', { cwd: '/tmp', cols: 80, rows: 24 }) + await mgr.checkpoint('chmodless', makeSnapshot()) + await mgr.dispose() + } finally { + vi.doUnmock('node:fs') + vi.resetModules() + } + + expect(readFileSync(sessionPath(base, 'chmodless', 'checkpoint.json'), 'utf-8')).toContain( + 'secret scrollback' + ) + }) + }) +}) diff --git a/src/main/daemon/terminal-history-recovery-freezes.ts b/src/main/daemon/terminal-history-recovery-freezes.ts new file mode 100644 index 00000000000..1ec1eff432d --- /dev/null +++ b/src/main/daemon/terminal-history-recovery-freezes.ts @@ -0,0 +1,48 @@ +import { join } from 'node:path' +import { getHistorySessionDirName } from './history-paths' +import { + markTerminalHistorySessionRecoveryFrozen, + unmarkTerminalHistorySessionRecoveryFrozen, + type ActiveHistoryRecoveryFreeze +} from './terminal-history-recovery-quarantine' + +/** The recovery freezes one HistoryManager holds, each paired with the process-wide hold that keeps + * the backlog permission sweep off a tree whose fingerprint has already been taken. Paired here so + * the in-memory freeze and that hold cannot drift apart across the manager's many release paths. */ +export class TerminalHistoryRecoveryFreezes { + private readonly bySessionId = new Map() + + constructor(private readonly basePath: string) {} + + get(sessionId: string): ActiveHistoryRecoveryFreeze | undefined { + return this.bySessionId.get(sessionId) + } + + has(sessionId: string): boolean { + return this.bySessionId.has(sessionId) + } + + hold(sessionId: string, freeze: ActiveHistoryRecoveryFreeze): void { + this.bySessionId.set(sessionId, freeze) + // Why before the caller's first await: the sweep must see the hold before the freeze reads the + // fingerprint it later re-checks, or a chmod in between silently disables the session's writer. + markTerminalHistorySessionRecoveryFrozen(this.sessionDir(sessionId)) + } + + release(sessionId: string): void { + if (this.bySessionId.delete(sessionId)) { + unmarkTerminalHistorySessionRecoveryFrozen(this.sessionDir(sessionId)) + } + } + + /** Why: an outstanding hold would keep the sweep off that tree for the rest of the process. */ + releaseAll(): void { + for (const sessionId of this.bySessionId.keys()) { + this.release(sessionId) + } + } + + private sessionDir(sessionId: string): string { + return join(this.basePath, getHistorySessionDirName(sessionId)) + } +} diff --git a/src/main/daemon/terminal-history-recovery-quarantine.ts b/src/main/daemon/terminal-history-recovery-quarantine.ts index 414fe4f58e5..ba60b951c52 100644 --- a/src/main/daemon/terminal-history-recovery-quarantine.ts +++ b/src/main/daemon/terminal-history-recovery-quarantine.ts @@ -1,14 +1,7 @@ import { createHash, randomUUID } from 'node:crypto' -import { - existsSync, - lstatSync, - mkdirSync, - readdirSync, - renameSync, - unlinkSync, - writeFileSync -} from 'node:fs' -import { join } from 'node:path' +import { existsSync, lstatSync, readdirSync, renameSync, unlinkSync, writeFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { ensurePrivateDir, PRIVATE_FILE_MODE } from './daemon-private-file-modes' import { getHistorySessionDirName } from './history-paths' const QUARANTINE_DIR_NAME = '.recovery-quarantine' @@ -35,6 +28,39 @@ export function getTerminalHistoryQuarantineOwnerDir(basePath: string, sessionId return join(basePath, QUARANTINE_DIR_NAME, sessionHash) } +// Why process-wide and not a HistoryManager field: the freeze lives in this process's memory while +// the backlog permission sweep walks the same tree from an unrelated module, and its chmod moves the +// `mode`/`ctimeMs` that fingerprintTerminalHistorySession hashes. Refcounted because the legacy and +// current daemon adapters each hold their own HistoryManager over one base path. +const recoveryFrozenSessionDirs = new Map() + +export function markTerminalHistorySessionRecoveryFrozen(sessionDir: string): void { + const key = resolve(sessionDir) + recoveryFrozenSessionDirs.set(key, (recoveryFrozenSessionDirs.get(key) ?? 0) + 1) +} + +export function unmarkTerminalHistorySessionRecoveryFrozen(sessionDir: string): void { + const key = resolve(sessionDir) + const held = recoveryFrozenSessionDirs.get(key) + if (held === undefined) { + return + } + if (held > 1) { + recoveryFrozenSessionDirs.set(key, held - 1) + } else { + recoveryFrozenSessionDirs.delete(key) + } +} + +/** True while a session tree must not be touched by anything outside its own recovery handshake: + * an open freeze holds a fingerprint of it, or a failed quarantine left it fail-closed on disk. */ +export function isTerminalHistorySessionDirRecoveryProtected(sessionDir: string): boolean { + return ( + recoveryFrozenSessionDirs.has(resolve(sessionDir)) || + existsSync(join(sessionDir, RECOVERY_PROTECTION_MARKER)) + ) +} + export function hasTerminalHistoryRecoveryProtection(basePath: string, sessionId: string): boolean { return existsSync(join(basePath, getHistorySessionDirName(sessionId), RECOVERY_PROTECTION_MARKER)) } @@ -83,8 +109,8 @@ export function quarantineTerminalHistorySession( const sessionDir = join(basePath, getHistorySessionDirName(sessionId)) const ownerDir = getTerminalHistoryQuarantineOwnerDir(basePath, sessionId) // Why: if rename is blocked, a later adapter must not attach a writer to the unreadable generation. - writeFileSync(join(sessionDir, RECOVERY_PROTECTION_MARKER), '') - mkdirSync(ownerDir, { recursive: true }) + writeFileSync(join(sessionDir, RECOVERY_PROTECTION_MARKER), '', { mode: PRIVATE_FILE_MODE }) + ensurePrivateDir(ownerDir) const quarantineDir = join(ownerDir, randomUUID()) renameSync(sessionDir, quarantineDir) return quarantineDir diff --git a/src/main/daemon/terminal-history-session-files.ts b/src/main/daemon/terminal-history-session-files.ts new file mode 100644 index 00000000000..fb11b02c11f --- /dev/null +++ b/src/main/daemon/terminal-history-session-files.ts @@ -0,0 +1,36 @@ +// The files one terminal-history session tree owns, and the whole-tree operations over them. +// Single list so the stale-file reset and the permission tightening cannot drift apart. + +import { unlinkSync } from 'node:fs' +import { join } from 'node:path' +import { PRIVATE_DIR_MODE, PRIVATE_FILE_MODE, tightenPathMode } from './daemon-private-file-modes' + +export const TERMINAL_HISTORY_SESSION_FILE_NAMES = [ + 'checkpoint.json', + 'output.log', + 'meta.json', + 'scrollback.bin' +] as const + +// meta.json survives: a reset re-anchors replayable state, not the session's identity. +const REPLAYABLE_SESSION_FILE_NAMES = ['checkpoint.json', 'scrollback.bin', 'output.log'] as const + +/** Why: a crash before the first checkpoint must not replay a cleanly ended prior session. */ +export function clearReplayableTerminalHistorySessionFiles(dir: string): void { + for (const name of REPLAYABLE_SESSION_FILE_NAMES) { + try { + unlinkSync(join(dir, name)) + } catch { + // ENOENT is expected for new sessions. + } + } +} + +/** Idempotent and ~5 syscalls: tighten one session tree as it is opened for writing. Needed because + * `mode` on writeFile only applies at creation, so files an older daemon left at umask stay open. */ +export function tightenTerminalHistorySessionDirMode(dir: string): void { + tightenPathMode(dir, PRIVATE_DIR_MODE) + for (const name of TERMINAL_HISTORY_SESSION_FILE_NAMES) { + tightenPathMode(join(dir, name), PRIVATE_FILE_MODE) + } +} diff --git a/src/main/daemon/terminal-history-session-tombstone.ts b/src/main/daemon/terminal-history-session-tombstone.ts index d72f9e8f798..d557bdac03c 100644 --- a/src/main/daemon/terminal-history-session-tombstone.ts +++ b/src/main/daemon/terminal-history-session-tombstone.ts @@ -3,9 +3,10 @@ // so the stop-and-wait path is metadata-only, and drain the queue off the critical path. import { randomUUID } from 'node:crypto' -import { existsSync, mkdirSync, readdirSync, renameSync } from 'node:fs' +import { existsSync, readdirSync, renameSync } from 'node:fs' import { join } from 'node:path' import { removeHostTree } from '../host-tree-removal' +import { ensurePrivateDir } from './daemon-private-file-modes' import { getHistorySessionDirName } from './history-paths' import { getTerminalHistoryQuarantineOwnerDir } from './terminal-history-recovery-quarantine' @@ -29,7 +30,7 @@ function getPendingDeleteRoot(basePath: string): string { function tombstoneSessionTree(basePath: string, dir: string): string | null { const pendingRoot = getPendingDeleteRoot(basePath) try { - mkdirSync(pendingRoot, { recursive: true }) + ensurePrivateDir(pendingRoot) const tombstone = join(pendingRoot, randomUUID()) renameSync(dir, tombstone) return tombstone diff --git a/src/main/daemon/terminal-history-session-writer.ts b/src/main/daemon/terminal-history-session-writer.ts index 3abe8f7b485..a562f35e1ec 100644 --- a/src/main/daemon/terminal-history-session-writer.ts +++ b/src/main/daemon/terminal-history-session-writer.ts @@ -18,6 +18,8 @@ import { clearTerminalHistoryRecoveryProtection } from './terminal-history-recov import type { PendingOutputRecord, TerminalSnapshot } from './types' import { TERMINAL_HISTORY_CHECKPOINT_MAX_BYTES } from './terminal-history-file-limits' import { serializeTerminalCheckpointWithinLimit } from './terminal-checkpoint-serializer' +import { PRIVATE_FILE_MODE, tightenPathMode } from './daemon-private-file-modes' +import { tightenTerminalHistorySessionDirMode } from './terminal-history-session-files' // Why 5MB: bounds cold-restore replay time and per-session disk; hitting the cap triggers one checkpoint that resets the log. const LOG_MAX_BYTES = 5 * 1024 * 1024 @@ -37,6 +39,8 @@ export class TerminalHistorySessionWriter { this.logPath = join(dir, 'output.log') this.logGeneration = fresh ? 0 : null this.logBytes = fresh ? 0 : null + // Why here: a warm attach reuses files an older daemon created at umask, which `mode` cannot fix. + tightenTerminalHistorySessionDirMode(dir) } async appendIncrements( @@ -50,10 +54,12 @@ export class TerminalHistorySessionWriter { return 'needs-checkpoint' } if (this.logBytes === 0) { - await fsPromises.writeFile(this.logPath, encodeLogHeader(this.logGeneration ?? 0)) + await fsPromises.writeFile(this.logPath, encodeLogHeader(this.logGeneration ?? 0), { + mode: PRIVATE_FILE_MODE + }) this.logBytes = LOG_HEADER_BYTES } - await fsPromises.appendFile(this.logPath, batch) + await fsPromises.appendFile(this.logPath, batch, { mode: PRIVATE_FILE_MODE }) this.logBytes = (this.logBytes ?? LOG_HEADER_BYTES) + batch.length return 'ok' } @@ -87,9 +93,14 @@ export class TerminalHistorySessionWriter { } } const tmpPath = `${this.checkpointPath}.tmp` - await fsPromises.writeFile(tmpPath, data) + // Mode on the tmp file, not after the rename: the checkpoint is never briefly world-readable. + await fsPromises.writeFile(tmpPath, data, { mode: PRIVATE_FILE_MODE }) + // A tmp left behind by a pre-fix crash is reused in place, where `mode` no longer applies. + tightenPathMode(tmpPath, PRIVATE_FILE_MODE) await fsPromises.rename(tmpPath, this.checkpointPath) - await fsPromises.writeFile(this.logPath, encodeLogHeader(generation)) + await fsPromises.writeFile(this.logPath, encodeLogHeader(generation), { + mode: PRIVATE_FILE_MODE + }) this.logGeneration = generation this.logBytes = LOG_HEADER_BYTES clearTerminalHistoryRecoveryProtection(this.dir) diff --git a/src/main/git/command-runner/git-exec-options.ts b/src/main/git/command-runner/git-exec-options.ts index 4390d84658d..75ced0d3030 100644 --- a/src/main/git/command-runner/git-exec-options.ts +++ b/src/main/git/command-runner/git-exec-options.ts @@ -1,7 +1,10 @@ // Why: cap execFile output to prevent an uncatchable V8 string overflow; match relay MAX_GIT_BUFFER. export const DEFAULT_GIT_MAX_BUFFER = 10 * 1024 * 1024 -export type GitAdmissionTier = 'interactive' | 'status' | 'background' +// Why: the admission tier is a wire value, so it is declared with its params schema. +import type { GitAdmissionTier } from '../../../shared/rpc-contract/git-admission-tier-params' + +export type { GitAdmissionTier } export type GitExecOptions = { cwd: string diff --git a/src/main/global-fetch-call-site-audit.test.ts b/src/main/global-fetch-call-site-audit.test.ts index 023f7323b1a..39bd157fc1d 100644 --- a/src/main/global-fetch-call-site-audit.test.ts +++ b/src/main/global-fetch-call-site-audit.test.ts @@ -23,6 +23,7 @@ const AUDITED_GLOBAL_FETCH_LINES = new Map([ ['main/orca-profiles/profile-cloud-client.ts', 1], ['main/orca-profiles/profile-cloud-org-members-client.ts', 1], ['main/rate-limits/codex-fetcher.ts', 3], + ['main/runtime/push/push-gateway-client.ts', 1], ['main/runtime/relay/relay-http-client.ts', 2], ['main/runtime/relay/relay-region-catalog-fetch.ts', 1], ['main/runtime/relay/relay-region-preference.ts', 2], diff --git a/src/main/host/electron-runtime-desktop-surface.ts b/src/main/host/electron-runtime-desktop-surface.ts index f709804955c..056db03176a 100644 --- a/src/main/host/electron-runtime-desktop-surface.ts +++ b/src/main/host/electron-runtime-desktop-surface.ts @@ -1,8 +1,10 @@ -import { BrowserWindow, ipcMain, Notification } from 'electron' +import { BrowserWindow, ipcMain, Notification, powerMonitor } from 'electron' +import { readDesktopAwayState } from '../notifications/desktop-away-state' import type { RuntimeDesktopSurface } from '../runtime/runtime-desktop-surface' /** The desktop implementation of the runtime's optional desktop facilities. */ export const electronRuntimeDesktopSurface: RuntimeDesktopSurface = { + isAwayForMobileNotifications: () => readDesktopAwayState(powerMonitor), showNotification: ({ title, body }) => { if (!Notification.isSupported()) { return false diff --git a/src/main/ipc/agent-hooks.test.ts b/src/main/ipc/agent-hooks.test.ts index 411a6084124..001380e934d 100644 --- a/src/main/ipc/agent-hooks.test.ts +++ b/src/main/ipc/agent-hooks.test.ts @@ -152,6 +152,37 @@ describe('agentStatus:getSnapshot IPC', () => { expect(handler!({})).toEqual(snapshot) }) + // The half-migration seam: until PR 2 retires the renderer's own feed bridge, main must not + // publish structured rows to the renderer at all — one pane key, one writer. + it('omits structured rows the renderer feed bridge still owns', async () => { + getStatusSnapshot.mockReturnValue([ + { + paneKey: PANE_KEY, + state: 'done', + prompt: 'hook row', + agentType: 'claude', + connectionId: null, + receivedAt: 1_700_000_000_000, + stateStartedAt: 1_699_999_999_000 + }, + { + paneKey: CHILD_PANE_KEY, + state: 'working', + prompt: 'native chat row', + agentType: 'codex', + connectionId: null, + structuredHost: 'owned', + receivedAt: 1_700_000_001_000, + stateStartedAt: 1_700_000_000_500 + } + ]) + const { registerAgentHookHandlers } = await import('./agent-hooks') + registerAgentHookHandlers() + + const rows = handleHandlers.get('agentStatus:getSnapshot')!({}) as { paneKey: string }[] + expect(rows.map((row) => row.paneKey)).toEqual([PANE_KEY]) + }) + it('enriches the hook cache snapshot with runtime lineage metadata', async () => { const snapshot = [ { diff --git a/src/main/ipc/agent-hooks.ts b/src/main/ipc/agent-hooks.ts index ee858e3cca8..f460be06b5f 100644 --- a/src/main/ipc/agent-hooks.ts +++ b/src/main/ipc/agent-hooks.ts @@ -49,9 +49,13 @@ export function registerAgentHookHandlers( // Why: the renderer pulls this after workspace hydration, so startup cannot // lose replayed statuses while its local store is still empty. Match the // live push enrichment in main/index.ts so parent/child rows survive replay. - return agentHookServer - .getStatusSnapshot() - .map((entry) => enrichAgentStatusIpcPayload(entry, runtime)) + return ( + agentHookServer + .getStatusSnapshot() + // Same rule as the live push: the renderer's feed bridge owns structured rows for now. + .filter((entry) => entry.structuredHost === undefined) + .map((entry) => enrichAgentStatusIpcPayload(entry, runtime)) + ) }) ipcMain.handle('agentStatus:inferInterrupt', (_event, request: unknown): boolean => { if (typeof request !== 'object' || request === null) { diff --git a/src/main/ipc/notification-burst-cooldown.ts b/src/main/ipc/notification-burst-cooldown.ts index e7616c57746..91e879a7e47 100644 --- a/src/main/ipc/notification-burst-cooldown.ts +++ b/src/main/ipc/notification-burst-cooldown.ts @@ -1,37 +1 @@ -const NOTIFICATION_COOLDOWN_MS = 5000 -const MAX_RECENT_NOTIFICATION_KEYS = 50 - -function pruneRecentNotifications(recentNotifications: Map, now: number): void { - if (recentNotifications.size <= MAX_RECENT_NOTIFICATION_KEYS) { - return - } - - for (const [key, ts] of recentNotifications) { - if (now - ts >= NOTIFICATION_COOLDOWN_MS) { - recentNotifications.delete(key) - } - } - - while (recentNotifications.size > MAX_RECENT_NOTIFICATION_KEYS) { - const oldest = recentNotifications.keys().next() - if (oldest.done) { - break - } - recentNotifications.delete(oldest.value) - } -} - -export function reserveNotificationCooldown( - recentNotifications: Map, - dedupeKey: string, - now: number -): boolean { - const lastSentAt = recentNotifications.get(dedupeKey) ?? 0 - if (now - lastSentAt < NOTIFICATION_COOLDOWN_MS) { - return false - } - recentNotifications.delete(dedupeKey) - recentNotifications.set(dedupeKey, now) - pruneRecentNotifications(recentNotifications, now) - return true -} +export { reserveNotificationCooldown } from '../../shared/notification-burst-cooldown' diff --git a/src/main/ipc/notification-options.ts b/src/main/ipc/notification-options.ts index a2553f05a3c..deb0b93fd2f 100644 --- a/src/main/ipc/notification-options.ts +++ b/src/main/ipc/notification-options.ts @@ -1,3 +1,4 @@ +import { translateMain } from '../i18n/main-i18n' import type { NotificationDispatchRequest } from '../../shared/notification-settings-types' const NOTIFICATION_AGENT_LABEL_MAX_LENGTH = 40 @@ -57,12 +58,7 @@ function buildAgentTaskCompleteNotificationOptions( const agentLabel = formatNotificationAgentLabel(args.agentType) const worktreeContext = formatNotificationWorktreeContext(args) - const statusText = - args.agentState === 'blocked' || args.agentState === 'waiting' - ? 'needs input' - : args.agentState === 'done' && args.agentInterrupted - ? 'stopped' - : 'finished' + const statusText = formatAgentNotificationStatusText(args) return { title: `${worktreeContext} - ${agentLabel} ${statusText}`, @@ -70,6 +66,21 @@ function buildAgentTaskCompleteNotificationOptions( } } +// Why (#4375): a still-working agent must never be announced as finished. Only an +// explicit terminal state, or no state at all (the hook snapshot expired and the +// notification itself is the completion signal), may say "finished". +function formatAgentNotificationStatusText(args: NotificationDispatchRequest): string { + if (args.agentState === 'blocked' || args.agentState === 'waiting') { + return translateMain('notifications.agentStatus.needsInput', 'needs input') + } + if (args.agentState === 'working') { + return translateMain('notifications.agentStatus.working', 'working') + } + return args.agentState === 'done' && args.agentInterrupted + ? translateMain('notifications.agentStatus.stopped', 'stopped') + : translateMain('notifications.agentStatus.finished', 'finished') +} + function formatNotificationWorktreeContext(args: NotificationDispatchRequest): string { const worktreeLabel = normalizeNotificationText( args.worktreeLabel, diff --git a/src/main/ipc/notifications-message-formatting.test.ts b/src/main/ipc/notifications-message-formatting.test.ts index 4fcbc3e0b64..677c3131203 100644 --- a/src/main/ipc/notifications-message-formatting.test.ts +++ b/src/main/ipc/notifications-message-formatting.test.ts @@ -278,6 +278,73 @@ describe('registerNotificationHandlers', () => { expect(options.body.length).toBeLessThanOrEqual(180) }) + it.each([ + { agentState: 'working', expected: 'feat/notis - Claude working' }, + { agentState: 'blocked', expected: 'feat/notis - Claude needs input' }, + { agentState: 'waiting', expected: 'feat/notis - Claude needs input' }, + { agentState: 'done', expected: 'feat/notis - Claude finished' }, + { agentState: undefined, expected: 'feat/notis - Claude finished' } + ])('titles agentState $agentState without claiming a false finish', async (scenario) => { + registerNotificationHandlers({ + getSettings: () => ({ + notifications: { + enabled: true, + agentTaskComplete: true, + terminalBell: false, + suppressWhenFocused: true + } + }) + } as never) + + const handler = getDispatchHandler() + await handler( + {}, + { + source: 'agent-task-complete', + worktreeLabel: 'feat/notis', + agentType: 'claude', + ...(scenario.agentState ? { agentState: scenario.agentState } : {}), + agentLastAssistantMessage: 'Ran the suite.' + } + ) + + expect(notificationCtorMock).toHaveBeenCalledWith( + expectedNativeNotificationOptions({ title: scenario.expected, body: 'Ran the suite.' }) + ) + }) + + it('reports an interrupted finish as stopped', async () => { + registerNotificationHandlers({ + getSettings: () => ({ + notifications: { + enabled: true, + agentTaskComplete: true, + terminalBell: false, + suppressWhenFocused: true + } + }) + } as never) + + const handler = getDispatchHandler() + await handler( + {}, + { + source: 'agent-task-complete', + worktreeLabel: 'feat/notis', + agentType: 'claude', + agentState: 'done', + agentInterrupted: true + } + ) + + expect(notificationCtorMock).toHaveBeenCalledWith( + expectedNativeNotificationOptions({ + title: 'feat/notis - Claude stopped', + body: 'Claude stopped.' + }) + ) + }) + it('uses tool context before falling back when no prompt or assistant preview exists', async () => { registerNotificationHandlers({ getSettings: () => ({ @@ -308,7 +375,7 @@ describe('registerNotificationHandlers', () => { expect(notificationCtorMock).toHaveBeenCalledWith( expectedNativeNotificationOptions({ - title: 'feat/notis - Agent finished', + title: 'feat/notis - Agent working', body: 'Using Bash: pnpm test' }) ) diff --git a/src/main/ipc/notifications-mobile-fanout.test.ts b/src/main/ipc/notifications-mobile-fanout.test.ts index 94d2535a3cc..ab797293042 100644 --- a/src/main/ipc/notifications-mobile-fanout.test.ts +++ b/src/main/ipc/notifications-mobile-fanout.test.ts @@ -71,15 +71,17 @@ describe('registerNotificationHandlers', () => { expect(dispatchMobileNotification).toHaveBeenCalledWith({ type: 'notification', + emittedAt: expect.any(Number), source: 'agent-task-complete', title: 'feat/notis - Hermes finished', body: 'The diff updates notification formatting.', - worktreeId: 'repo::wt1' + worktreeId: 'repo::wt1', + agentState: 'done' }) expect(notificationCtorMock).not.toHaveBeenCalled() }) - it('does not dispatch mobile notifications when notifications are disabled', async () => { + it('offers disabled desktop events to independently configured phones', async () => { const dispatchMobileNotification = vi.fn() registerNotificationHandlers( { @@ -101,10 +103,12 @@ describe('registerNotificationHandlers', () => { reason: 'disabled' }) - expect(dispatchMobileNotification).not.toHaveBeenCalled() + expect(dispatchMobileNotification).toHaveBeenCalledWith( + expect.objectContaining({ desktopAllowed: false }) + ) }) - it('does not dispatch mobile notifications when the source is disabled', async () => { + it('marks a disabled desktop source for phones following desktop settings', async () => { const dispatchMobileNotification = vi.fn() registerNotificationHandlers( { @@ -126,7 +130,9 @@ describe('registerNotificationHandlers', () => { reason: 'source-disabled' }) - expect(dispatchMobileNotification).not.toHaveBeenCalled() + expect(dispatchMobileNotification).toHaveBeenCalledWith( + expect.objectContaining({ desktopAllowed: false }) + ) }) it('dispatches one mobile notification when the active worktree is focused on desktop', async () => { @@ -173,7 +179,7 @@ describe('registerNotificationHandlers', () => { expect(notificationCtorMock).not.toHaveBeenCalled() }) - it('does not dispatch mobile notifications for cooldown-suppressed bursts', async () => { + it('preserves different mobile event categories before per-phone burst suppression', async () => { const dispatchMobileNotification = vi.fn() registerNotificationHandlers( { @@ -198,7 +204,7 @@ describe('registerNotificationHandlers', () => { reason: 'cooldown' }) - expect(dispatchMobileNotification).toHaveBeenCalledTimes(1) + expect(dispatchMobileNotification).toHaveBeenCalledTimes(2) expect(dispatchMobileNotification).toHaveBeenCalledWith( expect.objectContaining({ source: 'agent-task-complete', worktreeId: 'repo::wt1' }) ) diff --git a/src/main/ipc/notifications.ts b/src/main/ipc/notifications.ts index 28f6bfd95e5..274ab2719d8 100644 --- a/src/main/ipc/notifications.ts +++ b/src/main/ipc/notifications.ts @@ -1,4 +1,5 @@ -import { BrowserWindow, Notification, ipcMain } from 'electron' +import { BrowserWindow, Notification, ipcMain, powerMonitor } from 'electron' +import { readDesktopAwayState } from '../notifications/desktop-away-state' import type { Store } from '../persistence' import type { NotificationDeliveryProbeResult, @@ -26,6 +27,8 @@ import { } from './notification-permission-probe' export function registerNotificationHandlers(store: Store, runtime?: OrcaRuntimeService): void { + ipcMain.removeHandler('notifications:getDesktopAwayState') + ipcMain.handle('notifications:getDesktopAwayState', () => readDesktopAwayState(powerMonitor)) const recentDesktopNotifications = new Map() const recentMobileNotifications = new Map() resetNotificationPermissionEvidence() @@ -119,34 +122,43 @@ export function registerNotificationHandlers(store: Store, runtime?: OrcaRuntime } const settings = store.getSettings().notifications - if (!settings.enabled) { - return { delivered: false, reason: 'disabled' } - } - - if ( - (args.source === 'agent-task-complete' && !settings.agentTaskComplete) || - (args.source === 'terminal-bell' && !settings.terminalBell) - ) { - return { delivered: false, reason: 'source-disabled' } - } + const desktopAllowed = + settings.enabled && + (args.source !== 'agent-task-complete' || settings.agentTaskComplete) && + (args.source !== 'terminal-bell' || settings.terminalBell) const notificationOptions = buildNotificationOptions(args) // Why: desktop focus only means this computer sees the worktree; the paired phone may still need the alert. if (runtime && args.source !== 'test') { const dedupeKey = args.worktreeId ?? args.worktreeLabel ?? 'global' - if (reserveNotificationCooldown(recentMobileNotifications, dedupeKey, Date.now())) { + if ( + reserveNotificationCooldown( + recentMobileNotifications, + JSON.stringify([desktopAllowed, args.source, args.agentState, dedupeKey]), + Date.now() + ) + ) { runtime.dispatchMobileNotification({ type: 'notification', + emittedAt: Date.now(), source: args.source, + ...(!desktopAllowed ? { desktopAllowed: false } : {}), title: notificationOptions.title, body: notificationOptions.body, worktreeId: args.worktreeId, - ...(args.notificationId ? { notificationId: args.notificationId } : {}) + ...(args.notificationId ? { notificationId: args.notificationId } : {}), + // Why: background push needs the agent's real state to pick "needs input" + // vs "finished" — and to stay silent while the agent is still working. + ...(args.agentState ? { agentState: args.agentState } : {}) }) } } + if (!desktopAllowed) { + return { delivered: false, reason: settings.enabled ? 'source-disabled' : 'disabled' } + } + const browserWindow = BrowserWindow.getAllWindows().find((window) => !window.isDestroyed()) ?? null if ( diff --git a/src/main/ipc/runtime-environment-capability-evidence.test.ts b/src/main/ipc/runtime-environment-capability-evidence.test.ts index 4671326ef02..8c9bce5f272 100644 --- a/src/main/ipc/runtime-environment-capability-evidence.test.ts +++ b/src/main/ipc/runtime-environment-capability-evidence.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it } from 'vitest' import type { PairingOffer } from '../../shared/pairing' import { advanceRuntimeEnvironmentCapabilityIncarnation, @@ -17,7 +17,6 @@ describe('runtime environment capability evidence', () => { it('accepts evidence by dispatch order instead of completion order', () => { const older = captureRuntimeEnvironmentCapabilityEvidence('env', pairing()) const newer = captureRuntimeEnvironmentCapabilityEvidence('env', pairing()) - const pause = vi.fn() expect( applyRuntimeEnvironmentCapabilityVerdict({ @@ -30,12 +29,10 @@ describe('runtime environment capability evidence', () => { applyRuntimeEnvironmentCapabilityVerdict({ evidence: older, verdict: 'absent', - runtimeId: 'runtime-old', - onAbsent: pause + runtimeId: 'runtime-old' }) ).toBe(false) - expect(pause).not.toHaveBeenCalled() expect(isRuntimeEnvironmentCapabilityPaused('env')).toBe(false) }) diff --git a/src/main/ipc/runtime-environment-capability-evidence.ts b/src/main/ipc/runtime-environment-capability-evidence.ts index d32bda584e9..197ec71f4f8 100644 --- a/src/main/ipc/runtime-environment-capability-evidence.ts +++ b/src/main/ipc/runtime-environment-capability-evidence.ts @@ -68,8 +68,6 @@ export function applyRuntimeEnvironmentCapabilityVerdict(args: { evidence: RuntimeEnvironmentCapabilityEvidence verdict: RuntimeEnvironmentCapabilityVerdict runtimeId: string - onCapable?: () => void - onAbsent?: () => void }): boolean { const state = stateFor(args.evidence.environmentId) if ( @@ -83,11 +81,6 @@ export function applyRuntimeEnvironmentCapabilityVerdict(args: { verdict: args.verdict, runtimeId: args.runtimeId } - if (args.verdict === 'capable') { - args.onCapable?.() - } else { - args.onAbsent?.() - } return true } diff --git a/src/main/ipc/runtime-environment-connectivity-handlers.ts b/src/main/ipc/runtime-environment-connectivity-handlers.ts index 1e267d8675a..bfbc63847c4 100644 --- a/src/main/ipc/runtime-environment-connectivity-handlers.ts +++ b/src/main/ipc/runtime-environment-connectivity-handlers.ts @@ -20,6 +20,8 @@ import { verifyAndAddRuntimeEnvironmentFromPairingCode } from './runtime-environ import { clearRuntimeEnvironmentCapabilityEvidence } from './runtime-environment-capability-evidence' import { closeRemoteRuntimeRequestConnection, + getRuntimeEnvironmentStatusOwner, + getRuntimeEnvironmentStatusSnapshots, retryRemoteRuntimeSharedControlConnectionNow } from './runtime-environment-request-connections' import { @@ -29,7 +31,6 @@ import { } from './runtime-environment-manual-disconnect' import { callRuntimeEnvironment, - clearSharedControlSupport, getRuntimeEnvironmentStatus } from './runtime-environment-transport-routing' @@ -60,6 +61,9 @@ export function registerRuntimeEnvironmentConnectivityHandlers({ getUserDataPath, invalidateTransport }: ConnectivityHandlerOptions): void { + ipcMain.handle('runtimeEnvironments:getStatusSnapshots', () => + getRuntimeEnvironmentStatusSnapshots() + ) ipcMain.handle('runtimeEnvironments:list', () => listEnvironments(getUserDataPath()).map(redactRuntimeEnvironment) ) @@ -80,6 +84,12 @@ export function registerRuntimeEnvironmentConnectivityHandlers({ const result = await verifyAndAddRuntimeEnvironmentFromPairingCode(getUserDataPath(), args) if (result.ok) { clearRuntimeEnvironmentManualDisconnect(result.environment.id) + getRuntimeEnvironmentStatusOwner(getUserDataPath(), result.environment.id).acceptVerified({ + id: 'status.get', + ok: true, + result: result.runtimeStatus, + _meta: { runtimeId: result.runtimeStatus.runtimeId } + }) } return result } @@ -121,6 +131,8 @@ export function registerRuntimeEnvironmentConnectivityHandlers({ markRuntimeEnvironmentManuallyDisconnected(environment.id) invalidateTransport(environment.id) closeLegacySelectorTransport(args.selector, environment.id) + // Retain disconnected evidence for renderers that missed the teardown event. + getRuntimeEnvironmentStatusOwner(getUserDataPath(), environment.id) return { disconnected: redactRuntimeEnvironment(environment) } } ) @@ -132,7 +144,9 @@ export function registerRuntimeEnvironmentConnectivityHandlers({ ): Promise> => { const environment = resolveEnvironment(getUserDataPath(), args.selector) clearRuntimeEnvironmentManualDisconnect(environment.id) - return getRuntimeEnvironmentStatus(getUserDataPath(), environment.id, args.timeoutMs) + return getRuntimeEnvironmentStatus(getUserDataPath(), environment.id, args.timeoutMs, { + reconnect: true + }) } ) ipcMain.handle( @@ -156,7 +170,6 @@ function closeLegacySelectorTransport(selector: string, environmentId: string): return } closeRemoteRuntimeRequestConnection(selector) - clearSharedControlSupport(selector) } function registerPassiveStatusHandler(getUserDataPath: () => string): void { diff --git a/src/main/ipc/runtime-environment-federated-read-routing.test.ts b/src/main/ipc/runtime-environment-federated-read-routing.test.ts index c58a404fd39..51c77510a01 100644 --- a/src/main/ipc/runtime-environment-federated-read-routing.test.ts +++ b/src/main/ipc/runtime-environment-federated-read-routing.test.ts @@ -1,3 +1,5 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' +vi.mock('electron', () => ({ BrowserWindow: { getAllWindows: () => [] } })) import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -20,14 +22,17 @@ vi.mock('../../shared/remote-runtime-client', () => ({ sendRemoteRuntimeRequest: sendRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: vi.fn(), - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - reconnectRemoteRuntimeSharedControlConnection: vi.fn(), - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn() -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: vi.fn(), + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + reconnectRemoteRuntimeSharedControlConnection: vi.fn(), + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn() + }) +}) import { callRuntimeEnvironment, @@ -55,6 +60,7 @@ describe('federated read RPC transport routing', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) diff --git a/src/main/ipc/runtime-environment-handler-channels.ts b/src/main/ipc/runtime-environment-handler-channels.ts index 0b63dea943a..23b40fe5183 100644 --- a/src/main/ipc/runtime-environment-handler-channels.ts +++ b/src/main/ipc/runtime-environment-handler-channels.ts @@ -9,6 +9,7 @@ export const RUNTIME_ENVIRONMENT_HANDLER_CHANNELS = [ 'runtimeEnvironments:retryControlConnection', 'runtimeEnvironments:prepareBrowserClientHostPlacement', 'runtimeEnvironments:getStatus', + 'runtimeEnvironments:getStatusSnapshots', 'runtimeEnvironments:call', 'runtimeEnvironments:subscribe', 'runtimeEnvironments:unsubscribe' diff --git a/src/main/ipc/runtime-environment-request-connections.test.ts b/src/main/ipc/runtime-environment-request-connections.test.ts index 750d1becc6a..b02d1b06fb5 100644 --- a/src/main/ipc/runtime-environment-request-connections.test.ts +++ b/src/main/ipc/runtime-environment-request-connections.test.ts @@ -47,9 +47,9 @@ describe('runtime environment shared-control connection cache', () => { applyRuntimeEnvironmentCapabilityVerdict({ evidence: absent, verdict: 'absent', - runtimeId: 'runtime-test', - onAbsent: () => pauseRemoteRuntimeSharedControlRetry(ENVIRONMENT_ID) + runtimeId: 'runtime-test' }) + pauseRemoteRuntimeSharedControlRetry(ENVIRONMENT_ID) expect(getRemoteRuntimeSharedControlDiagnostics(ENVIRONMENT_ID)?.state).toBe('closed') await delay(400) expect(server.connectionCount()).toBe(1) @@ -58,12 +58,10 @@ describe('runtime environment shared-control connection cache', () => { applyRuntimeEnvironmentCapabilityVerdict({ evidence: capable, verdict: 'capable', - runtimeId: 'runtime-test', - onCapable: () => { - ensureRemoteRuntimeSharedControlConnection(ENVIRONMENT_ID, server.pairing) - reconnectRemoteRuntimeSharedControlConnection(ENVIRONMENT_ID) - } + runtimeId: 'runtime-test' }) + ensureRemoteRuntimeSharedControlConnection(ENVIRONMENT_ID, server.pairing) + reconnectRemoteRuntimeSharedControlConnection(ENVIRONMENT_ID) await waitFor(() => server.connectionCount() === 2) }) @@ -119,9 +117,9 @@ describe('runtime environment shared-control connection cache', () => { applyRuntimeEnvironmentCapabilityVerdict({ evidence, verdict: 'absent', - runtimeId: 'runtime-test', - onAbsent: () => pauseRemoteRuntimeSharedControlRetry(ENVIRONMENT_ID) + runtimeId: 'runtime-test' }) + pauseRemoteRuntimeSharedControlRetry(ENVIRONMENT_ID) expect(getRemoteRuntimeSharedControlDiagnostics(ENVIRONMENT_ID)?.state).toBe('reconnecting') await waitFor(() => server.connectionCount() === 2) diff --git a/src/main/ipc/runtime-environment-request-connections.ts b/src/main/ipc/runtime-environment-request-connections.ts index c1f855697e5..6ba9f273c1e 100644 --- a/src/main/ipc/runtime-environment-request-connections.ts +++ b/src/main/ipc/runtime-environment-request-connections.ts @@ -1,4 +1,9 @@ import type { PairingOffer } from '../../shared/pairing' +import { resolveEnvironment } from '../../shared/runtime-environment-store' +import { getPreferredPairingOffer } from '../../shared/runtime-environments' +import type { RuntimeHostStatusOwner } from '../../shared/runtime-host-status-owner' +import type { RuntimeStatus } from '../../shared/runtime-types' +import { createRuntimeEnvironmentStatusOwner } from './runtime-environment-status-owner' import { ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES } from '../../shared/protocol-version' import type { RuntimeOrchestrationEnvelope, @@ -30,6 +35,56 @@ type CachedSharedControlConnection = { const requestConnections = new Map() const sharedControlConnections = new Map() +const statusOwners = new Map() + +export function getRuntimeEnvironmentStatusOwner( + userDataPath: string, + selector: string +): RuntimeHostStatusOwner { + const environment = resolveEnvironment(userDataPath, selector) + const pairing = getPreferredPairingOffer(environment) + const key = `${userDataPath}\0${environment.pairingRevision ?? environment.createdAt}\0${getPairingKey(pairing)}` + let cached = statusOwners.get(environment.id) + if (!cached || cached.key !== key || cached.owner.read().retired) { + if (cached) { + closeRemoteRuntimeRequestConnection(environment.id) + } + const owner = createRuntimeEnvironmentStatusOwner(userDataPath, environment, { + isReady: () => getRemoteRuntimeSharedControlDiagnostics(environment.id)?.state === 'ready', + request: (signal) => + sendRemoteRuntimeSharedControlRequest( + environment.id, + pairing, + 'status.get', + undefined, + 15_000, + undefined, + signal + ), + establish: () => { + ensureRemoteRuntimeSharedControlConnection(environment.id, pairing) + reconnectRemoteRuntimeSharedControlConnection(environment.id) + }, + pause: () => pauseRemoteRuntimeSharedControlRetry(environment.id) + }) + cached = { key, owner } + statusOwners.set(environment.id, cached) + if (isRuntimeEnvironmentManuallyDisconnected(environment.id)) { + owner.dispose() + } + } + return cached.owner +} + +export function resetRuntimeEnvironmentStatusOwners(): void { + for (const id of statusOwners.keys()) { + closeRemoteRuntimeRequestConnection(id) + } +} + +export function getRuntimeEnvironmentStatusSnapshots() { + return [...statusOwners.values()].map(({ owner }) => owner.read()) +} export function sendRemoteRuntimeConnectionRequest( environmentId: string, @@ -56,6 +111,9 @@ export function sendRemoteRuntimeConnectionRequest( } export function closeRemoteRuntimeRequestConnection(environmentId: string): void { + const status = statusOwners.get(environmentId) + statusOwners.delete(environmentId) + status?.owner.dispose() const cached = requestConnections.get(environmentId) requestConnections.delete(environmentId) cached?.connection.close() @@ -166,6 +224,16 @@ function getSharedControlConnection( transportGeneration, diagnostics }) + statusOwners + .get(environmentId) + ?.owner.connectionChanged( + diagnostics.state === 'ready' + ? 'ready' + : diagnostics.state === 'closed' || diagnostics.state === 'reconnecting' + ? 'disconnected' + : 'connecting', + diagnostics + ) } }) } diff --git a/src/main/ipc/runtime-environment-shared-control-support.ts b/src/main/ipc/runtime-environment-shared-control-support.ts index 29513e2970a..0de6a35e0f6 100644 --- a/src/main/ipc/runtime-environment-shared-control-support.ts +++ b/src/main/ipc/runtime-environment-shared-control-support.ts @@ -1,39 +1,23 @@ -import { - ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES, - REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY -} from '../../shared/protocol-version' -import { sendRemoteRuntimeRequest } from '../../shared/remote-runtime-client' -import { markEnvironmentUsed } from '../../shared/runtime-environment-store' import type { getPreferredPairingOffer, KnownRuntimeEnvironment } from '../../shared/runtime-environments' -import type { RuntimeStatus } from '../../shared/runtime-types' +import { RemoteRuntimeClientError } from '../../shared/remote-runtime-client-error' import { - applyRuntimeEnvironmentCapabilityVerdict, - captureRuntimeEnvironmentCapabilityEvidence, getAcceptedRuntimeEnvironmentCapabilityOutcome, - isRuntimeEnvironmentCapabilityOutcomeCurrent, - runtimeEnvironmentCapabilityOutcome, resetRuntimeEnvironmentCapabilityEvidence, type RuntimeEnvironmentCapabilityOutcome } from './runtime-environment-capability-evidence' -import { pauseRemoteRuntimeSharedControlRetry } from './runtime-environment-request-connections' - -const sharedControlSupport = new Map< - string, - { cacheKey: string; check: Promise } ->() +import { + getRuntimeEnvironmentStatusOwner, + resetRuntimeEnvironmentStatusOwners +} from './runtime-environment-request-connections' export function resetSharedControlSupport(): void { - sharedControlSupport.clear() + resetRuntimeEnvironmentStatusOwners() resetRuntimeEnvironmentCapabilityEvidence() } -export function clearSharedControlSupport(environmentId: string): void { - sharedControlSupport.delete(environmentId) -} - export async function supportsSharedControl( userDataPath: string, environment: KnownRuntimeEnvironment, @@ -48,85 +32,17 @@ export async function supportsSharedControl( if (accepted) { return accepted } - const cacheKey = getSharedControlSupportCacheKey(environment, pairing) - const cached = sharedControlSupport.get(environment.id) - if (cached?.cacheKey === cacheKey) { - const outcome = await cached.check - if (isRuntimeEnvironmentCapabilityOutcomeCurrent(outcome)) { - return outcome - } - if (sharedControlSupport.get(environment.id)?.check === cached.check) { - sharedControlSupport.delete(environment.id) - } - return { kind: 'stale_incarnation' } + const response = await getRuntimeEnvironmentStatusOwner(userDataPath, environment.id).refresh({ + timeoutMs + }) + if (!response.ok) { + throw new RemoteRuntimeClientError(response.error.code, response.error.message) } - let resolvedCacheKey = cacheKey - const evidence = captureRuntimeEnvironmentCapabilityEvidence(environment.id, pairing) - const check = (async () => { - const response = await sendRemoteRuntimeRequest( + return ( + getAcceptedRuntimeEnvironmentCapabilityOutcome( + environment.id, pairing, - 'status.get', - undefined, - timeoutMs, - undefined, - undefined, - ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES - ) - if (response.ok === true) { - const verdict = response.result.capabilities?.includes( - REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY - ) - ? 'capable' - : 'absent' - const acceptedEvidence = applyRuntimeEnvironmentCapabilityVerdict({ - evidence, - verdict, - runtimeId: response._meta.runtimeId, - onAbsent: () => pauseRemoteRuntimeSharedControlRetry(environment.id) - }) - if (!acceptedEvidence) { - return { kind: 'stale_incarnation' } as const - } - markEnvironmentUsed(userDataPath, environment.id, { runtimeId: response._meta.runtimeId }) - resolvedCacheKey = getSharedControlSupportCacheKey( - environment, - pairing, - response._meta.runtimeId - ) - return runtimeEnvironmentCapabilityOutcome(evidence, verdict, response._meta.runtimeId) - } - return runtimeEnvironmentCapabilityOutcome( - evidence, - 'absent', - environment.runtimeId ?? 'unknown-runtime' - ) - })() - // Why: support belongs to the saved pairing/runtime identity, not its mutable display name. - sharedControlSupport.set(environment.id, { cacheKey, check }) - try { - const outcome = await check - const cachedAfterCheck = sharedControlSupport.get(environment.id) - if (cachedAfterCheck?.check === check && cachedAfterCheck.cacheKey !== resolvedCacheKey) { - sharedControlSupport.set(environment.id, { cacheKey: resolvedCacheKey, check }) - } - return outcome - } catch (error) { - if (sharedControlSupport.get(environment.id)?.check === check) { - sharedControlSupport.delete(environment.id) - } - throw error - } -} - -function getSharedControlSupportCacheKey( - environment: KnownRuntimeEnvironment, - pairing: ReturnType, - runtimeId = environment.runtimeId -): string { - return [ - runtimeId ?? 'unknown-runtime', - pairing.endpoint, - pairing.deviceToken, - pairing.publicKeyB64 - ].join('\0') + response._meta.runtimeId + ) ?? { kind: 'stale_incarnation' } + ) } diff --git a/src/main/ipc/runtime-environment-status-connection.test.ts b/src/main/ipc/runtime-environment-status-connection.test.ts new file mode 100644 index 00000000000..8d4fad6f9b9 --- /dev/null +++ b/src/main/ipc/runtime-environment-status-connection.test.ts @@ -0,0 +1,65 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, expect, it, vi } from 'vitest' +import { encodePairingOffer } from '../../shared/pairing' +import { addEnvironmentFromPairingCode } from '../../shared/runtime-environment-store' +import { REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY } from '../../shared/protocol-version' +import { + createSharedControlTestServer, + closeSharedControlTestServers +} from '../../shared/remote-runtime-shared-control-test-server' +import { getRuntimeEnvironmentStatus } from './runtime-environment-transport-routing' +import { + getRuntimeEnvironmentStatusOwner, + resetRuntimeEnvironmentStatusOwners +} from './runtime-environment-request-connections' + +vi.mock('electron', () => ({ BrowserWindow: { getAllWindows: () => [] } })) +const profiles: string[] = [] +afterEach(async () => { + resetRuntimeEnvironmentStatusOwners() + await closeSharedControlTestServers() + profiles.splice(0).forEach((profile) => rmSync(profile, { recursive: true, force: true })) +}) + +it('publishes real same-socket verification after every authenticated reconnect', async () => { + let runtimeId = 'host-before' + const server = await createSharedControlTestServer({ + resultForRequest: () => ({ + runtimeId, + capabilities: [REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY] + }) + }) + const profile = mkdtempSync(join(tmpdir(), 'orca-status-socket-')) + profiles.push(profile) + const environment = addEnvironmentFromPairingCode(profile, { + name: 'host', + pairingCode: encodePairingOffer(server.pairing) + }) + await getRuntimeEnvironmentStatus(profile, environment.id) + const owner = getRuntimeEnvironmentStatusOwner(profile, environment.id) + await vi.waitFor( + () => { + expect(owner.read()).toMatchObject({ transport: 'ready', verification: 'verified' }) + expect(server.requests).toHaveLength(2) + }, + { timeout: 3_000 } + ) + expect(server.connectionCount()).toBe(2) // Bootstrap plus persistent control. + runtimeId = 'host-after' + server.closeClients() + await vi.waitFor( + () => { + expect(owner.read().status?.runtimeId).toBe('host-after') + expect(owner.read().verification).toBe('verified') + }, + { timeout: 3_000 } + ) + expect(server.connectionCount()).toBe(3) + expect(server.requests.map((request) => request.method)).toEqual([ + 'status.get', + 'status.get', + 'status.get' + ]) +}) diff --git a/src/main/ipc/runtime-environment-status-owner.ts b/src/main/ipc/runtime-environment-status-owner.ts new file mode 100644 index 00000000000..4ac3c067f74 --- /dev/null +++ b/src/main/ipc/runtime-environment-status-owner.ts @@ -0,0 +1,89 @@ +import { BrowserWindow } from 'electron' +import { sendRemoteRuntimeRequest } from '../../shared/remote-runtime-client' +import { + ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES, + REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY +} from '../../shared/protocol-version' +import { + getPreferredPairingOffer, + type KnownRuntimeEnvironment +} from '../../shared/runtime-environments' +import { markEnvironmentUsed } from '../../shared/runtime-environment-store' +import { RuntimeHostStatusOwner } from '../../shared/runtime-host-status-owner' +import { + RUNTIME_HOST_STATUS_CHANNEL, + type RuntimeHostStatusResponse +} from '../../shared/runtime-host-status' +import { + applyRuntimeEnvironmentCapabilityVerdict, + getAcceptedRuntimeEnvironmentCapabilityOutcome, + captureRuntimeEnvironmentCapabilityEvidence +} from './runtime-environment-capability-evidence' +import { isRuntimeEnvironmentManuallyDisconnected } from './runtime-environment-manual-disconnect' + +export function createRuntimeEnvironmentStatusOwner( + userDataPath: string, + environment: KnownRuntimeEnvironment, + transport: { + isReady: () => boolean + request: (signal: AbortSignal) => Promise + establish: () => void + pause: () => void + } +): RuntimeHostStatusOwner { + const pairing = getPreferredPairingOffer(environment) + let evidence = captureRuntimeEnvironmentCapabilityEvidence(environment.id, pairing) + return new RuntimeHostStatusOwner({ + environmentId: environment.id, + pairingRevision: environment.pairingRevision ?? environment.createdAt, + request: (signal) => { + evidence = captureRuntimeEnvironmentCapabilityEvidence(environment.id, pairing) + return transport.isReady() && + getAcceptedRuntimeEnvironmentCapabilityOutcome(environment.id, pairing, null)?.kind === + 'supported' + ? transport.request(signal) + : sendRemoteRuntimeRequest( + pairing, + 'status.get', + undefined, + 15_000, + undefined, + signal, + ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES + ) + }, + verified: (response, active) => { + const capable = + response.result.capabilities?.includes(REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY) ?? false + const accepted = applyRuntimeEnvironmentCapabilityVerdict({ + evidence, + verdict: capable ? 'capable' : 'absent', + runtimeId: response._meta.runtimeId + }) + if (accepted && active && !isRuntimeEnvironmentManuallyDisconnected(environment.id)) { + markEnvironmentUsed(userDataPath, environment.id, { + runtimeId: response._meta.runtimeId, + pairedDeviceId: response.result.pairedDeviceId + }) + if (capable) { + transport.establish() + } else { + transport.pause() + } + } + return capable && active + }, + publish: (snapshot) => { + for (const window of BrowserWindow.getAllWindows()) { + if (window.isDestroyed()) { + continue + } + try { + window.webContents.send(RUNTIME_HOST_STATUS_CHANNEL, snapshot) + } catch { + /* A renderer can close during publication. */ + } + } + } + }) +} diff --git a/src/main/ipc/runtime-environment-status-recovery.test.ts b/src/main/ipc/runtime-environment-status-recovery.test.ts new file mode 100644 index 00000000000..82e94ea62e0 --- /dev/null +++ b/src/main/ipc/runtime-environment-status-recovery.test.ts @@ -0,0 +1,89 @@ +import { REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY } from '../../shared/protocol-version' +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { addEnvironmentFromPairingCode } from '../../shared/runtime-environment-store' +import { pairingCode } from './runtime-environments-ipc-test-harness' +import { + getRuntimeEnvironmentStatus, + resetSharedControlSupport +} from './runtime-environment-transport-routing' + +const { request, publish } = vi.hoisted(() => ({ request: vi.fn(), publish: vi.fn() })) +vi.mock('../../shared/remote-runtime-client', () => ({ + sendRemoteRuntimeRequest: request, + subscribeRemoteRuntimeRequest: vi.fn() +})) +vi.mock('electron', () => ({ + BrowserWindow: { + getAllWindows: () => [ + { + isDestroyed: () => false, + webContents: { send: publish } + } + ] + } +})) + +let profile: string +beforeEach(() => { + vi.useFakeTimers() + request.mockReset() + publish.mockReset() + profile = mkdtempSync(join(tmpdir(), 'orca-status-recovery-')) +}) +afterEach(() => { + resetSharedControlSupport() + vi.useRealTimers() + rmSync(profile, { recursive: true, force: true }) +}) + +it('recovers a saved host after its first status check fails, without another UI request', async () => { + const environment = addEnvironmentFromPairingCode(profile, { + name: 'offline-at-startup', + pairingCode: pairingCode() + }) + request + .mockRejectedValueOnce( + Object.assign(new Error('host offline'), { code: 'runtime_unavailable' }) + ) + .mockResolvedValue({ + id: 'status', + ok: true, + result: { runtimeId: 'host-1', graphStatus: 'ready', capabilities: [] }, + _meta: { runtimeId: 'host-1' } + }) + expect((await getRuntimeEnvironmentStatus(profile, environment.id)).ok).toBe(false) + await vi.advanceTimersByTimeAsync(3_000) + expect(request).toHaveBeenCalledTimes(2) + expect(publish).toHaveBeenCalledWith( + 'runtimeEnvironments:statusChanged', + expect.objectContaining({ + environmentId: environment.id, + verification: 'verified', + status: expect.objectContaining({ runtimeId: 'host-1' }) + }) + ) + await vi.advanceTimersByTimeAsync(300_000) + expect(request).toHaveBeenCalledTimes(2) +}) + +it('a passive capability check does not strand later active bootstrap recovery', async () => { + const environment = addEnvironmentFromPairingCode(profile, { + name: 'passive-first', + pairingCode: pairingCode() + }) + request + .mockResolvedValueOnce({ + id: 'status', + ok: true, + result: { runtimeId: 'host-1', capabilities: [REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY] }, + _meta: { runtimeId: 'host-1' } + }) + .mockRejectedValue(new Error('host offline')) + await getRuntimeEnvironmentStatus(profile, environment.id, undefined, { observeOnly: true }) + await getRuntimeEnvironmentStatus(profile, environment.id) + await vi.advanceTimersByTimeAsync(3_000) + expect(request).toHaveBeenCalledTimes(3) +}) diff --git a/src/main/ipc/runtime-environment-support-routing.test.ts b/src/main/ipc/runtime-environment-support-routing.test.ts index feb09ee481a..8fdde15a81c 100644 --- a/src/main/ipc/runtime-environment-support-routing.test.ts +++ b/src/main/ipc/runtime-environment-support-routing.test.ts @@ -57,7 +57,6 @@ describe('runtime environment support routing', () => { ).resolves.toMatchObject({ ok: true }) expect(supportsMock).toHaveBeenCalledTimes(2) - expect(clearSupportMock).toHaveBeenCalledOnce() expect(supported).toHaveBeenCalledOnce() expect(unsupported).not.toHaveBeenCalled() }) diff --git a/src/main/ipc/runtime-environment-support-routing.ts b/src/main/ipc/runtime-environment-support-routing.ts index e2503ad4445..9566b2fc1b7 100644 --- a/src/main/ipc/runtime-environment-support-routing.ts +++ b/src/main/ipc/runtime-environment-support-routing.ts @@ -18,10 +18,7 @@ import { type RuntimeEnvironmentCapabilityOutcome } from './runtime-environment-capability-evidence' import { runtimeEnvironmentRevisionFailure } from './runtime-environment-revision-guard' -import { - clearSharedControlSupport, - supportsSharedControl -} from './runtime-environment-shared-control-support' +import { supportsSharedControl } from './runtime-environment-shared-control-support' import { sendRemoteRuntimeRequestAbortable, sendRemoteRuntimeSharedControlRequestAbortable @@ -205,7 +202,6 @@ export async function routeRuntimeEnvironmentCallBySupport(args: { } return response } - clearSharedControlSupport(environment.id) environment = resolveEnvironment(args.userDataPath, environment.id) } return runtimeEnvironmentChangedFailure(environment, args.method) diff --git a/src/main/ipc/runtime-environment-transport-routing-tailscale-hint.test.ts b/src/main/ipc/runtime-environment-transport-routing-tailscale-hint.test.ts index a44f5df84c6..bb2bbdac322 100644 --- a/src/main/ipc/runtime-environment-transport-routing-tailscale-hint.test.ts +++ b/src/main/ipc/runtime-environment-transport-routing-tailscale-hint.test.ts @@ -1,20 +1,23 @@ import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { generateKeyPair, publicKeyToBase64 } from '../../shared/e2ee-crypto' import { encodePairingOffer, type PairingOffer } from '../../shared/pairing' import { addEnvironmentFromPairingCode } from '../../shared/runtime-environment-store' import { callRuntimeEnvironment, getRuntimeEnvironmentStatus, - subscribeRuntimeEnvironment + subscribeRuntimeEnvironment, + resetSharedControlSupport } from './runtime-environment-transport-routing' // Why: prove the wiring, not just the helper — an unreachable endpoint exercises // the real WebSocket failure → reject → Tailscale-hint join points the settings // probe (returned ok:false) and in-use calls (thrown) actually use. +vi.mock('electron', () => ({ BrowserWindow: { getAllWindows: () => [] } })) + let userDataPath: string function seedEnvironment(name: string, endpoint: string): string { @@ -39,6 +42,7 @@ beforeEach(() => { }) afterEach(() => { + resetSharedControlSupport() rmSync(userDataPath, { recursive: true, force: true }) }) diff --git a/src/main/ipc/runtime-environment-transport-routing.ts b/src/main/ipc/runtime-environment-transport-routing.ts index 70f49816603..b19b0d9e376 100644 --- a/src/main/ipc/runtime-environment-transport-routing.ts +++ b/src/main/ipc/runtime-environment-transport-routing.ts @@ -1,8 +1,5 @@ import { getPreferredPairingOffer } from '../../shared/runtime-environments' -import { - ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES, - REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY -} from '../../shared/protocol-version' +import { ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES } from '../../shared/protocol-version' import { resolveEnvironment, markEnvironmentUsed } from '../../shared/runtime-environment-store' import { isOrchestrationMutation } from '../../shared/orchestration-rpc-contract' import type { @@ -11,33 +8,22 @@ import type { } from '../../shared/runtime-rpc-envelope' import type { RuntimeStatus } from '../../shared/runtime-types' import { - sendRemoteRuntimeRequest, subscribeRemoteRuntimeRequest, type RemoteRuntimeSubscription } from '../../shared/remote-runtime-client' import { withRemoteRuntimeTailscaleHint } from '../../shared/remote-runtime-tailscale-hint' import { enqueueRuntimeCall } from './runtime-environment-call-queue' -import { - ensureRemoteRuntimeSharedControlConnection, - pauseRemoteRuntimeSharedControlRetry, - reconnectRemoteRuntimeSharedControlConnection -} from './runtime-environment-request-connections' +import { getRuntimeEnvironmentStatusOwner } from './runtime-environment-request-connections' import { sendRemoteRuntimeConnectionRequestAbortable, sendRemoteRuntimeRequestAbortable } from './runtime-environment-abortable-requests' import { attachRemoteControlDiagnostics } from './runtime-environment-status-diagnostics' -import { - applyRuntimeEnvironmentCapabilityVerdict, - captureRuntimeEnvironmentCapabilityEvidence -} from './runtime-environment-capability-evidence' + import { isRuntimeEnvironmentManuallyDisconnected } from './runtime-environment-manual-disconnect' import { runtimeEnvironmentRevisionFailure } from './runtime-environment-revision-guard' import { withTailscaleHintForResponse } from './runtime-environment-tailscale-response' -import { - clearSharedControlSupport, - resetSharedControlSupport -} from './runtime-environment-shared-control-support' +import { resetSharedControlSupport } from './runtime-environment-shared-control-support' import { executeSupportRoutedCall, shouldRouteCallBySupport, @@ -47,72 +33,31 @@ import { const DEFAULT_REMOTE_RUNTIME_TIMEOUT_MS = 15_000 -export { clearSharedControlSupport, resetSharedControlSupport } +export { resetSharedControlSupport } export async function getRuntimeEnvironmentStatus( userDataPath: string, selector: string, timeoutMs?: number, - options?: { observeOnly?: true } + options?: { observeOnly?: true; signal?: AbortSignal; reconnect?: true } ): Promise> { const environment = resolveEnvironment(userDataPath, selector) - const pairing = getPreferredPairingOffer(environment) - const evidence = captureRuntimeEnvironmentCapabilityEvidence(environment.id, pairing) - let response: RuntimeRpcResponse - try { - response = await sendRemoteRuntimeRequest( - pairing, - 'status.get', - undefined, - timeoutMs ?? DEFAULT_REMOTE_RUNTIME_TIMEOUT_MS, - undefined, - undefined, - ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES - ) - } catch (error) { - // Why: the status UI needs shared-control diagnostics most when the - // fresh status probe failed and the host is reconnecting/offline. - return attachRemoteControlDiagnostics( - withTailscaleHintForResponse( - { - id: 'status.get', - ok: false, - error: { - code: 'runtime_unavailable', - message: error instanceof Error ? error.message : String(error) - }, - _meta: { runtimeId: environment.runtimeId } - }, - pairing.endpoint - ), - environment.id - ) - } - if (response.ok === true) { - const verdict = response.result.capabilities?.includes(REMOTE_RUNTIME_SHARED_CONTROL_CAPABILITY) - ? 'capable' - : 'absent' - const accepted = applyRuntimeEnvironmentCapabilityVerdict({ - evidence, - verdict, - runtimeId: response._meta.runtimeId, - onCapable: () => { - if (!options?.observeOnly && !isRuntimeEnvironmentManuallyDisconnected(environment.id)) { - ensureRemoteRuntimeSharedControlConnection(environment.id, pairing) - reconnectRemoteRuntimeSharedControlConnection(environment.id) - } - }, - onAbsent: () => pauseRemoteRuntimeSharedControlRetry(environment.id) - }) - if (accepted && !options?.observeOnly) { - markEnvironmentUsed(userDataPath, environment.id, { - runtimeId: response._meta.runtimeId, - pairedDeviceId: response.result.pairedDeviceId - }) + if (isRuntimeEnvironmentManuallyDisconnected(environment.id)) { + return { + id: 'status.get', + ok: false, + error: { + code: 'runtime_manually_disconnected', + message: 'Runtime environment is manually disconnected.' + } } } + const response = await getRuntimeEnvironmentStatusOwner(userDataPath, environment.id).refresh({ + timeoutMs, + ...options + }) return attachRemoteControlDiagnostics( - withTailscaleHintForResponse(response, pairing.endpoint), + withTailscaleHintForResponse(response, getPreferredPairingOffer(environment).endpoint), environment.id ) } @@ -127,6 +72,15 @@ export async function callRuntimeEnvironment( envelope?: RuntimeOrchestrationEnvelope, options?: { signal?: AbortSignal } ): Promise> { + if (method === 'status.get') { + const environment = resolveEnvironment(userDataPath, selector) + const failure = runtimeEnvironmentRevisionFailure( + environment, + expectedEnvironmentPairingRevision, + method + ) + return failure ?? getRuntimeEnvironmentStatus(userDataPath, selector, timeoutMs, options) + } const environment = resolveEnvironment(userDataPath, selector) // Why: connection failures reject (they don't resolve as ok:false), so the // Tailscale hint is applied to the thrown error here — wrapping the resolved diff --git a/src/main/ipc/runtime-environments-call-routing.test.ts b/src/main/ipc/runtime-environments-call-routing.test.ts index ef92dc66826..e6f8946527d 100644 --- a/src/main/ipc/runtime-environments-call-routing.test.ts +++ b/src/main/ipc/runtime-environments-call-routing.test.ts @@ -1,3 +1,4 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -44,6 +45,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -58,18 +60,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn(), - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn(), + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) import { registerRuntimeEnvironmentHandlers } from './runtime-environments' import { channelHandlerLookup, pairingCode } from './runtime-environments-ipc-test-harness' @@ -112,6 +119,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) @@ -339,7 +347,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { undefined, 15_000, undefined, - undefined, + expect.any(AbortSignal), ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES ) expect(sendRemoteRuntimeSharedControlRequestMock).toHaveBeenCalledWith( @@ -451,7 +459,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { } ) - it('keeps uncoded call failures on the rejected IPC fallback path', async () => { + it('returns uncoded status failures through the owner response', async () => { registerRuntimeEnvironmentHandlers(store as never) sendRemoteRuntimeRequestMock.mockRejectedValue(new Error('shared down')) @@ -464,9 +472,10 @@ describe('registerRuntimeEnvironmentHandlers', () => { 'runtimeEnvironments:call' ) - await expect(call(null, { selector: 'desk', method: 'status.get' })).rejects.toThrow( - 'shared down' - ) + await expect(call(null, { selector: 'desk', method: 'status.get' })).resolves.toMatchObject({ + ok: false, + error: { code: 'runtime_unavailable', message: 'shared down' } + }) }) it('does not fall back after a shared-control request fails on a supported runtime', async () => { diff --git a/src/main/ipc/runtime-environments-capability-cache.test.ts b/src/main/ipc/runtime-environments-capability-cache.test.ts index 8ac11d69dd1..32f724df981 100644 --- a/src/main/ipc/runtime-environments-capability-cache.test.ts +++ b/src/main/ipc/runtime-environments-capability-cache.test.ts @@ -1,3 +1,4 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -37,6 +38,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -51,18 +53,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn(), - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn(), + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) import { registerRuntimeEnvironmentHandlers } from './runtime-environments' import { channelHandlerLookup, pairingCode } from './runtime-environments-ipc-test-harness' @@ -105,6 +112,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) @@ -182,9 +190,10 @@ describe('registerRuntimeEnvironmentHandlers', () => { { selector: string; method: string; params?: unknown; timeoutMs?: number }, { ok: true; result: unknown } >('runtimeEnvironments:call') - await expect(call(null, { selector: 'desk', method: 'repo.list' })).rejects.toThrow( - 'probe failed' - ) + await expect(call(null, { selector: 'desk', method: 'repo.list' })).resolves.toMatchObject({ + ok: false, + error: { code: 'runtime_unavailable', message: 'probe failed' } + }) await expect(call(null, { selector: 'desk', method: 'repo.list' })).resolves.toMatchObject({ ok: true, result: { repos: [] } diff --git a/src/main/ipc/runtime-environments-ipc-test-harness.ts b/src/main/ipc/runtime-environments-ipc-test-harness.ts index 016352793cb..e97e45ede1a 100644 --- a/src/main/ipc/runtime-environments-ipc-test-harness.ts +++ b/src/main/ipc/runtime-environments-ipc-test-harness.ts @@ -1,6 +1,62 @@ import { expect } from 'vitest' import type { Mock } from 'vitest' +import { getPreferredPairingOffer } from '../../shared/runtime-environments' import { encodePairingOffer } from '../../shared/pairing' +import { resolveEnvironment } from '../../shared/runtime-environment-store' +import { createRuntimeEnvironmentStatusOwner } from './runtime-environment-status-owner' +import type { RuntimeHostStatusOwner } from '../../shared/runtime-host-status-owner' +import { isRuntimeEnvironmentManuallyDisconnected } from './runtime-environment-manual-disconnect' + +/** Keep IPC tests on the production owner while replacing only its transport. */ +export function withRuntimeStatusOwners>(transport: T) { + const owners = new Map() + return { + ...transport, + getRuntimeEnvironmentStatusOwner: (profile: string, selector: string) => { + const environment = resolveEnvironment(profile, selector) + let owner = owners.get(environment.id) + if (!owner || owner.read().retired) { + owner = createRuntimeEnvironmentStatusOwner(profile, environment, { + isReady: () => + transport.getRemoteRuntimeSharedControlDiagnostics?.(environment.id)?.state === 'ready', + request: (signal) => + transport.sendRemoteRuntimeSharedControlRequest( + environment.id, + undefined, + 'status.get', + undefined, + 15_000, + undefined, + signal + ), + establish: () => { + transport.ensureRemoteRuntimeSharedControlConnection?.( + environment.id, + getPreferredPairingOffer(environment) + ) + transport.reconnectRemoteRuntimeSharedControlConnection?.(environment.id) + }, + pause: () => transport.pauseRemoteRuntimeSharedControlRetry?.(environment.id) + }) + owners.set(environment.id, owner) + if (isRuntimeEnvironmentManuallyDisconnected(environment.id)) { + owner.dispose() + } + } + return owner + }, + getRuntimeEnvironmentStatusSnapshots: () => [...owners.values()].map((owner) => owner.read()), + resetRuntimeEnvironmentStatusOwners: () => { + owners.forEach((owner) => owner.dispose()) + owners.clear() + }, + closeRemoteRuntimeRequestConnection: (...args: unknown[]) => { + owners.get(args[0] as string)?.dispose() + owners.delete(args[0] as string) + transport.closeRemoteRuntimeRequestConnection(...args) + } + } +} export function pairingCode(endpoint = 'ws://127.0.0.1:6768'): string { return encodePairingOffer({ diff --git a/src/main/ipc/runtime-environments-pairing.test.ts b/src/main/ipc/runtime-environments-pairing.test.ts index 87d6c2698ab..ce492a7a740 100644 --- a/src/main/ipc/runtime-environments-pairing.test.ts +++ b/src/main/ipc/runtime-environments-pairing.test.ts @@ -1,3 +1,5 @@ +import type { RuntimeHostStatusSnapshot } from '../../shared/runtime-host-status' +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -44,6 +46,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -58,18 +61,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: retryRemoteRuntimeSharedControlConnectionNowMock, - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn(), - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: retryRemoteRuntimeSharedControlConnectionNowMock, + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn(), + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) import { registerRuntimeEnvironmentHandlers } from './runtime-environments' import { channelHandlerLookup, pairingCode } from './runtime-environments-ipc-test-harness' @@ -125,6 +133,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) @@ -132,6 +141,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { registerRuntimeEnvironmentHandlers(store as never) expect(handleMock.mock.calls.map((call) => call[0])).toEqual([ + 'runtimeEnvironments:getStatusSnapshots', 'runtimeEnvironments:list', 'runtimeEnvironments:addFromPairingCode', 'runtimeEnvironments:verifyAndAddFromPairingCode', @@ -166,6 +176,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { 'runtimeEnvironments:retryControlConnection', 'runtimeEnvironments:prepareBrowserClientHostPlacement', 'runtimeEnvironments:getStatus', + 'runtimeEnvironments:getStatusSnapshots', 'runtimeEnvironments:call', 'runtimeEnvironments:subscribe', 'runtimeEnvironments:unsubscribe', @@ -467,6 +478,13 @@ describe('registerRuntimeEnvironmentHandlers', () => { ok: false, error: { code: 'runtime_manually_disconnected' } }) + const getSnapshots = handler( + 'runtimeEnvironments:getStatusSnapshots' + ) + // A new renderer only has the snapshot read, not the earlier disconnect event. + expect(await getSnapshots(null, undefined)).toMatchObject([ + { environmentId: added.environment.id, retired: true, transport: 'disconnected' } + ]) const call = handler< { selector: string; method: string }, { ok: boolean; error?: { code: string } } @@ -492,6 +510,10 @@ describe('registerRuntimeEnvironmentHandlers', () => { result: { runtimeId: 'runtime-remote' } }) expect(sendRemoteRuntimeRequestMock).toHaveBeenCalledOnce() + expect(await getSnapshots(null, undefined)).toMatchObject([ + { environmentId: added.environment.id, verification: 'verified' } + ]) + expect((await getSnapshots(null, undefined))[0].retired).not.toBe(true) }) it('marks environments owned by ephemeral VM runtimes in the public list', async () => { diff --git a/src/main/ipc/runtime-environments-status-diagnostics.test.ts b/src/main/ipc/runtime-environments-status-diagnostics.test.ts index b210e7c209b..9fe3d6baf0e 100644 --- a/src/main/ipc/runtime-environments-status-diagnostics.test.ts +++ b/src/main/ipc/runtime-environments-status-diagnostics.test.ts @@ -1,3 +1,4 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -44,6 +45,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -58,18 +60,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: ensureRemoteRuntimeSharedControlConnectionMock, - pauseRemoteRuntimeSharedControlRetry: pauseRemoteRuntimeSharedControlRetryMock, - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: ensureRemoteRuntimeSharedControlConnectionMock, + pauseRemoteRuntimeSharedControlRetry: pauseRemoteRuntimeSharedControlRetryMock, + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) import { registerRuntimeEnvironmentHandlers } from './runtime-environments' import { channelHandlerLookup, pairingCode } from './runtime-environments-ipc-test-harness' @@ -114,6 +121,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) @@ -148,9 +156,9 @@ describe('registerRuntimeEnvironmentHandlers', () => { expect.objectContaining({ endpoint: 'ws://127.0.0.1:6768', deviceToken: 'device-token' }), 'status.get', undefined, - 50, - undefined, + 15_000, undefined, + expect.any(AbortSignal), ELECTRON_REMOTE_RUNTIME_CLIENT_CAPABILITIES ) expect(reconnectRemoteRuntimeSharedControlConnectionMock).toHaveBeenCalledWith( @@ -319,36 +327,41 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) }) - it('returns shared-control diagnostics when saved remote runtime status throws', async () => { - registerRuntimeEnvironmentHandlers(store as never) - getRemoteRuntimeSharedControlDiagnosticsMock.mockReturnValue({ - state: 'reconnecting', - pendingRequestCount: 0, - subscriptionCount: 1, - reconnectAttempt: 2, - lastConnectedAt: 123, - lastClose: { code: 1006, reason: '' }, - lastError: 'closed' - }) - sendRemoteRuntimeRequestMock.mockRejectedValue(new Error('socket closed')) + it.each(['runtimeEnvironments:getStatus', 'runtimeEnvironments:connect'])( + 'preserves failure diagnostics and guidance on %s', + async (channel) => { + registerRuntimeEnvironmentHandlers(store as never) + getRemoteRuntimeSharedControlDiagnosticsMock.mockReturnValue({ + state: 'reconnecting', + pendingRequestCount: 0, + subscriptionCount: 1, + reconnectAttempt: 2, + lastConnectedAt: 123, + lastClose: { code: 1006, reason: '' }, + lastError: 'closed' + }) + sendRemoteRuntimeRequestMock.mockRejectedValue( + new Error('Could not connect to the remote Orca runtime.') + ) - const add = handler< - { name: string; pairingCode: string }, - { environment: { id: string; name: string } } - >('runtimeEnvironments:addFromPairingCode') - await add(null, { name: 'desk', pairingCode: pairingCode() }) + const add = handler< + { name: string; pairingCode: string }, + { environment: { id: string; name: string } } + >('runtimeEnvironments:addFromPairingCode') + await add(null, { name: 'desk', pairingCode: pairingCode() }) - const getStatus = handler< - { selector: string; timeoutMs?: number }, - { ok: false; error: { message: string; data?: { remoteControl?: { state: string } } } } - >('runtimeEnvironments:getStatus') + const getStatus = handler< + { selector: string; timeoutMs?: number }, + { ok: false; error: { message: string; data?: { remoteControl?: { state: string } } } } + >(channel) - await expect(getStatus(null, { selector: 'desk' })).resolves.toMatchObject({ - ok: false, - error: { - message: 'socket closed', - data: { remoteControl: { state: 'reconnecting' } } - } - }) - }) + await expect(getStatus(null, { selector: 'desk' })).resolves.toMatchObject({ + ok: false, + error: { + message: expect.stringContaining('connect both devices to Tailscale'), + data: { remoteControl: { state: 'reconnecting' } } + } + }) + } + ) }) diff --git a/src/main/ipc/runtime-environments-subscription-lifecycle.test.ts b/src/main/ipc/runtime-environments-subscription-lifecycle.test.ts index 53a07442c3b..ed4dcd62182 100644 --- a/src/main/ipc/runtime-environments-subscription-lifecycle.test.ts +++ b/src/main/ipc/runtime-environments-subscription-lifecycle.test.ts @@ -1,3 +1,4 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -38,6 +39,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -52,18 +54,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn(), - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn(), + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) import { invalidateRuntimeEnvironmentTransport, @@ -109,6 +116,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) diff --git a/src/main/ipc/runtime-environments-subscription-routing.test.ts b/src/main/ipc/runtime-environments-subscription-routing.test.ts index 494f0b9ea6b..0ef70f7ac96 100644 --- a/src/main/ipc/runtime-environments-subscription-routing.test.ts +++ b/src/main/ipc/runtime-environments-subscription-routing.test.ts @@ -1,3 +1,4 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -40,6 +41,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -54,18 +56,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn(), - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn(), + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) import { registerRuntimeEnvironmentHandlers } from './runtime-environments' import { channelHandlerLookup, pairingCode } from './runtime-environments-ipc-test-harness' @@ -108,6 +115,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) diff --git a/src/main/ipc/runtime-environments-subscription-teardown.test.ts b/src/main/ipc/runtime-environments-subscription-teardown.test.ts index 13a98e1057a..afb1adf457a 100644 --- a/src/main/ipc/runtime-environments-subscription-teardown.test.ts +++ b/src/main/ipc/runtime-environments-subscription-teardown.test.ts @@ -1,3 +1,4 @@ +import { resetRuntimeEnvironmentStatusOwners } from './runtime-environment-request-connections' import { mkdtempSync, rmSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -38,6 +39,7 @@ const { })) vi.mock('electron', () => ({ + BrowserWindow: { getAllWindows: () => [] }, app: { getPath: getPathMock }, ipcMain: { handle: handleMock, @@ -52,18 +54,23 @@ vi.mock('../../shared/remote-runtime-client', () => ({ subscribeRemoteRuntimeRequest: subscribeRemoteRuntimeRequestMock })) -vi.mock('./runtime-environment-request-connections', () => ({ - sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, - sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, - subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, - getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, - reconnectRemoteRuntimeSharedControlConnection: reconnectRemoteRuntimeSharedControlConnectionMock, - retryRemoteRuntimeSharedControlConnectionsNow: retryRemoteRuntimeSharedControlConnectionsNowMock, - retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), - ensureRemoteRuntimeSharedControlConnection: vi.fn(), - pauseRemoteRuntimeSharedControlRetry: vi.fn(), - closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock -})) +vi.mock('./runtime-environment-request-connections', async () => { + const { withRuntimeStatusOwners } = await import('./runtime-environments-ipc-test-harness') + return withRuntimeStatusOwners({ + sendRemoteRuntimeConnectionRequest: sendRemoteRuntimeConnectionRequestMock, + sendRemoteRuntimeSharedControlRequest: sendRemoteRuntimeSharedControlRequestMock, + subscribeRemoteRuntimeSharedControlRequest: subscribeRemoteRuntimeSharedControlRequestMock, + getRemoteRuntimeSharedControlDiagnostics: getRemoteRuntimeSharedControlDiagnosticsMock, + reconnectRemoteRuntimeSharedControlConnection: + reconnectRemoteRuntimeSharedControlConnectionMock, + retryRemoteRuntimeSharedControlConnectionsNow: + retryRemoteRuntimeSharedControlConnectionsNowMock, + retryRemoteRuntimeSharedControlConnectionNow: vi.fn(), + ensureRemoteRuntimeSharedControlConnection: vi.fn(), + pauseRemoteRuntimeSharedControlRetry: vi.fn(), + closeRemoteRuntimeRequestConnection: closeRemoteRuntimeRequestConnectionMock + }) +}) vi.mock('../browser/paired-runtime-browser-client-host-runtime', () => ({ retirePairedRuntimeBrowserClientHostEnvironment: retirePairedRuntimeBrowserClientHostEnvironmentMock @@ -115,6 +122,7 @@ describe('registerRuntimeEnvironmentHandlers', () => { }) afterEach(() => { + resetRuntimeEnvironmentStatusOwners() rmSync(userDataPath, { recursive: true, force: true }) }) diff --git a/src/main/ipc/runtime-environments.ts b/src/main/ipc/runtime-environments.ts index 7d9a261eef9..ac7a4107bc8 100644 --- a/src/main/ipc/runtime-environments.ts +++ b/src/main/ipc/runtime-environments.ts @@ -1,6 +1,6 @@ import { app, ipcMain } from 'electron' import { randomUUID } from 'node:crypto' -import { resolveEnvironment } from '../../shared/runtime-environment-store' +import { listEnvironments, resolveEnvironment } from '../../shared/runtime-environment-store' import type { RemoteRuntimeSubscription } from '../../shared/remote-runtime-client' import type { Store } from '../persistence' import { @@ -8,14 +8,16 @@ import { registerRuntimeEnvironmentConnectivityHandlers, registerRuntimeEnvironmentPassiveHandlers } from './runtime-environment-connectivity-handlers' -import { closeRemoteRuntimeRequestConnection } from './runtime-environment-request-connections' +import { + closeRemoteRuntimeRequestConnection, + getRuntimeEnvironmentStatusOwner +} from './runtime-environment-request-connections' import { registerRuntimeEnvironmentRecoveryHandler } from './runtime-environment-recovery-handler' import { advanceRuntimeEnvironmentTransportGeneration, getRuntimeEnvironmentTransportGeneration } from './runtime-environment-transport-generation' import { - clearSharedControlSupport, resetSharedControlSupport, subscribeRuntimeEnvironment } from './runtime-environment-transport-routing' @@ -64,7 +66,6 @@ export function invalidateRuntimeEnvironmentTransport(environmentId: string): Pr advanceRuntimeEnvironmentCapabilityIncarnation(environmentId) advanceRuntimeEnvironmentTransportGeneration(environmentId) closeRemoteRuntimeRequestConnection(environmentId) - clearSharedControlSupport(environmentId) closeSubscriptionsForEnvironment(environmentId) return retirePairedRuntimeBrowserClientHostEnvironment( environmentId, @@ -97,6 +98,11 @@ export function registerRuntimeEnvironmentHandlers(store: Store): void { }) registerRuntimeEnvironmentRecoveryHandler() registerRuntimeEnvironmentPassiveHandlers(getUserDataPath) + for (const environment of listEnvironments(getUserDataPath())) { + if (!isRuntimeEnvironmentManuallyDisconnected(environment.id)) { + getRuntimeEnvironmentStatusOwner(getUserDataPath(), environment.id).activate() + } + } ipcMain.handle( 'runtimeEnvironments:subscribe', async ( diff --git a/src/main/ipc/runtime.ts b/src/main/ipc/runtime.ts index 6237b8d040d..ff55c8dbab2 100644 --- a/src/main/ipc/runtime.ts +++ b/src/main/ipc/runtime.ts @@ -11,7 +11,10 @@ import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' import { TERMINAL_FIT_RESTORE_DEADLINE_MS } from '../../shared/terminal-fit-restore-deadline' import { + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' @@ -82,6 +85,9 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { connectionId: desktopSenders.connectionIdFor(event.sender), clientCapabilities: [ AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY, + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ] @@ -131,6 +137,9 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { connectionId, clientCapabilities: [ AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY, + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, + AGENT_SESSION_TURN_ITEM_CAPABILITY, + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ] diff --git a/src/main/ipc/worktrees/removal/worktree-removal-ownership.ts b/src/main/ipc/worktrees/removal/worktree-removal-ownership.ts index ec2198bcdb0..ecc982076c5 100644 --- a/src/main/ipc/worktrees/removal/worktree-removal-ownership.ts +++ b/src/main/ipc/worktrees/removal/worktree-removal-ownership.ts @@ -40,11 +40,17 @@ export async function stopPtysForDestructiveWorktreeRemoval( ...(allowUnverifiedStop ? { allowUnverifiedStop: true } : {}), ...(connectionId ? { includeLocalRegistry: false } : {}) }) + // Structured sessions are counted here too: closing a user's chat is now an ordinary outcome + // of this verb, and a removal that closed one but no PTY would otherwise log nothing at all. + const structuredStopped = teardownResult.structuredStopped ?? 0 const total = - teardownResult.runtimeStopped + teardownResult.providerStopped + teardownResult.registryStopped + teardownResult.runtimeStopped + + teardownResult.providerStopped + + teardownResult.registryStopped + + structuredStopped if (total > 0) { console.info( - `[worktree-teardown] ${worktreeId} killed runtime=${teardownResult.runtimeStopped} provider=${teardownResult.providerStopped} registry=${teardownResult.registryStopped}` + `[worktree-teardown] ${worktreeId} killed runtime=${teardownResult.runtimeStopped} provider=${teardownResult.providerStopped} registry=${teardownResult.registryStopped} structured=${structuredStopped}` ) } } diff --git a/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts b/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts index a7f54a14a4c..d9ca68de6fa 100644 --- a/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-crash-boundary.test.ts @@ -15,6 +15,11 @@ import type { AgentJournalMessageItem, AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import { hasUnansweredStructuredAgentSessionDispatch } from '../../../shared/structured-agent-session-projection' +import { + DISPATCH_DOUBT_RETRY_IN_PROGRESS, + dispatchDoubtProvesUndelivered +} from './journal-dispatch-doubt-reasons' import { digestPayload } from './journal-payload-bounds' import { reconcileSubmissions, @@ -110,6 +115,8 @@ describe('crash between provider accept and journal commit', () => { expect(restarted.pendingSubmissions().map((entry) => entry.clientMessageId)).toEqual(['cm_1']) await restarted.markPendingSubmissionsUnknown(2) expect(restarted.submissions()[0]?.dispatchState).toBe('unknown') + // Marks the send as outlived by its writer, so no reader reports it as still working. + expect(restarted.submissions()[0]?.recovered).toBe(true) const [outcome] = reconcileSubmissions({ submissions: restarted.submissions(), @@ -139,6 +146,82 @@ describe('crash between provider accept and journal commit', () => { expect(restarted.receiptFor('cm_1')?.providerItemId).toBe(agentJournalItemKey(outcome.identity)) }) + it('retires an ack timeout on restart without changing its delivery verdict', async () => { + const journal = await open() + await journal.appendSubmission({ + clientMessageId: 'cm_timeout', + payloadFingerprint: digestPayload('slow'), + body: userMessage('slow'), + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'cm_timeout', + state: 'unknown', + reason: 'ack timeout', + fence: 1 + }) + expect(hasUnansweredStructuredAgentSessionDispatch(journal.submissions())).toBe(true) + const restarted = await open() + await restarted.markPendingSubmissionsUnknown(2) + expect(restarted.submissions()[0]?.dispatchState).toBe('unknown') + expect(hasUnansweredStructuredAgentSessionDispatch(restarted.submissions())).toBe(false) + const cursor = restarted.cursor() + await restarted.markPendingSubmissionsUnknown(2) + expect(restarted.cursor()).toEqual(cursor) + }) + + it('preserves a proven write failure while retiring its live dispatch', async () => { + const journal = await open() + await journal.appendSubmission({ + clientMessageId: 'cm_write_failed', + payloadFingerprint: digestPayload('safe to retry'), + body: userMessage('safe to retry'), + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'cm_write_failed', + state: 'unknown', + reason: 'provider_write_failed: broken pipe', + fence: 1 + }) + + const restarted = await open() + await restarted.markPendingSubmissionsUnknown(2) + + expect(restarted.submissions()[0]).toMatchObject({ + dispatchState: 'unknown', + reason: 'provider_write_failed: broken pipe', + recovered: true + }) + expect(dispatchDoubtProvesUndelivered(restarted.submissions()[0]?.reason)).toBe(true) + }) + + it('turns an interrupted retry marker into recovery doubt', async () => { + const journal = await open() + await journal.appendSubmission({ + clientMessageId: 'cm_retrying', + payloadFingerprint: digestPayload('retry interrupted'), + body: userMessage('retry interrupted'), + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'cm_retrying', + state: 'unknown', + reason: DISPATCH_DOUBT_RETRY_IN_PROGRESS, + fence: 1 + }) + + const restarted = await open() + await restarted.markPendingSubmissionsUnknown(2, 'provider_exited_before_acknowledgement') + + expect(restarted.submissions()[0]).toMatchObject({ + dispatchState: 'unknown', + reason: 'provider_exited_before_acknowledgement', + recovered: true + }) + expect(dispatchDoubtProvesUndelivered(restarted.submissions()[0]?.reason)).toBe(false) + }) + it('reports a rejected submission as never delivered, and never re-sends it', async () => { const journal = await open() await journal.appendSubmission({ diff --git a/src/main/native-chat/agent-session-journal/journal-dispatch-doubt-reasons.ts b/src/main/native-chat/agent-session-journal/journal-dispatch-doubt-reasons.ts new file mode 100644 index 00000000000..3ac02fc89df --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-dispatch-doubt-reasons.ts @@ -0,0 +1,74 @@ +// Why a submission is in doubt, and whether Orca may put the message on the +// wire a second time. + +import type { AgentJournalDispatchState } from '../../../shared/agent-session-journal-types' +// +// `unknown` is never raised by elapsed time; what survives is a process fact. +// But a process fact that ends the WAIT is not the same claim as one that +// proves the message never reached a provider, and only the second justifies a +// re-delivery. The allowlist below names the reasons that carry the stronger +// claim, and it is deliberately FAIL-CLOSED: a reason nobody adds to it is +// refused. Refusing a legitimate retry costs the user one re-typed message; +// allowing an illegitimate one silently sends the model a second copy, which is +// the harm this whole path exists to remove. When those two are in tension, +// choose the re-type. + +/** A previous process wrote the message and died before learning its outcome. */ +export const DISPATCH_DOUBT_HOST_RESTARTED = 'host_restarted_before_acknowledgement' + +/** The child that would have acknowledged the message exited first. */ +export const DISPATCH_DOUBT_PROVIDER_EXITED = 'provider_exited_before_acknowledgement' + +/** The adapter took the message and only the journal write failed after it. */ +export const DISPATCH_DOUBT_PERSISTENCE_FAILED = 'dispatch_result_persistence_failed' + +/** A retry was durably armed but had not yet recorded its dispatch outcome. */ +export const DISPATCH_DOUBT_RETRY_IN_PROGRESS = 'dispatch_retry_in_progress' + +/** Codex owns a turn it started but did not name, because its turn-start still + * settles on a deadline. Delete this once Codex settles on the app-server's + * turn-start response instead; until then this reason is never re-delivered, + * which is what the allowlist below already does by omitting it. */ +export const DISPATCH_DOUBT_CODEX_TURN_UNNAMED = + 'codex app-server started a turn it did not name in time' + +/** The transport refused the frame; the underlying error follows the colon. */ +export const DISPATCH_DOUBT_WRITE_FAILED = 'provider_write_failed' + +/** The SDK took the frame, but its input pump did not prove whether the write completed. */ +export const DISPATCH_DOUBT_WRITE_OUTCOME_UNKNOWN = 'provider_write_outcome_unknown' + +export function dispatchWriteFailureReason(error: unknown): string { + const detail = error instanceof Error ? error.message : String(error) + return `${DISPATCH_DOUBT_WRITE_FAILED}: ${detail}` +} + +export function dispatchWriteOutcomeUnknownReason(error: unknown): string { + const detail = error instanceof Error ? error.message : String(error) + return `${DISPATCH_DOUBT_WRITE_OUTCOME_UNKNOWN}: ${detail}` +} + +/** + * The allowlist. True only where the frame is known never to have been taken by + * a provider, so sending it again is a first delivery rather than a second. + * + * A dead child and a dead host are NOT on this list. Both end the wait, neither + * proves non-delivery: the message was already written to that child's stdin, + * and Claude resumes the same provider session by id, so a message that child + * processed before dying is in the conversation Orca resumes. Deciding those + * needs the message matched against provider history — which is exactly what + * `journal-submission-reconciler.ts` does, and that module has no caller yet. + */ +export function dispatchDoubtProvesUndelivered(reason: string | null | undefined): boolean { + return ( + reason === DISPATCH_DOUBT_WRITE_FAILED || + reason?.startsWith(`${DISPATCH_DOUBT_WRITE_FAILED}: `) === true + ) +} + +export function dispatchMayMatchProviderEcho( + state: AgentJournalDispatchState, + reason: string | null +): boolean { + return state !== 'rejected' && !(state === 'unknown' && dispatchDoubtProvesUndelivered(reason)) +} diff --git a/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts b/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts index e95ff821058..5a5f39e5a62 100644 --- a/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts +++ b/src/main/native-chat/agent-session-journal/journal-epoch-rollover.ts @@ -5,7 +5,7 @@ // repair marker the superseded epoch was carrying. Superseded rows are DELETED // rather than retained — nothing would ever shed them. -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { journalRowSchemaVersion } from '../../../shared/agent-session-journal-types' import type { AgentSessionProviderHandle } from '../../../shared/agent-session-journal-types' import type Database from '../../sqlite/sync-database' import type { JournalLoad } from './journal-open' @@ -33,7 +33,8 @@ export function publishNewEpoch(input: { kind: 'epoch', reason: input.reason, providerHandle: input.providerHandle, - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + // Carries no body: an older host must keep reading a turn-free session past row 1. + v: journalRowSchemaVersion([]), epoch: input.epoch, seq: 1, fence: input.fence, diff --git a/src/main/native-chat/agent-session-journal/journal-item-revision.ts b/src/main/native-chat/agent-session-journal/journal-item-revision.ts new file mode 100644 index 00000000000..14b45368886 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-item-revision.ts @@ -0,0 +1,14 @@ +import type { JournalReducerState } from './journal-reducer' + +export function journalItemRevisionIsStale( + state: JournalReducerState, + itemId: string, + revision: number +): boolean { + const tombstoned = state.tombstones.get(itemId) + const existing = state.items.get(itemId) + return ( + (tombstoned !== undefined && revision <= tombstoned) || + (existing !== undefined && revision <= existing.revision) + ) +} diff --git a/src/main/native-chat/agent-session-journal/journal-lifecycle-batch-partition.ts b/src/main/native-chat/agent-session-journal/journal-lifecycle-batch-partition.ts index 009f44be784..305fa462f60 100644 --- a/src/main/native-chat/agent-session-journal/journal-lifecycle-batch-partition.ts +++ b/src/main/native-chat/agent-session-journal/journal-lifecycle-batch-partition.ts @@ -1,5 +1,5 @@ import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { journalRowSchemaVersion } from '../../../shared/agent-session-journal-types' import type { JournalLifecycleMutationInput } from './journal-row-builders' import type { JournalLifecycleBatchRow, JournalLifecycleMutation } from './journal-row-schema' import { @@ -54,7 +54,9 @@ function serializedLifecycleBatchFits( mutations: readonly JournalLifecycleMutationInput[] ): boolean { const row: JournalLifecycleBatchRow = { - v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, + v: journalRowSchemaVersion( + mutations.flatMap((mutation) => (mutation.kind === 'item' ? [mutation.body] : [])) + ), kind: 'lifecycle-batch', epoch: '00000000-0000-4000-8000-000000000000', seq: Number.MAX_SAFE_INTEGER, diff --git a/src/main/native-chat/agent-session-journal/journal-pending-submission-recovery.ts b/src/main/native-chat/agent-session-journal/journal-pending-submission-recovery.ts index 76bc00394f2..c0a2bee5431 100644 --- a/src/main/native-chat/agent-session-journal/journal-pending-submission-recovery.ts +++ b/src/main/native-chat/agent-session-journal/journal-pending-submission-recovery.ts @@ -1,18 +1,37 @@ +import { + DISPATCH_DOUBT_HOST_RESTARTED, + DISPATCH_DOUBT_RETRY_IN_PROGRESS +} from './journal-dispatch-doubt-reasons' import type { AgentSessionJournal } from './journal-store' +/** Settles every submission a process fact left unanswerable. The retry policy + * separately decides whether that fact proves the provider never received it. */ export async function markJournalPendingSubmissionsUnknown( journal: AgentSessionJournal, - fence: number + fence: number, + reason: string = DISPATCH_DOUBT_HOST_RESTARTED ): Promise { - const pending = journal.pendingSubmissions().map((entry) => entry.clientMessageId) - for (const clientMessageId of pending) { + const unresolved = journal + .submissions() + .filter( + (entry) => + entry.dispatchState === 'pending' || + (entry.dispatchState === 'unknown' && entry.recovered !== true) + ) + for (const entry of unresolved) { + const resolvedReason = + entry.dispatchState === 'unknown' && + entry.reason !== null && + entry.reason !== DISPATCH_DOUBT_RETRY_IN_PROGRESS + ? entry.reason + : reason await journal.resolveDispatch({ - clientMessageId, + clientMessageId: entry.clientMessageId, state: 'unknown', - reason: 'host_restarted_before_acknowledgement', + reason: resolvedReason, fence, recovered: true }) } - return pending + return unresolved.map((entry) => entry.clientMessageId) } diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts index 255e4ed184c..bbe78e41b64 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.test.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.test.ts @@ -216,6 +216,106 @@ describe('submission and dispatch state machine', () => { expect(items[0]?.revision).toBe(1) }) + it('durably accepts a pending submission from the provider echo row itself', () => { + const body = userText('hi') + const state = fold([ + { ...submission, payloadFingerprint: sendFingerprint(body) }, + { + kind: 'item', + itemId: 'claude:session-1:user-1', + revision: 1, + body, + ...base(2) + } + ]) + + expect(state.submissions.get('cm_1')).toMatchObject({ + dispatchState: 'accepted', + providerItemId: 'claude:session-1:user-1', + resolvedAt: 1_002 + }) + expect(state.receipts.get('cm_1')).toMatchObject({ + providerItemId: 'claude:session-1:user-1', + cursor: { epoch: EPOCH, sequence: 2 } + }) + }) + + it('does not give a newer identical echo to an older proven-undelivered submission', () => { + const body = userText('same message') + const state = fold([ + { + ...submission, + body, + payloadFingerprint: sendFingerprint(body) + }, + { + kind: 'dispatch', + clientMessageId: 'cm_1', + state: 'unknown', + providerItemId: null, + reason: 'provider_write_failed: closed before enqueue', + ...base(2) + }, + { + ...submission, + clientMessageId: 'cm_2', + body, + payloadFingerprint: sendFingerprint(body), + ...base(3) + }, + { + kind: 'item', + itemId: 'claude:session-1:user-1', + revision: 1, + body, + ...base(4) + } + ]) + + expect(state.submissions.get('cm_1')?.dispatchState).toBe('unknown') + expect(state.submissions.get('cm_2')).toMatchObject({ + dispatchState: 'accepted', + providerItemId: 'claude:session-1:user-1' + }) + expect(state.receipts.has('cm_1')).toBe(false) + expect(state.receipts.get('cm_2')?.providerItemId).toBe('claude:session-1:user-1') + }) + + it('does not accept a submission from a stale provider item behind its tombstone', () => { + const body = userText('hi') + const providerItemId = 'claude:session-1:user-1' + const state = fold([ + { ...submission, payloadFingerprint: sendFingerprint(body) }, + { kind: 'tombstone', itemId: providerItemId, revision: 2, ...base(2) }, + { kind: 'item', itemId: providerItemId, revision: 1, body, ...base(3) } + ]) + + expect(state.submissions.get('cm_1')?.dispatchState).toBe('pending') + expect(state.receipts.has('cm_1')).toBe(false) + expect(state.aliases.has(providerItemId)).toBe(false) + }) + + it('does not accept a submission from a stale lifecycle item behind its tombstone', () => { + const body = userText('hi') + const providerItemId = 'claude:session-1:user-1' + const state = fold([ + { ...submission, payloadFingerprint: sendFingerprint(body) }, + { + kind: 'lifecycle-batch', + settlementId: 'settlement-1', + mutations: [ + { kind: 'tombstone', itemId: providerItemId, revision: 2 }, + { kind: 'item', itemId: providerItemId, revision: 1, body } + ], + ...base(2) + } + ]) + + expect(state.submissions.get('cm_1')?.dispatchState).toBe('pending') + expect(state.receipts.has('cm_1')).toBe(false) + expect(state.aliases.has(providerItemId)).toBe(false) + }) + it.each(['codex:thread-1:turn-1:0', 'claude:session-1:user-1'])( 'preserves submitted text and attachments when %s is restored', (providerItemId) => { @@ -384,6 +484,36 @@ describe('submission and dispatch state machine', () => { expect(state.receipts.get('cm_1')).toBeTruthy() }) + it('returns a proven retry to pending without moving its original submission', () => { + const state = fold([ + submission, + { + kind: 'dispatch', + clientMessageId: 'cm_1', + state: 'unknown', + providerItemId: null, + reason: 'provider_write_failed: closed before enqueue', + ...base(2) + }, + { + kind: 'dispatch', + clientMessageId: 'cm_1', + state: 'pending', + providerItemId: null, + reason: null, + ...base(3) + } + ]) + + expect(state.submissions.get('cm_1')).toMatchObject({ + dispatchState: 'pending', + submittedAt: submission.ts, + reason: null, + resolvedAt: null + }) + expect(renderJournalState(state).items[0]?.sequence).toBe(submission.seq) + }) + it('ignores a dispatch for a submission this epoch never saw', () => { const state = fold([ { diff --git a/src/main/native-chat/agent-session-journal/journal-reducer.ts b/src/main/native-chat/agent-session-journal/journal-reducer.ts index 41625792aa0..9759f449f2e 100644 --- a/src/main/native-chat/agent-session-journal/journal-reducer.ts +++ b/src/main/native-chat/agent-session-journal/journal-reducer.ts @@ -18,6 +18,8 @@ import { parseAgentJournalItemKey } from '../../../shared/agent-session-journal-item-key' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' +import { dispatchMayMatchProviderEcho } from './journal-dispatch-doubt-reasons' +import { journalItemRevisionIsStale } from './journal-item-revision' import type { JournalRow } from './journal-row-schema' export const MAX_JOURNAL_APPLIED_SETTLEMENT_IDS = 4_096 @@ -66,7 +68,11 @@ export function applyJournalRow(state: JournalReducerState, row: JournalRow): vo } state.lastActivityAt = Math.max(state.lastActivityAt, row.ts) if (row.kind === 'item') { + if (journalItemRevisionIsStale(state, row.itemId, row.revision)) { + return + } const itemId = resolveJournalItemId(state, row.itemId, row.body) + acceptSubmissionFromProviderItem(state, row.itemId, itemId, row) upsertItem(state, itemId, row.revision, { itemId, revision: row.revision, @@ -87,7 +93,11 @@ export function applyJournalRow(state: JournalReducerState, row: JournalRow): vo } for (const mutation of row.mutations) { if (mutation.kind === 'item') { + if (journalItemRevisionIsStale(state, mutation.itemId, mutation.revision)) { + continue + } const itemId = resolveJournalItemId(state, mutation.itemId, mutation.body) + acceptSubmissionFromProviderItem(state, mutation.itemId, itemId, row) upsertItem(state, itemId, mutation.revision, { itemId, revision: mutation.revision, @@ -151,12 +161,12 @@ export function resolveJournalItemId( // Exact payload plus queue order preserves repeated identical sends one-for-one. const submission = [...state.submissions.values()] .sort((left, right) => left.submittedAt - right.submittedAt) - .find((candidate) => { - if (candidate.dispatchState === 'rejected' || candidate.payloadFingerprint !== fingerprint) { - return false - } - return state.items.get(agentJournalSubmissionKey(candidate.clientMessageId))?.revision === 0 - }) + .find( + (candidate) => + dispatchMayMatchProviderEcho(candidate.dispatchState, candidate.reason) && + candidate.payloadFingerprint === fingerprint && + state.items.get(agentJournalSubmissionKey(candidate.clientMessageId))?.revision === 0 + ) if (!submission) { return itemId } @@ -256,10 +266,16 @@ function applyDispatch( if (submission.dispatchState === 'rejected' || submission.dispatchState === 'accepted') { return } + submission.fence = row.fence submission.dispatchState = row.state submission.providerItemId = row.providerItemId submission.reason = row.reason - submission.resolvedAt = row.ts + submission.resolvedAt = row.state === 'pending' ? null : row.ts + if (row.recovered) { + submission.recovered = row.recovered + } else { + delete submission.recovered + } if (row.state !== 'accepted' || !row.providerItemId) { return } @@ -272,6 +288,39 @@ function applyDispatch( }) } +function acceptSubmissionFromProviderItem( + state: JournalReducerState, + providerItemId: string, + resolvedItemId: string, + row: Pick +): void { + if (providerItemId === resolvedItemId) { + return + } + const submission = [...state.submissions.values()].find( + (candidate) => agentJournalSubmissionKey(candidate.clientMessageId) === resolvedItemId + ) + if ( + !submission || + submission.dispatchState === 'accepted' || + submission.dispatchState === 'rejected' + ) { + return + } + submission.fence = row.fence + submission.dispatchState = 'accepted' + submission.providerItemId = providerItemId + submission.reason = null + submission.resolvedAt = row.ts + delete submission.recovered + state.receipts.set(submission.clientMessageId, { + clientMessageId: submission.clientMessageId, + providerItemId, + cursor: { epoch: row.epoch, sequence: row.seq }, + acceptedAt: row.ts + }) +} + /** Project the folded state into the client-facing snapshot. */ export function renderJournalState(state: JournalReducerState): AgentJournalSnapshot { // Sequence is the sole ordering key; map insertion order is not, because a diff --git a/src/main/native-chat/agent-session-journal/journal-row-builders.ts b/src/main/native-chat/agent-session-journal/journal-row-builders.ts index db82e86318f..be2c2552775 100644 --- a/src/main/native-chat/agent-session-journal/journal-row-builders.ts +++ b/src/main/native-chat/agent-session-journal/journal-row-builders.ts @@ -5,7 +5,7 @@ import type { AgentJournalMessageItem, AgentSessionProviderHandle } from '../../../shared/agent-session-journal-types' -import { AGENT_SESSION_JOURNAL_SCHEMA_VERSION } from '../../../shared/agent-session-journal-types' +import { journalRowSchemaVersion } from '../../../shared/agent-session-journal-types' import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' import type { JournalReducerState } from './journal-reducer' import type { @@ -76,7 +76,8 @@ export function journalDispatchRowBuilder( clientMessageId: input.clientMessageId, dispatchState: input.state, providerItemId, - reason: input.state === 'accepted' ? null : (input.reason ?? null), + reason: + input.state === 'accepted' || input.state === 'pending' ? null : (input.reason ?? null), seq, fence: input.fence, ts, @@ -118,7 +119,13 @@ export function journalLifecycleBatchRowBuilder( kind: 'lifecycle-batch', settlementId, mutations: built, - ...journalRowBase(current.epoch, seq, options.fence, ts), + ...journalRowBase( + current.epoch, + seq, + options.fence, + ts, + built.flatMap((mutation) => (mutation.kind === 'item' ? [mutation.body] : [])) + ), ...(options.recovered ? { recovered: options.recovered } : {}) } if (Buffer.byteLength(JSON.stringify(row), 'utf8') + 1 > MAX_JOURNAL_LIFECYCLE_BATCH_BYTES) { @@ -132,9 +139,10 @@ export function journalRowBase( epoch: string, seq: number, fence: number, - ts: number + ts: number, + bodies: readonly { kind: string }[] = [] ): { v: number; epoch: string; seq: number; fence: number; ts: number } { - return { v: AGENT_SESSION_JOURNAL_SCHEMA_VERSION, epoch, seq, fence, ts } + return { v: journalRowSchemaVersion(bodies), epoch, seq, fence, ts } } export function buildJournalItemRow(input: { @@ -160,7 +168,7 @@ export function buildJournalItemRow(input: { itemId, revision, body: input.body, - ...journalRowBase(input.state.epoch, input.seq, input.fence, input.ts), + ...journalRowBase(input.state.epoch, input.seq, input.fence, input.ts, [input.body]), ...(input.recovered ? { recovered: input.recovered } : {}) } } @@ -212,7 +220,7 @@ export function buildJournalSubmissionRow(input: { export function buildJournalDispatchRow(input: { state: JournalReducerState clientMessageId: string - dispatchState: Exclude + dispatchState: AgentJournalDispatchState providerItemId: string | null reason: string | null seq: number diff --git a/src/main/native-chat/agent-session-journal/journal-row-schema-version.test.ts b/src/main/native-chat/agent-session-journal/journal-row-schema-version.test.ts new file mode 100644 index 00000000000..b764ebd5ec9 --- /dev/null +++ b/src/main/native-chat/agent-session-journal/journal-row-schema-version.test.ts @@ -0,0 +1,66 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { agentJournalTurnBody } from '../../../shared/agent-session-turn-record' +import { openJournalDatabase } from './journal-database' +import { journalDatabaseFile } from './journal-paths' +import { createTrackedJournalOpener } from './journal-store-test-open' + +// Which rows an older host can still read: only rows that carry a turn item +// are stamped with the version it does not know, and the epoch row never is. +// Read raw: the reader upcasts every row to the current version, so only the +// stored row_json says what an older build would see. +describe('journal row schema versions', () => { + let root = '' + const opener = createTrackedJournalOpener() + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-journal-row-version-')) + }) + afterEach(async () => { + await opener.closeAll() + await rm(root, { recursive: true, force: true }) + }) + + it('stamps v3 only on rows that carry a turn item', async () => { + const journal = await opener.open({ + identity: { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + now: () => 1_000, + journalDir: join(root, 'session-1') + }) + const identity = { provider: 'orca' as const, clientMessageId: 'm1' } + await journal.appendItem( + identity, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] }, + { fence: 1 } + ) + await journal.appendItem( + { provider: 'legacy', agent: 'codex', sessionId: 'session-1', recordId: 'turn-lifecycle:t1' }, + agentJournalTurnBody({ turnId: 't1', state: 'running', startedAt: 1_000 }), + { fence: 1 } + ) + await journal.close() + const opened = openJournalDatabase(journalDatabaseFile(join(root, 'session-1'))) + try { + const stored = opened.db + .prepare('SELECT row_json FROM journal_rows ORDER BY seq') + .all() + .map((row) => JSON.parse(String((row as { row_json: string }).row_json))) + .map((row: { kind: string; v: number }) => [row.kind, row.v]) + expect(stored).toEqual([ + ['epoch', 2], + ['item', 2], + ['item', 3] + ]) + } finally { + opened.db.close() + } + }) +}) diff --git a/src/main/native-chat/agent-session-journal/journal-row-schema.ts b/src/main/native-chat/agent-session-journal/journal-row-schema.ts index dd8b0ce9f3e..7dc02dd197b 100644 --- a/src/main/native-chat/agent-session-journal/journal-row-schema.ts +++ b/src/main/native-chat/agent-session-journal/journal-row-schema.ts @@ -74,7 +74,7 @@ export type JournalSubmissionRow = JournalRowBase & { export type JournalDispatchRow = JournalRowBase & { kind: 'dispatch' clientMessageId: string - state: Exclude + state: AgentJournalDispatchState /** Provider item identity adopted on accept. */ providerItemId: string | null reason: string | null diff --git a/src/main/native-chat/agent-session-journal/journal-store-contracts.ts b/src/main/native-chat/agent-session-journal/journal-store-contracts.ts index 22e3a4c7cca..80c806b02e3 100644 --- a/src/main/native-chat/agent-session-journal/journal-store-contracts.ts +++ b/src/main/native-chat/agent-session-journal/journal-store-contracts.ts @@ -29,6 +29,7 @@ export type ResolveDispatchInput = { recovered?: true } & ( | { state: 'accepted'; providerIdentity: AgentJournalItemIdentity } + | { state: 'pending' } | { state: 'rejected' | 'unknown'; reason?: string | null } ) diff --git a/src/main/native-chat/agent-session-journal/journal-store.ts b/src/main/native-chat/agent-session-journal/journal-store.ts index 3c16800099b..3be64d9ceca 100644 --- a/src/main/native-chat/agent-session-journal/journal-store.ts +++ b/src/main/native-chat/agent-session-journal/journal-store.ts @@ -232,7 +232,7 @@ export class AgentSessionJournal { } /** - * Advance a submission to exactly one of accepted / rejected / unknown. + * Record a dispatch transition, including a proven retry returning to pending. * * Accepting REQUIRES the provider identity rather than a free-form id: the * adopted key is what the provider's echo will upsert into, so a mismatched @@ -245,10 +245,9 @@ export class AgentSessionJournal { })) } - /** On restart every `pending` submission becomes `unknown` before the session - * accepts a writer. Orca never re-sends on the user's behalf. */ - async markPendingSubmissionsUnknown(fence: number): Promise { - return markJournalPendingSubmissionsUnknown(this, fence) + /** Retire unanswered sends after their execution owner ended, without assuming delivery. */ + async markPendingSubmissionsUnknown(fence: number, reason?: string): Promise { + return markJournalPendingSubmissionsUnknown(this, fence, reason) } /** The escape hatch for corruption, an unreconcilable prefix, a forked handle, diff --git a/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts index ade04667174..1dd1115ebd1 100644 --- a/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts +++ b/src/main/native-chat/agent-session-journal/journal-terminal-settlement.ts @@ -1,4 +1,5 @@ import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' +import { isRunningAgentJournalTurn } from '../../../shared/agent-session-turn-record' /** True while an item is still awaiting the row that settles it, so a sink can * treat that row as lifecycle-critical rather than sheddable under pressure. */ @@ -9,5 +10,5 @@ export function requiresTerminalSettlement(body: AgentJournalItemBody): boolean if (body.kind === 'approval' || body.kind === 'question') { return body.resolution.state === 'pending' } - return body.kind === 'status' && body.turnLifecycle?.state === 'running' + return isRunningAgentJournalTurn(body) } diff --git a/src/main/native-chat/agent-session-wire/agent-session-empty-batch.ts b/src/main/native-chat/agent-session-wire/agent-session-empty-batch.ts new file mode 100644 index 00000000000..bc713f92171 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/agent-session-empty-batch.ts @@ -0,0 +1,7 @@ +import type { AgentJournalCursor } from '../../../shared/agent-session-journal-types' +import type { AgentSessionJournalBatch } from '../../../shared/agent-session-wire' + +/** A batch that advances nothing: the carrier for fence, handoff, roster, and clock updates. */ +export function emptyAgentSessionBatch(cursor: AgentJournalCursor): AgentSessionJournalBatch { + return { cursor, items: [], removedItemIds: [], submissions: [] } +} diff --git a/src/main/native-chat/agent-session-wire/provider-frame-activity.test.ts b/src/main/native-chat/agent-session-wire/provider-frame-activity.test.ts index 40504873282..c8860f227cc 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-activity.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-activity.test.ts @@ -38,36 +38,26 @@ describe('provider frame activity', () => { } }) - it('uses Claude descriptions and safe semantic status without exposing tool labels', () => { + it('leaves the Claude line on the generic fallback, since Claude never narrates its turn', () => { + // Prose on these frames belongs to a spawned task, not to this turn. + for (const [kind, payload] of [ + ['message:system:task_started', { description: 'Trace the activity channel' }], + ['message:system:task_progress', { summary: 'Checking remote compatibility' }], + ['message:system:task_updated', { patch: { description: 'Validating the renderer' } }], + ['message:system:control_request_progress', { status: 'api_retry' }], + ['message:tool_progress', { tool_name: 'ReadSecretFile' }] + ] as const) { + expect(claudeProviderFrameActivity(kind, payload)).toBeNull() + } + // `requesting` holds for nearly the whole turn and says no more than the fallback. expect( - claudeProviderFrameActivity('message:system:task_started', { - description: 'Trace the activity channel' - }) - ).toBe('Working on: Trace the activity channel') - expect( - claudeProviderFrameActivity('message:system:task_progress', { - description: 'Reading tests', - summary: 'Checking remote compatibility' - }) - ).toBe('Checking remote compatibility') - expect( - claudeProviderFrameActivity('message:system:task_updated', { - patch: { description: 'Validating the renderer' } - }) - ).toBe('Validating the renderer') + claudeProviderFrameActivity('message:system:status', { status: 'requesting' }) + ).toBeNull() expect(claudeProviderFrameActivity('message:system:status', { status: 'compacting' })).toBe( 'Compacting the conversation' ) - expect( - claudeProviderFrameActivity('message:system:control_request_progress', { - status: 'api_retry' - }) - ).toBe('Retrying a side question') - expect( - claudeProviderFrameActivity('message:tool_progress', { - tool_name: 'ReadSecretFile' - }) - ).toBeNull() + // An unmodeled frame still declines to answer, so it cannot clear live copy. + expect(claudeProviderFrameActivity('message:system:unknown_frame', {})).toBeUndefined() }) it('falls through on protocol noise and bounds long copy', () => { diff --git a/src/main/native-chat/agent-session-wire/provider-frame-activity.ts b/src/main/native-chat/agent-session-wire/provider-frame-activity.ts index 336170a4cfe..3cd6c4abaf0 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-activity.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-activity.ts @@ -100,40 +100,29 @@ export function codexProviderFrameActivity( return itemType ? (CODEX_ITEM_ACTIVITY[itemType] ?? null) : null } +/** + * Claude does not narrate its own turn, so the activity line stays the generic fallback. + * + * Codex names each item it starts, which is what makes its line worth reading. Claude's only + * turn-wide frame is `system/status`, whose payload is a bare token — every sentence Orca ever + * put on this line for it was Orca's own wording for `requesting`, which is true for nearly the + * whole turn and says no more than the fallback does. Its `task_*` frames do carry prose, but + * they are keyed by task id and subagent type: they describe a spawned task, not this turn, and + * the background-tasks strip already owns that. Compaction is the one exception kept — a real, + * rare state that explains an otherwise unexplained wait, and the Codex map reports it too. + */ export function claudeProviderFrameActivity(kind: string, payload: unknown): ActivityText { const source = record(payload) - if (kind === 'message:system:task_started') { - if (source?.ambient === true || source?.skip_transcript === true) { - return null - } - const description = providerActivityText(stringField(source, 'description')) - return description ? providerActivityText(`Working on: ${description}`) : null - } - if (kind === 'message:system:task_progress') { - return providerActivityText( - stringField(source, 'summary') ?? stringField(source, 'description') - ) - } - if (kind === 'message:system:task_updated') { - return providerActivityText(stringField(record(source?.patch), 'description')) - } if (kind === 'message:system:status') { - const status = stringField(source, 'status') - return status === 'compacting' - ? 'Compacting the conversation' - : status === 'requesting' - ? 'Requesting a response' - : null + return stringField(source, 'status') === 'compacting' ? 'Compacting the conversation' : null } - if (kind === 'message:system:control_request_progress') { - const status = stringField(source, 'status') - return status === 'started' - ? 'Exploring a side question' - : status === 'api_retry' - ? 'Retrying a side question' - : null - } - if (kind === 'message:tool_progress') { + if ( + kind === 'message:system:task_started' || + kind === 'message:system:task_progress' || + kind === 'message:system:task_updated' || + kind === 'message:system:control_request_progress' || + kind === 'message:tool_progress' + ) { return null } return undefined diff --git a/src/main/native-chat/agent-session-wire/provider-turn-activity-routing.test.ts b/src/main/native-chat/agent-session-wire/provider-turn-activity-routing.test.ts index a66ac567a4d..0067aea8ed5 100644 --- a/src/main/native-chat/agent-session-wire/provider-turn-activity-routing.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-turn-activity-routing.test.ts @@ -248,10 +248,11 @@ describe('provider turn activity routing', () => { }) ) expect(state.rows).toHaveLength(turnRows) + // Only compaction reaches the line; task and side-question prose is not this turn's work. expect(state.activities.slice(-3)).toEqual([ - { turnId: TURN_ID, text: 'Checking the renderer state' }, + null, { turnId: TURN_ID, text: 'Compacting the conversation' }, - { turnId: TURN_ID, text: 'Exploring a side question' } + null ]) translator.handle(claudeMessage({ type: 'tool_progress', tool_name: 'SecretReader' })) @@ -262,6 +263,7 @@ describe('provider turn activity routing', () => { claudeMessage({ type: 'result', subtype: 'success', is_error: false, result: 'Done' }) ) expect(state.activities.at(-1)).toBeNull() - expect(state.tombstones).toHaveLength(1) + expect(state.tombstones).toHaveLength(0) + expect(state.rows.at(-1)).toMatchObject({ kind: 'turn', turnId: TURN_ID, state: 'completed' }) }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts index 9a480438c25..3bd14a057aa 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts @@ -91,6 +91,13 @@ export function isAgentSessionPreSpawnError(error: unknown): error is AgentSessi export type AgentSessionDispatchOutcome = /** The provider owns the turn now, under this identity. */ | { state: 'accepted'; providerIdentity: AgentJournalItemIdentity } + /** + * The provider transport took the message; identity settles later, out of band. + * The submission stays `pending`: a message queued behind a running turn is + * acknowledged only when that turn starts, so elapsed time is not evidence of + * anything and never promotes this to `unknown`. + */ + | { state: 'admitted' } | { state: 'rejected'; reason: string } /** The call did not settle. Never re-send on the user's behalf. */ | { state: 'unknown'; reason: string } @@ -102,6 +109,8 @@ export type StructuredAgentSessionLifecycleEvent = { cause: 'unexpected-exit' | 'requested-close' fence: number acquisitionGeneration: string + /** Host receipt of the child exit, retained across settlement retries. */ + observedAt?: number /** Translator could not admit terminal rows; host recovery must append its bounded fallback. */ settlementRetryRequired?: boolean } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts index 986fcefadbf..4b9b46716cb 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts @@ -38,5 +38,8 @@ export type StructuredAgentSessionAttachContext = { reconcileLeases: (sessionId: string) => Promise serialize: (sessionId: string, task: () => Promise) => Promise now: () => number + /** Paired with `sessions.delete` by `forgetStructuredAgentSession`; a failed attach that only + * deleted would leave the store's row behind. */ + forgetStatus: (sessionId: string) => void publishStatus?: (sessionId: string) => void } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts index bb889ad23e3..ad84b9997be 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts @@ -46,10 +46,13 @@ export type AttachFlowInput = { callerKey: string params: AgentSessionAttachParams now: () => number - /** Publishes the journal before clients can send against the new owner. */ + /** Publishes the journal before clients can send against the new owner. `acquiredOwner` is + * true only when this attach spawned the provider child, so a re-attach to a live one is not + * mistaken for a cold acquire. */ onAttached: ( attached: AttachedJournal, - acquisitionGeneration: string | null + acquisitionGeneration: string | null, + acquiredOwner: boolean ) => Promise | void /** Host-owned provider sink, bound to the journal inside `onAttached`. */ eventSink?: StructuredAgentSessionEventSink @@ -84,6 +87,7 @@ export async function performAttach( let record: AgentSessionRecord let acquisitionGeneration: string | null = null + let acquiredOwner = false let reservedRecord: AgentSessionRecord | null = null let unsupportedReservationSettlementAttempted = false let replayed = false @@ -139,6 +143,7 @@ export async function performAttach( const acquired = await acquireOwner(input, record) record = acquired.record acquisitionGeneration = acquired.acquisitionGeneration + acquiredOwner = true } } catch (error) { const spawnToken = reservedRecord?.lease.reservedSpawnToken @@ -213,7 +218,7 @@ export async function performAttach( adapter: input.adapter }) await importAdoptedTranscript(params, attached, record, preparedTranscript.items) - await input.onAttached(attached, acquisitionGeneration) + await input.onAttached(attached, acquisitionGeneration, acquiredOwner) await store.recordOperationOutcome({ callerKey: input.callerKey, operationId: params.envelope.clientOperationId, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts index 3a58d71b625..85e02f2fea7 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts @@ -22,7 +22,9 @@ import { } from './structured-agent-session-launch-env' import { refuseAgentSessionMutation } from './structured-agent-session-mutation-admission' import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' +import { settleStaleRunningTurnsOnAcquire } from './structured-agent-session-stale-turn-verdict' import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' +import { forgetStructuredAgentSession } from './structured-agent-session-host-lifetime' import type { DeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' @@ -90,18 +92,26 @@ export function attachStructuredAgentSession( // Site 9: this closes the PRIOR map entry it drops, never the provisional // journal — it has no reference to that one. `onAttached` owns that. onAttachFailed: async () => { - await context.sessions.get(sessionId)?.journal.close() - context.sessions.delete(sessionId) + await forgetStructuredAgentSession(context, sessionId) eventSink.close() context.runtimeState.discardEventSink(sessionId) }, - onAttached: async (attached, acquisitionGeneration) => { + onAttached: async (attached, acquisitionGeneration, acquiredOwner) => { const fence = context.deps.store.getRecord(sessionId)?.lease.runtimeFence ?? 0 const previous = context.sessions.get(sessionId) const previousFence = previous?.fence // Site 8: the provisional journal has no owner until the map takes it, // and the barrier below throws by design. try { + if (acquiredOwner) { + // Before the drain: the buffered events are the new child's, never a stale row's. + await settleStaleRunningTurnsOnAcquire({ + journal: attached.journal, + sessionId, + fence, + acquisitionGeneration + }) + } await bindAndDrain(eventSink, attached.journal, fence, (activity) => context.subscribers.publish(sessionId, attached.journal, activity) ) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts index 4592dea26e9..6dfacbd2ff0 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts @@ -21,7 +21,10 @@ export class StructuredAgentSessionBackgroundTaskChannel { private readonly requireSession: (sessionId: string) => StructuredAgentSessionHostSession, private readonly handoffStatus: ( sessionId: string - ) => Parameters[0]['handoff'] + ) => Parameters[0]['handoff'], + /** Task edges change the status summary too; the feed's equality check + * keeps a no-op re-projection from reaching subscribers. */ + private readonly onPublished: (sessionId: string) => void ) {} history(request: AgentSessionHistoryRequest): AgentSessionHistoryResult { @@ -31,9 +34,15 @@ export class StructuredAgentSessionBackgroundTaskChannel { request }) const backgroundTasks = this.state(request.sessionId) - return backgroundTasks === undefined - ? result - : { ...result, page: { ...result.page, backgroundTasks } } + const hostNow = this.deps.now?.() ?? Date.now() + return { + ...result, + page: { + ...result.page, + hostNow, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + } + } } subscribe(input: AgentSessionSubscribeInput): () => void { @@ -53,6 +62,7 @@ export class StructuredAgentSessionBackgroundTaskChannel { const state = publishedState !== undefined ? publishedState : this.state(sessionId) if (session && state !== undefined) { this.subscribers.backgroundTasks(sessionId, state, session.fence) + this.onPublished(sessionId) } } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-client-delivery.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-client-delivery.ts new file mode 100644 index 00000000000..92a4f2653e0 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-client-delivery.ts @@ -0,0 +1,75 @@ +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { AgentSessionSubscribers } from './structured-agent-session-subscribers' +import type { + StructuredAgentSessionHostDeps, + StructuredAgentSessionHostSession +} from './structured-agent-session-host-types' +import { AGENT_SESSION_NOT_ATTACHED } from './structured-agent-session-mutation-admission' +import { StructuredAgentSessionSendSettlement } from './structured-agent-session-send-settlement' +import { + createStructuredAgentSessionHostStatusFeed, + type StructuredAgentSessionStatusSubscriber +} from './structured-agent-session-status-feed' + +/** Owns every host-to-client publication edge, including compatibility waits. */ +export class StructuredAgentSessionClientDelivery { + readonly subscribers: AgentSessionSubscribers + readonly waitForSendSettlement: StructuredAgentSessionSendSettlement['wait'] + private readonly statusFeed + private readonly sendSettlement + + constructor( + private readonly sessions: Map, + now: () => number, + deps: () => StructuredAgentSessionHostDeps + ) { + this.statusFeed = createStructuredAgentSessionHostStatusFeed({ sessions, now, deps }) + this.sendSettlement = new StructuredAgentSessionSendSettlement((sessionId) => + this.requireJournal(sessionId) + ) + this.waitForSendSettlement = this.sendSettlement.wait + this.subscribers = new AgentSessionSubscribers({ + readCommands: (sessionId) => deps().adapter.readCommands?.(sessionId), + onJournalPublished: (sessionId, journal) => this.publishJournal(sessionId, journal) + }) + } + + publishStatus = (sessionId: string): void => this.statusFeed.publish(sessionId) + + publishStatusAndSettlement = (sessionId: string): void => { + this.statusFeed.publish(sessionId) + const journal = this.sessions.get(sessionId)?.journal + if (journal) { + this.sendSettlement.publish(sessionId, journal) + } + } + + publishRestored = (sessionId: string): void => + this.statusFeed.publish(sessionId, undefined, { replay: true }) + + subscribeStatus = (subscriber: StructuredAgentSessionStatusSubscriber): (() => void) => + this.statusFeed.subscribe(subscriber) + forgetStatus = (sessionId: string): void => this.statusFeed.forget(sessionId) + + closeSession(sessionId: string): void { + this.sendSettlement.closeSession(sessionId) + this.statusFeed.close(sessionId) + } + + closeAll(): void { + this.sendSettlement.closeAll() + } + + private publishJournal(sessionId: string, journal: AgentSessionJournal): void { + this.statusFeed.publish(sessionId, journal) + this.sendSettlement.publish(sessionId, journal) + } + + private requireJournal(sessionId: string): AgentSessionJournal { + const journal = this.sessions.get(sessionId)?.journal + if (!journal) { + throw new Error(AGENT_SESSION_NOT_ATTACHED.code) + } + return journal + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts index a4666b4f045..b79091c7f9f 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-close-retry.test.ts @@ -131,7 +131,8 @@ function attachContext( tasks: { trackAttach: (task: Promise) => task }, reconcileLeases: async () => null, serialize: (_sessionId: string, task: () => Promise) => task(), - now: () => 1 + now: () => 1, + forgetStatus: () => undefined } as unknown as StructuredAgentSessionAttachContext } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts index 6952b6d93e6..784e3aa7b20 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-event-sink.ts @@ -27,6 +27,8 @@ export type StructuredAgentSessionAppendOptions = { coalescingKey?: string /** Marks a critical lifecycle operation for lifecycle barriers and diagnostics. */ lifecycle?: boolean + /** Host clock to stamp on the row instead of its append time. */ + observedAt?: number } export type StructuredAgentSessionEventSink = { @@ -162,7 +164,11 @@ export function createDeferredStructuredAgentSessionEventSink( { bytes: estimateStructuredAgentSessionItemBytes(identity, body), coalescingKey: options.coalescingKey, - run: (bound) => bound.journal.appendItem(identity, body, { fence: bound.fence }) + run: (bound) => + bound.journal.appendItem(identity, body, { + fence: bound.fence, + ...(options.observedAt === undefined ? {} : { observedAt: options.observedAt }) + }) }, options ) @@ -172,7 +178,11 @@ export function createDeferredStructuredAgentSessionEventSink( { bytes: estimateStructuredAgentSessionItemBytes(identity, body), coalescingKey: options.coalescingKey, - run: (bound) => bound.journal.appendItem(identity, body, { fence: bound.fence }) + run: (bound) => + bound.journal.appendItem(identity, body, { + fence: bound.fence, + ...(options.observedAt === undefined ? {} : { observedAt: options.observedAt }) + }) }, options ), diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-forget-status.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-forget-status.test.ts new file mode 100644 index 00000000000..d74a729dc2f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-forget-status.test.ts @@ -0,0 +1,162 @@ +// Removing a session from the host's map and removing its status row are ONE operation. +// +// The store keeps a row until told to drop it, and `structuredHostOwned` bypasses the staleness +// check in `agent-status-freshness.ts` — so a deletion path that skipped the forget leaves a +// permanently `working` agent in `worktree ps` and on mobile, with no UI able to clear it. +// +// This drives the real orchestration closure for the failed attach, against the real store. + +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import { AgentHookServer } from '../../agent-hooks/server' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import { attachStructuredAgentSession } from './structured-agent-session-attach-orchestration' +import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' +import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' +import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed' + +// Everything before the journal is out of scope here; what matters is that the orchestration's +// own `onAttachFailed` runs, which is the real one. +vi.mock('./structured-agent-session-attach-flow', () => ({ + performAttach: async (input: { onAttachFailed?: () => Promise }) => { + await input.onAttachFailed?.() + throw new Error('attach failed after acquisition') + } +})) + +const SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' +const TURN = { provider: 'codex', threadId: 'thread-1', turnId: 'turn-1', ordinal: 0 } as const +const PROMPT = { ...TURN, ordinal: 1 } + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: SESSION, + workspaceId: 'repo-1::/workspace/app', + hostId: 'host-1', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: SESSION } +} + +let root: string +const journals = createTrackedJournalOpener() + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-forget-status-')) +}) + +afterEach(async () => { + await journals.closeAll() + await rm(root, { recursive: true, force: true }) +}) + +/** A session the store already lists as a host-owned working agent. */ +async function workingSession(): Promise<{ + server: AgentHookServer + feed: StructuredAgentSessionStatusFeed + sessions: Map + journal: AgentSessionJournal +}> { + const journal = await journals.open({ identity: IDENTITY, journalDir: join(root, SESSION) }) + await journal.appendItem( + PROMPT, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'ship it' }] }, + { fence: 1 } + ) + await journal.appendItem( + TURN, + { kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 1 } + ) + const sessions = new Map([ + [ + SESSION, + { + journal, + params: { location: { workspaceId: IDENTITY.workspaceId }, provider: 'codex' }, + fence: 1, + hasProviderChild: true, + acquisitionGeneration: null + } as unknown as StructuredAgentSessionHostSession + ] + ]) + const server = new AgentHookServer() + const feed = new StructuredAgentSessionStatusFeed({ + sessions, + getRecord: () => null, + now: () => 1, + statusSink: () => ({ + publish: (summary) => server.ingestStructuredStatus(summary), + forget: (sessionId) => server.dropStructuredStatus(sessionId) + }) + }) + feed.publish(SESSION, journal) + expect(server.getStatusSnapshot()).toEqual([ + expect.objectContaining({ state: 'working', structuredHost: 'owned' }) + ]) + return { server, feed, sessions, journal } +} + +function attachContext( + sessions: Map, + feed: StructuredAgentSessionStatusFeed +): StructuredAgentSessionAttachContext { + const eventSink = { + sink: {}, + drained: async () => ({ ok: true }) as const, + unbind: () => undefined, + bind: () => undefined, + close: () => undefined + } + return { + deps: { store: { getRecord: () => null }, claimKeyId: 'key-1', journalRoot: root }, + runtimeState: { + resolveRecovery: async () => undefined, + eventSinkFor: () => eventSink, + probeOwner: async () => ({ outcome: 'pid-absent' }), + discardEventSink: () => undefined + }, + sessions, + subscribers: { reset: () => undefined, snapshot: () => undefined, publish: () => undefined }, + tasks: { trackAttach: (task: Promise) => task }, + reconcileLeases: async () => null, + serialize: (_sessionId: string, task: () => Promise) => task(), + now: () => 1, + forgetStatus: (sessionId: string) => feed.forget(sessionId) + } as unknown as StructuredAgentSessionAttachContext +} + +const attachParams = { + envelope: { sessionId: SESSION, clientOperationId: 'op-1' } +} as unknown as Parameters[2] + +describe('a session that leaves the host without an explicit close', () => { + it('leaves the agent-status store with it when an attach fails', async () => { + const { server, feed, sessions } = await workingSession() + + await expect( + attachStructuredAgentSession(attachContext(sessions, feed), 'caller-1', attachParams) + ).rejects.toThrow('attach failed after acquisition') + + expect(sessions.has(SESSION)).toBe(false) + expect(server.getStatusSnapshot()).toEqual([]) + }) + + // The feed's own cache deliberately retains the projection for reload history; only the store + // is a roster, which is why the forget has to be explicit rather than derived from the cache. + it('keeps the projection a reloading renderer still needs', async () => { + const { feed, sessions } = await workingSession() + + await expect( + attachStructuredAgentSession(attachContext(sessions, feed), 'caller-1', attachParams) + ).rejects.toThrow('attach failed after acquisition') + + const events: unknown[] = [] + feed.subscribe({ id: 'list-late', emit: (event) => events.push(event) }) + expect(events).toEqual([ + { type: 'snapshot', sessions: [expect.objectContaining({ sessionId: SESSION })] } + ]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts index 3495409f127..113940ff0f4 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-handoff.ts @@ -86,6 +86,13 @@ export function createStructuredAgentSessionHostHandoff( host.publishStatus?.(sessionId) try { await host.flush(sessionId) + const session = host.session(sessionId) + await session.journal.markPendingSubmissionsUnknown( + session.fence, + 'provider_exited_before_acknowledgement' + ) + host.subscribers.publish(sessionId, session.journal) + host.publishStatus?.(sessionId) host.eventSink(sessionId).unbind() return { state: 'stopped' } } catch (error) { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts index d1664bc985b..e2f75297a5a 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-lifetime.ts @@ -27,6 +27,19 @@ export type StructuredAgentSessionLifetimeContext = { runtimeState: StructuredAgentSessionHostRuntimeState sessions: Map now: () => number + /** Drops the session's row from the agent-status store; see `forgetStructuredAgentSession`. */ + forgetStatus: (sessionId: string) => void +} + +/** Dropping a session and dropping its status row are ONE operation: the store keeps the row until + * told, so a caller that only deletes strands a live-looking row no reader can ever decay. */ +export async function forgetStructuredAgentSession( + context: StructuredAgentSessionLifetimeContext, + sessionId: string +): Promise { + await context.sessions.get(sessionId)?.journal.close() + context.sessions.delete(sessionId) + context.forgetStatus(sessionId) } function hasProviderChild( @@ -50,10 +63,7 @@ export async function evictHeldStructuredAgentSession( hasProviderChild: hasProviderChild(context, sessionId), eventSink: context.runtimeState.eventSinkFor(sessionId), adapter: context.deps.adapter, - forget: async () => { - await context.sessions.get(sessionId)?.journal.close() - context.sessions.delete(sessionId) - }, + forget: () => forgetStructuredAgentSession(context, sessionId), discardSink: () => context.runtimeState.discardEventSink(sessionId), releaseLease: () => releaseStoredStructuredAgentSessionOwner({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts index 91e9ac91fa1..9d50adaaa1a 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts @@ -193,3 +193,25 @@ export async function settleStructuredAgentSessionLateDispatch( }) context.publish(input.sessionId, session.journal) } + +/** The host's thin mutation surface. Each call re-reads the context, so a session + * map or fence that moves between calls is never captured by a stale closure. */ +export function structuredAgentSessionMutationDelegates( + context: () => StructuredAgentSessionMutationContext +) { + return { + cancel: ( + caller: StructuredAgentSessionCaller, + params: Parameters[2] + ) => cancelStructuredAgentSessionTurn(context(), caller, params), + respondToPrompt: ( + caller: StructuredAgentSessionCaller, + params: Parameters[2] + ) => respondToStructuredAgentSessionPrompt(context(), caller, params), + setOption: ( + caller: StructuredAgentSessionCaller, + params: Parameters[2] + ) => setStructuredAgentSessionOption(context(), caller, params), + readOptions: (sessionId: string) => readStructuredAgentSessionOptions(context(), sessionId) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-test-harness.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-test-harness.ts new file mode 100644 index 00000000000..c67a8fabf58 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-test-harness.ts @@ -0,0 +1,197 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, expect, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionMutationEnvelope } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { journalDirectoryFor } from '../agent-session-journal/journal-paths' +import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' +import type { + AgentSessionDispatchOutcome, + StructuredAgentSessionAdapter +} from './structured-agent-session-adapter' +import type { AgentSessionAttachParams } from './structured-agent-session-attach' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' + +const journals = createTrackedJournalOpener() + +const CALLER = { callerKey: 'client-1' } + +function envelope( + method: string, + fields: Record, + overrides: Partial = {} +): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }), + ...overrides + } +} + +const attachParams = ( + overrides: Partial = {} +): AgentSessionAttachParams => hostTestAttachParams(null, overrides) + +const ensureParams = (fence: number): AgentSessionAttachParams => hostTestAttachParams(fence) + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let releaseAcquisition: Mock> +let dispatch: Mock +let cancelTurn: Mock +let answerPrompt: Mock +let setOption: Mock +let ordinal = 0 + +function accepted(): AgentSessionDispatchOutcome { + ordinal += 1 + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal } + } +} + +function adapter(): StructuredAgentSessionAdapter { + return { + acquire, + releaseAcquisition, + dispatch, + cancelTurn, + answerPrompt, + setOption + } +} + +async function attach(): Promise { + const result = await host.attach(CALLER, attachParams()) + expect(result.ok).toBe(true) + return store.getRecord(SESSION) +} + +/** Puts a pending approval in the journal BEFORE attach, which is the only way + * 1d can stage one: the adapter that would emit it is phase 2's. */ +async function seedApproval(optionId = 'allow'): Promise<{ itemId: string; revision: number }> { + const identity = { provider: 'codex' as const, threadId: THREAD, turnId: 'turn-1', ordinal: 99 } + const journalDir = journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir + }) + const appended = await journal.appendItem( + identity, + { + kind: 'approval', + title: 'Run the command?', + detail: null, + options: [{ id: optionId, label: 'Allow' }], + resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } + }, + { fence: 1 } + ) + return { itemId: appended.itemId, revision: appended.revision } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-wire-host-')) + resetHostTestOperationIds() + ordinal = 0 + acquire = vi.fn(async ({ fence }) => ({ + process: { + hostId: 'local', + pid: 4242, + processStartTimeMs: 1_700_000_000_000, + spawnToken: store.getRecord(SESSION)?.lease.reservedSpawnToken ?? 'spawn-a' + }, + link: { + linkId: `link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: store.getRecord(SESSION)?.providerHandleChain.length ? 'resumed' : 'created', + mintedAtFence: fence, + observedAt: NOW + } + })) + releaseAcquisition = vi.fn(async () => true) + dispatch = vi.fn(async () => accepted()) + cancelTurn = vi.fn(async () => ({ cancelled: true })) + answerPrompt = vi.fn(async () => undefined) + setOption = vi.fn(async () => undefined) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-a', + now: () => NOW + }) +}) + +afterEach(async () => { + await journals.closeAll() + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +/** A restarted process swaps the store and the host under the same directories. + * The helpers here close over both, so they have to be told. */ +export function replaceHostTestState(next: { + store: AgentSessionRecordStore + host: StructuredAgentSessionHost +}): void { + store = next.store + host = next.host +} + +/** The live per-test state. Read it in a `beforeEach` so a suite's test bodies + * keep using bare `host` / `store` / `dispatch` exactly as they did when this + * setup was inline. */ +export function hostTestState() { + return { + root, + store, + host, + acquire, + releaseAcquisition, + dispatch, + cancelTurn, + answerPrompt, + setOption + } +} + +export { + CALLER, + accepted, + adapter, + attach, + attachParams, + ensureParams, + envelope, + journals, + seedApproval +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-types.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-types.ts index 7a321668c46..50b913ee8bd 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-types.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-types.ts @@ -8,6 +8,7 @@ import type { AgentSessionJournal } from '../agent-session-journal/journal-store import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import type { AgentSessionAttachParams } from './structured-agent-session-attach' import type { StructuredAgentSessionHandoffTransport } from './structured-agent-session-handoff-types' +import type { StructuredAgentSessionStatusSink } from './structured-agent-session-status-feed' export type StructuredAgentSessionCaller = { callerKey: string } @@ -69,5 +70,9 @@ export type StructuredAgentSessionHostDeps = { summary: AgentSessionStatusSummary, options: { replay: boolean } ) => void + /** The agent-status store every held session's projection is written to and, on close, + * removed from. Both production hosts pass one — the desktop and headless `orcad`; absent, + * every reader of that store simply lists no structured session. */ + statusSink?: StructuredAgentSessionStatusSink handoffTransport?: StructuredAgentSessionHandoffTransport } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts index 5354670fa0b..d53c3c30e50 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.test.ts @@ -1,63 +1,31 @@ -import { mkdtemp, rm } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { beforeEach, describe, expect, it, vi, type Mock } from 'vitest' import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' import type { AgentSessionOwnerProbe } from '../../../shared/agent-session-lease-adjudication' -import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' import type { AgentSessionRecord } from '../../../shared/agent-session-record' -import type { - AgentSessionMutationEnvelope, - AgentSessionSubscribeEvent -} from '../../../shared/agent-session-wire' +import type { AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' +import { join } from 'node:path' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' -import { journalDirectoryFor } from '../agent-session-journal/journal-paths' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' -import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' -import type { - AgentSessionDispatchOutcome, - StructuredAgentSessionAdapter -} from './structured-agent-session-adapter' -import type { AgentSessionAttachParams } from './structured-agent-session-attach' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + adapter, + attach, + attachParams, + CALLER, + ensureParams, + envelope, + hostTestState, + replaceHostTestState, + seedApproval +} from './structured-agent-session-host-test-harness' import { HOST_TEST_NOW as NOW, HOST_TEST_SESSION as SESSION, HOST_TEST_THREAD as THREAD, - hostTestAttachParams, - hostTestMessage, - hostTestOperationId, - resetHostTestOperationIds + hostTestMessage } from './structured-agent-session-host-test-data' -const journals = createTrackedJournalOpener() - -const CALLER = { callerKey: 'client-1' } - -function envelope( - method: string, - fields: Record, - overrides: Partial = {} -): AgentSessionMutationEnvelope { - return { - sessionId: SESSION, - clientOperationId: hostTestOperationId(), - expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1, - payloadFingerprint: computeAgentSessionPayloadFingerprint({ - method, - sessionId: SESSION, - fields - }), - ...overrides - } -} - -const attachParams = ( - overrides: Partial = {} -): AgentSessionAttachParams => hostTestAttachParams(null, overrides) - -const ensureParams = (fence: number): AgentSessionAttachParams => hostTestAttachParams(fence) - let root: string let store: AgentSessionRecordStore let host: StructuredAgentSessionHost @@ -67,101 +35,19 @@ let dispatch: Mock let cancelTurn: Mock let answerPrompt: Mock let setOption: Mock -let ordinal = 0 -function accepted(): AgentSessionDispatchOutcome { - ordinal += 1 - return { - state: 'accepted', - providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal } - } -} - -function adapter(): StructuredAgentSessionAdapter { - return { +beforeEach(() => { + ;({ + root, + store, + host, acquire, releaseAcquisition, dispatch, cancelTurn, answerPrompt, setOption - } -} - -async function attach(): Promise { - const result = await host.attach(CALLER, attachParams()) - expect(result.ok).toBe(true) - return store.getRecord(SESSION) -} - -/** Puts a pending approval in the journal BEFORE attach, which is the only way - * 1d can stage one: the adapter that would emit it is phase 2's. */ -async function seedApproval(optionId = 'allow'): Promise<{ itemId: string; revision: number }> { - const identity = { provider: 'codex' as const, threadId: THREAD, turnId: 'turn-1', ordinal: 99 } - const journalDir = journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) - const journal = await journals.open({ - identity: { - sessionId: SESSION, - workspaceId: 'workspace-1', - hostId: 'local', - agent: 'codex', - providerHandle: { kind: 'codex', threadId: THREAD } - }, - journalDir - }) - const appended = await journal.appendItem( - identity, - { - kind: 'approval', - title: 'Run the command?', - detail: null, - options: [{ id: optionId, label: 'Allow' }], - resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } - }, - { fence: 1 } - ) - return { itemId: appended.itemId, revision: appended.revision } -} - -beforeEach(async () => { - root = await mkdtemp(join(tmpdir(), 'orca-wire-host-')) - resetHostTestOperationIds() - ordinal = 0 - acquire = vi.fn(async ({ fence }) => ({ - process: { - hostId: 'local', - pid: 4242, - processStartTimeMs: 1_700_000_000_000, - spawnToken: store.getRecord(SESSION)?.lease.reservedSpawnToken ?? 'spawn-a' - }, - link: { - linkId: `link-${fence}`, - handle: { provider: 'codex', threadId: THREAD }, - origin: store.getRecord(SESSION)?.providerHandleChain.length ? 'resumed' : 'created', - mintedAtFence: fence, - observedAt: NOW - } - })) - releaseAcquisition = vi.fn(async () => true) - dispatch = vi.fn(async () => accepted()) - cancelTurn = vi.fn(async () => ({ cancelled: true })) - answerPrompt = vi.fn(async () => undefined) - setOption = vi.fn(async () => undefined) - store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) - host = new StructuredAgentSessionHost({ - store, - adapter: adapter(), - journalRoot: root, - claimKeyId: 'key-1', - mintSpawnToken: () => 'spawn-a', - now: () => NOW - }) -}) - -afterEach(async () => { - await journals.closeAll() - await host.flushAllStreamedEvents() - await rm(root, { recursive: true, force: true }) + } = hostTestState()) }) describe('attach', () => { @@ -308,150 +194,6 @@ describe('attach', () => { }) }) -describe('send', () => { - it('writes the submission before dispatching and resolves it accepted', async () => { - await attach() - const body = hostTestMessage('add a retry') - const result = await host.send(CALLER, { - envelope: envelope('agentSession.send', { body }), - body - }) - if (!result.ok) { - throw new Error(`expected a send, got ${result.refusal.code}`) - } - expect(result.value.submission.dispatchState).toBe('accepted') - expect(dispatch).toHaveBeenCalledTimes(1) - const page = host.history({ sessionId: SESSION, direction: 'tail' }) - expect(page.ok && page.page.items).toHaveLength(1) - expect(page.ok && page.page.fence).toBe(1) - expect(page.providerSession).toEqual({ key: 'session_id', id: THREAD }) - }) - - it('settles a thrown dispatch as unknown, never as a rejection', async () => { - await attach() - dispatch.mockRejectedValueOnce(new Error('socket closed')) - const body = hostTestMessage('add a retry') - const result = await host.send(CALLER, { - envelope: envelope('agentSession.send', { body }), - body - }) - expect(result).toMatchObject({ ok: true, value: { submission: { dispatchState: 'unknown' } } }) - }) - - it('replays a retried send from the journal without dispatching twice', async () => { - await attach() - const body = hostTestMessage('add a retry') - const params = { envelope: envelope('agentSession.send', { body }), body } - await host.send(CALLER, params) - const retry = await host.send(CALLER, params) - expect(retry).toMatchObject({ ok: true, replayed: true }) - expect(dispatch).toHaveBeenCalledTimes(1) - }) - - it('redispatches an explicitly retried durable unknown without appending a second submission', async () => { - await attach() - dispatch - .mockRejectedValueOnce(new Error('socket closed')) - .mockImplementationOnce(async () => accepted()) - const body = hostTestMessage('possibly delivered') - const params = { envelope: envelope('agentSession.send', { body }), body } - - const first = await host.send(CALLER, params) - expect(first).toMatchObject({ - ok: true, - value: { submission: { dispatchState: 'unknown' } } - }) - const retried = await host.send(CALLER, { ...params, retryUnknown: true }) - - expect(retried).toMatchObject({ - ok: true, - replayed: false, - value: { submission: { dispatchState: 'accepted' } } - }) - expect(dispatch).toHaveBeenCalledTimes(2) - const state = host.history({ sessionId: SESSION, direction: 'tail' }) - expect(state.ok && state.page.submissions).toHaveLength(1) - }) - - it('advances an explicit retry after a ledger-unknown send is reconciled in the journal', async () => { - await attach() - const journal = ( - host as unknown as { sessions: Map } - ).sessions.get(SESSION)!.journal - vi.spyOn(journal, 'resolveDispatch').mockRejectedValueOnce(new Error('journal resolve failed')) - const body = hostTestMessage('possibly delivered before persistence failed') - const params = { envelope: envelope('agentSession.send', { body }), body } - - await expect(host.send(CALLER, params)).rejects.toThrow('journal resolve failed') - expect(journal.submissions()).toMatchObject([ - { clientMessageId: params.envelope.clientOperationId, dispatchState: 'unknown' } - ]) - expect( - store.listOperationRows().find((row) => row.operationId === params.envelope.clientOperationId) - ?.outcome - ).toEqual({ status: 'unknown' }) - expect(dispatch).toHaveBeenCalledTimes(1) - - await journal.markPendingSubmissionsUnknown(store.getRecord(SESSION)?.lease.runtimeFence ?? 1) - await expect(host.send(CALLER, params)).resolves.toMatchObject({ - ok: false, - refusal: { code: 'agent_session_operation_unknown' } - }) - expect(dispatch).toHaveBeenCalledTimes(1) - - await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ - ok: true, - replayed: false, - value: { submission: { dispatchState: 'accepted' } } - }) - expect(dispatch).toHaveBeenCalledTimes(2) - expect(journal.submissions()).toHaveLength(1) - }) - - it('refuses a stale fence and hands back the current one', async () => { - const record = await attach() - const body = hostTestMessage('add a retry') - const result = await host.send(CALLER, { - envelope: envelope( - 'agentSession.send', - { body }, - { expectedRuntimeFence: (record?.lease.runtimeFence ?? 1) + 5 } - ), - body - }) - expect(result).toMatchObject({ - ok: false, - refusal: { code: 'agent_session_checkpoint_stale', currentFence: record?.lease.runtimeFence } - }) - }) - - it('does not let a refused call leave a ledger row that replays past the fence', async () => { - const record = await attach() - const body = hostTestMessage('add a retry') - const params = { - envelope: envelope( - 'agentSession.send', - { body }, - { expectedRuntimeFence: (record?.lease.runtimeFence ?? 1) + 5 } - ), - body - } - await host.send(CALLER, params) - expect(await host.send(CALLER, params)).toMatchObject({ - ok: false, - refusal: { code: 'agent_session_checkpoint_stale' } - }) - expect(dispatch).not.toHaveBeenCalled() - }) - - it('refuses any mutation against a session this host has not attached', async () => { - const body = hostTestMessage('add a retry') - expect( - await host.send(CALLER, { envelope: envelope('agentSession.send', { body }), body }) - ).toMatchObject({ ok: false, refusal: { code: 'agent_session_ownership_unknown' } }) - }) -}) - describe('cancel', () => { it('records the request acknowledgement as a status item keyed by the operation id', async () => { await attach() @@ -663,6 +405,7 @@ describe('restart', () => { probeOwner, now: () => NOW }) + replaceHostTestState({ store, host }) } /** The refusal a restarted host owes a client holding the dead generation's diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 2df2fdf5412..176cd66b674 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -10,10 +10,7 @@ import type * as SessionWire from '../../../shared/agent-session-wire' import type { AgentSessionAttachParams } from './structured-agent-session-attach' import { AGENT_SESSION_NOT_ATTACHED } from './structured-agent-session-mutation-admission' import { createRestartReconciler } from './structured-agent-session-restart-reconcile' -import { - AgentSessionSubscribers, - type AgentSessionSubscribeInput -} from './structured-agent-session-subscribers' +import type { AgentSessionSubscribeInput } from './structured-agent-session-subscribers' import { StructuredAgentSessionTaskQueue } from './structured-agent-session-task-queue' import * as providerSupport from './structured-agent-session-provider-support' import { createStructuredAgentSessionHostRestore } from './structured-agent-session-reveal' @@ -36,10 +33,7 @@ import type { import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' import { listStructuredAgentSessionTabs } from './structured-agent-session-host-tabs' import { - cancelStructuredAgentSessionTurn, - readStructuredAgentSessionOptions, - respondToStructuredAgentSessionPrompt, - setStructuredAgentSessionOption, + structuredAgentSessionMutationDelegates, settleStructuredAgentSessionLateDispatch, type StructuredAgentSessionMutationContext } from './structured-agent-session-host-mutations' @@ -53,9 +47,10 @@ import type { StructuredAgentSessionHostSession, StructuredAgentSessionReveal } from './structured-agent-session-host-types' -import { StructuredAgentSessionStatusFeed } from './structured-agent-session-status-feed' +import type { StructuredAgentSessionStatusSubscriber } from './structured-agent-session-status-feed' import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery' import { StructuredAgentSessionBackgroundTaskChannel } from './structured-agent-session-background-task-channel' +import { StructuredAgentSessionClientDelivery } from './structured-agent-session-client-delivery' export type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' export class StructuredAgentSessionHost { @@ -64,16 +59,12 @@ export class StructuredAgentSessionHost { this ) private readonly sessions = new Map() - private readonly statusFeed = new StructuredAgentSessionStatusFeed({ - sessions: this.sessions, - getRecord: (sessionId) => this.deps.store.getRecord(sessionId), - now: () => this.now(), - onStatusChanged: (summary, options) => this.deps.onSessionStatusChanged?.(summary, options) - }) - private readonly subscribers = new AgentSessionSubscribers({ - readCommands: (sessionId) => this.deps.adapter.readCommands?.(sessionId), - onJournalPublished: (sessionId, journal) => this.statusFeed.publish(sessionId, journal) - }) + private readonly clientDelivery = new StructuredAgentSessionClientDelivery( + this.sessions, + () => this.now(), + () => this.deps + ) + private readonly subscribers = this.clientDelivery.subscribers private readonly tasks = new StructuredAgentSessionTaskQueue() private readonly runtimeState: StructuredAgentSessionHostRuntimeState private readonly reconcileLeases: ( @@ -91,7 +82,8 @@ export class StructuredAgentSessionHost { this.sessions, this.subscribers, (sessionId) => this.requireSession(sessionId), - (sessionId) => this.handoffs.status(sessionId) + (sessionId) => this.handoffs.status(sessionId), + this.clientDelivery.publishStatus ) this.runtimeState = new StructuredAgentSessionHostRuntimeState( deps, @@ -117,7 +109,7 @@ export class StructuredAgentSessionHost { flush: (sessionId) => this.flushStreamedEvents(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), subscribers: this.subscribers, - publishStatus: (sessionId) => this.statusFeed.publish(sessionId), + publishStatus: this.clientDelivery.publishStatus, now: this.now }) this.holds = createStructuredAgentSessionHolds(this.lifetimeContext(), { @@ -134,7 +126,7 @@ export class StructuredAgentSessionHost { // `hasSession` inside the same serialized step as this `set`. onReadable: (sessionId, restored) => { this.sessions.set(sessionId, restored) - this.statusFeed.publish(sessionId, undefined, { replay: true }) + this.clientDelivery.publishRestored(sessionId) }, restoreHandoff: (sessionId) => this.handoffs.restore(sessionId) }) @@ -145,7 +137,7 @@ export class StructuredAgentSessionHost { flushLifecycle: (sessionId) => this.runtimeState.lifecycleBarrier(sessionId), publishFence: (sessionId, session) => this.subscribers.snapshot(sessionId, session.journal, session.fence), - publishStatus: (sessionId) => this.statusFeed.publish(sessionId), + publishStatus: this.clientDelivery.publishStatusAndSettlement, hasResumeCapableHolder: (sessionId) => this.holds.hasResumeCapableHolder(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), now: () => this.now(), @@ -179,7 +171,8 @@ export class StructuredAgentSessionHost { deps: this.deps, runtimeState: this.runtimeState, sessions: this.sessions, - now: () => this.now() + now: () => this.now(), + forgetStatus: this.clientDelivery.forgetStatus } } @@ -191,7 +184,7 @@ export class StructuredAgentSessionHost { tasks: this.tasks, reconcileLeases: (sessionId) => this.reconcileLeases(sessionId), serialize: (sessionId, task) => this.serialize(sessionId, task), - publishStatus: (sessionId) => this.statusFeed.publish(sessionId) + publishStatus: this.clientDelivery.publishStatus } } /** Releases a session's resources without ending the conversation: the record and journal stay @@ -200,7 +193,7 @@ export class StructuredAgentSessionHost { return this.serialize(sessionId, async () => { await this.handoffs.closeRetainedTuiOwner(sessionId) await evictHeldStructuredAgentSession(this.lifetimeContext(), sessionId) - this.statusFeed.revokeLive(sessionId) + this.clientDelivery.closeSession(sessionId) // Whoever asked for the close, the surfaces that were holding this session are looking at a // session that no longer exists. A failed eviction throws above and keeps them. this.holds.forget(sessionId) @@ -211,11 +204,6 @@ export class StructuredAgentSessionHost { providerSupport.adapterSupportsCreate(this.deps.adapter, location, agent) listSessionTabs = () => listStructuredAgentSessionTabs(this.sessions) - - /** Last projected status for every structured session this host still holds, for non-subscribing - * readers. The retained projections of forgotten sessions are deliberately not included. */ - readonly liveSessionStatusSummaries = () => this.statusFeed.liveSessionSummaries() - getPersistedVisibleSessionTabIndex = () => this.deps.store.getVisibleSessionTabIndex() setSessionTabVisibility = (sessionId: string, visible: boolean): Promise => @@ -264,7 +252,7 @@ export class StructuredAgentSessionHost { tasks: this.tasks }), sessions: this.sessions - }) + }).finally(() => this.clientDelivery.closeAll()) } private mutationContext(): StructuredAgentSessionMutationContext { @@ -281,23 +269,13 @@ export class StructuredAgentSessionHost { send = (...args: Parameters) => this.conversationCommands.send(...args) - cancel = ( - caller: StructuredAgentSessionCaller, - params: Parameters[2] - ): ReturnType => - cancelStructuredAgentSessionTurn(this.mutationContext(), caller, params) + waitForSendSettlement = this.clientDelivery.waitForSendSettlement - respondToPrompt = ( - caller: StructuredAgentSessionCaller, - params: Parameters[2] - ): ReturnType => - respondToStructuredAgentSessionPrompt(this.mutationContext(), caller, params) - - setOption = ( - caller: StructuredAgentSessionCaller, - params: Parameters[2] - ): ReturnType => - setStructuredAgentSessionOption(this.mutationContext(), caller, params) + private mutations = structuredAgentSessionMutationDelegates(() => this.mutationContext()) + cancel = this.mutations.cancel + respondToPrompt = this.mutations.respondToPrompt + setOption = this.mutations.setOption + readOptions = this.mutations.readOptions requestHandoff = ( caller: StructuredAgentSessionCaller, @@ -305,9 +283,6 @@ export class StructuredAgentSessionHost { ): Promise> => this.handoffs.request(caller.callerKey, params) - readOptions = (sessionId: string): Promise => - readStructuredAgentSessionOptions(this.mutationContext(), sessionId) - rewind = (caller: StructuredAgentSessionCaller, params: AgentSessionRewindParams) => rewindStructuredAgentSession(this.mutationContext(), this.attachContext(), caller, params) @@ -329,8 +304,8 @@ export class StructuredAgentSessionHost { history: StructuredAgentSessionBackgroundTaskChannel['history'] = (request) => this.backgroundTasks.history(request) - /** The fully reduced timeline, for readers that cannot tolerate a page's ambiguity — a settled - * turn is tombstoned, so an item's ABSENCE from a bounded page proves nothing. */ + /** The fully reduced timeline, for readers that cannot tolerate a page's ambiguity — rows are + * revised or tombstoned in place, so an item's ABSENCE from a bounded page proves nothing. */ journalSnapshot = (sessionId: string): AgentJournalSnapshot => this.requireSession(sessionId).journal.snapshot() @@ -345,8 +320,8 @@ export class StructuredAgentSessionHost { unsubscribe = (sessionId: string, id: string): void => this.subscribers.close(sessionId, id) /** Every session's projected status for session lists; unlike `subscribe`, retains nothing. */ - subscribeStatus: StructuredAgentSessionStatusFeed['subscribe'] = (subscriber) => - this.statusFeed.subscribe(subscriber) + subscribeStatus = (subscriber: StructuredAgentSessionStatusSubscriber): (() => void) => + this.clientDelivery.subscribeStatus(subscriber) private requireSession(sessionId: string): StructuredAgentSessionHostSession { const session = this.sessions.get(sessionId) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-late-settlement.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-late-settlement.test.ts index a75d2ea6512..293f6ab2d8c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-late-settlement.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-late-settlement.test.ts @@ -8,6 +8,7 @@ import type { AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { AgentSessionDispatchOutcome, StructuredAgentSessionAdapter @@ -63,6 +64,12 @@ function submissions(): unknown { return state.ok ? state.page.submissions : null } +function journal(): AgentSessionJournal { + return ( + host as unknown as { sessions: Map } + ).sessions.get(SESSION)!.journal +} + beforeEach(async () => { root = await mkdtemp(join(tmpdir(), 'orca-wire-late-settle-')) resetHostTestOperationIds() @@ -197,6 +204,36 @@ describe('settling a send the provider proves it received after the ack window', expect(dispatch).toHaveBeenCalledTimes(1) }) + it('accepts from the durable echo row when the direct settlement write fails', async () => { + dispatch.mockResolvedValueOnce({ state: 'admitted' }) + const params = sendParams('settle from provider echo') + await host.send(CALLER, params) + vi.spyOn(journal(), 'resolveDispatch').mockRejectedValueOnce( + new Error('direct settlement write failed') + ) + + await expect( + host.settleLateDispatch({ + sessionId: SESSION, + clientMessageId: params.envelope.clientOperationId, + providerIdentity: { provider: 'claude', sessionId: THREAD, uuid: 'echo-row' } + }) + ).rejects.toThrow('direct settlement write failed') + await journal().appendItem( + { provider: 'claude', sessionId: THREAD, uuid: 'echo-row' }, + params.body, + { fence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1 } + ) + + expect(submissions()).toMatchObject([ + { + clientMessageId: params.envelope.clientOperationId, + dispatchState: 'accepted', + providerItemId: `claude:${THREAD}:echo-row` + } + ]) + }) + it('leaves an already accepted send alone', async () => { const params = sendParams('ordinary send') await host.send(CALLER, params) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-lease-release.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-lease-release.ts index 2db0fb0ff81..8d6cfc39ae2 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-lease-release.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-lease-release.ts @@ -40,6 +40,7 @@ export async function releaseStoredStructuredAgentSessionOwnerAfterUnexpectedExi expectedAcquisitionGeneration: string acquisitionGeneration: string | null now: number + exitObservedAt?: number settlementRetry?: { settlementId: string; detail: string } }): Promise { if (input.acquisitionGeneration !== input.expectedAcquisitionGeneration) { @@ -57,6 +58,7 @@ export async function releaseStoredStructuredAgentSessionOwnerAfterUnexpectedExi sessionId: input.sessionId, expectedFence: input.expectedFence, now: input.now, + exitObservedAt: input.exitObservedAt, ...(input.settlementRetry ? { settlementRetry: input.settlementRetry } : {}) }) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-admission.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-admission.ts index ac6dc23385a..4d0e0faab30 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-admission.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-admission.ts @@ -13,6 +13,7 @@ import type { AgentSessionMutationResult, AgentSessionWireRefusal } from '../../../shared/agent-session-wire' +import { AGENT_SESSION_UNATTACHED_REFUSAL_CODE } from '../../../shared/structured-agent-session-read-refusal' import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' @@ -21,8 +22,10 @@ import { runSettledAgentSessionMutation } from './structured-agent-session-opera import { resolveAgentSessionReplayOutcome } from './structured-agent-session-replay-outcome' import type { AgentSessionTurnContext } from './structured-agent-session-turns' +// The code is shared with the client so a read that refuses this way can be told apart from a +// transcript that failed to load; the two must never drift apart. export const AGENT_SESSION_NOT_ATTACHED: AgentSessionWireRefusal = { - code: 'agent_session_ownership_unknown', + code: AGENT_SESSION_UNATTACHED_REFUSAL_CODE, message: 'This host holds no attached session by that id.' } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.test.ts index 554b8d34395..c75301c6461 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.test.ts @@ -420,6 +420,56 @@ describe('host rewind', () => { expect(await host.rewind(caller, params(target))).toMatchObject({ ok: true }) }) + it('keeps the host-stamped turn rows before the boundary through a Codex provider hydration', async () => { + expect(await host.attach(caller, hostTestAttachParams(null))).toMatchObject({ ok: true }) + const message = (turnId: string) => ({ + provider: 'codex' as const, + threadId: HOST_TEST_THREAD, + turnId, + ordinal: 0 + }) + const turnRow = (turnId: string) => ({ + provider: 'legacy' as const, + agent: 'codex', + sessionId: HOST_TEST_SESSION, + recordId: `turn-lifecycle:${turnId}` + }) + const keptTurn = { + kind: 'turn' as const, + turnId: 'kept', + state: 'completed' as const, + userItemId: agentJournalItemKey(message('kept')), + startedAt: HOST_TEST_NOW - 9_000, + completedAt: HOST_TEST_NOW - 4_000, + durationMs: 5_000 + } + sink.appendItem(message('kept'), hostTestMessage('kept')) + sink.appendItem(turnRow('kept'), keptTurn) + sink.appendItem(message('drop'), hostTestMessage('drop')) + sink.appendItem(turnRow('drop'), { ...keptTurn, turnId: 'drop', durationMs: 1_000 }) + sink.appendItem(message('tip'), { ...hostTestMessage('tip'), role: 'assistant' }) + await host.flushStreamedEvents(HOST_TEST_SESSION) + // The provider preflight knows only its own items, never the host's turn rows. + const items = [{ identity: message('kept'), body: hostTestMessage('kept from provider') }] + rewind.mockImplementationOnce(async (input) => { + await input.onPrepared?.(items) + await input.onReverted?.() + return { ok: true, items } + }) + + expect(await host.rewind(caller, params(agentJournalItemKey(message('drop'))))).toMatchObject({ + ok: true + }) + + expect( + host.journalSnapshot(HOST_TEST_SESSION).items.map(({ itemId, body }) => ({ itemId, body })) + ).toEqual([ + { itemId: agentJournalItemKey(message('kept')), body: hostTestMessage('kept from provider') }, + { itemId: agentJournalItemKey(turnRow('kept')), body: keptTurn } + ]) + expect(store.getRecord(HOST_TEST_SESSION)?.rewind?.phase).toBe('completed') + }) + it('recovers against the complete provider preflight when the local journal omitted an older turn', async () => { const target = await seed() const items = ['older', 'kept'].map((turnId) => ({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.ts index 053707b9206..417a42fe414 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-rewind.ts @@ -1,3 +1,4 @@ +import { readAgentJournalTurn } from '../../../shared/agent-session-turn-record' import { agentJournalItemKey, agentJournalSubmissionKey, @@ -19,6 +20,7 @@ import { conversationCommandBlocked } from './structured-conversation-command-ad import { rewindRefusal } from './structured-rewind-refusal' import { persistRewindRecord, recoverStructuredRewind } from './structured-rewind-recovery' import { replaceClaudeRewindOwner } from './structured-rewind-claude-owner' +import { mergeRetainedTurnRows } from './structured-rewind-retained-turns' export async function rewindStructuredAgentSession( context: StructuredAgentSessionMutationContext, @@ -108,7 +110,7 @@ export async function rewindStructuredAgentSession( (identity?.provider === 'codex' && identity.threadId === key.threadId && identity.turnId === key.turnId) || - (item.body.kind === 'status' && item.body.turnLifecycle?.turnId === key.turnId) + readAgentJournalTurn(item.body)?.turnId === key.turnId ) }) } else if (key.provider === 'claude' && head.provider === 'claude') { @@ -172,11 +174,14 @@ export async function rewindStructuredAgentSession( fence: ctx.fence, beforeTurnId: key.provider === 'codex' ? key.turnId : '', onPrepared: async (items) => { - const retained = items.map(({ identity, body }) => ({ - itemId: agentJournalItemKey(identity), - body, - observedAt: ctx.now() - })) + const retained = mergeRetainedTurnRows( + prepared.retained, + items.map(({ identity, body }) => ({ + itemId: agentJournalItemKey(identity), + body, + observedAt: ctx.now() + })) + ) if ( retained.length > 10_000 || Buffer.byteLength(JSON.stringify(retained), 'utf8') > @@ -215,11 +220,14 @@ export async function rewindStructuredAgentSession( return rewindRefusal(reason) } const confirmed = provider.items - ? provider.items.map(({ identity, body }) => ({ - itemId: agentJournalItemKey(identity), - body, - observedAt: ctx.now() - })) + ? mergeRetainedTurnRows( + prepared.retained, + provider.items.map(({ identity, body }) => ({ + itemId: agentJournalItemKey(identity), + body, + observedAt: ctx.now() + })) + ) : prepared.retained if ( Buffer.byteLength(JSON.stringify(confirmed), 'utf8') > diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts index 581743633c0..14df2b3df6a 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-send-idempotency.test.ts @@ -3,6 +3,7 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' +import { hasUnansweredStructuredAgentSessionDispatch } from '../../../shared/structured-agent-session-projection' import { structuredAgentSessionPayloadFingerprint } from '../../../shared/structured-agent-session-mutation' import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' @@ -34,6 +35,39 @@ afterEach(async () => { }) describe('structured send idempotency', () => { + it('publishes a recovered retry as working before waiting for its provider', async () => { + const body: AgentJournalMessageItem = { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'retry' }] + } + const input = { clientMessageId: 'retry-id', payloadFingerprint: 'fingerprint', body } + await journal.appendSubmission({ ...input, fence: 1 }) + await journal.markPendingSubmissionsUnknown(2, 'provider_write_failed: broken pipe') + const originalItem = journal.snapshot().items[0] + const publish = vi.fn() + const dispatch = vi.fn(async () => { + expect(publish).toHaveBeenCalledOnce() + expect(hasUnansweredStructuredAgentSessionDispatch(journal.submissions(), 2)).toBe(true) + return { state: 'unknown' as const, reason: 'ack timeout' } + }) + await performSend( + { + sessionId: 'session-1', + journal, + fence: 2, + adapter: { dispatch } as unknown as StructuredAgentSessionAdapter, + persistOptions: async () => undefined, + resolvedBy: 'caller', + publish, + now: () => 1 + }, + { ...input, retryUnknown: true } + ) + expect(hasUnansweredStructuredAgentSessionDispatch(journal.submissions(), 2)).toBe(true) + expect(journal.snapshot().items).toEqual([originalItem]) + }) + it('does not redispatch one send id reused across caller ledgers', async () => { const body: AgentJournalMessageItem = { kind: 'message', diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-send-settlement.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-send-settlement.test.ts new file mode 100644 index 00000000000..b461d508f42 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-send-settlement.test.ts @@ -0,0 +1,133 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { StructuredAgentSessionSendSettlement } from './structured-agent-session-send-settlement' + +function journal(dispatchState: 'pending' | 'accepted' | 'unknown'): AgentSessionJournal { + return { + cursor: () => ({ epoch: 'epoch-1', sequence: dispatchState === 'pending' ? 1 : 2 }), + submissions: () => [ + { + clientMessageId: 'client-1', + fence: 1, + payloadFingerprint: 'fingerprint', + dispatchState, + providerItemId: dispatchState === 'accepted' ? 'provider-1' : null, + reason: dispatchState === 'unknown' ? 'provider exited' : null, + submittedAt: 1, + resolvedAt: dispatchState === 'pending' ? null : 2 + } + ] + } as AgentSessionJournal +} + +function emptyJournal(): AgentSessionJournal { + return { + cursor: () => ({ epoch: 'epoch-1', sequence: 2 }), + submissions: () => [] + } as unknown as AgentSessionJournal +} + +describe('structured send settlement compatibility wait', () => { + afterEach(() => vi.useRealTimers()) + + it('returns a settlement already present in the journal', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('accepted')) + + await expect(settlements.wait('session-1', 'client-1')).resolves.toMatchObject({ + value: { submission: { dispatchState: 'accepted' } } + }) + }) + + it('rejects when the send is absent from the current session generation', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => emptyJournal()) + + await expect(settlements.wait('session-1', 'client-1')).rejects.toThrow( + 'agent session send disappeared before settlement' + ) + }) + + it('resolves from a journal publication after durable admission', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const pending = settlements.wait('session-1', 'client-1') + + settlements.publish('session-1', journal('accepted')) + + await expect(pending).resolves.toMatchObject({ + cursor: { sequence: 2 }, + value: { submission: { dispatchState: 'accepted' } } + }) + }) + + it('removes an abandoned wait on transport cancellation', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const controller = new AbortController() + const pending = settlements.wait('session-1', 'client-1', controller.signal) + + controller.abort(new Error('transport closed')) + await expect(pending).rejects.toThrow('transport closed') + settlements.publish('session-1', journal('accepted')) + }) + + it('expires only the compatibility observer when the client leaves its socket open', async () => { + vi.useFakeTimers() + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const pending = settlements.wait('session-1', 'client-1') + + await vi.advanceTimersByTimeAsync(30_000) + + await expect(pending).resolves.toBeUndefined() + settlements.publish('session-1', journal('accepted')) + }) + + it('caps compatibility observers retained for one session', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const retained = Array.from({ length: 64 }, () => + settlements.wait('session-1', 'client-1').catch(() => undefined) + ) + + await expect(settlements.wait('session-1', 'client-1')).resolves.toBeUndefined() + settlements.closeAll() + await Promise.all(retained) + }) + + it('caps compatibility observers retained across sessions', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const retained = Array.from({ length: 1_024 }, (_, index) => + settlements.wait(`session-${index}`, 'client-1').catch(() => undefined) + ) + + await expect(settlements.wait('session-overflow', 'client-1')).resolves.toBeUndefined() + settlements.closeAll() + await Promise.all(retained) + }) + + it('ends only the compatibility observation when the session closes', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const pending = settlements.wait('session-1', 'client-1') + + settlements.closeSession('session-1') + + await expect(pending).resolves.toBeUndefined() + settlements.publish('session-1', journal('accepted')) + }) + + it('rejects a wait when an authoritative publication drops the submission', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const pending = settlements.wait('session-1', 'client-1') + + settlements.publish('session-1', emptyJournal()) + + await expect(pending).rejects.toThrow('agent session send disappeared before settlement') + }) + + it('ends every compatibility observation when the host closes', async () => { + const settlements = new StructuredAgentSessionSendSettlement(() => journal('pending')) + const first = settlements.wait('session-1', 'client-1') + const second = settlements.wait('session-2', 'client-1') + + settlements.closeAll() + + await expect(first).resolves.toBeUndefined() + await expect(second).resolves.toBeUndefined() + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-send-settlement.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-send-settlement.ts new file mode 100644 index 00000000000..6150e3412a6 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-send-settlement.ts @@ -0,0 +1,164 @@ +import type { + AgentJournalCursor, + AgentJournalSubmission +} from '../../../shared/agent-session-journal-types' +import type { AgentSessionSendResult } from '../../../shared/agent-session-wire' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' + +type SettledSend = { + cursor: AgentJournalCursor + value: AgentSessionSendResult +} + +type SendSettlement = SettledSend | 'pending' | 'missing' + +type SendSettlementWaiter = { + clientMessageId: string + resolve: (result: SettledSend | undefined) => void + reject: (error: Error) => void + timer: ReturnType + signal?: AbortSignal + onAbort?: () => void +} + +// Known legacy clients abandon the RPC after 15s without cancelling its socket dispatch. +const SEND_SETTLEMENT_WAIT_TIMEOUT_MS = 30_000 +const MAX_SEND_SETTLEMENT_WAITERS_PER_SESSION = 64 +const MAX_SEND_SETTLEMENT_WAITERS = 1_024 + +function settledSend( + journal: AgentSessionJournal, + clientMessageId: string, + submission: AgentJournalSubmission | undefined = journal + .submissions() + .find((candidate) => candidate.clientMessageId === clientMessageId) +): SendSettlement { + if (!submission) { + return 'missing' + } + return submission.dispatchState === 'pending' + ? 'pending' + : { cursor: journal.cursor(), value: { clientMessageId, submission } } +} + +function abortError(signal: AbortSignal): Error { + return signal.reason instanceof Error + ? signal.reason + : new Error('agent session send settlement wait aborted') +} + +/** Best-effort settlement observation for clients that predate admitted pending replies. */ +export class StructuredAgentSessionSendSettlement { + private readonly waiters = new Map>() + private waiterCount = 0 + + constructor(private readonly journalFor: (sessionId: string) => AgentSessionJournal) {} + + wait = ( + sessionId: string, + clientMessageId: string, + signal?: AbortSignal + ): Promise => { + if (signal?.aborted) { + return Promise.reject(abortError(signal)) + } + const immediate = settledSend(this.journalFor(sessionId), clientMessageId) + if (immediate === 'missing') { + return Promise.reject(new Error('agent session send disappeared before settlement')) + } + if (immediate !== 'pending') { + return Promise.resolve(immediate) + } + const existingSession = this.waiters.get(sessionId) + if ( + this.waiterCount >= MAX_SEND_SETTLEMENT_WAITERS || + (existingSession?.size ?? 0) >= MAX_SEND_SETTLEMENT_WAITERS_PER_SESSION + ) { + return Promise.resolve(undefined) + } + return new Promise((resolve, reject) => { + const waiter: SendSettlementWaiter = { + clientMessageId, + resolve, + reject, + timer: setTimeout(() => { + this.remove(sessionId, waiter) + resolve(undefined) + }, SEND_SETTLEMENT_WAIT_TIMEOUT_MS) + } + waiter.timer.unref?.() + const session = existingSession ?? new Set() + session.add(waiter) + this.waiters.set(sessionId, session) + this.waiterCount += 1 + if (signal) { + const onAbort = (): void => { + this.remove(sessionId, waiter) + reject(abortError(signal)) + } + waiter.signal = signal + waiter.onAbort = onAbort + signal.addEventListener('abort', onAbort, { once: true }) + if (signal.aborted) { + onAbort() + } + } + }) + } + + publish(sessionId: string, journal: AgentSessionJournal): void { + const waiters = this.waiters.get(sessionId) + if (!waiters) { + return + } + const submissions = new Map( + journal.submissions().map((submission) => [submission.clientMessageId, submission]) + ) + for (const waiter of waiters) { + const result = settledSend( + journal, + waiter.clientMessageId, + submissions.get(waiter.clientMessageId) + ) + if (result !== 'pending') { + this.remove(sessionId, waiter) + if (result === 'missing') { + waiter.reject(new Error('agent session send disappeared before settlement')) + } else { + waiter.resolve(result) + } + } + } + } + + closeSession(sessionId: string): void { + const waiters = this.waiters.get(sessionId) + if (!waiters) { + return + } + for (const waiter of waiters) { + this.remove(sessionId, waiter) + waiter.resolve(undefined) + } + } + + closeAll(): void { + for (const sessionId of this.waiters.keys()) { + this.closeSession(sessionId) + } + } + + private remove(sessionId: string, waiter: SendSettlementWaiter): void { + clearTimeout(waiter.timer) + if (waiter.signal && waiter.onAbort) { + waiter.signal.removeEventListener('abort', waiter.onAbort) + } + const session = this.waiters.get(sessionId) + if (session?.delete(waiter)) { + this.waiterCount -= 1 + } + if (session?.size === 0) { + this.waiters.delete(sessionId) + } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-send.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-send.test.ts new file mode 100644 index 00000000000..7f62a01c864 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-send.test.ts @@ -0,0 +1,332 @@ +// What one `agentSession.send` writes, and when a user's Retry is allowed to +// put the same message on the wire a second time. + +import { beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionHost } from './structured-agent-session-host' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { + accepted, + attach, + CALLER, + envelope, + hostTestState +} from './structured-agent-session-host-test-harness' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestMessage +} from './structured-agent-session-host-test-data' + +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let dispatch: Mock + +beforeEach(() => { + ;({ store, host, dispatch } = hostTestState()) +}) + +describe('send', () => { + it('writes the submission before dispatching and resolves it accepted', async () => { + await attach() + const body = hostTestMessage('add a retry') + const result = await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + if (!result.ok) { + throw new Error(`expected a send, got ${result.refusal.code}`) + } + expect(result.value.submission.dispatchState).toBe('accepted') + expect(dispatch).toHaveBeenCalledTimes(1) + const page = host.history({ sessionId: SESSION, direction: 'tail' }) + expect(page.ok && page.page.items).toHaveLength(1) + expect(page.ok && page.page.fence).toBe(1) + expect(page.page.hostNow).toBe(NOW) + expect(page.providerSession).toEqual({ key: 'session_id', id: THREAD }) + }) + + it('settles a thrown dispatch as unknown, never as a rejection', async () => { + await attach() + dispatch.mockRejectedValueOnce(new Error('socket closed')) + const body = hostTestMessage('add a retry') + const result = await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + expect(result).toMatchObject({ ok: true, value: { submission: { dispatchState: 'unknown' } } }) + }) + + it('replays a retried send from the journal without dispatching twice', async () => { + await attach() + const body = hostTestMessage('add a retry') + const params = { envelope: envelope('agentSession.send', { body }), body } + await host.send(CALLER, params) + const retry = await host.send(CALLER, params) + expect(retry).toMatchObject({ ok: true, replayed: true }) + expect(dispatch).toHaveBeenCalledTimes(1) + }) + + it('refuses to redeliver an explicitly retried unknown from a thrown adapter call', async () => { + await attach() + dispatch.mockRejectedValueOnce(new Error('socket closed')) + const body = hostTestMessage('possibly delivered') + const params = { envelope: envelope('agentSession.send', { body }), body } + + const first = await host.send(CALLER, params) + expect(first).toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + // A thrown adapter call is indistinguishable from a lost reply, so it is not + // on the allowlist: Retry replays the recorded outcome. + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + expect(dispatch).toHaveBeenCalledTimes(1) + const state = host.history({ sessionId: SESSION, direction: 'tail' }) + expect(state.ok && state.page.submissions).toHaveLength(1) + }) + + it('redispatches an explicitly retried unknown the write itself refused', async () => { + await attach() + dispatch + .mockImplementationOnce(async () => ({ + state: 'unknown' as const, + reason: 'provider_write_failed: broken pipe' + })) + .mockImplementationOnce(async () => accepted()) + const body = hostTestMessage('never written') + const params = { envelope: envelope('agentSession.send', { body }), body } + + await host.send(CALLER, params) + // The only doubt on the allowlist: the transport refused the frame, so this + // is a first delivery and not a second. + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + replayed: false, + value: { submission: { dispatchState: 'accepted' } } + }) + expect(dispatch).toHaveBeenCalledTimes(2) + const state = host.history({ sessionId: SESSION, direction: 'tail' }) + expect(state.ok && state.page.submissions).toHaveLength(1) + }) + + it('returns an admitted retry to pending until the provider echo accepts it', async () => { + await attach() + dispatch + .mockImplementationOnce(async () => ({ + state: 'unknown' as const, + reason: 'provider_write_failed: connection closed before enqueue' + })) + .mockImplementationOnce(async () => ({ state: 'admitted' as const })) + const body = hostTestMessage('admitted on retry') + const params = { envelope: envelope('agentSession.send', { body }), body } + + await host.send(CALLER, params) + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + replayed: false, + value: { + submission: { dispatchState: 'pending', reason: null, resolvedAt: null } + } + }) + expect(dispatch).toHaveBeenCalledTimes(2) + }) + + it('refuses to redeliver a retry for a turn the provider already owns', async () => { + await attach() + dispatch.mockImplementationOnce(async () => ({ + state: 'unknown' as const, + reason: 'codex app-server started a turn it did not name in time' + })) + const body = hostTestMessage('a turn codex owns but did not name') + const params = { envelope: envelope('agentSession.send', { body }), body } + + const first = await host.send(CALLER, params) + expect(first).toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + // The turn is running; a second delivery would be a duplicate, so Retry + // replays the recorded outcome instead of re-sending. + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + expect(dispatch).toHaveBeenCalledTimes(1) + }) + + it('never reopens a submission the provider already proved delivered', async () => { + await attach() + dispatch.mockImplementationOnce(async () => accepted()) + const body = hostTestMessage('settled for good') + const params = { envelope: envelope('agentSession.send', { body }), body } + await host.send(CALLER, params) + const journal = ( + host as unknown as { sessions: Map } + ).sessions.get(SESSION)!.journal + const fence = store.getRecord(SESSION)?.lease.runtimeFence ?? 1 + + // Every later signal that could assert doubt: the attach sweep, and a + // direct unknown resolution. Neither may unsettle an accepted answer. + await journal.markPendingSubmissionsUnknown(fence) + await journal.resolveDispatch({ + clientMessageId: params.envelope.clientOperationId, + state: 'unknown', + reason: 'provider_write_failed: late transport error', + fence, + recovered: true + }) + + expect(journal.submissions()).toMatchObject([{ dispatchState: 'accepted', reason: null }]) + expect(journal.receiptFor(params.envelope.clientOperationId)).not.toBeNull() + }) + + it('leaves an admitted send pending and writes no dispatch row', async () => { + await attach() + dispatch.mockImplementationOnce(async () => ({ state: 'admitted' as const })) + const body = hostTestMessage('queued behind a running turn') + const params = { envelope: envelope('agentSession.send', { body }), body } + + await expect(host.send(CALLER, params)).resolves.toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'pending', reason: null, resolvedAt: null } } + }) + const journal = ( + host as unknown as { sessions: Map } + ).sessions.get(SESSION)!.journal + expect(journal.pendingSubmissions()).toHaveLength(1) + }) + + it('refuses to redeliver an admitted send a host restart left unanswered', async () => { + await attach() + dispatch.mockImplementationOnce(async () => ({ state: 'admitted' as const })) + const body = hostTestMessage('written, never acknowledged') + const params = { envelope: envelope('agentSession.send', { body }), body } + await host.send(CALLER, params) + const journal = ( + host as unknown as { sessions: Map } + ).sessions.get(SESSION)!.journal + + await journal.markPendingSubmissionsUnknown(store.getRecord(SESSION)?.lease.runtimeFence ?? 1) + expect(journal.submissions()).toMatchObject([ + { dispatchState: 'unknown', reason: 'host_restarted_before_acknowledgement' } + ]) + + // The frame was already written to the dead child's stdin, and Claude resumes + // the same provider session by id, so the restart ends the wait without + // proving non-delivery. Re-typing costs a message; redelivering costs a + // duplicate in the model's conversation. + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + expect(dispatch).toHaveBeenCalledTimes(1) + expect(journal.submissions()).toHaveLength(1) + }) + + it('refuses to redeliver an admitted send whose child exited first', async () => { + await attach() + dispatch.mockImplementationOnce(async () => ({ state: 'admitted' as const })) + const body = hostTestMessage('written, then the child died') + const params = { envelope: envelope('agentSession.send', { body }), body } + await host.send(CALLER, params) + const journal = ( + host as unknown as { sessions: Map } + ).sessions.get(SESSION)!.journal + + await journal.markPendingSubmissionsUnknown( + store.getRecord(SESSION)?.lease.runtimeFence ?? 1, + 'provider_exited_before_acknowledgement' + ) + + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + expect(dispatch).toHaveBeenCalledTimes(1) + }) + + it('advances an explicit retry after a ledger-unknown send is reconciled in the journal', async () => { + await attach() + const journal = ( + host as unknown as { sessions: Map } + ).sessions.get(SESSION)!.journal + vi.spyOn(journal, 'resolveDispatch').mockRejectedValueOnce(new Error('journal resolve failed')) + const body = hostTestMessage('possibly delivered before persistence failed') + const params = { envelope: envelope('agentSession.send', { body }), body } + + await expect(host.send(CALLER, params)).rejects.toThrow('journal resolve failed') + expect(journal.submissions()).toMatchObject([ + { clientMessageId: params.envelope.clientOperationId, dispatchState: 'unknown' } + ]) + expect( + store.listOperationRows().find((row) => row.operationId === params.envelope.clientOperationId) + ?.outcome + ).toEqual({ status: 'unknown' }) + expect(dispatch).toHaveBeenCalledTimes(1) + + await journal.markPendingSubmissionsUnknown(store.getRecord(SESSION)?.lease.runtimeFence ?? 1) + await expect(host.send(CALLER, params)).resolves.toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_unknown' } + }) + expect(dispatch).toHaveBeenCalledTimes(1) + + // The adapter took the message before the journal write failed, so the + // provider may already have it: an explicit retry replays, never redelivers. + await expect(host.send(CALLER, { ...params, retryUnknown: true })).resolves.toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'unknown' } } + }) + expect(dispatch).toHaveBeenCalledTimes(1) + expect(journal.submissions()).toHaveLength(1) + }) + + it('refuses a stale fence and hands back the current one', async () => { + const record = await attach() + const body = hostTestMessage('add a retry') + const result = await host.send(CALLER, { + envelope: envelope( + 'agentSession.send', + { body }, + { expectedRuntimeFence: (record?.lease.runtimeFence ?? 1) + 5 } + ), + body + }) + expect(result).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_checkpoint_stale', currentFence: record?.lease.runtimeFence } + }) + }) + + it('does not let a refused call leave a ledger row that replays past the fence', async () => { + const record = await attach() + const body = hostTestMessage('add a retry') + const params = { + envelope: envelope( + 'agentSession.send', + { body }, + { expectedRuntimeFence: (record?.lease.runtimeFence ?? 1) + 5 } + ), + body + } + await host.send(CALLER, params) + expect(await host.send(CALLER, params)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_checkpoint_stale' } + }) + expect(dispatch).not.toHaveBeenCalled() + }) + + it('refuses any mutation against a session this host has not attached', async () => { + const body = hostTestMessage('add a retry') + expect( + await host.send(CALLER, { envelope: envelope('agentSession.send', { body }), body }) + ).toMatchObject({ ok: false, refusal: { code: 'agent_session_ownership_unknown' } }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-settled-attach-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-settled-attach-retry.test.ts index 24dc81f4684..278619c2c61 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-settled-attach-retry.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-settled-attach-retry.test.ts @@ -310,16 +310,18 @@ describe('settled attach retry', () => { }) }) - it('restores an unknown submission without redispatch before a distinct send', async () => { + it('settles a submission the host restart left pending, and never redelivers it', async () => { expect((await host.attach(CALLER, hostTestAttachParams(null))).ok).toBe(true) - dispatch.mockRejectedValueOnce(new Error('socket closed')) - const body = hostTestMessage('possibly delivered') + // Admitted: written to the child, acknowledgement still outstanding. The + // restart below is the process fact that ends the wait, not a stopwatch. + dispatch.mockImplementationOnce(async () => ({ state: 'admitted' as const })) + const body = hostTestMessage('written before the host died') const unknownParams = { envelope: envelope('agentSession.send', { body }), body } const first = await host.send(CALLER, unknownParams) - expect(first).toMatchObject({ ok: true, value: { submission: { dispatchState: 'unknown' } } }) + expect(first).toMatchObject({ ok: true, value: { submission: { dispatchState: 'pending' } } }) await host.flushAllStreamedEvents() store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) @@ -360,6 +362,8 @@ describe('settled attach retry', () => { )?.dispatchState ).toBe('unknown') + // A restart ends the wait without proving the dead child never took the + // frame, so even an explicit retry replays rather than sending a second copy. const explicitRetry = await host.send(CALLER, { ...unknownParams, envelope: { @@ -370,9 +374,9 @@ describe('settled attach retry', () => { }) expect(explicitRetry).toMatchObject({ ok: true, - value: { submission: { dispatchState: 'accepted' } } + value: { submission: { dispatchState: 'unknown' } } }) - expect(dispatch).toHaveBeenCalledTimes(3) + expect(dispatch).toHaveBeenCalledTimes(2) }) it('records proven acquisition cleanup as durable death evidence', async () => { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-settlement-retry.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-settlement-retry.ts index 9fc68a9fca2..4e5de3f2566 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-settlement-retry.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-settlement-retry.ts @@ -4,6 +4,7 @@ import type { StructuredAgentSessionHostDeps, StructuredAgentSessionHostSession } from './structured-agent-session-host-types' +import { turnVerdictFromDeathEvidence } from './structured-agent-session-stale-turn-verdict' import { retryUnexpectedExitSettlement, type StructuredAgentSessionUnexpectedExitContext @@ -80,7 +81,9 @@ export async function retryLoadedStructuredAgentSessionSettlement(input: { acquisitionGeneration: retrySession.acquisitionGeneration ?? 'recovery' }, session: retrySession, - stableSettlementId: record.lease.settlementRetryId + stableSettlementId: record.lease.settlementRetryId, + // Only an observed exit earns an end time; a probe-proven death never saw one. + verdict: turnVerdictFromDeathEvidence(record.lease.deathEvidence) }) if (!ok) { return false diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-stale-turn-verdict.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-stale-turn-verdict.test.ts new file mode 100644 index 00000000000..8ffb7acf6ce --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-stale-turn-verdict.test.ts @@ -0,0 +1,173 @@ +import { describe, expect, it, vi } from 'vitest' +import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import type { AgentJournalRenderItem } from '../../../shared/agent-session-journal-types' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { + runningTurnLifecycleRevisions, + settleStaleRunningTurnsOnAcquire, + turnVerdictFromDeathEvidence +} from './structured-agent-session-stale-turn-verdict' + +const THREAD = 'thread-1' +const RUNNING_IDENTITY = { + provider: 'codex' as const, + threadId: THREAD, + turnId: 'turn-2', + ordinal: 0 +} + +function lifecycleItem( + turnId: string, + state: 'running' | 'completed', + sequence: number, + extra: { startedAt?: number; completedAt?: number } = {} +): AgentJournalRenderItem { + return { + itemId: agentJournalItemKey({ provider: 'codex', threadId: THREAD, turnId, ordinal: 0 }), + revision: 1, + sequence, + observedAt: sequence, + body: { kind: 'turn', turnId, state, ...extra } + } +} + +/** The status-form carrier an older host wrote; still read, never written back. */ +function legacyLifecycleItem(turnId: string, startedAt: number): AgentJournalRenderItem { + return { + ...lifecycleItem(turnId, 'running', 2), + body: { + kind: 'status', + text: 'Working', + turnLifecycle: { turnId, state: 'running', startedAt } + } + } +} + +describe('turn verdict from death evidence', () => { + it('earns an end time only from an observed exit', () => { + expect( + turnVerdictFromDeathEvidence({ kind: 'exit-observed', detail: 'exit', observedAt: 500 }) + ).toEqual({ state: 'interrupted', completedAt: 500 }) + expect( + turnVerdictFromDeathEvidence({ kind: 'pid-absent', detail: 'gone', observedAt: 500 }) + ).toEqual({ state: 'unverifiable' }) + expect( + turnVerdictFromDeathEvidence({ kind: 'identity-mismatch', detail: 'pid', observedAt: 500 }) + ).toEqual({ state: 'unverifiable' }) + expect(turnVerdictFromDeathEvidence(null)).toEqual({ state: 'unverifiable' }) + }) +}) + +describe('running turn lifecycle revisions', () => { + it('revises only running rows in place and carries an end time only for an observed exit', () => { + const items = [ + lifecycleItem('turn-1', 'completed', 1, { startedAt: 10, completedAt: 20 }), + // A stray end on a running row is never carried into the verdict. + lifecycleItem('turn-2', 'running', 2, { startedAt: 30, completedAt: 99 }) + ] + expect(runningTurnLifecycleRevisions(items, { state: 'interrupted', completedAt: 40 })).toEqual( + [ + { + kind: 'item', + identity: RUNNING_IDENTITY, + body: { + kind: 'turn', + turnId: 'turn-2', + state: 'interrupted', + startedAt: 30, + completedAt: 40 + } + } + ] + ) + expect(runningTurnLifecycleRevisions(items, { state: 'unverifiable' })).toEqual([ + expect.objectContaining({ + body: { kind: 'turn', turnId: 'turn-2', state: 'unverifiable', startedAt: 30 } + }) + ]) + }) + + it('revises a legacy status-form running row from an older host into a typed turn', () => { + expect( + runningTurnLifecycleRevisions([legacyLifecycleItem('turn-2', 30)], { state: 'unverifiable' }) + ).toEqual([ + { + kind: 'item', + identity: RUNNING_IDENTITY, + body: { kind: 'turn', turnId: 'turn-2', state: 'unverifiable', startedAt: 30 } + } + ]) + }) + + it('skips rows without a parseable identity', () => { + const item = { ...lifecycleItem('turn-2', 'running', 2), itemId: 'not-an-item-key' } + expect(runningTurnLifecycleRevisions([item], { state: 'unverifiable' })).toEqual([]) + }) +}) + +describe('stale running turns on a cold acquire', () => { + function journalWith(items: AgentJournalRenderItem[]) { + const appendLifecycleBatch = vi.fn(async () => ({ epoch: 'epoch-1', sequence: 9 })) + const journal = { + snapshot: () => ({ items }), + cursor: () => ({ epoch: 'epoch-1', sequence: 8 }), + appendLifecycleBatch + } as unknown as AgentSessionJournal + return { journal, appendLifecycleBatch } + } + + it('marks a running row from the dead generation unverifiable without an end time', async () => { + const { journal, appendLifecycleBatch } = journalWith([ + lifecycleItem('turn-1', 'completed', 1, { startedAt: 10, completedAt: 20 }), + lifecycleItem('turn-2', 'running', 2, { startedAt: 30 }) + ]) + + await expect( + settleStaleRunningTurnsOnAcquire({ + journal, + sessionId: 'session-1', + fence: 14, + acquisitionGeneration: 'generation-2' + }) + ).resolves.toBe(1) + + expect(appendLifecycleBatch).toHaveBeenCalledExactlyOnceWith({ + settlementId: 'stale-turn:session-1:14:generation-2', + fence: 14, + recovered: true, + mutations: [ + { + kind: 'item', + identity: RUNNING_IDENTITY, + body: { kind: 'turn', turnId: 'turn-2', state: 'unverifiable', startedAt: 30 } + } + ] + }) + }) + + it('writes nothing when no turn is running and keys on the journal position without a generation', async () => { + const idle = journalWith([ + lifecycleItem('turn-1', 'completed', 1, { startedAt: 10, completedAt: 20 }) + ]) + await expect( + settleStaleRunningTurnsOnAcquire({ + journal: idle.journal, + sessionId: 'session-1', + fence: 14, + acquisitionGeneration: null + }) + ).resolves.toBe(0) + expect(idle.appendLifecycleBatch).not.toHaveBeenCalled() + + const running = journalWith([lifecycleItem('turn-2', 'running', 2)]) + await settleStaleRunningTurnsOnAcquire({ + journal: running.journal, + sessionId: 'session-1', + fence: 14, + acquisitionGeneration: null + }) + expect(running.appendLifecycleBatch).toHaveBeenCalledWith( + expect.objectContaining({ settlementId: 'stale-turn:session-1:14:seq-8' }) + ) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-stale-turn-verdict.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-stale-turn-verdict.ts new file mode 100644 index 00000000000..940b0c8be17 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-stale-turn-verdict.ts @@ -0,0 +1,102 @@ +// What the host may durably say about a turn whose provider child is gone. +// +// `interrupted` requires the host to have seen the child exit; that receipt is the only end time it +// is allowed to record. Everything weaker — a pid probe, an identity mismatch, a journal found +// running on a cold acquire — is `unverifiable` and carries no end at all. + +import { parseAgentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import type { + AgentJournalRenderItem, + AgentJournalTurnLifecycle +} from '../../../shared/agent-session-journal-types' +import { + agentJournalTurnBody, + readAgentJournalTurn +} from '../../../shared/agent-session-turn-record' +import type { AgentSessionDeathEvidence } from '../../../shared/agent-session-record' +import { partitionJournalLifecycleMutations } from '../agent-session-journal/journal-lifecycle-batch-partition' +import type { JournalLifecycleMutationInput } from '../agent-session-journal/journal-row-builders' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' + +export type StructuredAgentSessionTurnVerdict = + | { state: 'interrupted'; completedAt: number } + | { state: 'unverifiable' } + +export const UNVERIFIABLE_TURN_VERDICT: StructuredAgentSessionTurnVerdict = { + state: 'unverifiable' +} + +export function turnVerdictFromDeathEvidence( + evidence: AgentSessionDeathEvidence | null | undefined +): StructuredAgentSessionTurnVerdict { + return evidence?.kind === 'exit-observed' + ? { state: 'interrupted', completedAt: evidence.observedAt } + : UNVERIFIABLE_TURN_VERDICT +} + +/** Revises every still-running lifecycle item in place, keeping its identity and start. */ +export function runningTurnLifecycleRevisions( + items: readonly AgentJournalRenderItem[], + verdict: StructuredAgentSessionTurnVerdict +): JournalLifecycleMutationInput[] { + const revisions: JournalLifecycleMutationInput[] = [] + for (const item of items) { + const turn = readAgentJournalTurn(item.body) + if (turn?.state !== 'running') { + continue + } + const identity = parseAgentJournalItemKey(item.itemId) + if (!identity) { + continue + } + revisions.push({ + kind: 'item', + identity, + body: agentJournalTurnBody(settledLifecycle(turn, verdict)) + }) + } + return revisions +} + +function settledLifecycle( + lifecycle: AgentJournalTurnLifecycle, + verdict: StructuredAgentSessionTurnVerdict +): AgentJournalTurnLifecycle { + const settled: AgentJournalTurnLifecycle = { turnId: lifecycle.turnId, state: verdict.state } + if (lifecycle.userItemId !== undefined) { + settled.userItemId = lifecycle.userItemId + } + if (lifecycle.startedAt !== undefined) { + settled.startedAt = lifecycle.startedAt + } + if (verdict.state === 'interrupted') { + settled.completedAt = verdict.completedAt + } + return settled +} + +/** A running row found when a NEW child is acquired belongs to a generation whose exit nobody + * observed. Must run before that child's buffered events land, or a live turn would be judged. */ +export async function settleStaleRunningTurnsOnAcquire(input: { + journal: AgentSessionJournal + sessionId: string + fence: number + acquisitionGeneration: string | null +}): Promise { + const { journal } = input + const revisions = runningTurnLifecycleRevisions( + journal.snapshot().items, + UNVERIFIABLE_TURN_VERDICT + ) + const generation = input.acquisitionGeneration ?? `seq-${journal.cursor().sequence}` + const settlementId = `stale-turn:${input.sessionId}:${input.fence}:${generation}` + for (const chunk of partitionJournalLifecycleMutations(settlementId, revisions)) { + await journal.appendLifecycleBatch({ + settlementId: chunk.settlementId, + fence: input.fence, + recovered: true, + mutations: chunk.mutations + }) + } + return revisions.length +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts index 1e3b9bb25a3..b085cbd3012 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.test.ts @@ -1,16 +1,21 @@ import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { AgentSessionRecord } from '../../../shared/agent-session-record' -import type { AgentSessionStatusEvent } from '../../../shared/agent-session-wire' +import type { + AgentSessionBackgroundTask, + AgentSessionStatusEvent, + AgentSessionStatusSummary +} from '../../../shared/agent-session-wire' import { createClaudeJournalTranslator } from '../../claude/claude-structured-journal-translation' import { publishCodexTurnLifecycle } from '../../codex/codex-structured-journal-translation-turns' import { createDeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { createTrackedJournalOpener } from '../agent-session-journal/journal-store-test-open' import { StructuredAgentSessionStatusFeed, - type StructuredAgentSessionStatusFeedDeps + type StructuredAgentSessionStatusFeedDeps, + type StructuredAgentSessionStatusSink } from './structured-agent-session-status-feed' const SESSION = 'status-session' @@ -56,9 +61,11 @@ async function openJournal(sessionId = SESSION, now?: () => number) { function indexed(session: { journal: Awaited> hasProviderChild?: boolean + fence?: number }) { return { journal: session.journal, + fence: session.fence ?? 1, ...(session.hasProviderChild !== undefined ? { hasProviderChild: session.hasProviderChild } : {}), @@ -69,14 +76,18 @@ function indexed(session: { function feedFor( sessions: Map< string, - { journal: Awaited>; hasProviderChild?: boolean } + { journal: Awaited>; hasProviderChild?: boolean; fence?: number } >, record: Partial | null = null, - onStatusChanged?: StructuredAgentSessionStatusFeedDeps['onStatusChanged'] + onStatusChanged?: StructuredAgentSessionStatusFeedDeps['onStatusChanged'], + readBackgroundTasks?: StructuredAgentSessionStatusFeedDeps['readBackgroundTasks'], + statusSink?: StructuredAgentSessionStatusSink ) { let now = 1_000 const feed = new StructuredAgentSessionStatusFeed({ ...(onStatusChanged ? { onStatusChanged } : {}), + ...(statusSink ? { statusSink: () => statusSink } : {}), + ...(readBackgroundTasks ? { readBackgroundTasks } : {}), sessions: { get: (sessionId: string) => { const session = sessions.get(sessionId) @@ -153,6 +164,59 @@ describe('StructuredAgentSessionStatusFeed', () => { ]) }) + it('stops projecting an old-host unknown submission after the owner fence advances', async () => { + const journal = await openJournal() + const session = { journal, fence: 1 } + const { feed, events } = feedFor(new Map([[SESSION, session]])) + await journal.appendSubmission({ + clientMessageId: 'old-host', + payloadFingerprint: 'fp', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'slow' }] }, + fence: 1 + }) + await journal.resolveDispatch({ + clientMessageId: 'old-host', + state: 'unknown', + reason: 'ack timeout', + fence: 1 + }) + feed.publish(SESSION) + expect(events.at(-1)).toMatchObject({ session: { status: 'working' } }) + session.fence = 2 + feed.publish(SESSION) + expect(events.at(-1)).toMatchObject({ session: { status: 'idle' } }) + }) + + it('publishes working from the pending submission, before the provider replays the turn', async () => { + const journal = await openJournal() + const { feed, events } = feedFor(new Map([[SESSION, { journal }]])) + events.length = 0 + await journal.appendSubmission({ + clientMessageId: 'client-1', + payloadFingerprint: 'fingerprint-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'write a poem' }] }, + fence: 1 + }) + + feed.publish(SESSION) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'working' }) + }) + + await journal.resolveDispatch({ + clientMessageId: 'client-1', + state: 'accepted', + providerIdentity: USER_IDENTITY, + fence: 1 + }) + feed.publish(SESSION) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ status: 'idle' }) + }) + }) + it('publishes working, then idle once the running marker is tombstoned, and never a repeat', async () => { const journal = await openJournal() const { feed, events } = feedFor(new Map([[SESSION, { journal }]])) @@ -566,6 +630,147 @@ describe('StructuredAgentSessionStatusFeed', () => { session: expect.objectContaining({ status: 'idle', latestPrompt: 'hello' }) }) }) + + it('reuses the journal projection across task progress and invalidates on journal changes', async () => { + const journal = await openJournal() + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'fan out' }] }, + { fence: 1 } + ) + await journal.appendItem( + TURN_IDENTITY, + { kind: 'status', text: 'Working', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 1 } + ) + const snapshot = vi.spyOn(journal, 'snapshot') + let taskState: 'working' | 'waiting' = 'working' + const { feed, events } = feedFor(new Map([[SESSION, { journal }]]), null, undefined, () => ({ + state: 'monitoring', + tasks: [{ id: 'child', kind: 'agent', state: taskState }] + })) + for (let tick = 1; tick <= 100; tick++) { + taskState = tick % 2 === 1 ? 'waiting' : 'working' + feed.publish(SESSION) + } + expect(events).toHaveLength(101) + expect(snapshot).toHaveBeenCalledTimes(1) + expect(events.at(-1)).toMatchObject({ + type: 'status', + session: { status: 'working', backgroundTasks: [{ state: 'working' }] } + }) + await journal.appendTombstone(TURN_IDENTITY, { fence: 1 }) + feed.publish(SESSION) + expect(snapshot).toHaveBeenCalledTimes(2) + expect(events.at(-1)).toMatchObject({ type: 'status', session: { status: 'idle' } }) + }) + + it('invalidates cached status on unreadability and keeps record metadata live', async () => { + const journal = await openJournal() + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + const record = { options: { model: 'first-model' }, providerHandleChain: [] } + const { feed, events } = feedFor(new Map([[SESSION, { journal }]]), record) + record.options.model = 'second-model' + feed.publish(SESSION) + expect(events.at(-1)).toMatchObject({ + type: 'status', + session: { status: 'idle', model: 'second-model' } + }) + const readOnly = vi.spyOn(journal, 'isReadOnly', 'get').mockReturnValue(true) + feed.publish(SESSION) + expect(events.at(-1)).toMatchObject({ type: 'status', session: { status: null } }) + readOnly.mockRestore() + feed.publish(SESSION) + expect(events.at(-1)).toMatchObject({ type: 'status', session: { status: 'idle' } }) + }) + + it('projects live background tasks and republishes a task-only state change', async () => { + const journal = await openJournal() + let tasks = [ + { id: 'task-1', kind: 'agent' as const, name: 'deep_review', state: 'working' as const } + ] + const { feed, events } = feedFor(new Map([[SESSION, { journal }]]), null, undefined, () => ({ + state: 'monitoring', + tasks + })) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'fan out' }] }, + { fence: 1 } + ) + feed.publish(SESSION, journal) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ + backgroundTasks: [{ id: 'task-1', kind: 'agent', name: 'deep_review', state: 'working' }] + }) + }) + + // No journal change: only the task state moved. + tasks = [{ id: 'task-1', kind: 'agent', name: 'deep_review', state: 'waiting' as never }] + const before = events.length + feed.publish(SESSION, journal) + expect(events).toHaveLength(before + 1) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ + backgroundTasks: [expect.objectContaining({ state: 'waiting' })] + }) + }) + + // An identical projection is suppressed. + feed.publish(SESSION, journal) + expect(events).toHaveLength(before + 1) + }) + + it('omits task usage so a progress tick never re-broadcasts the summary', async () => { + const journal = await openJournal() + let tasks: AgentSessionBackgroundTask[] = [ + { id: 'task-1', kind: 'agent', name: 'deep_review', state: 'working', totalTokens: 10 } + ] + const { feed, events } = feedFor(new Map([[SESSION, { journal }]]), null, undefined, () => ({ + state: 'monitoring', + tasks + })) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'fan out' }] }, + { fence: 1 } + ) + feed.publish(SESSION, journal) + const before = events.length + + // A `task_progress` frame moves only usage, which no status-summary reader renders; + // re-broadcasting the whole summary per frame would cost every remote subscriber. + tasks = [ + { id: 'task-1', kind: 'agent', name: 'deep_review', state: 'working', totalTokens: 4_200 } + ] + feed.publish(SESSION, journal) + expect(events).toHaveLength(before) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ + backgroundTasks: [{ id: 'task-1', kind: 'agent', name: 'deep_review', state: 'working' }] + }) + }) + + // A state change on the same task still reaches subscribers. + tasks = [ + { id: 'task-1', kind: 'agent', name: 'deep_review', state: 'waiting', totalTokens: 4_200 } + ] + feed.publish(SESSION, journal) + expect(events).toHaveLength(before + 1) + expect(events.at(-1)).toEqual({ + type: 'status', + session: expect.objectContaining({ + backgroundTasks: [expect.objectContaining({ state: 'waiting' })] + }) + }) + }) }) /** @@ -574,23 +779,46 @@ describe('StructuredAgentSessionStatusFeed', () => { * it lists every session this host has ever opened. Eviction's `forget-session` step deletes the * session from the live map and touches nothing else, so a poller has to intersect with that map. */ -describe('the polling reader answers from the live sessions, not the retained cache', () => { - it('drops an evicted session from the poll while a late subscriber still sees it', async () => { +describe('the status sink sees the roster the broadcast cache deliberately lacks', () => { + function sinkFor() { + const published: AgentSessionStatusSummary[] = [] + const forgotten: string[] = [] + const sink: StructuredAgentSessionStatusSink = { + publish: (summary) => published.push(summary), + forget: (sessionId) => forgotten.push(sessionId) + } + return { sink, published, forgotten } + } + + it('receives every change once, ownership revocation, and the forget edge', async () => { const journal = await openJournal() - const sessions = new Map([[SESSION, { journal }]]) - const { feed } = feedFor(sessions) + const sessions = new Map([[SESSION, { journal, hasProviderChild: true }]]) + const { sink, published, forgotten } = sinkFor() + const { feed } = feedFor(sessions, null, undefined, undefined, sink) await journal.appendItem( USER_IDENTITY, { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, { fence: 1 } ) feed.publish(SESSION, journal) - expect(feed.liveSessionSummaries().map((summary) => summary.sessionId)).toEqual([SESSION]) + // A second identical publication is deduped for the sink exactly as for subscribers, so the + // sink saw two writes: the opening projection the harness's subscriber triggered, then this. + feed.publish(SESSION, journal) + expect(published.map((summary) => summary.status)).toEqual([null, 'idle']) + expect(published.at(-1)).toMatchObject({ + sessionId: SESSION, + status: 'idle', + hostExecutionOwned: true + }) - // Exactly what eviction's `forget-session` step does; nothing else touches the feed. + feed.revokeLive(SESSION) + expect(published.at(-1)).toMatchObject({ sessionId: SESSION, status: 'idle' }) + expect(published.at(-1)?.hostExecutionOwned).toBeUndefined() + + // Exactly what `close` does after eviction: the cache keeps the projection, the sink does not. sessions.delete(SESSION) - - expect(feed.liveSessionSummaries()).toEqual([]) + feed.forget(SESSION) + expect(forgotten).toEqual([SESSION]) const late: AgentSessionStatusEvent[] = [] feed.subscribe({ id: 'list-2', emit: (event) => late.push(event) }) expect(late).toEqual([ @@ -600,4 +828,34 @@ describe('the polling reader answers from the live sessions, not the retained ca } ]) }) + + it('keeps publishing to subscribers when the sink throws', async () => { + const journal = await openJournal() + const sink: StructuredAgentSessionStatusSink = { + publish: () => { + throw new Error('store down') + }, + forget: () => { + throw new Error('store down') + } + } + const { feed, events } = feedFor( + new Map([[SESSION, { journal }]]), + null, + undefined, + undefined, + sink + ) + await journal.appendItem( + USER_IDENTITY, + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hello' }] }, + { fence: 1 } + ) + feed.publish(SESSION, journal) + expect(() => feed.forget(SESSION)).not.toThrow() + expect(events.at(-1)).toMatchObject({ + type: 'status', + session: expect.objectContaining({ sessionId: SESSION, status: 'idle' }) + }) + }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts index 61d9649587e..220a8116160 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-status-feed.ts @@ -14,9 +14,11 @@ import { agentProviderSessionsEqual } from '../../../shared/agent-session-resume import type { AgentSessionRecord } from '../../../shared/agent-session-record' import { normalizeOptionalField } from '../../../shared/agent-status-field-normalization' import { AGENT_MODEL_MAX_LENGTH } from '../../../shared/agent-status-types' -import type { - AgentSessionStatusEvent, - AgentSessionStatusSummary +import { + agentSessionBackgroundTasksEqual, + type AgentSessionBackgroundTaskState, + type AgentSessionStatusEvent, + type AgentSessionStatusSummary } from '../../../shared/agent-session-wire' import { projectStructuredAgentSessionStatusSummary } from '../../../shared/structured-agent-session-projection' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' @@ -31,6 +33,15 @@ type StatusFeedSession = { journal: AgentSessionJournal params: { location: { workspaceId: string }; provider: AgentSessionRecord['provider'] } hasProviderChild?: boolean + fence?: number +} + +/** Where the host's projections land for readers that see every agent alike (`worktree ps`, + * mobile, the hook store's own fanout). `forget` is the roster edge the broadcast cache + * deliberately never has. */ +export type StructuredAgentSessionStatusSink = { + publish: (summary: AgentSessionStatusSummary) => void + forget: (sessionId: string) => void } export type StructuredAgentSessionStatusFeedDeps = { @@ -40,6 +51,12 @@ export type StructuredAgentSessionStatusFeedDeps = { /** Every projection change, whether or not anyone is subscribed. `replay` marks a re-projection * of state the host already knew (restore, an arriving subscriber) rather than a journal edge. */ onStatusChanged?: (summary: AgentSessionStatusSummary, options: { replay: boolean }) => void + /** Resolved on every call: the host builds this feed in a field initializer, before its own + * deps are assigned. */ + statusSink?: () => StructuredAgentSessionStatusSink | undefined + /** Live provider-owned background tasks for the summary, so session lists can + * render subagent children. Optional: a provider without the hook projects none. */ + readBackgroundTasks?: (sessionId: string) => AgentSessionBackgroundTaskState | null | undefined } function summariesEqual(a: AgentSessionStatusSummary, b: AgentSessionStatusSummary): boolean { @@ -56,13 +73,54 @@ function summariesEqual(a: AgentSessionStatusSummary, b: AgentSessionStatusSumma a.toolName === b.toolName && a.toolInput === b.toolInput && a.lastAssistantMessage === b.lastAssistantMessage && + agentSessionBackgroundTasksEqual(a.backgroundTasks, b.backgroundTasks) && agentProviderSessionsEqual(undefined, a.providerSession, b.providerSession) ) } +/** Wire the host's own deps into a feed; keeps the host at one call site. + * `deps` is a thunk because the host builds the feed in a field initializer, + * before its constructor parameters are assigned. */ +export function createStructuredAgentSessionHostStatusFeed(args: { + sessions: StructuredAgentSessionStatusFeedDeps['sessions'] + now: () => number + deps: () => { + store: { getRecord: (sessionId: string) => AgentSessionRecord | null } + adapter: { + backgroundTaskState?: ( + sessionId: string + ) => AgentSessionBackgroundTaskState | null | undefined + } + onSessionStatusChanged?: StructuredAgentSessionStatusFeedDeps['onStatusChanged'] + statusSink?: StructuredAgentSessionStatusSink + } +}): StructuredAgentSessionStatusFeed { + return new StructuredAgentSessionStatusFeed({ + sessions: args.sessions, + getRecord: (sessionId) => args.deps().store.getRecord(sessionId), + now: args.now, + onStatusChanged: (summary, options) => args.deps().onSessionStatusChanged?.(summary, options), + readBackgroundTasks: (sessionId) => args.deps().adapter.backgroundTaskState?.(sessionId), + // Resolved per call for the same reason the other deps are: the host builds this feed in a + // field initializer, before its constructor parameters are assigned. + statusSink: () => args.deps().statusSink + }) +} + export class StructuredAgentSessionStatusFeed { private readonly subscribers = new Map() private readonly published = new Map() + // Task progress must not sort and scan an unchanged conversation. Journal identity owns cleanup. + private readonly journalProjections = new WeakMap< + AgentSessionJournal, + { + epoch: string + sequence: number + readOnly: boolean + fence: number | undefined + summary: ReturnType + } + >() constructor(private readonly deps: StructuredAgentSessionStatusFeedDeps) {} @@ -78,25 +136,20 @@ export class StructuredAgentSessionStatusFeed { return () => this.unsubscribe(subscriber.id) } - /** - * Summaries for the sessions this host still holds, for readers that poll instead of subscribing. - * - * `published` never retracts, so it is a broadcast cache and not a roster: enumerating it lists - * every session ever opened here. A caller asking what is running gets the live intersection, - * while the retained view a subscriber opens on stays whole. - * - * Deliberately does NOT re-project: a subscriber's snapshot is the live read, and re-running the - * journal reduction per caller would make an enumerating command pay for every session it lists. - */ - liveSessionSummaries(): AgentSessionStatusSummary[] { - const summaries: AgentSessionStatusSummary[] = [] - for (const [sessionId] of this.deps.sessions) { - const summary = this.published.get(sessionId) - if (summary) { - summaries.push(summary) - } + /** The host stopped holding the session: ownership leaves the retained projection, and the + * row leaves the sink. `published` keeps the projection for reload history. */ + close(sessionId: string): void { + this.revokeLive(sessionId) + this.forget(sessionId) + } + + /** The sink lists what is running; a forgotten session must not be in it. */ + forget(sessionId: string): void { + try { + this.deps.statusSink?.()?.forget(sessionId) + } catch (error) { + console.warn('[structured-session-status] status sink forget failed', error) } - return summaries } unsubscribe(id: string): void { @@ -124,6 +177,7 @@ export class StructuredAgentSessionStatusFeed { type: 'status', session: retained }) + this.sink(retained) } /** Re-projects one session after its journal changed; equal projections are not re-sent. */ @@ -139,6 +193,7 @@ export class StructuredAgentSessionStatusFeed { } this.published.set(sessionId, summary) this.broadcast({ type: 'status', session: summary }) + this.sink(summary) try { this.deps.onStatusChanged?.(summary, { replay: options?.replay === true }) } catch (error) { @@ -153,27 +208,68 @@ export class StructuredAgentSessionStatusFeed { journal: AgentSessionJournal ): AgentSessionStatusSummary { // An unreadable journal projects as "no turn": the chat itself shows the reset. - const items = journal.isReadOnly ? [] : journal.snapshot().items + const cursor = journal.cursor() + const readOnly = journal.isReadOnly + const fence = session.fence + let projection = this.journalProjections.get(journal) + if ( + !projection || + projection.epoch !== cursor.epoch || + projection.sequence !== cursor.sequence || + projection.readOnly !== readOnly || + projection.fence !== fence + ) { + // A journalled submission bumps `lastSequence`, so the send-time working + // signal reaches the cache; the lease fence does not, hence the extra key. + const snapshot = readOnly ? null : journal.snapshot() + projection = { + ...cursor, + readOnly, + fence, + summary: projectStructuredAgentSessionStatusSummary( + snapshot?.items ?? [], + snapshot?.submissions ?? [], + fence + ) + } + this.journalProjections.set(journal, projection) + } const record = this.deps.getRecord(sessionId) const providerSession = structuredAgentSessionProviderSessionMetadata(record) // The journal has no model: the record's acknowledged options are where an owner // handoff or a mid-session switch lands, so the row follows whichever is in force. const model = normalizeOptionalField(record?.options?.model, AGENT_MODEL_MAX_LENGTH) + // Usage is dropped here on purpose: a `task_progress` tick would otherwise fail the + // equality check and re-broadcast a full summary to every remote subscriber for a + // number no session list renders. Tokens stay live on the background-task channel. + const backgroundTasks = this.deps + .readBackgroundTasks?.(sessionId) + ?.tasks?.map(({ totalTokens: _totalTokens, ...task }) => task) return { sessionId, workspaceId: session.params.location.workspaceId, agent: session.params.provider, ...(session.hasProviderChild ? { hostExecutionOwned: true as const } : {}), - ...projectStructuredAgentSessionStatusSummary(items), + ...projection.summary, ...(record?.rewind?.phase === 'prepared' || record?.rewind?.phase === 'provider-succeeded' ? { rewindBlockedReason: 'outcome-unknown' as const } : {}), ...(model ? { model } : {}), + ...(backgroundTasks && backgroundTasks.length > 0 ? { backgroundTasks } : {}), ...(providerSession ? { providerSession } : {}), updatedAt: journal.lastActivityAt() || this.deps.now() } } + /** A failing sink must never cost the subscribers their status event. */ + private sink(summary: AgentSessionStatusSummary): void { + try { + this.deps.statusSink?.()?.publish(summary) + } catch (error) { + console.warn('[structured-session-status] status sink publish failed', error) + } + } + private broadcast(event: AgentSessionStatusEvent): void { // A Map skips entries deleted mid-iteration, so a failing subscriber can drop itself here. for (const subscriber of this.subscribers.values()) { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts index a32db8786d4..4f22b56397c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts @@ -68,11 +68,57 @@ describe('AgentSessionSubscribers', () => { submissions: [] }, fence: 7, + hostNow: expect.any(Number), activity: null } ]) }) + it('stamps the host clock once per published frame', async () => { + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: 'thread-1' } + }, + journalDir: join(root, 'clock-journal') + }) + let now = 1_000 + const events: AgentSessionSubscribeEvent[] = [] + const subscribers = new AgentSessionSubscribers({ now: () => (now += 1) }) + const emit = (event: AgentSessionSubscribeEvent): void => { + events.push(event) + } + subscribers.open({ id: 'one', sessionId: SESSION, journal, fence: 1, emit }) + subscribers.open({ id: 'two', sessionId: SESSION, journal, fence: 1, emit }) + await journal.appendItem( + { provider: 'orca', clientMessageId: 'clocked' }, + { kind: 'status', text: 'Clocked' }, + { fence: 1 } + ) + subscribers.publish(SESSION, journal) + subscribers.handoff(SESSION, 1, { owner: 'native' } as AgentSessionHandoffStatus) + subscribers.reset(SESSION, journal, 'epoch_changed', 1) + + expect(events.map((event) => ('hostNow' in event ? event.hostNow : null))).toEqual([ + 1_001, 1_002, + // Both subscribers of one publication read the same clock sample. + 1_003, 1_003, 1_004, 1_004, 1_005, 1_005 + ]) + expect(events.map((event) => event.type)).toEqual([ + 'snapshot', + 'snapshot', + 'batch', + 'batch', + 'batch', + 'batch', + 'reset', + 'reset' + ]) + }) + it('includes catalogs on reconnect and sends an idle checkpoint without journal work', async () => { const journal = await journals.open({ identity: { @@ -101,6 +147,7 @@ describe('AgentSessionSubscribers', () => { type: 'batch', sessionId: SESSION, fence: 7, + hostNow: expect.any(Number), commands, batch: { cursor: journal.cursor(), items: [], removedItemIds: [], submissions: [] } }) @@ -245,6 +292,7 @@ describe('AgentSessionSubscribers', () => { submissions: [] }, fence: 2, + hostNow: expect.any(Number), handoff }) }) @@ -284,6 +332,7 @@ describe('AgentSessionSubscribers', () => { sessionId: SESSION, batch: { cursor, items: [], removedItemIds: [], submissions: [] }, fence: 2, + hostNow: expect.any(Number), backgroundTasks }) @@ -330,6 +379,7 @@ describe('AgentSessionSubscribers', () => { sessionId: SESSION, batch: { cursor, items: [], removedItemIds: [], submissions: [] }, fence: 1, + hostNow: expect.any(Number), activity: { turnId: 'turn-1', text: 'Inspecting the session wire' } }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts index a5062373dad..24415df1d3c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts @@ -17,6 +17,7 @@ import { type AgentSessionTurnActivity } from '../../../shared/agent-session-wire' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { emptyAgentSessionBatch } from './agent-session-empty-batch' import { createAgentSessionCatchUpReader, readAgentSessionHydrationPage @@ -44,6 +45,8 @@ export type AgentSessionSubscribersHooks = { /** Fires after any publication that can change journal content, whether or not anyone * is subscribed to the transcript: session lists project status from this same edge. */ onJournalPublished?: (sessionId: string, journal: AgentSessionJournal) => void + /** Host wall clock, stamped once per published frame as `hostNow`. */ + now?: () => number } export class AgentSessionSubscribers { @@ -76,8 +79,9 @@ export class AgentSessionSubscribers { session.set(input.id, subscriber) this.bySession.set(input.sessionId, session) + const hostNow = this.now() if (input.cursor) { - this.deliver(subscriber, input.journal, input.handoff, true, input.backgroundTasks) + this.deliver(subscriber, input.journal, hostNow, input.handoff, true, input.backgroundTasks) } else { const page = readAgentSessionHydrationPage(input.journal, input.fence) this.emit(subscriber, { @@ -85,6 +89,7 @@ export class AgentSessionSubscribers { sessionId: input.sessionId, page, fence: input.fence, + hostNow, ...(input.handoff ? { handoff: input.handoff } : {}), ...(input.backgroundTasks !== undefined ? { backgroundTasks: input.backgroundTasks } : {}), ...this.activityField(input.sessionId) @@ -121,8 +126,9 @@ export class AgentSessionSubscribers { this.activityBySession.delete(sessionId) } } + const hostNow = this.now() for (const subscriber of this.subscribers(sessionId)) { - this.deliver(subscriber, journal, undefined, false, undefined, activity) + this.deliver(subscriber, journal, hostNow, undefined, false, undefined, activity) } if (activity === undefined) { this.hooks.onJournalPublished?.(sessionId, journal) @@ -138,21 +144,7 @@ export class AgentSessionSubscribers { fence: number, backgroundTasks?: AgentSessionBackgroundTaskState | null ): void { - const page = readAgentSessionHydrationPage(journal, fence) - for (const subscriber of this.subscribers(sessionId)) { - this.emit(subscriber, { - type: 'reset', - sessionId, - reset: reason, - page, - fence, - ...(backgroundTasks !== undefined ? { backgroundTasks } : {}), - ...this.activityField(sessionId) - }) - subscriber.cursor = page.liveCursor ?? page.window.nextCursor - subscriber.fence = fence - } - this.hooks.onJournalPublished?.(sessionId, journal) + this.replay(sessionId, journal, fence, backgroundTasks, { type: 'reset', reset: reason }) } snapshot( @@ -160,14 +152,26 @@ export class AgentSessionSubscribers { journal: AgentSessionJournal, fence: number, backgroundTasks?: AgentSessionBackgroundTaskState | null + ): void { + this.replay(sessionId, journal, fence, backgroundTasks, { type: 'snapshot' }) + } + + private replay( + sessionId: string, + journal: AgentSessionJournal, + fence: number, + backgroundTasks: AgentSessionBackgroundTaskState | null | undefined, + frame: { type: 'snapshot' } | { type: 'reset'; reset: AgentJournalResetReason } ): void { const page = readAgentSessionHydrationPage(journal, fence) + const hostNow = this.now() for (const subscriber of this.subscribers(sessionId)) { this.emit(subscriber, { - type: 'snapshot', + ...frame, sessionId, page, fence, + hostNow, ...(backgroundTasks !== undefined ? { backgroundTasks } : {}), ...this.activityField(sessionId) }) @@ -178,18 +182,15 @@ export class AgentSessionSubscribers { } handoff(sessionId: string, fence: number, handoff: AgentSessionHandoffStatus): void { + const hostNow = this.now() for (const subscriber of this.subscribers(sessionId)) { this.emit(subscriber, { type: 'batch', sessionId, - batch: { - cursor: subscriber.cursor, - items: [], - removedItemIds: [], - submissions: [] - }, + batch: emptyAgentSessionBatch(subscriber.cursor), fence, - handoff + handoff, + hostNow }) subscriber.fence = fence } @@ -200,18 +201,15 @@ export class AgentSessionSubscribers { state: AgentSessionBackgroundTaskState | null, fence: number ): void { + const hostNow = this.now() for (const subscriber of this.subscribers(sessionId)) { this.emit(subscriber, { type: 'batch', sessionId, - batch: { - cursor: subscriber.cursor, - items: [], - removedItemIds: [], - submissions: [] - }, + batch: emptyAgentSessionBatch(subscriber.cursor), fence, - backgroundTasks: state + backgroundTasks: state, + hostNow }) subscriber.fence = fence } @@ -224,6 +222,7 @@ export class AgentSessionSubscribers { private deliver( subscriber: Subscriber, journal: AgentSessionJournal, + hostNow: number, handoff?: AgentSessionHandoffStatus, emitCheckpoint = false, backgroundTasks?: AgentSessionBackgroundTaskState | null, @@ -249,6 +248,7 @@ export class AgentSessionSubscribers { reset: result.reset, page, fence: subscriber.fence, + hostNow, ...(handoff ? { handoff } : {}), ...(backgroundTasks !== undefined ? { backgroundTasks } : {}), ...(publishedActivity !== undefined ? { activity: publishedActivity } : {}) @@ -266,13 +266,9 @@ export class AgentSessionSubscribers { this.emit(subscriber, { type: 'batch', sessionId: subscriber.sessionId, - batch: { - cursor: page.window.nextCursor, - items: [], - removedItemIds: [], - submissions: [] - }, + batch: emptyAgentSessionBatch(page.window.nextCursor), fence: subscriber.fence, + hostNow, ...(handoff ? { handoff } : {}), ...(backgroundTasks !== undefined ? { backgroundTasks } : {}), ...(publishedActivity !== undefined ? { activity: publishedActivity } : {}) @@ -290,6 +286,7 @@ export class AgentSessionSubscribers { submissions: page.submissions }, fence: subscriber.fence, + hostNow, ...(handoff ? { handoff } : {}), ...(backgroundTasks !== undefined ? { backgroundTasks } : {}), ...(publishedActivity !== undefined ? { activity: publishedActivity } : {}) @@ -301,6 +298,8 @@ export class AgentSessionSubscribers { } } + private now = (): number => this.hooks.now?.() ?? Date.now() + private isActive = (subscriber: Subscriber): boolean => this.bySession.get(subscriber.sessionId)?.get(subscriber.id) === subscriber diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-surface-lifetime.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-surface-lifetime.test.ts index 230e39cdaef..678b790ed4e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-surface-lifetime.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-surface-lifetime.test.ts @@ -8,11 +8,13 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' import type { AgentSessionOwnerProbe } from '../../../shared/agent-session-lease-adjudication' +import { hasUnansweredStructuredAgentSessionDispatch } from '../../../shared/structured-agent-session-projection' import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' import type { AgentSessionMutationEnvelope, AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' +import { AGENT_SESSION_UNATTACHED_REFUSAL_CODE } from '../../../shared/structured-agent-session-read-refusal' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import type { StructuredAgentSessionEventSink } from './structured-agent-session-event-sink' @@ -179,6 +181,25 @@ describe('a chat that closes', () => { expect(host.hasSession(SESSION)).toBe(true) }) + // The pane outlives the close by a few frames — a workspace delete closes the chats inside it + // while their panes are still mounted — so whatever a read raises in that window is what the user + // sees. This is the code the client narrows on to keep that window off the pane; a host that + // starts raising a different one there puts the red error back. + it('answers a read from the pane that outlived it with the code the client treats as transitional', async () => { + await attach() + await host.hold(SESSION, SURFACE) + + await host.close(SESSION) + + expect(host.hasSession(SESSION)).toBe(false) + expect(() => host.history({ sessionId: SESSION, direction: 'tail' })).toThrow( + AGENT_SESSION_UNATTACHED_REFUSAL_CODE + ) + expect(() => + host.subscribe({ id: 'sub-1', sessionId: SESSION, emit: () => undefined }) + ).toThrow(AGENT_SESSION_UNATTACHED_REFUSAL_CODE) + }) + it('does not lose the session to a release the client sent twice', async () => { await attach() await host.hold(SESSION, SURFACE) @@ -192,6 +213,28 @@ describe('a chat that closes', () => { expect(closeSession).not.toHaveBeenCalled() expect(host.hasSession(SESSION)).toBe(true) }) + + it('releases a compatibility wait when the session is evicted', async () => { + await attach() + dispatch.mockResolvedValueOnce({ state: 'admitted' }) + const body = hostTestMessage('pending until close') + const result = await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + expect(result).toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'pending' } } + }) + if (!result.ok) { + throw new Error('send was refused') + } + const settlement = host.waitForSendSettlement(SESSION, result.value.clientMessageId) + + await host.close(SESSION) + + await expect(settlement).resolves.toBeUndefined() + }) }) describe('a session with a turn in flight', () => { @@ -273,6 +316,38 @@ describe('a session evicted and opened again', () => { }) describe('an unexpected provider exit', () => { + it('publishes terminal settlement to a waiting older client', async () => { + await attach() + dispatch.mockResolvedValueOnce({ state: 'admitted' }) + const body = hostTestMessage('pending until provider exit') + const result = await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + expect(result).toMatchObject({ + ok: true, + value: { submission: { dispatchState: 'pending' } } + }) + if (!result.ok) { + throw new Error('send was refused') + } + const settlement = host.waitForSendSettlement(SESSION, result.value.clientMessageId) + const exitedFence = store.getRecord(SESSION)?.lease.runtimeFence ?? 0 + + await host.handleAdapterEvent({ + type: 'ended', + sessionId: SESSION, + reason: 'provider exited', + cause: 'unexpected-exit', + fence: exitedFence, + acquisitionGeneration: 'generation-1' + }) + + await expect(settlement).resolves.toMatchObject({ + value: { submission: { dispatchState: 'unknown' } } + }) + }) + it('turns a journal sink failure into observed-exit settlement and lease release', async () => { await attach() const session = ( @@ -334,6 +409,11 @@ describe('an unexpected provider exit', () => { acquisitionGeneration: 'generation-1' }) + const recoveredHistory = host.history({ sessionId: SESSION, direction: 'tail' }) + expect( + recoveredHistory.ok && + hasUnansweredStructuredAgentSessionDispatch(recoveredHistory.page.submissions) + ).toBe(false) expect(acquire).toHaveBeenCalledTimes(2) expect(dispatch).toHaveBeenCalledOnce() expect(store.getRecord(SESSION)?.lease).toMatchObject({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts index 3bd98a61735..35ca56a451c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts @@ -6,12 +6,20 @@ // row the next attach settles as `unknown`, whereas the reverse would lose a // turn the provider already accepted. -import type { AgentJournalMessageItem } from '../../../shared/agent-session-journal-types' +import type { + AgentJournalMessageItem, + AgentJournalSubmission +} from '../../../shared/agent-session-journal-types' import type { AgentSessionCancelResult, AgentSessionSendResult, AgentSessionWireRefusal } from '../../../shared/agent-session-wire' +import { + DISPATCH_DOUBT_PERSISTENCE_FAILED, + DISPATCH_DOUBT_RETRY_IN_PROGRESS, + dispatchDoubtProvesUndelivered +} from '../agent-session-journal/journal-dispatch-doubt-reasons' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { AgentSessionDispatchOutcome, @@ -73,6 +81,16 @@ async function appendStatus( ctx.publish() } +/** + * Whether a user's Retry may put this message on the wire again: only where the + * recorded doubt proves the frame never reached a provider. Everything else + * replays the recorded outcome instead — one message reached the model five + * times through this path. Orca never re-sends on its own either way. + */ +function retryWouldRedeliver(existing: AgentJournalSubmission | undefined): boolean { + return existing?.dispatchState === 'unknown' && dispatchDoubtProvesUndelivered(existing.reason) +} + export async function performSend( ctx: AgentSessionTurnContext, input: { @@ -88,18 +106,48 @@ export async function performSend( if (existing && existing.payloadFingerprint !== input.payloadFingerprint) { return invalid(`Message id ${input.clientMessageId} was already used for another send.`) } - if (existing && !(input.retryUnknown && existing.dispatchState === 'unknown')) { + const redeliver = input.retryUnknown === true && retryWouldRedeliver(existing) + if (existing && !redeliver) { return { ok: true, value: { clientMessageId: input.clientMessageId, submission: existing } } } - if (!(input.retryUnknown && existing?.dispatchState === 'unknown')) { + if (!redeliver) { await ctx.journal.appendSubmission({ ...input, fence: ctx.fence }) ctx.publish() + } else { + // Retry resumes work without moving or duplicating the original message. + await ctx.journal.resolveDispatch({ + clientMessageId: input.clientMessageId, + state: 'unknown', + reason: DISPATCH_DOUBT_RETRY_IN_PROGRESS, + fence: ctx.fence + }) + ctx.publish() } const outcome = await dispatchSafely(ctx, input.clientMessageId, input.body) + // A first admission needs no dispatch row: the submission is already pending. + // A retry must durably clear the old doubt so clients do not mistake a + // successful re-admission for a refused redelivery. + if (outcome.state === 'admitted') { + if (redeliver) { + await ctx.journal.resolveDispatch({ + clientMessageId: input.clientMessageId, + state: 'pending', + fence: ctx.fence + }) + } + ctx.publish() + return { + ok: true, + value: { + clientMessageId: input.clientMessageId, + submission: requireSubmission(ctx, input.clientMessageId) + } + } + } try { await ctx.journal.resolveDispatch( outcome.state === 'accepted' @@ -123,9 +171,8 @@ export async function performSend( await ctx.journal.resolveDispatch({ clientMessageId: input.clientMessageId, state: 'unknown', - reason: 'dispatch_result_persistence_failed', - fence: ctx.fence, - recovered: true + reason: DISPATCH_DOUBT_PERSISTENCE_FAILED, + fence: ctx.fence }) } catch { // Nothing further to record; the pending row is settled on the next attach. @@ -134,14 +181,26 @@ export async function performSend( throw error } ctx.publish() + return { + ok: true, + value: { + clientMessageId: input.clientMessageId, + submission: requireSubmission(ctx, input.clientMessageId) + } + } +} +function requireSubmission( + ctx: AgentSessionTurnContext, + clientMessageId: string +): AgentJournalSubmission { const submission = ctx.journal .submissions() - .find((entry) => entry.clientMessageId === input.clientMessageId) + .find((entry) => entry.clientMessageId === clientMessageId) if (!submission) { throw new Error('agent_session_submission_lost') } - return { ok: true, value: { clientMessageId: input.clientMessageId, submission } } + return submission } export async function performCancel( diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts index 084ffc7547d..813e2f8f3e8 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.test.ts @@ -1,6 +1,9 @@ import { describe, expect, it, vi } from 'vitest' +import { agentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import type { AgentJournalRenderItem } from '../../../shared/agent-session-journal-types' import type { AgentSessionRecord } from '../../../shared/agent-session-record' import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' +import { retryLoadedStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import { isStructuredAgentSessionRecoveryTicketCurrent, settleUnexpectedStructuredAgentSessionExit, @@ -42,14 +45,124 @@ function recoveryContext(input: { } as never } +function lifecycleItem( + turnId: string, + sequence: number, + turnLifecycle: { state: 'running' | 'completed'; startedAt: number; completedAt?: number } +): AgentJournalRenderItem { + return { + itemId: agentJournalItemKey({ provider: 'codex', threadId: 'thread-1', turnId, ordinal: 0 }), + revision: 1, + sequence, + observedAt: sequence, + body: { kind: 'turn', turnId, ...turnLifecycle } + } +} + describe('provider-exit recovery tickets', () => { - it('uses the fallback when the one-shot translator admission was rejected', async () => { - const appendLifecycleBatch = vi.fn(async () => ({ epoch: 'epoch-1', sequence: 1 })) + it.each([undefined, 2_000])('keeps exit receipt %s on retry', async (observedAt) => { + let now = observedAt === undefined ? 2_000 : 30_000 + let record = { + lease: { + handoffStage: null, + runtimeFence: 7, + runtimeKind: 'native', + claimStatus: 'live', + ownerProcess: 'provider', + reservedSpawnToken: null, + processlessAt: null + } + } as unknown as AgentSessionRecord + const store = { + getRecord: () => record, + transitionHandoff: async ( + _sessionId: string, + transition: (current: AgentSessionRecord) => AgentSessionRecord + ) => (record = transition(record)) + } + const appendLifecycleBatch = vi + .fn() + .mockRejectedValueOnce(new Error('journal unavailable')) + .mockResolvedValue({ epoch: 'epoch-1', sequence: 2 }) const session = { hasProviderChild: true, fence: 7, acquisitionGeneration: GENERATION, - journal: { snapshot: () => ({ items: [] }), appendLifecycleBatch } + journal: { + snapshot: () => ({ + items: [lifecycleItem('turn-1', 1, { state: 'running', startedAt: 1_000 })] + }), + appendLifecycleBatch, + markPendingSubmissionsUnknown: vi.fn(async () => []) + } + } as unknown as StructuredAgentSessionHostSession + + await settleUnexpectedStructuredAgentSessionExit( + { + store, + sessions: new Map([[SESSION, session]]), + flushLifecycle: async () => { + now = 60_000 + return { ok: false, error: new Error('sink unavailable') } + }, + publishFence: vi.fn(), + hasResumeCapableHolder: () => true, + serialize: async (_sessionId, task) => task(), + now: () => now + } as never, + { + type: 'ended', + sessionId: SESSION, + reason: 'provider exited', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: GENERATION, + observedAt + } + ) + expect(record.lease.settlementRetryRequired).toBe(true) + expect(record.lease.deathEvidence?.observedAt).toBe(2_000) + expect(record.lease.lastRenewedAt).toBe(60_000) + expect(record.updatedAt).toBe(60_000) + + now = 120_000 + await expect( + retryLoadedStructuredAgentSessionSettlement({ + deps: { store } as never, + sessionId: SESSION, + session: { journal: session.journal, fence: 8, acquisitionGeneration: null }, + now: () => now + }) + ).resolves.toBe(true) + expect(appendLifecycleBatch.mock.calls.at(-1)?.[0].mutations).toContainEqual( + expect.objectContaining({ + body: { + kind: 'turn', + turnId: 'turn-1', + state: 'interrupted', + startedAt: 1_000, + completedAt: 2_000 + } + }) + ) + expect(record.lease.settlementRetryRequired).toBeUndefined() + }) + + it('uses the fallback when the one-shot translator admission was rejected, revising the running turn in place', async () => { + const appendLifecycleBatch = vi.fn(async () => ({ epoch: 'epoch-1', sequence: 3 })) + const items = [ + lifecycleItem('turn-1', 1, { state: 'completed', startedAt: 10, completedAt: 20 }), + lifecycleItem('turn-2', 2, { state: 'running', startedAt: 30 }) + ] + const session = { + hasProviderChild: true, + fence: 7, + acquisitionGeneration: GENERATION, + journal: { + snapshot: () => ({ items }), + appendLifecycleBatch, + markPendingSubmissionsUnknown: vi.fn(async () => []) + } } as unknown as StructuredAgentSessionHostSession const store = { getRecord: () => ({ @@ -74,7 +187,7 @@ describe('provider-exit recovery tickets', () => { publishFence: vi.fn(), hasResumeCapableHolder: () => true, serialize: async (_sessionId, task) => task(), - now: () => 1 + now: () => 1_234 } as never, { type: 'ended', @@ -88,8 +201,90 @@ describe('provider-exit recovery tickets', () => { ) expect(result).toMatchObject({ settlementRetryRequired: false, releasedFence: 8 }) - expect(appendLifecycleBatch).toHaveBeenCalledOnce() + expect(session.journal.markPendingSubmissionsUnknown).toHaveBeenCalledWith( + 7, + 'provider_exited_before_acknowledgement' + ) expect(session.hasProviderChild).toBe(false) + // The running row is revised to interrupted at exit receipt, never tombstoned. + expect(appendLifecycleBatch).toHaveBeenCalledExactlyOnceWith({ + settlementId: `provider-exit:${SESSION}:7:${GENERATION}`, + fence: 7, + recovered: true, + mutations: [ + { + kind: 'item', + identity: { + provider: 'orca', + clientMessageId: `provider-exit:${SESSION}:7:${GENERATION}` + }, + body: { kind: 'status', text: 'Provider exited: provider exited' } + }, + { + kind: 'item', + identity: { provider: 'codex', threadId: 'thread-1', turnId: 'turn-2', ordinal: 0 }, + body: { + kind: 'turn', + turnId: 'turn-2', + state: 'interrupted', + startedAt: 30, + completedAt: 1_234 + } + } + ] + }) + }) + + it('settles a submission the dead child never acknowledged', async () => { + const markPendingSubmissionsUnknown = vi.fn(async () => ['client-1']) + const session = { + hasProviderChild: true, + fence: 7, + acquisitionGeneration: GENERATION, + journal: { + snapshot: () => ({ items: [] }), + appendLifecycleBatch: vi.fn(async () => ({ epoch: 'epoch-1', sequence: 1 })), + markPendingSubmissionsUnknown + } + } as unknown as StructuredAgentSessionHostSession + + await settleUnexpectedStructuredAgentSessionExit( + { + store: { + getRecord: () => ({ + lease: { + handoffStage: null, + runtimeFence: 7, + runtimeKind: 'native', + claimStatus: 'live', + ownerProcess: 'provider', + reservedSpawnToken: null, + processlessAt: null + } + }), + transitionHandoff: async () => ({ lease: { runtimeFence: 8 } }) + }, + sessions: new Map([[SESSION, session]]), + flushLifecycle: async () => ({ ok: true }), + publishFence: vi.fn(), + hasResumeCapableHolder: () => true, + serialize: async (_sessionId, task: () => Promise) => task(), + now: () => 1 + } as never, + { + type: 'ended', + sessionId: SESSION, + reason: 'provider exited', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: GENERATION + } + ) + + expect(markPendingSubmissionsUnknown).toHaveBeenCalledWith( + 7, + 'provider_exited_before_acknowledgement' + ) }) it('does not release or reacquire while terminal settlement retry is still failing', async () => { @@ -98,6 +293,7 @@ describe('provider-exit recovery tickets', () => { fence: 7, acquisitionGeneration: GENERATION, journal: { + markPendingSubmissionsUnknown: vi.fn(async () => []), snapshot: () => ({ items: [] }), appendLifecycleBatch: vi.fn(async () => { throw new Error('journal still unavailable') diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts index ddc4af9f2d3..c4b32f71647 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-unexpected-exit.ts @@ -1,4 +1,8 @@ import { parseAgentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import { + runningTurnLifecycleRevisions, + type StructuredAgentSessionTurnVerdict +} from './structured-agent-session-stale-turn-verdict' import type { AgentJournalItemBody, AgentJournalRenderItem @@ -47,6 +51,8 @@ export async function settleUnexpectedStructuredAgentSessionExit( return null } const unexpectedEvent = event as UnexpectedExitLifecycleEvent + // Receipt of the exit is the one end time the host may record for a running turn. + const observedAt = event.observedAt ?? context.now() return context.serialize(unexpectedEvent.sessionId, async () => { const session = context.sessions.get(unexpectedEvent.sessionId) if ( @@ -81,12 +87,22 @@ export async function settleUnexpectedStructuredAgentSessionExit( settlementRetryRequired = true context.onBarrierError?.(unexpectedEvent.sessionId, error) } + try { + await session.journal.markPendingSubmissionsUnknown( + session.fence, + 'provider_exited_before_acknowledgement' + ) + } catch (error) { + settlementRetryRequired = true + context.onBarrierError?.(unexpectedEvent.sessionId, error) + } if (unexpectedEvent.settlementRetryRequired || settlementRetryRequired) { const retried = await retryUnexpectedExitSettlement({ context, event: unexpectedEvent, session, - stableSettlementId + stableSettlementId, + verdict: { state: 'interrupted', completedAt: observedAt } }) if (!retried) { settlementFailed = true @@ -106,6 +122,7 @@ export async function settleUnexpectedStructuredAgentSessionExit( expectedAcquisitionGeneration: unexpectedEvent.acquisitionGeneration, acquisitionGeneration: session.acquisitionGeneration, now: context.now(), + exitObservedAt: observedAt, ...(settlementFailed ? { settlementRetry: { @@ -168,12 +185,18 @@ export async function retryUnexpectedExitSettlement(input: { event: UnexpectedExitLifecycleEvent session: Pick stableSettlementId: string + verdict: StructuredAgentSessionTurnVerdict }): Promise { try { + await input.session.journal.markPendingSubmissionsUnknown( + input.session.fence, + 'provider_exited_before_acknowledgement' + ) const mutations = unexpectedExitFallbackMutations( input.event, input.session, - input.stableSettlementId + input.stableSettlementId, + input.verdict ) for (const chunk of partitionJournalLifecycleMutations(input.stableSettlementId, mutations)) { await input.session.journal.appendLifecycleBatch({ @@ -193,11 +216,12 @@ export async function retryUnexpectedExitSettlement(input: { function unexpectedExitFallbackMutations( event: UnexpectedExitLifecycleEvent, session: Pick, - stableSettlementId: string + stableSettlementId: string, + verdict: StructuredAgentSessionTurnVerdict ): JournalLifecycleMutationInput[] { const mutations: JournalLifecycleMutationInput[] = [] - const tombstones: JournalLifecycleMutationInput[] = [] - for (const item of session.journal.snapshot().items) { + const { items } = session.journal.snapshot() + for (const item of items) { const identity = parseAgentJournalItemKey(item.itemId) if (!identity) { continue @@ -206,16 +230,14 @@ function unexpectedExitFallbackMutations( if (terminal) { mutations.push({ kind: 'item', identity, body: terminal }) } - if (item.body.kind === 'status' && item.body.turnLifecycle?.state === 'running') { - tombstones.push({ kind: 'tombstone', identity }) - } } mutations.push({ kind: 'item', identity: { provider: 'orca', clientMessageId: stableSettlementId }, body: { kind: 'status', text: boundJournalStatusText(`Provider exited: ${event.reason}`) } }) - mutations.push(...tombstones) + // Lifecycle rows settle last, in place: the turn's endpoints outlive the child. + mutations.push(...runningTurnLifecycleRevisions(items, verdict)) return mutations } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-wedged-profile-migration.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-wedged-profile-migration.test.ts index dfeaa650129..1305c23313b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-wedged-profile-migration.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-wedged-profile-migration.test.ts @@ -16,6 +16,7 @@ import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' import { evaluateAgentSessionAcquisition } from '../../../shared/agent-session-lease-adjudication' import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import { readAgentJournalTurn } from '../../../shared/agent-session-turn-record' import type { AgentSessionClaimStatus, AgentSessionHandoffStage, @@ -197,16 +198,19 @@ async function seedRunningTurn(provider: 'codex' | 'claude' = 'codex'): Promise< provider === 'codex' ? { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 0 } : { provider: 'claude', sessionId: 'provider-session-alpha-1', uuid: 'uuid-running' }, - { - kind: 'status', - text: 'Agent is working...', - turnLifecycle: { turnId: 'turn-1', state: 'running' } - }, + { kind: 'turn', turnId: 'turn-1', state: 'running', startedAt: NOW - 5_000 }, { fence: 13 } ) await journal.close() } +function turnLifecycle(turnId: string) { + const item = restoredJournal() + .snapshot() + .items.find((candidate) => readAgentJournalTurn(candidate.body)?.turnId === turnId) + return item ? { ...readAgentJournalTurn(item.body), recovered: item.recovered } : null +} + function restoredJournal(): AgentSessionJournal { const restored = ( host as unknown as { sessions: Map } @@ -289,6 +293,13 @@ describe('already-wedged profiles become usable on load', () => { expect(acquire).toHaveBeenCalledOnce() expect(activeStructuredAgentSessionTurnId(restoredJournal().snapshot().items)).toBe(null) + // A pid probe proved the owner gone; nobody saw it exit, so the turn has no end. + expect(turnLifecycle('turn-1')).toEqual({ + turnId: 'turn-1', + state: 'unverifiable', + startedAt: NOW - 5_000, + recovered: true + }) expect(store.getRecord(SESSION)?.lease).toMatchObject({ claimStatus: 'live', handoffStage: null, @@ -314,6 +325,14 @@ describe('already-wedged profiles become usable on load', () => { expect(acquire).toHaveBeenCalledOnce() expect(activeStructuredAgentSessionTurnId(restoredJournal().snapshot().items)).toBe(null) + // The exit was observed, so its receipt is the turn's end. + expect(turnLifecycle('turn-1')).toEqual({ + turnId: 'turn-1', + state: 'interrupted', + startedAt: NOW - 5_000, + completedAt: NOW - 1_000, + recovered: true + }) expect(store.getRecord(SESSION)?.lease).toMatchObject({ claimStatus: 'live', handoffStage: null, @@ -322,6 +341,53 @@ describe('already-wedged profiles become usable on load', () => { }) }) + it('marks a running turn left behind by a released lease unverifiable on a cold acquire', async () => { + // No settlement latch: the record was released cleanly, but the journal still says a turn is + // running. The child that wrote it is gone and nothing observed its exit. + await seedStore(wedgedRecord({ claimStatus: 'released', handoffStage: null })) + await seedRunningTurn() + openHost() + + expect(await host.attach(CALLER, hostTestAttachParams(13))).toMatchObject({ ok: true }) + + expect(acquire).toHaveBeenCalledOnce() + expect(turnLifecycle('turn-1')).toEqual({ + turnId: 'turn-1', + state: 'unverifiable', + startedAt: NOW - 5_000, + recovered: true + }) + expect( + restoredJournal() + .snapshot() + .items.some( + (item) => item.body.kind === 'status' && item.body.text.startsWith('Provider exited') + ) + ).toBe(false) + }) + + it("leaves the live generation's running turn alone on a re-attach", async () => { + await seedStore(wedgedRecord({ claimStatus: 'released', handoffStage: null })) + // The child this host spawns stays provably alive across the second attach. + openHost({ + probeOwner: async () => ({ outcome: 'identity-matched', matchedOn: ['spawn-token'] }) + }) + const params = hostTestAttachParams(13) + expect(await host.attach(CALLER, params)).toMatchObject({ ok: true }) + const fence = store.getRecord(SESSION)!.lease.runtimeFence + await restoredJournal().appendItem( + { provider: 'codex', threadId: THREAD, turnId: 'turn-2', ordinal: 0 }, + { kind: 'turn', turnId: 'turn-2', state: 'running', startedAt: NOW }, + { fence } + ) + + // A reconnecting client replays its attach; the same operation admits the live owner. + expect(await host.attach(CALLER, params)).toMatchObject({ ok: true, replayed: true }) + + expect(acquire).toHaveBeenCalledOnce() + expect(turnLifecycle('turn-2')).toEqual({ turnId: 'turn-2', state: 'running', startedAt: NOW }) + }) + it('re-adjudicates a conflicted manual-recovery record whose owner is provably gone', async () => { // A crash can leave a conflicted current-schema row in manual recovery; positive death proof // must make it acquirable again without discarding the provider handle. diff --git a/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts b/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts index 42d03f46b5b..5a3370c910f 100644 --- a/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-conversation-command-admission.test.ts @@ -45,4 +45,26 @@ describe('conversationCommandBlocked background tasks', () => { ) expect(blocked).toBe('Wait for background tasks to finish before using this command.') }) + + it('still refuses on the open turn, not on the work the strip now shows', () => { + // The strip reports subagents while a turn runs. That must not change which + // refusal the user sees: an open turn already refuses, and it refuses first, + // so a live fan-out never re-labels the reason or blocks anything new. + const ctx = contextWith({ state: 'monitoring', supportsTaskStop: true }) + ctx.journal.snapshot = () => + ({ + items: [ + { + id: 'turn-1', + body: { + kind: 'status', + turnLifecycle: { turnId: 'turn-1', state: 'running' } + } + } + ] + }) as unknown as ReturnType + expect(conversationCommandBlocked(ctx, RECORD)).toBe( + 'Wait for the current turn to finish before using this command.' + ) + }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.test.ts b/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.test.ts index 3e6b70c2bcc..e93aef352fe 100644 --- a/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.test.ts @@ -34,6 +34,39 @@ describe('rewind recovery of newer durable records', () => { text: JSON.stringify(status) }) }) + it.each(['interrupted', 'unverifiable'] as const)( + 'keeps a %s turn and its recorded endpoints', + (state) => { + const status = { + kind: 'status' as const, + text: 'Working', + turnLifecycle: { + turnId: 'turn', + state, + startedAt: 10, + ...(state === 'interrupted' ? { completedAt: 20 } : {}) + } + } + expect(restoreRewindJournalBody(status)).toEqual(status) + } + ) + it('accepts a canonical turn body with a known state and keeps an unknown one as evidence', () => { + const turn = { + kind: 'turn' as const, + turnId: 'turn', + state: 'completed', + userItemId: 'codex:thread:turn:0', + startedAt: 10, + completedAt: 20, + durationMs: 10 + } + expect(restoreRewindJournalBody(turn)).toEqual(turn) + const unknown = { ...turn, state: 'future-state' } + expect(restoreRewindJournalBody(unknown)).toEqual({ + kind: 'status', + text: JSON.stringify(unknown) + }) + }) it('does not reject a saved recovery prefix over a newer refusal reason', () => { expect( AgentSessionRewindRecordSchema.safeParse({ diff --git a/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.ts b/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.ts index b1c32633af4..e23058a7d7b 100644 --- a/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.ts +++ b/src/main/native-chat/agent-session-wire/structured-rewind-journal-body.ts @@ -1,5 +1,8 @@ import { isAdmissibleAgentJournalItemBody } from '../../../shared/agent-session-journal-schemas' -import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' +import { + AGENT_JOURNAL_TURN_LIFECYCLE_STATES, + type AgentJournalItemBody +} from '../../../shared/agent-session-journal-types' import type { AgentSessionRewindRecord } from '../../../shared/agent-session-rewind' import { NATIVE_CHAT_ROLES } from '../../../shared/native-chat-types' @@ -47,10 +50,10 @@ export function restoreRewindJournalBody(body: StoredBody): AgentJournalItemBody ) { normalized = fallback() } else if ( - body.kind === 'status' && - body.turnLifecycle && - body.turnLifecycle.state !== 'running' && - body.turnLifecycle.state !== 'completed' + (body.kind === 'turn' || (body.kind === 'status' && body.turnLifecycle)) && + !(AGENT_JOURNAL_TURN_LIFECYCLE_STATES as readonly string[]).includes( + body.kind === 'turn' ? body.state : body.turnLifecycle!.state + ) ) { normalized = fallback() } diff --git a/src/main/native-chat/agent-session-wire/structured-rewind-recovery.ts b/src/main/native-chat/agent-session-wire/structured-rewind-recovery.ts index 41ccab4ae28..8f5a8942a86 100644 --- a/src/main/native-chat/agent-session-wire/structured-rewind-recovery.ts +++ b/src/main/native-chat/agent-session-wire/structured-rewind-recovery.ts @@ -1,4 +1,5 @@ import { restoreRewindJournalBody } from './structured-rewind-journal-body' +import { isRetainedTurnRow, mergeRetainedTurnRows } from './structured-rewind-retained-turns' import { isDeepStrictEqual } from 'node:util' import { agentJournalItemKey, @@ -60,7 +61,10 @@ export async function recoverStructuredRewind( } throw new Error(`agent_session_rewind:${recovered?.reason ?? 'outcome-unknown'}`) } - const expectedItems = new Set(rewind.retained.map((item) => item.itemId)) + // Turn rows are the host's, never the provider's; the proof covers provider items only. + const expectedItems = new Set( + rewind.retained.filter((item) => !isRetainedTurnRow(item)).map((item) => item.itemId) + ) const observedItems = new Set() for (const { identity } of recovered.items) { const itemId = agentJournalItemKey(identity) @@ -76,11 +80,14 @@ export async function recoverStructuredRewind( if (observedItems.size !== expectedItems.size) { throw new Error('agent_session_rewind:proof-mismatch') } - const retained = recovered.items.map(({ identity, body }) => ({ - itemId: agentJournalItemKey(identity), - body, - observedAt: now() - })) + const retained = mergeRetainedTurnRows( + rewind.retained, + recovered.items.map(({ identity, body }) => ({ + itemId: agentJournalItemKey(identity), + body, + observedAt: now() + })) + ) if ( retained.length > 10_000 || Buffer.byteLength(JSON.stringify(retained), 'utf8') > AGENT_SESSION_HISTORY_MAX_PAGE_BYTES diff --git a/src/main/native-chat/agent-session-wire/structured-rewind-retained-turns.ts b/src/main/native-chat/agent-session-wire/structured-rewind-retained-turns.ts new file mode 100644 index 00000000000..aa0753b502c --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-rewind-retained-turns.ts @@ -0,0 +1,34 @@ +// The Codex preflight returns provider items only. The host's turn rows are its own record, so a +// rewind that takes the provider's list as the new epoch would drop every duration before the +// boundary unless those rows are spliced back beside the item each one followed. + +import type { AgentJournalItemBody } from '../../../shared/agent-session-journal-types' +import type { AgentSessionRewindRecord } from '../../../shared/agent-session-rewind' +import { readAgentJournalTurn } from '../../../shared/agent-session-turn-record' + +type RetainedRow = AgentSessionRewindRecord['retained'][number] + +export function isRetainedTurnRow(item: Pick): boolean { + return readAgentJournalTurn(item.body as AgentJournalItemBody) !== null +} + +/** `reference` fixes where each turn row sits; the provider items are the spine and keep their + * own order, including turns the local journal never saw. */ +export function mergeRetainedTurnRows( + reference: readonly RetainedRow[], + providerItems: readonly RetainedRow[] +): RetainedRow[] { + const spineIndex = new Map(providerItems.map((item, index) => [item.itemId, index])) + const rowsAfter = new Map() + let anchor = -1 + for (const item of reference) { + if (!isRetainedTurnRow(item)) { + anchor = spineIndex.get(item.itemId) ?? anchor + } else if (!spineIndex.has(item.itemId)) { + rowsAfter.set(anchor, [...(rowsAfter.get(anchor) ?? []), item]) + } + } + const merged = [...(rowsAfter.get(-1) ?? [])] + providerItems.forEach((item, index) => merged.push(item, ...(rowsAfter.get(index) ?? []))) + return merged +} diff --git a/src/main/native-chat/transcript-line-decoders-codex.ts b/src/main/native-chat/transcript-line-decoders-codex.ts index 229ace2a461..ddc748dde03 100644 --- a/src/main/native-chat/transcript-line-decoders-codex.ts +++ b/src/main/native-chat/transcript-line-decoders-codex.ts @@ -92,10 +92,13 @@ function codexResponseItem( payload.type === 'custom_tool_call' ) { const name = extractString(payload.name) ?? 'tool' + const callId = extractString(payload.call_id) return { id, role: 'assistant', - blocks: [{ type: 'tool-call', name, input: codexCallInput(payload) }], + blocks: [ + { type: 'tool-call', name, input: codexCallInput(payload), ...(callId ? { callId } : {}) } + ], timestamp, source: 'transcript' } diff --git a/src/main/native-chat/transcript-reader-codex-history-mode.test.ts b/src/main/native-chat/transcript-reader-codex-history-mode.test.ts index 5d18bec8c85..9a9b2b76094 100644 --- a/src/main/native-chat/transcript-reader-codex-history-mode.test.ts +++ b/src/main/native-chat/transcript-reader-codex-history-mode.test.ts @@ -240,7 +240,7 @@ describe('Codex transcript history modes', () => { expect(call).toMatchObject({ id: 'call-1', role: 'assistant', - blocks: [{ type: 'tool-call', name: 'exec', input: 'pwd' }] + blocks: [{ type: 'tool-call', name: 'exec', input: 'pwd', callId: 'durable-call-1' }] }) expect(output).toMatchObject({ id: 'fallback-output', diff --git a/src/main/native-chat/transcript-reader.test.ts b/src/main/native-chat/transcript-reader.test.ts index ff48548804d..eb333cd8fda 100644 --- a/src/main/native-chat/transcript-reader.test.ts +++ b/src/main/native-chat/transcript-reader.test.ts @@ -67,7 +67,7 @@ describe('readNativeChatTranscript (claude)', () => { timestamp: '2026-06-01T10:05:00.000Z', message: { role: 'assistant', - content: [{ type: 'tool_use', name: 'Bash', input: { command: 'ls' } }] + content: [{ type: 'tool_use', id: 'tool-call-1', name: 'Bash', input: { command: 'ls' } }] } }) records.push({ @@ -98,7 +98,8 @@ describe('readNativeChatTranscript (claude)', () => { expect(toolCall?.blocks[0]).toEqual({ type: 'tool-call', name: 'Bash', - input: { command: 'ls' } + input: { command: 'ls' }, + callId: 'tool-call-1' }) const toolResult = result.messages.at(-1) diff --git a/src/main/native-chat/transcript-record-blocks.ts b/src/main/native-chat/transcript-record-blocks.ts index 6355df277e5..b82ff9672ca 100644 --- a/src/main/native-chat/transcript-record-blocks.ts +++ b/src/main/native-chat/transcript-record-blocks.ts @@ -81,7 +81,8 @@ function claudeContentBlock(record: Record): NativeChatBlock | } case 'tool_use': { const name = extractString(record.name) ?? 'tool' - return { type: 'tool-call', name, input: record.input } + const callId = extractString(record.id) + return { type: 'tool-call', name, input: record.input, ...(callId ? { callId } : {}) } } case 'tool_result': return toolResultBlock(record) diff --git a/src/main/notifications/desktop-away-state.test.ts b/src/main/notifications/desktop-away-state.test.ts new file mode 100644 index 00000000000..702539d6447 --- /dev/null +++ b/src/main/notifications/desktop-away-state.test.ts @@ -0,0 +1,25 @@ +import { expect, it } from 'vitest' +import { readDesktopAwayState } from './desktop-away-state' + +it.each([ + [179, false], + [180, true], + [181, true] +])('checks the three-minute boundary at %s seconds', (idle, away) => { + expect( + readDesktopAwayState({ getSystemIdleState: () => 'active', getSystemIdleTime: () => idle }) + ).toBe(away) +}) +it('allows immediate delivery when locked and fails open when presence cannot be read', () => { + expect( + readDesktopAwayState({ getSystemIdleState: () => 'locked', getSystemIdleTime: () => 0 }) + ).toBe(true) + expect( + readDesktopAwayState({ + getSystemIdleState: () => { + throw new Error('unsupported') + }, + getSystemIdleTime: () => 0 + }) + ).toBeUndefined() +}) diff --git a/src/main/notifications/desktop-away-state.ts b/src/main/notifications/desktop-away-state.ts new file mode 100644 index 00000000000..ccfbaeb5760 --- /dev/null +++ b/src/main/notifications/desktop-away-state.ts @@ -0,0 +1,20 @@ +export const MOBILE_NOTIFICATION_AWAY_SECONDS = 180 + +type IdleMonitor = { + getSystemIdleState(threshold: number): string + getSystemIdleTime(): number +} + +export function readDesktopAwayState(monitor: IdleMonitor): boolean | undefined { + try { + const state = monitor.getSystemIdleState(MOBILE_NOTIFICATION_AWAY_SECONDS) + if (state === 'locked' || state === 'idle') { + return true + } + const idle = monitor.getSystemIdleTime() + return Number.isFinite(idle) && idle >= 0 ? idle >= MOBILE_NOTIFICATION_AWAY_SECONDS : undefined + } catch { + // Unknown presence must not silence a phone. + return undefined + } +} diff --git a/src/main/observability/local-file-sink-memory.test.ts b/src/main/observability/local-file-sink-memory.test.ts index fef5b8d00dd..2c6644694b9 100644 --- a/src/main/observability/local-file-sink-memory.test.ts +++ b/src/main/observability/local-file-sink-memory.test.ts @@ -2,11 +2,7 @@ import { mkdtempSync, readFileSync, rmSync, statSync } from 'node:fs' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { - createLocalFileSink, - DROPPED_RECORD_TYPE, - type LocalFileSink -} from './local-file-sink' +import { createLocalFileSink, DROPPED_RECORD_TYPE, type LocalFileSink } from './local-file-sink' function parseLine(raw: string): Record { return JSON.parse(raw) as Record diff --git a/src/main/orca-profiles/profile-cloud-auth-config.ts b/src/main/orca-profiles/profile-cloud-auth-config.ts index 09cfd8dfc6b..4e0e75bdd8e 100644 --- a/src/main/orca-profiles/profile-cloud-auth-config.ts +++ b/src/main/orca-profiles/profile-cloud-auth-config.ts @@ -1,4 +1,9 @@ import { app } from 'electron' +import { + cleanCloudServiceUrl as cleanUrl, + cleanCloudServiceOrigin as cleanOrigin +} from '../../shared/cloud-service-url' +import { resolvePushGatewayOrigin } from '../runtime/push/push-gateway-origin' export type OrcaCloudAuthConfig = { apiBaseUrl: string @@ -30,39 +35,10 @@ function isPackagedOrcaBuild(): boolean { } } -function cleanUrl(value: string | undefined, allowLoopbackHttp: boolean): string | null { - const trimmed = value?.trim() - if (!trimmed) { - return null - } - try { - const parsed = new URL(trimmed) - const loopbackHost = - parsed.hostname === '127.0.0.1' || - parsed.hostname === 'localhost' || - parsed.hostname === '[::1]' - if (parsed.protocol !== 'https:' && !(loopbackHost && allowLoopbackHttp)) { - return null - } - return parsed.toString().replace(/\/$/, '') - } catch { - return null - } -} - function endpoint(baseUrl: string, path: string): string { return new URL(path, `${baseUrl}/`).toString() } -function cleanOrigin(value: string | undefined, allowLoopbackHttp: boolean): string | null { - const cleaned = cleanUrl(value, allowLoopbackHttp) - if (!cleaned) { - return null - } - const parsed = new URL(cleaned) - return parsed.pathname === '/' && !parsed.search && !parsed.hash ? parsed.origin : null -} - export function getOrcaCloudAuthConfig( env: NodeJS.ProcessEnv = process.env, packaged: boolean = isPackagedOrcaBuild() @@ -124,6 +100,18 @@ export function getOrcaCloudAuthConfig( } } +/** + * Where the host registers phones for background push. Deliberately outside + * OrcaCloudAuthConfig: the push gateway authenticates with the host keypair, so an + * accountless host reaches it on exactly the same path as a signed-in one. + */ +export function getOrcaPushGatewayUrl( + env: NodeJS.ProcessEnv = process.env, + packaged: boolean = isPackagedOrcaBuild() +): string { + return resolvePushGatewayOrigin(env, packaged) +} + export function allowsPlaintextOrcaCloudSession( env: NodeJS.ProcessEnv = process.env, packaged: boolean = isPackagedOrcaBuild() diff --git a/src/main/orca-profiles/profile-cloud-auth-status.test.ts b/src/main/orca-profiles/profile-cloud-auth-status.test.ts new file mode 100644 index 00000000000..7a28783e9a9 --- /dev/null +++ b/src/main/orca-profiles/profile-cloud-auth-status.test.ts @@ -0,0 +1,106 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { ActiveOrcaProfileState } from './profile-index-store' +import type { OrcaCloudSessionReadResult } from './profile-cloud-session-store' +import { getOrcaProfileAuthStatusFromProfile } from './profile-cloud-auth-status' + +const { readSession, configuration } = vi.hoisted(() => ({ + readSession: vi.fn<() => OrcaCloudSessionReadResult>(), + configuration: { configured: true } +})) + +vi.mock('./profile-cloud-session-store', () => ({ readOrcaCloudSession: readSession })) +vi.mock('./profile-cloud-auth-config', () => ({ + getOrcaCloudAuthConfig: () => configuration, + isOrcaCloudDevAuthEnabled: () => false +})) + +function activeProfile(linked: boolean): ActiveOrcaProfileState { + const profile: ActiveOrcaProfileState['profile'] = { + id: 'profile-1', + name: 'Personal', + avatar: { kind: 'initials', initials: 'P', color: 'neutral' }, + kind: linked ? 'cloud-linked' : 'local', + createdAt: 0, + updatedAt: 0, + lastOpenedAt: 0, + ...(linked + ? { + cloud: { + cloudProfileId: 'cloud-1', + userId: 'user-1', + email: 'a@example.com', + linkedAt: 0 + } + } + : {}) + } + return { + profile, + index: { schemaVersion: 1, activeProfileId: profile.id, profiles: [profile] }, + dataFile: '', + profileDirectory: '' + } +} + +const absentSessions: OrcaCloudSessionReadResult[] = [ + { status: 'missing', persistence: 'none' }, + { status: 'decrypt-failed', persistence: 'none', error: 'Cannot decrypt' }, + { status: 'unreadable', persistence: 'none', error: 'Permission denied' } +] + +describe('unexpected sign-out auth evidence', () => { + beforeEach(() => { + readSession.mockReset() + configuration.configured = true + }) + + it.each(absentSessions)('requires a preserved cloud link for $status credentials', (session) => { + readSession.mockReturnValue(session) + const linked = activeProfile(true) + expect(getOrcaProfileAuthStatusFromProfile(linked, '')).toMatchObject({ + state: 'reconnect-required', + cloud: linked.profile.cloud, + persistence: 'none', + credentialError: 'error' in session ? session.error : undefined + }) + readSession.mockClear() + const signedOut = getOrcaProfileAuthStatusFromProfile(activeProfile(false), '') + expect(signedOut.state).toBe('local') + expect(signedOut.cloud).toBeUndefined() + expect(readSession).not.toHaveBeenCalled() + }) + + it.each(absentSessions)( + 'keeps unconfigured linked profiles out of reconnect for $status', + (session) => { + configuration.configured = false + readSession.mockReturnValue(session) + expect(getOrcaProfileAuthStatusFromProfile(activeProfile(true), '').state).toBe( + 'unconfigured' + ) + expect(getOrcaProfileAuthStatusFromProfile(activeProfile(false), '').state).toBe( + 'unconfigured' + ) + } + ) + + it('treats a live memory-only session as connected, then reconnects after its loss', () => { + readSession.mockReturnValue({ + status: 'found', + persistence: 'memory-only', + session: { + accessToken: 'access', + refreshToken: 'refresh', + expiresAt: Date.now() + 60_000, + capabilities: { flags: {}, refreshedAt: 0 } + } + }) + const linked = activeProfile(true) + expect(getOrcaProfileAuthStatusFromProfile(linked, '')).toMatchObject({ + state: 'connected', + persistence: 'memory-only' + }) + readSession.mockReturnValue({ status: 'missing', persistence: 'none' }) + expect(getOrcaProfileAuthStatusFromProfile(linked, '').state).toBe('reconnect-required') + }) +}) diff --git a/src/main/orcad/orcad-entry.ts b/src/main/orcad/orcad-entry.ts index 3b10ca985e9..79e01a13163 100644 --- a/src/main/orcad/orcad-entry.ts +++ b/src/main/orcad/orcad-entry.ts @@ -5,7 +5,7 @@ * desktop uses, installs a PTY controller via `registerHeadlessPtyRuntime`, and * serves runtime RPC. See docs/design/node-only-runtime-backend.html. * - * Desktop UI surfaces stay uninstalled: no notifications, no renderer window. The + * Desktop UI surfaces stay uninstalled: no native notifications, no renderer window. The * renderer window is faked as a destroyed one because `registerPtyHandlers` takes a * non-null `BrowserWindow`. Browser automation is different — it is installed through * the runtime factory, but only when an Electron serve sidecar or an operator-supplied @@ -146,6 +146,11 @@ async function startOrcadRuntime( const { startOrcadDaemon, stopOrcadDaemon } = await import('./orcad-daemon-supervision') const { daemonOwnsFreshPersistentPtys } = await import('../daemon/daemon-init') const { collectOrcadHealth } = await import('./orcad-health') + // Why importable here: the store is an in-memory singleton whose module tree never reaches + // Electron, and its file paths come from `start()`, which orcad never calls. + const { agentHookServer } = await import('../agent-hooks/server') + const { DesktopPushService } = await import('../runtime/push/desktop-push-service') + const { resolvePushGatewayOrigin } = await import('../runtime/push/push-gateway-origin') const runtimeUserDataPath = getAppEnvironment().getPath('userData') initOrcaProfilePaths() @@ -180,7 +185,15 @@ async function startOrcadRuntime( // Why 'blocked': `'openable'` means a desktop window can be opened here, which is // what powers serve→desktop promotion. A Node host can never do that, and the // constructor's default would advertise it. - getDesktopWindowStatus: () => 'blocked' + getDesktopWindowStatus: () => 'blocked', + // Why here too and not only on the desktop: orcad serves `worktree.ps` and `agentSession.*`, + // so without these a headless host publishes its structured chats nowhere and lists no agents. + getAgentStatusSnapshot: () => + agentHookServer.getStatusSnapshot().filter((entry) => entry.providerSessionOnly !== true), + structuredAgentStatusSink: { + publish: (summary) => agentHookServer.ingestStructuredStatus(summary), + forget: (sessionId) => agentHookServer.dropStructuredStatus(sessionId) + } }) // Why the headless entry point rather than registerPtyHandlers directly: this is the @@ -213,6 +226,13 @@ async function startOrcadRuntime( ...(options.port !== undefined ? { wsPort: options.port, preferPinnedWsPort: true } : {}) }) await rpc.start() + const pushService = DesktopPushService.create({ + runtime, + runtimeRpc: rpc, + gatewayUrl: resolvePushGatewayOrigin(process.env, getAppEnvironment().isPackaged()) + }) + pushService?.start() + getAppEnvironment().onWillQuit(() => pushService?.stop()) console.error(`[orcad] ${describeOrcadBindExposure(bindHost)}`) const boundEndpoint = rpc.getWebSocketEndpoint() diff --git a/src/main/orcad/orcad-push-startup.test.ts b/src/main/orcad/orcad-push-startup.test.ts new file mode 100644 index 00000000000..a8fbbc9fe16 --- /dev/null +++ b/src/main/orcad/orcad-push-startup.test.ts @@ -0,0 +1,147 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, expect, it, vi } from 'vitest' +import { DeviceRegistry } from '../runtime/device-registry' +import { RuntimeMobileNotificationController } from '../runtime/runtime-mobile-notification-controller' +import { PushUnregisterOutbox } from '../runtime/push/push-unregister-outbox' +import { createPushHostKeypair } from '../runtime/push/push-host-challenge-fixtures' + +const state = vi.hoisted(() => ({ + root: '', + controller: null as RuntimeMobileNotificationController | null, + registry: null as DeviceRegistry | null, + rpcStarted: false, + register: vi.fn(async () => ({ ok: true, registrationId: 'headless-registration' })), + send: vi.fn(async () => ({ ok: true, results: [] })) +})) +vi.mock('./orcad-app-paths', () => ({ + resolveOrcadInstallRoot: () => state.root, + resolveOrcadPath: () => state.root, + resolveUserDataPath: () => state.root +})) +vi.mock('./orcad-browser-provider', () => ({ resolveOrcadBrowserProvider: async () => null })) +vi.mock('./orcad-instance-lock', () => ({ acquireOrcadInstanceLock: () => ({ release() {} }) })) +vi.mock('./orcad-daemon-supervision', () => ({ + startOrcadDaemon: async () => {}, + stopOrcadDaemon: async () => {} +})) +vi.mock('./orcad-health', () => ({ collectOrcadHealth: async () => ({}) })) +vi.mock('../daemon/daemon-init', () => ({ daemonOwnsFreshPersistentPtys: () => false })) +vi.mock('../ipc/pty', () => ({ + registerHeadlessPtyRuntime: async () => {}, + getLocalPtyProvider: () => null, + getSshPtyProvider: () => null +})) +vi.mock('../persistence/loading-store/store', () => ({ + Store: class { + getSettings() { + return {} + } + } +})) +vi.mock('../orca-profiles/profile-index-store', () => ({ + initOrcaProfilePaths() {}, + ensureActiveOrcaProfile: () => ({ dataFile: join(state.root, 'profile.json') }) +})) +vi.mock('../ssh/ssh-host-key-store', () => ({ initSshHostKeyStoreFile() {} })) +vi.mock('../server/serve-readiness', () => ({ + ServeReadinessPublisher: class { + async publish() {} + } +})) +vi.mock('../runtime/orca-runtime', () => ({ + OrcaRuntimeService: class { + getRuntimeId() { + return 'headless-runtime' + } + rehydrateClientHostedBrowserPages() {} + async refreshRestoredOrchestrationAuthority() {} + async reconcileLegacyWorkerTerminals() {} + setMobilePushRegistrar( + registrar: Parameters[0] + ) { + state.controller!.setPushRegistrar(registrar) + } + onNotificationDispatched( + listener: Parameters[0] + ) { + return state.controller!.onDispatched(listener) + } + } +})) +vi.mock('../runtime/runtime-rpc', () => ({ + OrcaRuntimeRpcServer: class { + async start() { + state.rpcStarted = true + } + async stop() { + state.rpcStarted = false + } + getWebSocketEndpoint() { + return null + } + getE2EEKeypair() { + expect(state.rpcStarted).toBe(true) + return createPushHostKeypair() + } + getDeviceRegistry() { + return state.registry + } + getPushUnregisterOutbox() { + return new PushUnregisterOutbox(state.root) + } + setOnPushUnregisterQueued() {} + } +})) +vi.mock('../runtime/push/push-gateway-client', () => ({ + PushGatewayClient: class { + registerDevice = state.register + send = state.send + async deleteDevice() { + return { deleted: true, retryable: false } + } + } +})) + +afterEach(() => { + rmSync(state.root, { recursive: true, force: true }) + vi.clearAllMocks() +}) + +it('starts push after RPC identity is available and stops dispatch on shutdown', async () => { + state.root = mkdtempSync(join(tmpdir(), 'orca-headless-push-')) + state.controller = new RuntimeMobileNotificationController() + state.registry = new DeviceRegistry(state.root) + const phone = state.registry.addDevice('headless-phone', 'mobile') + const { startOrcad } = await import('./orcad-entry') + const host = await startOrcad({ noPairing: true, json: true }) + try { + const result = await state.controller.registerPushDevice({ + deviceId: phone.deviceId, + platform: 'android', + token: 'test-token', + filter: { + onlyWhenDesktopAway: true + } + }) + expect(result).toMatchObject({ registered: true }) + expect(state.registry.getDevice(phone.deviceId)?.pushRegistration?.expiresAt).toBeGreaterThan( + Date.now() + ) + state.controller.dispatch({ + type: 'notification', + source: 'agent-task-complete', + title: 'QA', + body: 'QA' + }) + await new Promise((resolve) => setImmediate(resolve)) + expect(state.send).toHaveBeenCalledTimes(1) + } finally { + await host.stop() + } + expect(state.controller.getListenerCount()).toBe(0) + expect(await state.controller.registerPushDevice({} as never)).toMatchObject({ + registered: false + }) +}) diff --git a/src/main/pi/agent-status-extension-test-harness.ts b/src/main/pi/agent-status-extension-test-harness.ts index eec2b615017..810bc3d04d5 100644 --- a/src/main/pi/agent-status-extension-test-harness.ts +++ b/src/main/pi/agent-status-extension-test-harness.ts @@ -24,6 +24,7 @@ type FakeCurlChild = { } export type AgentStatusExtensionHarness = { + killMock: ReturnType fetchMock: ReturnType spawnMock: ReturnType spawnedChildren: FakeCurlChild[] @@ -57,6 +58,7 @@ export const AGENT_STATUS_EXTENSION_SELF_PID = 4242 export function createAgentStatusExtensionHarness(args: { kind: 'pi' | 'omp' | 'prime-agent' + killImpl?: (pid: number, signal: number) => void env?: Record pid?: number title?: string @@ -115,7 +117,9 @@ export function createAgentStatusExtensionHarness(args: { throw new Error(`unexpected require(${specifier})`) }) + const killMock = vi.fn(args.killImpl ?? (() => undefined)) const processMock = { + kill: killMock, env: { ...BASE_ENV, ...(args.kind === 'prime-agent' ? { PRIME_AGENT_INTERNAL_DAEMON_WORKER: '1' } : {}), @@ -172,6 +176,7 @@ export function createAgentStatusExtensionHarness(args: { return { fetchMock, + killMock, spawnMock, spawnedChildren, fsMock, diff --git a/src/main/pi/agent-status-handler-source.ts b/src/main/pi/agent-status-handler-source.ts index 9d02abbd78d..5a778a1c81f 100644 --- a/src/main/pi/agent-status-handler-source.ts +++ b/src/main/pi/agent-status-handler-source.ts @@ -88,13 +88,31 @@ export function getPiAgentStatusHandlerSourceLines(kind: PiAgentKind): string[] '// etc.), so we forward the raw object verbatim under the same field', '// names Claude uses (tool_name / tool_input) and let the server pick the', '// preview. Keeps tool-name knowledge centralized on the receiver side.', + '// Why: a restarted agent inherits the previous owner PID through env, so a', + '// dead owner must be claimable or the pane goes silent for good. Only ESRCH', + '// proves the owner is gone -- every other probe result keeps suppression, so', + '// a live foreign owner still cannot double-report. Mirrors the tri-state in', + '// main/agent-hooks/managed-hook-owner-identity.ts, which this runtime cannot', + '// import (the extension loads inside pi/omp with no Orca deps).', + 'function isStatusOwnerAlive(pid: string): boolean {', + ' const parsed = Number(pid)', + ' if (!Number.isSafeInteger(parsed) || parsed < 1 || parsed > 0x7fffffff) return false', + " if (typeof process.kill !== 'function') return true", + ' try {', + ' process.kill(parsed, 0)', + ' return true', + ' } catch (err: unknown) {', + " return (err as { code?: string } | null)?.code !== 'ESRCH'", + ' }', + '}', + '', "// Why: child agents inherit the lead's pane env; only its process may", '// register status hooks. PID identity keeps in-process reloads reporting.', 'export default function (pi): void {', ...primeDaemonWorkerGuard, ` const ownerPid = process.env.${ownerEnv}`, ' const selfPid = String(process.pid)', - ' if (ownerPid && ownerPid !== selfPid) return', + ' if (ownerPid && ownerPid !== selfPid && isStatusOwnerAlive(ownerPid)) return', ` process.env.${ownerEnv} = selfPid`, ...sessionStartHandler, ` pi.on('before_agent_start', (event${ctxParam}) => {`, diff --git a/src/main/pi/agent-status-owner-recovery.test.ts b/src/main/pi/agent-status-owner-recovery.test.ts new file mode 100644 index 00000000000..d176bcb8dae --- /dev/null +++ b/src/main/pi/agent-status-owner-recovery.test.ts @@ -0,0 +1,89 @@ +import { describe, expect, it } from 'vitest' +import { + createAgentStatusExtensionHarness as createHarness, + AGENT_STATUS_EXTENSION_SELF_PID as SELF_PID +} from './agent-status-extension-test-harness' + +describe('Pi status owner recovery', () => { + it.each(['pi', 'omp', 'prime-agent'] as const)( + 'claims the pane for a restarted %s agent whose inherited owner PID is dead', + async (kind) => { + // Why: STA-5245 -- a restart leaves a dead owner PID in the inherited env. + // Without a liveness probe the guard suppresses every later load, so the + // pane never reports status again. + const ownerKey = + kind === 'prime-agent' ? 'ORCA_PRIME_AGENT_STATUS_OWNED' : 'ORCA_PI_STATUS_OWNED' + const harness = createHarness({ + kind, + pid: SELF_PID, + env: { [ownerKey]: String(SELF_PID - 1) }, + killImpl: () => { + throw Object.assign(new Error('ESRCH'), { code: 'ESRCH' }) + } + }) + + expect(harness.killMock).toHaveBeenCalledWith(SELF_PID - 1, 0) + expect(harness.handlers.agent_end).toBeTypeOf('function') + expect(harness.processEnv[ownerKey]).toBe(String(SELF_PID)) + + await harness.callHook('agent_end') + expect(harness.fetchMock).toHaveBeenCalledTimes(1) + } + ) + + it.each(['EPERM', 'EACCES', 'EINVAL', undefined])( + 'keeps suppression for unverifiable probe error %s', + (code) => { + // Why: EPERM means the owner exists but belongs to another user, so + // claiming the pane there would reintroduce double-reporting. + const harness = createHarness({ + kind: 'pi', + pid: SELF_PID, + env: { ORCA_PI_STATUS_OWNED: String(SELF_PID - 1) }, + killImpl: () => { + throw Object.assign(new Error('probe failed'), { code }) + } + }) + + expect(harness.handlers).toEqual({}) + expect(harness.processEnv.ORCA_PI_STATUS_OWNED).toBe(String(SELF_PID - 1)) + } + ) + + it('claims the pane when the inherited owner PID is not a usable pid', () => { + // Why: a truncated/garbage marker is not evidence of a live owner. + const harness = createHarness({ + kind: 'pi', + pid: SELF_PID, + env: { ORCA_PI_STATUS_OWNED: 'not-a-pid' } + }) + + expect(harness.killMock).not.toHaveBeenCalled() + expect(harness.handlers.agent_end).toBeTypeOf('function') + expect(harness.processEnv.ORCA_PI_STATUS_OWNED).toBe(String(SELF_PID)) + }) + + it('claims the pane when the inherited owner PID exceeds safe integer precision', () => { + const harness = createHarness({ + kind: 'pi', + pid: SELF_PID, + env: { ORCA_PI_STATUS_OWNED: '99999999999999999999999' } + }) + + expect(harness.killMock).not.toHaveBeenCalled() + expect(harness.handlers.agent_end).toBeTypeOf('function') + expect(harness.processEnv.ORCA_PI_STATUS_OWNED).toBe(String(SELF_PID)) + }) + + it('claims the pane when the inherited owner PID exceeds the process API range', () => { + const harness = createHarness({ + kind: 'pi', + pid: SELF_PID, + env: { ORCA_PI_STATUS_OWNED: String(2 ** 31) } + }) + + expect(harness.killMock).not.toHaveBeenCalled() + expect(harness.handlers.agent_end).toBeTypeOf('function') + expect(harness.processEnv.ORCA_PI_STATUS_OWNED).toBe(String(SELF_PID)) + }) +}) diff --git a/src/main/pi/titlebar-extension-lifetime-source.ts b/src/main/pi/titlebar-extension-lifetime-source.ts new file mode 100644 index 00000000000..1d7c4abd304 --- /dev/null +++ b/src/main/pi/titlebar-extension-lifetime-source.ts @@ -0,0 +1,30 @@ +export function getPiTitlebarLifetimeSourceLines(): string[] { + return [ + ' // Why: replacement factories share the process realm; retire the old owner before painting.', + " const ownersKey = Symbol.for('orca.pi.titlebar.owners')", + ' const owners = globalThis[ownersKey] ??= new Map()', + ' const paneKey = process.env.ORCA_PANE_KEY', + ' owners.get(paneKey)?.()', + ' let disposed = false', + ' function clearOwnedTimers() {', + ' clearPendingAgentEndCheck()', + ' clearAnimation()', + ' stopMarkerReassert()', + ' }', + '', + ' function dispose() {', + ' disposed = true', + ' clearOwnedTimers()', + ' resetPromptState()', + ' if (owners.get(paneKey) === dispose) owners.delete(paneKey)', + ' }', + ' owners.set(paneKey, dispose)', + '', + ' function on(name, handler) {', + ' pi.on(name, (event, ctx) => {', + ' if (!disposed) return handler(event, ctx)', + ' })', + ' }', + '' + ] +} diff --git a/src/main/pi/titlebar-extension-service.test.ts b/src/main/pi/titlebar-extension-service.test.ts index 3ad73bba6eb..69884d6591b 100644 --- a/src/main/pi/titlebar-extension-service.test.ts +++ b/src/main/pi/titlebar-extension-service.test.ts @@ -39,6 +39,7 @@ vi.mock('os', async (importOriginal) => { }) import { PiTitlebarExtensionService, isSafeDescendCandidate } from './titlebar-extension-service' +import { getPiTitlebarExtensionSource } from './titlebar-extension-source' function legacyOverlayPath(kind: 'pi' | 'omp', ptyId: string): string { const rootDir = kind === 'pi' ? 'pi-agent-overlays' : 'omp-agent-overlays' @@ -458,6 +459,16 @@ describe('PiTitlebarExtensionService', () => { expectPiHomeIntact() }) + it('refreshes a managed spinner in an explicitly selected senpi home', () => { + const agentDir = join(userDataDir, '.omo', 'agent') + const extensionPath = join(agentDir, 'extensions', 'orca-titlebar-spinner.ts') + mkdirSync(join(agentDir, 'extensions'), { recursive: true }) + writeFileSync(extensionPath, '// @orca-managed-pi-extension\nstale spinner') + const svc = new PiTitlebarExtensionService() + svc.buildPtyEnv('pty-senpi', agentDir, 'pi') + expect(readFileSync(extensionPath, 'utf8')).toContain(getPiTitlebarExtensionSource()) + }) + it('rebuilding updates Orca-owned extensions while preserving user files', () => { const svc = new PiTitlebarExtensionService() svc.buildPtyEnv('pty-refresh-1', piHome, 'pi') diff --git a/src/main/pi/titlebar-extension-source.test.ts b/src/main/pi/titlebar-extension-source.test.ts index be21f8c6a16..eb826c1903e 100644 --- a/src/main/pi/titlebar-extension-source.test.ts +++ b/src/main/pi/titlebar-extension-source.test.ts @@ -35,6 +35,8 @@ function createHarness( processTitle?: string cwdImpl?: () => string sessionNameImpl?: () => string + setTitle?: (title: string) => void + globals?: Record env?: Record } = {} ): Harness { @@ -42,6 +44,7 @@ function createHarness( const ctx: TitlebarContext = { ui: { setTitle: (title: string) => { + options.setTitle?.(title) titles.push(title) } }, @@ -75,7 +78,7 @@ function createHarness( setTimeout: (...args: Parameters) => setTimeout(...args), clearTimeout: (timer: ReturnType) => clearTimeout(timer) } as Record - context.globalThis = context + context.globalThis = options.globals ?? context const output = ts.transpileModule(getPiTitlebarExtensionSource(options.kind ?? 'pi'), { compilerOptions: { module: ts.ModuleKind.CommonJS, target: ts.ScriptTarget.ES2020 } @@ -588,4 +591,148 @@ describe('getPiTitlebarExtensionSource', () => { expect(harness.handlers.ui_prompt_start).toBeDefined() expect(() => harness.handlers.ui_prompt_start?.({}, undefined)).not.toThrow() }) + + it.each(['getter', 'title'] as const)( + 'retires a stale %s during animation without throwing or rescheduling', + async (failure) => { + let stale = false + const harness = createHarness({ + sessionNameImpl: () => { + if (stale && failure === 'getter') { + throw new Error('expired session') + } + return SESSION + }, + setTitle: () => { + if (stale && failure === 'title') { + throw new Error('expired UI') + } + } + }) + await harness.callHook('agent_start') + stale = true + expect(() => vi.advanceTimersByTime(80)).not.toThrow() + expect(vi.getTimerCount()).toBe(0) + await harness.callHook('agent_start') + expect(vi.getTimerCount()).toBe(0) + stale = false + await harness.callHook('agent_start') + expect(vi.getTimerCount()).toBe(1) + } + ) + + it.each(['getter', 'title'] as const)('contains stale %s during shutdown', async (failure) => { + let stale = false + const harness = createHarness({ + sessionNameImpl: () => { + if (stale && failure === 'getter') { + throw new Error('expired session') + } + return SESSION + }, + setTitle: () => { + if (stale && failure === 'title') { + throw new Error('expired UI') + } + }, + isIdle: () => false + }) + await harness.callHook('agent_start') + await harness.callHook('agent_end') + stale = true + await expect(harness.callHook('session_shutdown')).resolves.toBeUndefined() + expect(vi.getTimerCount()).toBe(0) + }) + + it('does not schedule a timer when the first frame fails', async () => { + const harness = createHarness({ + sessionNameImpl: () => { + throw new Error('expired') + } + }) + await harness.callHook('agent_start') + expect(vi.getTimerCount()).toBe(0) + }) + + it('retires animation when an idle recheck loses its session', async () => { + const harness = createHarness({ + isIdle: () => { + throw new Error('expired') + } + }) + await harness.callHook('agent_start') + await harness.callHook('agent_end') + expect(() => vi.advanceTimersByTime(1)).not.toThrow() + expect(vi.getTimerCount()).toBe(0) + }) + + it('clears animation and pending idle checks on session replacement', async () => { + const harness = createHarness({ isIdle: () => false }) + await harness.callHook('agent_start') + await harness.callHook('agent_end') + await harness.callHook('session_shutdown') + await harness.callHook('session_start') + expect(vi.getTimerCount()).toBe(0) + await harness.callHook('agent_start') + expect(vi.getTimerCount()).toBe(1) + }) + + it('reload replaces only its pane owner and ignores late old-generation events', async () => { + const globals = {} + const old = createHarness({ globals, isIdle: () => false }) + const other = createHarness({ globals, paneKey: 'pane-2' }) + await old.callHook('agent_start') + await old.callHook('ui_prompt_start') + await old.callHook('agent_end') + await other.callHook('agent_start') + const replacement = createHarness({ globals }) + expect(vi.getTimerCount()).toBe(1) + await replacement.callHook('agent_start') + const oldCount = old.titles.length + await old.callHook('session_shutdown') + await old.callHook('agent_start') + vi.advanceTimersByTime(80) + expect(old.titles).toHaveLength(oldCount) + expect(vi.getTimerCount()).toBe(2) + expect(replacement.lastTitle()).toMatch(BRAILLE_RE) + expect(other.lastTitle()).toMatch(BRAILLE_RE) + const third = createHarness({ globals }) + expect(vi.getTimerCount()).toBe(1) + await third.callHook('agent_start') + await replacement.callHook('session_shutdown') + expect(vi.getTimerCount()).toBe(2) + }) + + it('stops spinner, prompt reassertion and idle recheck together on invalidation', async () => { + let stale = false + const harness = createHarness({ + isIdle: () => false, + sessionNameImpl: () => { + if (stale) { + throw new Error('stale generation') + } + return SESSION + } + }) + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + await harness.callHook('agent_end') + expect(vi.getTimerCount()).toBe(3) + stale = true + await vi.advanceTimersByTimeAsync(80) + expect(vi.getTimerCount()).toBe(0) + }) + + it('clears prompt and idle timers at session_start without needing shutdown', async () => { + const harness = createHarness({ isIdle: () => false }) + await harness.callHook('agent_start') + await harness.callHook('ui_prompt_start') + await harness.callHook('agent_end') + expect(vi.getTimerCount()).toBe(3) + await harness.callHook('session_start') + expect(vi.getTimerCount()).toBe(0) + await harness.callHook('agent_start') + expect(harness.lastTitle()).toMatch(BRAILLE_RE) + expect(vi.getTimerCount()).toBe(1) + }) }) diff --git a/src/main/pi/titlebar-extension-source.ts b/src/main/pi/titlebar-extension-source.ts index 7fc15c191bc..570d23a15d3 100644 --- a/src/main/pi/titlebar-extension-source.ts +++ b/src/main/pi/titlebar-extension-source.ts @@ -1,3 +1,4 @@ +import { getPiTitlebarLifetimeSourceLines } from './titlebar-extension-lifetime-source' import type { PiAgentKind } from '../../shared/pi-agent-kind' import { getPiOmpRuntimeDetectionSourceLines } from './agent-status-runtime-detection-source' @@ -10,7 +11,7 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { const uiPromptHandlers = kind === 'pi' ? [ - " pi.on('ui_prompt_start', async (_event, ctx) => {", + " on('ui_prompt_start', async (_event, ctx) => {", ' if (isOmpRuntime() || !ownsMarker) return', ' promptDepth++', ' // Why: retry on every open rather than only the outermost, so an outer ctx', @@ -25,7 +26,7 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' startMarkerReassert(painter)', ' })', '', - " pi.on('ui_prompt_end', async (_event, ctx) => {", + " on('ui_prompt_end', async (_event, ctx) => {", ' if (isOmpRuntime() || !ownsMarker || promptDepth === 0) return', ' promptDepth--', ' if (promptDepth > 0) return', @@ -98,20 +99,6 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' }', '}', '', - '// Why: buildTitle runs inside the try because it is not safe either — getSessionName()', - '// calls assertActive() and process.cwd() throws ENOENT once the worktree is deleted.', - '// Most call sites are timer callbacks, where an escape is an uncaught exception and pi', - '// exits(1) through its own uncaughtException handler.', - 'function paintTitle(ctx, buildTitle) {', - ' if (!ctx) return false', - ' try {', - ' ctx.ui.setTitle(buildTitle())', - ' return true', - ' } catch {', - ' return false', - ' }', - '}', - '', 'export default function (pi) {', ' if (!process.env.ORCA_PANE_KEY) return', ...(kind === 'pi' @@ -125,6 +112,7 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ] : []), + ...getPiTitlebarLifetimeSourceLines(), ' let timer = null', ' let frameIndex = 0', ' // Why: only idle maintenance owns a spinner of its own. A threshold compaction runs', @@ -144,6 +132,21 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' let pendingAgentEndContext = null', ' let agentEndIdleRecheckMs = AGENT_END_IDLE_RECHECK_MS', '', + '// Why: buildTitle runs inside the try because it is not safe either — getSessionName()', + '// calls assertActive() and process.cwd() throws ENOENT once the worktree is deleted.', + '// Most call sites are timer callbacks, where an escape is an uncaught exception and pi', + '// exits(1) through its own uncaughtException handler.', + ' function paintTitle(ctx, buildTitle) {', + ' if (disposed || !ctx) return false', + ' try {', + ' ctx.ui.setTitle(buildTitle())', + ' return true', + ' } catch {', + ' clearOwnedTimers()', + ' return false', + ' }', + ' }', + '', ' function resetPromptState() {', ' stopMarkerReassert()', ' promptDepth = 0', @@ -199,23 +202,24 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' // otherwise wipe the marker with nothing to restore it. The frame still counts,', ' // so the cap above keeps accruing in wall-clock.', ' if (markerPainted) {', - " paintTitle(ctx, () => getMarkedTitle(pi, '!'))", + " const painted = paintTitle(ctx, () => getMarkedTitle(pi, '!'))", ' frameIndex++', - ' return', + ' return painted', ' }', - ' paintTitle(ctx, () => {', + ' const painted = paintTitle(ctx, () => {', ' const frame = BRAILLE_FRAMES[frameIndex % BRAILLE_FRAMES.length]', ' const cwd = process.cwd().split(/[\\\\/]/).filter(Boolean).at(-1) || process.cwd()', ' const session = pi.getSessionName()', ' return session ? `${frame} \\u03c0 - ${session} - ${cwd}` : `${frame} \\u03c0 - ${cwd}`', ' })', ' frameIndex++', + ' return painted', ' }', '', ' function startAnimation(ctx) {', ' clearPendingAgentEndCheck()', ' clearAnimation()', - ' renderFrame(ctx)', + ' if (!renderFrame(ctx)) return', ' timer = setInterval(() => renderFrame(ctx), FRAME_INTERVAL_MS)', ' }', '', @@ -230,7 +234,7 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' return', ' }', ' } catch {', - ' pendingAgentEndContext = null', + ' clearOwnedTimers()', ' return', ' }', ' pendingAgentEndCheck = setTimeout(checkPendingAgentEnd, agentEndIdleRecheckMs)', @@ -238,7 +242,7 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' agentEndIdleRecheckMs = Math.min(agentEndIdleRecheckMs * 2, AGENT_END_IDLE_RECHECK_MAX_MS)', ' }', '', - " pi.on('agent_start', async (_event, ctx) => {", + " on('agent_start', async (_event, ctx) => {", ' resetPromptState()', ' startAnimation(ctx)', ' })', @@ -246,17 +250,18 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' // Why: pi drops an open dialog through resetExtensionUI without resolving its promise,', ' // so a replaced or reloaded session never sends the matching close. Both boundaries', ' // prove no dialog from the old session is still on screen.', - " pi.on('session_start', async () => {", + " on('session_start', async () => {", + ' clearOwnedTimers()', ' resetPromptState()', ' })', '', ' // Why: modern Pi/OMP emit agent_end mid-run and only settle later, so settlement is the', ' // authoritative completion boundary. Legacy runtimes never emit it, so agent_end stays.', - " pi.on('agent_settled', async (_event, ctx) => {", + " on('agent_settled', async (_event, ctx) => {", ' stopAnimation(ctx)', ' })', '', - " pi.on('agent_end', async (event, ctx) => {", + " on('agent_end', async (event, ctx) => {", ' if (event?.willContinue === true) {', ' clearPendingAgentEndCheck()', ' return', @@ -273,7 +278,7 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' })', '', ...uiPromptHandlers, - " pi.on('auto_compaction_start', async (event, ctx) => {", + " on('auto_compaction_start', async (event, ctx) => {", " if (event?.reason !== 'idle') return", ' // Why: the idle worker can fire against a turn that just started, and reason alone does', ' // not prove the pane is idle. Adopting a live agent spinner would let the matching', @@ -283,12 +288,12 @@ export function getPiTitlebarExtensionSource(kind: PiAgentKind = 'pi'): string { ' idleCompactionOwnsSpinner = true', ' })', '', - " pi.on('auto_compaction_end', async (_event, ctx) => {", + " on('auto_compaction_end', async (_event, ctx) => {", ' if (!idleCompactionOwnsSpinner) return', ' stopAnimation(ctx)', ' })', '', - " pi.on('session_shutdown', async (_event, ctx) => {", + " on('session_shutdown', async (_event, ctx) => {", ' resetPromptState()', ' stopAnimation(ctx)', ' })', diff --git a/src/main/runtime/__fixtures__/antigravity-busy-mid-turn.meta.json b/src/main/runtime/__fixtures__/antigravity-busy-mid-turn.meta.json new file mode 100644 index 00000000000..4e875eb047a --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-busy-mid-turn.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T06:10:52.713Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy TUI 1.2.0; recording stopped ~0.3s after submit, while the spinner was live; no shutdown repaint in the file", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-busy-mid-turn.txt b/src/main/runtime/__fixtures__/antigravity-busy-mid-turn.txt new file mode 100644 index 00000000000..8f3645800f7 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-busy-mid-turn.txt @@ -0,0 +1,38 @@ +[?2026$p[?2027$p[>4m[=0;1u[?1049h[?25l[?5W[?2004h[>4;2m[=1;1u[?u +▄▀▀▄ +▀▀▀▀▀▀ +▀▀▀▀▀▀▀▀ + ▄▀▀ ▀▀▄ + ▄▀▀ ▀▀▄ + + Welcome to the Antigravity CLI. You are currently not signed in. + + ⣾ Signing in... No authentication methods available. + + Press ctrl+c or ctrl+d twice to exit.[>4m[=0;1u[?1049l[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) + ▄▀▀ ▀▀▄ ~ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[?25lI[?25h[?25ln ab + + G[?25h[?25lout 8[?25h[?25l0 wo[?25h[?25lrds,[?25h[?25lexpla[?25h[?25lin w[?25h[?25lhat a[?25h[?25l pse[?25h[?25lud[?25h[?25loter[?25h[?25lminal[?25h[?25l is.[?25h[?25l[?25h[?25l + +? for shortcuts[?25h[?25lM +> In about 80 words, explain what a pseudoterminal is. +⣷ Generating... +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +esc to cancelGemini 3.7 Flash · low [?25h[?25lng + +[?25h[?25l ⣯ Generating + +[?25h[?25l ⣟ Generating. + +[?25h \ No newline at end of file diff --git a/src/main/runtime/__fixtures__/antigravity-busy-turn-ended.meta.json b/src/main/runtime/__fixtures__/antigravity-busy-turn-ended.meta.json new file mode 100644 index 00000000000..084be8e54bc --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-busy-turn-ended.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T06:13:00.364Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy TUI 1.2.0; recording stopped after the turn ended and the composer returned, with the process still alive. This account's API key cannot complete a turn, so the turn ends in a backend error", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-busy-turn-ended.txt b/src/main/runtime/__fixtures__/antigravity-busy-turn-ended.txt new file mode 100644 index 00000000000..e10de85d361 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-busy-turn-ended.txt @@ -0,0 +1,42 @@ +[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) + ▄▀▀ ▀▀▄ ~ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[?25lIn + + G[?25h[?25labo[?25h[?25lut 80[?25h[?25l wo[?25h[?25lrds[?25h[?25l, ex[?25h[?25lpla[?25h[?25lin wh[?25h[?25lat a[?25h[?25lpseudo[?25h[?25ltermi[?25h[?25lnal is[?25h[?25l.[?25h[?25l + +? for shortcuts[?25h[?25lM +> In about 80 words, explain what a pseudoterminal is. +⣾ Generating... +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +esc to cancelGemini 3.7 Flash · low [?25h[?25l ⣷ Generatin + +[?25h[?25l ⣯ Generating + +[?25h[?25l ⣟ Generating. + +[?25h[?25l ⡿ Generating... + +[?25h[?25l ⢿ Generatin + +[?25h[?25l  +⚠ Agent execution terminated due to error. +Error ID: 00000000-0000-4000-8000-000000000000-2 +⢿ Generating... +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +esc to cancelGemini 3.7 Flash · low [?25h[?25l  + + + +? for shortcuts[?25h \ No newline at end of file diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-command-palette.meta.json b/src/main/runtime/__fixtures__/antigravity-dialog-command-palette.meta.json new file mode 100644 index 00000000000..e098a1677ab --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-command-palette.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T04:34:32.974Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy TUI 1.2.0; slash-command palette live, unanswered", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-command-palette.txt b/src/main/runtime/__fixtures__/antigravity-dialog-command-palette.txt new file mode 100644 index 00000000000..9bf02cc0ff9 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-command-palette.txt @@ -0,0 +1,41 @@ +[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) + ▄▀▀ ▀▀▄ ~ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[?25l/ + +> /add-dir  Add a directory to the workspace + /agents List available custom agents + /artifact View and review artifacts + /btw Ask a side question without interrupting the current task + /changelog Show release notes and changes + ↓ 50 more + + ↑/↓ Navigate · enter Select · tab Complete + Gemini 3.7 Flash · low [?25h[?25l + + + + + + + + + +esc to cancel[?25h[>4m[=0;1u + + + + + + + + + +[?2004l[0 q \ No newline at end of file diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-dismissed.meta.json b/src/main/runtime/__fixtures__/antigravity-dialog-dismissed.meta.json new file mode 100644 index 00000000000..8e8d5043fdf --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-dismissed.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T04:35:06.866Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy TUI 1.2.0; /model picker opened then dismissed with esc, settled before stop", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-dismissed.txt b/src/main/runtime/__fixtures__/antigravity-dialog-dismissed.txt new file mode 100644 index 00000000000..bb35ae33af2 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-dismissed.txt @@ -0,0 +1,54 @@ +[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) + ▄▀▀ ▀▀▄ ~ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[?25l/mod + +> /model Set a model, or run a single prompt on another model + /permissioned-github Guidelines for interacting with GitHub and request permissions from the user when commands f... + + ↑/↓ Navigate · enter Select · tab Complete +esc to cancelGemini 3.7 Flash · low [?25h[?25l + + + + +/model + +  + + ↑/↓ Navigate · enter Select · tab Complete +esc to cancelGemini 3.7 Flash · low [?25h[?25l[0 q + +Switch Model + + Gemini 3.8 Flash +> Gemini 3.7 Flash (current) + Gemini 3.6 Flash + Gemini 3.1 Pro + + Effort ◂  ◉──────────────○──────────────○  ▸ +  low  medium high  + Faster responses, lighter reasoning — great for simpler tasks + +Keyboard: ↑/↓ Navigate ←/→ Effort enter Select esc Go Back + + Gemini 3.7 Flash · low [0 q> /model + ⎿ Exited /model command + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +Gemini 3.7 Flash · low [?25h[?25l + +? for shortcuts[?25h[>4m[=0;1u + +[?2004l[0 q +Resume with -c (or command below): +agy --conversation=00000000-0000-4000-8000-000000000000 diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-model-picker.meta.json b/src/main/runtime/__fixtures__/antigravity-dialog-model-picker.meta.json new file mode 100644 index 00000000000..9a4e5c0c8e1 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-model-picker.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T04:34:10.855Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy TUI 1.2.0; /model picker live, unanswered, killed while it owns the screen", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-model-picker.txt b/src/main/runtime/__fixtures__/antigravity-dialog-model-picker.txt new file mode 100644 index 00000000000..6a09f6082f8 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-model-picker.txt @@ -0,0 +1,56 @@ +[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) + ▄▀▀ ▀▀▄ ~ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[?25l/mo + +> /model Set a model, or run a single prompt on another model + /migrate-workflows Automatically migrate legacy workflows to modern skills across global and workspace configur... + /permissions Manage tool permissions + /agy-customizations Comprehensive guide and reference for the Antigravity Customization System. Use to explain h... + /permissioned-github Guidelines for interacting with GitHub and request permissions from the user when commands f... + + ↑/↓ Navigate · enter Select · tab Complete +? for shortcutsGemini 3.7 Flash · low [?25h[?25l + + + + +/model + +  + + ↑/↓ Navigate · enter Select · tab Complete +esc to cancelGemini 3.7 Flash · low [?25h[?25l[0 q + +Switch Model + +> Gemini 3.8 Flash + Gemini 3.7 Flash (current) + Gemini 3.6 Flash + Gemini 3.1 Pro + + Effort ◂  ●━━━━━━━━━━━━━━◉──────────────○  ▸ +  low  medium  high  + Balanced speed and reasoning quality for most tasks + +Keyboard: ↑/↓ Navigate ←/→ Effort enter Select esc Go Back + +? for shortcutsGemini 3.7 Flash · low  Gemini 3.8 Flash +> Gemini 3.7 Flash + + + +◂  ◉──────────────○ + low  medium  +Faster responses, lighter reasoning — great for simpler tasks + + + +  G[>4m[=0;1u [?25h[?2004l \ No newline at end of file diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-trust-workspace.meta.json b/src/main/runtime/__fixtures__/antigravity-dialog-trust-workspace.meta.json new file mode 100644 index 00000000000..07fb15ab6f7 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-trust-workspace.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T04:35:20.989Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy TUI 1.2.0; workspace trust dialog live and unanswered in a throwaway untrusted directory", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-dialog-trust-workspace.txt b/src/main/runtime/__fixtures__/antigravity-dialog-trust-workspace.txt new file mode 100644 index 00000000000..b2e1b342199 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-dialog-trust-workspace.txt @@ -0,0 +1,12 @@ +[?2026$p[?2027$p[>4m[=0;1u[?1049h[?25l[?5W[?2004h[>4;2m[=1;1u[?uAccessing workspace: + +/private/tmp/agy-trust-scratch-77950 + +Do you trust the contents of this project? + +Antigravity CLI requires permission to read, edit, and execute files here. + +> Yes, I trust this folder + No, exit + + ↑/↓ Navigate · enter ConfirmGemini 3.7 Flash · low[>4m[=0;1u [?1049l[?25h[?2004l \ No newline at end of file diff --git a/src/main/runtime/__fixtures__/antigravity-ready-account-info-hidden.meta.json b/src/main/runtime/__fixtures__/antigravity-ready-account-info-hidden.meta.json new file mode 100644 index 00000000000..9607841cf6e --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-ready-account-info-hidden.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T04:33:34.954Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "same session as antigravity-ready-api-key-gemini-model but with AGY_CLI_HIDE_ACCOUNT_INFO=1", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-ready-account-info-hidden.txt b/src/main/runtime/__fixtures__/antigravity-ready-account-info-hidden.txt new file mode 100644 index 00000000000..b93514374e0 --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-ready-account-info-hidden.txt @@ -0,0 +1,13 @@ +[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini 3.7 Flash (Low) +▀▀▀▀▀▀▀▀ ~ + ▄▀▀ ▀▀▄ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[>4m[=0;1u + +[?2004l[0 q \ No newline at end of file diff --git a/src/main/runtime/__fixtures__/antigravity-ready-api-key-gemini-model.meta.json b/src/main/runtime/__fixtures__/antigravity-ready-api-key-gemini-model.meta.json new file mode 100644 index 00000000000..97a54e107dc --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-ready-api-key-gemini-model.meta.json @@ -0,0 +1,9 @@ +{ + "capturedAt": "2026-09-11T04:33:14.819Z", + "platform": "darwin", + "command": ["agy"], + "cols": 120, + "rows": 40, + "note": "agy binary 1.1.25, TUI banner 1.2.0; Gemini API key identity (no OAuth sign-in); model Gemini 3.7 Flash (Low); workspace ~", + "exitCode": 0 +} diff --git a/src/main/runtime/__fixtures__/antigravity-ready-api-key-gemini-model.txt b/src/main/runtime/__fixtures__/antigravity-ready-api-key-gemini-model.txt new file mode 100644 index 00000000000..c9501f1caac --- /dev/null +++ b/src/main/runtime/__fixtures__/antigravity-ready-api-key-gemini-model.txt @@ -0,0 +1,13 @@ +[?2026$p[?2027$p[?5W[?2004h[>4;2m[=1;1u[?u[0 q  +▄▀▀▄ Antigravity CLI 1.2.0 +▀▀▀▀▀▀ Gemini API key +▀▀▀▀▀▀▀▀ Gemini 3.7 Flash (Low) + ▄▀▀ ▀▀▄ ~ + ▄▀▀ ▀▀▄ + +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +> +──────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── +? for shortcutsGemini 3.7 Flash · low [?25h[>4m[=0;1u + +[?2004l[0 q \ No newline at end of file diff --git a/src/main/runtime/agent-session-conversation-name-store.test.ts b/src/main/runtime/agent-session-conversation-name-store.test.ts new file mode 100644 index 00000000000..ce5d3fd2621 --- /dev/null +++ b/src/main/runtime/agent-session-conversation-name-store.test.ts @@ -0,0 +1,104 @@ +// The name is durable state on the record: the store is the only thing that writes it. +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { AgentSessionRecordStore } from './agent-session-record-store' +import type { AgentSessionReserveRequest } from './agent-session-reservation-admission' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-alpha' +const NATIVE: AgentSessionExecutionLocation = { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' +} + +let counter = 0 +/** Same shape the store's own suite uses: `-<32 hex>`. */ +function operationId(): string { + counter += 1 + return `${NOW}-${String(counter) + .padStart(32, '0') + .replaceAll(/[^0-9a-f]/g, '0')}` +} + +const reserveRequest = (): AgentSessionReserveRequest => ({ + sessionId: SESSION, + location: NATIVE, + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/dev/.claude-work' }, + runtimeKind: 'native', + expectedFence: null, + spawnToken: 'spawn-a', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'indeterminate', reason: 'no answer' }, + operation: { callerKey: 'client-1', operationId: operationId(), fingerprint: 'fp-1' }, + now: NOW +}) + +let directory: string + +beforeEach(async () => { + directory = await mkdtemp(join(tmpdir(), 'orca-conversation-name-store-')) +}) +afterEach(async () => { + await rm(directory, { recursive: true, force: true }) +}) + +async function reservedStore(): Promise { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + await store.reserveOwner(reserveRequest()) + return store +} + +describe('AgentSessionRecordStore.setConversationName', () => { + it('stores the name and survives a reload, so the record is where it lives', async () => { + const store = await reservedStore() + + await store.setConversationName(SESSION, 'Fix the lease probe') + + const reloaded = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + expect(reloaded.getRecord(SESSION)?.conversationName).toBe('Fix the lease probe') + }) + + it('normalizes at the boundary, so no caller can persist an invalid record', async () => { + const store = await reservedStore() + + await store.setConversationName(SESSION, `Fix\nthe ${'x'.repeat(400)}`) + + const name = store.getRecord(SESSION)?.conversationName ?? '' + expect(name).toHaveLength(200) + expect(name.startsWith('Fix the ')).toBe(true) + // A reload validates every record; an over-long name would be dropped as unreadable. + const reloaded = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + expect(reloaded.getRecord(SESSION)?.conversationName).toBe(name) + }) + + it('clears the name with null', async () => { + const store = await reservedStore() + await store.setConversationName(SESSION, 'Fix the lease probe') + + await store.setConversationName(SESSION, null) + + expect(store.getRecord(SESSION)?.conversationName).toBeUndefined() + }) + + it('does not need the lease: an unfenced rename never contends with the writer', async () => { + const store = await reservedStore() + + // No fence argument exists to pass, and no fence error is raised. + await expect(store.setConversationName(SESSION, 'Fix the lease probe')).resolves.toBeDefined() + }) + + it('refuses a session it has no record for', async () => { + const store = await reservedStore() + + await expect(store.setConversationName('missing', 'A name')).rejects.toThrow( + 'agent_session_identity_required' + ) + }) +}) diff --git a/src/main/runtime/agent-session-record-conversation-name.test.ts b/src/main/runtime/agent-session-record-conversation-name.test.ts new file mode 100644 index 00000000000..3c5c55d83c6 --- /dev/null +++ b/src/main/runtime/agent-session-record-conversation-name.test.ts @@ -0,0 +1,129 @@ +import { describe, expect, it } from 'vitest' +import { isAgentSessionRecord } from '../../shared/agent-session-record' +import { agentSessionRecordFixture } from '../../shared/agent-session-record.test-fixture' +import { setAgentSessionRecordConversationName } from './agent-session-record-conversation-name' + +const NOW = 9_000 + +describe('agent session record conversationName validation', () => { + it('accepts a record carrying a bounded name', () => { + expect( + isAgentSessionRecord({ + ...agentSessionRecordFixture(), + conversationName: 'Fix the lease probe' + }) + ).toBe(true) + }) + + it('accepts a record with no name at all', () => { + expect(isAgentSessionRecord(agentSessionRecordFixture())).toBe(true) + }) + + it('rejects a name past the stored maximum', () => { + expect( + isAgentSessionRecord({ ...agentSessionRecordFixture(), conversationName: 'a'.repeat(201) }) + ).toBe(false) + }) + + it('rejects a name that is not a string', () => { + expect(isAgentSessionRecord({ ...agentSessionRecordFixture(), conversationName: 42 })).toBe( + false + ) + expect(isAgentSessionRecord({ ...agentSessionRecordFixture(), conversationName: '' })).toBe( + false + ) + }) + + it('rejects persisted names that bypassed canonical normalization', () => { + expect( + isAgentSessionRecord({ + ...agentSessionRecordFixture(), + conversationName: 'Fix\u202Egnp.exe probe' + }) + ).toBe(false) + expect( + isAgentSessionRecord({ ...agentSessionRecordFixture(), conversationName: 'Fix\nthe probe' }) + ).toBe(false) + }) +}) + +describe('setAgentSessionRecordConversationName', () => { + it('sets the name and stamps the update', () => { + const next = setAgentSessionRecordConversationName( + agentSessionRecordFixture(), + 'Fix the lease probe', + NOW + ) + + expect(next.conversationName).toBe('Fix the lease probe') + expect(next.updatedAt).toBe(NOW) + expect(isAgentSessionRecord(next)).toBe(true) + }) + + it('normalizes on the way in so the record stays valid whatever the caller sent', () => { + const next = setAgentSessionRecordConversationName( + agentSessionRecordFixture(), + `Fix\nthe probe`, + NOW + ) + + expect(next.conversationName).toBe('Fix the probe') + expect(isAgentSessionRecord(next)).toBe(true) + }) + + it('bounds an over-long name rather than storing a record the validator would reject', () => { + const next = setAgentSessionRecordConversationName( + agentSessionRecordFixture(), + 'a'.repeat(1000), + NOW + ) + + expect(next.conversationName).toHaveLength(200) + expect(isAgentSessionRecord(next)).toBe(true) + }) + + it('clears the name via null, deleting the key rather than storing an empty string', () => { + const named = setAgentSessionRecordConversationName( + agentSessionRecordFixture(), + 'Fix the probe', + NOW + ) + + const cleared = setAgentSessionRecordConversationName(named, null, NOW + 1) + + expect(Object.hasOwn(cleared, 'conversationName')).toBe(false) + expect(cleared.updatedAt).toBe(NOW + 1) + expect(isAgentSessionRecord(cleared)).toBe(true) + }) + + it('treats a name that normalizes to nothing as a clear', () => { + const named = setAgentSessionRecordConversationName( + agentSessionRecordFixture(), + 'Fix the probe', + NOW + ) + + expect( + Object.hasOwn( + setAgentSessionRecordConversationName(named, ' ', NOW + 1), + 'conversationName' + ) + ).toBe(false) + }) + + it('returns the same object when the name is unchanged, so no write is provoked', () => { + const named = setAgentSessionRecordConversationName( + agentSessionRecordFixture(), + 'Fix the probe', + NOW + ) + + expect(setAgentSessionRecordConversationName(named, 'Fix the probe', NOW + 1)).toBe(named) + }) + + it('returns the same object when clearing a record that has no name', () => { + const record = agentSessionRecordFixture() + + expect(setAgentSessionRecordConversationName(record, null, NOW)).toBe(record) + }) +}) diff --git a/src/main/runtime/agent-session-record-conversation-name.ts b/src/main/runtime/agent-session-record-conversation-name.ts new file mode 100644 index 00000000000..db4ad25d10a --- /dev/null +++ b/src/main/runtime/agent-session-record-conversation-name.ts @@ -0,0 +1,27 @@ +import { normalizeAgentSessionConversationName } from '../../shared/agent-session-conversation-name' +import type { AgentSessionRecord } from '../../shared/agent-session-record' + +/** + * Set or clear the conversation name on one record. + * + * Deliberately unfenced: the name is a durable note, not ownership, so writing it never contends + * with the writer lease. Normalizing here — the only writer of the field — keeps the record's own + * validator satisfied no matter which caller supplied the text. + */ +export function setAgentSessionRecordConversationName( + record: AgentSessionRecord, + name: string | null, + now: number +): AgentSessionRecord { + const normalized = name === null ? null : normalizeAgentSessionConversationName(name) + if ((record.conversationName ?? null) === normalized) { + return record + } + const next = { ...record, updatedAt: now } + if (normalized === null) { + delete next.conversationName + return next + } + next.conversationName = normalized + return next +} diff --git a/src/main/runtime/agent-session-record-store.ts b/src/main/runtime/agent-session-record-store.ts index 4325410ed81..1bfc9bbcb3c 100644 --- a/src/main/runtime/agent-session-record-store.ts +++ b/src/main/runtime/agent-session-record-store.ts @@ -1,9 +1,9 @@ import { setVisibleSessionId } from './agent-session-visible-tab-index' import { commitConversationCommandRecord } from './agent-session-conversation-command-record' +import { setAgentSessionRecordConversationName } from './agent-session-record-conversation-name' /** Durable single-writer session records and their operation ledger. */ import { - agentSessionOperationKey, settleAgentSessionOperation, type AgentSessionOperationDecision, type AgentSessionOperationOutcome, @@ -51,10 +51,7 @@ import { type AgentSessionReservationProcesslessProof } from './agent-session-processless-reservation' import { - admitPendingAgentSessionReservationReplay, - applyAgentSessionReservation, - evaluateAgentSessionReserveOperation, - requireAgentSessionRecordForReplay, + commitAgentSessionReservation, type AgentSessionReserveRequest, type AgentSessionReserveResult } from './agent-session-reservation-admission' @@ -145,6 +142,13 @@ export class AgentSessionRecordStore { ) } + /** Unfenced on purpose: the name is a durable note, so writing it never contends with the + * writer lease. `null` clears it. */ + setConversationName = (sessionId: string, name: string | null): Promise => + this.mutate(sessionId, (record) => + setAgentSessionRecordConversationName(record, name, Date.now()) + ) + /** A record this build cannot validate: readable as present, never grantable as a writer. */ isSessionUnreadable(sessionId: string): boolean { return this.state.unreadableRecords.has(sessionId) @@ -165,31 +169,10 @@ export class AgentSessionRecordStore { ) } - /** - * Compare-and-swap reservation plus its client-operation row, committed together. A replayed - * operation returns the recorded outcome and never reaches the reservation. - */ async reserveOwner(request: AgentSessionReserveRequest): Promise { - return this.transact(() => { - const decision = evaluateAgentSessionReserveOperation(this.state, request) - if (decision.decision === 'refused') { - throw new Error(decision.code) - } - if (decision.decision === 'replay') { - let record = requireAgentSessionRecordForReplay(this.state, decision.row, request.sessionId) - if (decision.row.outcome.status === 'pending' && request.handoffOperationId !== null) { - record = admitPendingAgentSessionReservationReplay(record, request) - } - return { record, disposition: 'replayed' as const, operationRow: decision.row } - } - const result = applyAgentSessionReservation(this.state, request, AGENT_SESSION_LEASE_TTL_MS) - this.state.operations.set( - agentSessionOperationKey(request.operation.callerKey, request.operation.operationId), - decision.row - ) - this.state.records.set(result.record.sessionId, result.record) - return { ...result, operationRow: decision.row } - }) + return this.transact(() => + commitAgentSessionReservation(this.state, request, AGENT_SESSION_LEASE_TTL_MS) + ) } async commitProcessIdentity( diff --git a/src/main/runtime/agent-session-reservation-admission.ts b/src/main/runtime/agent-session-reservation-admission.ts index 82176a05297..662ea481921 100644 --- a/src/main/runtime/agent-session-reservation-admission.ts +++ b/src/main/runtime/agent-session-reservation-admission.ts @@ -2,11 +2,15 @@ * Reservation admission: what a reserve request means against the persisted state. * * Pure over a store snapshot so the compare-and-swap, the idempotency replay, and the - * location-immutability check can be reasoned about without touching the disk. The store applies - * the result inside one transaction; nothing here mutates. + * location-immutability check can be reasoned about without touching the disk. + * + * `commitAgentSessionReservation` is the one exception and the only writer here: it sequences + * those decisions and applies the winning one to the state it was handed. The store calls it + * inside a transaction, which is what makes the record and its operation row land together. */ import { + agentSessionOperationKey, evaluateAgentSessionOperation, pruneAgentSessionOperationRows, type AgentSessionOperationDecision, @@ -265,3 +269,32 @@ function createAgentSessionRecord( } } } + +/** + * Compare-and-swap reservation plus its client-operation row, committed together. A replayed + * operation returns the recorded outcome and never reaches the reservation. + */ +export function commitAgentSessionReservation( + state: AgentSessionStoreState, + request: AgentSessionReserveRequest, + leaseTtlMs: number +): AgentSessionReserveResult { + const decision = evaluateAgentSessionReserveOperation(state, request) + if (decision.decision === 'refused') { + throw new Error(decision.code) + } + if (decision.decision === 'replay') { + let record = requireAgentSessionRecordForReplay(state, decision.row, request.sessionId) + if (decision.row.outcome.status === 'pending' && request.handoffOperationId !== null) { + record = admitPendingAgentSessionReservationReplay(record, request) + } + return { record, disposition: 'replayed' as const, operationRow: decision.row } + } + const result = applyAgentSessionReservation(state, request, leaseTtlMs) + state.operations.set( + agentSessionOperationKey(request.operation.callerKey, request.operation.operationId), + decision.row + ) + state.records.set(result.record.sessionId, result.record) + return { ...result, operationRow: decision.row } +} diff --git a/src/main/runtime/agent-session-surface-release-transition.ts b/src/main/runtime/agent-session-surface-release-transition.ts index 894eaa6f75a..9fd3ef9cfda 100644 --- a/src/main/runtime/agent-session-surface-release-transition.ts +++ b/src/main/runtime/agent-session-surface-release-transition.ts @@ -27,6 +27,8 @@ export function releaseAgentSessionOwnerAfterSurfaceClose(args: { record: AgentSessionRecord expectedFence: number now: number + /** Exit receipt can precede a delayed journal settlement and lease release. */ + exitObservedAt?: number settlementRetry?: { settlementId: string; detail: string } }): AgentSessionRecord { const { record } = args @@ -48,7 +50,7 @@ export function releaseAgentSessionOwnerAfterSurfaceClose(args: { deathEvidence: { kind: 'exit-observed', detail: args.settlementRetry?.detail ?? 'the last surface holding this session released it', - observedAt: args.now + observedAt: args.exitObservedAt ?? args.now } }) } @@ -60,6 +62,7 @@ export function releaseStoredAgentSessionOwnerAfterSurfaceClose( sessionId: string expectedFence: number now: number + exitObservedAt?: number settlementRetry?: { settlementId: string; detail: string } } ): Promise { diff --git a/src/main/runtime/agent-transcript-pane-test-harness.ts b/src/main/runtime/agent-transcript-pane-test-harness.ts new file mode 100644 index 00000000000..f3a9a64793c --- /dev/null +++ b/src/main/runtime/agent-transcript-pane-test-harness.ts @@ -0,0 +1,79 @@ +// One pane builder for every suite that replays a captured agent transcript through the runtime. +import { vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' + +const TRANSCRIPT_PANE_LEAF_ID = '11111111-1111-4111-8111-111111111111' +const TRANSCRIPT_PANE_TAB_ID = 'tab-1' +const TRANSCRIPT_PANE_WORKTREE_ID = 'wt-1' +export const TRANSCRIPT_PANE_PTY_ID = 'pty-1' + +export type TranscriptPaneOptions = { + paneTitle: string + foregroundProcess: string | null + data: string + /** Set for a pane whose PTY lives on an SSH host or WSL distro rather than locally. */ + connectionId?: string + /** Simulates a PTY controller whose foreground probe never settles. */ + foregroundProbeHangs?: boolean + onForegroundProbe?: () => void +} + +export async function createTranscriptPane( + options: TranscriptPaneOptions +): Promise<{ runtime: OrcaRuntimeService; handle: string }> { + const runtime = new OrcaRuntimeService(null) + const internals = runtime as unknown as { + resolveTerminalWorkspaceLaunchScope: (selector: string) => Promise + } + vi.spyOn(internals, 'resolveTerminalWorkspaceLaunchScope').mockResolvedValue({ + id: TRANSCRIPT_PANE_WORKTREE_ID, + path: '/repo/app', + connectionId: options.connectionId ?? null, + repo: null, + folderWorkspace: null + }) + runtime.setPtyController({ + spawn: vi.fn().mockResolvedValue({ id: TRANSCRIPT_PANE_PTY_ID, incarnationId: 'inc-1' }), + write: () => true, + kill: () => true, + getForegroundProcess: (): Promise => { + options.onForegroundProbe?.() + return options.foregroundProbeHangs === true + ? new Promise(() => {}) + : Promise.resolve(options.foregroundProcess) + } + }) + const terminal = await runtime.createTerminal(`id:${TRANSCRIPT_PANE_WORKTREE_ID}`, { + tabId: TRANSCRIPT_PANE_TAB_ID, + leafId: TRANSCRIPT_PANE_LEAF_ID, + title: 'Terminal' + }) + runtime.attachWindow(1) + runtime.syncWindowGraph(1, { + tabs: [ + { + tabId: TRANSCRIPT_PANE_TAB_ID, + worktreeId: TRANSCRIPT_PANE_WORKTREE_ID, + title: 'Terminal', + activeLeafId: TRANSCRIPT_PANE_LEAF_ID, + layout: null + } + ], + leaves: [ + { + tabId: TRANSCRIPT_PANE_TAB_ID, + worktreeId: TRANSCRIPT_PANE_WORKTREE_ID, + leafId: TRANSCRIPT_PANE_LEAF_ID, + paneRuntimeId: 1, + ptyId: TRANSCRIPT_PANE_PTY_ID, + paneTitle: options.paneTitle + } + ] + }) + // Why the guard: a restore seed is only applied to a never-written record, so the restore + // cases must not write an empty chunk first. + if (options.data.length > 0) { + runtime.onPtyData(TRANSCRIPT_PANE_PTY_ID, options.data, Date.now()) + } + return { runtime, handle: terminal.handle } +} diff --git a/src/main/runtime/antigravity-readiness-transcripts.test.ts b/src/main/runtime/antigravity-readiness-transcripts.test.ts new file mode 100644 index 00000000000..3ac7707565f --- /dev/null +++ b/src/main/runtime/antigravity-readiness-transcripts.test.ts @@ -0,0 +1,281 @@ +/** + * Pins Antigravity readiness to captured transcripts instead of hand-written fixtures. + * + * Five detector attempts were tuned against a five-line screen someone typed from memory, and + * three of them shipped worse behaviour than the bug they replaced. Nothing here asserts what + * Antigravity prints: the transcripts do. Six are recorded from a live `agy`; the rest name + * themselves as skipped until someone can reach them. + * + * Four cases are pinned as KNOWN DEFECT: on real output the shipped detector refuses the ready + * screen and accepts the live model picker. Those assert what it does, not what it should. + * + * Capture protocol: docs/reference/agent-pty-transcript-capture.md + * What each transcript decides: docs/reference/antigravity-readiness-evidence.md + */ +import { existsSync, readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { createTranscriptPane } from './agent-transcript-pane-test-harness' +import { extractLastOscTitle } from '../../shared/osc-title-extraction' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +const FIXTURE_DIR = join(__dirname, '__fixtures__') +const EVIDENCE_DOC = join( + __dirname, + '..', + '..', + '..', + 'docs', + 'reference', + 'antigravity-readiness-evidence.md' +) +// Why asymmetric: a ready verdict has to survive the settle window, while a refusal only has to +// hold for one poll. Keeping the refusal short keeps seven transcripts off the suite's clock. +const READY_TIMEOUT_MS = 2_000 +const REFUSAL_TIMEOUT_MS = 600 +/** Antigravity's binary, as Orca launches and probes it (`tui-agent-config.ts` detectCmd). */ +const ANTIGRAVITY_COMMAND = 'agy' +// String.fromCharCode, not a literal: the formatter rewrites an escape sequence into a raw +// control byte in source, which is unreadable and survives badly in diffs. +const ESC = String.fromCharCode(27) + +type TranscriptCase = { + /** Fixture basename; `.txt` under `__fixtures__/`. */ + name: string + /** Capture in docs/reference/antigravity-readiness-evidence.md. */ + capture: string + what: string + /** What a correct detector must answer. Not what the shipped one answers. */ + expectReady: boolean + /** + * Set where the shipped detector contradicts the transcript. The case then runs inverted, so + * CI pins the defect instead of going permanently red — and flips to failing the moment + * someone fixes it, which is exactly when these expectations need re-reading. + */ + knownDefect?: string +} + +const TRANSCRIPTS: readonly TranscriptCase[] = [ + { + name: 'antigravity-ready-api-key-gemini-model', + capture: 'B', + what: 'ready screen, API-key identity — the account row reads "Gemini API key", not an email', + expectReady: true, + knownDefect: 'refused: the model row never starts a line, the logo shares it' + }, + { + name: 'antigravity-ready-account-info-hidden', + capture: 'B', + what: 'ready screen with AGY_CLI_HIDE_ACCOUNT_INFO=1 — no account row at all', + expectReady: true, + knownDefect: 'refused: same line-start defect, and no account row exists to require' + }, + { + name: 'antigravity-dialog-trust-workspace', + capture: 'C', + what: 'workspace trust dialog owning the screen', + expectReady: false + }, + { + name: 'antigravity-dialog-model-picker', + capture: 'C', + what: 'model picker owning the screen', + expectReady: false, + knownDefect: "accepted: the picker's own `Gemini 3.x Flash` rows satisfy the model rule" + }, + { + name: 'antigravity-dialog-command-palette', + capture: 'C', + what: 'slash-command palette owning the screen', + expectReady: false + }, + { + name: 'antigravity-busy-mid-turn', + capture: 'E', + what: 'mid-turn, spinner live — the pane is working, not waiting for a prompt', + expectReady: false + }, + { + // Expected ready because the turn is over and the composer is back on screen. The captured + // turn ends in a backend error, which is the only ending this account's key can produce. + name: 'antigravity-busy-turn-ended', + capture: 'E', + what: 'the turn has ended and the composer has returned, process still alive', + expectReady: true, + knownDefect: 'refused: the retained tail ends on the error block, with no composer row in it' + }, + { + name: 'antigravity-dialog-dismissed', + capture: 'D', + what: 'the screen immediately after the model picker is dismissed', + expectReady: true, + knownDefect: 'refused: the banner is not reprinted and no model row starts a line' + }, + // Not captured: this machine's agy has no OAuth session and offers only Gemini models, and + // reaching the rest would mean signing the operator out or deleting their config. See + // docs/reference/antigravity-readiness-evidence.md § What could not be captured. + { + name: 'antigravity-ready-business-non-gemini', + capture: 'A', + what: 'ready screen, Business account, non-Gemini model', + expectReady: true + }, + { + name: 'antigravity-dialog-sign-in', + capture: 'C', + what: 'sign-in dialog owning the screen', + expectReady: false + }, + { + name: 'antigravity-dialog-theme-picker', + capture: 'C', + what: 'theme picker owning the screen', + expectReady: false + }, + { + name: 'antigravity-dialog-privacy-notice', + capture: 'C', + what: 'privacy notice owning the screen', + expectReady: false + }, + { + name: 'antigravity-dialog-update-banner', + capture: 'C', + what: 'update banner owning the screen', + expectReady: false + } +] + +function fixturePath(name: string): string { + return join(FIXTURE_DIR, `${name}.txt`) +} + +/** + * A `tui-idle` wait ends three ways, and only one of them is readiness: it resolves satisfied, it + * resolves unsatisfied with a blocked reason, or it rejects with `timeout` because nothing ever + * looked ready. The orchestrator treats the last two identically — no prompt is delivered — so + * they are both `ready: false` here. This is the shape `worker-start` sees. + */ +async function readinessVerdict( + transcript: string, + timeoutMs: number +): Promise<{ ready: boolean; blockedReason: unknown; outcome: string }> { + const { runtime, handle } = await createTranscriptPane({ + // Why the transcript's own title: every attempt guessed at Antigravity's title. A raw + // capture carries the OSC bytes, so the pane wears whatever the CLI actually set. + paneTitle: extractLastOscTitle(transcript) ?? ANTIGRAVITY_COMMAND, + foregroundProcess: ANTIGRAVITY_COMMAND, + data: transcript + }) + try { + const result = (await runtime.waitForTerminal(handle, { + condition: 'tui-idle', + timeoutMs + })) as { satisfied?: boolean; blockedReason?: unknown } + return { + ready: result.satisfied === true, + blockedReason: result.blockedReason ?? null, + outcome: result.satisfied === true ? 'satisfied' : 'unsatisfied' + } + } catch (error) { + return { ready: false, blockedReason: null, outcome: `rejected: ${String(error)}` } + } +} + +describe('Antigravity readiness, decided by captured transcripts', () => { + for (const transcript of TRANSCRIPTS) { + const path = fixturePath(transcript.name) + const captured = existsSync(path) + const label = `capture ${transcript.capture}: ${transcript.what}` + + // A pinned defect asserts what the detector DOES, so CI is honest rather than permanently + // red; fixing the detector flips this case to failing, which is when these expectations + // need re-reading. The correct answer stays in `expectReady` and in the test's name. + const shipped = + transcript.knownDefect === undefined ? transcript.expectReady : !transcript.expectReady + const verdictName = + transcript.knownDefect === undefined + ? `${label} → ${transcript.expectReady ? 'ready' : 'not ready'}` + : `${label} → must be ${transcript.expectReady ? 'ready' : 'not ready'}; KNOWN DEFECT, ${transcript.knownDefect}` + + it.skipIf(!captured)( + verdictName, + async () => { + // A refusal only has to hold for one poll; a ready verdict has to survive the settle + // window. Keeping the refusal short keeps eleven transcripts off the suite's clock. + const verdict = await readinessVerdict( + readFileSync(path, 'utf8'), + transcript.expectReady ? READY_TIMEOUT_MS : REFUSAL_TIMEOUT_MS + ) + // A silent dialog carries no blocked-signal wording, so the assertion is only that Orca + // does not call the pane ready and type a prompt into a dialog that owns the screen. + expect({ ready: verdict.ready, outcome: verdict.outcome }).toMatchObject({ + ready: shipped + }) + }, + READY_TIMEOUT_MS + 10_000 + ) + + it.skipIf(!captured)(`${label} was captured raw, not pasted from a rendered screen`, () => { + const text = readFileSync(path, 'utf8') + // Why: a transcript with no escape bytes went through a terminal's renderer and a + // human's clipboard. It cannot answer what the caret or chrome looked like. + expect(text).toContain(ESC) + }) + } + + it('documents every transcript the detector is allowed to depend on', () => { + // Why a test: the doc is the operator's checklist. A name that drifts out of it is a + // transcript nobody will capture, and a case that silently skips forever. + const doc = readFileSync(EVIDENCE_DOC, 'utf8') + for (const transcript of TRANSCRIPTS) { + expect(doc).toContain(`${transcript.name}.txt`) + } + }) + + it('reports how much evidence exists, so a fully skipped run is visible', () => { + const missing = TRANSCRIPTS.filter( + (transcript) => !existsSync(fixturePath(transcript.name)) + ).map((transcript) => `${transcript.name}.txt`) + if (missing.length > 0) { + console.info( + `Antigravity transcripts: ${TRANSCRIPTS.length - missing.length}/${TRANSCRIPTS.length} captured. Missing: ${missing.join(', ')}` + ) + } + expect(missing.length).toBeLessThanOrEqual(TRANSCRIPTS.length) + }) +}) + +describe('scaffold self-check', () => { + // Why these two live here: when a transcript lands and fails, the failure has to mean the + // capture disagreed with the detector — not that the harness or the timeouts are broken. + // Neither case is evidence about Antigravity; both are shapes the current detector already + // decides, used only to prove the plumbing reaches a verdict. + it('reaches a ready verdict through the harness', async () => { + const verdict = await readinessVerdict( + [ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Gemini 3.5 Flash (High)', + '~/orca/workspaces/orca/agy-dispatch-issue', + '>' + ].join('\n'), + READY_TIMEOUT_MS + ) + expect(verdict.ready).toBe(true) + }) + + it('reaches a not-ready verdict through the harness', async () => { + const verdict = await readinessVerdict( + 'Do you trust this workspace directory?\nPress t to trust\n', + REFUSAL_TIMEOUT_MS + ) + expect(verdict.ready).toBe(false) + }) +}) diff --git a/src/main/runtime/device-registry.ts b/src/main/runtime/device-registry.ts index b2d5de8ef41..e3d848405f0 100644 --- a/src/main/runtime/device-registry.ts +++ b/src/main/runtime/device-registry.ts @@ -15,6 +15,10 @@ import { DEVICE_REGISTRY_FILENAME } from './mobile-pairing-files' import type { RelayDeviceBinding } from './relay/relay-revoke-outbox' import type { MobilePairingConnectionMode } from '../../shared/mobile-pairing-connection-mode' import type { RuntimePairingReach } from '../../shared/runtime-pairing-reach' +import { + parseMobilePushRegistration, + type MobilePushRegistration +} from '../../shared/mobile-push-contract' export type { DeviceScope } @@ -30,6 +34,9 @@ export type DeviceEntry = { // Why: STA-2370 — a grant minted for "This computer only" proves nothing about off-host reach when its // client connects, so the bind decision must be able to tell it apart from a LAN/phone grant. pairingReach?: RuntimePairingReach + // Why: survives a desktop restart so the host can keep pushing without the phone + // re-registering. Absent on every registry written before background push existed. + pushRegistration?: MobilePushRegistration } function validRelayBinding(value: unknown, deviceId: string): RelayDeviceBinding | undefined { @@ -179,6 +186,26 @@ export class DeviceRegistry { return true } + /** Passing null clears the registration (unregister, or a token the gateway reported dead). */ + setPushRegistration(deviceId: string, registration: MobilePushRegistration | null): boolean { + const index = this.devices.findIndex((candidate) => candidate.deviceId === deviceId) + if (index === -1 || this.devices[index]?.scope !== 'mobile') { + return false + } + const nextDevices = this.devices.map((device, candidateIndex) => { + if (candidateIndex !== index) { + return device + } + const { pushRegistration: _dropped, ...rest } = device + return registration ? { ...rest, pushRegistration: registration } : rest + }) + // Why: persist before the memory swap so a failed write cannot leave the dispatcher + // pushing to a registration disk says is gone (or vice versa on reload). + this.save(nextDevices) + this.devices = nextDevices + return true + } + setMobilePairingConnectionMode(deviceId: string, mode: MobilePairingConnectionMode): boolean { const index = this.devices.findIndex((candidate) => candidate.deviceId === deviceId) if (index === -1 || this.devices[index]?.scope !== 'mobile') { @@ -297,7 +324,10 @@ export class DeviceRegistry { device.mobilePairingConnectionMode === 'local-only' ? 'local-only' : 'automatic', // Why: registries written before this field existed only ever held network-reach grants (phones and // LAN links), so a missing value must keep binding every interface on reconnect. - pairingReach: device.pairingReach === 'this-computer' ? 'this-computer' : 'network' + pairingReach: device.pairingReach === 'this-computer' ? 'this-computer' : 'network', + // Why: a malformed row must degrade to "no background push", never fail the load + // and strand every paired device. + pushRegistration: parseMobilePushRegistration(device.pushRegistration) })) this.registryUnreadable = false } catch (error) { diff --git a/src/main/runtime/host-challenge-envelope.ts b/src/main/runtime/host-challenge-envelope.ts new file mode 100644 index 00000000000..6a00381c158 --- /dev/null +++ b/src/main/runtime/host-challenge-envelope.ts @@ -0,0 +1,139 @@ +// Why: the relay and the push gateway both authenticate this host with the same +// sealed-box challenge shape (the host keypair is X25519, so it cannot sign). +// Only the domain strings and the transcript fields differ, so the envelope +// handling lives here and each protocol owns its own field validation. +import { createHmac, timingSafeEqual } from 'node:crypto' +import nacl from 'tweetnacl' + +const textEncoder = new TextEncoder() +const textDecoder = new TextDecoder() + +export function decodeCanonicalBase64(value: string, expectedBytes: number): Uint8Array | null { + if (!/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(value)) { + return null + } + const decoded = Buffer.from(value, 'base64') + return decoded.byteLength === expectedBytes && decoded.toString('base64') === value + ? decoded + : null +} + +export function encodeUint64(value: number): Uint8Array { + const bytes = new Uint8Array(8) + new DataView(bytes.buffer).setBigUint64(0, BigInt(value), false) + return bytes +} + +export function equalBytes(left: Uint8Array | undefined, right: Uint8Array): boolean { + return Boolean(left && left.byteLength === right.byteLength && timingSafeEqual(left, right)) +} + +export function encodeText(value: string): Uint8Array { + return textEncoder.encode(value) +} + +/** Length-prefixed field map: u32be(len(name)) || name || u32be(len(value)) || value. */ +export function parseHostChallengeTranscript( + transcript: Uint8Array +): Map | null { + const fields = new Map() + const view = new DataView(transcript.buffer, transcript.byteOffset, transcript.byteLength) + let offset = 0 + try { + while (offset < transcript.byteLength) { + const nameLength = view.getUint32(offset, false) + offset += 4 + const name = textDecoder.decode(transcript.slice(offset, offset + nameLength)) + offset += nameLength + const valueLength = view.getUint32(offset, false) + offset += 4 + if (fields.has(name) || offset + valueLength > transcript.byteLength) { + return null + } + fields.set(name, transcript.slice(offset, offset + valueLength)) + offset += valueLength + } + } catch { + return null + } + return offset === transcript.byteLength ? fields : null +} + +export function readTranscriptUint64(value: Uint8Array | undefined): number | null { + if (!value || value.byteLength !== 8) { + return null + } + const parsed = new DataView(value.buffer, value.byteOffset, value.byteLength).getBigUint64( + 0, + false + ) + return parsed <= BigInt(Number.MAX_SAFE_INTEGER) ? Number(parsed) : null +} + +export type HostChallengeEnvelope = { + transcript: Uint8Array + secret: Uint8Array + peerEphemeralPublicKey: Uint8Array + nonce: Uint8Array +} + +/** + * Opens the sealed challenge and splits out the transcript and the 32-byte secret. + * Returns null for any malformed or undecryptable challenge; the caller still has + * to validate the transcript's fields before answering. + */ +export function openHostChallengeEnvelope(input: { + peerEphemeralPublicKeyB64: string + nonceB64: string + ciphertextB64: string + hostSecretKey: Uint8Array + plaintextDomain: string + /** Reports the failing check by name only; never receives field values. */ + onInvalid?: (reason: string) => void +}): HostChallengeEnvelope | null { + const peerKey = decodeCanonicalBase64(input.peerEphemeralPublicKeyB64, 32) + const nonce = decodeCanonicalBase64(input.nonceB64, 24) + const ciphertext = Buffer.from(input.ciphertextB64, 'base64') + if (!peerKey || !nonce || ciphertext.toString('base64') !== input.ciphertextB64) { + return null + } + const plaintext = nacl.box.open(ciphertext, nonce, peerKey, input.hostSecretKey) + if (!plaintext) { + input.onInvalid?.('challenge-box-open') + return null + } + const domain = textEncoder.encode(`${input.plaintextDomain}\0`) + if ( + !equalBytes(plaintext.slice(0, domain.byteLength), domain) || + plaintext.byteLength < domain.byteLength + 36 + ) { + return null + } + const transcriptLength = new DataView( + plaintext.buffer, + plaintext.byteOffset + domain.byteLength, + 4 + ).getUint32(0, false) + const transcriptStart = domain.byteLength + 4 + const secretStart = transcriptStart + transcriptLength + if (secretStart + 32 !== plaintext.byteLength) { + return null + } + return { + transcript: plaintext.slice(transcriptStart, secretStart), + secret: plaintext.slice(secretStart), + peerEphemeralPublicKey: peerKey, + nonce + } +} + +export function hostChallengeAckProof(input: { + secret: Uint8Array + transcript: Uint8Array + proofDomain: string +}): string { + return createHmac('sha256', input.secret) + .update(textEncoder.encode(`${input.proofDomain}\0ack\0`)) + .update(input.transcript) + .digest('base64') +} diff --git a/src/main/runtime/mobile-notification-dismissal-read-failure.test.ts b/src/main/runtime/mobile-notification-dismissal-read-failure.test.ts new file mode 100644 index 00000000000..a665e9b4444 --- /dev/null +++ b/src/main/runtime/mobile-notification-dismissal-read-failure.test.ts @@ -0,0 +1,38 @@ +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import type * as fs from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { expect, it, vi } from 'vitest' +import { MobileNotificationDismissalStore } from './mobile-notification-dismissal-store' +vi.mock('node:fs', async (original) => { + const f = await original() + return { ...f, readFileSync: vi.fn(f.readFileSync) } +}) +it('preserves dismissal history after EIO', () => { + const dir = mkdtempSync(join(tmpdir(), 'push-comment-')) + try { + const store = new MobileNotificationDismissalStore(dir) + store.record({ + type: 'dismiss', + notificationId: 'old', + notificationEpoch: 'epoch', + notificationSeq: 1 + }) + const path = join(dir, 'mobile-notification-dismissals.json') + const before = readFileSync(path, 'utf8') + vi.mocked(readFileSync).mockImplementationOnce(() => { + throw Object.assign(new Error('read failed'), { code: 'EIO' }) + }) + const restarted = new MobileNotificationDismissalStore(dir) + restarted.record({ + type: 'dismiss', + notificationId: 'new', + notificationEpoch: 'epoch', + notificationSeq: 2 + }) + expect(readFileSync(path, 'utf8')).toBe(before) + } finally { + vi.restoreAllMocks() + rmSync(dir, { recursive: true, force: true }) + } +}) diff --git a/src/main/runtime/mobile-notification-dismissal-store.test.ts b/src/main/runtime/mobile-notification-dismissal-store.test.ts new file mode 100644 index 00000000000..7679c55d31d --- /dev/null +++ b/src/main/runtime/mobile-notification-dismissal-store.test.ts @@ -0,0 +1,57 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, expect, it, vi } from 'vitest' +import { MobileNotificationDismissalStore } from './mobile-notification-dismissal-store' +const paths: string[] = [] +afterEach(() => { + paths.splice(0).forEach((path) => rmSync(path, { recursive: true, force: true })) + vi.restoreAllMocks() +}) +function fixture() { + const path = mkdtempSync(join(tmpdir(), 'orca-dismissals-')) + paths.push(path) + return { path, store: new MobileNotificationDismissalStore(path) } +} +const shown = { notificationId: 'same', notificationEpoch: 'old', notificationSeq: 12 } +const alert = { + type: 'notification' as const, + source: 'terminal-bell' as const, + title: 'QA', + body: '' +} +it('reconciles an old delivered alert after desktop restart and preserves unrelated identities', () => { + const h = fixture() + h.store.record({ ...alert, ...shown }) + const restarted = new MobileNotificationDismissalStore(h.path) + restarted.record({ + type: 'dismiss', + notificationId: 'same', + notificationEpoch: 'new', + notificationSeq: 1 + }) + const loaded = new MobileNotificationDismissalStore(h.path) + expect( + loaded.reconcile([ + shown, + { ...shown, notificationEpoch: 'other' }, + { ...shown, notificationId: 'other' }, + { ...shown, notificationSeq: 13 } + ]) + ).toEqual([shown]) +}) +it('does not dismiss a newer replacement and does not treat missing or expired history as dismissal', () => { + const h = fixture() + const now = Date.now() + vi.spyOn(Date, 'now').mockReturnValue(now) + h.store.record({ ...alert, ...shown }) + h.store.record({ type: 'dismiss', ...shown, notificationSeq: 13 }) + expect(h.store.reconcile([shown])).toEqual([shown]) + h.store.record({ ...alert, ...shown, notificationSeq: 14 }) + expect(h.store.reconcile([{ ...shown, notificationSeq: 14 }])).toEqual([]) + expect(h.store.reconcile([shown])).toEqual([shown]) + h.store.record({ type: 'dismiss', ...shown, notificationSeq: 15 }) + vi.mocked(Date.now).mockReturnValue(now + 7 * 86400_000) + expect(h.store.reconcile([shown])).toEqual([]) + expect(new MobileNotificationDismissalStore(`${h.path}-unknown`).reconcile([shown])).toEqual([]) +}) diff --git a/src/main/runtime/mobile-notification-dismissal-store.ts b/src/main/runtime/mobile-notification-dismissal-store.ts new file mode 100644 index 00000000000..984a8c9a1e0 --- /dev/null +++ b/src/main/runtime/mobile-notification-dismissal-store.ts @@ -0,0 +1,114 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { + writeSecureJsonFile, + hardenExistingSecureFile, + isUnreadableError +} from '../../shared/secure-file' +import type { MobileNotificationEvent } from './runtime-mobile-notification-controller' + +export type DeliveredNotificationIdentity = { + notificationId: string + notificationEpoch: string + notificationSeq: number +} +type RecordEntry = DeliveredNotificationIdentity & { dismissedThrough: number; expiresAt: number } +const LIMIT = 4096 +const RETENTION_MS = 7 * 86400_000 + +export class MobileNotificationDismissalStore { + private readonly path: string + private entries: RecordEntry[] = [] + private unreadable = false + constructor(userDataPath: string) { + this.path = join(userDataPath, 'mobile-notification-dismissals.json') + try { + hardenExistingSecureFile(this.path) + const value: unknown = JSON.parse(readFileSync(this.path, 'utf8')) + if (Array.isArray(value)) { + this.entries = value.filter(isEntry).slice(-LIMIT) + } + } catch (error) { + this.unreadable = isUnreadableError(error) + // Missing history cannot establish that a delivered alert was dismissed. + } + } + + record( + event: MobileNotificationEvent & { notificationEpoch: string; notificationSeq: number } + ): void { + if (!event.notificationId) { + return + } + const now = Date.now() + const kept = this.entries.filter((entry) => entry.expiresAt > now) + const same = (entry: RecordEntry) => + entry.notificationId === event.notificationId && + entry.notificationEpoch === event.notificationEpoch + let next: RecordEntry[] + if (event.type === 'notification') { + next = [ + ...kept.filter((entry) => !same(entry)), + { + notificationId: event.notificationId, + notificationEpoch: event.notificationEpoch, + notificationSeq: event.notificationSeq, + dismissedThrough: kept.find(same)?.dismissedThrough ?? -1, + expiresAt: now + RETENTION_MS + } + ] + } else { + next = kept + .filter((entry) => !same(entry)) + .map((entry) => + entry.notificationId === event.notificationId + ? { ...entry, dismissedThrough: entry.notificationSeq, expiresAt: now + RETENTION_MS } + : entry + ) + next.push({ + notificationId: event.notificationId, + notificationEpoch: event.notificationEpoch, + notificationSeq: event.notificationSeq, + dismissedThrough: event.notificationSeq, + expiresAt: now + RETENTION_MS + }) + } + next = next.slice(-LIMIT) + if (!this.unreadable) { + writeSecureJsonFile(this.path, next) + } + this.entries = next + } + + reconcile(delivered: readonly DeliveredNotificationIdentity[]): DeliveredNotificationIdentity[] { + const now = Date.now() + return delivered.filter((item) => + this.entries.some( + (entry) => + entry.dismissedThrough >= 0 && + entry.expiresAt > now && + entry.notificationId === item.notificationId && + entry.notificationEpoch === item.notificationEpoch && + entry.dismissedThrough >= item.notificationSeq + ) + ) + } +} + +function isEntry(value: unknown): value is RecordEntry { + if (!value || typeof value !== 'object') { + return false + } + const item = value as RecordEntry + return ( + typeof item.notificationId === 'string' && + item.notificationId.length > 0 && + typeof item.notificationEpoch === 'string' && + item.notificationEpoch.length > 0 && + Number.isSafeInteger(item.notificationSeq) && + item.notificationSeq >= 0 && + Number.isSafeInteger(item.dismissedThrough) && + item.dismissedThrough >= -1 && + Number.isFinite(item.expiresAt) + ) +} diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 3dbd8809018..ade25ad1975 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -10,7 +10,6 @@ import { applyRuntimeWorktreePsTerminalActivity } from './runtime-worktree-ps-activity' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' -import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import { compareWorktreePs } from './runtime-worktree-status-projection' import type { AgentSessionRecord } from '../../shared/agent-session-record' import type { Repo } from '../../shared/repo-types' @@ -109,9 +108,8 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent mirroredWorktreeIdByTabId, connectedPtyEvidence, retainedSnapshots: this.agentRows.values(), - hookSnapshots: this.getAgentStatusSnapshotFn?.() ?? [], - // Broadcast history outlives closed sessions; only the host roster is eligible. - structuredSummaries: getStructuredAgentSessionHost()?.liveSessionStatusSummaries() ?? [] + // Structured sessions are in here too: the host publishes them into the same store. + hookSnapshots: this.getAgentStatusSnapshotFn?.() ?? [] }), orchestrationByPaneKey: this.agentOrchestrationProjection.buildByPaneKey(), getSummary: (summaryMap, pathIndex, missingIds, worktreeId) => @@ -173,6 +171,7 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent firstWorkRenameDeps(this.requireStore(), this) ) }, + ...(this.structuredAgentStatusSinkFn ? { statusSink: this.structuredAgentStatusSinkFn } : {}), handoffTransport: this.createStructuredAgentSessionHandoffTransport() }) } diff --git a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts index 2e1357e3450..33b4ace7772 100644 --- a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts +++ b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts @@ -10,6 +10,7 @@ import type { } from './runtime-terminal-contracts' import type { TerminalSideEffectBatch } from '../../shared/terminal-side-effect-facts' import type { AgentStatusIpcPayload } from '../../shared/agent-status-types' +import type { StructuredAgentSessionStatusSink } from '../native-chat/agent-session-wire/structured-agent-session-status-feed' import type { ObservedAgentStatusPaneIdentity } from '../ipc/agent-status-ipc-boundary' import type { AgentHookAuthorityAttestation } from '../agent-hooks/server' import type { RuntimeDesktopWindowStatus } from '../../shared/runtime-types' @@ -66,6 +67,8 @@ export class OrcaRuntimeWithPreservedBranchCleanup extends OrcaRuntimeWithTermin protected readonly getAgentStatusSnapshotFn: (() => AgentStatusIpcPayload[]) | null + protected readonly structuredAgentStatusSinkFn: StructuredAgentSessionStatusSink | null + protected readonly readObservedAgentStatusPaneIdentityFn: ( paneKey: string ) => ObservedAgentStatusPaneIdentity diff --git a/src/main/runtime/orca-runtime-pty-foreground-process-reads.ts b/src/main/runtime/orca-runtime-pty-foreground-process-reads.ts index a7e10247fed..84161b64789 100644 --- a/src/main/runtime/orca-runtime-pty-foreground-process-reads.ts +++ b/src/main/runtime/orca-runtime-pty-foreground-process-reads.ts @@ -149,13 +149,17 @@ export class OrcaRuntimeWithPtyForegroundProcessReads extends OrcaRuntimeWithSta ...(allowUnverifiedStop ? { allowUnverifiedStop: true } : {}), ...(connectionId ? { includeLocalRegistry: false } : {}) }) + // Structured sessions are counted here too, mirroring the IPC path: closing a user's chat is + // now an ordinary outcome of this verb, and a removal that closed one but no PTY logged nothing. + const structuredStopped = teardownResult.structuredStopped ?? 0 const total = teardownResult.runtimeStopped + teardownResult.providerStopped + - teardownResult.registryStopped + teardownResult.registryStopped + + structuredStopped if (total > 0) { console.info( - `[worktree-teardown] ${worktreeId} killed runtime=${teardownResult.runtimeStopped} provider=${teardownResult.providerStopped} registry=${teardownResult.registryStopped}` + `[worktree-teardown] ${worktreeId} killed runtime=${teardownResult.runtimeStopped} provider=${teardownResult.providerStopped} registry=${teardownResult.registryStopped} structured=${structuredStopped}` ) } } diff --git a/src/main/runtime/orca-runtime-state-fields.ts b/src/main/runtime/orca-runtime-state-fields.ts index 1884df7ed12..876de03fe76 100644 --- a/src/main/runtime/orca-runtime-state-fields.ts +++ b/src/main/runtime/orca-runtime-state-fields.ts @@ -6,6 +6,7 @@ import type { IPtyProvider } from '../providers/types' import type { RuntimeTerminalAgentStatusEvent } from './runtime-terminal-contracts' import type { TerminalSideEffectBatch } from '../../shared/terminal-side-effect-facts' import type { AgentStatusIpcPayload } from '../../shared/agent-status-types' +import type { StructuredAgentSessionStatusSink } from '../native-chat/agent-session-wire/structured-agent-session-status-feed' import type { ObservedAgentStatusPaneIdentity } from '../ipc/agent-status-ipc-boundary' import type { AgentHookAuthorityAttestation } from '../agent-hooks/server' import type { @@ -49,6 +50,9 @@ export class OrcaRuntimeWithStateFields extends OrcaRuntimeWithLinearCommands { // terminal output. worktree.ps reads this at query time so mobile shows the // same inline agent rows the desktop sidebar does — same source, 1:1. getAgentStatusSnapshot?: () => AgentStatusIpcPayload[] + /** Where structured (native chat) sessions publish into that same store, so the snapshot + * above lists them like every other agent. */ + structuredAgentStatusSink?: StructuredAgentSessionStatusSink /** The identity the runtime resolved for a pane as each status arrived. Without it the * fleet path reminted cached rows against whatever the pane owns now. */ readObservedAgentStatusPaneIdentity?: (paneKey: string) => ObservedAgentStatusPaneIdentity @@ -188,6 +192,7 @@ export class OrcaRuntimeWithStateFields extends OrcaRuntimeWithLinearCommands { this.stats = stats } this.getAgentStatusSnapshotFn = deps?.getAgentStatusSnapshot ?? null + this.structuredAgentStatusSinkFn = deps?.structuredAgentStatusSink ?? null this.readObservedAgentStatusPaneIdentityFn = deps?.readObservedAgentStatusPaneIdentity ?? (() => ({ kind: 'unobserved' })) this.getAgentProviderSessionSnapshotFn = diff --git a/src/main/runtime/orca-runtime-structured-status-sink-wiring.test.ts b/src/main/runtime/orca-runtime-structured-status-sink-wiring.test.ts new file mode 100644 index 00000000000..decaa5cc599 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-status-sink-wiring.test.ts @@ -0,0 +1,93 @@ +import { describe, expect, it, vi } from 'vitest' + +const installed = vi.hoisted(() => ({ deps: null as Record | null })) + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +vi.mock('./structured-agent-session-runtime', () => ({ + ensureStructuredAgentSessionHost: vi.fn(async (deps: Record) => { + installed.deps = deps + }) +})) + +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { OrcaRuntimeService } from './orca-runtime' +import type { StructuredAgentSessionStatusSink } from '../native-chat/agent-session-wire/structured-agent-session-status-feed' + +type OrcaRuntimeDeps = NonNullable[2]> + +/** Renaming either option reddens this list, and dropping either from a host's construction + * reddens the assertion below. Both are needed: the runtime class does not typecheck its own + * `this` calls, and each entry point wires the store separately. */ +const AGENT_STATUS_STORE_DEPS = [ + 'getAgentStatusSnapshot', + 'structuredAgentStatusSink' +] as const satisfies readonly (keyof OrcaRuntimeDeps)[] + +const MAIN_ROOT = join(import.meta.dirname, '..') + +/** The text of the `new OrcaRuntimeService(...)` call in one entry point. */ +function runtimeConstruction(relativePath: string): string { + const source = readFileSync(join(MAIN_ROOT, relativePath), 'utf8') + const start = source.indexOf('new OrcaRuntimeService(') + expect(start).toBeGreaterThanOrEqual(0) + let depth = 0 + for (let index = source.indexOf('(', start); index < source.length; index += 1) { + const character = source[index] + if (character === '(') { + depth += 1 + } else if (character === ')') { + depth -= 1 + if (depth === 0) { + return source.slice(start, index + 1) + } + } + } + throw new Error(`unbalanced OrcaRuntimeService construction in ${relativePath}`) +} + +/** The runtime class this wiring lives on does not typecheck its own `this` calls, so a misnamed + * field here would install a host that never writes to the agent-status store — and every reader + * of that store would simply list no structured sessions. Pin it behaviourally. */ +/** `worktree ps` reads structured rows only from the agent-status store, so an entry point that + * constructs a runtime without these lists no agents at all — and `orcad` serves `worktree.ps` + * and `agentSession.*` exactly like the desktop does. */ +describe('every host that constructs a runtime wires the agent-status store', () => { + it.each([['orcad/orcad-entry.ts'], ['startup/main-process-runtime-service.ts']])( + '%s passes both store deps', + (relativePath) => { + const construction = runtimeConstruction(relativePath) + for (const dep of AGENT_STATUS_STORE_DEPS) { + expect(construction).toContain(`${dep}:`) + } + } + ) +}) + +describe('structured status sink wiring', () => { + it('hands the host the sink the runtime was constructed with', async () => { + installed.deps = null + const sink: StructuredAgentSessionStatusSink = { publish: vi.fn(), forget: vi.fn() } + const runtime = new OrcaRuntimeService(null, undefined, { structuredAgentStatusSink: sink }) + + await runtime.ensureStructuredAgentSessionHost() + + expect(installed.deps?.['statusSink']).toBe(sink) + }) + + it('installs without a sink when none was provided', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService() + + await runtime.ensureStructuredAgentSessionHost() + + expect(installed.deps).not.toBeNull() + expect('statusSink' in (installed.deps ?? {})).toBe(false) + }) +}) diff --git a/src/main/runtime/orca-runtime-tail-wait-memo.test.ts b/src/main/runtime/orca-runtime-tail-wait-memo.test.ts index f61431b7b86..5ce8d9710cd 100644 --- a/src/main/runtime/orca-runtime-tail-wait-memo.test.ts +++ b/src/main/runtime/orca-runtime-tail-wait-memo.test.ts @@ -134,7 +134,7 @@ describe('onPtyData tail wait memoization', () => { '' ) expect(blocked.fromTail).toBe(true) - expect(blocked.signal?.reason).toBe('codex-update-prompt') + expect(blocked.signal?.reason).toBe('agent-update-prompt') }) it('does not rebuild or repeatedly scan an ordinary saturated tail', () => { diff --git a/src/main/runtime/orca-runtime-tests/agent-status-and-waits.spec.ts b/src/main/runtime/orca-runtime-tests/agent-status-and-waits.spec.ts index 6fed887e741..7c0310554ef 100644 --- a/src/main/runtime/orca-runtime-tests/agent-status-and-waits.spec.ts +++ b/src/main/runtime/orca-runtime-tests/agent-status-and-waits.spec.ts @@ -247,7 +247,7 @@ describe('OrcaRuntimeService', () => { runtime.waitForTerminal(terminal.handle, { condition: 'tui-idle', timeoutMs: 1_000 }) ).resolves.toMatchObject({ satisfied: false, - blockedReason: 'codex-interactive-prompt' + blockedReason: 'agent-interactive-prompt' }) }) diff --git a/src/main/runtime/orca-runtime-tests/terminal-creation-and-readiness-part-07.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-creation-and-readiness-part-07.spec.ts index e829be07d9f..682661f0fce 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-creation-and-readiness-part-07.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-creation-and-readiness-part-07.spec.ts @@ -146,11 +146,11 @@ describe('OrcaRuntimeService', () => { condition: 'tui-idle', satisfied: false, status: 'running', - blockedReason: 'codex-hooks-review-prompt' + blockedReason: 'agent-hooks-review-prompt' }) }) - it('returns a blocked wait result for Codex update prompts', async () => { + it('returns an agent-neutral blocked wait result for update prompts', async () => { const runtime = new OrcaRuntimeService(store) runtime.setPtyController({ spawn: vi.fn().mockResolvedValue({ id: 'pty-bg' }), @@ -177,11 +177,11 @@ describe('OrcaRuntimeService', () => { condition: 'tui-idle', satisfied: false, status: 'running', - blockedReason: 'codex-update-prompt' + blockedReason: 'agent-update-prompt' }) }) - it('returns a blocked wait result for Codex workspace trust prompts', async () => { + it('returns an agent-neutral blocked wait result for workspace trust prompts', async () => { const runtime = new OrcaRuntimeService(store) runtime.setPtyController({ spawn: vi.fn().mockResolvedValue({ id: 'pty-bg' }), @@ -203,7 +203,7 @@ describe('OrcaRuntimeService', () => { condition: 'tui-idle', satisfied: false, status: 'running', - blockedReason: 'codex-trust-workspace' + blockedReason: 'agent-trust-workspace' }) }) @@ -270,7 +270,7 @@ describe('OrcaRuntimeService', () => { ).rejects.toThrow('timeout') }) - it('returns a blocked wait result for Codex cwd selection prompts', async () => { + it('returns an agent-neutral blocked wait result for cwd selection prompts', async () => { const runtime = new OrcaRuntimeService(store) runtime.setPtyController({ spawn: vi.fn().mockResolvedValue({ id: 'pty-bg' }), @@ -297,7 +297,7 @@ describe('OrcaRuntimeService', () => { condition: 'tui-idle', satisfied: false, status: 'running', - blockedReason: 'codex-cwd-prompt' + blockedReason: 'agent-cwd-prompt' }) }) @@ -359,11 +359,11 @@ describe('OrcaRuntimeService', () => { condition: 'tui-idle', satisfied: false, status: 'running', - blockedReason: 'codex-hooks-review-prompt' + blockedReason: 'agent-hooks-review-prompt' }) }) - it('returns a blocked wait result for generic Codex interactive prompts', async () => { + it('returns an agent-neutral blocked wait result for generic interactive prompts', async () => { const runtime = new OrcaRuntimeService(store) runtime.setPtyController({ spawn: vi.fn().mockResolvedValue({ id: 'pty-bg' }), @@ -390,7 +390,7 @@ describe('OrcaRuntimeService', () => { condition: 'tui-idle', satisfied: false, status: 'running', - blockedReason: 'codex-interactive-prompt' + blockedReason: 'agent-interactive-prompt' }) }) diff --git a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts index a2d45395b26..ecf2ab53678 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-output-and-worker-recovery-part-02.spec.ts @@ -623,7 +623,7 @@ describe('OrcaRuntimeService', () => { runtime.waitForTerminal(terminal.handle, { condition: 'tui-idle', timeoutMs: 100 }) ).resolves.toMatchObject({ satisfied: false, - blockedReason: 'codex-trust-workspace' + blockedReason: 'agent-trust-workspace' }) serializeProviderBuffer.mockImplementationOnce(() => new Promise(() => {})) await expect( diff --git a/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts index 58e7de1b674..e6db4b6af0a 100644 --- a/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts +++ b/src/main/runtime/orca-runtime-tests/worktree-ps-structured-host.spec.ts @@ -1,146 +1,75 @@ -import { afterEach, describe, expect, it, vi } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' import { OrcaRuntimeService } from '../orca-runtime-test-mocks.spec' import { TEST_WORKTREE_ID, store } from '../orca-runtime-test-fixtures.spec' -import type { AgentJournalRenderItem } from '../../../shared/agent-session-journal-types' -import type { - AgentSessionStatusEvent, - AgentSessionStatusSummary -} from '../../../shared/agent-session-wire' -import type { AgentSessionJournal } from '../../native-chat/agent-session-journal/journal-store' -import type { StructuredAgentSessionHost } from '../../native-chat/agent-session-wire/structured-agent-session-host' -import { - getStructuredAgentSessionHost, - setStructuredAgentSessionHost -} from '../../native-chat/agent-session-wire/structured-agent-session-registry' -import { StructuredAgentSessionStatusFeed } from '../../native-chat/agent-session-wire/structured-agent-session-status-feed' +import { AgentHookServer, _internals } from '../../agent-hooks/server' + +vi.mock('../../telemetry/client', () => ({ track: vi.fn() })) +vi.mock('../../telemetry/cohort-classifier', () => ({ + getCohortAtEmit: vi.fn(() => ({ nth_repo_added: 2 })) +})) /** - * The production wiring, not the projection. Both structured-row suites call - * `attachRuntimeWorktreeAgentRows` directly with summaries they built themselves, so nothing - * executed `getWorktreePs`'s own `getStructuredAgentSessionHost()?.liveSessionStatusSummaries()` - * — and that file carries `@ts-nocheck`, so renaming the accessor stayed green in typecheck AND - * in the suite while `orca worktree ps` and mobile's poll would throw for every user. + * The production read, not the projection. The structured-row suites feed + * `attachRuntimeWorktreeAgentRows` a snapshot they built themselves; this executes `getWorktreePs` + * in `orca-runtime-get-worktree-ps.ts`, which is `@ts-nocheck`, so a renamed dependency there stays + * green in typecheck while `orca worktree ps` and mobile's poll would list nothing. */ +const SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' -const HELD_SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' -const FORGOTTEN_SESSION = 'b2c3d4e5-f6a7-4b8c-9d0e-1f2a3b4c5d6e' -const OBSERVED_AT = 1_757_030_400_000 +beforeEach(() => { + _internals.resetCachesForTests() +}) -function runningTurn(prompt: string): AgentJournalRenderItem[] { - return [ - { - itemId: 'user-1', - sequence: 1, - revision: 1, - observedAt: OBSERVED_AT, - body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: prompt }] } - }, - { - itemId: 'turn-1', - sequence: 2, - revision: 1, - observedAt: OBSERVED_AT, - body: { - kind: 'status', - text: 'Working', - turnLifecycle: { turnId: 'turn-1', state: 'running' } - } - } - ] -} +describe('worktree ps reads structured sessions from the agent-status store', () => { + it('lists a host-held structured session with no terminal behind it', async () => { + const statusStore = new AgentHookServer() + statusStore.ingestStructuredStatus({ + sessionId: SESSION, + workspaceId: TEST_WORKTREE_ID, + agent: 'claude', + status: 'working', + hostExecutionOwned: true, + latestPrompt: 'ship the thing', + updatedAt: 1_757_030_400_000 + }) + const getAgentStatusSnapshot = vi.fn(() => statusStore.getStatusSnapshot()) -function journalWith(prompt: string): AgentSessionJournal { - return { - isReadOnly: false, - lastActivityAt: () => OBSERVED_AT, - snapshot: () => ({ items: runningTurn(prompt) }) - } as unknown as AgentSessionJournal -} - -/** A real feed the host still holds one session on, having forgotten the other. Its `published` - * cache never retracts, so the two views genuinely differ. */ -function statusFeed(): StructuredAgentSessionStatusFeed { - const session = (prompt: string) => ({ - journal: journalWith(prompt), - params: { location: { workspaceId: TEST_WORKTREE_ID }, provider: 'claude' as const } - }) - const sessions = new Map([ - [HELD_SESSION, session('ship the thing')], - [FORGOTTEN_SESSION, session('rm the branch')] - ]) - const feed = new StructuredAgentSessionStatusFeed({ - sessions, - getRecord: () => null, - now: () => OBSERVED_AT - }) - feed.publish(HELD_SESSION) - feed.publish(FORGOTTEN_SESSION) - // `forget-session`, the last eviction step, drops the session and leaves the feed alone. - sessions.delete(FORGOTTEN_SESSION) - return feed -} - -/** Everything the feed retained, read through the snapshot a subscriber opens on. No production - * code reads this; it is here so swapping the call site back to a whole-cache read is a one-line - * edit that this suite must catch. */ -function retainedSummaries(feed: StructuredAgentSessionStatusFeed): AgentSessionStatusSummary[] { - let retained: AgentSessionStatusSummary[] = [] - feed.subscribe({ - id: 'retained-probe', - emit: (event: AgentSessionStatusEvent) => { - if (event.type === 'snapshot') { - retained = event.sessions - } - } - })() - return retained -} - -function installHost(feed: StructuredAgentSessionStatusFeed) { - const liveSessionStatusSummaries = vi.fn(() => feed.liveSessionSummaries()) - const retainedSessionStatusSummaries = vi.fn(() => retainedSummaries(feed)) - // Typed against the real host, so renaming the accessor on the class reddens `tc` here — the - // caller cannot, because `orca-runtime-get-worktree-ps.ts` is `@ts-nocheck`. - const host: Pick & { - retainedSessionStatusSummaries: () => AgentSessionStatusSummary[] - } = { liveSessionStatusSummaries, retainedSessionStatusSummaries } - setStructuredAgentSessionHost(host as unknown as StructuredAgentSessionHost) - return { liveSessionStatusSummaries, retainedSessionStatusSummaries } -} - -describe('worktree ps reads the installed structured host', () => { - afterEach(() => { - setStructuredAgentSessionHost(null) - }) - - it('reports the held session and asks the host for its live summaries', async () => { - const feed = statusFeed() - const { liveSessionStatusSummaries } = installHost(feed) - - const { worktrees } = await new OrcaRuntimeService(store).getWorktreePs() + const { worktrees } = await new OrcaRuntimeService(store, undefined, { + getAgentStatusSnapshot + }).getWorktreePs() const worktree = worktrees.find((entry) => entry.worktreeId === TEST_WORKTREE_ID) expect(worktree).toBeDefined() - // Exactly one: the forgotten session is still in the feed's retained cache, so a call site - // that enumerated that cache instead would report two. expect(worktree?.agents).toHaveLength(1) expect(worktree?.agents[0]).toMatchObject({ state: 'working', agentType: 'claude', - prompt: 'ship the thing' + prompt: 'ship the thing', + structuredHostOwned: true }) - // Pins the call site to the live-intersecting accessor, not merely to some accessor. - expect(liveSessionStatusSummaries).toHaveBeenCalledTimes(1) + expect(worktree?.status).toBe('working') + expect(getAgentStatusSnapshot).toHaveBeenCalledTimes(1) }) - it('succeeds with no structured rows when no host is installed', async () => { - // Guard the guard: these specs share one module registry, so state the premise. - expect(getStructuredAgentSessionHost()).toBeNull() + it('lists nothing once the host has dropped the session', async () => { + const statusStore = new AgentHookServer() + statusStore.ingestStructuredStatus({ + sessionId: SESSION, + workspaceId: TEST_WORKTREE_ID, + agent: 'claude', + status: 'attention', + latestPrompt: 'rm the branch', + updatedAt: 1_757_030_400_000 + }) + statusStore.dropStructuredStatus(SESSION) - const { worktrees } = await new OrcaRuntimeService(store).getWorktreePs() + const { worktrees } = await new OrcaRuntimeService(store, undefined, { + getAgentStatusSnapshot: () => statusStore.getStatusSnapshot() + }).getWorktreePs() const worktree = worktrees.find((entry) => entry.worktreeId === TEST_WORKTREE_ID) expect(worktree).toBeDefined() expect(worktree?.agents).toEqual([]) + expect(worktree?.status).not.toBe('permission') }) }) diff --git a/src/main/runtime/orchestration/groups.ts b/src/main/runtime/orchestration/groups.ts index a4e548f06d0..7c63a9bf475 100644 --- a/src/main/runtime/orchestration/groups.ts +++ b/src/main/runtime/orchestration/groups.ts @@ -3,7 +3,9 @@ import type { OrchestrationAddressableAgent } from './structured-worker-group-ad // Why: group addresses enable broadcast messaging to logical groups of agents. // Resolution is done at send-time: one message record per recipient, same thread_id, -// so each recipient gets their own read-tracking (Section 4.5). +// so each recipient gets their own read-tracking (Section 4.5). The caller picks the +// candidates: the sender's Run for every group but `@worktree:`, which names one +// workspace explicitly. There is no host-wide candidate set. const AGENT_NAME_GROUPS = [ 'claude', @@ -71,7 +73,7 @@ export function resolveGroupAddress( const group = to.toLowerCase() if (group === '@all') { - // Why: @all broadcasts to every terminal except the sender to avoid self-delivery loops. + // Why: every candidate except the sender, to avoid self-delivery loops. return terminals.map((t) => t.handle).filter((h) => h !== senderHandle) } diff --git a/src/main/runtime/orchestration/mailbox-pointer-delivery.ts b/src/main/runtime/orchestration/mailbox-pointer-delivery.ts index 11fbc08229d..b16e5c474b3 100644 --- a/src/main/runtime/orchestration/mailbox-pointer-delivery.ts +++ b/src/main/runtime/orchestration/mailbox-pointer-delivery.ts @@ -9,6 +9,10 @@ import { OrchestrationMailboxPointerState, type OrchestrationMailboxDeliveryFlight } from './mailbox-pointer-state' +import { + MAILBOX_POINTER_RESERVED, + MAILBOX_POINTER_WRITE_ATTEMPTED +} from './db/messages/mailbox-pointer-enter-state' import { resumePendingOrchestrationMailboxPointer } from './mailbox-pointer-resume' import { stageOrchestrationMailboxPointer } from './mailbox-pointer-stage' @@ -163,7 +167,19 @@ export class OrchestrationMailboxPointerDelivery { db.close() }) }) + +describe('retiring a pty mid-delivery', () => { + // Why: an Enter that was already written may have landed. Releasing it would send the same + // mail a second time, so only phases that provably never submitted become redeliverable. + it('leaves an attempted Enter at its phase instead of making it redeliverable', async () => { + vi.useFakeTimers() + const db = new OrchestrationDb(':memory:') + const settlements: ((settlement: WriteSettlement) => void)[] = [] + const writePty = vi.fn( + () => + new Promise((resolve) => { + settlements.push(resolve) + }) as unknown as WriteSettlement + ) + try { + const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'mail' }) + const delivery = new OrchestrationMailboxPointerDelivery(pointerDeps(db, writePty) as never) + delivery.deliver(LEAF, { mailboxHandle: 'run:run-1' }) + + // Settle the pointer write, so the pane reaches WRITE_ATTEMPTED and arms the Enter. + settlements[0]?.(WRITE_ACCEPTED) + await vi.advanceTimersByTimeAsync(0) + expect(db.getMessageById(message.id)?.pointer_enter_pending).toBe(2) + + // Fire the Enter but never settle it: this is the ambiguous state. + await vi.advanceTimersByTimeAsync(600) + expect(db.getMessageById(message.id)?.pointer_enter_pending).toBe(3) + + delivery.retirePty('pty-1') + expect(db.getMessageById(message.id)).toMatchObject({ + pointer_enter_pending: 3, + read: 0 + }) + } finally { + db.close() + vi.useRealTimers() + } + }) + + it('releases a pointer whose Enter never fired', async () => { + vi.useFakeTimers() + const db = new OrchestrationDb(':memory:') + const settlements: ((settlement: WriteSettlement) => void)[] = [] + const writePty = vi.fn( + () => + new Promise((resolve) => { + settlements.push(resolve) + }) as unknown as WriteSettlement + ) + try { + const message = db.insertMessage({ from: 'a', to: 'run:run-1', subject: 'mail' }) + const delivery = new OrchestrationMailboxPointerDelivery(pointerDeps(db, writePty) as never) + delivery.deliver(LEAF, { mailboxHandle: 'run:run-1' }) + settlements[0]?.(WRITE_ACCEPTED) + await vi.advanceTimersByTimeAsync(0) + expect(db.getMessageById(message.id)?.pointer_enter_pending).toBe(2) + + delivery.retirePty('pty-1') + expect(db.getMessageById(message.id)).toMatchObject({ + pointer_enter_pending: 0, + read: 0, + delivered_at: null + }) + } finally { + db.close() + vi.useRealTimers() + } + }) +}) diff --git a/src/main/runtime/orchestration/mailbox-pointer-stage.ts b/src/main/runtime/orchestration/mailbox-pointer-stage.ts index aed3b8e06b4..3c0bea2baad 100644 --- a/src/main/runtime/orchestration/mailbox-pointer-stage.ts +++ b/src/main/runtime/orchestration/mailbox-pointer-stage.ts @@ -58,6 +58,7 @@ export function stageOrchestrationMailboxPointer message.id) try { if ( diff --git a/src/main/runtime/orchestration/mailbox-pointer-state.ts b/src/main/runtime/orchestration/mailbox-pointer-state.ts index 149d25057b5..d52374a4888 100644 --- a/src/main/runtime/orchestration/mailbox-pointer-state.ts +++ b/src/main/runtime/orchestration/mailbox-pointer-state.ts @@ -6,6 +6,8 @@ export type OrchestrationMailboxDeliveryFlight = { submitEnter: (() => void) | null deferredUntilIdle: boolean idleObservedWhileDeferred: boolean + /** The incarnation that staged this flight, so retirement can name the rows it owns. */ + processIncarnation?: string } export type ParkedOrchestrationMailboxDelivery = { diff --git a/src/main/runtime/orchestration/structured-mailbox-pointer-host.ts b/src/main/runtime/orchestration/structured-mailbox-pointer-host.ts index 4b80b8df5d7..7704e549d2f 100644 --- a/src/main/runtime/orchestration/structured-mailbox-pointer-host.ts +++ b/src/main/runtime/orchestration/structured-mailbox-pointer-host.ts @@ -33,10 +33,10 @@ export function structuredSessionPointerCallerKey(sessionId: string): string { /** * The idle gate for a structured session, read off its FULL reduced timeline. * - * Never a bounded page. Settlement tombstones the running turn's lifecycle item rather than - * rewriting it to `completed`, so on any tail window an idle session and a busy one whose - * lifecycle item scrolled off look identical — and idle-with-history is the normal steady state of - * a working agent. Shared so the pointer lane and group addressing cannot disagree about it. + * Never a bounded page. A settled turn's lifecycle item is revised in place, so on any tail window + * an idle session and a busy one whose lifecycle item scrolled off look identical — and + * idle-with-history is the normal steady state of a working agent. Shared so the pointer lane and + * group addressing cannot disagree about it. */ export function readStructuredSessionGateFacts( sessionId: string diff --git a/src/main/runtime/orchestration/structured-worker-group-addressing.test.ts b/src/main/runtime/orchestration/structured-worker-group-addressing.test.ts index ce9a9fbaf37..93fa58fd6b5 100644 --- a/src/main/runtime/orchestration/structured-worker-group-addressing.test.ts +++ b/src/main/runtime/orchestration/structured-worker-group-addressing.test.ts @@ -180,7 +180,7 @@ describe('sendGroupMessage actually composes structured workers in', () => { * describes could come straight back. This test owns that seam. */ it('addresses a structured worker that only the call site can enumerate', async () => { - const handle = registerWorker() + const handle = registerWorker('wt_2') installHost({}) const inserted: { to: string }[] = [] const db = { @@ -206,7 +206,9 @@ describe('sendGroupMessage actually composes structured workers in', () => { runtime: runtime as never, db: db as never, from: 'term_sender', - groupAddress: '@all', + // `@worktree:` is the one group that still enumerates agents; the Run groups read + // Dispatch rows, where a structured worker's identity is looked up by this same handle. + groupAddress: '@worktree:wt_2', senderPaneKey: undefined, senderRunId: undefined, explicitRunId: undefined, @@ -217,4 +219,63 @@ describe('sendGroupMessage actually composes structured workers in', () => { } as never) expect(inserted.map((row) => row.to)).toEqual([handle]) }) + + it('reads a structured worker identity for @codex off its Dispatch row handle', async () => { + const handle = registerWorker() + installHost({}) + const inserted: { to: string }[] = [] + const db = { + getLegacyAdoptedRunMailboxOwner: () => null, + getCurrentRunForPane: () => undefined, + getActiveDispatchForIdentity: () => ({ run_id: 'run_1' }), + getActiveDispatchMailboxOwners: () => [], + getRunMailboxOwnerIdsForHandle: () => [], + listWorkerTerminalResources: () => [ + { + dispatchId: 'ctx_structured', + runId: 'run_1', + dispatchStatus: 'dispatched', + agentTerminalHandle: handle, + worktreeId: 'wt_1' + }, + { + dispatchId: 'ctx_pty_claude', + runId: 'run_1', + dispatchStatus: 'dispatched', + agentTerminalHandle: 'term_claude', + worktreeId: 'wt_1' + } + ], + listFederatedDispatchesByIds: () => [], + getDispatchContextById: () => undefined, + insertMessages: (rows: { to: string }[]) => { + inserted.push(...rows) + return rows.map((row, index) => ({ id: `m${index}`, to_handle: row.to, type: 'status' })) + } + } + const runtime = { + listTerminals: async () => ({ + terminals: [{ handle: 'term_claude', worktreeId: 'wt_1', agentIdentity: 'claude' }] + }), + getAgentStatusForHandle: () => 'idle', + getLiveTerminalPaneKey: () => null, + getOrchestrationDb: () => db, + notifyMessageArrived: () => {} + } + await sendGroupMessage({ + params: { subject: 's', body: 'b', type: 'status', priority: 'normal' }, + runtime: runtime as never, + db: db as never, + from: 'term_sender', + groupAddress: '@codex', + senderPaneKey: undefined, + senderRunId: 'run_1', + explicitRunId: undefined, + legacyCoordinatorRunId: undefined, + revalidateLegacyCoordinator: undefined, + recordMutationReceipt: undefined, + withSendWarnings: (receipt) => receipt + } as never) + expect(inserted.map((row) => row.to)).toEqual(['dispatch:ctx_structured']) + }) }) diff --git a/src/main/runtime/push/desktop-push-service-unreadable-outbox.test.ts b/src/main/runtime/push/desktop-push-service-unreadable-outbox.test.ts new file mode 100644 index 00000000000..b3e98418460 --- /dev/null +++ b/src/main/runtime/push/desktop-push-service-unreadable-outbox.test.ts @@ -0,0 +1,89 @@ +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import type * as fs from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { expect, it, vi } from 'vitest' +import { DeviceRegistry } from '../device-registry' +import { DesktopPushService } from './desktop-push-service' +import { createPushHostKeypair } from './push-host-challenge-fixtures' +import { PushUnregisterOutbox } from './push-unregister-outbox' + +vi.mock('node:fs', async (importOriginal) => { + const original = await importOriginal() + return { ...original, readFileSync: vi.fn(original.readFileSync) } +}) + +it('refuses registration until unreadable cleanup is recovered and settled on restart', async () => { + const dir = mkdtempSync(join(tmpdir(), 'orca-push-unreadable-')) + let service: DesktopPushService | null = null + try { + let registry = new DeviceRegistry(dir) + const { deviceId } = registry.addDevice('phone', 'mobile') + const queued = new PushUnregisterOutbox(dir).enqueue({ deviceId, registrationId: 'stable-id' }) + const path = join(dir, 'mobile-push-unregister-outbox.json') + const bytes = readFileSync(path, 'utf-8') + vi.mocked(readFileSync).mockImplementationOnce(() => { + throw Object.assign(new Error('temporarily unavailable'), { code: 'EIO' }) + }) + const unreadable = new PushUnregisterOutbox(dir) + let gatewayLive = true + const calls: string[] = [] + const client = { + registerDevice: vi.fn(async () => { + calls.push('register') + gatewayLive = true + return { ok: true, registrationId: 'stable-id' } as const + }), + deleteDevice: vi.fn(async () => { + calls.push('delete') + gatewayLive = false + return true + }) + } + const createService = (outbox: PushUnregisterOutbox): DesktopPushService => + DesktopPushService.create({ + runtime: { + setMobilePushRegistrar: vi.fn(), + onNotificationDispatched: () => () => {} + } as never, + runtimeRpc: { + getE2EEKeypair: createPushHostKeypair, + getDeviceRegistry: () => registry, + getPushUnregisterOutbox: () => outbox, + setOnPushUnregisterQueued: vi.fn() + } as never, + client: client as never, + gatewayUrl: 'https://push.invalid', + scheduleRetry: vi.fn() + })! + const input = { deviceId, platform: 'android' as const, token: 'synthetic', filter: {} } + service = createService(unreadable) + service.start() + expect(await service.register(input)).toEqual({ + registered: false, + reason: 'registration_storage_failed' + }) + expect(client.registerDevice).not.toHaveBeenCalled() + expect(client.deleteDevice).not.toHaveBeenCalled() + expect(registry.getDevice(deviceId)?.pushRegistration).toBeUndefined() + expect(readFileSync(path, 'utf-8')).toBe(bytes) + service.stop() + + registry = new DeviceRegistry(dir) + const recovered = new PushUnregisterOutbox(dir) + expect(recovered.pending()).toEqual([queued]) + service = createService(recovered) + service.start() + expect(await service.register(input)).toEqual({ registered: true, registrationId: 'stable-id' }) + await service.flushUnregisterOutbox() + expect(calls).toEqual(['delete', 'register']) + expect(gatewayLive).toBe(true) + expect(new DeviceRegistry(dir).getDevice(deviceId)?.pushRegistration?.registrationId).toBe( + 'stable-id' + ) + expect(new PushUnregisterOutbox(dir).pending()).toEqual([]) + } finally { + service?.stop() + rmSync(dir, { recursive: true, force: true }) + } +}) diff --git a/src/main/runtime/push/desktop-push-service.test.ts b/src/main/runtime/push/desktop-push-service.test.ts new file mode 100644 index 00000000000..1300b43c3fb --- /dev/null +++ b/src/main/runtime/push/desktop-push-service.test.ts @@ -0,0 +1,315 @@ +import { mkdtempSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import type { MobileNotificationEvent } from '../runtime-mobile-notification-controller' +import { DeviceRegistry } from '../device-registry' +import { DesktopPushService } from './desktop-push-service' +import { PushRegisterThrottle } from './push-register-throttle' +import { PushUnregisterOutbox } from './push-unregister-outbox' +import { createPushHostKeypair } from './push-host-challenge-fixtures' + +const REGISTER_INPUT = { + platform: 'android' as const, + token: 'fcm-token', + filter: {} +} + +function createService( + options: { + registerFails?: boolean + deleteFails?: boolean + /** Runs before each delete resolves, so a suite can queue work mid-flush. */ + onDelete?: (registrationId: string) => void + now?: () => number + } = {} +): { + service: DesktopPushService + registry: DeviceRegistry + outbox: PushUnregisterOutbox + deviceId: string + deletes: string[] + send: ReturnType + dispatch: (event: MobileNotificationEvent) => void + retries: { run: () => void; delayMs: number }[] +} { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-push-service-')) + const registry = new DeviceRegistry(userDataPath) + const outbox = new PushUnregisterOutbox(userDataPath) + const device = registry.addDevice('phone', 'mobile') + const deletes: string[] = [] + let listener: ((event: MobileNotificationEvent) => void) | null = null + + const runtime = { + setMobilePushRegistrar: vi.fn(), + onNotificationDispatched: vi.fn((next: (event: MobileNotificationEvent) => void) => { + listener = next + return () => { + listener = null + } + }) + } + const runtimeRpc = { + getE2EEKeypair: () => createPushHostKeypair(), + getDeviceRegistry: () => registry, + getPushUnregisterOutbox: () => outbox, + setOnPushUnregisterQueued: vi.fn() + } + // A stub gateway keeps the suite on the service's own persistence decisions. + const client = { + registerDevice: vi.fn(async () => + options.registerFails + ? ({ ok: false, reason: 'unreachable' } as const) + : ({ ok: true, registrationId: 'reg-1' } as const) + ), + deleteDevice: vi.fn(async (registrationId: string) => { + deletes.push(registrationId) + options.onDelete?.(registrationId) + return !options.deleteFails + }), + send: vi.fn(async () => ({ ok: true, results: [] }) as const) + } + const retries: { run: () => void; delayMs: number }[] = [] + const service = DesktopPushService.create({ + runtime: runtime as never, + runtimeRpc: runtimeRpc as never, + gatewayUrl: 'https://push.onorca.dev', + client: client as never, + scheduleRetry: (run, delayMs) => { + retries.push({ run, delayMs }) + }, + ...(options.now ? { registerThrottle: new PushRegisterThrottle({ now: options.now }) } : {}) + })! + + service.start() + return { + service, + registry, + outbox, + deviceId: device.deviceId, + deletes, + send: client.send, + dispatch: (event) => listener?.(event), + retries + } +} + +describe('DesktopPushService', () => { + it('persists the registration the gateway hands back', async () => { + const harness = createService() + + expect( + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + ).toEqual({ registered: true, registrationId: 'reg-1' }) + expect(harness.registry.getDevice(harness.deviceId)?.pushRegistration).toMatchObject({ + registrationId: 'reg-1', + filter: REGISTER_INPUT.filter + }) + }) + + it('persists nothing when the gateway is unreachable', async () => { + const harness = createService({ registerFails: true }) + + expect( + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + ).toEqual({ registered: false, reason: 'gateway_unreachable' }) + expect(harness.registry.getDevice(harness.deviceId)?.pushRegistration).toBeUndefined() + }) + + it('refuses to register a device that is not a paired phone', async () => { + const harness = createService() + + expect(await harness.service.register({ deviceId: 'not-a-device', ...REGISTER_INPUT })).toEqual( + { + registered: false, + reason: 'not_mobile' + } + ) + }) + + it('clears the local registration and deletes at the gateway on unregister', async () => { + const harness = createService() + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + + expect(await harness.service.unregister(harness.deviceId)).toEqual({ unregistered: true }) + await harness.service.flushUnregisterOutbox() + expect(harness.registry.getDevice(harness.deviceId)?.pushRegistration).toBeUndefined() + expect(harness.deletes).toEqual(['reg-1']) + expect(harness.outbox.pending()).toEqual([]) + }) + + it('keeps the delete queued when the gateway cannot be reached', async () => { + const harness = createService({ deleteFails: true }) + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + + await harness.service.unregister(harness.deviceId) + + expect(harness.registry.getDevice(harness.deviceId)?.pushRegistration).toBeUndefined() + expect(harness.outbox.pending()).toEqual([ + expect.objectContaining({ registrationId: 'reg-1', deviceId: harness.deviceId }) + ]) + }) + + it('reports nothing to unregister for a device that never enabled push', async () => { + const harness = createService() + expect(await harness.service.unregister(harness.deviceId)).toEqual({ unregistered: false }) + }) + + it('drains a delete queued before this launch', async () => { + const harness = createService() + harness.outbox.enqueue({ registrationId: 'reg-stale', deviceId: 'device-gone' }) + + await harness.service.flushUnregisterOutbox() + + expect(harness.deletes).toEqual(['reg-stale']) + expect(harness.outbox.pending()).toEqual([]) + }) + + it('unregisters at the gateway when the device stopped being a phone mid-register', async () => { + const harness = createService() + vi.spyOn(harness.registry, 'setPushRegistration').mockReturnValue(false) + + expect( + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + ).toEqual({ registered: false, reason: 'not_mobile' }) + // register() kicks the flush off without awaiting it; join the same run. + await harness.service.flushUnregisterOutbox() + expect(harness.deletes).toEqual(['reg-1']) + expect(harness.outbox.pending()).toEqual([]) + }) + + it('unregisters at the gateway when the registration cannot be written', async () => { + const harness = createService({ deleteFails: true }) + vi.spyOn(harness.registry, 'setPushRegistration').mockImplementation(() => { + throw new Error('disk full') + }) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + expect( + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + ).toEqual({ registered: false, reason: 'registration_storage_failed' }) + // The gateway kept the token, so the delete stays queued until it lands. + expect(harness.outbox.pending()).toEqual([ + expect.objectContaining({ registrationId: 'reg-1', deviceId: harness.deviceId }) + ]) + warn.mockRestore() + }) + + it('drains a delete queued while a flush is already running', async () => { + let queued = false + const harness = createService({ + onDelete: () => { + if (queued) { + return + } + queued = true + harness.outbox.enqueue({ registrationId: 'reg-late', deviceId: 'device-late' }) + // Mirrors unregister(): the trigger arrives while the flush is mid-await. + void harness.service.flushUnregisterOutbox() + } + }) + harness.outbox.enqueue({ registrationId: 'reg-first', deviceId: 'device-first' }) + + await harness.service.flushUnregisterOutbox() + + expect(harness.deletes).toEqual(['reg-first', 'reg-late']) + expect(harness.outbox.pending()).toEqual([]) + }) + + it('retries a failed drain on a capped backoff instead of waiting for a relaunch', async () => { + const harness = createService({ deleteFails: true }) + harness.outbox.enqueue({ registrationId: 'reg-stuck', deviceId: 'device-1' }) + + await harness.service.flushUnregisterOutbox() + expect(harness.retries.map((entry) => entry.delayMs)).toEqual([30_000]) + + harness.retries[0]?.run() + await new Promise((resolve) => setImmediate(resolve)) + expect(harness.deletes).toEqual(['reg-stuck', 'reg-stuck']) + expect(harness.retries.map((entry) => entry.delayMs)).toEqual([30_000, 60_000]) + expect(harness.outbox.pending()).toHaveLength(1) + }) + + it('stops re-arming the retry once the service is stopped', async () => { + const harness = createService({ deleteFails: true }) + harness.outbox.enqueue({ registrationId: 'reg-stuck', deviceId: 'device-1' }) + await harness.service.flushUnregisterOutbox() + + harness.service.stop() + harness.retries[0]?.run() + await new Promise((resolve) => setImmediate(resolve)) + + expect(harness.retries).toHaveLength(1) + }) + + it('throttles a device that registers in a loop and lets it back in a minute later', async () => { + let clock = 1_700_000_000_000 + const harness = createService({ now: () => clock }) + const input = { deviceId: harness.deviceId, ...REGISTER_INPUT } + + for (let index = 0; index < 10; index++) { + expect(await harness.service.register(input)).toEqual({ + registered: true, + registrationId: 'reg-1' + }) + } + expect(await harness.service.register(input)).toEqual({ + registered: false, + reason: 'throttled' + }) + // The registration it already made stands; only the new write is refused. + expect(harness.registry.getDevice(harness.deviceId)?.pushRegistration?.registrationId).toBe( + 'reg-1' + ) + + clock += 60_000 + expect(await harness.service.register(input)).toEqual({ + registered: true, + registrationId: 'reg-1' + }) + }) + + it('pushes a dispatched notification through the subscribed dispatcher', async () => { + const harness = createService() + await harness.service.register({ deviceId: harness.deviceId, ...REGISTER_INPUT }) + + harness.dispatch({ + type: 'notification', + source: 'agent-task-complete', + title: 'feat/x - Claude finished', + body: 'Done.', + notificationSeq: 3, + notificationEpoch: 'epoch-1', + agentState: 'done' + }) + await new Promise((resolve) => setImmediate(resolve)) + + expect(harness.send).toHaveBeenCalledWith( + expect.objectContaining({ registrationIds: ['reg-1'] }) + ) + }) +}) + +it('renews a seven-day mobile lease only on explicit registration', async () => { + const now = 1_800_000_000_000 + const clock = vi.spyOn(Date, 'now').mockReturnValue(now) + const h = createService() + try { + await h.service.register({ + deviceId: h.deviceId, + ...REGISTER_INPUT + }) + expect(h.registry.getDevice(h.deviceId)?.pushRegistration?.expiresAt).toBe(now + 7 * 86400_000) + clock.mockReturnValue(now + 86400_000) + h.dispatch({ type: 'notification', source: 'terminal-bell', title: 'QA', body: 'QA' }) + expect(h.registry.getDevice(h.deviceId)?.pushRegistration?.expiresAt).toBe(now + 7 * 86400_000) + await h.service.register({ + deviceId: h.deviceId, + ...REGISTER_INPUT + }) + expect(h.registry.getDevice(h.deviceId)?.pushRegistration?.expiresAt).toBe(now + 8 * 86400_000) + } finally { + h.service.stop() + clock.mockRestore() + } +}) diff --git a/src/main/runtime/push/desktop-push-service.ts b/src/main/runtime/push/desktop-push-service.ts new file mode 100644 index 00000000000..69bf810d026 --- /dev/null +++ b/src/main/runtime/push/desktop-push-service.ts @@ -0,0 +1,284 @@ +// Why: owns the desktop half of background push — the gateway session, the +// registration each paired phone asked for, and the durable delete queue. Built +// alongside DesktopRelayService but deliberately not gated on cloud sign-in: the +// gateway authenticates with the host keypair, so accountless hosts push too. +import type { + MobilePushRegisterInput, + MobilePushRegisterResult +} from '../../../shared/mobile-push-contract' +import { runKeyedSerializedOperation } from '../../cli/keyed-promise-queue' +import type { DeviceRegistry } from '../device-registry' +import type { OrcaRuntimeService } from '../orca-runtime' +import type { OrcaRuntimeRpcServer } from '../runtime-rpc' +import { PushDispatcher } from './push-dispatcher' +import { PushGatewayClient } from './push-gateway-client' +import { PushRegisterThrottle } from './push-register-throttle' +import type { PushUnregisterOutbox } from './push-unregister-outbox' + +const OUTBOX_RETRY_BASE_MS = 30_000 +const OUTBOX_RETRY_MAX_MS = 10 * 60_000 + +type RegisterStorageFailure = 'not_mobile' | 'registration_storage_failed' + +type DesktopPushServiceOptions = { + runtime: OrcaRuntimeService + runtimeRpc: OrcaRuntimeRpcServer + gatewayUrl: string + /** Test seam: lets a suite drive the service without a live gateway. */ + client?: PushGatewayClient + /** Test seam: lets a suite drive the outbox backoff without real timers. */ + scheduleRetry?: (run: () => void, delayMs: number) => void + /** Test seam: lets a suite drive the per-device register bucket on its own clock. */ + registerThrottle?: PushRegisterThrottle +} + +export class DesktopPushService { + private readonly runtime: OrcaRuntimeService + private readonly runtimeRpc: OrcaRuntimeRpcServer + private readonly registry: DeviceRegistry + private readonly outbox: PushUnregisterOutbox + private readonly client: PushGatewayClient + private readonly dispatcher: PushDispatcher + private readonly registerThrottle: PushRegisterThrottle + private readonly scheduleRetry: (run: () => void, delayMs: number) => void + private unsubscribe: (() => void) | null = null + private flushLoop: Promise | null = null + private flushRequested = false + private retryArmed = false + private retryDelayMs = OUTBOX_RETRY_BASE_MS + private stopped = false + private readonly deviceOperations = new Map>() + + private constructor( + options: DesktopPushServiceOptions, + registry: DeviceRegistry, + client: PushGatewayClient + ) { + this.runtime = options.runtime + this.runtimeRpc = options.runtimeRpc + this.registry = registry + this.client = client + this.outbox = options.runtimeRpc.getPushUnregisterOutbox() + this.dispatcher = new PushDispatcher({ client, registry }) + this.registerThrottle = options.registerThrottle ?? new PushRegisterThrottle() + this.scheduleRetry = + options.scheduleRetry ?? + ((run, delayMs) => { + // Why: a queued gateway delete must never hold the app open at quit. + setTimeout(run, delayMs).unref?.() + }) + } + + /** Returns null when the mobile runtime never came up, so there is nothing to push for. */ + static create(options: DesktopPushServiceOptions): DesktopPushService | null { + const keypair = options.runtimeRpc.getE2EEKeypair() + const registry = options.runtimeRpc.getDeviceRegistry() + if (!keypair || !registry) { + return null + } + const client = + options.client ?? new PushGatewayClient({ gatewayUrl: options.gatewayUrl, keypair }) + return new DesktopPushService(options, registry, client) + } + + start(): void { + this.stopped = false + this.dispatcher.start() + this.runtime.setMobilePushRegistrar(this) + this.unsubscribe = this.runtime.onNotificationDispatched((event) => { + this.dispatcher.enqueue(event) + }) + // Unpairing queues a delete without going through this service; drain on that too. + this.runtimeRpc.setOnPushUnregisterQueued(() => { + void this.flushUnregisterOutbox() + }) + // Deletes queued while the gateway was unreachable — including across restarts. + void this.flushUnregisterOutbox() + } + + stop(): void { + this.stopped = true + this.dispatcher.stop() + this.unsubscribe?.() + this.unsubscribe = null + this.runtimeRpc.setOnPushUnregisterQueued(null) + this.runtime.setMobilePushRegistrar(null) + } + + async register(input: MobilePushRegisterInput): Promise { + if (this.registry.getDevice(input.deviceId)?.scope !== 'mobile') { + return { registered: false, reason: 'not_mobile' } + } + // Unregister needs no bucket: with nothing registered it is a lookup, and + // with something registered it can only run once per successful register. + if (!this.registerThrottle.allow(input.deviceId)) { + return { registered: false, reason: 'throttled' } + } + return runKeyedSerializedOperation(this.deviceOperations, input.deviceId, () => + this.registerAfterCleanup(input) + ) + } + + private async registerAfterCleanup( + input: MobilePushRegisterInput + ): Promise { + if (this.outbox.isUnreadable()) { + return { registered: false, reason: 'registration_storage_failed' } + } + // A stable gateway ID must not inherit a delete from an earlier registration. + for (const item of this.outbox.pending().filter((entry) => entry.deviceId === input.deviceId)) { + if (!(await this.deleteQueued(item.reqId, item.registrationId))) { + this.scheduleFlushRetry() + return { registered: false, reason: 'gateway_unreachable' } + } + } + if (this.registry.getDevice(input.deviceId)?.scope !== 'mobile') { + return { registered: false, reason: 'not_mobile' } + } + if (this.stopped) { + return { registered: false, reason: 'gateway_unreachable' } + } + const result = await this.client.registerDevice(input) + if (!result.ok) { + return { + registered: false, + reason: result.reason === 'unreachable' ? 'gateway_unreachable' : 'gateway_rejected' + } + } + const failure = this.storeRegistration(input, result.registrationId) + if (failure) { + // Why: the gateway now holds a token this host will never push to. Queue its + // delete instead of leaking it until the phone happens to register again. + this.outbox.enqueue({ registrationId: result.registrationId, deviceId: input.deviceId }) + } + void this.flushUnregisterOutbox() + return failure + ? { registered: false, reason: failure } + : { registered: true, registrationId: result.registrationId } + } + + async unregister(deviceId: string): Promise<{ unregistered: boolean }> { + return runKeyedSerializedOperation(this.deviceOperations, deviceId, async () => + this.unregisterCurrent(deviceId) + ) + } + + private unregisterCurrent(deviceId: string): { unregistered: boolean } { + const registrationId = this.registry.getDevice(deviceId)?.pushRegistration?.registrationId + if (!registrationId) { + return { unregistered: false } + } + // Persist cleanup before forgetting its ID; neither write waits on the gateway. + this.outbox.enqueue({ registrationId, deviceId }) + try { + this.registry.setPushRegistration(deviceId, null) + } finally { + void this.flushUnregisterOutbox() + } + return { unregistered: true } + } + + /** Joining an in-flight drain still waits for the item this call queued. */ + async flushUnregisterOutbox(): Promise { + if (this.stopped) { + return + } + this.flushRequested = true + this.flushLoop ??= this.runFlushLoop() + await this.flushLoop + } + + private async runFlushLoop(): Promise { + try { + while (this.flushRequested && !this.stopped) { + // Cleared before the pass, so a delete queued mid-drain earns another one. + this.flushRequested = false + if (await this.drainPending()) { + this.scheduleFlushRetry() + } else { + this.retryDelayMs = OUTBOX_RETRY_BASE_MS + } + } + } finally { + // Clear ownership before the runner settles, so a late request starts a new drain. + this.flushLoop = null + } + } + + /** Returns the refusal reason when a gateway-accepted registration cannot be stored. */ + private storeRegistration( + input: MobilePushRegisterInput, + registrationId: string + ): RegisterStorageFailure | null { + try { + const stored = this.registry.setPushRegistration(input.deviceId, { + registrationId, + filter: input.filter, + expiresAt: Date.now() + 7 * 24 * 60 * 60_000 + }) + // False means the device was removed or left mobile scope while the gateway + // call was in flight. + return stored ? null : 'not_mobile' + } catch (error) { + console.warn('[push] Failed to persist a push registration:', error) + return 'registration_storage_failed' + } + } + + /** Returns true when the pass left behind an item the gateway may still accept. */ + private async drainPending(): Promise { + let retryable = false + // Every enqueue requests a flush; the outer loop owns work added during this pass. + for (const item of this.outbox.pending()) { + try { + const deleted = await runKeyedSerializedOperation( + this.deviceOperations, + item.deviceId, + () => { + // Failed local removal must not delete a still-attached gateway registration. + if ( + this.registry.getDevice(item.deviceId)?.pushRegistration?.registrationId === + item.registrationId + ) { + return Promise.resolve(false) + } + return this.deleteQueued(item.reqId, item.registrationId) + } + ) + if (!deleted) { + retryable = true + } + } catch (error) { + // One bad delete must not strand the rest of the queue. + console.warn('[push] Failed to drain the push unregister outbox:', error) + retryable = true + } + } + return retryable + } + + private async deleteQueued(reqId: string, registrationId: string): Promise { + if (!this.outbox.pending().some((item) => item.reqId === reqId)) { + return true + } + const deleted = await this.client.deleteDevice(registrationId) + if (!deleted) { + return false + } + this.outbox.remove(reqId) + return true + } + + private scheduleFlushRetry(): void { + if (this.retryArmed || this.stopped) { + return + } + this.retryArmed = true + const delayMs = this.retryDelayMs + this.retryDelayMs = Math.min(delayMs * 2, OUTBOX_RETRY_MAX_MS) + this.scheduleRetry(() => { + this.retryArmed = false + void this.flushUnregisterOutbox() + }, delayMs) + } +} diff --git a/src/main/runtime/push/push-agent-state.test.ts b/src/main/runtime/push/push-agent-state.test.ts new file mode 100644 index 00000000000..e56d39ffb01 --- /dev/null +++ b/src/main/runtime/push/push-agent-state.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it } from 'vitest' +import { mapPushAgentState } from './push-dispatcher' + +describe('mapPushAgentState', () => { + it.each([ + ['blocked', 'needs-input'], + ['waiting', 'needs-input'], + ['done', 'finished'], + [undefined, 'finished'] + ] as const)('maps agent-task-complete %s to %s', (agentState, expected) => { + expect(mapPushAgentState('agent-task-complete', agentState)).toBe(expected) + }) + + it('suppresses a still-working agent', () => { + expect(mapPushAgentState('agent-task-complete', 'working')).toBeUndefined() + }) + + it('leaves non-agent sources without a state', () => { + expect(mapPushAgentState('terminal-bell', undefined)).toBeNull() + }) +}) diff --git a/src/main/runtime/push/push-cleanup-auth-expiry.test.ts b/src/main/runtime/push/push-cleanup-auth-expiry.test.ts new file mode 100644 index 00000000000..dc7ce5e1883 --- /dev/null +++ b/src/main/runtime/push/push-cleanup-auth-expiry.test.ts @@ -0,0 +1,41 @@ +import { createHash } from 'node:crypto' +import { expect, it } from 'vitest' +import { PushGatewayClient } from './push-gateway-client' +import { buildPushChallengeFixture, createPushHostKeypair } from './push-host-challenge-fixtures' + +it('retains a delete when its session proof expires before the DELETE is attempted', async () => { + const keypair = createPushHostKeypair() + const hostFingerprint = createHash('sha256') + .update(keypair.publicKey) + .digest('base64url') + .slice(0, 16) + let now = 1_770_000_000_000 + let deletes = 0 + const client = new PushGatewayClient({ + gatewayUrl: 'https://push.example.test', + keypair, + now: () => now, + fetch: (async (url, init) => { + if (String(url).endsWith('/challenge')) { + const fixture = buildPushChallengeFixture({ + hostKeypair: keypair, + hostFingerprint, + gatewayOrigin: 'https://push.example.test', + issuedAt: now, + challengeId: 'challenge-1' + }) + now += 11_000 + return Response.json(fixture.challenge) + } + if (String(url).endsWith('/session')) { + return Response.json({ error: 'invalid_proof' }, { status: 401 }) + } + if (init?.method === 'DELETE') { + deletes++ + } + return new Response(null, { status: 204 }) + }) as typeof fetch + }) + expect(await client.deleteDevice('registration-1')).toEqual(false) + expect(deletes).toBe(0) +}) diff --git a/src/main/runtime/push/push-delivery-policy.test.ts b/src/main/runtime/push/push-delivery-policy.test.ts new file mode 100644 index 00000000000..4451a9b9d85 --- /dev/null +++ b/src/main/runtime/push/push-delivery-policy.test.ts @@ -0,0 +1,48 @@ +import { afterEach, expect, it, vi } from 'vitest' +import { createHarness, flush, notification, registration } from './push-dispatcher.test-fixture' +import { parseMobilePushRegistration } from '../../../shared/mobile-push-contract' + +afterEach(() => vi.useRealTimers()) + +it('does not send or consume cooldown while the desktop is active', async () => { + const reg = registration() + reg.filter = { ...reg.filter, onlyWhenDesktopAway: true } + const { dispatcher, sends } = createHarness({ + devices: [{ deviceId: 'phone', pushRegistration: reg }] + }) + dispatcher.enqueue(notification({ desktopAway: false, emittedAt: 10_000 })) + dispatcher.enqueue(notification({ desktopAway: true, emittedAt: 10_001 })) + await flush() + expect(sends).toHaveLength(1) +}) + +it('expires per phone at the boundary, preserves leases across persistence, and permits renewal', async () => { + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(50_000) + const expired = parseMobilePushRegistration(registration({ expiresAt: 50_000 }))! + const devices = [{ deviceId: 'phone', pushRegistration: expired }] + const { dispatcher, sends } = createHarness({ devices }) + dispatcher.enqueue(notification()) + await flush() + expect(sends).toHaveLength(0) + devices[0].pushRegistration = registration({ expiresAt: 50_001 }) + dispatcher.enqueue(notification()) + await flush() + expect(sends).toHaveLength(1) +}) + +it('rechecks expiry before a retry', async () => { + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(10) + const { dispatcher, sends, runRetry } = createHarness({ + devices: [{ deviceId: 'phone', pushRegistration: registration({ expiresAt: 20 }) }], + sendImpl: async () => ({ ok: false, reason: 'unreachable' }) as never + }) + dispatcher.enqueue(notification()) + await flush() + vi.setSystemTime(20) + runRetry() + await flush() + expect(sends).toHaveLength(1) + expect(parseMobilePushRegistration({ ...registration(), expiresAt: undefined })).toBeUndefined() +}) diff --git a/src/main/runtime/push/push-device-registration-persistence.test.ts b/src/main/runtime/push/push-device-registration-persistence.test.ts new file mode 100644 index 00000000000..1e2d3d6e51c --- /dev/null +++ b/src/main/runtime/push/push-device-registration-persistence.test.ts @@ -0,0 +1,123 @@ +import { mkdtempSync, readFileSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { DeviceRegistry } from '../device-registry' +import { DEVICE_REGISTRY_FILENAME } from '../mobile-pairing-files' +import type { MobilePushRegistration } from '../../../shared/mobile-push-contract' + +const REGISTRATION: MobilePushRegistration = { + registrationId: 'reg-1', + filter: {}, + expiresAt: Date.now() + 7 * 86400_000 +} + +function userDataDir(): string { + return mkdtempSync(join(tmpdir(), 'orca-push-registry-')) +} + +function rewriteRegistry(dir: string, mutate: (devices: Record[]) => void): void { + const path = join(dir, DEVICE_REGISTRY_FILENAME) + const devices: Record[] = JSON.parse(readFileSync(path, 'utf-8')) + mutate(devices) + writeFileSync(path, JSON.stringify(devices)) +} + +describe('DeviceRegistry push registrations', () => { + it('persists a registration across a restart', () => { + const dir = userDataDir() + const device = new DeviceRegistry(dir).addDevice('phone', 'mobile') + expect(new DeviceRegistry(dir).setPushRegistration(device.deviceId, REGISTRATION)).toBe(true) + + expect(new DeviceRegistry(dir).getDevice(device.deviceId)?.pushRegistration).toEqual( + REGISTRATION + ) + }) + + it('clears a registration when the gateway reports the token dead', () => { + const dir = userDataDir() + const registry = new DeviceRegistry(dir) + const device = registry.addDevice('phone', 'mobile') + registry.setPushRegistration(device.deviceId, REGISTRATION) + + expect(registry.setPushRegistration(device.deviceId, null)).toBe(true) + expect(new DeviceRegistry(dir).getDevice(device.deviceId)?.pushRegistration).toBeUndefined() + }) + + it.each([1_770_000_000_000, 'unused'])( + 'ignores the obsolete registeredAt field (%s)', + (registeredAt) => { + const dir = userDataDir() + const device = new DeviceRegistry(dir).addDevice('phone', 'mobile') + rewriteRegistry(dir, (devices) => { + for (const entry of devices) { + entry.pushRegistration = { ...REGISTRATION, registeredAt } + } + }) + + expect(new DeviceRegistry(dir).getDevice(device.deviceId)?.pushRegistration).toEqual( + REGISTRATION + ) + } + ) + + it('refuses to register a runtime-scoped device', () => { + const dir = userDataDir() + const registry = new DeviceRegistry(dir) + const cli = registry.addDevice('cli', 'runtime') + + expect(registry.setPushRegistration(cli.deviceId, REGISTRATION)).toBe(false) + }) + + it('loads a registry written before push existed', () => { + const dir = userDataDir() + const device = new DeviceRegistry(dir).addDevice('phone', 'mobile') + rewriteRegistry(dir, (devices) => { + for (const entry of devices) { + delete entry.pushRegistration + } + }) + + const reloaded = new DeviceRegistry(dir) + expect(reloaded.listDevices()).toHaveLength(1) + expect(reloaded.getDevice(device.deviceId)?.pushRegistration).toBeUndefined() + }) + + it.each([ + ['a malformed registration', { registrationId: 'reg-1' }], + ['a missing expiry', { ...REGISTRATION, expiresAt: undefined }], + ['a non-finite expiry', { ...REGISTRATION, expiresAt: Infinity }], + ['an array filter', { ...REGISTRATION, filter: [] }], + ['a missing filter', { ...REGISTRATION, filter: undefined }], + ['a non-object', 'nonsense'] + ])('keeps the device but drops %s', (_name, pushRegistration) => { + const dir = userDataDir() + const device = new DeviceRegistry(dir).addDevice('phone', 'mobile') + rewriteRegistry(dir, (devices) => { + for (const entry of devices) { + entry.pushRegistration = pushRegistration + } + }) + + const reloaded = new DeviceRegistry(dir) + expect(reloaded.listDevices()).toHaveLength(1) + expect(reloaded.getDevice(device.deviceId)?.pushRegistration).toBeUndefined() + }) + + it('drops only the unknown members of a stored filter', () => { + const dir = userDataDir() + const device = new DeviceRegistry(dir).addDevice('phone', 'mobile') + rewriteRegistry(dir, (devices) => { + for (const entry of devices) { + entry.pushRegistration = { + ...REGISTRATION, + filter: { sound: false, unknownSetting: true } + } + } + }) + + expect(new DeviceRegistry(dir).getDevice(device.deviceId)?.pushRegistration?.filter).toEqual({ + sound: false + }) + }) +}) diff --git a/src/main/runtime/push/push-dispatcher.test-fixture.ts b/src/main/runtime/push/push-dispatcher.test-fixture.ts new file mode 100644 index 00000000000..16baada47d8 --- /dev/null +++ b/src/main/runtime/push/push-dispatcher.test-fixture.ts @@ -0,0 +1,88 @@ +import { vi } from 'vitest' +import type { MobilePushRegistration } from '../../../shared/mobile-push-contract' +import type { MobileNotificationEvent } from '../runtime-mobile-notification-controller' +import type { PushGatewayClient, PushSendResult } from './push-gateway-client' +import { PushDispatcher, type PushDispatcherRegistry } from './push-dispatcher' + +export function registration( + overrides: Partial = {} +): MobilePushRegistration { + return { + registrationId: 'reg-1', + filter: {}, + expiresAt: Date.now() + 7 * 86400_000, + ...overrides + } +} + +export type SendCall = Parameters[0] + +export function createHarness(options: { + devices: { deviceId: string; pushRegistration?: MobilePushRegistration }[] + results?: PushSendResult[] + sendImpl?: () => Promise +}): { + dispatcher: PushDispatcher + sends: SendCall[] + cleared: (string | null)[] + runRetry: () => void +} { + const sends: SendCall[] = [] + const cleared: (string | null)[] = [] + let retry: (() => void) | null = null + const client = { + send: vi.fn(async (input: SendCall) => { + sends.push(input) + if (options.sendImpl) { + return await options.sendImpl() + } + return { + ok: true as const, + results: + options.results ?? + input.registrationIds.map((registrationId) => ({ + registrationId, + status: 'queued' as const + })) + } + }) + } as unknown as PushGatewayClient + const registry: PushDispatcherRegistry = { + listDevices: () => options.devices, + setPushRegistration: (deviceId, value) => { + cleared.push(value === null ? deviceId : null) + return true + } + } + return { + dispatcher: new PushDispatcher({ + client, + registry, + scheduleRetry: (run) => { + retry = run + } + }), + sends, + cleared, + runRetry: () => retry?.() + } +} + +export function notification( + overrides: Partial = {} +): MobileNotificationEvent { + return { + type: 'notification', + source: 'agent-task-complete', + title: 'feat/x - Claude finished', + body: 'All done.', + worktreeId: 'repo::wt1', + notificationId: 'agent:one', + notificationSeq: 7, + notificationEpoch: 'epoch-1', + agentState: 'done', + ...overrides + } as MobileNotificationEvent +} + +export const flush = (): Promise => new Promise((resolve) => setImmediate(resolve)) diff --git a/src/main/runtime/push/push-dispatcher.test.ts b/src/main/runtime/push/push-dispatcher.test.ts new file mode 100644 index 00000000000..47373045f28 --- /dev/null +++ b/src/main/runtime/push/push-dispatcher.test.ts @@ -0,0 +1,188 @@ +import { describe, expect, it, vi } from 'vitest' +import type { PushGatewayClient } from './push-gateway-client' +import { PushDispatcher } from './push-dispatcher' +import { + createHarness, + flush, + notification, + registration, + type SendCall +} from './push-dispatcher.test-fixture' + +describe('PushDispatcher', () => { + it('batches every matching registration into one send', async () => { + const harness = createHarness({ + devices: [ + { deviceId: 'a', pushRegistration: registration({ registrationId: 'reg-a' }) }, + { deviceId: 'b', pushRegistration: registration({ registrationId: 'reg-b' }) }, + { deviceId: 'c' } + ] + }) + + harness.dispatcher.enqueue(notification()) + await flush() + + expect(harness.sends).toHaveLength(1) + expect(harness.sends[0]?.registrationIds).toEqual(['reg-a', 'reg-b']) + expect(harness.sends[0]?.notification).toMatchObject({ + source: 'agent-task-complete', + agentState: 'finished', + notificationSeq: 7, + notificationEpoch: 'epoch-1', + worktreeId: 'repo::wt1' + }) + }) + + it('fans out past the per-request cap instead of starving the extra devices', async () => { + const devices = Array.from({ length: 25 }, (_, index) => ({ + deviceId: `device-${index}`, + pushRegistration: registration({ registrationId: `reg-${index}` }) + })) + const harness = createHarness({ devices }) + + harness.dispatcher.enqueue(notification()) + await flush() + + expect(harness.sends).toHaveLength(2) + expect(harness.sends[0]?.registrationIds).toHaveLength(20) + expect(harness.sends[1]?.registrationIds).toEqual([ + 'reg-20', + 'reg-21', + 'reg-22', + 'reg-23', + 'reg-24' + ]) + }) + + it('drops a dead registration reported by a later chunk', async () => { + const devices = Array.from({ length: 25 }, (_, index) => ({ + deviceId: `device-${index}`, + pushRegistration: registration({ registrationId: `reg-${index}` }) + })) + const harness = createHarness({ + devices, + results: [{ registrationId: 'reg-24', status: 'dead' }] + }) + + harness.dispatcher.enqueue(notification()) + await flush() + + expect(harness.cleared).toEqual(['device-24']) + }) + + it('pushes a silent dismissal with an absolute expiry', async () => { + const harness = createHarness({ + devices: [{ deviceId: 'a', pushRegistration: registration() }] + }) + + harness.dispatcher.enqueue({ + type: 'dismiss', + notificationId: 'agent:one', + notificationSeq: 8, + notificationEpoch: 'epoch-1' + }) + await flush() + + expect(harness.sends).toHaveLength(1) + expect(harness.sends[0]?.notification).toMatchObject({ + kind: 'dismiss', + sound: false, + notificationId: 'agent:one', + expiresAt: expect.any(Number) + }) + }) + + it('stays silent while the agent is still working', async () => { + const harness = createHarness({ + devices: [{ deviceId: 'a', pushRegistration: registration() }] + }) + + harness.dispatcher.enqueue(notification({ agentState: 'working' })) + await flush() + + expect(harness.sends).toHaveLength(0) + }) + + it('drops a registration the gateway reports dead', async () => { + const harness = createHarness({ + devices: [ + { deviceId: 'a', pushRegistration: registration({ registrationId: 'reg-a' }) }, + { deviceId: 'b', pushRegistration: registration({ registrationId: 'reg-b' }) } + ], + results: [ + { registrationId: 'reg-a', status: 'dead' }, + { registrationId: 'reg-b', status: 'queued' } + ] + }) + + harness.dispatcher.enqueue(notification()) + await flush() + + expect(harness.cleared).toEqual(['a']) + }) + + it('retries once when the gateway is unreachable', async () => { + const sends: SendCall[] = [] + const client = { + send: vi.fn(async (input: SendCall) => { + sends.push(input) + return { ok: false as const, reason: 'unreachable' as const } + }) + } as unknown as PushGatewayClient + const scheduled: (() => void)[] = [] + const devices = [{ deviceId: 'a', pushRegistration: registration() }] + const dispatcher = new PushDispatcher({ + client, + registry: { + listDevices: () => devices, + setPushRegistration: () => true + }, + scheduleRetry: (run, delayMs) => { + expect(delayMs).toBe(2_000) + scheduled.push(run) + } + }) + + dispatcher.enqueue(notification()) + await flush() + expect(sends).toHaveLength(1) + expect(scheduled).toHaveLength(1) + + scheduled[0]?.() + await flush() + expect(sends).toHaveLength(2) + // The second attempt is the last one; a further retry is never scheduled. + expect(scheduled).toHaveLength(1) + }) + + it('never throws into the caller when the client rejects', async () => { + const harness = createHarness({ + devices: [{ deviceId: 'a', pushRegistration: registration() }], + sendImpl: async () => { + throw new Error('boom') + } + }) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + expect(() => harness.dispatcher.enqueue(notification())).not.toThrow() + await flush() + expect(warn).toHaveBeenCalled() + warn.mockRestore() + }) + + it('never throws when the registry itself fails', async () => { + const dispatcher = new PushDispatcher({ + client: { send: vi.fn() } as unknown as PushGatewayClient, + registry: { + listDevices: () => { + throw new Error('registry unavailable') + }, + setPushRegistration: () => true + } + }) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + expect(() => dispatcher.enqueue(notification())).not.toThrow() + warn.mockRestore() + }) +}) diff --git a/src/main/runtime/push/push-dispatcher.ts b/src/main/runtime/push/push-dispatcher.ts new file mode 100644 index 00000000000..e028b80d858 --- /dev/null +++ b/src/main/runtime/push/push-dispatcher.ts @@ -0,0 +1,271 @@ +import { reserveNotificationCooldown } from '../../../shared/notification-burst-cooldown' +// Why: the out-of-band leg of the mobile notification fan-out. Every event that +// already went to connected sockets is offered to the push gateway so a phone +// with Orca closed still hears about it. Fire-and-forget by construction: the +// socket fan-out must never wait on, or fail because of, a push. +import { + MOBILE_PUSH_SOURCES, + type MobilePushAgentState, + type MobilePushRegistration +} from '../../../shared/mobile-push-contract' +import type { MobileNotificationEvent } from '../runtime-mobile-notification-controller' +import type { PushGatewayClient, PushSendNotification } from './push-gateway-client' +import { PushOutcomeCounters } from './push-outcome-counters' + +const PUSH_RETRY_DELAY_MS = 2_000 +// The gateway rejects a whole request above this, so a host with more paired +// phones fans out across several sends rather than starving the extras. +const MAX_REGISTRATIONS_PER_SEND = 20 +const PUSH_TITLE_MAX_LENGTH = 80 +const PUSH_BODY_MAX_LENGTH = 180 + +export type PushDispatcherRegistry = { + listDevices(): readonly { deviceId: string; pushRegistration?: MobilePushRegistration }[] + setPushRegistration(deviceId: string, registration: MobilePushRegistration | null): boolean +} + +type PushDispatcherOptions = { + client: PushGatewayClient + registry: PushDispatcherRegistry + /** Test seam: lets a suite drive the single retry without real time. */ + scheduleRetry?: (run: () => void, delayMs: number) => void +} + +type PushTarget = { deviceId: string; registration: MobilePushRegistration } + +function clip(value: string, maxLength: number): string { + const normalized = value.replace(/\s+/g, ' ').trim() + return normalized.length <= maxLength ? normalized : `${normalized.slice(0, maxLength - 1)}…` +} + +export function mapPushAgentState( + source: string, + state: string | undefined +): MobilePushAgentState | null | undefined { + if (source !== 'agent-task-complete') { + return null + } + if (state === 'blocked' || state === 'waiting' || state === 'needs-input') { + return 'needs-input' + } + return state === undefined || state === 'done' || state === 'finished' ? 'finished' : undefined +} + +function allowsPushDelivery( + registration: MobilePushRegistration, + event: MobileNotificationEvent +): boolean { + return ( + event.type === 'notification' && + event.desktopAllowed !== false && + (!registration.filter.onlyWhenDesktopAway || event.desktopAway !== false) + ) +} + +export class PushDispatcher { + private readonly recentNotifications = new Map() + private readonly outcomes = new PushOutcomeCounters() + private stopped = false + private readonly client: PushGatewayClient + private readonly registry: PushDispatcherRegistry + private readonly scheduleRetry: (run: () => void, delayMs: number) => void + + constructor(options: PushDispatcherOptions) { + this.client = options.client + this.registry = options.registry + this.scheduleRetry = + options.scheduleRetry ?? + ((run, delayMs) => { + // Why: a pending push retry must never hold the app open at quit. + setTimeout(run, delayMs).unref?.() + }) + } + + start(): void { + this.stopped = false + } + + stop(): void { + this.stopped = true + this.outcomes.flush() + } + + enqueue(event: MobileNotificationEvent): void { + if (this.stopped) { + return + } + try { + const plan = this.planSend(event) + if (!plan) { + return + } + for (const sound of [true, false]) { + const targets = plan.targets.filter( + (target) => (target.registration.filter.sound !== false) === sound + ) + for (let start = 0; start < targets.length; start += MAX_REGISTRATIONS_PER_SEND) { + void this.deliver( + targets.slice(start, start + MAX_REGISTRATIONS_PER_SEND), + { ...plan.notification, ...(!sound ? { sound: false } : {}) }, + 0 + ) + } + } + } catch (error) { + console.warn('[push] Failed to prepare a push notification:', error) + } + } + + private planSend( + event: MobileNotificationEvent + ): { targets: PushTarget[]; notification: PushSendNotification } | null { + if (event.type === 'dismiss') { + if (event.notificationSeq === undefined || !event.notificationEpoch) { + return null + } + const targets = this.registry + .listDevices() + .flatMap(({ deviceId, pushRegistration: registration }) => + registration && registration.expiresAt > Date.now() ? [{ deviceId, registration }] : [] + ) + return { + targets, + notification: { + kind: 'dismiss', + expiresAt: Date.now() + 300_000, + notificationId: event.notificationId, + notificationSeq: event.notificationSeq, + notificationEpoch: event.notificationEpoch, + source: 'agent-task-complete', + agentState: null, + title: 'Orca', + body: '', + sound: false + } + } + } + const source = MOBILE_PUSH_SOURCES.find((candidate) => candidate === event.source) + if (!source || event.notificationSeq === undefined || event.notificationEpoch === undefined) { + return null + } + const agentState = mapPushAgentState(source, event.agentState) + if (agentState === undefined) { + return null + } + const targets = this.registry.listDevices().flatMap((device) => { + const registration = device.pushRegistration + if ( + !registration || + registration.expiresAt <= Date.now() || + !allowsPushDelivery(registration, event) + ) { + return [] + } + if ( + event.emittedAt !== undefined && + !reserveNotificationCooldown( + this.recentNotifications, + JSON.stringify([device.deviceId, event.worktreeId ?? 'global']), + event.emittedAt + ) + ) { + return [] + } + return [{ deviceId: device.deviceId, registration }] + }) + if (targets.length === 0) { + return null + } + return { + targets, + notification: { + expiresAt: Date.now() + 300_000, + ...(event.notificationId ? { notificationId: event.notificationId } : {}), + notificationSeq: event.notificationSeq, + notificationEpoch: event.notificationEpoch, + source, + agentState, + title: clip(event.title, PUSH_TITLE_MAX_LENGTH), + body: clip(event.body, PUSH_BODY_MAX_LENGTH), + ...(event.worktreeId ? { worktreeId: event.worktreeId } : {}) + } + } + } + + private async deliver( + targets: readonly PushTarget[], + notification: PushSendNotification, + attempt: number + ): Promise { + if (this.stopped) { + return + } + const currentTargets = targets.filter((target) => + this.registry + .listDevices() + .some( + (device) => + device.deviceId === target.deviceId && + device.pushRegistration === target.registration && + target.registration.expiresAt > Date.now() + ) + ) + if (!currentTargets.length) { + return + } + try { + const result = await this.client.send({ + registrationIds: currentTargets.map((target) => target.registration.registrationId), + notification + }) + if (this.stopped) { + return + } + if (result.ok) { + for (const entry of result.results) { + if (entry.status === 'error' || entry.status === 'rate_limited') { + this.outcomes.record(entry.status) + } + } + this.dropDeadRegistrations(targets, result.results) + return + } + this.outcomes.record(result.reason) + // Only a transport-level miss is worth repeating; a gateway that refused + // this payload will refuse the identical retry. + if (attempt === 0 && result.reason === 'unreachable') { + this.scheduleRetry(() => { + void this.deliver(targets, notification, attempt + 1) + }, PUSH_RETRY_DELAY_MS) + } + } catch (error) { + console.warn('[push] Push send failed:', error) + } + } + + private dropDeadRegistrations( + targets: readonly PushTarget[], + results: readonly { registrationId: string; status: string }[] + ): void { + for (const result of results) { + if (result.status !== 'dead') { + continue + } + const target = targets.find( + (entry) => entry.registration.registrationId === result.registrationId + ) + if ( + !target || + this.registry.listDevices().find((device) => device.deviceId === target.deviceId) + ?.pushRegistration !== target.registration + ) { + continue + } + try { + this.registry.setPushRegistration(target.deviceId, null) + } catch (error) { + console.warn('[push] Failed to drop a dead push registration:', error) + } + } + } +} diff --git a/src/main/runtime/push/push-gateway-client.test.ts b/src/main/runtime/push/push-gateway-client.test.ts new file mode 100644 index 00000000000..0707f686fae --- /dev/null +++ b/src/main/runtime/push/push-gateway-client.test.ts @@ -0,0 +1,256 @@ +import { describe, expect, it, vi } from 'vitest' +import { createHash } from 'node:crypto' +import { buildPushChallengeFixture, createPushHostKeypair } from './push-host-challenge-fixtures' +import { PushGatewayClient } from './push-gateway-client' + +const GATEWAY_URL = 'https://push.onorca.dev' +const NOW = 1_770_000_000_000 + +type Recorded = { + url: string + method: string + authorization: string | null + body: unknown + redirect: RequestRedirect | undefined +} + +function fingerprintOf(publicKey: Uint8Array): string { + return createHash('sha256').update(publicKey).digest('base64url').slice(0, 16) +} + +function jsonResponse(status: number, body: unknown): Response { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' } + }) +} + +function createFakeGateway( + options: { sessionTtlMs?: number; devicesStatus?: number; rejectBearer?: boolean } = {} +): { + client: PushGatewayClient + calls: Recorded[] + expireSession: () => void + now: { value: number } +} { + const hostKeypair = createPushHostKeypair() + const hostFingerprint = fingerprintOf(hostKeypair.publicKey) + const now = { value: NOW } + const calls: Recorded[] = [] + const liveTokens = new Set() + const knownRegistrations = new Set() + let issued = 0 + let pendingProof: string | null = null + + const fetchImpl = (async (input: string, init?: RequestInit): Promise => { + const url = String(input) + const headers = new Headers(init?.headers) + const body: unknown = init?.body ? JSON.parse(String(init.body)) : undefined + calls.push({ + url, + method: init?.method ?? 'GET', + authorization: headers.get('authorization'), + body, + redirect: init?.redirect + }) + if (url.endsWith('/v1/host/challenge')) { + const built = buildPushChallengeFixture({ + hostKeypair, + gatewayOrigin: GATEWAY_URL, + hostFingerprint, + issuedAt: now.value, + challengeId: `challenge-${++issued}` + }) + pendingProof = built.proof + return jsonResponse(200, built.challenge) + } + if (url.endsWith('/v1/host/session')) { + const params = body as { proofB64: string } + if (params.proofB64 !== pendingProof) { + return jsonResponse(401, { error: 'bad_proof' }) + } + const sessionToken = `session-${issued}` + liveTokens.add(sessionToken) + return jsonResponse(200, { + sessionToken, + expiresAt: now.value + (options.sessionTtlMs ?? 24 * 60 * 60_000), + hostFingerprint + }) + } + const bearer = headers.get('authorization')?.replace('Bearer ', '') ?? '' + if (options.rejectBearer || !liveTokens.has(bearer)) { + return jsonResponse(401, { error: 'session_expired' }) + } + if (url.endsWith('/v1/devices')) { + if (options.devicesStatus) { + return jsonResponse(options.devicesStatus, { error: 'nope' }) + } + knownRegistrations.add('reg-1') + return jsonResponse(200, { registrationId: 'reg-1' }) + } + if (url.endsWith('/v1/send')) { + return jsonResponse(200, { results: [{ registrationId: 'reg-1', status: 'queued' }] }) + } + // Why explicit: a catch-all 204 would report every delete as accepted and + // leave the 404 branch of deleteDevice untested. + const deleted = /\/v1\/devices\/([^/]+)$/.exec(url) + if (deleted && init?.method === 'DELETE') { + const registrationId = decodeURIComponent(deleted[1] ?? '') + return new Response(null, { status: knownRegistrations.has(registrationId) ? 204 : 404 }) + } + throw new Error(`unexpected request: ${init?.method ?? 'GET'} ${url}`) + }) as unknown as typeof globalThis.fetch + + return { + client: new PushGatewayClient({ + gatewayUrl: GATEWAY_URL, + keypair: hostKeypair, + fetch: fetchImpl, + now: () => now.value + }), + calls, + expireSession: () => liveTokens.clear(), + now + } +} + +const REGISTER_INPUT = { + deviceId: 'device-1', + platform: 'ios' as const, + token: 'a'.repeat(64), + apnsEnvironment: 'sandbox' as const +} + +describe('PushGatewayClient', () => { + it('runs the challenge handshake once and reuses the cached session', async () => { + const gateway = createFakeGateway() + + expect(await gateway.client.registerDevice(REGISTER_INPUT)).toEqual({ + ok: true, + registrationId: 'reg-1' + }) + expect( + await gateway.client.send({ + registrationIds: ['reg-1'], + notification: { + notificationSeq: 1, + notificationEpoch: 'epoch-1', + source: 'agent-task-complete', + agentState: 'finished', + title: 'Done', + body: 'Body' + } + }) + ).toEqual({ ok: true, results: [{ registrationId: 'reg-1', status: 'queued' }] }) + + const handshakes = gateway.calls.filter((call) => call.url.includes('/v1/host/')) + expect(handshakes).toHaveLength(2) + expect(gateway.calls.at(-1)?.authorization).toBe('Bearer session-1') + }) + + it('re-authenticates once when the gateway rejects the cached session', async () => { + const gateway = createFakeGateway() + await gateway.client.registerDevice(REGISTER_INPUT) + gateway.expireSession() + + expect(await gateway.client.registerDevice(REGISTER_INPUT)).toEqual({ + ok: true, + registrationId: 'reg-1' + }) + expect(gateway.calls.filter((call) => call.url.endsWith('/v1/host/challenge'))).toHaveLength(2) + expect(gateway.calls.at(-1)?.authorization).toBe('Bearer session-2') + }) + + it('re-authenticates before a session that is about to expire', async () => { + const gateway = createFakeGateway({ sessionTtlMs: 90_000 }) + await gateway.client.registerDevice(REGISTER_INPUT) + gateway.now.value += 60_000 + + await gateway.client.registerDevice(REGISTER_INPUT) + expect(gateway.calls.filter((call) => call.url.endsWith('/v1/host/challenge'))).toHaveLength(2) + }) + + it('shares one handshake across concurrent calls', async () => { + const gateway = createFakeGateway() + await Promise.all([ + gateway.client.registerDevice(REGISTER_INPUT), + gateway.client.registerDevice(REGISTER_INPUT) + ]) + expect(gateway.calls.filter((call) => call.url.endsWith('/v1/host/challenge'))).toHaveLength(1) + }) + + it('reports an unreachable gateway instead of throwing', async () => { + const keypair = createPushHostKeypair() + const client = new PushGatewayClient({ + gatewayUrl: GATEWAY_URL, + keypair, + fetch: vi.fn(async () => { + throw new Error('network down') + }) as unknown as typeof globalThis.fetch, + now: () => NOW + }) + expect(await client.registerDevice(REGISTER_INPUT)).toEqual({ + ok: false, + reason: 'unreachable' + }) + }) + + it('reports a refused registration as rejected', async () => { + const gateway = createFakeGateway({ devicesStatus: 400 }) + expect(await gateway.client.registerDevice(REGISTER_INPUT)).toEqual({ + ok: false, + reason: 'rejected' + }) + }) + + it('never follows a redirect, on the handshake or on an authorized call', async () => { + const gateway = createFakeGateway() + + await gateway.client.registerDevice(REGISTER_INPUT) + await gateway.client.deleteDevice('reg-1') + + // A 307 would replay the host proof, then the phone's token, to whatever + // origin the redirect named. + expect(gateway.calls.length).toBeGreaterThanOrEqual(4) + expect(gateway.calls.every((call) => call.redirect === 'error')).toBe(true) + }) + + it('reports a gateway 5xx as unreachable so the caller can retry', async () => { + const gateway = createFakeGateway({ devicesStatus: 503 }) + expect(await gateway.client.registerDevice(REGISTER_INPUT)).toEqual({ + ok: false, + reason: 'unreachable' + }) + }) + + it('treats a delete the gateway accepted as done', async () => { + const gateway = createFakeGateway() + await gateway.client.registerDevice(REGISTER_INPUT) + + expect(await gateway.client.deleteDevice('reg-1')).toEqual(true) + expect(gateway.calls.at(-1)).toMatchObject({ method: 'DELETE' }) + }) + + it('treats a delete of an unknown registration as done', async () => { + const gateway = createFakeGateway() + + expect(await gateway.client.deleteDevice('reg-gone')).toEqual(true) + }) + + it('reports a 401 that survives the forced re-auth as unreachable', async () => { + const gateway = createFakeGateway({ rejectBearer: true }) + + expect(await gateway.client.registerDevice(REGISTER_INPUT)).toEqual({ + ok: false, + reason: 'unreachable' + }) + // Exactly one forced re-auth, not a handshake loop. + expect(gateway.calls.filter((call) => call.url.endsWith('/v1/host/challenge'))).toHaveLength(2) + }) + + it('keeps an unreachable-classified 401 retryable for a queued delete', async () => { + const gateway = createFakeGateway({ rejectBearer: true }) + + expect(await gateway.client.deleteDevice('reg-1')).toEqual(false) + }) +}) diff --git a/src/main/runtime/push/push-gateway-client.ts b/src/main/runtime/push/push-gateway-client.ts new file mode 100644 index 00000000000..05fb9a1ffef --- /dev/null +++ b/src/main/runtime/push/push-gateway-client.ts @@ -0,0 +1,172 @@ +// Why: talks to the Orca push gateway (cloud/packages/push-contract/src). +// Every method returns a result instead of throwing — push is best-effort and +// must never break the socket fan-out it rides along with. +import { z } from 'zod' +import { cancelUnreadResponseBody } from '../../lib/unread-response-body' +import type { E2EEKeypair } from '../e2ee-keypair' +import type { + MobilePushAgentState, + MobilePushApnsEnvironment, + MobilePushPlatform, + MobilePushSource +} from '../../../shared/mobile-push-contract' +import { + PUSH_REQUEST_DEADLINE_MS, + readPushGatewayJson, + type PushGatewayFailure, + type PushGatewayResponse, + type PushGatewayResult +} from './push-gateway-response' +import { PushGatewaySession } from './push-gateway-session' + +export type { PushGatewayFailure, PushGatewayResult } + +const RegisterResponseSchema = z.object({ registrationId: z.string().min(1).max(512) }) + +const SendResponseSchema = z.object({ + results: z + .array( + z.object({ + registrationId: z.string().min(1).max(512), + status: z.enum(['queued', 'dead', 'rate_limited', 'error']) + }) + ) + .max(64) +}) + +export type PushSendResult = z.infer['results'][number] + +export type PushSendNotification = { + kind?: 'alert' | 'dismiss' + expiresAt?: number + sound?: boolean + notificationId?: string + notificationSeq: number + notificationEpoch: string + source: MobilePushSource + agentState: MobilePushAgentState | null + title: string + body: string + worktreeId?: string +} + +type PushGatewayClientOptions = { + gatewayUrl: string + keypair: E2EEKeypair + fetch?: typeof globalThis.fetch + now?: () => number +} + +type AuthorizedResponse = { ok: true; response: Response; token: string } | PushGatewayFailure + +export class PushGatewayClient { + private readonly origin: string + private readonly fetchImpl: typeof globalThis.fetch + private readonly session: PushGatewaySession + + constructor(options: PushGatewayClientOptions) { + this.origin = new URL(options.gatewayUrl).origin + this.fetchImpl = options.fetch ?? globalThis.fetch + this.session = new PushGatewaySession({ + origin: this.origin, + keypair: options.keypair, + fetchImpl: this.fetchImpl, + now: options.now ?? Date.now + }) + } + + async registerDevice(input: { + deviceId: string + platform: MobilePushPlatform + token: string + apnsEnvironment?: MobilePushApnsEnvironment + }): Promise> { + const response = await this.authorized('/v1/devices', { + method: 'POST', + body: { + v: 1, + deviceId: input.deviceId, + platform: input.platform, + token: input.token, + ...(input.apnsEnvironment ? { apnsEnvironment: input.apnsEnvironment } : {}) + } + }) + const parsed = await readPushGatewayJson(response, RegisterResponseSchema) + return parsed.ok ? { ok: true, registrationId: parsed.value.registrationId } : parsed + } + + async deleteDevice(registrationId: string): Promise { + const response = await this.authorized(`/v1/devices/${encodeURIComponent(registrationId)}`, { + method: 'DELETE' + }) + if (!response.ok) { + return false + } + await cancelUnreadResponseBody(response.response) + // A gateway that no longer knows the registration is as deleted as it gets. + return response.response.ok || response.response.status === 404 + } + + async send(input: { + registrationIds: readonly string[] + notification: PushSendNotification + }): Promise> { + const response = await this.authorized('/v1/send', { + method: 'POST', + body: { + v: 1, + registrationIds: [...input.registrationIds], + notification: input.notification + } + }) + const parsed = await readPushGatewayJson(response, SendResponseSchema) + return parsed.ok ? { ok: true, results: parsed.value.results } : parsed + } + + private async authorized( + path: string, + init: { method: string; body?: unknown } + ): Promise { + const first = await this.sendAuthorized(path, init, null) + if (!first.ok || first.response.status !== 401) { + return first + } + // A 401 means that one session died server-side; one forced re-auth, then stop. + await cancelUnreadResponseBody(first.response) + const retried = await this.sendAuthorized(path, init, first.token) + if (retried.ok && retried.response.status === 401) { + await cancelUnreadResponseBody(retried.response) + // A 401 that survives a freshly minted session is the gateway being unusable + // right now, not this request being wrong: register should report it as + // unreachable, and send should still spend its one retry. + return { ok: false, reason: 'unreachable' } + } + return retried + } + + private async sendAuthorized( + path: string, + init: { method: string; body?: unknown }, + staleToken: string | null + ): Promise { + const outcome = await this.session.ensure(staleToken) + if (!outcome.ok) { + return outcome + } + try { + const response = await this.fetchImpl(`${this.origin}${path}`, { + method: init.method, + headers: { + authorization: `Bearer ${outcome.session.token}`, + ...(init.body === undefined ? {} : { 'content-type': 'application/json' }) + }, + redirect: 'error', + signal: AbortSignal.timeout(PUSH_REQUEST_DEADLINE_MS), + ...(init.body === undefined ? {} : { body: JSON.stringify(init.body) }) + }) + return { ok: true, response, token: outcome.session.token } + } catch { + return { ok: false, reason: 'unreachable' } + } + } +} diff --git a/src/main/runtime/push/push-gateway-origin.ts b/src/main/runtime/push/push-gateway-origin.ts new file mode 100644 index 00000000000..f4f78f2ed38 --- /dev/null +++ b/src/main/runtime/push/push-gateway-origin.ts @@ -0,0 +1,5 @@ +import { cleanCloudServiceOrigin } from '../../../shared/cloud-service-url' + +export function resolvePushGatewayOrigin(env: NodeJS.ProcessEnv, packaged: boolean): string { + return cleanCloudServiceOrigin(env.ORCA_PUSH_GATEWAY_URL, !packaged) ?? 'https://push.onorca.dev' +} diff --git a/src/main/runtime/push/push-gateway-response.ts b/src/main/runtime/push/push-gateway-response.ts new file mode 100644 index 00000000000..12a901b2943 --- /dev/null +++ b/src/main/runtime/push/push-gateway-response.ts @@ -0,0 +1,61 @@ +// Why: the authorized request path and the handshake that authorizes it must +// classify a gateway response identically — otherwise the same 503 means "retry" +// on one leg and "give up" on the other, and register/send disagree about why. +import type { z } from 'zod' +import { cancelUnreadResponseBody } from '../../lib/unread-response-body' + +export const PUSH_REQUEST_DEADLINE_MS = 15_000 + +export type PushGatewayFailure = { ok: false; reason: 'unreachable' | 'rejected' } +export type PushGatewayResult = ({ ok: true } & T) | PushGatewayFailure +export type PushGatewayResponse = { ok: true; response: Response } | PushGatewayFailure + +/** Unauthenticated POST; the handshake legs run before any session exists. */ +export async function postPushGatewayJson( + fetchImpl: typeof globalThis.fetch, + url: string, + body: unknown +): Promise { + try { + const response = await fetchImpl(url, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + // A 307 would replay the proof, and later the phone's token, to whatever + // origin the redirect named. + redirect: 'error', + signal: AbortSignal.timeout(PUSH_REQUEST_DEADLINE_MS), + body: JSON.stringify(body) + }) + return { ok: true, response } + } catch { + return { ok: false, reason: 'unreachable' } + } +} + +export async function readPushGatewayJson( + result: PushGatewayResponse, + schema: TSchema +): Promise<{ ok: true; value: z.infer } | PushGatewayFailure> { + if (!result.ok) { + return result + } + const { response } = result + if (!response.ok) { + await cancelUnreadResponseBody(response) + // 5xx and 429 are worth another attempt later; anything else is the gateway + // refusing this request as written. + return { + ok: false, + reason: response.status >= 500 || response.status === 429 ? 'unreachable' : 'rejected' + } + } + let payload: unknown + try { + payload = await response.json() + } catch { + await cancelUnreadResponseBody(response) + return { ok: false, reason: 'unreachable' } + } + const parsed = schema.safeParse(payload) + return parsed.success ? { ok: true, value: parsed.data } : { ok: false, reason: 'rejected' } +} diff --git a/src/main/runtime/push/push-gateway-session.test.ts b/src/main/runtime/push/push-gateway-session.test.ts new file mode 100644 index 00000000000..8527430365a --- /dev/null +++ b/src/main/runtime/push/push-gateway-session.test.ts @@ -0,0 +1,169 @@ +import { createHash } from 'node:crypto' +import { describe, expect, it, vi } from 'vitest' +import { buildPushChallengeFixture, createPushHostKeypair } from './push-host-challenge-fixtures' +import { PushGatewaySession, type PushSessionOutcome } from './push-gateway-session' + +const GATEWAY_ORIGIN = 'https://push.onorca.dev' +const NOW = 1_770_000_000_000 + +function jsonResponse(status: number, body: unknown): Response { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' } + }) +} + +function tokenOf(outcome: PushSessionOutcome): string | null { + return outcome.ok ? outcome.session.token : null +} + +function createSessionHarness( + options: { sessionStatus?: number; challengeStatus?: number; wrongFingerprint?: boolean } = {} +): { + session: PushGatewaySession + challenges: () => number + requests: () => number + now: { value: number } +} { + const hostKeypair = createPushHostKeypair() + const hostFingerprint = createHash('sha256') + .update(hostKeypair.publicKey) + .digest('base64url') + .slice(0, 16) + const now = { value: NOW } + let issued = 0 + let requests = 0 + let pendingProof: string | null = null + + const fetchImpl = (async (input: string, init?: RequestInit): Promise => { + const url = String(input) + requests += 1 + if (url.endsWith('/v1/host/challenge')) { + if (options.challengeStatus) { + return jsonResponse(options.challengeStatus, { error: 'rate_limited' }) + } + const built = buildPushChallengeFixture({ + hostKeypair, + gatewayOrigin: GATEWAY_ORIGIN, + hostFingerprint, + issuedAt: now.value, + challengeId: `challenge-${++issued}` + }) + pendingProof = built.proof + return jsonResponse(200, built.challenge) + } + if (options.sessionStatus) { + return jsonResponse(options.sessionStatus, { error: 'nope' }) + } + const body = init?.body ? (JSON.parse(String(init.body)) as { proofB64: string }) : null + if (body?.proofB64 !== pendingProof) { + return jsonResponse(401, { error: 'bad_proof' }) + } + return jsonResponse(200, { + sessionToken: `session-${issued}`, + expiresAt: now.value + 24 * 60 * 60_000, + hostFingerprint: options.wrongFingerprint ? 'someone-else' : hostFingerprint + }) + }) as unknown as typeof globalThis.fetch + + return { + session: new PushGatewaySession({ + origin: GATEWAY_ORIGIN, + keypair: hostKeypair, + fetchImpl, + now: () => now.value + }), + challenges: () => issued, + requests: () => requests, + now + } +} + +describe('PushGatewaySession', () => { + it('reuses the cached session until it nears expiry', async () => { + const harness = createSessionHarness() + + expect(tokenOf(await harness.session.ensure(null))).toBe('session-1') + expect(tokenOf(await harness.session.ensure(null))).toBe('session-1') + expect(harness.challenges()).toBe(1) + }) + + it('drops only the exact session that received the 401', async () => { + const harness = createSessionHarness() + expect(tokenOf(await harness.session.ensure(null))).toBe('session-1') + + // A request that 401ed on session-1 forces a fresh handshake. + expect(tokenOf(await harness.session.ensure('session-1'))).toBe('session-2') + // A second request whose 401 also named session-1 must keep the new token. + expect(tokenOf(await harness.session.ensure('session-1'))).toBe('session-2') + expect(harness.challenges()).toBe(2) + }) + + it('reports a refused handshake as rejected rather than unreachable', async () => { + const harness = createSessionHarness({ sessionStatus: 403 }) + + expect(await harness.session.ensure(null)).toEqual({ ok: false, reason: 'rejected' }) + }) + + it('reports a session minted for another host as rejected', async () => { + const harness = createSessionHarness({ wrongFingerprint: true }) + + expect(await harness.session.ensure(null)).toEqual({ ok: false, reason: 'rejected' }) + }) + + it('caches a refusal briefly instead of re-handshaking on every call', async () => { + const harness = createSessionHarness({ sessionStatus: 403 }) + + await harness.session.ensure(null) + await harness.session.ensure(null) + expect(harness.challenges()).toBe(1) + + harness.now.value += 30_000 + await harness.session.ensure(null) + expect(harness.challenges()).toBe(2) + }) + + it('never caches a transport failure, which may clear on the next try', async () => { + const fetchImpl = vi.fn(async () => { + throw new Error('network down') + }) as unknown as typeof globalThis.fetch + const session = new PushGatewaySession({ + origin: GATEWAY_ORIGIN, + keypair: createPushHostKeypair(), + fetchImpl, + now: () => NOW + }) + + expect(await session.ensure(null)).toEqual({ ok: false, reason: 'unreachable' }) + expect(await session.ensure(null)).toEqual({ ok: false, reason: 'unreachable' }) + expect(fetchImpl).toHaveBeenCalledTimes(2) + }) + + it('reports a rate-limited challenge as unreachable and backs off', async () => { + const harness = createSessionHarness({ challengeStatus: 429 }) + + expect(await harness.session.ensure(null)).toEqual({ ok: false, reason: 'unreachable' }) + expect(await harness.session.ensure(null)).toEqual({ ok: false, reason: 'unreachable' }) + expect(harness.requests()).toBe(1) + + harness.now.value += 60_000 + await harness.session.ensure(null) + expect(harness.requests()).toBe(2) + }) + + it('reports a rate-limited session mint as unreachable, not refused', async () => { + const harness = createSessionHarness({ sessionStatus: 429 }) + + expect(await harness.session.ensure(null)).toEqual({ ok: false, reason: 'unreachable' }) + // Cached for a minute, so the next dispatch does not spend more of the bucket. + expect(await harness.session.ensure(null)).toEqual({ ok: false, reason: 'unreachable' }) + expect(harness.challenges()).toBe(1) + }) + + it('shares one handshake across concurrent callers', async () => { + const harness = createSessionHarness() + + await Promise.all([harness.session.ensure(null), harness.session.ensure(null)]) + expect(harness.challenges()).toBe(1) + }) +}) diff --git a/src/main/runtime/push/push-gateway-session.ts b/src/main/runtime/push/push-gateway-session.ts new file mode 100644 index 00000000000..dd50b813f1d --- /dev/null +++ b/src/main/runtime/push/push-gateway-session.ts @@ -0,0 +1,157 @@ +// Why: the challenge/proof handshake every push request rides on, split out of +// push-gateway-client.ts so the session cache and its refusal cache stay readable +// next to the request methods rather than buried under them. +import { z } from 'zod' +import { cancelUnreadResponseBody } from '../../lib/unread-response-body' +import type { E2EEKeypair } from '../e2ee-keypair' +import { deriveRelayHostId } from '../relay/relay-http-client' +import { answerPushHostChallenge } from './push-host-proof' +import { + postPushGatewayJson, + readPushGatewayJson, + type PushGatewayFailure +} from './push-gateway-response' + +// Re-auth a little early so a send never spends its one retry on a token that +// expired between the check and the request. +const SESSION_RENEWAL_MARGIN_MS = 60_000 +// Why: a gateway that refuses this host's proof refuses the identical next one, +// so without this every dispatch pays two full handshake round trips to relearn it. +const HANDSHAKE_REFUSAL_TTL_MS = 30_000 +// Why: the handshake routes sit behind a per-IP bucket. Backing off keeps this +// host from spending the whole bucket on challenges it will never get to use. +const HANDSHAKE_RATE_LIMIT_TTL_MS = 60_000 + +const ChallengeResponseSchema = z + .object({ + challengeId: z.string().min(1).max(512), + gatewayEphemeralPublicKeyB64: z.string().min(1).max(128), + nonceB64: z.string().min(1).max(128), + ciphertextB64: z + .string() + .min(1) + .max(8 * 1024), + expiresAt: z.number().int().positive().max(Number.MAX_SAFE_INTEGER) + }) + .strict() + +const SessionResponseSchema = z + .object({ + sessionToken: z.string().min(1).max(1024), + expiresAt: z.number().int().positive().max(Number.MAX_SAFE_INTEGER), + hostFingerprint: z.string().min(1).max(64) + }) + .strict() + +export type PushSession = { token: string; expiresAt: number } +export type PushSessionOutcome = { ok: true; session: PushSession } | PushGatewayFailure + +type PushGatewaySessionOptions = { + origin: string + keypair: E2EEKeypair + fetchImpl: typeof globalThis.fetch + now: () => number +} + +export class PushGatewaySession { + private readonly origin: string + private readonly keypair: E2EEKeypair + private readonly fetchImpl: typeof globalThis.fetch + private readonly now: () => number + readonly hostFingerprint: string + private session: PushSession | null = null + private pending: Promise | null = null + private negative: { until: number; reason: PushGatewayFailure['reason'] } | null = null + + constructor(options: PushGatewaySessionOptions) { + this.origin = options.origin + this.keypair = options.keypair + this.fetchImpl = options.fetchImpl + this.now = options.now + this.hostFingerprint = deriveRelayHostId(options.keypair.publicKey) + } + + /** + * `staleToken` is the token that just received a 401. Only that exact session is + * dropped: a concurrent request may already have installed a good one, and + * clearing unconditionally would throw it away and re-handshake for nothing. + */ + async ensure(staleToken: string | null): Promise { + if (staleToken !== null && this.session?.token === staleToken) { + this.session = null + } + const cached = this.session + if (cached && cached.expiresAt - SESSION_RENEWAL_MARGIN_MS > this.now()) { + return { ok: true, session: cached } + } + if (this.negative && this.negative.until > this.now()) { + return { ok: false, reason: this.negative.reason } + } + // Concurrent sends must not each burn a challenge; share one handshake. + this.pending ??= this.open().finally(() => { + this.pending = null + }) + return await this.pending + } + + private async open(): Promise { + const challenge = await this.handshakePost( + '/v1/host/challenge', + { v: 1, hostPublicKeyB64: this.keypair.publicKeyB64 }, + ChallengeResponseSchema + ) + if (!challenge.ok) { + return this.remember(challenge) + } + const proofB64 = answerPushHostChallenge(challenge.value, { + gatewayOrigin: this.origin, + hostFingerprint: this.hostFingerprint, + hostPublicKey: this.keypair.publicKey, + hostSecretKey: this.keypair.secretKey, + now: this.now + }) + if (!proofB64) { + // A challenge this host cannot answer is a refusal, not a dropped packet. + return this.remember({ ok: false, reason: 'rejected' }) + } + const parsed = await this.handshakePost( + '/v1/host/session', + { v: 1, challengeId: challenge.value.challengeId, proofB64 }, + SessionResponseSchema + ) + if (!parsed.ok) { + return this.remember(parsed) + } + if (parsed.value.hostFingerprint !== this.hostFingerprint) { + // The gateway answered for some other host; that token is never usable here. + return this.remember({ ok: false, reason: 'rejected' }) + } + this.session = { token: parsed.value.sessionToken, expiresAt: parsed.value.expiresAt } + this.negative = null + return { ok: true, session: this.session } + } + + private async handshakePost( + path: string, + body: unknown, + schema: TSchema + ): Promise<{ ok: true; value: z.infer } | PushGatewayFailure> { + const response = await postPushGatewayJson(this.fetchImpl, `${this.origin}${path}`, body) + if (response.ok && response.response.status === 429) { + await cancelUnreadResponseBody(response.response) + // Rate limiting refuses the moment, not this host: back off, stay retryable + // so register reports gateway_unreachable and send keeps its one retry. + this.negative = { until: this.now() + HANDSHAKE_RATE_LIMIT_TTL_MS, reason: 'unreachable' } + return { ok: false, reason: 'unreachable' } + } + return await readPushGatewayJson(response, schema) + } + + /** Caches refusals only: a transport failure may clear on the very next try. */ + private remember(failure: PushGatewayFailure): PushGatewayFailure { + if (failure.reason === 'rejected') { + this.negative = { until: this.now() + HANDSHAKE_REFUSAL_TTL_MS, reason: 'rejected' } + } + return failure + } +} diff --git a/src/main/runtime/push/push-host-challenge-fixtures.ts b/src/main/runtime/push/push-host-challenge-fixtures.ts new file mode 100644 index 00000000000..e48dec33c7a --- /dev/null +++ b/src/main/runtime/push/push-host-challenge-fixtures.ts @@ -0,0 +1,136 @@ +// Test fixtures: builds the sealed challenge the push gateway would issue, so the +// proof answerer and the gateway client can both be exercised against a real box. +import { createHmac, randomBytes } from 'node:crypto' +import nacl from 'tweetnacl' +import type { E2EEKeypair } from '../e2ee-keypair' +import type { PushHostChallenge, PushHostProofContext } from './push-host-proof' + +const encoder = new TextEncoder() +export const PUSH_PROOF_DOMAIN = 'orca-push-host-proof/v1' +export const PUSH_CHALLENGE_DOMAIN = 'orca-push-host-challenge/v1' + +function concat(parts: readonly Uint8Array[]): Uint8Array { + const output = new Uint8Array(parts.reduce((total, part) => total + part.byteLength, 0)) + let offset = 0 + for (const part of parts) { + output.set(part, offset) + offset += part.byteLength + } + return output +} + +function uint32(value: number): Uint8Array { + const bytes = new Uint8Array(4) + new DataView(bytes.buffer).setUint32(0, value, false) + return bytes +} + +function uint64(value: number): Uint8Array { + const bytes = new Uint8Array(8) + new DataView(bytes.buffer).setBigUint64(0, BigInt(value), false) + return bytes +} + +function field(name: string, value: Uint8Array): Uint8Array { + const encodedName = encoder.encode(name) + return concat([uint32(encodedName.byteLength), encodedName, uint32(value.byteLength), value]) +} + +export function text(value: string): Uint8Array { + return encoder.encode(value) +} + +export type PushTranscriptInput = { + gatewayOrigin: string + gatewayKey: Uint8Array + nonce: Uint8Array + challengeId: string + issuedAt: number + expiresAt: number + hostFingerprint: string + hostKey: Uint8Array +} + +export function buildPushTranscript(input: PushTranscriptInput): Uint8Array { + return concat([ + field('protocol', text(PUSH_PROOF_DOMAIN)), + field('version', new Uint8Array([1])), + field('gatewayOrigin', text(input.gatewayOrigin)), + field('gatewayEphemeralPublicKey', input.gatewayKey), + field('challengeNonce', input.nonce), + field('challengeId', text(input.challengeId)), + field('issuedAt', uint64(input.issuedAt)), + field('expiresAt', uint64(input.expiresAt)), + field('hostFingerprint', text(input.hostFingerprint)), + field('hostPublicKey', input.hostKey) + ]) +} + +export function pushAckProof(secret: Uint8Array, transcript: Uint8Array): string { + return createHmac('sha256', secret) + .update(text(`${PUSH_PROOF_DOMAIN}\0ack\0`)) + .update(transcript) + .digest('base64') +} + +export function createPushHostKeypair(): E2EEKeypair { + const keys = nacl.box.keyPair() + return { + publicKey: keys.publicKey, + secretKey: keys.secretKey, + publicKeyB64: Buffer.from(keys.publicKey).toString('base64') + } +} + +/** Seals a challenge for `hostPublicKey`; overrides let a suite corrupt one field at a time. */ +export function buildPushChallengeFixture(input: { + hostKeypair: E2EEKeypair + gatewayOrigin: string + hostFingerprint: string + issuedAt: number + challengeId?: string + transcript?: Partial + challenge?: Partial +}): { challenge: PushHostChallenge; context: Omit; proof: string } { + const gatewayKeys = nacl.box.keyPair() + const nonce = randomBytes(24) + const secret = randomBytes(32) + const expiresAt = input.issuedAt + 10_000 + const challengeId = input.challengeId ?? 'challenge-1' + const transcript = buildPushTranscript({ + gatewayOrigin: input.gatewayOrigin, + gatewayKey: gatewayKeys.publicKey, + nonce, + challengeId, + issuedAt: input.issuedAt, + expiresAt, + hostFingerprint: input.hostFingerprint, + hostKey: input.hostKeypair.publicKey, + ...input.transcript + }) + const plaintext = concat([ + text(`${PUSH_CHALLENGE_DOMAIN}\0`), + uint32(transcript.byteLength), + transcript, + secret + ]) + return { + challenge: { + challengeId, + gatewayEphemeralPublicKeyB64: Buffer.from(gatewayKeys.publicKey).toString('base64'), + nonceB64: nonce.toString('base64'), + ciphertextB64: Buffer.from( + nacl.box(plaintext, nonce, input.hostKeypair.publicKey, gatewayKeys.secretKey) + ).toString('base64'), + expiresAt, + ...input.challenge + }, + context: { + gatewayOrigin: input.gatewayOrigin, + hostFingerprint: input.hostFingerprint, + hostPublicKey: input.hostKeypair.publicKey, + hostSecretKey: input.hostKeypair.secretKey + }, + proof: pushAckProof(secret, transcript) + } +} diff --git a/src/main/runtime/push/push-host-proof-vector.test.ts b/src/main/runtime/push/push-host-proof-vector.test.ts new file mode 100644 index 00000000000..6a012d9cd05 --- /dev/null +++ b/src/main/runtime/push/push-host-proof-vector.test.ts @@ -0,0 +1,30 @@ +import { describe, expect, it } from 'vitest' +import { createHmac } from 'node:crypto' +import vector from '../../../../cloud/packages/push-contract/src/push-host-proof-vector.json' +import { answerPushHostChallenge } from './push-host-proof' + +// Why: the gateway builds the challenge and this file answers it, in two +// workspaces that cannot import each other in CI. Both replay one checked-in +// vector; a transcript field drift on either side fails here and in the +// gateway's copy of this test. +describe('push host proof vector', () => { + it('answers the checked-in gateway challenge with the expected proof', () => { + const secret = Buffer.from(vector.challengeSecretB64, 'base64') + const transcript = Buffer.from(vector.transcriptB64, 'base64') + const expected = createHmac('sha256', secret) + .update(Buffer.from('orca-push-host-proof/v1\0ack\0')) + .update(transcript) + .digest('base64') + const reasons: string[] = [] + const proof = answerPushHostChallenge(vector.challenge, { + gatewayOrigin: vector.gatewayOrigin, + hostFingerprint: vector.hostFingerprint, + hostPublicKey: Buffer.from(vector.hostPublicKeyB64, 'base64'), + hostSecretKey: Buffer.from(vector.hostSecretKeyB64, 'base64'), + now: () => vector.issuedAt + 1_000, + onInvalid: (reason) => reasons.push(reason) + }) + expect(reasons).toEqual([]) + expect(proof).toBe(expected) + }) +}) diff --git a/src/main/runtime/push/push-host-proof.test.ts b/src/main/runtime/push/push-host-proof.test.ts new file mode 100644 index 00000000000..7ec59f3a1b4 --- /dev/null +++ b/src/main/runtime/push/push-host-proof.test.ts @@ -0,0 +1,106 @@ +import { describe, expect, it } from 'vitest' +import nacl from 'tweetnacl' +import { + buildPushChallengeFixture, + createPushHostKeypair, + type PushTranscriptInput +} from './push-host-challenge-fixtures' +import { answerPushHostChallenge, type PushHostProofContext } from './push-host-proof' + +const GATEWAY_ORIGIN = 'https://push.onorca.dev' +const HOST_FINGERPRINT = 'abcdef0123456789' +const ISSUED_AT = 1_770_000_000_000 + +function fixture( + overrides: { + transcript?: Partial + challenge?: Partial[0]> + context?: Partial + } = {} +): { + challenge: Parameters[0] + context: PushHostProofContext + proof: string +} { + const built = buildPushChallengeFixture({ + hostKeypair: createPushHostKeypair(), + gatewayOrigin: GATEWAY_ORIGIN, + hostFingerprint: HOST_FINGERPRINT, + issuedAt: ISSUED_AT, + transcript: overrides.transcript, + challenge: overrides.challenge + }) + return { + challenge: built.challenge, + context: { ...built.context, now: () => ISSUED_AT + 1_000, ...overrides.context }, + proof: built.proof + } +} + +describe('answerPushHostChallenge', () => { + it('answers a well-formed challenge with the ack HMAC', () => { + const { challenge, context, proof } = fixture() + expect(answerPushHostChallenge(challenge, context)).toBe(proof) + }) + + it('tolerates clock skew inside the 30s allowance', () => { + const { challenge, context, proof } = fixture({ context: { now: () => ISSUED_AT - 20_000 } }) + expect(answerPushHostChallenge(challenge, context)).toBe(proof) + }) + + it('refuses a challenge whose secret was sealed to another host', () => { + const { challenge, context } = fixture() + expect( + answerPushHostChallenge(challenge, { + ...context, + hostSecretKey: nacl.box.keyPair().secretKey + }) + ).toBeNull() + }) + + it.each([ + ['gatewayOrigin', { gatewayOrigin: 'https://push.evil.example' }], + ['hostFingerprint', { hostFingerprint: 'ffffffffffffffff' }], + ['challengeId', { challengeId: 'challenge-other' }], + ['issuedAt', { issuedAt: ISSUED_AT + 120_000 }] + ] as const)('refuses a transcript whose %s does not match the challenge', (_name, transcript) => { + const invalid: string[] = [] + const { challenge, context } = fixture({ + transcript, + context: { onInvalid: (reason) => invalid.push(reason) } + }) + expect(answerPushHostChallenge(challenge, context)).toBeNull() + expect(invalid.join(',')).toContain('transcript') + }) + + it('refuses a transcript that swaps in a different gateway ephemeral key', () => { + const { challenge, context } = fixture({ + transcript: { gatewayKey: nacl.box.keyPair().publicKey } + }) + expect(answerPushHostChallenge(challenge, context)).toBeNull() + }) + + it('refuses an expired challenge beyond the skew allowance', () => { + const { challenge, context } = fixture({ + context: { now: () => ISSUED_AT + 10_000 + 30_001 } + }) + expect(answerPushHostChallenge(challenge, context)).toBeNull() + }) + + it('refuses a challenge whose declared expiry disagrees with the transcript', () => { + const { challenge, context } = fixture() + expect( + answerPushHostChallenge({ ...challenge, expiresAt: challenge.expiresAt + 1 }, context) + ).toBeNull() + }) + + it('refuses a non-canonical base64 ephemeral key without opening the box', () => { + const { challenge, context } = fixture() + expect( + answerPushHostChallenge( + { ...challenge, gatewayEphemeralPublicKeyB64: 'not base64!' }, + context + ) + ).toBeNull() + }) +}) diff --git a/src/main/runtime/push/push-host-proof.ts b/src/main/runtime/push/push-host-proof.ts new file mode 100644 index 00000000000..33da29f4a03 --- /dev/null +++ b/src/main/runtime/push/push-host-proof.ts @@ -0,0 +1,113 @@ +// Why: the push gateway authenticates this host the same way the relay does — +// a sealed box the host can only open with its X25519 E2EE secret key — but with +// its own domain strings and a transcript that names the host by fingerprint +// instead of by account. See cloud/packages/push-contract/src. +import { + encodeText, + equalBytes, + hostChallengeAckProof, + openHostChallengeEnvelope, + parseHostChallengeTranscript, + readTranscriptUint64 +} from '../host-challenge-envelope' + +const PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN = 'orca-push-host-proof/v1' +const PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN = 'orca-push-host-challenge/v1' +const PUSH_HOST_PROOF_CLOCK_SKEW_MS = 30_000 +const MAX_PUSH_HOST_PROOF_CHALLENGE_WINDOW_MS = 10_000 +const PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT = 10 + +export type PushHostChallenge = { + challengeId: string + gatewayEphemeralPublicKeyB64: string + nonceB64: string + ciphertextB64: string + expiresAt: number +} + +export type PushHostProofContext = { + gatewayOrigin: string + hostFingerprint: string + hostPublicKey: Uint8Array + hostSecretKey: Uint8Array + now?: () => number + /** Reports the failing check by name only; never receives field values. */ + onInvalid?: (reason: string) => void +} + +function validateTranscript( + transcript: Uint8Array, + challenge: PushHostChallenge, + context: PushHostProofContext, + gatewayKey: Uint8Array, + nonce: Uint8Array +): boolean { + const fields = parseHostChallengeTranscript(transcript) + if (!fields || fields.size !== PUSH_HOST_PROOF_TRANSCRIPT_FIELD_COUNT) { + context.onInvalid?.('transcript-structure') + return false + } + const now = (context.now ?? Date.now)() + const issuedAt = readTranscriptUint64(fields.get('issuedAt')) + const expiresAt = readTranscriptUint64(fields.get('expiresAt')) + const checks: [string, boolean][] = [ + ['issuedAt-readable', issuedAt !== null], + ['issuedAt-not-future', issuedAt === null || issuedAt - PUSH_HOST_PROOF_CLOCK_SKEW_MS <= now], + ['not-expired', now - PUSH_HOST_PROOF_CLOCK_SKEW_MS <= challenge.expiresAt], + ['issuedAt-before-expiry', issuedAt === null || issuedAt <= challenge.expiresAt], + [ + 'window', + issuedAt === null || challenge.expiresAt - issuedAt <= MAX_PUSH_HOST_PROOF_CHALLENGE_WINDOW_MS + ], + ['expiry-consistent', expiresAt === challenge.expiresAt], + ['protocol', equalBytes(fields.get('protocol'), encodeText(PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN))], + ['version', equalBytes(fields.get('version'), new Uint8Array([1]))], + ['gatewayOrigin', equalBytes(fields.get('gatewayOrigin'), encodeText(context.gatewayOrigin))], + ['gatewayEphemeralPublicKey', equalBytes(fields.get('gatewayEphemeralPublicKey'), gatewayKey)], + ['challengeNonce', equalBytes(fields.get('challengeNonce'), nonce)], + ['challengeId', equalBytes(fields.get('challengeId'), encodeText(challenge.challengeId))], + [ + 'hostFingerprint', + equalBytes(fields.get('hostFingerprint'), encodeText(context.hostFingerprint)) + ], + ['hostPublicKey', equalBytes(fields.get('hostPublicKey'), context.hostPublicKey)] + ] + const failed = checks.filter(([, ok]) => !ok).map(([name]) => name) + if (failed.length > 0) { + context.onInvalid?.(`transcript:${failed.join('+')}`) + return false + } + return true +} + +/** Returns the base64 HMAC proof for a valid challenge, or null for anything else. */ +export function answerPushHostChallenge( + challenge: PushHostChallenge, + context: PushHostProofContext +): string | null { + const envelope = openHostChallengeEnvelope({ + peerEphemeralPublicKeyB64: challenge.gatewayEphemeralPublicKeyB64, + nonceB64: challenge.nonceB64, + ciphertextB64: challenge.ciphertextB64, + hostSecretKey: context.hostSecretKey, + plaintextDomain: PUSH_HOST_CHALLENGE_PLAINTEXT_DOMAIN, + onInvalid: context.onInvalid + }) + if ( + !envelope || + !validateTranscript( + envelope.transcript, + challenge, + context, + envelope.peerEphemeralPublicKey, + envelope.nonce + ) + ) { + return null + } + return hostChallengeAckProof({ + secret: envelope.secret, + transcript: envelope.transcript, + proofDomain: PUSH_HOST_PROOF_TRANSCRIPT_DOMAIN + }) +} diff --git a/src/main/runtime/push/push-outcome-counters.test.ts b/src/main/runtime/push/push-outcome-counters.test.ts new file mode 100644 index 00000000000..67ccc475cfc --- /dev/null +++ b/src/main/runtime/push/push-outcome-counters.test.ts @@ -0,0 +1,25 @@ +import { expect, it, vi } from 'vitest' +import { PushOutcomeCounters } from './push-outcome-counters' +it('limits failure logs while retaining category counts', () => { + let now = 0 + const log = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + const counters = new PushOutcomeCounters(() => now) + counters.record('rejected') + counters.record('error') + counters.record('error') + expect(log).toHaveBeenCalledTimes(1) + now += 60_000 + counters.record('rate_limited') + expect(JSON.parse(String(log.mock.calls[1]![0]))).toEqual({ + event: 'orca_desktop_push_failures', + error: 2, + rate_limited: 1 + }) + counters.record('unreachable') + counters.flush() + expect(log).toHaveBeenCalledTimes(3) + } finally { + log.mockRestore() + } +}) diff --git a/src/main/runtime/push/push-outcome-counters.ts b/src/main/runtime/push/push-outcome-counters.ts new file mode 100644 index 00000000000..6b2507e5a18 --- /dev/null +++ b/src/main/runtime/push/push-outcome-counters.ts @@ -0,0 +1,27 @@ +type PushOutcome = 'error' | 'rate_limited' | 'rejected' | 'unreachable' + +export class PushOutcomeCounters { + private readonly counts = new Map() + private nextLogAt = 0 + + constructor(private readonly now: () => number = Date.now) {} + + record(outcome: PushOutcome): void { + this.counts.set(outcome, (this.counts.get(outcome) ?? 0) + 1) + if (this.now() < this.nextLogAt) { + return + } + this.nextLogAt = this.now() + 60_000 + this.flush() + } + + flush(): void { + if (!this.counts.size) { + return + } + console.warn( + JSON.stringify({ event: 'orca_desktop_push_failures', ...Object.fromEntries(this.counts) }) + ) + this.counts.clear() + } +} diff --git a/src/main/runtime/push/push-policy-pipeline.integration.test.ts b/src/main/runtime/push/push-policy-pipeline.integration.test.ts new file mode 100644 index 00000000000..47ba42052f8 --- /dev/null +++ b/src/main/runtime/push/push-policy-pipeline.integration.test.ts @@ -0,0 +1,141 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, expect, it, vi } from 'vitest' +import { readDesktopAwayState } from '../../notifications/desktop-away-state' +import { DeviceRegistry } from '../device-registry' +import { RuntimeMobileNotificationController } from '../runtime-mobile-notification-controller' +import { setRuntimeDesktopSurface } from '../runtime-desktop-surface' +import { DesktopPushService } from './desktop-push-service' +import { PushUnregisterOutbox } from './push-unregister-outbox' +import { createPushHostKeypair } from './push-host-challenge-fixtures' + +const paths: string[] = [] +const services: DesktopPushService[] = [] +const filter = { + onlyWhenDesktopAway: true +} +const flush = () => new Promise((resolve) => setImmediate(resolve)) + +afterEach(() => { + services.splice(0).forEach((service) => service.stop()) + paths.splice(0).forEach((path) => rmSync(path, { recursive: true, force: true })) + setRuntimeDesktopSurface(null) + vi.restoreAllMocks() +}) + +async function pipeline() { + const path = mkdtempSync(join(tmpdir(), 'orca-push-policy-')) + paths.push(path) + const registry = new DeviceRegistry(path) + const device = registry.addDevice('policy-phone', 'mobile') + const controller = new RuntimeMobileNotificationController() + const client = { + registerDevice: vi.fn(async () => ({ ok: true, registrationId: 'policy-registration' })), + deleteDevice: vi.fn(async () => true), + send: vi.fn(async () => ({ ok: true, results: [] })) + } + const service = DesktopPushService.create({ + runtime: { + setMobilePushRegistrar: controller.setPushRegistrar.bind(controller), + onNotificationDispatched: controller.onDispatched.bind(controller) + } as never, + runtimeRpc: { + getE2EEKeypair: () => createPushHostKeypair(), + getDeviceRegistry: () => registry, + getPushUnregisterOutbox: () => new PushUnregisterOutbox(path), + setOnPushUnregisterQueued: () => {} + } as never, + gatewayUrl: 'https://push.onorca.dev', + client: client as never + })! + services.push(service) + service.start() + const register = () => + controller.registerPushDevice({ + deviceId: device.deviceId, + platform: 'ios', + token: 'test-token', + filter + }) + expect(await register()).toMatchObject({ registered: true }) + const dispatch = () => + controller.dispatch({ + type: 'notification', + source: 'agent-task-complete', + agentState: 'done', + notificationId: 'policy-event', + title: 'Policy test', + body: 'Policy test' + }) + return { path, registry, device, controller, client, register, dispatch } +} + +it('carries the native idle boundary through replay and push dispatch', async () => { + let idle = 179 + setRuntimeDesktopSurface({ + isAwayForMobileNotifications: () => + readDesktopAwayState({ + getSystemIdleState: () => 'active', + getSystemIdleTime: () => idle + }), + showNotification: () => false, + findWindowById: () => null, + onIpc: () => {}, + removeIpcListener: () => {} + }) + const h = await pipeline() + h.dispatch() + await flush() + expect(h.client.send).not.toHaveBeenCalled() + idle = 180 + h.dispatch() + await flush() + expect(h.client.send).toHaveBeenCalledTimes(1) + idle = 0 + h.dispatch() + await flush() + expect(h.client.send).toHaveBeenCalledTimes(1) + const replay = h.controller.getMissedSince(0) + expect(replay).toHaveLength(3) + expect(replay.map((event) => (event.type === 'notification' ? event.desktopAway : null))).toEqual( + [false, true, false] + ) +}) + +it('keeps headless presence unknown and legacy socket events readable', async () => { + setRuntimeDesktopSurface(null) + const h = await pipeline() + const events: unknown[] = [] + h.controller.onDispatched((event) => events.push(JSON.parse(JSON.stringify(event)))) + h.dispatch() + await flush() + expect(events[0]).not.toHaveProperty('desktopAway') + expect(h.client.send).toHaveBeenCalledTimes(1) +}) + +it('expires persisted registration at seven days despite host activity and renews explicitly', async () => { + const now = 1_800_000_000_000 + const clock = vi.spyOn(Date, 'now').mockReturnValue(now) + const h = await pipeline() + const deadline = now + 7 * 86400_000 + const persisted = new DeviceRegistry(h.path).getDevice(h.device.deviceId)?.pushRegistration + expect(persisted?.expiresAt).toBe(deadline) + clock.mockReturnValue(deadline - 1) + h.dispatch() + await flush() + expect(h.client.send).toHaveBeenCalledTimes(1) + expect(h.registry.getDevice(h.device.deviceId)?.pushRegistration?.expiresAt).toBe(deadline) + clock.mockReturnValue(deadline) + h.dispatch() + h.controller.dismiss('policy-event') + await flush() + expect(h.client.send).toHaveBeenCalledTimes(1) + await h.register() + expect(h.registry.getDevice(h.device.deviceId)?.pushRegistration?.expiresAt).toBe( + deadline + 7 * 86400_000 + ) + h.dispatch() + await flush() + expect(h.client.send).toHaveBeenCalledTimes(2) +}) diff --git a/src/main/runtime/push/push-preferences.test.ts b/src/main/runtime/push/push-preferences.test.ts new file mode 100644 index 00000000000..947cde4fa67 --- /dev/null +++ b/src/main/runtime/push/push-preferences.test.ts @@ -0,0 +1,85 @@ +import { expect, it } from 'vitest' +import { createHarness, notification, registration, flush } from './push-dispatcher.test-fixture' + +it('applies desktop category eligibility regardless of phone sound preferences', async () => { + const harness = createHarness({ + devices: [ + { + deviceId: 'mirror', + pushRegistration: registration({ + registrationId: 'mirror' + }) + }, + { + deviceId: 'quiet', + pushRegistration: registration({ + registrationId: 'quiet', + filter: { sound: false } + }) + }, + { + deviceId: 'second-phone', + pushRegistration: registration({ registrationId: 'second-phone' }) + } + ] + }) + harness.dispatcher.enqueue(notification({ source: 'terminal-bell', desktopAllowed: false })) + await flush() + expect(harness.sends).toHaveLength(0) + + harness.dispatcher.enqueue(notification({ source: 'terminal-bell', desktopAllowed: true })) + await flush() + expect(harness.sends).toHaveLength(2) + expect(harness.sends[0]).toMatchObject({ registrationIds: ['mirror', 'second-phone'] }) + expect(harness.sends[1]).toMatchObject({ + registrationIds: ['quiet'], + notification: { sound: false } + }) +}) + +it('keeps sound preferences separate when several phones receive the same event', async () => { + const harness = createHarness({ + devices: [ + { deviceId: 'loud', pushRegistration: registration({ registrationId: 'loud' }) }, + { + deviceId: 'quiet', + pushRegistration: registration({ + registrationId: 'quiet', + filter: { ...registration().filter, sound: false } + }) + } + ] + }) + harness.dispatcher.enqueue(notification()) + await flush() + expect(harness.sends).toHaveLength(2) + expect(harness.sends[0]).toMatchObject({ registrationIds: ['loud'] }) + expect(harness.sends[0].notification.sound).toBeUndefined() + expect(harness.sends[1]).toMatchObject({ + registrationIds: ['quiet'], + notification: { sound: false } + }) +}) + +it('applies burst suppression independently to each eligible phone', async () => { + const harness = createHarness({ + devices: [ + { + deviceId: 'all', + pushRegistration: registration({ + registrationId: 'all' + }) + }, + { + deviceId: 'second-phone', + pushRegistration: registration({ + registrationId: 'second-phone' + }) + } + ] + }) + harness.dispatcher.enqueue(notification({ source: 'terminal-bell', emittedAt: 10000 })) + harness.dispatcher.enqueue(notification({ emittedAt: 10250 })) + await flush() + expect(harness.sends.map((send) => send.registrationIds)).toEqual([['all', 'second-phone']]) +}) diff --git a/src/main/runtime/push/push-register-throttle.ts b/src/main/runtime/push/push-register-throttle.ts new file mode 100644 index 00000000000..7cc31bbb11d --- /dev/null +++ b/src/main/runtime/push/push-register-throttle.ts @@ -0,0 +1,45 @@ +// Why: notifications.registerPush costs a gateway write and a synchronous +// registry write on the main thread, and a paired phone may call it as often +// as it likes. A phone legitimately registers on switch-on, on each host +// connect, and on a token change, so a small per-device bucket bounds a loop +// without getting in the way of any of those. +const DEFAULT_CAPACITY = 10 +const DEFAULT_WINDOW_MS = 60_000 + +type Bucket = { tokens: number; updatedAt: number } + +export type PushRegisterThrottleOptions = { + capacity?: number + windowMs?: number + now?: () => number +} + +export class PushRegisterThrottle { + private readonly buckets = new Map() + private readonly capacity: number + private readonly windowMs: number + private readonly now: () => number + + constructor(options: PushRegisterThrottleOptions = {}) { + this.capacity = options.capacity ?? DEFAULT_CAPACITY + this.windowMs = options.windowMs ?? DEFAULT_WINDOW_MS + this.now = options.now ?? Date.now + } + + allow(deviceId: string): boolean { + const now = this.now() + const bucket = this.buckets.get(deviceId) + const refilled = bucket + ? Math.min( + this.capacity, + bucket.tokens + Math.max(0, ((now - bucket.updatedAt) * this.capacity) / this.windowMs) + ) + : this.capacity + if (refilled < 1) { + this.buckets.set(deviceId, { tokens: refilled, updatedAt: now }) + return false + } + this.buckets.set(deviceId, { tokens: refilled - 1, updatedAt: now }) + return true + } +} diff --git a/src/main/runtime/push/push-registration-races.test.ts b/src/main/runtime/push/push-registration-races.test.ts new file mode 100644 index 00000000000..1cd946daea2 --- /dev/null +++ b/src/main/runtime/push/push-registration-races.test.ts @@ -0,0 +1,294 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, expect, it, vi } from 'vitest' +import { DeviceRegistry } from '../device-registry' +import { DesktopPushService } from './desktop-push-service' +import { PushUnregisterOutbox } from './push-unregister-outbox' +import { createPushHostKeypair } from './push-host-challenge-fixtures' +import { PushDispatcher } from './push-dispatcher' + +const paths: string[] = [] +afterEach(() => { + for (const path of paths.splice(0)) { + rmSync(path, { recursive: true, force: true }) + } +}) +const input = { + platform: 'android' as const, + token: 'synthetic', + filter: {} +} +const tick = () => new Promise((resolve) => setImmediate(resolve)) + +function harness() { + const path = mkdtempSync(join(tmpdir(), 'push-races-')) + paths.push(path) + const registry = new DeviceRegistry(path) + const deviceId = registry.addDevice('phone', 'mobile').deviceId + const outbox = new PushUnregisterOutbox(path) + const retries: { run: () => void; delayMs: number }[] = [] + let live = false + let reachable = true + const client = { + registerDevice: vi.fn(async () => { + live = true + return { ok: true, registrationId: 'stable-id' } + }), + deleteDevice: vi.fn(async (_registrationId: string) => { + if (!reachable) { + return false + } + live = false + return true + }), + send: vi.fn() + } + const service = DesktopPushService.create({ + gatewayUrl: 'https://push.example.test', + client: client as never, + scheduleRetry: (run, delayMs) => retries.push({ run, delayMs }), + runtime: { + setMobilePushRegistrar: () => {}, + onNotificationDispatched: () => () => {} + } as never, + runtimeRpc: { + getE2EEKeypair: createPushHostKeypair, + getDeviceRegistry: () => registry, + getPushUnregisterOutbox: () => outbox, + setOnPushUnregisterQueued: () => {} + } as never + })! + service.start() + return { + path, + retries, + registry, + deviceId, + outbox, + client, + service, + live: () => live, + reachable: (value: boolean) => { + reachable = value + } + } +} + +it('deletes obsolete gateway state before reporting successful re-enable', async () => { + const h = harness() + await h.service.register({ ...input, deviceId: h.deviceId }) + h.reachable(false) + await h.service.unregister(h.deviceId) + await h.service.flushUnregisterOutbox() + expect(h.outbox.pending()).toHaveLength(1) + expect(await h.service.register({ ...input, deviceId: h.deviceId })).toMatchObject({ + registered: false + }) + h.reachable(true) + expect(await h.service.register({ ...input, deviceId: h.deviceId })).toMatchObject({ + registered: true + }) + await h.service.flushUnregisterOutbox() + expect(h.live()).toBe(true) + expect(h.outbox.pending()).toEqual([]) +}) + +it('waits for an already-running delete before re-registering', async () => { + const h = harness() + await h.service.register({ ...input, deviceId: h.deviceId }) + let release!: () => void + const normalDelete = h.client.deleteDevice.getMockImplementation()! + h.client.deleteDevice.mockImplementationOnce(async () => { + await new Promise((resolve) => { + release = resolve + }) + return normalDelete('stable-id') + }) + await h.service.unregister(h.deviceId) + await tick() + const registration = h.service.register({ ...input, deviceId: h.deviceId }) + await tick() + expect(h.client.registerDevice).toHaveBeenCalledTimes(1) + release() + await registration + await h.service.flushUnregisterOutbox() + expect(h.live()).toBe(true) +}) + +it('orders unregister after a register already in flight', async () => { + const h = harness() + let release!: () => void + const normalRegister = h.client.registerDevice.getMockImplementation()! + h.client.registerDevice.mockImplementationOnce(async () => { + await new Promise((resolve) => { + release = resolve + }) + return normalRegister() + }) + const registered = h.service.register({ ...input, deviceId: h.deviceId }) + await tick() + const unregistered = h.service.unregister(h.deviceId) + release() + await Promise.all([registered, unregistered]) + await h.service.flushUnregisterOutbox() + expect(h.registry.getDevice(h.deviceId)?.pushRegistration).toBeUndefined() + expect(h.live()).toBe(false) +}) + +it('does not clear a replacement with the same ID and timestamp after a stale dead response', async () => { + const h = harness() + await h.service.register({ ...input, deviceId: h.deviceId }) + let finish!: (value: unknown) => void + h.client.send.mockImplementation( + () => + new Promise((resolve) => { + finish = resolve + }) + ) + const dispatcher = new PushDispatcher({ registry: h.registry, client: h.client as never }) + dispatcher.enqueue({ + type: 'notification', + source: 'plugin', + title: 'test', + body: '', + notificationEpoch: 'epoch', + notificationSeq: 1 + }) + const original = h.registry.getDevice(h.deviceId)!.pushRegistration! + h.registry.setPushRegistration(h.deviceId, { ...original }) + finish({ ok: true, results: [{ registrationId: 'stable-id', status: 'dead' }] }) + await tick() + expect(h.registry.getDevice(h.deviceId)?.pushRegistration).toEqual(original) +}) + +it('drains a cleanup queued as an empty flush is completing', async () => { + const h = harness() + // Let the startup drain return, but queue cleanup before its promise finalizer runs. + await Promise.resolve() + h.outbox.enqueue({ registrationId: 'orphan', deviceId: h.deviceId }) + await h.service.flushUnregisterOutbox() + expect(h.client.deleteDevice).toHaveBeenCalledWith('orphan') + expect(h.outbox.pending()).toEqual([]) +}) + +it('preserves the live route when clearing local registration fails, then cleans before re-registering', async () => { + const h = harness() + await h.service.register({ ...input, deviceId: h.deviceId }) + await h.service.flushUnregisterOutbox() + const persist = vi.spyOn(h.registry, 'setPushRegistration').mockImplementation(() => { + throw new Error('disk full') + }) + await expect(h.service.unregister(h.deviceId)).rejects.toThrow('disk full') + await tick() + expect(h.client.deleteDevice).not.toHaveBeenCalled() + expect(h.outbox.pending()).toHaveLength(1) + expect(h.live()).toBe(true) + persist.mockRestore() + expect(await h.service.register({ ...input, deviceId: h.deviceId })).toMatchObject({ + registered: true + }) + await h.service.flushUnregisterOutbox() + expect(h.client.deleteDevice).toHaveBeenCalledWith('stable-id') + expect(h.outbox.pending()).toEqual([]) + expect(h.live()).toBe(true) + h.service.stop() +}) + +it('retries an old failure before mid-drain work, then waits for the armed backoff', async () => { + const h = harness() + await h.service.flushUnregisterOutbox() + const deletes: string[] = [] + h.client.deleteDevice.mockImplementation(async (registrationId) => { + deletes.push(registrationId) + if (deletes.length === 1) { + h.outbox.enqueue({ registrationId: 'new', deviceId: 'new-phone' }) + void h.service.flushUnregisterOutbox() + } + return registrationId === 'new' + }) + h.outbox.enqueue({ registrationId: 'old', deviceId: h.deviceId }) + await h.service.flushUnregisterOutbox() + expect(deletes).toEqual(['old', 'old', 'new']) + expect(h.outbox.pending().map((item) => item.registrationId)).toEqual(['old']) + expect(h.retries.map((retry) => retry.delayMs)).toEqual([30_000]) + await tick() + expect(deletes).toHaveLength(3) + h.retries[0].run() + await tick() + expect(deletes).toEqual(['old', 'old', 'new', 'old']) + expect(h.retries.map((retry) => retry.delayMs)).toEqual([30_000, 60_000]) + h.service.stop() +}) + +it('skips a snapshot delete consumed by same-device registration cleanup', async () => { + const h = harness() + await h.service.flushUnregisterOutbox() + let release!: () => void + h.client.deleteDevice.mockImplementationOnce( + () => + new Promise((resolve) => { + release = () => resolve(true) + }) + ) + h.outbox.enqueue({ registrationId: 'blocker', deviceId: 'other-phone' }) + h.outbox.enqueue({ registrationId: 'stable-id', deviceId: h.deviceId }) + const flush = h.service.flushUnregisterOutbox() + await tick() + expect(await h.service.register({ ...input, deviceId: h.deviceId })).toMatchObject({ + registered: true + }) + expect(h.live()).toBe(true) + release() + await flush + expect(h.client.deleteDevice.mock.calls).toEqual([['blocker'], ['stable-id']]) + expect(h.outbox.pending()).toEqual([]) + expect(h.live()).toBe(true) + h.service.stop() +}) + +it('finishes the current snapshot on stop and leaves later work durable for restart', async () => { + const h = harness() + await h.service.flushUnregisterOutbox() + let release!: () => void + h.client.deleteDevice.mockImplementationOnce( + () => + new Promise((resolve) => { + release = () => resolve(false) + }) + ) + h.outbox.enqueue({ registrationId: 'blocked', deviceId: h.deviceId }) + h.outbox.enqueue({ registrationId: 'in-snapshot', deviceId: 'other-phone' }) + const flush = h.service.flushUnregisterOutbox() + await tick() + h.outbox.enqueue({ registrationId: 'late', deviceId: 'late-phone' }) + void h.service.flushUnregisterOutbox() + h.service.stop() + release() + await flush + expect(h.client.deleteDevice.mock.calls).toEqual([['blocked'], ['in-snapshot']]) + expect(h.retries).toEqual([]) + const recovered = new PushUnregisterOutbox(h.path) + expect(recovered.pending().map((item) => item.registrationId)).toEqual(['blocked', 'late']) + await h.service.flushUnregisterOutbox() + expect(h.client.deleteDevice).toHaveBeenCalledTimes(2) + h.service.start() + await h.service.flushUnregisterOutbox() + expect(h.outbox.pending()).toEqual([]) + h.service.stop() +}) + +it('reports shutdown as retryable and allows registration after restart', async () => { + const h = harness() + h.service.stop() + expect(await h.service.register({ ...input, deviceId: h.deviceId })).toEqual({ + registered: false, + reason: 'gateway_unreachable' + }) + expect(h.client.registerDevice).not.toHaveBeenCalled() + h.service.start() + expect(await h.service.register({ ...input, deviceId: h.deviceId })).toMatchObject({ + registered: true + }) + h.service.stop() +}) diff --git a/src/main/runtime/push/push-registration-rpc.test.ts b/src/main/runtime/push/push-registration-rpc.test.ts new file mode 100644 index 00000000000..78a91a238fa --- /dev/null +++ b/src/main/runtime/push/push-registration-rpc.test.ts @@ -0,0 +1,156 @@ +import { mkdtempSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { eraseRpcMethods, type RpcContext, type RpcMethod } from '../rpc/core' +import { NOTIFICATION_METHODS } from '../rpc/methods/notifications' +import { DeviceRegistry } from '../device-registry' +import { OrcaRuntimeRpcServer } from '../runtime-rpc' +import { OrcaRuntimeService } from '../orca-runtime' + +function method(name: string): RpcMethod { + const found = eraseRpcMethods(NOTIFICATION_METHODS).find((candidate) => candidate.name === name) + if (!found || 'stream' in found) { + throw new Error(`${name} is not a one-shot RPC method`) + } + return found +} + +const REGISTER_PARAMS = { + platform: 'ios', + token: 'a'.repeat(64), + apnsEnvironment: 'sandbox', + filter: {} +} + +function contextFor(overrides: Partial): RpcContext { + return { + runtime: { + registerMobilePushDevice: vi.fn(async () => ({ + registered: true, + registrationId: 'reg-1' + })), + unregisterMobilePushDevice: vi.fn(async () => ({ unregistered: true })) + }, + ...overrides + } as unknown as RpcContext +} + +describe('notifications.registerPush', () => { + it('registers under the authenticated paired device id', async () => { + const registerPush = method('notifications.registerPush') + const ctx = contextFor({ clientKind: 'mobile', pairedDeviceId: 'device-1' }) + + const result = await registerPush.handler(registerPush.params!.parse(REGISTER_PARAMS), ctx) + + expect(result).toEqual({ registered: true, registrationId: 'reg-1' }) + expect(ctx.runtime.registerMobilePushDevice).toHaveBeenCalledWith({ + deviceId: 'device-1', + platform: 'ios', + token: REGISTER_PARAMS.token, + apnsEnvironment: 'sandbox', + filter: REGISTER_PARAMS.filter + }) + }) + + it.each([ + ['a runtime-scoped caller', { clientKind: 'runtime' as const, pairedDeviceId: 'device-1' }], + ['an in-process caller', {}], + ['a mobile caller with no paired device', { clientKind: 'mobile' as const }] + ])('refuses %s', async (_name, overrides) => { + const registerPush = method('notifications.registerPush') + const ctx = contextFor(overrides) + + expect(await registerPush.handler(registerPush.params!.parse(REGISTER_PARAMS), ctx)).toEqual({ + registered: false, + reason: 'not_mobile' + }) + expect(ctx.runtime.registerMobilePushDevice).not.toHaveBeenCalled() + }) + + it('requires an APNs environment for an iOS token', () => { + const registerPush = method('notifications.registerPush') + expect( + registerPush.params!.safeParse({ ...REGISTER_PARAMS, apnsEnvironment: undefined }).success + ).toBe(false) + expect( + registerPush.params!.safeParse({ + ...REGISTER_PARAMS, + platform: 'android', + apnsEnvironment: undefined + }).success + ).toBe(true) + }) + + it('rejects a caller-supplied device id instead of dropping it', () => { + const registerPush = method('notifications.registerPush') + expect( + registerPush.params!.safeParse({ ...REGISTER_PARAMS, deviceId: 'device-9' }).success + ).toBe(false) + }) + + it('rejects a malformed phone preference', () => { + const registerPush = method('notifications.registerPush') + expect( + registerPush.params!.safeParse({ + ...REGISTER_PARAMS, + filter: { sound: 'yes' } + }).success + ).toBe(false) + }) +}) + +describe('notifications.unregisterPush', () => { + it('unregisters the authenticated paired device', async () => { + const unregisterPush = method('notifications.unregisterPush') + const ctx = contextFor({ clientKind: 'mobile', pairedDeviceId: 'device-1' }) + + expect(await unregisterPush.handler(undefined, ctx)).toEqual({ unregistered: true }) + expect(ctx.runtime.unregisterMobilePushDevice).toHaveBeenCalledWith('device-1') + }) + + it('refuses a non-mobile caller', async () => { + const unregisterPush = method('notifications.unregisterPush') + const ctx = contextFor({ clientKind: 'runtime', pairedDeviceId: 'device-1' }) + + expect(await unregisterPush.handler(undefined, ctx)).toEqual({ unregistered: false }) + expect(ctx.runtime.unregisterMobilePushDevice).not.toHaveBeenCalled() + }) +}) + +describe('revokeMobileDevice', () => { + it('queues the gateway delete before the device row disappears', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-push-revoke-')) + const server = new OrcaRuntimeRpcServer({ + runtime: new OrcaRuntimeService(), + userDataPath, + enableWebSocket: false + }) + server['deviceRegistry'] = new DeviceRegistry(userDataPath) + const device = server['deviceRegistry']!.addDevice('phone', 'mobile') + server['deviceRegistry']!.setPushRegistration(device.deviceId, { + registrationId: 'reg-1', + filter: {}, + expiresAt: Date.now() + 7 * 86400_000 + }) + + expect(await server.revokeMobileDevice(device.deviceId)).toBe(true) + expect(server.getPushUnregisterOutbox().pending()).toEqual([ + expect.objectContaining({ registrationId: 'reg-1', deviceId: device.deviceId }) + ]) + }) + + it('queues nothing for a device that never enabled push', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-push-revoke-')) + const server = new OrcaRuntimeRpcServer({ + runtime: new OrcaRuntimeService(), + userDataPath, + enableWebSocket: false + }) + server['deviceRegistry'] = new DeviceRegistry(userDataPath) + const device = server['deviceRegistry']!.addDevice('phone', 'mobile') + + expect(await server.revokeMobileDevice(device.deviceId)).toBe(true) + expect(server.getPushUnregisterOutbox().pending()).toEqual([]) + }) +}) diff --git a/src/main/runtime/push/push-unpair-persistence.test.ts b/src/main/runtime/push/push-unpair-persistence.test.ts new file mode 100644 index 00000000000..560afcccd6f --- /dev/null +++ b/src/main/runtime/push/push-unpair-persistence.test.ts @@ -0,0 +1,65 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { DeviceRegistry } from '../device-registry' +import { OrcaRuntimeService } from '../orca-runtime' +import { DesktopPushService } from './desktop-push-service' +import { createPushHostKeypair } from './push-host-challenge-fixtures' +import { OrcaRuntimeRpcServer } from '../runtime-rpc' + +describe('mobile revoke when the registry write fails', () => { + it('preserves a live route after failed unpair and deletes it after durable removal', async () => { + const userDataPath = mkdtempSync(join(tmpdir(), 'orca-revoke-write-failure-')) + const runtime = new OrcaRuntimeService() + const server = new OrcaRuntimeRpcServer({ runtime, userDataPath, enableWebSocket: false }) + const registry = new DeviceRegistry(userDataPath) + const device = registry.addDevice('phone', 'mobile') + registry.setPushRegistration(device.deviceId, { + registrationId: 'reg-live', + filter: {}, + expiresAt: Date.now() + 60_000 + }) + server['deviceRegistry'] = registry + server['e2eeKeypair'] = createPushHostKeypair() + + const deleted: string[] = [] + const client = { + registerDevice: vi.fn(), + deleteDevice: vi.fn(async (registrationId: string) => { + deleted.push(registrationId) + return true + }), + send: vi.fn(async () => ({ ok: true, results: [] }) as const) + } + const service = DesktopPushService.create({ + runtime, + runtimeRpc: server, + gatewayUrl: 'https://push.onorca.dev', + client: client as never + })! + service.start() + const save = registry['save'].bind(registry) + registry['save'] = vi.fn(() => { + throw new Error('disk full') + }) + + await expect(server.revokeMobileDevice(device.deviceId)).rejects.toThrow('disk full') + await service.flushUnregisterOutbox() + + expect(registry.getDevice(device.deviceId)?.pushRegistration?.registrationId).toBe('reg-live') + expect(deleted).toEqual([]) + expect(server.getPushUnregisterOutbox().pending()).toHaveLength(1) + service.stop() + service.start() + await service.flushUnregisterOutbox() + expect(deleted).toEqual([]) + registry['save'] = save + expect(await server.revokeMobileDevice(device.deviceId)).toBe(true) + await service.flushUnregisterOutbox() + expect(deleted).toEqual(['reg-live']) + expect(server.getPushUnregisterOutbox().pending()).toEqual([]) + service.stop() + rmSync(userDataPath, { recursive: true, force: true }) + }) +}) diff --git a/src/main/runtime/push/push-unregister-outbox.test.ts b/src/main/runtime/push/push-unregister-outbox.test.ts new file mode 100644 index 00000000000..5f2d856d6ab --- /dev/null +++ b/src/main/runtime/push/push-unregister-outbox.test.ts @@ -0,0 +1,93 @@ +import { mkdtempSync, readFileSync, writeFileSync } from 'node:fs' +import type * as fs from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { PushUnregisterOutbox } from './push-unregister-outbox' + +vi.mock('node:fs', async (importOriginal) => { + const original = await importOriginal() + return { ...original, readFileSync: vi.fn(original.readFileSync) } +}) + +const OUTBOX_FILENAME = 'mobile-push-unregister-outbox.json' + +function userDataDir(): string { + return mkdtempSync(join(tmpdir(), 'orca-push-outbox-')) +} + +describe('PushUnregisterOutbox', () => { + it('survives a restart with the queued delete intact', () => { + const dir = userDataDir() + const first = new PushUnregisterOutbox(dir) + const item = first.enqueue({ registrationId: 'reg-1', deviceId: 'device-1' }) + + const reopened = new PushUnregisterOutbox(dir) + expect(reopened.pending()).toEqual([item]) + }) + + it('coalesces repeat enqueues of the same registration', () => { + const dir = userDataDir() + const outbox = new PushUnregisterOutbox(dir) + const first = outbox.enqueue({ registrationId: 'reg-1', deviceId: 'device-1' }) + const second = outbox.enqueue({ registrationId: 'reg-1', deviceId: 'device-1' }) + + expect(second.reqId).toBe(first.reqId) + expect(outbox.pending()).toHaveLength(1) + }) + + it('keeps a removal durable across a restart', () => { + const dir = userDataDir() + const outbox = new PushUnregisterOutbox(dir) + const kept = outbox.enqueue({ registrationId: 'reg-keep', deviceId: 'device-1' }) + const dropped = outbox.enqueue({ registrationId: 'reg-drop', deviceId: 'device-2' }) + outbox.remove(dropped.reqId) + + expect(new PushUnregisterOutbox(dir).pending()).toEqual([kept]) + }) + + it('drops malformed rows instead of failing the whole load', () => { + const dir = userDataDir() + const valid = new PushUnregisterOutbox(dir).enqueue({ + registrationId: 'reg-1', + deviceId: 'device-1' + }) + const path = join(dir, OUTBOX_FILENAME) + const stored: unknown[] = JSON.parse(readFileSync(path, 'utf-8')) + writeFileSync( + path, + JSON.stringify([...stored, { reqId: 'broken' }, null, 'nope', { registrationId: '' }]) + ) + + expect(new PushUnregisterOutbox(dir).pending()).toEqual([valid]) + }) + + it('preserves unreadable pending deletes until the outbox can be reloaded', () => { + const dir = userDataDir() + const pending = new PushUnregisterOutbox(dir).enqueue({ + registrationId: 'reg-1', + deviceId: 'device-1' + }) + const path = join(dir, OUTBOX_FILENAME) + const original = readFileSync(path, 'utf-8') + vi.mocked(readFileSync).mockImplementationOnce(() => { + throw Object.assign(new Error('temporarily unavailable'), { code: 'EIO' }) + }) + const unreadable = new PushUnregisterOutbox(dir) + + expect(() => unreadable.enqueue({ registrationId: 'reg-2', deviceId: 'device-2' })).toThrow( + 'Cannot overwrite unreadable push unregister outbox' + ) + expect(readFileSync(path, 'utf-8')).toBe(original) + const recovered = new PushUnregisterOutbox(dir) + expect(recovered.pending()).toEqual([pending]) + recovered.enqueue({ registrationId: 'reg-2', deviceId: 'device-2' }) + expect(new PushUnregisterOutbox(dir).pending()).toHaveLength(2) + }) + + it('starts empty when the file is not JSON at all', () => { + const dir = userDataDir() + writeFileSync(join(dir, OUTBOX_FILENAME), 'not json') + expect(new PushUnregisterOutbox(dir).pending()).toEqual([]) + }) +}) diff --git a/src/main/runtime/push/push-unregister-outbox.ts b/src/main/runtime/push/push-unregister-outbox.ts new file mode 100644 index 00000000000..81b9e7a769b --- /dev/null +++ b/src/main/runtime/push/push-unregister-outbox.ts @@ -0,0 +1,93 @@ +// Why: a phone that turns background notifications off, or gets unpaired, must +// have its token deleted at the gateway even if the gateway is unreachable right +// then. Modelled on relay-revoke-outbox.ts: durable, hardened, drained on start. +import { randomUUID } from 'node:crypto' +import { existsSync, readFileSync } from 'node:fs' +import { join } from 'node:path' +import { + hardenExistingSecureFile, + isUnreadableError, + writeSecureJsonFile +} from '../../../shared/secure-file' + +export type PushUnregisterOutboxItem = { + reqId: string + registrationId: string + deviceId: string +} + +const OUTBOX_FILENAME = 'mobile-push-unregister-outbox.json' + +function isItem(value: unknown): value is PushUnregisterOutboxItem { + if (!value || typeof value !== 'object') { + return false + } + const item = value as Partial + return ( + typeof item.reqId === 'string' && + typeof item.registrationId === 'string' && + item.registrationId.length > 0 && + typeof item.deviceId === 'string' + ) +} + +export class PushUnregisterOutbox { + private readonly path: string + private outboxUnreadable = false + private items: PushUnregisterOutboxItem[] + + constructor(userDataPath: string) { + this.path = join(userDataPath, OUTBOX_FILENAME) + this.items = this.load() + } + + enqueue(entry: { registrationId: string; deviceId: string }): PushUnregisterOutboxItem { + const existing = this.items.find((item) => item.registrationId === entry.registrationId) + if (existing) { + return existing + } + const item = { ...entry, reqId: randomUUID() } + const next = [...this.items, item] + this.save(next) + this.items = next + return item + } + + isUnreadable(): boolean { + return this.outboxUnreadable + } + + pending(): readonly PushUnregisterOutboxItem[] { + return this.items + } + + remove(reqId: string): void { + const next = this.items.filter((item) => item.reqId !== reqId) + if (next.length === this.items.length) { + return + } + this.save(next) + this.items = next + } + + private load(): PushUnregisterOutboxItem[] { + if (!existsSync(this.path)) { + return [] + } + try { + hardenExistingSecureFile(this.path) + const parsed: unknown = JSON.parse(readFileSync(this.path, 'utf-8')) + return Array.isArray(parsed) ? parsed.filter(isItem) : [] + } catch (error) { + this.outboxUnreadable = isUnreadableError(error) + return [] + } + } + + private save(items: readonly PushUnregisterOutboxItem[]): void { + if (this.outboxUnreadable) { + throw new Error('Cannot overwrite unreadable push unregister outbox') + } + writeSecureJsonFile(this.path, items) + } +} diff --git a/src/main/runtime/relay/relay-host-proof.ts b/src/main/runtime/relay/relay-host-proof.ts index 59c028b1ab1..a169540b5ee 100644 --- a/src/main/runtime/relay/relay-host-proof.ts +++ b/src/main/runtime/relay/relay-host-proof.ts @@ -1,13 +1,18 @@ -import { createHmac, timingSafeEqual } from 'node:crypto' -import nacl from 'tweetnacl' +import { + encodeText, + encodeUint64, + equalBytes, + hostChallengeAckProof, + openHostChallengeEnvelope, + parseHostChallengeTranscript, + readTranscriptUint64 +} from '../host-challenge-envelope' const HOST_PROOF_TRANSCRIPT_DOMAIN = 'orca-relay-host-proof/v1' const HOST_CHALLENGE_PLAINTEXT_DOMAIN = 'orca-relay-host-challenge/v1' // Covers routine NTP drift without extending the signed challenge window. const RELAY_HOST_PROOF_CLOCK_SKEW_MS = 30_000 const MAX_HOST_PROOF_CHALLENGE_WINDOW_MS = 10_000 -const textEncoder = new TextEncoder() -const textDecoder = new TextDecoder() export type RelayHostChallenge = { challengeId: string @@ -33,61 +38,6 @@ export type RelayHostProofContext = { onInvalid?: (reason: string) => void } -function decodeCanonicalBase64(value: string, expectedBytes: number): Uint8Array | null { - if (!/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(value)) { - return null - } - const decoded = Buffer.from(value, 'base64') - return decoded.byteLength === expectedBytes && decoded.toString('base64') === value - ? decoded - : null -} - -function uint64(value: number): Uint8Array { - const bytes = new Uint8Array(8) - new DataView(bytes.buffer).setBigUint64(0, BigInt(value), false) - return bytes -} - -function equal(left: Uint8Array | undefined, right: Uint8Array): boolean { - return Boolean(left && left.byteLength === right.byteLength && timingSafeEqual(left, right)) -} - -function parseTranscript(transcript: Uint8Array): Map | null { - const fields = new Map() - const view = new DataView(transcript.buffer, transcript.byteOffset, transcript.byteLength) - let offset = 0 - try { - while (offset < transcript.byteLength) { - const nameLength = view.getUint32(offset, false) - offset += 4 - const name = textDecoder.decode(transcript.slice(offset, offset + nameLength)) - offset += nameLength - const valueLength = view.getUint32(offset, false) - offset += 4 - if (fields.has(name) || offset + valueLength > transcript.byteLength) { - return null - } - fields.set(name, transcript.slice(offset, offset + valueLength)) - offset += valueLength - } - } catch { - return null - } - return offset === transcript.byteLength ? fields : null -} - -function readUint64(value: Uint8Array | undefined): number | null { - if (!value || value.byteLength !== 8) { - return null - } - const parsed = new DataView(value.buffer, value.byteOffset, value.byteLength).getBigUint64( - 0, - false - ) - return parsed <= BigInt(Number.MAX_SAFE_INTEGER) ? Number(parsed) : null -} - function validateTranscript( transcript: Uint8Array, challenge: RelayHostChallenge, @@ -95,17 +45,19 @@ function validateTranscript( relayKey: Uint8Array, nonce: Uint8Array ): boolean { - const fields = parseTranscript(transcript) + const fields = parseHostChallengeTranscript(transcript) if (!fields || fields.size !== 16) { context.onInvalid?.('transcript-structure') return false } const now = (context.now ?? Date.now)() - const issuedAt = readUint64(fields.get('issuedAt')) - const expiresAt = readUint64(fields.get('expiresAt')) + const issuedAt = readTranscriptUint64(fields.get('issuedAt')) + const expiresAt = readTranscriptUint64(fields.get('expiresAt')) const previousGeneration = fields.get('previousGeneration') const expectedPrevious = - context.previousGeneration === undefined ? new Uint8Array() : uint64(context.previousGeneration) + context.previousGeneration === undefined + ? new Uint8Array() + : encodeUint64(context.previousGeneration) // Main's 30s skew bounds with named-check reporting kept from the incident // instrumentation; deltas are relative offsets only, never absolute values. const checks: [string, boolean][] = [ @@ -124,25 +76,28 @@ function validateTranscript( issuedAt === null || challenge.expiresAt - issuedAt <= MAX_HOST_PROOF_CHALLENGE_WINDOW_MS ], ['expiry-consistent', expiresAt === challenge.expiresAt], - ['protocol', equal(fields.get('protocol'), textEncoder.encode(HOST_PROOF_TRANSCRIPT_DOMAIN))], - ['version', equal(fields.get('version'), new Uint8Array([1]))], - ['relayOrigin', equal(fields.get('relayOrigin'), textEncoder.encode(context.relayOrigin))], - ['relayEphemeralPublicKey', equal(fields.get('relayEphemeralPublicKey'), relayKey)], - ['challengeNonce', equal(fields.get('challengeNonce'), nonce)], - ['challengeId', equal(fields.get('challengeId'), textEncoder.encode(challenge.challengeId))], - ['userId', equal(fields.get('userId'), textEncoder.encode(context.userId))], - ['profileId', equal(fields.get('profileId'), textEncoder.encode(context.profileId))], + ['protocol', equalBytes(fields.get('protocol'), encodeText(HOST_PROOF_TRANSCRIPT_DOMAIN))], + ['version', equalBytes(fields.get('version'), new Uint8Array([1]))], + ['relayOrigin', equalBytes(fields.get('relayOrigin'), encodeText(context.relayOrigin))], + ['relayEphemeralPublicKey', equalBytes(fields.get('relayEphemeralPublicKey'), relayKey)], + ['challengeNonce', equalBytes(fields.get('challengeNonce'), nonce)], + ['challengeId', equalBytes(fields.get('challengeId'), encodeText(challenge.challengeId))], + ['userId', equalBytes(fields.get('userId'), encodeText(context.userId))], + ['profileId', equalBytes(fields.get('profileId'), encodeText(context.profileId))], [ 'organizationId', - equal(fields.get('organizationId'), textEncoder.encode(context.organizationId)) + equalBytes(fields.get('organizationId'), encodeText(context.organizationId)) ], - ['relayHostId', equal(fields.get('relayHostId'), textEncoder.encode(context.relayHostId))], - ['hostPublicKey', equal(fields.get('hostPublicKey'), context.hostPublicKey)], - ['assignmentEpoch', equal(fields.get('assignmentEpoch'), uint64(context.assignmentEpoch))], - ['previousGeneration', equal(previousGeneration, expectedPrevious)], + ['relayHostId', equalBytes(fields.get('relayHostId'), encodeText(context.relayHostId))], + ['hostPublicKey', equalBytes(fields.get('hostPublicKey'), context.hostPublicKey)], + [ + 'assignmentEpoch', + equalBytes(fields.get('assignmentEpoch'), encodeUint64(context.assignmentEpoch)) + ], + ['previousGeneration', equalBytes(previousGeneration, expectedPrevious)], [ 'resumeRequested', - equal(fields.get('resumeRequested'), new Uint8Array([context.resumeRequested ? 1 : 0])) + equalBytes(fields.get('resumeRequested'), new Uint8Array([context.resumeRequested ? 1 : 0])) ] ] const failed = checks.filter(([, ok]) => !ok).map(([name]) => name) @@ -157,41 +112,29 @@ export function answerRelayHostChallenge( challenge: RelayHostChallenge, context: RelayHostProofContext ): string | null { - const relayKey = decodeCanonicalBase64(challenge.relayEphemeralPublicKeyB64, 32) - const nonce = decodeCanonicalBase64(challenge.nonceB64, 24) - const ciphertext = Buffer.from(challenge.ciphertextB64, 'base64') - if (!relayKey || !nonce || ciphertext.toString('base64') !== challenge.ciphertextB64) { - return null - } - const plaintext = nacl.box.open(ciphertext, nonce, relayKey, context.hostSecretKey) - if (!plaintext) { - context.onInvalid?.('challenge-box-open') - return null - } - const domain = textEncoder.encode(`${HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`) + const envelope = openHostChallengeEnvelope({ + peerEphemeralPublicKeyB64: challenge.relayEphemeralPublicKeyB64, + nonceB64: challenge.nonceB64, + ciphertextB64: challenge.ciphertextB64, + hostSecretKey: context.hostSecretKey, + plaintextDomain: HOST_CHALLENGE_PLAINTEXT_DOMAIN, + onInvalid: context.onInvalid + }) if ( - !equal(plaintext.slice(0, domain.byteLength), domain) || - plaintext.byteLength < domain.byteLength + 36 + !envelope || + !validateTranscript( + envelope.transcript, + challenge, + context, + envelope.peerEphemeralPublicKey, + envelope.nonce + ) ) { return null } - const transcriptLength = new DataView( - plaintext.buffer, - plaintext.byteOffset + domain.byteLength, - 4 - ).getUint32(0, false) - const transcriptStart = domain.byteLength + 4 - const secretStart = transcriptStart + transcriptLength - if (secretStart + 32 !== plaintext.byteLength) { - return null - } - const transcript = plaintext.slice(transcriptStart, secretStart) - if (!validateTranscript(transcript, challenge, context, relayKey, nonce)) { - return null - } - const secret = plaintext.slice(secretStart) - return createHmac('sha256', secret) - .update(textEncoder.encode(`${HOST_PROOF_TRANSCRIPT_DOMAIN}\0ack\0`)) - .update(transcript) - .digest('base64') + return hostChallengeAckProof({ + secret: envelope.secret, + transcript: envelope.transcript, + proofDomain: HOST_PROOF_TRANSCRIPT_DOMAIN + }) } diff --git a/src/main/runtime/remote-runtime-close-intent.integration.test.ts b/src/main/runtime/remote-runtime-close-intent.integration.test.ts index fec4a67388a..321d466e3f2 100644 --- a/src/main/runtime/remote-runtime-close-intent.integration.test.ts +++ b/src/main/runtime/remote-runtime-close-intent.integration.test.ts @@ -44,6 +44,7 @@ it( ] }) const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'close-intent-runtime-test', getStartedAt: () => 1, cleanupSubscriptionsForConnection: () => {}, diff --git a/src/main/runtime/remote-runtime-request-connection.integration.test.ts b/src/main/runtime/remote-runtime-request-connection.integration.test.ts index 7429b2aaa88..32dcb5d350e 100644 --- a/src/main/runtime/remote-runtime-request-connection.integration.test.ts +++ b/src/main/runtime/remote-runtime-request-connection.integration.test.ts @@ -44,6 +44,7 @@ describe('remote runtime request connection integration', () => { } ] const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'fetch-runtime-test', getStartedAt: () => 1, cleanupSubscriptionsForConnection: () => {}, @@ -115,6 +116,7 @@ describe('remote runtime request connection integration', () => { const clientEventListeners = new Set<(event: RuntimeClientEvent) => void>() const subscriptionCleanups = new Map void>() const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'events-runtime-test', getStartedAt: () => 1, cleanupSubscriptionsForConnection: (connectionId: string) => { @@ -280,6 +282,7 @@ describe('remote runtime request connection integration', () => { } } const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'remote-sleep-runtime-test', getStartedAt: () => 1, cleanupSubscriptionsForConnection: (connectionId: string) => { @@ -506,6 +509,7 @@ describe('remote runtime request connection integration', () => { tabs: [] } const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'shared-runtime-test', getStartedAt: () => 1, getStatus: () => ({ diff --git a/src/main/runtime/rpc/core-typed-method-contract.test.ts b/src/main/runtime/rpc/core-typed-method-contract.test.ts new file mode 100644 index 00000000000..fc1cdd1f97f --- /dev/null +++ b/src/main/runtime/rpc/core-typed-method-contract.test.ts @@ -0,0 +1,114 @@ +// The preserved types are the whole point of defineMethod, so they are asserted here: if a name +// widens to `string` or a result to `unknown`, these assertions fail at typecheck, not at runtime. +import { describe, expect, expectTypeOf, it } from 'vitest' +import { z } from 'zod' +import { + buildRegistry, + defineMethod, + defineStreamingMethod, + eraseRpcMethods, + isStreamingMethod, + type RpcContext, + type RpcMethod, + type RpcStreamingMethod +} from './core' +import type { ALL_RPC_METHODS } from './methods' +import { STATUS_METHODS } from './methods/status' +import type { HOST_CAPABILITY_METHODS } from './methods/host-capabilities' + +const ProbeParams = z.object({ id: z.string(), count: z.number().optional() }) + +const probe = defineMethod({ + name: 'test.typedProbe', + params: ProbeParams, + handler: (params) => ({ id: params.id, count: params.count ?? 0 }) +}) + +const schemalessProbe = defineMethod({ + name: 'test.schemalessProbe', + params: null, + handler: () => ['a', 'b'] +}) + +const streamingProbe = defineStreamingMethod({ + name: 'test.streamingProbe', + params: ProbeParams, + handler: async (params, _ctx, emit) => { + emit(params.id) + } +}) + +type ByName = Extract + +describe('defineMethod preserves the declared contract', () => { + it('keeps the literal method name', () => { + expectTypeOf(probe.name).toEqualTypeOf<'test.typedProbe'>() + expectTypeOf(streamingProbe.name).toEqualTypeOf<'test.streamingProbe'>() + expect(probe.name).toBe('test.typedProbe') + }) + + it('keeps the producer result type', () => { + expectTypeOf(probe.handler).returns.toEqualTypeOf<{ id: string; count: number }>() + expectTypeOf(schemalessProbe.handler).returns.toEqualTypeOf() + }) + + it('infers parsed params from the schema, and `void` without one', () => { + expectTypeOf(probe.handler) + .parameter(0) + .toEqualTypeOf<{ id: string; count?: number | undefined }>() + expectTypeOf(schemalessProbe.handler).parameter(0).toEqualTypeOf() + expectTypeOf(streamingProbe.handler) + .parameter(0) + .toEqualTypeOf<{ id: string; count?: number | undefined }>() + expectTypeOf(probe.params).toEqualTypeOf() + }) + + it('keeps a registered method addressable by its literal name', () => { + type StatusGet = ByName<(typeof STATUS_METHODS)[number], 'status.get'> + type ListDistros = ByName<(typeof HOST_CAPABILITY_METHODS)[number], 'host.wsl.listDistros'> + expectTypeOf().not.toBeNever() + expectTypeOf().returns.toExtend<{ runtimeId: string }>() + expectTypeOf().returns.toEqualTypeOf>() + // The manifest is the erasure boundary's input, so the literal names have to survive it too. + expectTypeOf>().not.toBeNever() + }) +}) + +describe('eraseRpcMethods is the registry boundary', () => { + it('erases to the shape the dispatcher calls, keeping the streaming split', () => { + expectTypeOf(eraseRpcMethods([probe])).toEqualTypeOf() + expectTypeOf(eraseRpcMethods([streamingProbe])).toEqualTypeOf() + expectTypeOf(eraseRpcMethods(STATUS_METHODS)).toEqualTypeOf() + expectTypeOf(eraseRpcMethods([probe])[0]!.handler) + .parameter(0) + .toEqualTypeOf() + }) + + it('returns the same methods, so nothing about the runtime value changes', () => { + const erased = eraseRpcMethods([probe, streamingProbe]) + + expect(erased[0]).toBe(probe) + expect(erased[1]).toBe(streamingProbe) + }) + + it('produces methods the registry accepts and the dispatcher can invoke', async () => { + const registry = buildRegistry([probe, streamingProbe, ...STATUS_METHODS]) + const registered = registry.get('test.typedProbe') + + expect(registered).toBe(probe) + expect(registry.get('status.get')).toBe(STATUS_METHODS[0]) + expect(isStreamingMethod(registry.get('test.streamingProbe')!)).toBe(true) + expect(registered && isStreamingMethod(registered)).toBe(false) + // The dispatcher parses params itself and then calls the erased handler with `unknown`. + const parsed: unknown = probe.params.parse({ id: 'a' }) + expect( + registered && !isStreamingMethod(registered) + ? await registered.handler(parsed, {} as RpcContext) + : undefined + ).toEqual({ id: 'a', count: 0 }) + }) + + it('rejects a duplicate name before erasure hides it', () => { + expect(() => buildRegistry([probe, probe])).toThrow('duplicate_rpc_method:test.typedProbe') + }) +}) diff --git a/src/main/runtime/rpc/core.ts b/src/main/runtime/rpc/core.ts index 702ea1b3aaa..c59c8da64d3 100644 --- a/src/main/runtime/rpc/core.ts +++ b/src/main/runtime/rpc/core.ts @@ -119,28 +119,27 @@ export type RpcContext = { ) => () => void } -export type RpcHandler = (params: TParams, ctx: RpcContext) => unknown +export type RpcHandler = (params: TParams, ctx: RpcContext) => TResult -// Why: RpcMethod erases the param type; centralizing the cast in defineMethod sidesteps RpcHandler's contravariance. -export type RpcMethod = { - readonly name: string - readonly params: ZodType | null - readonly handler: (params: unknown, ctx: RpcContext) => unknown +// Why: a schema-less method takes no params, so its handler must not be able to read the first argument. +type RpcParsedParams = TSchema extends ZodType + ? TSchema['_output'] + : void + +// Why: the authored shape — literal name, params schema, and producer result all survive for compile-time contracts. +export type RpcTypedMethod = { + readonly name: TName + readonly params: TSchema + readonly handler: RpcHandler, TResult> } -type DefineMethodSpec = { - name: string - params: TSchema - handler: RpcHandler -} - -export function defineMethod( - spec: DefineMethodSpec -): RpcMethod { +export function defineMethod( + spec: RpcTypedMethod +): RpcTypedMethod { return { name: spec.name, params: spec.params, - handler: spec.handler as RpcMethod['handler'] + handler: spec.handler } } @@ -150,6 +149,53 @@ export type RpcStreamingHandler = ( emit: (result: unknown) => void ) => Promise +// Why: emitted values stay `unknown` — the emit callback is an input, so there is no return position to infer them from. +export type RpcTypedStreamingMethod = { + readonly name: TName + readonly params: TSchema + readonly stream: true + readonly handler: RpcStreamingHandler> +} + +export function defineStreamingMethod( + spec: Omit, 'stream'> +): RpcTypedStreamingMethod { + return { + name: spec.name, + params: spec.params, + stream: true, + handler: spec.handler + } +} + +// Why `never` params: it makes the declaration a supertype of every parsed-params handler, so typed methods +// travel to the registry boundary — and only there get erased — without a cast in each methods module. +export type RpcMethodDeclaration = { + readonly name: string + readonly params: ZodType | null + readonly handler: (params: never, ctx: RpcContext) => unknown +} + +export type RpcStreamingMethodDeclaration = { + readonly name: string + readonly params: ZodType | null + readonly stream: true + readonly handler: ( + params: never, + ctx: RpcContext, + emit: (result: unknown) => void + ) => Promise +} + +export type RpcAnyMethodDeclaration = RpcMethodDeclaration | RpcStreamingMethodDeclaration + +// Why: RpcMethod is the registry's erased view; the dispatcher parses params itself and hands handlers `unknown`. +export type RpcMethod = { + readonly name: string + readonly params: ZodType | null + readonly handler: (params: unknown, ctx: RpcContext) => unknown +} + // Why: the `stream` flag lets the dispatcher route these to the emit-based path instead of the one-shot Promise path. export type RpcStreamingMethod = { readonly name: string @@ -162,34 +208,33 @@ export type RpcStreamingMethod = { ) => Promise } -type DefineStreamingMethodSpec = { - name: string - params: TSchema - handler: RpcStreamingHandler -} - -export function defineStreamingMethod( - spec: DefineStreamingMethodSpec -): RpcStreamingMethod { - return { - name: spec.name, - params: spec.params, - stream: true, - handler: spec.handler as RpcStreamingMethod['handler'] - } -} - export type RpcAnyMethod = RpcMethod | RpcStreamingMethod +// Why the overloads: erasure drops the parsed-params type, not the one-shot/streaming split the dispatcher routes on. +export function eraseRpcMethods(methods: readonly RpcMethodDeclaration[]): readonly RpcMethod[] +export function eraseRpcMethods( + methods: readonly RpcStreamingMethodDeclaration[] +): readonly RpcStreamingMethod[] +export function eraseRpcMethods( + methods: readonly RpcAnyMethodDeclaration[] +): readonly RpcAnyMethod[] +// Why: the one place the parsed-params type is dropped — contravariance makes it uncastable by assignment, and +// the dispatcher only ever calls a handler with an already-parsed `unknown`. Runtime value is untouched. +export function eraseRpcMethods( + methods: readonly RpcAnyMethodDeclaration[] +): readonly RpcAnyMethod[] { + return methods as readonly RpcAnyMethod[] +} + export function isStreamingMethod(method: RpcAnyMethod): method is RpcStreamingMethod { return 'stream' in method && method.stream === true } export type RpcRegistry = ReadonlyMap -export function buildRegistry(methods: readonly RpcAnyMethod[]): RpcRegistry { +export function buildRegistry(methods: readonly RpcAnyMethodDeclaration[]): RpcRegistry { const registry = new Map() - for (const method of methods) { + for (const method of eraseRpcMethods(methods)) { if (registry.has(method.name)) { throw new Error(`duplicate_rpc_method:${method.name}`) } diff --git a/src/main/runtime/rpc/dispatcher-request-parsing.ts b/src/main/runtime/rpc/dispatcher-request-parsing.ts index da4ec510e75..a4ada6df7c4 100644 --- a/src/main/runtime/rpc/dispatcher-request-parsing.ts +++ b/src/main/runtime/rpc/dispatcher-request-parsing.ts @@ -1,7 +1,7 @@ import { compile, type ZodType } from 'zod' import { formatZodError, - type RpcAnyMethod, + type RpcAnyMethodDeclaration, type RpcEnvelopeMeta, type RpcRequest, type RpcResponse @@ -12,7 +12,7 @@ const compiledParams = new WeakMap() export function parseRpcRequestParams( request: RpcRequest, - method: RpcAnyMethod, + method: RpcAnyMethodDeclaration, meta: RpcEnvelopeMeta ): { value: unknown; error?: undefined } | { value?: undefined; error: RpcResponse } { if (method.params === null) { diff --git a/src/main/runtime/rpc/dispatcher.ts b/src/main/runtime/rpc/dispatcher.ts index 73cfa596dd5..2c407207197 100644 --- a/src/main/runtime/rpc/dispatcher.ts +++ b/src/main/runtime/rpc/dispatcher.ts @@ -1,7 +1,7 @@ import { buildRegistry, isStreamingMethod, - type RpcAnyMethod, + type RpcAnyMethodDeclaration, type RpcEnvelopeMeta, type RpcRegistry, type RpcRequest, @@ -24,7 +24,10 @@ import { parseRpcRequestParams } from './dispatcher-request-parsing' import { RpcStreamingDispatcher } from './rpc-streaming-dispatcher' import { invokeDispatcherUnaryMethod } from './dispatcher-unary-method-invocation' -export type DispatcherOptions = { runtime: OrcaRuntimeService; methods?: readonly RpcAnyMethod[] } +export type DispatcherOptions = { + runtime: OrcaRuntimeService + methods?: readonly RpcAnyMethodDeclaration[] +} type DispatchCallOptions = RpcDispatchStreamingOptions diff --git a/src/main/runtime/rpc/methods/accounts.test.ts b/src/main/runtime/rpc/methods/accounts.test.ts index dc09f93fc22..ac8948422ac 100644 --- a/src/main/runtime/rpc/methods/accounts.test.ts +++ b/src/main/runtime/rpc/methods/accounts.test.ts @@ -2,11 +2,11 @@ import { describe, expect, it, vi } from 'vitest' import { tmpdir } from 'node:os' import { join } from 'node:path' import type { OrcaRuntimeService } from '../../orca-runtime' -import { isStreamingMethod } from '../core' +import { eraseRpcMethods, isStreamingMethod } from '../core' import { ACCOUNT_METHODS } from './accounts' function method(name: string) { - const found = ACCOUNT_METHODS.find((candidate) => candidate.name === name) + const found = eraseRpcMethods(ACCOUNT_METHODS).find((candidate) => candidate.name === name) if (!found) { throw new Error(`Missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/accounts.ts b/src/main/runtime/rpc/methods/accounts.ts index 41d82965d00..f7fe0af90ec 100644 --- a/src/main/runtime/rpc/methods/accounts.ts +++ b/src/main/runtime/rpc/methods/accounts.ts @@ -1,5 +1,14 @@ -import { z } from 'zod' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' +import { + AccountsUnsubscribeParams, + AddClaudeFromConfigDirParams, + AddCodexFromHomeParams, + ConsumeCodexResetCreditParams, + ListAccountsParams, + RemoveAccountParams, + SelectAccountParams, + SelectCodexAccountForTargetParams +} from '../../../../shared/rpc-contract/accounts-params' // Why: monotonically increasing per-process counter avoids the Date.now() // collision that fired when two near-simultaneous accounts.subscribe calls @@ -7,86 +16,6 @@ import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' // registerSubscriptionCleanup's existing-key eviction path. let accountsSubscriptionSeq = 0 -const CodexResetTarget = z.discriminatedUnion('runtime', [ - z.object({ runtime: z.literal('host'), wslDistro: z.null() }).strict(), - // Why: reset scope must identify one exact WSL distro; null means all slots only for selection. - z.object({ runtime: z.literal('wsl'), wslDistro: z.string().trim().min(1).max(255) }).strict() -]) - -const CodexSelectionTarget = z.discriminatedUnion('runtime', [ - z.object({ runtime: z.literal('host'), wslDistro: z.null() }).strict(), - z - .object({ - runtime: z.literal('wsl'), - // A null distro intentionally means all WSL selection slots. - wslDistro: z.string().trim().min(1).max(255).nullable() - }) - .strict() -]) - -const SelectAccountParams = z.object({ - accountId: z - .union([z.string().min(1, 'Missing accountId'), z.null()]) - .transform((v) => (v === null ? null : v)) -}) - -const SelectCodexAccountForTargetParams = SelectAccountParams.extend({ - target: CodexSelectionTarget -}) - -const RemoveAccountParams = z.object({ - accountId: z.string().min(1, 'Missing accountId') -}) - -const CodexResetExpectedScope = z - .object({ - target: CodexResetTarget, - accountId: z.string().min(1, 'Missing accountId').max(512), - accountRevision: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), - offerRevision: z.string().startsWith('v1:', 'Invalid offerRevision').max(4_096) - }) - .strict() - -const ConsumeCodexResetCreditParams = z - .object({ - // Why: the phone owns the logical attempt key so a lost response can be - // retried without spending a finite earned credit twice. - idempotencyKey: z.uuid('Invalid idempotencyKey'), - expectedScope: CodexResetExpectedScope - }) - .strict() - -const AddClaudeFromConfigDirParams = z.object({ - configDir: z.string().min(1, 'Missing configDir'), - runtime: z.enum(['host', 'wsl']).optional(), - wslDistro: z.string().nullish(), - previousLegacyCredentialsSha256: z - .string() - .regex(/^[a-f0-9]{64}$/, 'Invalid legacy credential digest') - .nullable() - .optional() -}) - -const AddCodexFromHomeParams = z.object({ - sourceHome: z.string().min(1, 'Missing sourceHome'), - runtime: z.enum(['host', 'wsl']).optional(), - wslDistro: z.string().nullish() -}) - -// Why: `orca account list` prints only emails and the active ids, so it opts out -// of the forced all-provider usage refresh below — that lane bypasses the poll -// throttle and Retry-After gate and costs one serial round-trip per account. -const ListAccountsParams = z.object({ - refreshUsage: z.boolean().default(true) -}) - -const AccountsUnsubscribeParams = z.object({ - subscriptionId: z - .unknown() - .transform((value) => (typeof value === 'string' && value.length > 0 ? value : '')) - .pipe(z.string().min(1, 'Missing subscriptionId')) -}) - // Why: bridges the desktop ClaudeAccountService / CodexAccountService / // RateLimitService into the WebSocket / local-socket RPC. Read + switch + // remove for all clients; interactive add/re-auth flows spawn `claude login` @@ -95,7 +24,7 @@ const AccountsUnsubscribeParams = z.object({ // captures an already-authenticated CLAUDE_CONFIG_DIR (no PTY) so the local // `orca account add` CLI can register accounts on a headless host; it is gated // to the local runtime connection, never a mobile device token. See #1438. -export const ACCOUNT_METHODS: readonly RpcAnyMethod[] = [ +export const ACCOUNT_METHODS = [ defineMethod({ name: 'accounts.list', params: ListAccountsParams, diff --git a/src/main/runtime/rpc/methods/agent-hooks.test.ts b/src/main/runtime/rpc/methods/agent-hooks.test.ts index 5781045a1f7..f78e7709d6e 100644 --- a/src/main/runtime/rpc/methods/agent-hooks.test.ts +++ b/src/main/runtime/rpc/methods/agent-hooks.test.ts @@ -1,6 +1,6 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import type { OrcaRuntimeService } from '../../orca-runtime' -import { isStreamingMethod, type RpcContext } from '../core' +import { eraseRpcMethods, isStreamingMethod, type RpcContext } from '../core' const { installForRuntimeHomeSerializedMock, realpathMock } = vi.hoisted(() => ({ installForRuntimeHomeSerializedMock: vi.fn(), @@ -23,7 +23,7 @@ const RUNTIME_HOME = '\\\\wsl.localhost\\Ubuntu-24.04\\home\\jin\\.local\\share\\orca\\codex-runtime-home\\home' function prepareMethod() { - const method = AGENT_HOOK_METHODS.find( + const method = eraseRpcMethods(AGENT_HOOK_METHODS).find( (candidate) => candidate.name === 'agentHooks.prepareCodexForWslPane' ) if (!method || isStreamingMethod(method)) { diff --git a/src/main/runtime/rpc/methods/agent-hooks.ts b/src/main/runtime/rpc/methods/agent-hooks.ts index b26b117bd32..4d9f4a7706c 100644 --- a/src/main/runtime/rpc/methods/agent-hooks.ts +++ b/src/main/runtime/rpc/methods/agent-hooks.ts @@ -1,21 +1,8 @@ -import { z } from 'zod' import { prepareManagedWslCodexHomeBeforeShellLaunch } from '../../../codex/managed-wsl-home-shell-preflight' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' +import { PrepareCodexForWslPaneParams } from '../../../../shared/rpc-contract/agent-hooks-params' -const PrepareCodexForWslPaneParams = z - .object({ - codexHome: z.string().max(4_096), - orcaCodexHome: z.string().max(4_096), - wslDistro: z - .string() - .trim() - .min(1) - .max(255) - .regex(/^[^\\/\r\n]+$/) - }) - .strict() - -export const AGENT_HOOK_METHODS: readonly RpcMethod[] = [ +export const AGENT_HOOK_METHODS = [ defineMethod({ name: 'agentHooks.prepareCodexForWslPane', params: PrepareCodexForWslPaneParams, diff --git a/src/main/runtime/rpc/methods/agent-session.ts b/src/main/runtime/rpc/methods/agent-session.ts index 08aaa91038e..84e008a7f55 100644 --- a/src/main/runtime/rpc/methods/agent-session.ts +++ b/src/main/runtime/rpc/methods/agent-session.ts @@ -1,9 +1,3 @@ -import { z } from 'zod' -import { - getAgentResumeArgv, - hasUnsafeProviderSessionIdChars, - RESUMABLE_TUI_AGENTS -} from '../../../../shared/agent-session-resume' import type { RuntimeAgentSessionRpcCaller, RuntimeCreateAgentSessionRequest, @@ -15,182 +9,13 @@ import { AGENT_SESSION_OPERATION_FUTURE_SKEW_MS, parseAgentSessionOperationTimestamp } from '../../../../shared/agent-session-host-authority' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' import type { OrcaRuntimeService } from '../../orca-runtime' -import { defineMethod, type RpcAnyMethod } from '../core' - -const MAX_WORKTREE_SELECTOR_LENGTH = 32_768 -const MAX_TRANSCRIPT_PATH_BYTES = 16 * 1024 -const MAX_PROMPT_BYTES = 256 * 1024 -const MAX_AGENT_ARGS_BYTES = 16 * 1024 -const MAX_LAUNCH_PREFERENCE_LENGTH = 512 - -const StrictNonEmptyString = (max: number, message: string) => - z - .string() - .min(1, message) - .max(max, message) - .refine((value) => value === value.trim(), `${message}; surrounding whitespace is invalid`) - -const WorktreeSelector = StrictNonEmptyString( - MAX_WORKTREE_SELECTOR_LENGTH, - 'Invalid worktree selector' -) - -const Presentation = z.enum(['background', 'focused']) - -const Placement = z - .object({ - tabId: z - .string() - .min(1) - .max(512) - .refine(isValidTerminalTabId, 'Invalid terminal tab ID') - .optional(), - leafId: z.string().min(1).max(128).optional() - }) - .strict() - .refine((value) => value.tabId !== undefined || value.leafId !== undefined, { - message: 'Placement must include a tab or leaf ID' - }) - -const LaunchPreferences = z - .object({ - model: StrictNonEmptyString( - MAX_LAUNCH_PREFERENCE_LENGTH, - 'Invalid model preference' - ).optional(), - effort: StrictNonEmptyString( - MAX_LAUNCH_PREFERENCE_LENGTH, - 'Invalid effort preference' - ).optional(), - mode: StrictNonEmptyString(MAX_LAUNCH_PREFERENCE_LENGTH, 'Invalid mode preference').optional() - }) - .strict() - -const PromptDelivery = z.enum(['auto-submit', 'draft']) - -const AgentArgs = z - .string() - .refine( - (value) => Buffer.byteLength(value, 'utf8') <= MAX_AGENT_ARGS_BYTES, - 'Agent arguments are too large' - ) - .nullable() - -const OmpResumeFilePath = z - .string() - .min(1) - .refine((value) => value === value.trim(), 'Invalid OMP resume path') - .refine( - (value) => - !hasUnsafeProviderSessionIdChars(value) && - Buffer.byteLength(value, 'utf8') <= MAX_TRANSCRIPT_PATH_BYTES, - 'Invalid OMP resume path' - ) - -const ProviderSession = z - .object({ - key: z.enum(['session_id', 'conversation_id']), - id: StrictNonEmptyString(512, 'Invalid provider session ID').refine( - (value) => !value.startsWith('-') && !hasUnsafeProviderSessionIdChars(value), - 'Invalid provider session ID' - ), - transcriptPath: z - .string() - .min(1) - .refine((value) => value === value.trim(), 'Invalid transcript path') - .refine( - (value) => - !hasUnsafeProviderSessionIdChars(value) && - Buffer.byteLength(value, 'utf8') <= MAX_TRANSCRIPT_PATH_BYTES, - 'Invalid transcript path' - ) - .optional() - }) - .strict() - -const AutomaticEnsure = z - .object({ - kind: z.literal('automatic'), - sleepingCheckpointId: z - .string() - .min(32) - .max(128) - .regex(/^[A-Za-z0-9_-]+$/), - presentation: Presentation.optional() - }) - .strict() - -const ExplicitEnsure = z - .object({ - kind: z.literal('explicit'), - worktree: WorktreeSelector, - agent: z.enum(RESUMABLE_TUI_AGENTS), - providerSession: ProviderSession, - ompResumeFilePath: OmpResumeFilePath.optional(), - agentArgs: AgentArgs.optional(), - launchPreferences: LaunchPreferences.optional(), - presentation: Presentation.optional(), - placement: Placement.optional() - }) - .strict() - .superRefine((value, context) => { - if (value.ompResumeFilePath !== undefined && value.agent !== 'omp') { - context.addIssue({ - code: z.ZodIssueCode.custom, - path: ['ompResumeFilePath'], - message: 'OMP resume path requires the OMP agent' - }) - } - if (getAgentResumeArgv(value.agent, value.providerSession, value.ompResumeFilePath) === null) { - context.addIssue({ - code: z.ZodIssueCode.custom, - path: ['providerSession'], - message: 'Provider session is not resumable for this agent' - }) - } - }) - -export const EnsureAgentSessionParams: z.ZodType = - z.discriminatedUnion('kind', [AutomaticEnsure, ExplicitEnsure]) - -export const CreateAgentSessionParams: z.ZodType = z - .object({ - clientOperationId: z - .string() - .refine( - (value) => parseAgentSessionOperationTimestamp(value) !== null, - 'Invalid agent operation ID' - ), - worktree: WorktreeSelector, - agent: z.string().refine(isTuiAgent, 'Unknown agent preset'), - prompt: z - .string() - .refine( - (value) => Buffer.byteLength(value, 'utf8') <= MAX_PROMPT_BYTES, - 'Prompt is too large' - ) - .optional(), - promptDelivery: PromptDelivery.optional(), - agentArgs: AgentArgs.optional(), - launchPreferences: LaunchPreferences.optional(), - startupCwd: z.string().min(1).max(MAX_WORKTREE_SELECTOR_LENGTH).optional(), - presentation: Presentation.optional(), - placement: Placement.optional(), - viewMode: z.enum(['terminal', 'chat']).optional() - }) - .strict() - .superRefine((value, context) => { - if (value.promptDelivery === 'draft' && !value.prompt?.trim()) { - context.addIssue({ - code: z.ZodIssueCode.custom, - path: ['prompt'], - message: 'Draft delivery requires a non-empty prompt' - }) - } - }) +import { defineMethod } from '../core' +import { + CreateAgentSessionParams, + EnsureAgentSessionParams +} from '../../../../shared/rpc-contract/agent-session-params' +export { CreateAgentSessionParams, EnsureAgentSessionParams } type AgentSessionRuntime = OrcaRuntimeService & { ensureAgentSession( @@ -233,7 +58,7 @@ function assertOperationTimestampWithinFutureSkew(clientOperationId: string): vo } } -export const AGENT_SESSION_METHODS: RpcAnyMethod[] = [ +export const AGENT_SESSION_METHODS = [ defineMethod({ name: 'terminal.ensureAgentSession', params: EnsureAgentSessionParams, diff --git a/src/main/runtime/rpc/methods/ai-vault.ts b/src/main/runtime/rpc/methods/ai-vault.ts index c689165924d..abc3aff2650 100644 --- a/src/main/runtime/rpc/methods/ai-vault.ts +++ b/src/main/runtime/rpc/methods/ai-vault.ts @@ -1,86 +1,21 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalBoolean } from '../schemas' +import { defineMethod } from '../core' import { restampAiVaultListResult } from '../../../ai-vault/session-list-results' -import { AI_VAULT_AGENTS, AI_VAULT_SCOPE_PATHS_MAX_COUNT } from '../../../../shared/ai-vault-types' -import { AI_VAULT_SESSION_TITLE_REQUEST_MAX_COUNT } from '../../../../shared/ai-vault-session-title' import type { AiVaultPrepareSessionResumeArgs } from '../../../../shared/ai-vault-resume-preparation' -import { LOCAL_EXECUTION_HOST_ID, parseExecutionHostId } from '../../../../shared/execution-host' +import { LOCAL_EXECUTION_HOST_ID } from '../../../../shared/execution-host' import { describeAiVaultScanError } from '../../../../shared/ai-vault-scan-error-message' import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { assertLegacyAiVaultResumeAllowed, projectStructuredAiVaultSessions } from '../../../ai-vault/structured-session-ownership' +import { + AiVaultListSessionsParams, + AiVaultPrepareSessionResumeParams, + AiVaultSessionTitlesParams +} from '../../../../shared/rpc-contract/ai-vault-params' +export { AiVaultListSessionsParams, AiVaultPrepareSessionResumeParams, AiVaultSessionTitlesParams } -// Why: bound limit + scopePaths so a client cannot force an unbounded scan. -// Each scopePath is a host-local match prefix (validated/capped, never used for -// traversal); the count/length caps mirror the worktree-schemas bounding style. -const AI_VAULT_SCOPE_PATH_MAX_LENGTH = 4096 -const AI_VAULT_LIMIT_MAX = 2000 - -const executionHostIdSchema = z.string().transform((value, ctx): `runtime:${string}` => { - const parsed = parseExecutionHostId(value) - if (parsed?.kind === 'runtime') { - return parsed.id - } - ctx.addIssue({ - code: 'custom', - message: 'Invalid runtime execution host id' - }) - return z.NEVER -}) - -export const AiVaultListSessionsParams = z - .object({ - limit: z - .unknown() - .transform((value) => - typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined - ) - .pipe(z.union([z.number().int(), z.undefined()])) - .optional(), - unlimited: OptionalBoolean, - force: OptionalBoolean, - scopePaths: z - .array(z.string().min(1).max(AI_VAULT_SCOPE_PATH_MAX_LENGTH)) - // Why: clamp instead of reject — scope paths only ever widen discovery, and - // rejecting would hard-break older/uncapped producers (web client, pre-cap - // desktop parents) that send more than the bound. - .transform((paths) => paths.slice(0, AI_VAULT_SCOPE_PATHS_MAX_COUNT)) - .optional(), - // Why: desktop/web callers name the runtime host they are addressing; mobile - // omits it. The scan itself is host-local either way, so the id must never - // change what is scanned — it only restamps the shared cached result. - executionHostId: executionHostIdSchema.optional() - }) - .superRefine((params, ctx) => { - if (params.unlimited !== true && params.limit && params.limit > AI_VAULT_LIMIT_MAX) { - ctx.addIssue({ code: 'custom', path: ['limit'], message: 'Limit exceeds maximum' }) - } - }) - -export const AiVaultPrepareSessionResumeParams = z.object({ - agent: z.enum(AI_VAULT_AGENTS), - sessionId: z.string().min(1).max(512).optional(), - filePath: z.string().min(1).max(AI_VAULT_SCOPE_PATH_MAX_LENGTH), - codexHome: z.string().min(1).max(AI_VAULT_SCOPE_PATH_MAX_LENGTH).nullable(), - executionHostId: z.string().optional() -}) - -export const AiVaultSessionTitlesParams = z.object({ - requests: z - .array( - z.object({ - agent: z.enum(['claude', 'codex']), - sessionId: z.string().min(1).max(512), - transcriptPath: z.string().min(1).max(32_768).optional() - }) - ) - .max(AI_VAULT_SESSION_TITLE_REQUEST_MAX_COUNT) -}) - -export const AI_VAULT_METHODS: RpcMethod[] = [ +export const AI_VAULT_METHODS = [ defineMethod({ name: 'aiVault.resolveSessionTitles', params: AiVaultSessionTitlesParams, diff --git a/src/main/runtime/rpc/methods/artifacts.ts b/src/main/runtime/rpc/methods/artifacts.ts index 4b5d7b5ab3a..b6b91af8b0b 100644 --- a/src/main/runtime/rpc/methods/artifacts.ts +++ b/src/main/runtime/rpc/methods/artifacts.ts @@ -1,47 +1,12 @@ -import { z } from 'zod' +import { defineMethod } from '../core' import { - ARTIFACT_MAX_CONTENT_BYTES, - ARTIFACT_MAX_REQUEST_BYTES, - artifactContentByteLength, - artifactWriteRequestByteLength -} from '../../../../shared/artifacts' -import { defineMethod, type RpcAnyMethod } from '../core' + ArtifactsDeleteParams, + ListOptions, + SourceRequest, + WriteRequest +} from '../../../../shared/rpc-contract/artifacts-params' -const CloudOptions = { - apiUrl: z.string().max(2_048).optional(), - authToken: z.string().max(16_384).optional() -} - -const ListOptions = z.object({ - ...CloudOptions, - cursor: z.string().min(1).max(2_048).optional() -}) - -const SourceRequest = z.object({ - sourceKey: z.string().min(1).max(32_768), - ...CloudOptions -}) - -const WriteRequest = z - .object({ - sourceKey: z.string().min(1).max(32_768), - content: z - .string() - .min(1) - .max(ARTIFACT_MAX_CONTENT_BYTES) - .refine((content) => artifactContentByteLength(content) <= ARTIFACT_MAX_CONTENT_BYTES, { - message: 'Artifact content exceeds the 10 MiB limit.' - }), - contentType: z.enum(['text/html', 'text/markdown']), - fileName: z.string().min(1).max(512), - title: z.string().max(512).optional(), - ...CloudOptions - }) - .refine((request) => artifactWriteRequestByteLength(request) <= ARTIFACT_MAX_REQUEST_BYTES, { - message: 'Artifact request exceeds the supported size.' - }) - -export const ARTIFACT_METHODS: readonly RpcAnyMethod[] = [ +export const ARTIFACT_METHODS = [ defineMethod({ name: 'artifacts.list', params: ListOptions, @@ -74,7 +39,7 @@ export const ARTIFACT_METHODS: readonly RpcAnyMethod[] = [ }), defineMethod({ name: 'artifacts.delete', - params: z.object({ id: z.string().min(1), ...CloudOptions }), + params: ArtifactsDeleteParams, handler: (params, { runtime }) => runtime.deleteArtifact(params.id, params) }) ] diff --git a/src/main/runtime/rpc/methods/automation-schemas.ts b/src/main/runtime/rpc/methods/automation-schemas.ts index f2c829c1a9d..23962b84ccc 100644 --- a/src/main/runtime/rpc/methods/automation-schemas.ts +++ b/src/main/runtime/rpc/methods/automation-schemas.ts @@ -1,193 +1,10 @@ // Why: the automation method table stays readable only if its field-level validation lives beside it rather than inside it. -import { z } from 'zod' -import { isValidAutomationSchedule } from '../../../../shared/automation-schedule-parsing' -import { - MAX_AUTOMATION_PRECHECK_TIMEOUT_SECONDS, - normalizeAutomationPrecheckTimeoutSeconds -} from '../../../../shared/automation-precheck' -import { normalizeExecutionHostId } from '../../../../shared/execution-host' -import type { TaskProviderIdentity as SharedTaskProviderIdentity } from '../../../../shared/task-source-context' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import { - OptionalBoolean, - OptionalPlainString, - OptionalPositiveInt, - OptionalString, - requiredNumber, - requiredString -} from '../schemas' - -const TuiAgent = requiredString('Missing provider').refine(isTuiAgent, { - message: 'Unknown provider' -}) - -const AutomationWorkspaceMode = z.enum(['existing', 'new_per_run']).optional() -const SetupDecision = z.enum(['inherit', 'run', 'skip']).optional() -const ExecutionHostId = requiredString('Missing host id').transform((value, ctx) => { - const hostId = normalizeExecutionHostId(value) - if (!hostId) { - ctx.addIssue({ code: 'custom', message: 'Invalid host id' }) - return z.NEVER - } - return hostId -}) - -const AutomationSchedule = requiredString('Missing trigger').refine(isValidAutomationSchedule, { - message: 'Invalid automation trigger' -}) - -const AutomationPrecheck = z - .object({ - command: requiredString('Missing precheck command'), - timeoutSeconds: OptionalPositiveInt.transform((value) => - normalizeAutomationPrecheckTimeoutSeconds(value) - ).refine((value) => value <= MAX_AUTOMATION_PRECHECK_TIMEOUT_SECONDS, { - message: 'Precheck timeout is too large' - }) - }) - .nullable() - .optional() - -const OptionalNullablePlainString = z - .unknown() - .transform((value) => (value === null || typeof value === 'string' ? value : undefined)) - .pipe(z.union([z.string(), z.null(), z.undefined()])) - .optional() - -const TaskProviderIdentity = z - .custom( - (value) => - value !== null && - typeof value === 'object' && - 'provider' in value && - ['github', 'gitlab', 'linear', 'jira'].includes(String(value.provider)) - ) - .optional() - .nullable() - -const TaskSourceContext = z - .object({ - kind: z.literal('task-source'), - provider: z.enum(['github', 'gitlab', 'linear', 'jira']), - projectId: requiredString('Missing source project id'), - hostId: ExecutionHostId, - projectHostSetupId: OptionalNullablePlainString, - repoId: OptionalNullablePlainString, - providerIdentity: TaskProviderIdentity, - accountLabel: OptionalNullablePlainString - }) - .optional() - .nullable() - -const WorkspaceRunContext = z - .object({ - kind: z.literal('workspace-run'), - projectId: requiredString('Missing run project id'), - hostId: ExecutionHostId, - projectHostSetupId: requiredString('Missing project host setup id'), - repoId: requiredString('Missing repo id'), - path: requiredString('Missing run path') - }) - .optional() - .nullable() - -const SshTargetGeneration = requiredNumber('Missing SSH target generation').refine( - (value) => Number.isSafeInteger(value) && value >= 1, - { message: 'Invalid SSH target generation' } -) - -const OwnedSshSelector = z.object({ - kind: z.literal('ssh'), - targetId: requiredString('Missing SSH target id'), - targetGeneration: SshTargetGeneration -}) - -/** Orphan is accepted here, unlike a destination: a record with no executable host is still deletable. */ -const OwnerPreconditionSelector = z.discriminatedUnion('kind', [ - z.object({ kind: z.literal('self') }), - OwnedSshSelector, - z.object({ kind: z.literal('orphan') }) -]) - -const DestinationSelector = z.discriminatedUnion('kind', [ - z.object({ kind: z.literal('self') }), - OwnedSshSelector -]) - -export const ExpectedOwner = z.object({ selector: OwnerPreconditionSelector }).optional() -export const Destination = z.object({ selector: DestinationSelector }).optional() - -const ListScopeSelector = z.discriminatedUnion('kind', [ - z.object({ kind: z.literal('self') }), - z.object({ - kind: z.literal('ssh'), - targetId: requiredString('Missing SSH target id'), - expectedTargetGeneration: SshTargetGeneration - }), - z.object({ kind: z.literal('orphan') }) -]) - -/** An omitted selector is the legacy request; old clients keep the authority's complete list. */ -export const AutomationList = z.object({ selector: ListScopeSelector.optional() }) - -export const AutomationId = z.object({ - id: requiredString('Missing automation id'), - expectedOwner: ExpectedOwner -}) - -export const AutomationRuns = z.object({ - automationId: OptionalString, - expectedOwner: ExpectedOwner, - limit: OptionalPositiveInt, - cursor: OptionalString -}) - -export const AutomationCreate = z.object({ - creationKey: OptionalString, - name: requiredString('Missing automation name'), - prompt: requiredString('Missing automation prompt'), - precheck: AutomationPrecheck, - agentId: TuiAgent, - runContext: WorkspaceRunContext, - sourceContext: TaskSourceContext, - repo: OptionalString, - workspace: OptionalString, - workspaceMode: AutomationWorkspaceMode, - baseBranch: OptionalPlainString, - setupDecision: SetupDecision, - reuseSession: OptionalBoolean, - timezone: OptionalString, - rrule: AutomationSchedule, - dtstart: requiredNumber('Missing trigger start time'), - enabled: OptionalBoolean, - missedRunGraceMinutes: OptionalPositiveInt, - destination: Destination -}) - -const AutomationUpdateFields = z.object({ - name: OptionalString, - prompt: OptionalString, - precheck: AutomationPrecheck, - agentId: TuiAgent.optional(), - runContext: WorkspaceRunContext, - sourceContext: TaskSourceContext, - repo: OptionalString, - workspace: OptionalString, - workspaceMode: AutomationWorkspaceMode, - // Why: update patches distinguish omitted from null so callers can clear a saved base branch. - baseBranch: OptionalNullablePlainString, - setupDecision: SetupDecision, - reuseSession: OptionalBoolean, - timezone: OptionalString, - rrule: AutomationSchedule.optional(), - dtstart: requiredNumber('Missing trigger start time').optional(), - enabled: OptionalBoolean, - missedRunGraceMinutes: OptionalPositiveInt -}) - -export const AutomationUpdate = z.object({ - id: requiredString('Missing automation id'), - updates: AutomationUpdateFields, - expectedOwner: ExpectedOwner, - destination: Destination -}) +export { + AutomationCreate, + AutomationId, + AutomationList, + AutomationRuns, + AutomationUpdate, + Destination, + ExpectedOwner +} from '../../../../shared/rpc-contract/automation-params' diff --git a/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts b/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts index f1d2081dc6b..6af7ba467e9 100644 --- a/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts +++ b/src/main/runtime/rpc/methods/automation-scoped-list-methods.test.ts @@ -4,14 +4,14 @@ * current callers also receive owner metadata. */ import { describe, expect, it, vi } from 'vitest' -import type { RpcContext, RpcRequest } from '../core' +import { eraseRpcMethods, type RpcContext, type RpcRequest } from '../core' import { RpcDispatcher } from '../dispatcher' import type { OrcaRuntimeService } from '../../orca-runtime' import { AUTOMATION_METHODS } from './automations' import { AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' function method(name: string) { - const found = AUTOMATION_METHODS.find((entry) => entry.name === name) + const found = eraseRpcMethods(AUTOMATION_METHODS).find((entry) => entry.name === name) if (!found?.params) { throw new Error(`missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/automations.ts b/src/main/runtime/rpc/methods/automations.ts index ff5daca315c..b1e3eebe6eb 100644 --- a/src/main/runtime/rpc/methods/automations.ts +++ b/src/main/runtime/rpc/methods/automations.ts @@ -1,6 +1,6 @@ import { AUTOMATION_OWNER_FENCING_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { AutomationOwnerPrecondition } from '../../../../shared/automation-owner-precondition' -import { defineMethod, type RpcContext, type RpcMethod } from '../core' +import { defineMethod, type RpcContext } from '../core' import { AutomationCreate, AutomationId, @@ -25,7 +25,7 @@ function mutationOwner( return context.runtime.automationOwnerPrecondition(id) ?? undefined } -export const AUTOMATION_METHODS: RpcMethod[] = [ +export const AUTOMATION_METHODS = [ defineMethod({ name: 'automation.list', params: AutomationList, diff --git a/src/main/runtime/rpc/methods/browser-client-file-channel.ts b/src/main/runtime/rpc/methods/browser-client-file-channel.ts index b2020488af2..362a1990531 100644 --- a/src/main/runtime/rpc/methods/browser-client-file-channel.ts +++ b/src/main/runtime/rpc/methods/browser-client-file-channel.ts @@ -7,7 +7,7 @@ import { BROWSER_CLIENT_HOST_RUNTIME_CAPABILITY } from '../../../../shared/proto import { getBrowserClientDownloadTransferStore } from '../../browser-client-download-transfer-store' import { getBrowserHostLeaseRegistry } from '../../browser-host-lease-registry-instance' import { getRuntimeBrowserPageRegistry } from '../../runtime-browser-page-registry' -import { defineMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, type RpcContext } from '../core' type FileChannelAuthorityParams = { browserHostClientId: string @@ -64,7 +64,7 @@ function requireFileChannelPage( return page } -export const BROWSER_CLIENT_FILE_CHANNEL_METHODS: RpcAnyMethod[] = [ +export const BROWSER_CLIENT_FILE_CHANNEL_METHODS = [ defineMethod({ name: 'browser.clientHost.fileChannel.read', params: BrowserClientFileChannelReadParams, diff --git a/src/main/runtime/rpc/methods/browser-client-host.ts b/src/main/runtime/rpc/methods/browser-client-host.ts index 525fd4fde96..e06632f7959 100644 --- a/src/main/runtime/rpc/methods/browser-client-host.ts +++ b/src/main/runtime/rpc/methods/browser-client-host.ts @@ -14,9 +14,9 @@ import { getRuntimeBrowserPageRegistry } from '../../runtime-browser-page-regist import { adoptRuntimeBrowserClientPagesFromInventory } from '../../runtime-browser-client-page-adoption' import { recoverUnavailableRuntimeBrowserClientPages } from '../../runtime-browser-client-page-recovery' import { releaseRuntimeBrowserClientPageRecord } from '../../runtime-browser-client-page-release' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' -export const BROWSER_CLIENT_HOST_METHODS: RpcAnyMethod[] = [ +export const BROWSER_CLIENT_HOST_METHODS = [ defineStreamingMethod({ name: 'browser.clientHost.attach', params: BrowserClientHostAttachParams, diff --git a/src/main/runtime/rpc/methods/browser-core.ts b/src/main/runtime/rpc/methods/browser-core.ts index 5e5fba227b6..780f40fb373 100644 --- a/src/main/runtime/rpc/methods/browser-core.ts +++ b/src/main/runtime/rpc/methods/browser-core.ts @@ -1,5 +1,5 @@ -import { defineMethod, type RpcMethod } from '../core' -import { BrowserTarget, requiredString } from '../schemas' +import { defineMethod } from '../core' +import { BrowserTarget } from '../schemas' import { Check, Drag, @@ -33,12 +33,9 @@ import { } from './browser-schemas' import { BrowserOpenUrlParams, BrowserTabCreateParams } from './browser-tab-create-schema' import { BROWSER_TEXT_METHODS } from './browser-text-rpc-methods' +import { CertificateProceed } from '../../../../shared/rpc-contract/browser-core-params' -const CertificateProceed = BrowserTarget.extend({ - challengeId: requiredString('Missing required challengeId') -}) - -export const BROWSER_CORE_METHODS: RpcMethod[] = [ +export const BROWSER_CORE_METHODS = [ defineMethod({ name: 'browser.snapshot', params: BrowserTarget, diff --git a/src/main/runtime/rpc/methods/browser-extras.ts b/src/main/runtime/rpc/methods/browser-extras.ts index 692c5e19b7f..000838c4938 100644 --- a/src/main/runtime/rpc/methods/browser-extras.ts +++ b/src/main/runtime/rpc/methods/browser-extras.ts @@ -1,7 +1,6 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { assertRpcClipboardTextWriteWithinLimit } from '../rpc-clipboard-text-validation' -import { BrowserTarget, OptionalFiniteNumber } from '../schemas' +import { BrowserTarget } from '../schemas' import { ClipboardWrite, CookieDelete, @@ -22,19 +21,9 @@ import { StorageKeyValue, Viewport } from './browser-schemas' +import { MouseClick } from '../../../../shared/rpc-contract/browser-extras-params' -const MouseModifiers = z - .unknown() - .transform((v) => (Array.isArray(v) ? v : undefined)) - .pipe(z.union([z.array(z.enum(['cmd', 'ctrl', 'alt', 'shift'])), z.undefined()])) - .optional() - -const MouseClick = MouseXY.merge(MouseButton).extend({ - radius: OptionalFiniteNumber, - modifiers: MouseModifiers -}) - -export const BROWSER_EXTRA_METHODS: RpcMethod[] = [ +export const BROWSER_EXTRA_METHODS = [ defineMethod({ name: 'browser.cookie.get', params: CookieGet, diff --git a/src/main/runtime/rpc/methods/browser-network-tunnel.ts b/src/main/runtime/rpc/methods/browser-network-tunnel.ts index 405419deef0..d610824b1f2 100644 --- a/src/main/runtime/rpc/methods/browser-network-tunnel.ts +++ b/src/main/runtime/rpc/methods/browser-network-tunnel.ts @@ -12,14 +12,14 @@ import { BROWSER_NETWORK_TUNNEL_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { getBrowserHostLeaseRegistry } from '../../browser-host-lease-registry-instance' -import { defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineStreamingMethod } from '../core' const outboundMemoryBudgets = new BrowserNetworkTunnelOutboundMemoryBudgetRegistry() export function createBrowserNetworkTunnelMethods( memoryBudgets: BrowserNetworkTunnelOutboundMemoryBudgetRegistry = outboundMemoryBudgets, resolveExecutionRoute: BrowserNetworkExecutionRouteResolver = resolveBrowserNetworkExecutionRoute -): RpcAnyMethod[] { +) { return [ defineStreamingMethod({ name: 'network.browserTunnel', diff --git a/src/main/runtime/rpc/methods/browser-schemas.ts b/src/main/runtime/rpc/methods/browser-schemas.ts index 03872b61cc0..034fee898ad 100644 --- a/src/main/runtime/rpc/methods/browser-schemas.ts +++ b/src/main/runtime/rpc/methods/browser-schemas.ts @@ -1,356 +1,55 @@ // Why: browser schemas stay separate from handler registration so both sides // remain under the line cap and dispatch wiring stays scannable. -import { z } from 'zod' -import { - BrowserTarget, - OptionalBoolean, - OptionalFiniteNumber, - OptionalPlainString, - OptionalString, - requiredStringAllowingEmpty, - requiredString -} from '../schemas' - -export const Element = BrowserTarget.extend({ - element: requiredString('Missing required --element') -}) - -export const Goto = BrowserTarget.extend({ - url: requiredString('Missing required --url') -}) - -export const Fill = BrowserTarget.extend({ - element: requiredString('Missing required --element'), - value: requiredStringAllowingEmpty('Missing required --value') -}) - -export const Type = BrowserTarget.extend({ - input: requiredString('Missing required --input') -}) - -export const Select = BrowserTarget.extend({ - element: requiredString('Missing required --element'), - value: z.custom((v) => typeof v === 'string', { - message: 'Missing required --value' - }) -}) - -export const Scroll = BrowserTarget.extend({ - direction: z.custom<'up' | 'down'>((v) => v === 'up' || v === 'down', { - message: 'Missing required --direction (up or down)' - }), - amount: z - .unknown() - .transform((v) => (typeof v === 'number' && v > 0 ? v : undefined)) - .pipe(z.union([z.number(), z.undefined()])) - .optional() -}) - -export const Screenshot = BrowserTarget.extend({ - format: z - .unknown() - .transform((v) => (v === 'png' || v === 'jpeg' ? v : undefined)) - .pipe(z.union([z.enum(['png', 'jpeg']), z.undefined()])) - .optional() -}) - -export const Screencast = BrowserTarget.extend({ - format: z - .unknown() - .optional() - .transform((v) => (v === 'png' ? 'png' : 'jpeg')) - .pipe(z.enum(['png', 'jpeg'])), - quality: OptionalFiniteNumber, - maxWidth: OptionalFiniteNumber, - maxHeight: OptionalFiniteNumber, - viewportWidth: OptionalFiniteNumber, - viewportHeight: OptionalFiniteNumber, - deviceScaleFactor: OptionalFiniteNumber, - mobile: OptionalBoolean, - everyNthFrame: OptionalFiniteNumber, - minFrameIntervalMs: OptionalFiniteNumber -}) - -export const FullScreenshot = BrowserTarget.extend({ - format: z - .unknown() - .optional() - .transform((v) => (v === 'jpeg' ? 'jpeg' : 'png')) - .pipe(z.enum(['png', 'jpeg'])) -}) - -export const Eval = BrowserTarget.extend({ - expression: requiredString('Missing required --expression') -}) - -export const TabList = z.object({ worktree: OptionalString }) -// Why: --index xor --page must be present. The refine guards that invariant -// so the dispatcher surfaces a single legible error instead of either shape -// leaking into the runtime. -// -// `focus` is opt-in: when true, the runtime sends `browser:pane-focus` to -// the renderer after the switch lands. The renderer surfaces the browser -// pane only if the user is already on the targeted worktree; otherwise it -// pre-stages per-worktree state silently. This avoids cross-worktree screen -// theft when multiple agents drive browsers in parallel worktrees. -export const TabSwitch = BrowserTarget.extend({ - index: z - .unknown() - .transform((v) => (typeof v === 'number' ? v : undefined)) - .pipe(z.union([z.number(), z.undefined()])) - .optional(), - focus: z.boolean().optional() -}).refine( - (val) => { - if (val.page !== undefined) { - return true - } - return val.index !== undefined && Number.isInteger(val.index) && val.index >= 0 - }, - { message: 'Missing required --index (non-negative integer) or --page' } -) - -export const TabShow = z.object({ - page: requiredString('Missing required --page'), - worktree: OptionalString -}) - -export const TabCurrent = z.object({ worktree: OptionalString }) - -export const TabClose = z.object({ - index: z - .unknown() - .transform((v) => (typeof v === 'number' ? v : undefined)) - .pipe(z.union([z.number(), z.undefined()])) - .optional(), - page: OptionalString, - worktree: OptionalString -}) - -export const TabSetProfile = BrowserTarget.extend({ - profileId: requiredString('Missing required --profile') -}) - -export const TabProfileClone = BrowserTarget.extend({ - profileId: requiredString('Missing required --profile') -}) - -export const ProfileCreate = z.object({ - label: requiredString('Missing required --label'), - // Strict enum so unknown scope values surface validation errors instead of being - // silently coerced to 'isolated' (pr-bug-scan finding from #1397). - scope: z.enum(['isolated', 'imported']), - userAgentMode: z.enum(['clean', 'native']).optional() -}) - -export const ProfileDelete = z.object({ profileId: requiredString('Missing required --profile') }) - -export const ProfileImportFromBrowser = z.object({ - profileId: requiredString('Missing required --profile'), - browserFamily: requiredString('Missing required --browser-family'), - browserProfile: OptionalString, - supportsPartitionSkippedCookies: z.literal(true).optional() -}) - -export const Drag = BrowserTarget.extend({ - from: requiredString('Missing required --from and --to element refs'), - to: requiredString('Missing required --from and --to element refs') -}) - -export const Upload = BrowserTarget.extend({ - element: requiredString('Missing required --element and --files'), - files: z.custom( - (v) => Array.isArray(v) && v.length > 0 && v.every((f) => typeof f === 'string'), - { message: 'Missing required --element and --files' } - ) -}) - -export const Wait = BrowserTarget.extend({ - selector: OptionalPlainString, - timeout: z - .unknown() - .transform((v) => (typeof v === 'number' && v > 0 ? v : undefined)) - .pipe(z.union([z.number(), z.undefined()])) - .optional(), - text: OptionalPlainString, - url: OptionalPlainString, - load: OptionalPlainString, - fn: OptionalPlainString, - state: OptionalPlainString -}) - -export const Check = BrowserTarget.extend({ - element: requiredString('Missing required --element'), - checked: z - .unknown() - .optional() - .transform((v) => (v === undefined ? true : v)) - .pipe(z.boolean()) -}) - -export const Keypress = BrowserTarget.extend({ - key: requiredString('Missing required --key') -}) - -export const SelectorPath = BrowserTarget.extend({ - selector: requiredString('Missing required --selector and --path'), - path: requiredString('Missing required --selector and --path') -}) - -export const Highlight = BrowserTarget.extend({ - selector: requiredString('Missing required --selector') -}) - -export const Exec = BrowserTarget.extend({ - command: requiredString('Missing required --command') -}) - -export const Get = BrowserTarget.extend({ - what: requiredString('Missing required --what'), - selector: OptionalString -}) - -export const Is = BrowserTarget.extend({ - what: z.custom((v) => typeof v === 'string' && v.length > 0, { - message: 'Missing required --what and --element' - }), - selector: z.custom((v) => typeof v === 'string' && v.length > 0, { - message: 'Missing required --what and --element' - }) -}) - -export const KeyboardInsert = BrowserTarget.extend({ - text: requiredString('Missing required --text') -}) - -export const LimitParam = BrowserTarget.extend({ - limit: OptionalFiniteNumber -}) - -export const Find = BrowserTarget.extend({ - locator: requiredString('Missing required --locator, --value, and --action'), - value: requiredString('Missing required --locator, --value, and --action'), - action: requiredString('Missing required --locator, --value, and --action'), - text: OptionalString -}) - -export const CookieGet = BrowserTarget.extend({ - url: OptionalPlainString -}) - -export const CookieSet = BrowserTarget.extend({ - name: z.custom((v) => typeof v === 'string' && v.length > 0, { - message: 'Missing name or value' - }), - value: z.custom((v) => typeof v === 'string', { - message: 'Missing name or value' - }), - domain: OptionalPlainString, - path: OptionalPlainString, - secure: OptionalBoolean, - httpOnly: OptionalBoolean, - sameSite: OptionalPlainString, - expires: OptionalFiniteNumber -}) - -export const CookieDelete = BrowserTarget.extend({ - name: requiredString('Missing cookie name'), - domain: OptionalPlainString, - url: OptionalPlainString -}) - -export const Viewport = BrowserTarget.extend({ - width: z.custom((v) => typeof v === 'number' && v > 0, { - message: 'Width and height must be positive numbers' - }), - height: z.custom((v) => typeof v === 'number' && v > 0, { - message: 'Width and height must be positive numbers' - }), - deviceScaleFactor: OptionalFiniteNumber, - mobile: OptionalBoolean -}) - -export const Geolocation = BrowserTarget.extend({ - latitude: z.custom((v) => typeof v === 'number', { - message: 'Missing latitude or longitude' - }), - longitude: z.custom((v) => typeof v === 'number', { - message: 'Missing latitude or longitude' - }), - accuracy: OptionalFiniteNumber -}) - -export const InterceptEnable = BrowserTarget.extend({ - patterns: z - .unknown() - .transform((v) => (Array.isArray(v) ? (v as string[]) : undefined)) - .pipe(z.union([z.array(z.string()), z.undefined()])) - .optional() -}) - -export const MouseXY = BrowserTarget.extend({ - x: z.custom((v) => typeof v === 'number', { - message: 'Missing required x and y coordinates' - }), - y: z.custom((v) => typeof v === 'number', { - message: 'Missing required x and y coordinates' - }) -}) - -export const MouseButton = BrowserTarget.extend({ - button: OptionalPlainString -}) - -export const MouseWheel = BrowserTarget.extend({ - dy: z.custom((v) => typeof v === 'number', { - message: 'Missing required --dy' - }), - dx: OptionalFiniteNumber -}) - -export const SetDevice = BrowserTarget.extend({ - name: requiredString('Missing required --name') -}) - -export const SetOffline = BrowserTarget.extend({ - state: OptionalPlainString -}) - -export const SetHeaders = BrowserTarget.extend({ - headers: requiredString('Missing required --headers (JSON string)') -}) - -export const SetCredentials = BrowserTarget.extend({ - user: z.custom((v) => typeof v === 'string' && v.length > 0, { - message: 'Missing required --user and --pass' - }), - pass: z.custom((v) => typeof v === 'string', { - message: 'Missing required --user and --pass' - }) -}) - -export const SetMedia = BrowserTarget.extend({ - colorScheme: OptionalPlainString, - reducedMotion: OptionalPlainString -}) - -export const ClipboardWrite = BrowserTarget.extend({ - text: requiredString('Missing required --text') -}) - -export const DialogAccept = BrowserTarget.extend({ - text: OptionalPlainString -}) - -export const StorageKey = BrowserTarget.extend({ - key: requiredString('Missing required --key') -}) - -export const StorageKeyValue = BrowserTarget.extend({ - key: z.custom((v) => typeof v === 'string' && v.length > 0, { - message: 'Missing required --key and --value' - }), - value: z.custom((v) => typeof v === 'string', { - message: 'Missing required --key and --value' - }) -}) +export { + Check, + ClipboardWrite, + CookieDelete, + CookieGet, + CookieSet, + DialogAccept, + Drag, + Element, + Eval, + Exec, + Fill, + Find, + FullScreenshot, + Geolocation, + Get, + Goto, + Highlight, + InterceptEnable, + Is, + KeyboardInsert, + Keypress, + LimitParam, + MouseButton, + MouseWheel, + MouseXY, + ProfileCreate, + ProfileDelete, + ProfileImportFromBrowser, + Screencast, + Screenshot, + Scroll, + Select, + SelectorPath, + SetCredentials, + SetDevice, + SetHeaders, + SetMedia, + SetOffline, + StorageKey, + StorageKeyValue, + TabClose, + TabCurrent, + TabList, + TabProfileClone, + TabSetProfile, + TabShow, + TabSwitch, + Type, + Upload, + Viewport, + Wait +} from '../../../../shared/rpc-contract/browser-params' diff --git a/src/main/runtime/rpc/methods/browser-screencast.ts b/src/main/runtime/rpc/methods/browser-screencast.ts index ea965a2b44a..798e1ed84d2 100644 --- a/src/main/runtime/rpc/methods/browser-screencast.ts +++ b/src/main/runtime/rpc/methods/browser-screencast.ts @@ -1,15 +1,11 @@ -import { z } from 'zod' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { Screencast } from './browser-schemas' import { BrowserError } from '../../../browser/browser-error' import { BROWSER_UNAVAILABLE_ERROR_CODE } from '../../../../shared/runtime-types' import { runtimeBrowserCommandsFactoryIsAvailable } from '../../runtime-browser-commands-factory' +import { ScreencastUnsubscribe } from '../../../../shared/rpc-contract/browser-screencast-params' -const ScreencastUnsubscribe = z.object({ - subscriptionId: z.string().min(1, 'Missing required --subscription-id') -}) - -export const BROWSER_SCREENCAST_METHODS: RpcAnyMethod[] = [ +export const BROWSER_SCREENCAST_METHODS = [ defineStreamingMethod({ name: 'browser.screencast', params: Screencast, diff --git a/src/main/runtime/rpc/methods/browser-tab-create-schema.ts b/src/main/runtime/rpc/methods/browser-tab-create-schema.ts index 799111b2a6b..ad7c0ee5b75 100644 --- a/src/main/runtime/rpc/methods/browser-tab-create-schema.ts +++ b/src/main/runtime/rpc/methods/browser-tab-create-schema.ts @@ -1,23 +1,4 @@ -import { z } from 'zod' -import { OptionalString } from '../schemas' -import { BrowserPageCreationPlacement } from '../../../../shared/browser-client-host-placement' -import { RUNTIME_NAVIGATION_TARGETS } from '../../../../shared/runtime-navigation' - -export const BrowserTabCreateParams = z.object({ - url: OptionalString, - worktree: OptionalString, - page: OptionalString, - profileId: OptionalString, - waitForRegistration: z.boolean().optional(), - activate: z.boolean().optional(), - // Why: `activate` says the caller wants the new tab selected; `navigation` says on whose screens. - // Absent, a paired caller means 'caller' — one device's create must not steer every other UI. - navigation: z.enum(RUNTIME_NAVIGATION_TARGETS).optional(), - targetGroupId: OptionalString, - placement: BrowserPageCreationPlacement.optional() -}) - -export const BrowserOpenUrlParams = z.object({ - url: z.url(), - worktree: z.string().min(1) -}) +export { + BrowserOpenUrlParams, + BrowserTabCreateParams +} from '../../../../shared/rpc-contract/browser-tab-create-params' diff --git a/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts b/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts index 813dbe9d2e2..18fd19a0e6a 100644 --- a/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts +++ b/src/main/runtime/rpc/methods/browser-text-rpc-methods.ts @@ -1,8 +1,8 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { assertRpcClipboardTextWriteWithinLimit } from '../rpc-clipboard-text-validation' import { Fill, KeyboardInsert, Type } from './browser-schemas' -export const BROWSER_TEXT_METHODS: RpcMethod[] = [ +export const BROWSER_TEXT_METHODS = [ defineMethod({ name: 'browser.fill', params: Fill, diff --git a/src/main/runtime/rpc/methods/client-events.test.ts b/src/main/runtime/rpc/methods/client-events.test.ts index bf5026c819e..cae682ea4b4 100644 --- a/src/main/runtime/rpc/methods/client-events.test.ts +++ b/src/main/runtime/rpc/methods/client-events.test.ts @@ -1,11 +1,16 @@ import { describe, expect, it, vi } from 'vitest' import type { RuntimeClientEvent } from '../../../../shared/runtime-client-events' import type { OrcaRuntimeService } from '../../orca-runtime' -import { isStreamingMethod, type RpcContext, type RpcStreamingMethod } from '../core' +import { + eraseRpcMethods, + isStreamingMethod, + type RpcContext, + type RpcStreamingMethod +} from '../core' // Why: importing client-events directly trips its module-init cycle through ipc/ssh; the index resolves it. import { ALL_RPC_METHODS } from './index' -const subscribeMethod = ALL_RPC_METHODS.find( +const subscribeMethod = eraseRpcMethods(ALL_RPC_METHODS).find( (method) => method.name === 'runtime.clientEvents.subscribe' && isStreamingMethod(method) ) as RpcStreamingMethod diff --git a/src/main/runtime/rpc/methods/client-events.ts b/src/main/runtime/rpc/methods/client-events.ts index 0c3a079262f..e7506ad2f59 100644 --- a/src/main/runtime/rpc/methods/client-events.ts +++ b/src/main/runtime/rpc/methods/client-events.ts @@ -1,18 +1,11 @@ -import { z } from 'zod' import { getRegisteredSshState, listRegisteredSshTargets } from '../../../ssh/ssh-target-registry' import { getPublicSshState } from '../../public-ssh-state' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' +import { ClientEventsUnsubscribeParams } from '../../../../shared/rpc-contract/client-events-params' let clientEventSubscriptionSeq = 0 -const ClientEventsUnsubscribeParams = z.object({ - subscriptionId: z - .unknown() - .transform((value) => (typeof value === 'string' && value.length > 0 ? value : '')) - .pipe(z.string().min(1, 'Missing subscriptionId')) -}) - -export const CLIENT_EVENT_METHODS: readonly RpcAnyMethod[] = [ +export const CLIENT_EVENT_METHODS = [ defineStreamingMethod({ name: 'runtime.clientEvents.subscribe', params: null, diff --git a/src/main/runtime/rpc/methods/client-settings-schemas.ts b/src/main/runtime/rpc/methods/client-settings-schemas.ts index e389ed9d12b..c8467636d9a 100644 --- a/src/main/runtime/rpc/methods/client-settings-schemas.ts +++ b/src/main/runtime/rpc/methods/client-settings-schemas.ts @@ -1,122 +1,5 @@ -import { z } from 'zod' -import { normalizePRBotAuthorOverrides } from '../../../../shared/pr-bot-author-overrides' -import { isTaskProvider } from '../../../../shared/task-providers' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import { - normalizeTuiAgentArgsRecord, - normalizeTuiAgentEnvRecord -} from '../../../../shared/tui-agent-launch-defaults' -import { normalizeDisabledTuiAgents } from '../../../../shared/tui-agent-selection' -import { WorktreeVisibilityDefaultsUpdate } from './worktree-visibility-defaults-schema' -import type { TaskProvider } from '../../../../shared/task-providers' - -const TaskProviderParam = z.custom(isTaskProvider, { - message: 'Unknown task provider' -}) - -export const PRBotAuthorOverrideUpdate = z - .object({ author: z.string(), isBot: z.boolean() }) - .strict() - -const NativeChatSessionOptionPickBase = { - modelId: z.string().trim().min(1).max(512), - adoptModelAsLaunchDefault: z.boolean().optional() -} - -const NativeChatSessionOptionPick = z.union([ - z - .object({ - ...NativeChatSessionOptionPickBase, - optionId: z.enum(['model', 'effort']), - value: z.string().trim().min(1).max(512) - }) - .strict(), - z - .object({ - ...NativeChatSessionOptionPickBase, - optionId: z.enum(['fastMode', 'thinking']), - value: z.boolean() - }) - .strict() -]) - -export const NativeChatSessionOptionsMutation = z.discriminatedUnion('type', [ - z - .object({ - type: z.literal('apply-picks'), - agent: z.enum(['claude', 'codex', 'gemini', 'cursor', 'grok']), - picks: z.array(NativeChatSessionOptionPick).min(1).max(8) - }) - .strict(), - z - .object({ - type: z.literal('clear-model-if-missing'), - agent: z.enum(['claude', 'codex', 'gemini', 'cursor', 'grok']), - availableModelIds: z.array(z.string().trim().min(1).max(512)).min(1).max(256) - }) - .strict() -]) - -const GitHubProjectRef = z - .object({ - owner: z.string(), - ownerType: z.enum(['organization', 'user']), - number: z.number().int(), - host: z.string().optional() - }) - .strict() -const GitHubProjectSettings = z - .object({ - pinned: z.array(GitHubProjectRef), - recent: z.array( - GitHubProjectRef.extend({ - lastOpenedAt: z.string() - }).strict() - ), - lastViewByProject: z.record(z.string(), z.object({ viewId: z.string() }).strict()), - activeProject: GitHubProjectRef.nullable() - }) - .strict() - -export const SettingsUpdate = z - .object({ - worktreeVisibilityDefaults: WorktreeVisibilityDefaultsUpdate.optional(), - defaultTuiAgent: z - .unknown() - .transform((value) => - value === null || value === 'blank' || isTuiAgent(value) ? value : undefined - ) - .optional(), - disabledTuiAgents: z - .unknown() - .transform((value) => normalizeDisabledTuiAgents(value)) - .optional(), - agentDefaultArgs: z - .unknown() - .transform((value) => normalizeTuiAgentArgsRecord(value)) - .optional(), - agentDefaultEnv: z - .unknown() - .transform((value) => normalizeTuiAgentEnvRecord(value)) - .optional(), - defaultTaskSource: TaskProviderParam.optional(), - visibleTaskProviders: z.array(TaskProviderParam).optional(), - defaultTaskViewPreset: z - .enum(['issues', 'my-issues', 'prs', 'my-prs', 'review', 'all']) - .optional(), - experimentalNewWorktreeCardStyle: z.boolean().optional(), - agentStatusHooksEnabled: z.boolean().optional(), - defaultRepoSelection: z.array(z.string()).nullable().optional(), - defaultLinearTeamSelection: z.array(z.string()).nullable().optional(), - compactWorktreeCards: z.boolean().optional(), - minimaxGroupId: z.string().optional(), - minimaxUsageModels: z.string().optional(), - minimaxEndpoint: z.enum(['overseas', 'cn']).optional(), - githubProjects: GitHubProjectSettings.optional(), - prBotAuthorOverrides: z - .unknown() - .transform((value) => normalizePRBotAuthorOverrides(value)) - .optional() - }) - .strict() - .default({}) +export { + NativeChatSessionOptionsMutation, + PRBotAuthorOverrideUpdate, + SettingsUpdate +} from '../../../../shared/rpc-contract/client-settings-params' diff --git a/src/main/runtime/rpc/methods/client-ui-schemas.ts b/src/main/runtime/rpc/methods/client-ui-schemas.ts index 943d0081fdf..f11bd7ca49c 100644 --- a/src/main/runtime/rpc/methods/client-ui-schemas.ts +++ b/src/main/runtime/rpc/methods/client-ui-schemas.ts @@ -1,244 +1,8 @@ -import { z } from 'zod' -import { - isFeatureInteractionId, - type FeatureInteractionId -} from '../../../../shared/feature-interactions' -import { - ACTIVITY_GROUP_BY_VALUES, - THREAD_READ_FILTER_VALUES -} from '../../../../shared/agents-view-thread-filters' -import { isFeatureTipId } from '../../../../shared/feature-tips' -import { isReleaseChannel, type ReleaseChannel } from '../../../../shared/release-channel' -import { - normalizeWorktreeCardProperties, - WORKTREE_CARD_PROPERTIES -} from '../../../../shared/worktree/card-properties' -import { isPluginPanelTabKey } from '../../../../shared/plugins/plugin-manifest' -import { ClientUiWorkspaceFilterFields } from './client-ui-workspace-filter-fields' -import { TaskResumeState } from './task-resume-state-schema' -import { WorkspaceCleanup } from './workspace-cleanup-ui-schema' -import { omitUndefinedValues, tolerateUnknownValues } from './ui-update-value-tolerance' - -const NullableString = z.string().nullable() -const StringArray = z.array(z.string()) -const FeatureTipIds = z.array(z.custom(isFeatureTipId, { message: 'Unknown feature tip id' })) -const UnknownRecord = z.record(z.string(), z.unknown()) -const UnknownRecordArray = z.array(UnknownRecord) -type StaticRightSidebarTab = (typeof STATIC_RIGHT_SIDEBAR_TABS)[number] -// Derived from the shared union so a new card property cannot drift out of the -// client schema — it previously omitted 'cli' and rejected the whole payload. -const WorktreeCardPropertyParam = z.enum(WORKTREE_CARD_PROPERTIES) -const WorktreeCardProperties = z - .array(WorktreeCardPropertyParam) - .transform((value) => normalizeWorktreeCardProperties(value)) -const STATIC_RIGHT_SIDEBAR_TABS = [ - 'explorer', - 'search', - 'vault', - 'workspaces', - 'pr-checks', - 'source-control', - 'checks', - 'ports' -] as const -// Plugin panels are open-ended `plugin:./` keys, so the -// schema validates their shape rather than enumerating them. -const RightSidebarTabParam = z.custom( - (value) => - typeof value === 'string' && - (STATIC_RIGHT_SIDEBAR_TABS.includes(value as StaticRightSidebarTab) || - isPluginPanelTabKey(value)), - { message: 'Unknown right sidebar tab' } -) -const AgentActivityDisplayMode = z.enum(['compact', 'full']) -const StatusBarItem = z.enum([ - 'claude', - 'codex', - 'gemini', - 'antigravity', - 'opencode-go', - 'kimi', - 'minimax', - 'grok', - 'ssh', - 'resource-usage', - 'ports' -]) -const WorkspaceStatusDefinition = z.object({ - id: z.string(), - label: z.string(), - color: z.string().optional(), - icon: z.string().optional() -}) -const FeatureInteractionRecord = z - .object({ - firstInteractedAt: z.number().finite().nonnegative(), - interactionCount: z.number().int().positive().optional() - }) - .strict() -const FeatureInteractions = z - .record(z.string(), FeatureInteractionRecord) - .superRefine((value, ctx) => { - for (const id of Object.keys(value)) { - if (!isFeatureInteractionId(id)) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: `Unknown feature interaction id: ${id}`, - path: [id] - }) - } - } - }) -export const FeatureInteractionIdParam = z.custom(isFeatureInteractionId, { - message: 'Unknown feature interaction id' -}) -const TopLevelViewSchema = z.enum([ - 'terminal', - 'settings', - 'tasks', - 'activity', - 'automations', - 'space', - 'skills', - 'artifacts', - 'mobile' -]) -const UiUpdateFields = z - .object({ - lastActiveRepoId: NullableString.optional(), - lastActiveWorktreeId: NullableString.optional(), - // Why: sync hydration ignores this persisted startup view, so paired windows stay put. - activeView: TopLevelViewSchema.optional(), - sidebarWidth: z.number().finite().optional(), - rightSidebarOpen: z.boolean().optional(), - rightSidebarTab: RightSidebarTabParam.optional(), - rightSidebarExplorerView: z.enum(['files', 'search']).optional(), - rightSidebarWidth: z.number().finite().optional(), - markdownTocPanelWidth: z.number().finite().optional(), - combinedDiffFileTreeWidth: z.number().finite().optional(), - groupBy: z.enum(['none', 'workspace-status', 'repo', 'pr-status']).optional(), - showWorkspaceLineage: z.boolean().optional(), - sortBy: z.enum(['name', 'smart', 'recent', 'repo', 'manual']).optional(), - projectOrderBy: z.enum(['manual', 'recent']).optional(), - showActiveOnly: z.boolean().optional(), - hideSleepingWorkspaces: z.boolean().optional(), - showSleepingWorkspaces: z.boolean().optional(), - showInactiveWorkspaces: z.boolean().optional(), - workspaceHostScope: z.string().optional(), - visibleWorkspaceHostIds: z.array(z.string()).nullable().optional(), - agentsVisibleHostIds: z.array(z.string()).nullable().optional(), - agentsFilterRepoIds: StringArray.optional(), - agentsShowChildAgents: z.boolean().optional(), - agentsCompactMode: z.boolean().optional(), - agentsShowSearch: z.boolean().optional(), - agentsReadFilter: z.enum(THREAD_READ_FILTER_VALUES).optional(), - agentsGroupBy: z.enum(ACTIVITY_GROUP_BY_VALUES).optional(), - workspaceHostOrder: z.array(z.string()).optional(), - automationHostFilter: z - .union([ - z.object({ kind: z.literal('all') }).strict(), - z.object({ kind: z.literal('host'), hostKey: z.string().min(1) }).strict() - ]) - .optional(), - manualRepoOrder: z - .array(z.object({ hostId: z.string(), repoId: z.string() }).strict()) - .optional(), - ...ClientUiWorkspaceFilterFields, - // Why: rides App.tsx's debounced writer, so omitting it rejected that entire - // payload (sidebar widths, filters, agent acks) for every paired client. - showDotfilesByWorktree: z.record(z.string(), z.boolean()).optional(), - collapsedGroups: StringArray.optional(), - uiZoomLevel: z.number().finite().optional(), - editorFontZoomLevel: z.number().finite().optional(), - worktreeCardProperties: WorktreeCardProperties.optional(), - _worktreeCardModeDefaulted: z.boolean().optional(), - agentActivityDisplayMode: AgentActivityDisplayMode.optional(), - workspaceStatuses: z.array(WorkspaceStatusDefinition).optional(), - workspaceBoardOpacity: z.number().finite().optional(), - workspaceBoardColumnWidth: z.number().finite().optional(), - syncTaskStatusFromWorkspaceBoard: z.boolean().optional(), - _workspaceStatusesDefaultOrderMigrated: z.boolean().optional(), - _workspaceStatusesReorderedDefaultRepaired: z.boolean().optional(), - _workspaceStatusesDefaultWorkflowMigrated: z.boolean().optional(), - _workspaceStatusesDefaultVisualsMigrated: z.boolean().optional(), - statusBarItems: z.array(StatusBarItem).optional(), - _portsStatusBarDefaultAdded: z.boolean().optional(), - _kimiStatusBarDefaultAdded: z.boolean().optional(), - _minimaxStatusBarDefaultAdded: z.boolean().optional(), - _antigravityStatusBarDefaultAdded: z.boolean().optional(), - _grokStatusBarDefaultAdded: z.boolean().optional(), - statusBarVisible: z.boolean().optional(), - usagePercentageDisplay: z.enum(['used', 'remaining']).optional(), - statusBarUsageMode: z.enum(['verbose', 'compact']).optional(), - dismissedUpdateVersion: NullableString.optional(), - lastUpdateCheckAt: z.number().finite().nullable().optional(), - pendingUpdateNudgeId: NullableString.optional(), - dismissedUpdateNudgeId: NullableString.optional(), - // Why the predicate rather than an inline z.enum: an enum here is a copy of - // RELEASE_CHANNELS, and a copy that drifts silently rejects the new - // channel's override on its way here — the picker moves, nothing installs. - releaseChannelOverride: z.custom(isReleaseChannel).nullable().optional(), - notificationPermissionRequested: z.boolean().optional(), - updateReassuranceSeen: z.boolean().optional(), - osc52ClipboardDefaultOnNoticePending: z.boolean().optional(), - acknowledgedAgentsByPaneKey: z.record(z.string(), z.number().finite()).optional(), - activityClearedAtByPaneKey: z.record(z.string(), z.number().finite()).optional(), - manuallyUnreadTurnsByPaneKey: z.record(z.string(), z.number().finite()).optional(), - browserDefaultUrl: NullableString.optional(), - browserDefaultSearchEngine: z - .enum(['google', 'duckduckgo', 'bing', 'kagi']) - .nullable() - .optional(), - browserDefaultZoomLevel: z.number().finite().optional(), - browserKagiSessionLink: NullableString.optional(), - windowBounds: z - .object({ - x: z.number().finite(), - y: z.number().finite(), - width: z.number().finite(), - height: z.number().finite() - }) - .nullable() - .optional(), - windowMaximized: z.boolean().optional(), - _sortBySmartMigrated: z.boolean().optional(), - _inlineAgentsDefaultedForExperiment: z.boolean().optional(), - _inlineAgentsDefaultedForAllUsers: z.boolean().optional(), - trustedOrcaHooks: z.record(z.string(), z.unknown()).optional(), - setupScriptPromptDismissedRepoIds: StringArray.optional(), - // Why: one-shot dismissals the renderer writes through ui.set; each was a - // whole-payload rejection for paired clients while unlisted. - setupGuideSidebarDismissed: z.boolean().optional(), - setupGuideBrowserMilestoneMigrated: z.boolean().optional(), - setupGuideBrowserMilestoneLegacyComplete: z.boolean().optional(), - browserImportHintHidden: z.boolean().optional(), - mobileEmulatorTabIntroDismissed: z.boolean().optional(), - mobileEmulatorAgentSetupDismissed: z.boolean().optional(), - projectOrderManualDefaultNoticeDismissed: z.boolean().optional(), - usagePercentageDisplayChangeNoticeDismissed: z.boolean().optional(), - usageEmptyStateDismissed: z.boolean().optional(), - petVisible: z.boolean().optional(), - petId: z.string().optional(), - customPets: UnknownRecordArray.optional(), - petSize: z.number().finite().optional(), - sidekickVisible: z.boolean().optional(), - sidekickId: z.string().optional(), - customSidekicks: UnknownRecordArray.optional(), - sidekickSize: z.number().finite().optional(), - taskResumeState: TaskResumeState.optional(), - workspaceCleanup: WorkspaceCleanup.optional(), - featureTipsSeenIds: FeatureTipIds.optional(), - featureInteractions: FeatureInteractions.optional(), - contextualToursSeenIds: StringArray.optional(), - contextualToursAutoEligible: z.boolean().optional() - }) - .strict() - -export const UiUpdate = z - .object(tolerateUnknownValues(UiUpdateFields.shape)) - .strict() - .default({}) - .transform(omitUndefinedValues) +import type { UiUpdateFields } from '../../../../shared/rpc-contract/client-ui-params' +export { + FeatureInteractionIdParam, + UiUpdate +} from '../../../../shared/rpc-contract/client-ui-params' // The key/value parity assertions over this live in ui-state-schema-parity-checks.ts. export type UiUpdateFieldsSchema = typeof UiUpdateFields diff --git a/src/main/runtime/rpc/methods/client-ui-workspace-filter-fields.ts b/src/main/runtime/rpc/methods/client-ui-workspace-filter-fields.ts index d0239b03ec3..be6bc445f6e 100644 --- a/src/main/runtime/rpc/methods/client-ui-workspace-filter-fields.ts +++ b/src/main/runtime/rpc/methods/client-ui-workspace-filter-fields.ts @@ -1,11 +1 @@ -import { z } from 'zod' - -export const ClientUiWorkspaceFilterFields = { - hideDefaultBranchWorkspace: z.boolean().optional(), - hideAutomationGeneratedWorkspaces: z.boolean().optional(), - hideCliCreatedWorkspaces: z.boolean().optional(), - hideDetachedHeadWorkspaces: z.boolean().optional(), - hideWorkspacesFromOtherDevices: z.boolean().optional(), - alwaysShowDefaultBranchWorkspace: z.boolean().optional(), - filterRepoIds: z.array(z.string()).optional() -} +export { ClientUiWorkspaceFilterFields } from '../../../../shared/rpc-contract/client-ui-workspace-filter-fields-params' diff --git a/src/main/runtime/rpc/methods/client-ui.test.ts b/src/main/runtime/rpc/methods/client-ui.test.ts index 39048611155..3a26b1c615d 100644 --- a/src/main/runtime/rpc/methods/client-ui.test.ts +++ b/src/main/runtime/rpc/methods/client-ui.test.ts @@ -607,6 +607,8 @@ describe('client UI RPC methods', () => { ], ['taskResumeState.jiraPreset', { taskResumeState: { jiraPreset: 'assigned' } }], ['taskResumeState.jiraQuery', { taskResumeState: { jiraQuery: 'ENG' } }], + ['dismissedUnexpectedSignoutVersion', { dismissedUnexpectedSignoutVersion: '1.2.3' }], + ['dismissedUnexpectedSignoutVersion null', { dismissedUnexpectedSignoutVersion: null }], ['activeView', { activeView: 'tasks' }], ['showDotfilesByWorktree', { showDotfilesByWorktree: { 'repo::/worktree': true } }], ['setupGuideSidebarDismissed', { setupGuideSidebarDismissed: true }], diff --git a/src/main/runtime/rpc/methods/client-ui.ts b/src/main/runtime/rpc/methods/client-ui.ts index ffd964b6be6..6ed36a6fe83 100644 --- a/src/main/runtime/rpc/methods/client-ui.ts +++ b/src/main/runtime/rpc/methods/client-ui.ts @@ -1,6 +1,6 @@ import { omitPairingLocalUiFields } from '../../../../shared/pairing-local-ui-fields' import type { PersistedUIState } from '../../../../shared/persisted-ui-state-types' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { NativeChatSessionOptionsMutation, PRBotAuthorOverrideUpdate, @@ -12,7 +12,7 @@ import { FeatureInteractionIdParam, UiUpdate } from './client-ui-schemas' import { TerminalQuickCommandsUpdate } from './terminal-quick-command-rpc-schema' -export const CLIENT_UI_METHODS: RpcMethod[] = [ +export const CLIENT_UI_METHODS = [ defineMethod({ name: 'settings.get', params: null, diff --git a/src/main/runtime/rpc/methods/clipboard.ts b/src/main/runtime/rpc/methods/clipboard.ts index e6b487d7761..3ec78265dc6 100644 --- a/src/main/runtime/rpc/methods/clipboard.ts +++ b/src/main/runtime/rpc/methods/clipboard.ts @@ -1,18 +1,18 @@ -import { z } from 'zod' -import { defineMethod, type RpcContext, type RpcMethod } from '../core' +import { defineMethod, type RpcContext } from '../core' import { saveClipboardImageBufferAsTempFile } from '../../../window/clipboard-image-temp-file' import { randomUUID } from 'node:crypto' -import { - CLIPBOARD_IMAGE_MAX_BASE64_CHARS, - CLIPBOARD_IMAGE_TOO_LARGE_ERROR -} from '../../../../shared/clipboard-image' import { recordMobileClipboardImagePath } from '../mobile-clipboard-image-provenance' - -const MAX_CLIPBOARD_IMAGE_BASE64_CHARS = CLIPBOARD_IMAGE_MAX_BASE64_CHARS -export const CLIPBOARD_IMAGE_UPLOAD_CHUNK_BASE64_CHARS = 512 * 1024 +import { + AbortImageUpload, + AppendImageUploadChunk, + CommitImageUpload, + SaveImageAsTempFile, + StartImageUpload, + isValidBase64 +} from '../../../../shared/rpc-contract/clipboard-params' +export { CLIPBOARD_IMAGE_UPLOAD_CHUNK_BASE64_CHARS } from '../../../../shared/rpc-contract/clipboard-params' export const CLIPBOARD_IMAGE_UPLOAD_MAX_CONCURRENT = 8 const CLIPBOARD_IMAGE_UPLOAD_TTL_MS = 5 * 60 * 1000 -const BASE64_PATTERN = /^[A-Za-z0-9+/]*={0,2}$/ type ClipboardImageUpload = { expectedBase64Length: number @@ -26,10 +26,6 @@ type ClipboardImageUpload = { const clipboardImageUploads = new Map() -function isValidBase64(value: string): boolean { - return value.length % 4 !== 1 && BASE64_PATTERN.test(value) -} - function pruneExpiredUploads(now = Date.now()): void { for (const [uploadId, upload] of clipboardImageUploads) { if (upload.expiresAt <= now) { @@ -99,59 +95,7 @@ function assertValidBase64Content(value: string): void { } } -function clipboardImageBase64Payload(maxChars: number, tooLargeMessage: string) { - return z.unknown().transform((value, ctx): string => { - if (typeof value !== 'string') { - ctx.addIssue({ code: 'custom', message: 'Missing image content' }) - return z.NEVER - } - if (value.length > maxChars) { - ctx.addIssue({ code: 'custom', message: tooLargeMessage }) - return z.NEVER - } - if (!isValidBase64(value)) { - ctx.addIssue({ code: 'custom', message: 'Clipboard image content must be base64' }) - return z.NEVER - } - return value - }) -} - -const SaveImageAsTempFile = z.object({ - contentBase64: clipboardImageBase64Payload( - MAX_CLIPBOARD_IMAGE_BASE64_CHARS, - CLIPBOARD_IMAGE_TOO_LARGE_ERROR - ), - connectionId: z.string().min(1).nullable().optional() -}) - -const StartImageUpload = z.object({ - expectedBase64Length: z - .number() - .int() - .nonnegative() - .max(MAX_CLIPBOARD_IMAGE_BASE64_CHARS, CLIPBOARD_IMAGE_TOO_LARGE_ERROR), - connectionId: z.string().min(1).nullable().optional() -}) - -const AppendImageUploadChunk = z.object({ - uploadId: z.string().min(1), - offset: z.number().int().nonnegative(), - contentBase64: clipboardImageBase64Payload( - CLIPBOARD_IMAGE_UPLOAD_CHUNK_BASE64_CHARS, - 'Clipboard image chunk is too large' - ) -}) - -const CommitImageUpload = z.object({ - uploadId: z.string().min(1) -}) - -const AbortImageUpload = z.object({ - uploadId: z.string().min(1) -}) - -export const CLIPBOARD_METHODS: RpcMethod[] = [ +export const CLIPBOARD_METHODS = [ defineMethod({ name: 'clipboard.saveImageAsTempFile', params: SaveImageAsTempFile, diff --git a/src/main/runtime/rpc/methods/computer-actions.test.ts b/src/main/runtime/rpc/methods/computer-actions.test.ts index 96dc374e459..fa2b7d83ee1 100644 --- a/src/main/runtime/rpc/methods/computer-actions.test.ts +++ b/src/main/runtime/rpc/methods/computer-actions.test.ts @@ -27,6 +27,7 @@ vi.mock('../../../computer/macos-computer-use-permissions', () => ({ })) import { COMPUTER_METHODS, resetComputerSessionsForTest } from './computer' +import { eraseRpcMethods } from '../core' describe('computer action RPC methods', () => { beforeEach(() => { @@ -269,7 +270,7 @@ describe('computer action RPC methods', () => { }) function findMethod(name: string) { - const method = COMPUTER_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(COMPUTER_METHODS).find((candidate) => candidate.name === name) if (!method) { throw new Error(`missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/computer-schemas.ts b/src/main/runtime/rpc/methods/computer-schemas.ts index e3ef7ca88f3..f949fa0459c 100644 --- a/src/main/runtime/rpc/methods/computer-schemas.ts +++ b/src/main/runtime/rpc/methods/computer-schemas.ts @@ -1,226 +1,15 @@ -import { z } from 'zod' -import { - computerUseClickModifiersValidationMessage, - computerUseHotkeyValidationMessage, - computerUsePressKeyValidationMessage -} from '../../../../shared/computer-use-key-spec' -import { - OptionalBoolean, - OptionalFiniteNumber, - OptionalString, - requiredString, - requiredStringAllowingEmpty -} from '../schemas' - -const OptionalNonNegativeInt = z.number().int().nonnegative().optional() -const OptionalPositiveInt = z.number().int().positive().optional() - -const ComputerTarget = z.object({ - app: requiredString('Missing app'), - session: OptionalString, - worktree: OptionalString -}) - -const ComputerObserveTargetBase = ComputerTarget.extend({ - noScreenshot: OptionalBoolean, - restoreWindow: OptionalBoolean, - windowId: OptionalNonNegativeInt, - windowIndex: OptionalNonNegativeInt -}) - -function validateWindowTarget( - value: { windowId?: number; windowIndex?: number }, - ctx: z.RefinementCtx -): void { - if (value.windowId !== undefined && value.windowIndex !== undefined) { - ctx.addIssue({ - code: 'custom', - message: 'Window targeting accepts either --window-id or --window-index, not both' - }) - } -} - -function validateComputerTarget( - value: { session?: string; worktree?: string; windowId?: number; windowIndex?: number }, - ctx: z.RefinementCtx -): void { - if (value.session !== undefined && value.worktree !== undefined) { - ctx.addIssue({ - code: 'custom', - message: 'Computer-use targeting accepts either session or worktree, not both' - }) - } - validateWindowTarget(value, ctx) -} - -export const ComputerObserveTarget = ComputerObserveTargetBase.superRefine(validateComputerTarget) - -export const ListApps = z.object({}).strict() - -export const ListWindows = z - .object({ - app: requiredString('Missing app') - }) - .strict() - -export const Click = ComputerObserveTargetBase.extend({ - elementIndex: OptionalNonNegativeInt, - x: OptionalFiniteNumber, - y: OptionalFiniteNumber, - clickCount: OptionalPositiveInt, - mouseButton: z.enum(['left', 'right', 'middle']).optional(), - modifiers: z.string().optional() -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - const hasElement = value.elementIndex !== undefined - const hasX = value.x !== undefined - const hasY = value.y !== undefined - if (!hasElement && !(hasX && hasY)) { - ctx.addIssue({ - code: 'custom', - message: 'Click requires --element-index or both --x and --y' - }) - } - if (hasX !== hasY) { - ctx.addIssue({ - code: 'custom', - message: 'Click coordinates require both --x and --y' - }) - } - if (hasElement && (hasX || hasY)) { - ctx.addIssue({ - code: 'custom', - message: 'Click accepts either --element-index or coordinate flags, not both' - }) - } - if (value.modifiers !== undefined) { - const message = computerUseClickModifiersValidationMessage(value.modifiers) - if (message) { - ctx.addIssue({ code: 'custom', message }) - } - } -}) - -export const PerformSecondaryAction = ComputerObserveTargetBase.extend({ - elementIndex: OptionalNonNegativeInt, - action: requiredString('Missing action') -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - if (value.elementIndex === undefined) { - ctx.addIssue({ code: 'custom', message: 'Missing element index' }) - } -}) - -export const Scroll = ComputerObserveTargetBase.extend({ - elementIndex: OptionalNonNegativeInt, - x: OptionalFiniteNumber, - y: OptionalFiniteNumber, - direction: z.enum(['up', 'down', 'left', 'right']), - pages: z.number().positive().optional() -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - const hasElement = value.elementIndex !== undefined - const hasX = value.x !== undefined - const hasY = value.y !== undefined - if (!hasElement && !(hasX && hasY)) { - ctx.addIssue({ - code: 'custom', - message: 'Scroll requires --element-index or both --x and --y' - }) - } - if (hasX !== hasY) { - ctx.addIssue({ - code: 'custom', - message: 'Scroll coordinates require both --x and --y' - }) - } - if (hasElement && (hasX || hasY)) { - ctx.addIssue({ - code: 'custom', - message: 'Scroll accepts either --element-index or coordinate flags, not both' - }) - } -}) - -export const Drag = ComputerObserveTargetBase.extend({ - fromElementIndex: OptionalNonNegativeInt, - toElementIndex: OptionalNonNegativeInt, - fromX: OptionalFiniteNumber, - fromY: OptionalFiniteNumber, - toX: OptionalFiniteNumber, - toY: OptionalFiniteNumber -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - const hasElementPair = value.fromElementIndex !== undefined && value.toElementIndex !== undefined - const hasPartialElementPair = - value.fromElementIndex !== undefined || value.toElementIndex !== undefined - const coordinateKeys = [value.fromX, value.fromY, value.toX, value.toY] - const hasCoordinatePair = coordinateKeys.every((coordinate) => coordinate !== undefined) - const hasPartialCoordinatePair = coordinateKeys.some((coordinate) => coordinate !== undefined) - if (hasElementPair && hasCoordinatePair) { - ctx.addIssue({ - code: 'custom', - message: 'Drag accepts either element indexes or coordinate flags, not both' - }) - } - if (!hasElementPair && !hasCoordinatePair) { - ctx.addIssue({ - code: 'custom', - message: 'Drag requires --from-element-index and --to-element-index, or all coordinate flags' - }) - } - if (hasPartialElementPair && !hasElementPair) { - ctx.addIssue({ - code: 'custom', - message: 'Drag element targeting requires both --from-element-index and --to-element-index' - }) - } - if (hasPartialCoordinatePair && !hasCoordinatePair) { - ctx.addIssue({ - code: 'custom', - message: 'Drag coordinates require --from-x, --from-y, --to-x, and --to-y' - }) - } -}) - -export const TypeText = ComputerObserveTargetBase.extend({ - text: requiredString('Missing text') -}).superRefine(validateComputerTarget) - -export const PressKey = ComputerObserveTargetBase.extend({ - key: requiredString('Missing key') -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - const message = computerUsePressKeyValidationMessage(value.key) - if (message) { - ctx.addIssue({ code: 'custom', message }) - } -}) - -export const Hotkey = ComputerObserveTargetBase.extend({ - key: requiredString('Missing key') -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - const message = computerUseHotkeyValidationMessage(value.key) - if (message) { - ctx.addIssue({ code: 'custom', message }) - } -}) - -export const ComputerPermissions = z.object({ - id: z.enum(['accessibility', 'screenshots']).optional() -}) - -export const PasteText = ComputerObserveTargetBase.extend({ - text: requiredString('Missing text') -}).superRefine(validateComputerTarget) - -export const SetValue = ComputerObserveTargetBase.extend({ - elementIndex: OptionalNonNegativeInt, - value: requiredStringAllowingEmpty('Missing value') -}).superRefine((value, ctx) => { - validateComputerTarget(value, ctx) - if (value.elementIndex === undefined) { - ctx.addIssue({ code: 'custom', message: 'Missing element index' }) - } -}) +export { + Click, + ComputerObserveTarget, + ComputerPermissions, + Drag, + Hotkey, + ListApps, + ListWindows, + PasteText, + PerformSecondaryAction, + PressKey, + Scroll, + SetValue, + TypeText +} from '../../../../shared/rpc-contract/computer-schemas-params' diff --git a/src/main/runtime/rpc/methods/computer.test.ts b/src/main/runtime/rpc/methods/computer.test.ts index 6a01fac7d0a..073a1c0a363 100644 --- a/src/main/runtime/rpc/methods/computer.test.ts +++ b/src/main/runtime/rpc/methods/computer.test.ts @@ -1,5 +1,5 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' -import { buildRegistry } from '../core' +import { eraseRpcMethods, buildRegistry } from '../core' import { CLIPBOARD_TEXT_WRITE_MAX_BYTES } from '../../../../shared/clipboard-text' const computerMocks = vi.hoisted(() => ({ @@ -249,7 +249,7 @@ describe('computer RPC methods', () => { }) function findMethod(name: string) { - const method = COMPUTER_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(COMPUTER_METHODS).find((candidate) => candidate.name === name) if (!method) { throw new Error(`missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/computer.ts b/src/main/runtime/rpc/methods/computer.ts index 1f928b97afe..e2708667633 100644 --- a/src/main/runtime/rpc/methods/computer.ts +++ b/src/main/runtime/rpc/methods/computer.ts @@ -1,4 +1,3 @@ -import { z } from 'zod' import { callComputerSidecarAction, callComputerSidecarCapabilities, @@ -7,7 +6,7 @@ import { callComputerSidecarSnapshot, resetComputerSidecarForTest } from '../../../computer/sidecar-client' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { Click, ComputerObserveTarget, @@ -23,15 +22,19 @@ import { SetValue, TypeText } from './computer-schemas' +import { + ComputerCapabilitiesParams, + ComputerPermissionsStatusParams +} from '../../../../shared/rpc-contract/computer-params' export function resetComputerSessionsForTest(): void { resetComputerSidecarForTest() } -export const COMPUTER_METHODS: RpcMethod[] = [ +export const COMPUTER_METHODS = [ defineMethod({ name: 'computer.capabilities', - params: z.object({}), + params: ComputerCapabilitiesParams, handler: async () => { return await callComputerSidecarCapabilities() } @@ -54,7 +57,7 @@ export const COMPUTER_METHODS: RpcMethod[] = [ }), defineMethod({ name: 'computer.permissionsStatus', - params: z.object({}), + params: ComputerPermissionsStatusParams, handler: async () => { const { getComputerUsePermissionStatus } = await import('../../../computer/macos-computer-use-permissions') diff --git a/src/main/runtime/rpc/methods/diagnostics.ts b/src/main/runtime/rpc/methods/diagnostics.ts index 4d158d98f63..953eccd5d51 100644 --- a/src/main/runtime/rpc/methods/diagnostics.ts +++ b/src/main/runtime/rpc/methods/diagnostics.ts @@ -1,6 +1,6 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' -export const DIAGNOSTICS_METHODS: RpcMethod[] = [ +export const DIAGNOSTICS_METHODS = [ defineMethod({ name: 'diagnostics.memory', params: null, diff --git a/src/main/runtime/rpc/methods/emulator.ts b/src/main/runtime/rpc/methods/emulator.ts index 50354f790e4..b472e539603 100644 --- a/src/main/runtime/rpc/methods/emulator.ts +++ b/src/main/runtime/rpc/methods/emulator.ts @@ -1,66 +1,26 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import path from 'node:path' import { z } from 'zod' - -// Minimal schemas for emulator commands (loose for initial testing; can be tightened like browser-schemas). -const WorktreeParam = z.object({ worktree: z.string().optional() }).partial() - -const TapParams = z.object({ - x: z.number().min(0).max(1), - y: z.number().min(0).max(1), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const GesturePoint = z.object({ - edge: z.number().int().min(0).max(4).optional(), - type: z.enum(['begin', 'move', 'end']), - x: z.number().min(0).max(1), - y: z.number().min(0).max(1) -}) - -const GestureParams = z.object({ - points: z.array(GesturePoint).min(2).max(64), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const TypeParams = z.object({ - text: z.string(), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const ButtonParams = z.object({ - name: z.string(), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const RotateOrientation = z.enum([ - 'portrait', - 'portrait_upside_down', - 'landscape_left', - 'landscape_right' -]) - -const RotateParams = z.object({ - orientation: RotateOrientation, - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const ExecParams = z.object({ - command: z.string(), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) +import { + AttachParams, + AxParams, + ButtonParams, + EmulatorAvailabilityParams, + EmulatorListDevicesParams, + EmulatorListSimulatorsParams, + EmulatorUnregisterActiveParams, + ExecParams, + GestureParams, + KillParams, + LaunchParams, + ListParams, + LogcatParams, + PermissionsParams, + RotateParams, + ShutdownParams, + TapParams, + TypeParams +} from '../../../../shared/rpc-contract/emulator-params' const InstallParams = z.object({ path: z.string().refine((value) => path.isAbsolute(value), { @@ -72,90 +32,7 @@ const InstallParams = z.object({ worktree: z.string().optional() }) -const LaunchParams = z.object({ - package: z.string(), - activity: z.string().optional(), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const PermissionsParams = z - .object({ - op: z.enum(['grant', 'revoke', 'reset']), - package: z.string().optional(), - permission: z.string().optional(), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() - }) - .superRefine((value, ctx) => { - if (value.op === 'reset') { - if (value.package) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['package'], - message: 'package is not allowed for reset' - }) - } - if (value.permission) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['permission'], - message: 'permission is not allowed for reset' - }) - } - return - } - if (!value.package) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['package'], - message: 'package is required for grant/revoke' - }) - } - if (!value.permission) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['permission'], - message: 'permission is required for grant/revoke' - }) - } - }) - -const AxParams = z.object({ - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const LogcatParams = z.object({ - lines: z.number().int().positive().optional(), - filters: z.array(z.string()).optional(), - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const AttachParams = z.object({ - device: z.string().optional(), - worktree: z.string().optional(), - focus: z.boolean().optional() -}) - -const KillParams = z.object({ - device: z.string().optional(), - emulator: z.string().optional(), - worktree: z.string().optional() -}) - -const ShutdownParams = KillParams.extend({ - managedOnly: z.boolean().optional() -}) - -const ListParams = WorktreeParam - -export const EMULATOR_METHODS: RpcMethod[] = [ +export const EMULATOR_METHODS = [ defineMethod({ name: 'emulator.list', params: ListParams, @@ -208,17 +85,17 @@ export const EMULATOR_METHODS: RpcMethod[] = [ }), defineMethod({ name: 'emulator.listSimulators', - params: z.object({ worktree: z.string().optional() }).partial(), + params: EmulatorListSimulatorsParams, handler: async (params, { runtime }) => runtime.emulatorListSimulators(params) }), defineMethod({ name: 'emulator.availability', - params: z.object({ worktree: z.string().optional() }).partial(), + params: EmulatorAvailabilityParams, handler: async (params, { runtime }) => runtime.emulatorAvailability(params) }), defineMethod({ name: 'emulator.listDevices', - params: z.object({ worktree: z.string().optional() }).partial(), + params: EmulatorListDevicesParams, handler: async (params, { runtime }) => runtime.emulatorListDevices(params) }), defineMethod({ @@ -248,7 +125,7 @@ export const EMULATOR_METHODS: RpcMethod[] = [ }), defineMethod({ name: 'emulator.unregisterActive', - params: z.object({ worktree: z.string().optional() }).partial(), + params: EmulatorUnregisterActiveParams, handler: async (params, { runtime }) => runtime.emulatorUnregisterActive(params) }) ] diff --git a/src/main/runtime/rpc/methods/files-mutation-methods.ts b/src/main/runtime/rpc/methods/files-mutation-methods.ts index 1add0232055..eb734d5c8d6 100644 --- a/src/main/runtime/rpc/methods/files-mutation-methods.ts +++ b/src/main/runtime/rpc/methods/files-mutation-methods.ts @@ -1,14 +1,14 @@ -import { z } from 'zod' -import { defineMethod, type RpcAnyMethod } from '../core' -import { FileOpen, WorktreeSelector } from './files-target-schemas' - -const RUNTIME_FILE_BASE64_PATTERN = /^[A-Za-z0-9+/]*={0,2}$/ - -function isValidRuntimeFileBase64(value: unknown): value is string { - return ( - typeof value === 'string' && value.length % 4 !== 1 && RUNTIME_FILE_BASE64_PATTERN.test(value) - ) -} +import { defineMethod } from '../core' +import { + FileCommitUpload, + FileCopy, + FileDelete, + FileMutationOpen, + FileRename, + FileWrite, + FileWriteBase64, + FileWriteBase64Chunk +} from '../../../../shared/rpc-contract/files-mutation-params' type SshMutationParams = { expectedExecutionHostId?: string @@ -33,81 +33,7 @@ function sshMutationArguments( ] } -const FileMutationOpen = FileOpen.extend({ - expectedExecutionHostId: z.string().min(1).optional(), - expectedSshTargetId: z.string().min(1).optional(), - expectedSshConnectionGeneration: z.number().int().nonnegative().optional() -}) - -// Why: write content must be a real string. Coercing a missing/non-string value -// to '' silently truncated the target file to empty instead of erroring. An -// explicit '' is still accepted (writing an empty file is legitimate). -const FileWrite = FileMutationOpen.extend({ - content: z - .unknown() - .refine((v): v is string => typeof v === 'string', { message: 'Missing file content' }) -}) - -const FileWriteBase64 = FileMutationOpen.extend({ - contentBase64: z - .unknown() - .refine((v): v is string => typeof v === 'string', { message: 'Missing file content' }) - // Why: Buffer.from(..., 'base64') accepts malformed input by dropping - // invalid bytes, which can silently create empty or corrupt uploaded files. - .refine(isValidRuntimeFileBase64, 'File content must be base64') -}) - -const FileWriteBase64Chunk = FileWriteBase64.extend({ - append: z.boolean().optional() -}) - -const FileRename = WorktreeSelector.extend({ - expectedExecutionHostId: z.string().min(1).optional(), - expectedSshTargetId: z.string().min(1).optional(), - expectedSshConnectionGeneration: z.number().int().nonnegative().optional(), - oldRelativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing source path')), - newRelativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing destination path')) -}) - -const FileCopy = WorktreeSelector.extend({ - expectedExecutionHostId: z.string().min(1).optional(), - expectedSshTargetId: z.string().min(1).optional(), - expectedSshConnectionGeneration: z.number().int().nonnegative().optional(), - sourceRelativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing source path')), - destinationRelativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing destination path')) -}) - -const FileCommitUpload = WorktreeSelector.extend({ - expectedExecutionHostId: z.string().min(1).optional(), - expectedSshTargetId: z.string().min(1).optional(), - expectedSshConnectionGeneration: z.number().int().nonnegative().optional(), - tempRelativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing temporary path')), - finalRelativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing final path')) -}) - -const FileDelete = FileMutationOpen.extend({ - recursive: z.boolean().optional() -}) - -export const FILE_MUTATION_METHODS: RpcAnyMethod[] = [ +export const FILE_MUTATION_METHODS = [ defineMethod({ name: 'files.write', params: FileWrite, diff --git a/src/main/runtime/rpc/methods/files-target-schemas.ts b/src/main/runtime/rpc/methods/files-target-schemas.ts index 6c546b3605e..945d3c6aa85 100644 --- a/src/main/runtime/rpc/methods/files-target-schemas.ts +++ b/src/main/runtime/rpc/methods/files-target-schemas.ts @@ -1,15 +1 @@ -import { z } from 'zod' - -export const WorktreeSelector = z.object({ - worktree: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing worktree selector')) -}) - -export const FileOpen = WorktreeSelector.extend({ - relativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing relative path')) -}) +export { FileOpen, WorktreeSelector } from '../../../../shared/rpc-contract/files-target-params' diff --git a/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts b/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts index 08925e08689..20d52eb424d 100644 --- a/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts +++ b/src/main/runtime/rpc/methods/files-terminal-artifact-methods.ts @@ -1,26 +1,11 @@ -import { z } from 'zod' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { remoteFileContentBudget } from './files-remote-content-budget' -import { WorktreeSelector } from './files-target-schemas' +import { + TerminalArtifactFile, + TerminalArtifactFileWrite +} from '../../../../shared/rpc-contract/files-terminal-artifact-params' -const TerminalArtifactFile = WorktreeSelector.extend({ - grantId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing terminal artifact grant')), - absolutePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing terminal artifact path')) -}) - -const TerminalArtifactFileWrite = TerminalArtifactFile.extend({ - content: z - .unknown() - .refine((v): v is string => typeof v === 'string', { message: 'Missing file content' }) -}) - -export const FILE_TERMINAL_ARTIFACT_METHODS: RpcAnyMethod[] = [ +export const FILE_TERMINAL_ARTIFACT_METHODS = [ defineMethod({ name: 'files.readTerminalArtifact', params: TerminalArtifactFile, diff --git a/src/main/runtime/rpc/methods/files.ts b/src/main/runtime/rpc/methods/files.ts index ef349a22f84..055c02b3265 100644 --- a/src/main/runtime/rpc/methods/files.ts +++ b/src/main/runtime/rpc/methods/files.ts @@ -1,115 +1,27 @@ -import { z } from 'zod' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { runFileWatchStream } from './file-watch-stream-lifecycle' import { FILE_MUTATION_METHODS } from './files-mutation-methods' import { remoteFileContentBudget } from './files-remote-content-budget' -import { - QUICK_OPEN_REMOTE_QUERY_MAX_CODE_UNITS, - QUICK_OPEN_SEARCH_VERSION -} from '../../../../shared/quick-open-path-search' +import { QUICK_OPEN_SEARCH_VERSION } from '../../../../shared/quick-open-path-search' import { limitQuickOpenSearchReplyBySerializedBytes } from '../../../../shared/quick-open-transport-budget' import { FileOpen, WorktreeSelector } from './files-target-schemas' import { FILE_TERMINAL_ARTIFACT_METHODS } from './files-terminal-artifact-methods' +import { + DocPreviewFileRead, + FileListAll, + FileOpenDiff, + FilePathSearch, + FileReadChunk, + FileSearch, + FileTreePath, + FileUnwatch, + ResolveTerminalPath, + ServerDirectoryBrowse +} from '../../../../shared/rpc-contract/files-params' let filesWatchSubscriptionSeq = 0 -const FilePathSearch = WorktreeSelector.extend({ - query: z.string().max(QUICK_OPEN_REMOTE_QUERY_MAX_CODE_UNITS).default(''), - limit: z.number().int().positive().max(32).default(16), - excludePaths: z.array(z.string()).optional(), - mode: z.literal('quick-open').optional() -}) - -const ResolveTerminalPath = WorktreeSelector.extend({ - pathText: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing path text')), - terminal: z - .unknown() - .transform((v) => (typeof v === 'string' && v.length > 0 ? v : null)) - .nullable() - .optional(), - cwd: z - .unknown() - .transform((v) => (typeof v === 'string' && v.length > 0 ? v : null)) - .nullable() - .optional(), - crossWorkspace: z - .unknown() - .transform((v) => v === true) - .optional(), - nativeChatContext: z - .object({ - tabId: z.string().min(1), - sessionId: z.string().min(1) - }) - .optional() -}) - -const FileOpenDiff = FileOpen.extend({ - staged: z.boolean().optional() -}) - -const DocPreviewFileRead = FileOpen.extend({ - entryRelativePath: z.string().min(1), - implicitRootRelativePath: z.string().nullable(), - authorizedRootRelativePaths: z.array(z.string()) -}) - -const FileTreePath = WorktreeSelector.extend({ - relativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string()) -}) - -const ServerDirectoryBrowse = z.object({ - path: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string()) -}) - -const FileReadChunk = FileOpen.extend({ - offset: z.number().int().nonnegative(), - length: z - .number() - .int() - .positive() - .max(512 * 1024) -}) - -const FileSearch = WorktreeSelector.extend({ - query: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing search query')), - caseSensitive: z.boolean().optional(), - wholeWord: z.boolean().optional(), - useRegex: z.boolean().optional(), - includePattern: z.string().optional(), - excludePattern: z.string().optional(), - maxResults: z.number().int().positive().optional() -}) - -// Why: `maxResults` is a new optional field (wire rule 1) — an older host strips it and keeps its -// own default. It existed only on the Electron IPC hop, so "the client names its cap and a full page -// means there is more" was true for desktop and merely incidental for web and mobile, which were -// saved by `remoteFileContentBudget` defaulting the cap inside `listRuntimeFiles`. -const FileListAll = WorktreeSelector.extend({ - excludePaths: z.array(z.string()).optional(), - maxResults: z.number().int().positive().optional() -}) - -const FileUnwatch = z.object({ - subscriptionId: z - .unknown() - .transform((value) => (typeof value === 'string' && value.length > 0 ? value : '')) - .pipe(z.string().min(1, 'Missing subscriptionId')) -}) - -export const FILE_METHODS: RpcAnyMethod[] = [ +export const FILE_METHODS = [ defineMethod({ name: 'files.list', params: WorktreeSelector, diff --git a/src/main/runtime/rpc/methods/folder-workspace.ts b/src/main/runtime/rpc/methods/folder-workspace.ts index aa39654d9f8..a2e178b8ed9 100644 --- a/src/main/runtime/rpc/methods/folder-workspace.ts +++ b/src/main/runtime/rpc/methods/folder-workspace.ts @@ -1,92 +1,13 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../schemas' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import { TaskSourceContextSchema } from '../../../../shared/task-source-context-schema' -import { WorkspaceLinkedItemSchema } from '../../../../shared/workspace-linked-item-schema' -import { isWorkspaceLinkedItemSourceContextMatch } from '../../../../shared/workspace-linked-item-source-context' +import { defineMethod } from '../core' import { resolveRpcWorkspaceCreatorProvenance } from '../workspace-creator-context' -import { DiffCommentSchema } from '../../../../shared/diff-comment-schema' +import { + FolderWorkspaceCreate, + FolderWorkspacePathStatus, + FolderWorkspaceSelector, + FolderWorkspaceUpdate +} from '../../../../shared/rpc-contract/folder-workspace-params' -const FolderWorkspaceLinkedTask = WorkspaceLinkedItemSchema.nullable() - -function assertLinkedTaskSourceContextMatch( - value: { - linkedTask?: z.infer - linkedTaskSourceContext?: z.infer | null - }, - ctx: z.RefinementCtx -): void { - if ( - value.linkedTask && - value.linkedTaskSourceContext && - !isWorkspaceLinkedItemSourceContextMatch(value.linkedTask, value.linkedTaskSourceContext) - ) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Linked task and source context identities must match' - }) - } -} - -const FolderWorkspaceCreate = z - .object({ - projectGroupId: requiredString('Missing project group id'), - name: OptionalString, - folderPath: OptionalString.nullable().optional(), - connectionId: OptionalString.nullable().optional(), - linkedTask: FolderWorkspaceLinkedTask.optional(), - linkedTaskSourceContext: TaskSourceContextSchema.nullable().optional(), - createdWithAgent: z.string().refine(isTuiAgent).optional(), - pendingFirstAgentMessageRename: z.boolean().optional() - }) - .superRefine(assertLinkedTaskSourceContextMatch) - -const FolderWorkspaceUpdate = z.object({ - folderWorkspaceId: requiredString('Missing folder workspace id'), - updates: z - .object({ - name: OptionalString, - folderPath: OptionalString, - linkedTask: FolderWorkspaceLinkedTask.optional(), - linkedTaskSourceContext: TaskSourceContextSchema.nullable().optional(), - comment: z.string().optional(), - isArchived: z.boolean().optional(), - isUnread: z.boolean().optional(), - isPinned: z.boolean().optional(), - sortOrder: OptionalFiniteNumber, - manualOrder: OptionalFiniteNumber, - workspaceStatus: OptionalString, - createdWithAgent: z.string().refine(isTuiAgent).optional(), - pendingFirstAgentMessageRename: z.boolean().optional(), - firstAgentMessageRenameError: z.string().nullable().optional(), - lastActivityAt: OptionalFiniteNumber, - diffComments: z.array(DiffCommentSchema).optional() - }) - .superRefine(assertLinkedTaskSourceContextMatch) -}) - -const FolderWorkspaceSelector = z.object({ - folderWorkspaceId: requiredString('Missing folder workspace id') -}) - -const FolderWorkspacePathStatus = z.discriminatedUnion('scope', [ - z.object({ - scope: z.literal('folder-workspace'), - folderWorkspaceId: requiredString('Missing folder workspace id') - }), - z.object({ - scope: z.literal('project-group'), - projectGroupId: requiredString('Missing project group id') - }), - z.object({ - scope: z.literal('path'), - path: requiredString('Missing folder path'), - connectionId: OptionalString.nullable().optional() - }) -]) - -export const FOLDER_WORKSPACE_METHODS: RpcMethod[] = [ +export const FOLDER_WORKSPACE_METHODS = [ defineMethod({ name: 'folderWorkspace.list', params: null, diff --git a/src/main/runtime/rpc/methods/git-admission-tier-schema.ts b/src/main/runtime/rpc/methods/git-admission-tier-schema.ts index 926aba5b6f9..c5366441b4b 100644 --- a/src/main/runtime/rpc/methods/git-admission-tier-schema.ts +++ b/src/main/runtime/rpc/methods/git-admission-tier-schema.ts @@ -1,11 +1 @@ -import { z } from 'zod' -import type { GitAdmissionTier } from '../../../git/command-runner/git-exec-options' - -export const OptionalGitAdmissionTier = z - .unknown() - .optional() - .transform((value): GitAdmissionTier | undefined => { - return value === 'interactive' || value === 'status' || value === 'background' - ? value - : undefined - }) +export { OptionalGitAdmissionTier } from '../../../../shared/rpc-contract/git-admission-tier-params' diff --git a/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts b/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts index 787c8581dea..1dffcfb2211 100644 --- a/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts +++ b/src/main/runtime/rpc/methods/git-commit-message-generation-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import type { GlobalSettings } from '../../../../shared/global-settings-types' import type { ResolvedSourceControlAiGenerationParams } from '../../../../shared/source-control-ai' import { @@ -58,7 +58,7 @@ function buildCommitMessageGenerationOverride(params: { } } -export const GIT_COMMIT_MESSAGE_GENERATION_METHODS: RpcMethod[] = [ +export const GIT_COMMIT_MESSAGE_GENERATION_METHODS = [ defineMethod({ name: 'git.generateCommitMessage', params: GitGenerateCommitMessage, diff --git a/src/main/runtime/rpc/methods/git-diff-methods.ts b/src/main/runtime/rpc/methods/git-diff-methods.ts index 7b5c0661c0e..edcaebbdf42 100644 --- a/src/main/runtime/rpc/methods/git-diff-methods.ts +++ b/src/main/runtime/rpc/methods/git-diff-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { remoteRpcContentBudget } from '../../../../shared/remote-rpc-content-budget' import { GitBranchDiff, GitCommitDiff, GitDiff } from './git-params' @@ -11,7 +11,7 @@ function remoteDiffContentBudget( return clientKind && requestId ? remoteRpcContentBudget(requestId) : undefined } -export const GIT_DIFF_METHODS: RpcMethod[] = [ +export const GIT_DIFF_METHODS = [ defineMethod({ name: 'git.diff', params: GitDiff, diff --git a/src/main/runtime/rpc/methods/git-params.ts b/src/main/runtime/rpc/methods/git-params.ts index f69b01cd053..ad80955ea76 100644 --- a/src/main/runtime/rpc/methods/git-params.ts +++ b/src/main/runtime/rpc/methods/git-params.ts @@ -1,271 +1,25 @@ -import { z } from 'zod' -import { OptionalGitAdmissionTier } from './git-admission-tier-schema' - -export const WorktreeSelector = z.object({ - worktree: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing worktree selector')) -}) - -export const GitStatusParams = WorktreeSelector.extend({ - admissionTier: OptionalGitAdmissionTier, - includeIgnored: z.boolean().optional(), - includeLineStats: z.boolean().optional(), - bypassEffectiveUpstreamNegativeCache: z.boolean().optional(), - reuseLineStats: z.boolean().optional(), - // Shape is re-validated host-side before it reaches a git argv. - branchLineTotalMergeBase: z.string().optional() -}) - -export const GitCheckIgnored = WorktreeSelector.extend({ - paths: z.array(z.string().min(1, 'Missing path')).max(2000) -}) - -export const GitSubmoduleStatus = WorktreeSelector.extend({ - submodulePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe( - z - .string() - .min(1, 'Missing submodule path') - // Why: never let a submodule path be parsed as a git flag (arg injection). - .refine((value) => !value.startsWith('-'), 'Submodule path must not start with -') - ), - // Why: submodule expansion is requested from a Source Control row; the row - // area determines whether the gitlink range is HEAD->index or index->worktree. - area: z.enum(['staged', 'unstaged', 'untracked']).optional() -}) - -export const GitFilePath = WorktreeSelector.extend({ - filePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing file path')) -}) - -export const GitDiff = GitFilePath.extend({ - staged: z.boolean(), - compareAgainstHead: z.boolean().optional() -}) - -export const GitBranchCompare = WorktreeSelector.extend({ - admissionTier: OptionalGitAdmissionTier, - baseRef: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe( - z - .string() - .min(1, 'Missing base ref') - .refine((value) => !value.startsWith('-'), 'Base ref must not start with -') - ) -}) - -const FullGitObjectId = z - .string() - .regex(/^(?:[0-9a-fA-F]{40}|[0-9a-fA-F]{64})$/, 'Expected a full git object id') - -export const GitCommitCompare = WorktreeSelector.extend({ - commitId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(FullGitObjectId) -}) - -export const GitHistory = WorktreeSelector.extend({ - limit: z.number().int().min(1).max(200).optional(), - baseRef: z.string().nullable().optional() -}) - -export const GitBranchDiff = GitFilePath.extend({ - compare: z.object({ - baseRef: z.string().optional(), - baseOid: FullGitObjectId.optional(), - headOid: FullGitObjectId, - mergeBase: FullGitObjectId - }), - oldPath: z.string().optional() -}) - -export const GitCommitDiff = GitFilePath.extend({ - commitOid: FullGitObjectId, - parentOid: FullGitObjectId.nullable().optional(), - oldPath: z.string().optional() -}) - -export const GitCommit = WorktreeSelector.extend({ - message: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing commit message')) -}) - -const CommitMessageModelCapability = z.object({ - id: z.string(), - label: z.string(), - thinkingLevels: z.array(z.object({ id: z.string(), label: z.string() })).optional(), - defaultThinkingLevel: z.string().optional() -}) - -const CommitMessageAiSettings = z.object({ - enabled: z.boolean(), - agentId: z.string().nullable(), - selectedModelByAgent: z.record(z.string(), z.string()), - selectedModelByAgentByHost: z.record(z.string(), z.record(z.string(), z.string())).optional(), - discoveredModelsByAgent: z.record(z.string(), z.array(CommitMessageModelCapability)).optional(), - discoveredModelsByAgentByHost: z - .record(z.string(), z.record(z.string(), z.array(CommitMessageModelCapability))) - .optional(), - selectedThinkingByModel: z.record(z.string(), z.string()), - customPrompt: z.string(), - customAgentCommand: z.string() -}) - -const SourceControlAiSettings = CommitMessageAiSettings.omit({ customPrompt: true }).extend({ - actions: z - .record( - z.string(), - z.object({ - agentId: z.string().nullable().optional(), - commandInputTemplate: z.string().optional(), - agentArgs: z.string().optional() - }) - ) - .optional(), - instructionsByOperation: z.record(z.string(), z.string()).optional(), - modelOverridesByOperation: z - .record( - z.string(), - z.object({ - selectedModelByAgent: z.record(z.string(), z.string()).optional(), - selectedModelByAgentByHost: z - .record(z.string(), z.record(z.string(), z.string())) - .optional(), - selectedThinkingByModel: z.record(z.string(), z.string()).optional() - }) - ) - .optional(), - prCreationDefaults: z - .object({ - draft: z.boolean().optional(), - useTemplate: z.boolean().optional(), - generateDetailsOnOpen: z.boolean().optional(), - openAfterCreate: z.boolean().optional() - }) - .optional(), - launchActionDefaults: z - .record( - z.string(), - z.object({ - agentId: z.string().nullable().optional(), - commandInputTemplate: z.string().optional(), - agentArgs: z.string().optional() - }) - ) - .optional() -}) - -const ResolvedSourceControlAiGenerationParams = z.object({ - agentId: z.string(), - model: z.string(), - thinkingLevel: z.string().optional(), - customPrompt: z.string().optional(), - commandInputTemplate: z.string().optional(), - agentArgs: z.string().optional(), - customAgentCommand: z.string().optional(), - agentCommandOverride: z.string().optional() -}) - -export const GitGenerateCommitMessage = WorktreeSelector.extend({ - commitMessageAi: CommitMessageAiSettings.optional(), - sourceControlAi: SourceControlAiSettings.optional(), - sourceControlAiResolvedParams: ResolvedSourceControlAiGenerationParams.optional(), - agentCmdOverrides: z.record(z.string(), z.string()).optional(), - commitMessageDiscoveryHostKey: z.string().optional() -}) - -export const GitDiscoverCommitMessageModels = WorktreeSelector.extend({ - agentId: z.string().min(1, 'Missing agent id'), - agentCmdOverrides: z.record(z.string(), z.string()).optional() -}) - -export const GitGeneratePullRequestFields = GitGenerateCommitMessage.extend({ - base: z.string().min(1, 'Missing base branch'), - title: z.string(), - body: z.string(), - draft: z.boolean(), - provider: z - .enum(['github', 'gitlab', 'bitbucket', 'azure-devops', 'gitea', 'unsupported']) - .optional(), - useTemplate: z.boolean().optional() -}) - -export const GitBulkPaths = WorktreeSelector.extend({ - filePaths: z.array(z.string().min(1, 'Missing file path')) -}) - -const GitPushTargetParam = z.object({ - remoteName: z.string(), - branchName: z.string(), - remoteUrl: z.string().optional(), - remoteCreated: z.boolean().optional() -}) - -export const GitPush = WorktreeSelector.extend({ - publish: z.boolean().optional(), - forceWithLease: z.boolean().optional(), - pushTarget: GitPushTargetParam.optional() -}) - -export const GitTargetedRemote = WorktreeSelector.extend({ - pushTarget: GitPushTargetParam.optional() -}) - -export const GitForkSync = WorktreeSelector.extend({ - expectedUpstream: z.object({ - owner: z.string().trim().min(1), - repo: z.string().trim().min(1) - }) -}) - -export const GitRebaseFromBase = WorktreeSelector.extend({ - baseRef: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe( - z - .string() - .min(1, 'Missing base ref') - .refine((value) => !value.startsWith('-'), 'Base ref must not start with -') - ) -}) - -export const GitCheckout = WorktreeSelector.extend({ - branch: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe( - z - .string() - .min(1, 'Missing branch') - // Why: never let a branch arg be parsed as a git flag (arg injection). - .refine((value) => !value.startsWith('-'), 'Branch must not start with -') - ) -}) - -export const GitRemoteFileUrl = WorktreeSelector.extend({ - relativePath: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing relative path')), - line: z.number().int().min(1) -}) - -export const GitRemoteCommitUrl = WorktreeSelector.extend({ - sha: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(FullGitObjectId) -}) +export { + GitBranchCompare, + GitBranchDiff, + GitBulkPaths, + GitCheckIgnored, + GitCheckout, + GitCommit, + GitCommitCompare, + GitCommitDiff, + GitDiff, + GitDiscoverCommitMessageModels, + GitFilePath, + GitForkSync, + GitGenerateCommitMessage, + GitGeneratePullRequestFields, + GitHistory, + GitPush, + GitRebaseFromBase, + GitRemoteCommitUrl, + GitRemoteFileUrl, + GitStatusParams, + GitSubmoduleStatus, + GitTargetedRemote, + WorktreeSelector +} from '../../../../shared/rpc-contract/git-params' diff --git a/src/main/runtime/rpc/methods/git.ts b/src/main/runtime/rpc/methods/git.ts index ddfbe7bf273..20102d71e28 100644 --- a/src/main/runtime/rpc/methods/git.ts +++ b/src/main/runtime/rpc/methods/git.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { GIT_COMMIT_MESSAGE_GENERATION_METHODS } from './git-commit-message-generation-methods' import { GIT_DIFF_METHODS } from './git-diff-methods' import { @@ -21,7 +21,7 @@ import { WorktreeSelector } from './git-params' -export const GIT_METHODS: RpcMethod[] = [ +export const GIT_METHODS = [ defineMethod({ name: 'git.status', params: GitStatusParams, diff --git a/src/main/runtime/rpc/methods/github-issue-methods.ts b/src/main/runtime/rpc/methods/github-issue-methods.ts index 75eb75c81b6..86300017741 100644 --- a/src/main/runtime/rpc/methods/github-issue-methods.ts +++ b/src/main/runtime/rpc/methods/github-issue-methods.ts @@ -1,33 +1,12 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { requiredString } from '../schemas' -import { IssueUpdate } from './github-issue-update-schema' -import { RepoSelector, SlugRepo } from './github-repo-target-schemas' +import { defineMethod } from '../core' +import { + CreateIssue, + Issue, + IssueComment, + UpdateIssue +} from '../../../../shared/rpc-contract/github-issue-params' -const Issue = RepoSelector.extend({ - number: z.number().int().positive() -}) - -const CreateIssue = RepoSelector.extend({ - title: requiredString('Missing title'), - body: z.string(), - labels: z.array(z.string()).optional(), - assignees: z.array(z.string()).optional() -}) - -const UpdateIssue = RepoSelector.extend({ - number: z.number().int().positive(), - updates: IssueUpdate -}) - -const IssueComment = RepoSelector.extend({ - number: z.number().int().positive(), - body: requiredString('Comment body required'), - type: z.enum(['issue', 'pr']).optional(), - prRepo: SlugRepo.nullable().optional() -}) - -export const GITHUB_ISSUE_METHODS: RpcMethod[] = [ +export const GITHUB_ISSUE_METHODS = [ defineMethod({ name: 'github.issue', params: Issue, diff --git a/src/main/runtime/rpc/methods/github-issue-update-schema.ts b/src/main/runtime/rpc/methods/github-issue-update-schema.ts index 7b3020e91b5..6a9f860113b 100644 --- a/src/main/runtime/rpc/methods/github-issue-update-schema.ts +++ b/src/main/runtime/rpc/methods/github-issue-update-schema.ts @@ -1,13 +1 @@ -import { z } from 'zod' -import { OptionalString } from '../schemas' - -// Why: repo-selector and slug-addressed issue updates must accept the identical field set. -export const IssueUpdate = z.object({ - state: z.enum(['open', 'closed']).optional(), - title: OptionalString, - body: OptionalString, - addLabels: z.array(z.string()).optional(), - removeLabels: z.array(z.string()).optional(), - addAssignees: z.array(z.string()).optional(), - removeAssignees: z.array(z.string()).optional() -}) +export { IssueUpdate } from '../../../../shared/rpc-contract/github-issue-update-params' diff --git a/src/main/runtime/rpc/methods/github-project-methods.ts b/src/main/runtime/rpc/methods/github-project-methods.ts index ce70c200218..c05086decbf 100644 --- a/src/main/runtime/rpc/methods/github-project-methods.ts +++ b/src/main/runtime/rpc/methods/github-project-methods.ts @@ -1,135 +1,26 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' -import { IssueUpdate } from './github-issue-update-schema' +import { defineMethod } from '../core' import { SlugRepo } from './github-repo-target-schemas' +import { + ClearProjectItemField, + GithubProjectListAccessibleParams, + ProjectItemField, + ProjectRef, + ProjectViewTable, + ProjectViews, + ProjectWorkItemDetailsBySlug, + SlugAssignableUsers, + SlugIssueComment, + SlugIssueCommentDelete, + SlugIssueCommentEdit, + SlugIssueTypeUpdate, + SlugIssueUpdate, + SlugPullRequestUpdate +} from '../../../../shared/rpc-contract/github-project-params' -const SlugAssignableUsers = SlugRepo.extend({ - seedLogins: z.array(z.string()).optional() -}) - -const ProjectOwnerType = z.enum(['organization', 'user']) - -const ProjectViewTable = z.object({ - owner: requiredString('Missing owner'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - ownerType: ProjectOwnerType, - projectNumber: z.number().int().positive(), - viewId: OptionalString, - viewNumber: z.number().int().positive().optional(), - viewName: OptionalString, - queryOverride: OptionalString -}) - -const ProjectWorkItemDetailsBySlug = SlugRepo.extend({ - number: z.number().int().positive(), - type: z.enum(['issue', 'pr']) -}) - -const ProjectRef = z.object({ - input: requiredString('Missing project reference'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString -}) - -const ProjectViews = z.object({ - owner: requiredString('Missing owner'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - ownerType: ProjectOwnerType, - projectNumber: z.number().int().positive() -}) - -const ProjectItemField = z.object({ - projectId: requiredString('Missing project ID'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - itemId: requiredString('Missing item ID'), - fieldId: requiredString('Missing field ID'), - value: z.any() -}) - -const ClearProjectItemField = z.object({ - projectId: requiredString('Missing project ID'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - itemId: requiredString('Missing item ID'), - fieldId: requiredString('Missing field ID') -}) - -const SlugIssueUpdate = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - number: z.number().int().positive(), - updates: IssueUpdate -}) - -const SlugPullRequestUpdate = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - number: z.number().int().positive(), - updates: z.object({ - state: z.enum(['open', 'closed']).optional(), - title: OptionalString, - body: OptionalString - }) -}) - -const SlugIssueTypeUpdate = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - number: z.number().int().positive(), - issueTypeId: z.string().nullable() -}) - -const SlugIssueComment = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - number: z.number().int().positive(), - body: requiredString('Comment body required') -}) - -const SlugIssueCommentEdit = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - commentId: z.number().int().positive(), - body: requiredString('Comment body required') -}) - -const SlugIssueCommentDelete = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - commentId: z.number().int().positive() -}) - -export const GITHUB_PROJECT_METHODS: RpcMethod[] = [ +export const GITHUB_PROJECT_METHODS = [ defineMethod({ name: 'github.project.listAccessible', - params: z.object({ host: OptionalString }), + params: GithubProjectListAccessibleParams, handler: async (params, { runtime }) => runtime.listGitHubProjects(params) }), defineMethod({ diff --git a/src/main/runtime/rpc/methods/github-pull-request-methods.ts b/src/main/runtime/rpc/methods/github-pull-request-methods.ts index 958e08c439a..0a7f2efd522 100644 --- a/src/main/runtime/rpc/methods/github-pull-request-methods.ts +++ b/src/main/runtime/rpc/methods/github-pull-request-methods.ts @@ -1,85 +1,17 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' -import { RepoSelector, SlugRepo } from './github-repo-target-schemas' -import type { GitHubPRRefreshReason } from '../../../../shared/github/pull-request-refresh-types' +import { defineMethod } from '../core' +import { + PRCommentReaction, + PrForBranch, + PullRequest, + PullRequestCheckDetails, + PullRequestChecks, + PullRequestFileContents, + PullRequestFileViewed, + RerunPullRequestChecks, + ReviewThread +} from '../../../../shared/rpc-contract/github-pull-request-params' -const OptionalPRRefreshReason = z - .unknown() - .optional() - .transform((value): GitHubPRRefreshReason | undefined => { - return value === 'visible' || - value === 'active' || - value === 'post-push' || - value === 'manual' || - value === 'swr' - ? value - : undefined - }) - -const PrForBranch = RepoSelector.extend({ - branch: requiredString('Missing branch'), - reason: OptionalPRRefreshReason, - linkedPRNumber: z.number().int().positive().nullable().optional(), - fallbackPRNumber: z.number().int().positive().nullable().optional(), - acceptMergedFallbackPR: z.boolean().optional(), - currentHeadOid: z.string().nullable().optional() -}) - -const PullRequest = RepoSelector.extend({ - prNumber: z.number().int().positive(), - noCache: z.boolean().optional(), - prRepo: SlugRepo.nullable().optional() -}) - -const PRCommentReaction = RepoSelector.extend({ - reactionSubjectId: requiredString('Missing reaction subject ID'), - content: z.enum(['+1', '-1', 'laugh', 'confused', 'heart', 'hooray', 'rocket', 'eyes']), - reacted: z.boolean(), - prRepo: SlugRepo.nullable().optional() -}) - -const PullRequestChecks = PullRequest.extend({ - headSha: OptionalString -}) - -const PullRequestCheckDetails = RepoSelector.extend({ - checkRunId: z.number().int().positive().optional(), - workflowRunId: z.number().int().positive().optional(), - checkName: OptionalString, - url: OptionalString.nullable().optional(), - prRepo: SlugRepo.nullable().optional() -}) - -const RerunPullRequestChecks = PullRequest.extend({ - headSha: OptionalString, - failedOnly: z.boolean().optional() -}) - -const PullRequestFileContents = RepoSelector.extend({ - prNumber: z.number().int().positive(), - prRepo: SlugRepo.nullable().optional(), - path: requiredString('Missing file path'), - oldPath: OptionalString, - status: z.enum(['added', 'removed', 'modified', 'renamed', 'copied', 'changed', 'unchanged']), - headSha: requiredString('Missing head SHA'), - baseSha: requiredString('Missing base SHA') -}) - -const PullRequestFileViewed = RepoSelector.extend({ - prRepo: SlugRepo.nullable().optional(), - pullRequestId: requiredString('Missing pull request ID'), - path: requiredString('Missing file path'), - viewed: z.boolean() -}) - -const ReviewThread = RepoSelector.extend({ - prRepo: SlugRepo.nullable().optional(), - threadId: requiredString('Missing thread ID'), - resolve: z.boolean() -}) - -export const GITHUB_PULL_REQUEST_METHODS: RpcMethod[] = [ +export const GITHUB_PULL_REQUEST_METHODS = [ defineMethod({ name: 'github.prForBranch', params: PrForBranch, diff --git a/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts b/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts index 4d34b89420e..59089b82015 100644 --- a/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts +++ b/src/main/runtime/rpc/methods/github-pull-request-update-methods.ts @@ -1,82 +1,18 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' -import { RepoSelector, SlugRepo } from './github-repo-target-schemas' +import { defineMethod } from '../core' +import { + MarkPrReadyForReview, + MergePr, + PRReviewComment, + PRReviewCommentReply, + RemovePrReviewers, + RequestPrReviewers, + SetPrAutoMerge, + UpdatePr, + UpdatePrState, + UpdatePrTitle +} from '../../../../shared/rpc-contract/github-pull-request-update-params' -const UpdatePrTitle = RepoSelector.extend({ - prNumber: z.number().int().positive(), - title: requiredString('Missing title'), - prRepo: SlugRepo.nullable().optional() -}) - -const UpdatePr = RepoSelector.extend({ - prNumber: z.number().int().positive(), - updates: z.object({ - title: OptionalString, - body: z.string().optional() - }), - prRepo: SlugRepo.nullable().optional() -}) - -const MergePr = RepoSelector.extend({ - prNumber: z.number().int().positive(), - method: z.enum(['merge', 'squash', 'rebase']).optional(), - prRepo: SlugRepo.nullable().optional() -}) - -const SetPrAutoMerge = RepoSelector.extend({ - prNumber: z.number().int().positive(), - enabled: z.boolean(), - method: z.enum(['merge', 'squash', 'rebase']).optional(), - prRepo: SlugRepo.nullable().optional() -}) - -const UpdatePrState = RepoSelector.extend({ - prNumber: z.number().int().positive(), - prRepo: SlugRepo.nullable().optional(), - updates: z.object({ - state: z.enum(['open', 'closed']) - }) -}) - -const MarkPrReadyForReview = RepoSelector.extend({ - prNumber: z.number().int().positive(), - prRepo: SlugRepo.nullable().optional() -}) - -const RequestPrReviewers = RepoSelector.extend({ - prNumber: z.number().int().positive(), - prRepo: SlugRepo.nullable().optional(), - reviewers: z.array(z.string()).min(1) -}) - -const RemovePrReviewers = RepoSelector.extend({ - prNumber: z.number().int().positive(), - prRepo: SlugRepo.nullable().optional(), - reviewers: z.array(z.string()).min(1) -}) - -const PRReviewComment = RepoSelector.extend({ - prNumber: z.number().int().positive(), - prRepo: SlugRepo.nullable().optional(), - commitId: requiredString('Missing PR head SHA'), - path: requiredString('File path required'), - line: z.number().int().positive(), - startLine: z.number().int().positive().optional(), - body: requiredString('Comment body required') -}) - -const PRReviewCommentReply = RepoSelector.extend({ - prNumber: z.number().int().positive(), - commentId: z.number().int().positive(), - body: requiredString('Comment body required'), - threadId: OptionalString, - path: OptionalString, - line: z.number().int().positive().optional(), - prRepo: SlugRepo.nullable().optional() -}) - -export const GITHUB_PULL_REQUEST_UPDATE_METHODS: RpcMethod[] = [ +export const GITHUB_PULL_REQUEST_UPDATE_METHODS = [ defineMethod({ name: 'github.updatePRTitle', params: UpdatePrTitle, diff --git a/src/main/runtime/rpc/methods/github-repo-target-schemas.ts b/src/main/runtime/rpc/methods/github-repo-target-schemas.ts index 3ea8b688910..06d47d47d9b 100644 --- a/src/main/runtime/rpc/methods/github-repo-target-schemas.ts +++ b/src/main/runtime/rpc/methods/github-repo-target-schemas.ts @@ -1,14 +1 @@ -import { z } from 'zod' -import { OptionalString, requiredString } from '../schemas' - -export const RepoSelector = z.object({ - repo: requiredString('Missing repo selector') -}) - -export const SlugRepo = z.object({ - owner: requiredString('Missing owner'), - repo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString -}) +export { RepoSelector, SlugRepo } from '../../../../shared/rpc-contract/github-repo-target-params' diff --git a/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts b/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts index 9602ba6cb70..6b2d83988d7 100644 --- a/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts +++ b/src/main/runtime/rpc/methods/github-repo-work-item-methods.ts @@ -1,45 +1,16 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' import { RepoSelector } from './github-repo-target-schemas' +import { + IssuesList, + RateLimit, + WorkItem, + WorkItemByOwnerRepo, + WorkItemDetails, + WorkItemsCount, + WorkItemsList +} from '../../../../shared/rpc-contract/github-repo-work-item-params' -const WorkItemsList = RepoSelector.extend({ - limit: OptionalFiniteNumber, - query: OptionalString, - page: z.number().int().positive().optional(), - noCache: z.boolean().optional() -}) - -const IssuesList = RepoSelector.extend({ - limit: OptionalFiniteNumber -}) - -const WorkItem = RepoSelector.extend({ - number: z.number().int().positive(), - type: z.enum(['issue', 'pr']).optional() -}) - -const WorkItemByOwnerRepo = RepoSelector.extend({ - owner: requiredString('Missing owner'), - ownerRepo: requiredString('Missing repo'), - // Why: Enterprise host identity must survive RPC parsing; Zod strips - // undeclared fields before the runtime can host-qualify gh requests. - host: OptionalString, - number: z.number().int().positive(), - type: z.enum(['issue', 'pr']) -}) - -const WorkItemDetails = WorkItem - -const WorkItemsCount = RepoSelector.extend({ - query: OptionalString -}) - -const RateLimit = z.object({ - force: z.boolean().optional() -}) - -export const GITHUB_REPO_WORK_ITEM_METHODS: RpcMethod[] = [ +export const GITHUB_REPO_WORK_ITEM_METHODS = [ defineMethod({ name: 'github.repoSlug', params: RepoSelector, diff --git a/src/main/runtime/rpc/methods/github.ts b/src/main/runtime/rpc/methods/github.ts index 1dd4cb28823..d2113f9ac22 100644 --- a/src/main/runtime/rpc/methods/github.ts +++ b/src/main/runtime/rpc/methods/github.ts @@ -1,11 +1,10 @@ -import type { RpcMethod } from '../core' import { GITHUB_ISSUE_METHODS } from './github-issue-methods' import { GITHUB_PROJECT_METHODS } from './github-project-methods' import { GITHUB_PULL_REQUEST_METHODS } from './github-pull-request-methods' import { GITHUB_PULL_REQUEST_UPDATE_METHODS } from './github-pull-request-update-methods' import { GITHUB_REPO_WORK_ITEM_METHODS } from './github-repo-work-item-methods' -export const GITHUB_METHODS: RpcMethod[] = [ +export const GITHUB_METHODS = [ ...GITHUB_REPO_WORK_ITEM_METHODS, ...GITHUB_ISSUE_METHODS, ...GITHUB_PULL_REQUEST_METHODS, diff --git a/src/main/runtime/rpc/methods/gitlab.ts b/src/main/runtime/rpc/methods/gitlab.ts index ac8fcad6991..93f73d5f05c 100644 --- a/src/main/runtime/rpc/methods/gitlab.ts +++ b/src/main/runtime/rpc/methods/gitlab.ts @@ -1,156 +1,29 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' import { normalizeGitLabIssueListArgs } from '../../../gitlab/gitlab-preload-args' import { toGitLabJobLogExcerptResult } from '../../../../shared/gitlab-job-log-excerpt' +import { + AddIssueComment, + AddMRComment, + AddMRInlineComment, + CreateIssue, + EmptyParams, + GitLabRateLimit, + IssuesList, + JobTrace, + MergeMr, + RepoSelector, + ResolveMRDiscussion, + RetryJob, + UpdateIssue, + UpdateMr, + UpdateMrReviewers, + UpdateMrState, + WorkItemByPath, + WorkItemDetails, + WorkItemsList +} from '../../../../shared/rpc-contract/gitlab-params' -const RepoSelector = z.object({ - repo: requiredString('Missing repo selector') -}) - -const EmptyParams = z.object({}).optional().default({}) -const GitLabRateLimit = z - .object({ - force: z.boolean().optional(), - host: OptionalString - }) - .optional() - .default({}) - -// nullish, not optional: renderer callers normalise a missing ref to `null` -// (`item.projectRef ?? null`), which a bare `.optional()` would reject outright. -const GitLabProjectRef = z - .object({ - host: requiredString('Missing GitLab host'), - path: requiredString('Missing GitLab project path') - }) - .nullish() - -const WorkItemsList = RepoSelector.extend({ - state: z.enum(['opened', 'merged', 'closed', 'all']).optional(), - page: OptionalFiniteNumber, - perPage: OptionalFiniteNumber, - query: OptionalString -}) - -const IssuesList = RepoSelector.extend({ - state: z.unknown().optional(), - assignee: OptionalString, - limit: OptionalFiniteNumber, - page: OptionalFiniteNumber -}) - -const CreateIssue = RepoSelector.extend({ - title: requiredString('Missing title'), - body: z.string() -}) - -const IssueUpdate = z.object({ - state: z.enum(['opened', 'closed']).optional(), - title: z.string().optional(), - body: z.string().optional(), - addLabels: z.array(z.string()).optional(), - removeLabels: z.array(z.string()).optional(), - addAssignees: z.array(z.string()).optional(), - removeAssignees: z.array(z.string()).optional() -}) - -const UpdateIssue = RepoSelector.extend({ - number: z.number().int().positive(), - updates: IssueUpdate, - projectRef: GitLabProjectRef -}) - -const UpdateMrState = RepoSelector.extend({ - iid: z.number().int().positive(), - state: z.enum(['opened', 'closed']), - projectRef: GitLabProjectRef -}) - -const UpdateMr = RepoSelector.extend({ - iid: z.number().int().positive(), - updates: z.object({ - title: z.string().optional(), - body: z.string().optional(), - addLabels: z.array(z.string()).optional(), - removeLabels: z.array(z.string()).optional(), - readyForReview: z.literal(true).optional() - }), - projectRef: GitLabProjectRef -}) - -const UpdateMrReviewers = RepoSelector.extend({ - iid: z.number().int().positive(), - reviewerIds: z.array(z.number().int().nonnegative()), - projectRef: GitLabProjectRef -}) - -const MergeMr = RepoSelector.extend({ - iid: z.number().int().positive(), - method: z.enum(['merge', 'squash', 'rebase']).optional(), - projectRef: GitLabProjectRef -}) - -const AddIssueComment = RepoSelector.extend({ - number: z.number().int().positive(), - body: requiredString('Comment body is required'), - projectRef: GitLabProjectRef -}) - -const AddMRComment = RepoSelector.extend({ - iid: z.number().int().positive(), - body: requiredString('Comment body is required'), - projectRef: GitLabProjectRef -}) - -const AddMRInlineComment = RepoSelector.extend({ - iid: z.number().int().positive(), - input: z.object({ - body: requiredString('Comment body is required'), - path: requiredString('File path is required'), - oldPath: z.string().optional(), - line: z.number().int().positive(), - baseSha: requiredString('Base SHA is required'), - startSha: requiredString('Start SHA is required'), - headSha: requiredString('Head SHA is required') - }), - projectRef: GitLabProjectRef -}) - -const ResolveMRDiscussion = RepoSelector.extend({ - iid: z.number().int().positive(), - discussionId: requiredString('Discussion id is required'), - resolved: z.boolean(), - projectRef: GitLabProjectRef -}) - -const JobTrace = RepoSelector.extend({ - jobId: z.number().int().positive(), - projectRef: GitLabProjectRef, - // Why: raw CI traces routinely exceed the 1 MB transport frame cap, so callers - // that only render an excerpt ask main to bound it before it crosses the wire. - logExcerpt: z.boolean().optional() -}) - -const RetryJob = RepoSelector.extend({ - jobId: z.number().int().positive(), - projectRef: GitLabProjectRef -}) - -const WorkItemDetails = RepoSelector.extend({ - iid: z.number().int().positive(), - type: z.enum(['issue', 'mr']), - projectRef: GitLabProjectRef -}) - -const WorkItemByPath = RepoSelector.extend({ - host: requiredString('Missing GitLab host'), - path: requiredString('Missing GitLab project path'), - iid: z.number().int().positive(), - type: z.enum(['issue', 'mr']) -}) - -export const GITLAB_METHODS: RpcMethod[] = [ +export const GITLAB_METHODS = [ defineMethod({ name: 'gitlab.listMRs', params: WorkItemsList, diff --git a/src/main/runtime/rpc/methods/host-capabilities.ts b/src/main/runtime/rpc/methods/host-capabilities.ts index afa1af88474..85a32fd1e5c 100644 --- a/src/main/runtime/rpc/methods/host-capabilities.ts +++ b/src/main/runtime/rpc/methods/host-capabilities.ts @@ -1,9 +1,9 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { isPwshAvailableAsync } from '../../../pwsh' import { isWslAvailableAsync, listWslDistrosAsync } from '../../../wsl' import { isGitBashAvailable } from '../../../git-bash' -export const HOST_CAPABILITY_METHODS: RpcMethod[] = [ +export const HOST_CAPABILITY_METHODS = [ defineMethod({ name: 'host.platform', params: null, diff --git a/src/main/runtime/rpc/methods/hosted-review.ts b/src/main/runtime/rpc/methods/hosted-review.ts index fae2e8eb162..51d663210d8 100644 --- a/src/main/runtime/rpc/methods/hosted-review.ts +++ b/src/main/runtime/rpc/methods/hosted-review.ts @@ -1,53 +1,11 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { requiredString } from '../schemas' -import { OptionalGitAdmissionTier } from './git-admission-tier-schema' +import { defineMethod } from '../core' +import { + HostedReviewCreate, + HostedReviewCreationEligibility, + HostedReviewForBranch +} from '../../../../shared/rpc-contract/hosted-review-params' -const HostedReviewForBranch = z.object({ - repo: requiredString('Missing repo selector'), - branch: requiredString('Missing branch'), - admissionTier: OptionalGitAdmissionTier, - currentHeadOid: z.string().nullable().optional(), - // Only the caller's selected worktree; the host caps how many earn the fast tier. - active: z.boolean().optional(), - linkedGitHubPR: z.number().int().positive().nullable().optional(), - fallbackGitHubPR: z.number().int().positive().nullable().optional(), - linkedGitLabMR: z.number().int().positive().nullable().optional(), - linkedBitbucketPR: z.number().int().positive().nullable().optional(), - linkedAzureDevOpsPR: z.number().int().positive().nullable().optional(), - linkedGiteaPR: z.number().int().positive().nullable().optional() -}) - -const HostedReviewCreationEligibility = z.object({ - repo: requiredString('Missing repo selector'), - worktree: z.string().min(1, 'Missing worktree selector').optional(), - branch: requiredString('Missing branch'), - base: z.string().nullable().optional(), - hasUncommittedChanges: z.boolean().optional(), - hasUpstream: z.boolean().optional(), - ahead: z.number().int().nonnegative().optional(), - behind: z.number().int().nonnegative().optional(), - linkedGitHubPR: z.number().int().positive().nullable().optional(), - fallbackGitHubPR: z.number().int().positive().nullable().optional(), - linkedGitLabMR: z.number().int().positive().nullable().optional(), - linkedBitbucketPR: z.number().int().positive().nullable().optional(), - linkedAzureDevOpsPR: z.number().int().positive().nullable().optional(), - linkedGiteaPR: z.number().int().positive().nullable().optional() -}) - -const HostedReviewCreate = z.object({ - repo: requiredString('Missing repo selector'), - worktree: z.string().min(1, 'Missing worktree selector').optional(), - provider: z.enum(['github', 'gitlab', 'bitbucket', 'azure-devops', 'gitea', 'unsupported']), - base: requiredString('Missing base branch'), - head: z.string().optional(), - title: requiredString('Missing title'), - body: z.string().optional(), - draft: z.boolean().optional(), - useTemplate: z.boolean().optional() -}) - -export const HOSTED_REVIEW_METHODS: RpcMethod[] = [ +export const HOSTED_REVIEW_METHODS = [ defineMethod({ name: 'hostedReview.forBranch', params: HostedReviewForBranch, diff --git a/src/main/runtime/rpc/methods/index.ts b/src/main/runtime/rpc/methods/index.ts index ba77b94803e..3a53bccb9ce 100644 --- a/src/main/runtime/rpc/methods/index.ts +++ b/src/main/runtime/rpc/methods/index.ts @@ -1,4 +1,3 @@ -import type { RpcAnyMethod } from '../core' import { STATUS_METHODS } from './status' import { AI_VAULT_METHODS } from './ai-vault' import { AUTOMATION_METHODS } from './automations' @@ -50,7 +49,7 @@ import { AGENT_HOOK_METHODS } from './agent-hooks' // Why: a flat manifest keeps registration order explicit and provides one // grep-point for "what methods does the RPC server expose?" — useful when // auditing the security boundary or wiring new CLI commands. -export const ALL_RPC_METHODS: readonly RpcAnyMethod[] = [ +export const ALL_RPC_METHODS = [ ...STATUS_METHODS, ...AGENT_HOOK_METHODS, ...AI_VAULT_METHODS, diff --git a/src/main/runtime/rpc/methods/jira.ts b/src/main/runtime/rpc/methods/jira.ts index 087aa25f36e..0e959cf0f5e 100644 --- a/src/main/runtime/rpc/methods/jira.ts +++ b/src/main/runtime/rpc/methods/jira.ts @@ -1,109 +1,24 @@ -import { z } from 'zod' import { JIRA_PAYLOAD_CHUNK_CHARS, JIRA_PAYLOAD_MAX_CHARS } from '../../../../shared/jira-payload-stream' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { - OptionalFiniteNumber, - OptionalPlainString, - OptionalString, - requiredString -} from '../schemas' - -const VALID_FILTERS = ['assigned', 'reported', 'all', 'done'] as const - -const SiteSelection = z - .object({ - siteId: OptionalString - }) - .optional() - -const Connect = z.object({ - siteUrl: requiredString('Site URL is required'), - // Self-hosted PAT auth needs no email; connect() enforces it for Cloud. - email: OptionalPlainString, - apiToken: requiredString('API token is required'), - authType: z.enum(['cloud', 'server']).optional() -}) - -const SelectSite = z.object({ - siteId: requiredString('Site ID is required') -}) - -const SearchIssues = z.object({ - jql: requiredString('Missing JQL'), - limit: OptionalFiniteNumber, - siteId: OptionalString -}) - -const ListIssues = z - .object({ - filter: z.enum(VALID_FILTERS).optional(), - limit: OptionalFiniteNumber, - siteId: OptionalString - }) - .optional() - -const IssueKey = z.object({ - key: requiredString('Issue key is required'), - siteId: OptionalString -}) - -const CreateIssue = z.object({ - siteId: OptionalString, - projectId: requiredString('Project is required'), - issueTypeId: requiredString('Issue type is required'), - title: requiredString('Title is required'), - description: OptionalPlainString, - customFields: z.record(z.string(), z.unknown()).optional(), - userFieldKeys: z.array(z.string()).optional() -}) - -const IssueUpdate = z.object({ - key: requiredString('Issue key is required'), - siteId: OptionalString, - updates: z.object({ - title: OptionalString, - labels: z.array(z.string()).optional(), - assigneeAccountId: z.union([z.string(), z.null()]).optional(), - priorityId: z.union([z.string(), z.null()]).optional(), - transitionId: OptionalString - }) -}) - -const IssueComment = z.object({ - key: requiredString('Issue key is required'), - body: requiredString('Comment body is required'), - siteId: OptionalString -}) - -const ProjectIssueTypes = z.object({ - projectIdOrKey: requiredString('Project is required'), - siteId: OptionalString -}) - -const ProjectIssueTypeFields = z.object({ - projectIdOrKey: requiredString('Project is required'), - issueTypeId: requiredString('Issue type is required'), - siteId: OptionalString -}) - -const AssignableUsers = z.object({ - key: requiredString('Issue key is required'), - query: OptionalPlainString, - siteId: OptionalString -}) - -const UserSearch = z.object({ - query: OptionalPlainString, - siteId: OptionalString -}) - -const ProjectStatusOrder = z.object({ - projectKey: requiredString('Project key is required'), - siteId: OptionalString -}) + AssignableUsers, + Connect, + CreateIssue, + IssueComment, + IssueKey, + IssueUpdate, + ListIssues, + ProjectIssueTypeFields, + ProjectIssueTypes, + ProjectStatusOrder, + SearchIssues, + SelectSite, + SiteSelection, + UserSearch +} from '../../../../shared/rpc-contract/jira-params' /** Emits a Jira result over RPC, normalizing it to the shape clients decode. */ function emitJiraPayload(value: unknown, emit: (result: unknown) => void): void { @@ -119,7 +34,7 @@ function emitJiraPayload(value: unknown, emit: (result: unknown) => void): void emit({ type: 'end' }) } -export const JIRA_METHODS: RpcAnyMethod[] = [ +export const JIRA_METHODS = [ defineMethod({ name: 'jira.connect', params: Connect, diff --git a/src/main/runtime/rpc/methods/linear-agent-access.ts b/src/main/runtime/rpc/methods/linear-agent-access.ts index 50e7bfef3a2..2c46face39d 100644 --- a/src/main/runtime/rpc/methods/linear-agent-access.ts +++ b/src/main/runtime/rpc/methods/linear-agent-access.ts @@ -1,149 +1,22 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' import { linearError } from '../../../linear/issue-context-errors' import { isLinearUuid } from '../../../../shared/linear/uuid' - -const LINEAR_DUE_DATE_PATTERN = /^\d{4}-\d{2}-\d{2}$/ -const LinearDueDate = z.string().refine((value) => LINEAR_DUE_DATE_PATTERN.test(value), { - message: 'Linear due dates must use YYYY-MM-DD' -}) -const OptionalLinearDueDate = LinearDueDate.optional() -const OptionalLinearDueDateOrClear = z.union([LinearDueDate, z.null()]).optional() - -const AgentSearchIssues = z.object({ - query: requiredString('Missing query'), - limit: OptionalFiniteNumber, - workspaceId: z.union([z.string(), z.literal('all')]).optional() -}) - -const LinearWorkspaceRead = z.object({ - workspaceId: z.union([z.string(), z.literal('all')]).optional() -}) - -const LinearTeamLookup = z.object({ - teamInput: requiredString('Missing team'), - workspaceId: OptionalString.refine((value) => value !== 'all', { - message: '--workspace all is only valid for team list' - }) -}) - -const LinearIssueList = z.object({ - filter: z.enum(['assigned', 'created', 'all', 'completed', 'open']).optional(), - teamInput: OptionalString, - limit: OptionalFiniteNumber, - workspaceId: z.union([z.string(), z.literal('all')]).optional() -}) - -const LinearProjectList = z.object({ - query: OptionalString, - limit: OptionalFiniteNumber, - workspaceId: z.union([z.string(), z.literal('all')]).optional() -}) - -const LinearIncludeFlags = z.object({ - comments: z.boolean(), - children: z.boolean(), - attachments: z.boolean(), - relations: z.boolean(), - activity: z.boolean().default(false) -}) - -const LinearCurrentContext = z - .object({ - worktreeId: OptionalString, - terminalHandle: OptionalString, - cwd: OptionalString, - remote: z.boolean().optional() - }) - .optional() - -const LinearWriteTarget = z.object({ - input: OptionalString, - current: z.boolean().optional(), - workspaceId: OptionalString.refine((value) => value !== 'all', { - message: '--workspace all is not valid for Linear writes' - }), - context: LinearCurrentContext -}) - -const AgentIssueContext = z.object({ - input: OptionalString, - current: z.boolean().optional(), - workspaceId: OptionalString, - include: LinearIncludeFlags, - depth: z.number().int().min(0).max(5), - context: LinearCurrentContext -}) - -const LinearIssueSetState = LinearWriteTarget.extend({ - to: requiredString('Missing target state') -}) - -const LinearIssueUpdateTask = LinearWriteTarget.extend({ - operation: z.enum(['assignee', 'priority', 'estimate', 'dueDate', 'labels']), - assigneeId: z.string().nullable().optional(), - assigneeMe: z.boolean().optional(), - priority: z.number().int().min(0).max(4).optional(), - estimate: z.number().int().min(0).nullable().optional(), - dueDate: OptionalLinearDueDateOrClear, - labelMode: z.enum(['add', 'remove', 'set']).optional(), - labels: z.array(z.string()).optional() -}) - -const LinearIssueAddComment = LinearWriteTarget.extend({ - body: requiredString('Missing comment body'), - replyTo: OptionalString, - writeId: OptionalString -}) - -const LinearIssueRelationWrite = LinearWriteTarget.extend({ - relatedInput: requiredString('Missing related issue'), - relationship: z.enum(['blocks', 'blockedBy', 'relatedTo', 'duplicateOf']), - operation: z.enum(['add', 'remove']) -}) - -const LinearIssueAttachLink = LinearWriteTarget.extend({ - url: requiredString('Missing attachment URL'), - title: OptionalString, - writeId: OptionalString -}) - -const LinearIssueCreate = z.object({ - title: requiredString('Missing issue title'), - body: OptionalString, - teamInput: OptionalString, - teamKey: OptionalString, - state: OptionalString, - assignee: OptionalString, - priority: z.number().int().min(0).max(4).optional(), - estimate: z.number().int().min(0).optional(), - dueDate: OptionalLinearDueDate, - labels: z.array(z.string()).optional(), - projectInput: OptionalString, - parentInput: OptionalString, - parentCurrent: z.boolean().optional(), - workspaceId: OptionalString.refine((value) => value !== 'all', { - message: '--workspace all is not valid for Linear writes' - }), - writeId: OptionalString, - context: LinearCurrentContext -}) - -const LinearSaveIssue = LinearWriteTarget.extend({ - team: OptionalString, - title: OptionalString, - description: z.string().optional(), - state: OptionalString, - assignee: z.string().nullable().optional(), - priority: z.number().int().min(0).max(4).optional(), - estimate: z.number().min(0).nullable().optional(), - dueDate: OptionalLinearDueDateOrClear, - labels: z.array(z.string()).optional(), - project: z.string().nullable().optional(), - parentId: z.string().nullable().optional(), - writeId: OptionalString -}) +import { + AgentIssueContext, + AgentSearchIssues, + LinearCurrentContext, + LinearIssueAddComment, + LinearIssueAttachLink, + LinearIssueCreate, + LinearIssueList, + LinearIssueRelationWrite, + LinearIssueSetState, + LinearIssueUpdateTask, + LinearProjectList, + LinearSaveIssue, + LinearTeamLookup, + LinearWorkspaceRead +} from '../../../../shared/rpc-contract/linear-agent-access-params' function parseLinearWriteId(writeId: string | undefined): string | undefined { if (writeId === undefined) { @@ -155,7 +28,7 @@ function parseLinearWriteId(writeId: string | undefined): string | undefined { return writeId } -export const LINEAR_AGENT_ACCESS_METHODS: RpcMethod[] = [ +export const LINEAR_AGENT_ACCESS_METHODS = [ defineMethod({ name: 'linear.saveIssue', params: LinearSaveIssue, diff --git a/src/main/runtime/rpc/methods/linear-issue-attribute-filter-schema.ts b/src/main/runtime/rpc/methods/linear-issue-attribute-filter-schema.ts index 7b7176f48b8..b369a01a15c 100644 --- a/src/main/runtime/rpc/methods/linear-issue-attribute-filter-schema.ts +++ b/src/main/runtime/rpc/methods/linear-issue-attribute-filter-schema.ts @@ -1,35 +1 @@ -import { z } from 'zod' -import { - LINEAR_ISSUE_ATTRIBUTE_FILTER_ID_MAX_LENGTH, - LINEAR_ISSUE_ATTRIBUTE_FILTER_MAX_LABEL_IDS, - LINEAR_ISSUE_ATTRIBUTE_FILTER_MAX_PRIORITIES, - LINEAR_ISSUE_ATTRIBUTE_FILTER_MAX_STATE_IDS -} from '../../../../shared/linear/issue-attribute-filter' - -// Why: keep ListIssues param validation co-located with shared limits without -// pushing linear.ts past the max-lines ratchet. -const LinearAttributeFilterId = z - .string() - .trim() - .min(1) - .max(LINEAR_ISSUE_ATTRIBUTE_FILTER_ID_MAX_LENGTH) - -export const LinearIssueAttributeFilterSchema = z - .object({ - stateIds: z.array(LinearAttributeFilterId).max(LINEAR_ISSUE_ATTRIBUTE_FILTER_MAX_STATE_IDS), - priorities: z - .array(z.number().int().min(0).max(4)) - .max(LINEAR_ISSUE_ATTRIBUTE_FILTER_MAX_PRIORITIES), - assignee: z.union([ - z.object({ kind: z.literal('unassigned') }).strict(), - z - .object({ - kind: z.literal('user'), - id: LinearAttributeFilterId - }) - .strict(), - z.null() - ]), - labelIds: z.array(LinearAttributeFilterId).max(LINEAR_ISSUE_ATTRIBUTE_FILTER_MAX_LABEL_IDS) - }) - .strict() +export { LinearIssueAttributeFilterSchema } from '../../../../shared/rpc-contract/linear-issue-attribute-filter-params' diff --git a/src/main/runtime/rpc/methods/linear-issue-list-method.ts b/src/main/runtime/rpc/methods/linear-issue-list-method.ts index fa69b745377..9dd838c837a 100644 --- a/src/main/runtime/rpc/methods/linear-issue-list-method.ts +++ b/src/main/runtime/rpc/methods/linear-issue-list-method.ts @@ -1,42 +1,6 @@ -import { z } from 'zod' +import type { z } from 'zod' import { defineMethod } from '../core' -import { OptionalFiniteNumber, OptionalString } from '../schemas' -import { LinearIssueAttributeFilterSchema } from './linear-issue-attribute-filter-schema' - -const LegacyListIssues = z - .object({ - filter: z.enum(['assigned', 'created', 'all', 'completed']).optional(), - limit: OptionalFiniteNumber, - workspaceId: OptionalString, - attributeFilter: LinearIssueAttributeFilterSchema.optional() - }) - .strict() - .optional() - -const McpListIssues = z - .object({ - team: OptionalString, - cycle: OptionalString, - label: OptionalString, - limit: z.number().int().min(1).max(250).optional(), - query: OptionalString, - state: OptionalString, - cursor: OptionalString, - orderBy: z.enum(['createdAt', 'updatedAt']).optional(), - project: OptionalString, - release: OptionalString, - assignee: OptionalString, - delegate: OptionalString, - parentId: OptionalString, - priority: z.number().int().min(0).max(4).optional(), - createdAt: OptionalString, - updatedAt: OptionalString, - includeArchived: z.boolean().optional(), - workspaceId: OptionalString - }) - .strict() - -const ListIssues = z.union([McpListIssues, LegacyListIssues]) +import { ListIssues, McpListIssues } from '../../../../shared/rpc-contract/linear-issue-list-params' export const LINEAR_ISSUE_LIST_METHOD = defineMethod({ name: 'linear.listIssues', diff --git a/src/main/runtime/rpc/methods/linear-project-create.ts b/src/main/runtime/rpc/methods/linear-project-create.ts index 026c585aba2..8377b69acdf 100644 --- a/src/main/runtime/rpc/methods/linear-project-create.ts +++ b/src/main/runtime/rpc/methods/linear-project-create.ts @@ -1,25 +1,7 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' +import { CreateProject } from '../../../../shared/rpc-contract/linear-project-create-params' -const LinearPriority = z.number().int().min(0).max(4).optional() -const LinearLabelIds = z.array(requiredString('Invalid label ID')).optional() - -const CreateProject = z.object({ - name: requiredString('Project name is required'), - description: OptionalString, - content: OptionalString, - workspaceId: OptionalString, - teamIds: z.array(requiredString('Invalid team ID')).min(1, 'At least one team is required'), - leadId: z.union([z.string(), z.null()]).optional(), - memberIds: z.array(requiredString('Invalid member ID')).optional(), - labelIds: LinearLabelIds, - priority: LinearPriority, - startDate: OptionalString, - targetDate: OptionalString -}) - -export const LINEAR_PROJECT_CREATE_METHOD: RpcMethod = defineMethod({ +export const LINEAR_PROJECT_CREATE_METHOD = defineMethod({ name: 'linear.createProject', params: CreateProject, handler: async (params, { runtime }) => diff --git a/src/main/runtime/rpc/methods/linear.ts b/src/main/runtime/rpc/methods/linear.ts index 1f3e474e78d..456b4c5b4e9 100644 --- a/src/main/runtime/rpc/methods/linear.ts +++ b/src/main/runtime/rpc/methods/linear.ts @@ -1,126 +1,26 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' import { LINEAR_PROJECT_CREATE_METHOD } from './linear-project-create' import { LINEAR_ISSUE_LIST_METHOD, LINEAR_MCP_ISSUE_LIST_METHOD } from './linear-issue-list-method' +import { + Connect, + CreateIssue, + CustomViewContents, + CustomViewId, + IssueComment, + IssueId, + IssueUpdate, + LinearIssueCommentsParams, + ListCustomViews, + ListProjects, + ProjectId, + ProjectIssues, + SearchIssues, + SelectWorkspace, + TeamId, + WorkspaceSelection +} from '../../../../shared/rpc-contract/linear-params' -const VALID_CUSTOM_VIEW_MODELS = ['issue', 'project'] as const -const LinearPriority = z.number().int().min(0).max(4).optional() -const LinearLabelIds = z.array(requiredString('Invalid label ID')).optional() - -const Connect = z.object({ - apiKey: requiredString('Invalid API key') -}) - -const WorkspaceSelection = z - .object({ - workspaceId: OptionalString - }) - .optional() - -const ConcreteWorkspaceId = requiredString('Concrete Linear workspace ID is required').refine( - (value) => value !== 'all', - 'Concrete Linear workspace ID is required' -) - -const SelectWorkspace = z.object({ - workspaceId: requiredString('Workspace ID is required') -}) - -const SearchIssues = z.object({ - query: requiredString('Missing query'), - limit: OptionalFiniteNumber, - workspaceId: OptionalString -}) - -const CreateIssue = z.object({ - teamId: requiredString('Team ID is required'), - title: requiredString('Title is required'), - description: OptionalString, - workspaceId: OptionalString, - parentIssueId: OptionalString, - projectId: z.union([z.string(), z.null()]).optional(), - stateId: OptionalString, - priority: LinearPriority, - assigneeId: z.union([z.string(), z.null()]).optional(), - labelIds: LinearLabelIds -}) - -const IssueId = z.object({ - id: requiredString('Issue ID is required'), - workspaceId: OptionalString -}) - -const IssueComment = z.object({ - issueId: requiredString('Issue ID is required'), - body: requiredString('Comment body is required'), - workspaceId: OptionalString -}) - -const ListProjects = z - .object({ - query: OptionalString, - limit: OptionalFiniteNumber, - workspaceId: OptionalString, - force: z.boolean().optional() - }) - .optional() - -const ProjectId = z.object({ - id: requiredString('Project ID is required'), - workspaceId: ConcreteWorkspaceId, - force: z.boolean().optional() -}) - -const ProjectIssues = z.object({ - projectId: requiredString('Project ID is required'), - limit: OptionalFiniteNumber, - workspaceId: ConcreteWorkspaceId, - force: z.boolean().optional() -}) - -const ListCustomViews = z.object({ - model: z.enum(VALID_CUSTOM_VIEW_MODELS), - limit: OptionalFiniteNumber, - workspaceId: OptionalString, - force: z.boolean().optional() -}) - -const CustomViewId = z.object({ - viewId: requiredString('Custom view ID is required'), - model: z.enum(VALID_CUSTOM_VIEW_MODELS), - workspaceId: ConcreteWorkspaceId, - force: z.boolean().optional() -}) - -const CustomViewContents = z.object({ - viewId: requiredString('Custom view ID is required'), - limit: OptionalFiniteNumber, - workspaceId: ConcreteWorkspaceId, - force: z.boolean().optional() -}) - -const TeamId = z.object({ - teamId: requiredString('Team ID is required'), - workspaceId: OptionalString -}) - -const IssueUpdate = z.object({ - id: requiredString('Issue ID is required'), - workspaceId: OptionalString, - updates: z.object({ - stateId: OptionalString, - title: OptionalString, - description: z.string().optional(), - assigneeId: z.union([z.string(), z.null()]).optional(), - estimate: z.union([z.number().int().min(0), z.null()]).optional(), - priority: z.number().int().min(0).max(4).optional(), - labelIds: z.array(z.string()).optional(), - projectId: z.union([z.string(), z.null()]).optional() - }) -}) - -export const LINEAR_METHODS: RpcMethod[] = [ +export const LINEAR_METHODS = [ defineMethod({ name: 'linear.connect', params: Connect, @@ -193,10 +93,7 @@ export const LINEAR_METHODS: RpcMethod[] = [ }), defineMethod({ name: 'linear.issueComments', - params: z.object({ - issueId: requiredString('Issue ID is required'), - workspaceId: OptionalString - }), + params: LinearIssueCommentsParams, handler: async (params, { runtime }) => runtime.linearIssueComments(params.issueId.trim(), params.workspaceId) }), diff --git a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts index dcba8b7b64e..756d4719f22 100644 --- a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts +++ b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' -export const MOBILE_MARKDOWN_TAB_METHODS: RpcAnyMethod[] = [ +export const MOBILE_MARKDOWN_TAB_METHODS = [ defineMethod({ name: 'markdown.readTab', params: ActivateTab, diff --git a/src/main/runtime/rpc/methods/native-chat.ts b/src/main/runtime/rpc/methods/native-chat.ts index e1a92dd52db..05f8f7f8726 100644 --- a/src/main/runtime/rpc/methods/native-chat.ts +++ b/src/main/runtime/rpc/methods/native-chat.ts @@ -1,59 +1,17 @@ -import { z } from 'zod' -import type { NativeChatMessage, AgentType } from '../../../../shared/native-chat-types' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' import { readNativeChatTranscriptTail, subscribeNativeChatTranscript, type NativeChatTranscriptSubscription, type SubscribeNativeChatTranscriptArgs } from '../../../native-chat/transcript-watch' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, defineStreamingMethod, type RpcContext } from '../core' import { sanitizeNativeChatRpcBlock } from './native-chat-rpc-block-sanitize' - -// Why: native chat renders an agent's own transcript (Claude/Codex JSONL). The -// desktop reaches the readers via Electron IPC; mobile/web clients reach the -// same pure readers through these runtime RPC methods so the native chat view -// works over the paired connection, not just in the desktop renderer. - -const NativeChatSession = z.object({ - agent: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing agent')) - .transform((v) => v as AgentType), - sessionId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing session id')), - // How many of the most-recent messages to return. Clients start small for a - // fast first paint and raise it to page older history in as the user scrolls. - // Clamp (don't reject) a limit past the max window so a client paging beyond it - // gets the capped tail and pagination stops cleanly — a hard `.max` rejection - // would fail the read and stall "load earlier" at the boundary. - limit: z - .number() - .int() - .positive() - .transform((value) => Math.min(value, MOBILE_NATIVE_CHAT_MAX_WINDOW)) - .optional(), - // Optional client-supplied cleanup token. When present, the subscribe handler - // keys the fs-watcher cleanup under it so registration and unsubscribe derive - // from the SAME token (back-compat: falls back to `agent:sessionId` when absent, - // which is exactly what existing mobile clients rely on). - subscriptionId: z.string().min(1).optional(), - // Authoritative transcript path from the agent hook (providerSession), used to - // locate the file directly when the session id no longer names it (recent - // Claude Code). Optional for back-compat with older clients. - transcriptPath: z.string().min(1).optional(), - // A pending snapshot is not authoritative transcript history. Only clients - // that advertise this semantic may receive one; legacy clients treat it as a - // settled empty read and can overwrite retention / unblock launch drafts. - capabilities: z.object({ transcriptPending: z.literal(1).optional() }).optional(), - beforeOffset: z.number().int().nonnegative().optional() -}) - -const NativeChatUnsubscribe = z.object({ - subscriptionId: z.string().min(1).optional() -}) +import { + MOBILE_NATIVE_CHAT_MAX_WINDOW, + NativeChatSession, + NativeChatUnsubscribe +} from '../../../../shared/rpc-contract/native-chat-params' // Why: a long agent session can hold thousands of turns (with full tool I/O). // Shipping all of them over the paired connection and rendering them at once @@ -63,7 +21,6 @@ const NativeChatUnsubscribe = z.object({ // Small first page for a fast initial paint; the client raises `limit` to load // older history as the user scrolls back. const MOBILE_NATIVE_CHAT_DEFAULT_WINDOW = 40 -const MOBILE_NATIVE_CHAT_MAX_WINDOW = 2000 function sanitizeMessage( message: NativeChatMessage, @@ -106,7 +63,7 @@ function windowForClient( return windowed.map((message) => sanitizeMessage(message, clientKind)) } -export const NATIVE_CHAT_METHODS: readonly RpcAnyMethod[] = [ +export const NATIVE_CHAT_METHODS = [ defineMethod({ name: 'nativeChat.readSession', params: NativeChatSession, diff --git a/src/main/runtime/rpc/methods/notification-preferences.test.ts b/src/main/runtime/rpc/methods/notification-preferences.test.ts new file mode 100644 index 00000000000..5686682376f --- /dev/null +++ b/src/main/runtime/rpc/methods/notification-preferences.test.ts @@ -0,0 +1,84 @@ +import { expect, it } from 'vitest' +import { NOTIFICATION_METHODS } from './notifications' +import { RuntimeMobileNotificationController } from '../../runtime-mobile-notification-controller' +import type { RpcContext, RpcStreamingMethod, RpcMethod } from '../core' + +it('keeps desktop-disabled events out of legacy live and replay streams', async () => { + const controller = new RuntimeMobileNotificationController() + const cleanups: (() => void)[] = [] + const runtime = { + onNotificationDispatched: controller.onDispatched.bind(controller), + getMobileNotificationEpoch: controller.getEpoch.bind(controller), + getMissedNotificationsSince: controller.getMissedSince.bind(controller), + registerSubscriptionCleanup: (_id: string, cleanup: () => void) => cleanups.push(cleanup) + } + const ctx = { runtime } as unknown as RpcContext + const subscribe = NOTIFICATION_METHODS.find( + (method) => method.name === 'notifications.subscribe' + ) as RpcStreamingMethod + const replay = NOTIFICATION_METHODS.find( + (method) => method.name === 'notifications.getMissedSince' + ) as RpcMethod + const legacy: unknown[] = [] + const current: unknown[] = [] + const pending = [ + subscribe.handler({}, ctx, (event) => legacy.push(event)), + subscribe.handler({ includeDesktopSuppressed: true }, ctx, (event) => current.push(event)) + ] + controller.dispatch({ + type: 'notification', + source: 'terminal-bell', + title: 'bell', + body: '', + desktopAllowed: false + }) + controller.dispatch({ + type: 'notification', + source: 'agent-task-complete', + title: 'done', + body: '' + }) + expect(legacy).toHaveLength(2) + expect(current).toHaveLength(3) + expect(legacy[1]).toMatchObject({ title: 'done' }) + expect(current[1]).toMatchObject({ desktopAllowed: false }) + expect(await replay.handler({ lastSeenSeq: 0 }, ctx)).toMatchObject({ + notifications: [{ title: 'done' }] + }) + const result = (await replay.handler( + { lastSeenSeq: 0, includeDesktopSuppressed: true }, + ctx + )) as { notifications: unknown[] } + expect(result.notifications).toHaveLength(2) + cleanups.forEach((cleanup) => cleanup()) + await Promise.all(pending) +}) + +it('preserves legacy workspace cooldown while letting current phones filter before cooldown', async () => { + const { createNotificationStreamFilter } = await import('./notification-stream-policy') + const events = [ + { + type: 'notification' as const, + source: 'terminal-bell' as const, + title: '', + body: '', + worktreeId: 'folder', + emittedAt: 10000 + }, + { + type: 'notification' as const, + source: 'agent-task-complete' as const, + title: '', + body: '', + worktreeId: 'folder', + emittedAt: 10250 + } + ] + const controller = new RuntimeMobileNotificationController() + events.forEach((event) => controller.dispatch(event)) + const recorded = controller.getMissedSince(0) + expect(recorded.filter(createNotificationStreamFilter())).toEqual([ + expect.objectContaining(events[0]) + ]) + expect(recorded.filter(createNotificationStreamFilter(true))).toHaveLength(2) +}) diff --git a/src/main/runtime/rpc/methods/notification-reconnect-cooldown.test.ts b/src/main/runtime/rpc/methods/notification-reconnect-cooldown.test.ts new file mode 100644 index 00000000000..acdf1f08d2d --- /dev/null +++ b/src/main/runtime/rpc/methods/notification-reconnect-cooldown.test.ts @@ -0,0 +1,65 @@ +import { expect, it } from 'vitest' +import { RuntimeMobileNotificationController } from '../../runtime-mobile-notification-controller' +import { NOTIFICATION_METHODS } from './notifications' +import type { RpcContext, RpcMethod, RpcStreamingMethod } from '../core' + +it.each([0, 255])( + 'preserves the live cooldown decision after %i intervening replay entries', + async (filler) => { + const controller = new RuntimeMobileNotificationController() + let stop!: () => void + const ctx = { + runtime: { + onNotificationDispatched: controller.onDispatched.bind(controller), + getMobileNotificationEpoch: controller.getEpoch.bind(controller), + getMissedNotificationsSince: controller.getMissedSince.bind(controller), + registerSubscriptionCleanup: (_id: string, cleanup: () => void) => { + stop = cleanup + } + } + } as unknown as RpcContext + const subscribe = NOTIFICATION_METHODS.find( + (m) => m.name === 'notifications.subscribe' + ) as RpcStreamingMethod + const replay = NOTIFICATION_METHODS.find( + (m) => m.name === 'notifications.getMissedSince' + ) as RpcMethod + const live: unknown[] = [] + const pending = subscribe.handler(undefined, ctx, (e) => live.push(e)) + try { + controller.dispatch({ + type: 'notification', + source: 'terminal-bell', + title: 'first', + body: '', + worktreeId: 'folder', + emittedAt: 10000 + }) + controller.dispatch({ + type: 'notification', + source: 'agent-task-complete', + title: 'suppressed', + body: '', + worktreeId: 'folder', + emittedAt: 10250 + }) + expect(live).toHaveLength(2) + for (let i = 0; i < filler; i++) { + controller.dispatch({ type: 'dismiss', notificationId: `other-${i}` }) + } + const result = (await replay.handler( + { lastSeenSeq: 1, epoch: controller.getEpoch() }, + ctx + )) as { notifications: { type: string }[] } + expect(result.notifications.filter((e) => e.type === 'notification')).toEqual([]) + const all = (await replay.handler( + { lastSeenSeq: 1, epoch: controller.getEpoch(), includeDesktopSuppressed: true }, + ctx + )) as { notifications: { title?: string }[] } + expect(all.notifications.some((e) => e.title === 'suppressed')).toBe(true) + } finally { + stop() + await pending + } + } +) diff --git a/src/main/runtime/rpc/methods/notification-stream-policy.ts b/src/main/runtime/rpc/methods/notification-stream-policy.ts new file mode 100644 index 00000000000..faee99aea02 --- /dev/null +++ b/src/main/runtime/rpc/methods/notification-stream-policy.ts @@ -0,0 +1,8 @@ +import type { MobileNotificationEvent } from '../../runtime-mobile-notification-controller' + +export function createNotificationStreamFilter(includeDesktopSuppressed = false) { + return (event: MobileNotificationEvent): boolean => + includeDesktopSuppressed || + event.type !== 'notification' || + (event.desktopAllowed !== false && event.legacySocketAllowed !== false) +} diff --git a/src/main/runtime/rpc/methods/notifications.ts b/src/main/runtime/rpc/methods/notifications.ts index 80c6af7caec..8168da70cae 100644 --- a/src/main/runtime/rpc/methods/notifications.ts +++ b/src/main/runtime/rpc/methods/notifications.ts @@ -1,46 +1,29 @@ -import { z } from 'zod' -import { defineStreamingMethod, defineMethod, type RpcAnyMethod } from '../core' +import { createNotificationStreamFilter } from './notification-stream-policy' +import { defineStreamingMethod, defineMethod } from '../core' +import { + NotificationGetMissedSinceParams, + NotificationRegisterPushParams, + NotificationUnsubscribeParams, + NotificationsSubscribeParams +} from '../../../../shared/rpc-contract/notifications-params' // Why: monotonically increasing per-process counter eliminates the // Date.now() collision that could fire when two near-simultaneous // notifications.subscribe calls landed on the same millisecond. let notificationsSubscriptionSeq = 0 -const NotificationUnsubscribeParams = z.object({ - subscriptionId: z - .unknown() - .transform((value) => (typeof value === 'string' && value.length > 0 ? value : '')) - .pipe(z.string().min(1, 'Missing subscriptionId')) -}) - -// Why: notifications.getMissedSince is the catch-up RPC for mobile reconnect -// (#8129). The client passes the highest seq it has already delivered; the -// runtime returns only notifications dispatched after that seq. Because the -// desktop assigns a monotonic seq to every dispatched notification, the cut is -// exact and idempotent — re-requesting with the same watermark can never -// return an already-delivered event, so reconnects never duplicate local -// pushes (the adversarial-review gate for #8129). -// `epoch` names the counter lifetime lastSeenSeq came from (#8591). The desktop's -// seq restarts at 0 on every launch while the client's watermark is persisted, so -// without it a post-restart watermark silently cuts away everything. Optional: a -// client that predates the field keeps the seq-only cut. -const NotificationGetMissedSinceParams = z.object({ - lastSeenSeq: z.number().int().min(0, 'lastSeenSeq must be a non-negative integer'), - epoch: z.string().optional() -}) - -// Why: notifications.subscribe streams desktop notification events to mobile -// clients over WebSocket. The mobile client shows a local push notification -// for each event. This avoids requiring Firebase/APNs — the existing -// persistent WebSocket connection doubles as the push channel. -export const NOTIFICATION_METHODS: readonly RpcAnyMethod[] = [ +// Legacy callers retain filtered socket alerts; push clients opt into the full event stream. +export const NOTIFICATION_METHODS = [ defineStreamingMethod({ name: 'notifications.subscribe', - params: null, - handler: async (_params, { runtime, connectionId }, emit) => { + params: NotificationsSubscribeParams, + handler: async (params, { runtime, connectionId }, emit) => { + const shouldEmit = createNotificationStreamFilter(params?.includeDesktopSuppressed) await new Promise((resolve) => { const unsubscribe = runtime.onNotificationDispatched((event) => { - emit(event) + if (shouldEmit(event)) { + emit(event) + } }) // Why: scope by per-ws connectionId + per-process counter so @@ -79,7 +62,41 @@ export const NOTIFICATION_METHODS: readonly RpcAnyMethod[] = [ // client missed while its socket was reaped. handler: async (params, { runtime }) => { const missed = runtime.getMissedNotificationsSince(params.lastSeenSeq, params.epoch) - return { notifications: missed, epoch: runtime.getMobileNotificationEpoch() } + return { + notifications: missed.filter( + createNotificationStreamFilter(params.includeDesktopSuppressed) + ), + epoch: runtime.getMobileNotificationEpoch(), + ...(params.deliveredPushes + ? { dismissedPushes: runtime.reconcileDismissedPushes(params.deliveredPushes) } + : {}) + } + } + }), + defineMethod({ + name: 'notifications.registerPush', + params: NotificationRegisterPushParams, + // Why: the registration is keyed by the revocable paired device identity, never + // by anything the caller can assert, so an in-process or CLI caller has no device + // to register and is refused outright. + handler: async (params, { runtime, clientKind, pairedDeviceId }) => { + if (clientKind !== 'mobile' || !pairedDeviceId) { + return { registered: false, reason: 'not_mobile' } + } + // The paired identity is spread last so no parameter can ever override it. + return await runtime.registerMobilePushDevice({ ...params, deviceId: pairedDeviceId }) + } + }), + defineMethod({ + name: 'notifications.unregisterPush', + params: null, + // Deleting the gateway token is durable (outbox), so an offline gateway still + // reports success to the phone that asked to stop being pushed to. + handler: async (_params, { runtime, clientKind, pairedDeviceId }) => { + if (clientKind !== 'mobile' || !pairedDeviceId) { + return { unregistered: false } + } + return await runtime.unregisterMobilePushDevice(pairedDeviceId) } }) ] diff --git a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts index 9cd9f51e0ac..92dc5c644a8 100644 --- a/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts +++ b/src/main/runtime/rpc/methods/orchestration-worker-start-mode.ts @@ -1,41 +1,59 @@ /** - * `orchestration.workerStart`'s view of the shared launch-mode decision. + * Which kind of worker `orchestration.workerStart` starts, decided from the user's own settings. * - * The decision itself lives in `main/agent-launch/agent-launch-mode`, which every launch surface - * shares — a worker is not a special kind of launch, it is the same launch with a dispatch - * attached. All this module contributes is the noun orchestration puts in its receipts ("worker") - * and the `--terminal` wording, so a dispatch receipt reads the way it always has. + * There is no `--structured` flag: if the user's default is that a new agent tab opens as a + * structured native chat, an orchestration worker is one too. That default is a preference, not a + * demand, so a dispatch it cannot apply to falls back to an ordinary PTY terminal worker and the + * receipt says which mode ran and why — a routine `worker-start` must never fail because the user + * happens to have a chat preference on. + * + * The settings default and the per-launch feasibility both come from + * `shared/structured-native-chat-launch-route`, the same module the renderer's + * `resolveAgentLaunchRoute` uses. This adapter supplies placement facts and formats the receipt; + * it does not own a second feasibility policy. */ +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { RUNTIME_CAPABILITIES } from '../../../../shared/protocol-version' import { - decideAgentLaunchMode, - downgradeAgentLaunchModeForHost, - readAgentLaunchModeSettings, - resolveAgentLaunchModeOnHost, - type AgentLaunchMode, - type AgentLaunchModeReason, - type AgentLaunchModeReceipt, - type AgentLaunchModeSettings, - type AgentLaunchModeVocabulary -} from '../../../agent-launch/agent-launch-mode' + prefersStructuredNativeChatByDefault, + resolveStructuredNativeChatSupport, + type NativeChatDefaultSettings, + type StructuredNativeChatBlocker +} from '../../../../shared/structured-native-chat-launch-route' import type { TuiAgent } from '../../../../shared/tui-agent' +import { hasExplicitTuiLaunchCustomization } from '../../../../shared/tui-agent-launch-customization' import type { OrcaRuntimeService } from '../../orca-runtime' -export type WorkerStartMode = AgentLaunchMode -export type WorkerStartModeReason = AgentLaunchModeReason -export type WorkerStartModeReceipt = AgentLaunchModeReceipt +export type WorkerStartMode = 'structured' | 'terminal' -/** Orchestration's receipts are read next to dispatch records, so they name the worker and the - * flag that reused a terminal. Pinned here because the exact strings are asserted. */ -export const WORKER_START_VOCABULARY: AgentLaunchModeVocabulary = { - structured: 'a structured chat session worker', - terminal: 'a terminal agent worker', - detailOverrides: { - remote_execution_host: 'this worker runs on a remote execution host', - reused_terminal: '--terminal reuses a running terminal agent' - } +export type WorkerStartModeReason = + | 'user_default' + | 'remote_execution_host' + | 'reused_terminal' + | 'agent_without_structured_session' + | 'tui_launch_customization' + | 'structured_sessions_unavailable' + | 'structured_support_unknown' + | 'wsl_execution_runtime' + | 'codex_on_windows' + | 'structured_unsupported_on_host' + +export type WorkerStartModeReceipt = { + /** The mode the worker actually started in. */ + mode: WorkerStartMode + /** The user's settings default for a new agent tab. */ + preferred: WorkerStartMode + reason: WorkerStartModeReason + /** One sentence, always present, so a fallback is never silent. */ + detail: string } +type WorkerStartModeSettings = Partial< + NativeChatDefaultSettings & + Pick +> + /** The placement options that exist only on `worker-start`. `worktree`, `model` and `effort` are * listed but no longer read: a structured worker honours all three, and naming them here keeps * the set of options this decision has considered visible. */ @@ -48,35 +66,152 @@ type WorkerStartModePlacement = { effort?: string } -export function decideWorkerStartMode(args: { - params: WorkerStartModePlacement - settings: AgentLaunchModeSettings | null | undefined -}): WorkerStartModeReceipt { - return decideAgentLaunchMode({ - placement: args.params, - settings: args.settings, - vocabulary: WORKER_START_VOCABULARY - }) +const DOWNGRADE_DETAIL: Record, string> = { + remote_execution_host: 'this worker runs on a remote execution host', + reused_terminal: '--terminal reuses a running terminal agent', + agent_without_structured_session: 'this agent has no structured session', + tui_launch_customization: + 'this agent has a custom launch command, arguments or environment that only a terminal applies', + structured_sessions_unavailable: 'this runtime does not support structured agent sessions', + structured_support_unknown: 'the execution host has not established structured session support', + wsl_execution_runtime: 'this workspace runs under WSL', + codex_on_windows: 'Codex has no structured session on Windows', + structured_unsupported_on_host: 'the execution host cannot create one here' } +const BLOCKER_REASON: Record< + StructuredNativeChatBlocker, + Exclude +> = { + 'reused-terminal': 'reused_terminal', + 'agent-without-structured-session': 'agent_without_structured_session', + 'floating-workspace': 'structured_unsupported_on_host', + 'tui-launch-customization': 'tui_launch_customization', + 'remote-execution-host': 'remote_execution_host', + 'project-runtime': 'wsl_execution_runtime', + 'runtime-capability': 'structured_sessions_unavailable', + 'runtime-capability-unknown': 'structured_support_unknown' +} + +/** The host's own create-support verdict (`agentSession.createSupport`) in this vocabulary. */ +const HOST_SUPPORT_REASON: Record< + 'agent' | 'remote' | 'wsl', + Exclude +> = { + agent: 'structured_unsupported_on_host', + remote: 'remote_execution_host', + wsl: 'wsl_execution_runtime' +} + +export function decideWorkerStartMode(args: { + params: WorkerStartModePlacement + settings: WorkerStartModeSettings | null | undefined +}): WorkerStartModeReceipt { + const { params, settings } = args + if (!prefersStructuredNativeChatByDefault(settings)) { + return { + mode: 'terminal', + preferred: 'terminal', + reason: 'user_default', + detail: 'Started a terminal agent worker, the default for new agent tabs in your settings.' + } + } + const agent = params.agent as TuiAgent + const support = resolveStructuredNativeChatSupport({ + agent, + executionHostId: params.on ? `runtime:${params.on}` : 'local', + reusesTerminal: Boolean(params.terminal), + hostCapabilities: RUNTIME_CAPABILITIES, + // Orchestration resolves a managed worktree or folder workspace; a floating terminal is never + // a worker placement. WSL is left to the executing host's own create-support probe, which + // reads the resolved workspace rather than guessing from a client-side project runtime. + requiresTuiLaunchCustomization: hasExplicitTuiLaunchCustomization(settings, agent) + }) + if (!support.supported) { + return downgraded(BLOCKER_REASON[support.blocker]) + } + return { + mode: 'structured', + preferred: 'structured', + reason: 'user_default', + detail: + 'Started a structured chat session worker, the default for new agent tabs in your settings.' + } +} + +/** + * Second half of the decision, once the worktree is resolved: the host that will run the worker + * answers whether it can create a structured session there at all. Asked before anything is + * created, so a refusal becomes a terminal worker rather than a failed start. + */ export async function resolveWorkerStartModeOnHost( runtime: Pick, mode: WorkerStartModeReceipt, worktreeId: string | undefined, agent: TuiAgent | undefined ): Promise { - return resolveAgentLaunchModeOnHost(runtime, mode, worktreeId, agent, WORKER_START_VOCABULARY) + if (mode.mode !== 'structured' || !worktreeId) { + return mode + } + return downgradeWorkerStartModeForHost( + mode, + await readStructuredCreateSupport(runtime, worktreeId, agent) + ) } +/** A host that cannot answer has not proved it can create one, so the worker stays a PTY agent. */ +async function readStructuredCreateSupport( + runtime: Pick, + worktreeId: string, + agent: TuiAgent | undefined +): Promise<{ supported: boolean; reason?: 'agent' | 'remote' | 'wsl' } | null> { + if (agent !== 'claude' && agent !== 'codex') { + return { supported: false, reason: 'agent' } + } + try { + return await runtime.getStructuredAgentSessionCreateSupport(`id:${worktreeId}`, agent) + } catch { + return null + } +} + +/** + * Applies the executing host's `agentSession.createSupport` answer, which is the authority on WSL, + * remoteness and the Windows process-start-time gate for the resolved workspace. + */ export function downgradeWorkerStartModeForHost( receipt: WorkerStartModeReceipt, support: { supported: boolean; reason?: 'agent' | 'remote' | 'wsl' } | null ): WorkerStartModeReceipt { - return downgradeAgentLaunchModeForHost(receipt, support, WORKER_START_VOCABULARY) + if (receipt.mode !== 'structured' || support?.supported) { + return receipt + } + if (support === null) { + return downgraded(BLOCKER_REASON['runtime-capability-unknown']) + } + return downgraded( + support.reason ? HOST_SUPPORT_REASON[support.reason] : 'structured_unsupported_on_host' + ) } +function downgraded( + reason: Exclude +): WorkerStartModeReceipt { + return { + mode: 'terminal', + preferred: 'structured', + reason, + detail: `Your default is a structured chat session, but ${DOWNGRADE_DETAIL[reason]}; started a terminal agent worker instead.` + } +} + +/** The store can be missing on a runtime that never opened one; that reads as no preference. */ export function readWorkerStartModeSettings( runtime: Pick -): AgentLaunchModeSettings | null { - return readAgentLaunchModeSettings(runtime) +): WorkerStartModeSettings | null { + try { + return runtime.getClientSettings() + } catch { + return null + } } diff --git a/src/main/runtime/rpc/methods/orchestration.ts b/src/main/runtime/rpc/methods/orchestration.ts index ed89ae4519d..ab80b91e830 100644 --- a/src/main/runtime/rpc/methods/orchestration.ts +++ b/src/main/runtime/rpc/methods/orchestration.ts @@ -1,4 +1,3 @@ -import type { RpcMethod } from '../core' import { ORCHESTRATION_RUN_METHODS } from './orchestration/runs/runs' import { ORCHESTRATION_WORKER_METHODS } from './orchestration/worker/worker-methods' import { ORCHESTRATION_FEDERATION_METHODS } from './orchestration/federation/federation-methods' @@ -11,7 +10,7 @@ import { ORCHESTRATION_ASK_METHODS } from './orchestration/messaging/ask-methods import { ORCHESTRATION_GATE_METHODS } from './orchestration/gates/gates' import { ORCHESTRATION_RESET_METHODS } from './orchestration/runs/reset-methods' -export const ORCHESTRATION_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_METHODS = [ ...ORCHESTRATION_RUN_METHODS, ...ORCHESTRATION_WORKER_METHODS, ...ORCHESTRATION_FEDERATION_METHODS, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts index d5150dc052e..87ac8f65356 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federated-release-safety.test.ts @@ -3,6 +3,7 @@ import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../../../shared/protoco import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' const HOME_FINGERPRINT = 'home-peer' const PANE_KEY = 'tab_remote:bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb' @@ -191,7 +192,9 @@ describe('federated worker release ownership', () => { dispatchId: string, params: Record = { dispatchId } ): Promise { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts index 6091c1080fa..caec51d462e 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-control.ts @@ -1,9 +1,6 @@ -import { z } from 'zod' -import { ORCHESTRATION_WORKER_READ_SOURCES } from '../../../../../../shared/orchestration-worker-output' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import type { RemoteDispatchAttachmentRow } from '../../../../orchestration/types' -import { defineMethod, type RpcMethod } from '../../../core' -import { OptionalFiniteNumber, requiredString } from '../../../schemas' +import { defineMethod } from '../../../core' import { mapWithConcurrency } from '../../../../../../shared/map-with-concurrency' import { readExactWorkerOutput } from '../worker/worker-output' import { describeUnconfirmedAgentStop } from '../../../../../../shared/pty-liveness-verdict' @@ -12,24 +9,14 @@ import { readRemoteAttachmentArchive, releaseRemoteAttachment } from './federated-worker-release-host' +import { + FederationDispatchParams, + FederationFleetSnapshotParams, + FederationOutputReadParams, + FederationReadParams +} from '../../../../../../shared/rpc-contract/orchestration-federation-control-params' -const FederationDispatchParams = z.object({ - dispatchId: requiredString('Missing Dispatch ID') -}) -const FederationReadParams = FederationDispatchParams.extend({ - cursor: OptionalFiniteNumber, - limit: OptionalFiniteNumber -}) -const FederationOutputReadParams = FederationDispatchParams.extend({ - cursor: z.union([z.number().int().nonnegative(), z.string().min(1).max(2_048)]).optional(), - limit: OptionalFiniteNumber, - source: z.enum(ORCHESTRATION_WORKER_READ_SOURCES).optional() -}) -const FederationFleetSnapshotParams = z.object({ - dispatchIds: z.array(requiredString('Missing Dispatch ID')).min(1).max(100) -}) - -export const ORCHESTRATION_FEDERATION_CONTROL_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_CONTROL_METHODS = [ defineMethod({ name: 'orchestration.federationFleetSnapshot', params: FederationFleetSnapshotParams, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts index 7c75c52eb6c..dc5989fff2a 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-liveness-verdict.test.ts @@ -4,6 +4,7 @@ import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../../../shared/protoco import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' // The federation host runs its own copy of the observation and stop logic, so // it needs the same rule: lost contact with a worker's host is not an exit, and @@ -87,7 +88,9 @@ describe('federation host liveness verdicts', () => { afterEach(() => db.close()) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } @@ -138,7 +141,9 @@ describe('federation host liveness verdicts', () => { }) hostDb.markRemoteAttachmentReady(DISPATCH_ID) const callHost = async (name: string, params: Record) => { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts index fbdbda6c7ca..589b69f3934 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-methods.ts @@ -1,9 +1,8 @@ -import type { RpcMethod } from '../../../core' import { ORCHESTRATION_FEDERATION_CONTROL_METHODS } from './federation-control' import { ORCHESTRATION_FEDERATION_RELAY_METHODS } from './federation-relay' import { ORCHESTRATION_FEDERATION_ATTACH_METHODS } from './federation' -export const ORCHESTRATION_FEDERATION_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_METHODS = [ ...ORCHESTRATION_FEDERATION_ATTACH_METHODS, ...ORCHESTRATION_FEDERATION_RELAY_METHODS, ...ORCHESTRATION_FEDERATION_CONTROL_METHODS diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts index 8cb4e08f2f3..721232f194a 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-relay.ts @@ -1,4 +1,3 @@ -import { z } from 'zod' import { ORCHESTRATION_FEDERATION_CONTROL_MAIL_PROTOCOL_VERSION } from '../../../../../../shared/protocol-version' import { importFederatedControlMessage } from '../../../../orchestration/federation-control-message' import { OrchestrationError } from '../../../../orchestration/orchestration-error' @@ -8,54 +7,13 @@ import { type FederatedLifecycleSettlement } from '../../../../orchestration/federation-lifecycle-settlement' import { defineMethod, type RpcMethod } from '../../../core' -import { OptionalFiniteNumber, requiredString } from '../../../schemas' +import { + FederationAckParams, + FederationImportParams, + FederationPullParams +} from '../../../../../../shared/rpc-contract/orchestration-federation-relay-params' -const FederationPullParams = z.object({ - dispatchId: requiredString('Missing Dispatch ID'), - afterSequence: OptionalFiniteNumber, - replayUnacknowledged: z.boolean().optional(), - limit: OptionalFiniteNumber -}) - -const FederationAckParams = z.object({ - dispatchId: requiredString('Missing Dispatch ID'), - throughSequence: z.number().int().nonnegative(), - settlements: z - .array( - z.object({ - sequence: z.number().int().positive(), - lifecycle: z.discriminatedUnion('action', [ - z.object({ - action: z.enum(['completed', 'failed']), - authority: z.literal('run_home') - }), - z.object({ - action: z.literal('rejected'), - code: z.string(), - reason: z.string(), - authority: z.literal('run_home') - }) - ]) - }) - ) - .optional() -}) - -const FederationImportParams = z.object({ - dispatchId: requiredString('Missing Dispatch ID'), - items: z.array( - z.object({ - dispatch_id: requiredString('Missing item Dispatch ID'), - direction: z.literal('to_worker'), - sequence: z.number().int().positive(), - message_id: requiredString('Missing relay message ID'), - kind: requiredString('Missing relay kind'), - payload: requiredString('Missing relay payload') - }) - ) -}) - -export const ORCHESTRATION_FEDERATION_RELAY_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_RELAY_METHODS = [ defineMethod({ name: 'orchestration.federationPull', params: FederationPullParams, diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts index 84ed57d58cc..89b55c86a4c 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation-start-schema.ts @@ -1,31 +1,5 @@ -import { z } from 'zod' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../../../schemas' -import { OptionalWorkerLaunchPreference } from '../worker/worker-start-schema' - -export const FederationAttachStartParams = z.object({ - /** Omitted by v1.4.198 coordinators; the worker host then mints a stub home Run. */ - runId: OptionalString, - dispatchId: requiredString('Missing Dispatch ID'), - taskId: requiredString('Missing Task ID'), - taskSpec: requiredString('Missing Task spec'), - /** Depth stamped by the Run home; omitted by older clients and defaults to 1. */ - depth: z.number().int().min(1).optional(), - protocolVersion: z.union([z.literal(1), z.literal(2), z.literal(3)]), - worktree: requiredString('Missing remote worktree selector'), - name: OptionalString, - repo: OptionalString, - baseBranch: OptionalString, - displayName: OptionalString, - displayNameKind: z.enum(['generated', 'user']).optional(), - comment: OptionalString, - setup: z.enum(['run', 'skip', 'inherit']).optional(), - setupSource: z.enum(['explicit_request', 'orchestration_default']).optional(), - terminal: OptionalString, - agent: OptionalString, - model: OptionalWorkerLaunchPreference, - effort: OptionalWorkerLaunchPreference, - timeoutMs: OptionalFiniteNumber, - devMode: z.boolean().optional() -}) +import type { z } from 'zod' +import { FederationAttachStartParams } from '../../../../../../shared/rpc-contract/orchestration-federation-start-params' +export { FederationAttachStartParams } export type FederationAttachStartInput = z.infer diff --git a/src/main/runtime/rpc/methods/orchestration/federation/federation.ts b/src/main/runtime/rpc/methods/orchestration/federation/federation.ts index 785f6a67eec..8d9d92547c8 100644 --- a/src/main/runtime/rpc/methods/orchestration/federation/federation.ts +++ b/src/main/runtime/rpc/methods/orchestration/federation/federation.ts @@ -1,7 +1,8 @@ import type { TuiAgent } from '../../../../../../shared/tui-agent' +import { describeTerminalWaitBlockedReason } from '../../../../../../shared/terminal-wait-blocked-reason-legacy-alias' import { buildDispatchPreamble } from '../../../../orchestration/preamble' import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { assertOrchestrationWorktreeCreationSupported } from '../worker/folder-worktree-placement' import { appendFederationSetupEffect, @@ -24,7 +25,7 @@ import { } from '../../../../../../shared/orchestration-timing-budgets' import { assertWorkerStartTaskSpecWithinPromptBudget } from '../worker/worker-start-prompt-budget' -export const ORCHESTRATION_FEDERATION_ATTACH_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_FEDERATION_ATTACH_METHODS = [ defineMethod({ name: 'orchestration.federationAttachStart', params: FederationAttachStartParams, @@ -222,7 +223,7 @@ export const ORCHESTRATION_FEDERATION_ATTACH_METHODS: RpcMethod[] = [ } throw new Error( wait.blockedReason - ? `Agent startup blocked: ${wait.blockedReason}` + ? `Agent startup blocked: ${describeTerminalWaitBlockedReason(wait.blockedReason)}` : `Agent did not become ready (${wait.status}).` ) } diff --git a/src/main/runtime/rpc/methods/orchestration/gates/gates.ts b/src/main/runtime/rpc/methods/orchestration/gates/gates.ts index 76bfd23b76e..29be91b4324 100644 --- a/src/main/runtime/rpc/methods/orchestration/gates/gates.ts +++ b/src/main/runtime/rpc/methods/orchestration/gates/gates.ts @@ -1,49 +1,22 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../../../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../../../schemas' +import { defineMethod } from '../../../core' import type { GateStatus } from '../../../../orchestration/db' import { Coordinator } from '../../../../orchestration/coordinator' import { resolveRunScope } from '../runs/run-scope' import { taskNotFoundError } from '../../../../orchestration/task-dispatch-refusal' +import { + GateCreateParams, + GateListParams, + GateResolveParams, + RunParams, + RunStopParams +} from '../../../../../../shared/rpc-contract/orchestration-gates-params' // Why: the coordinator instance is stored at module scope so orchestration.runStop // can signal it to halt. Only one coordinator can run at a time (enforced by // the DB's active-run check), so a single reference suffices. let activeCoordinator: Coordinator | null = null -const RunParams = z.object({ - spec: requiredString('Missing --spec'), - from: OptionalString, - pollIntervalMs: OptionalFiniteNumber, - maxConcurrent: OptionalFiniteNumber, - worktree: OptionalString -}) - -const RunStopParams = z.object({}) - -const GateCreateParams = z.object({ - task: requiredString('Missing --task'), - question: requiredString('Missing --question'), - options: OptionalString, - from: OptionalString, - run: OptionalString -}) - -const GateResolveParams = z.object({ - id: requiredString('Missing --id'), - resolution: requiredString('Missing --resolution'), - from: OptionalString, - run: OptionalString -}) - -const GateListParams = z.object({ - task: OptionalString, - status: z.enum(['pending', 'resolved', 'timeout']).optional(), - from: OptionalString, - run: OptionalString -}) - -export const ORCHESTRATION_GATE_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_GATE_METHODS = [ // Why: Section 4.12 — orchestration.run returns immediately with a run ID. // The coordinator loop runs in the background; progress is queried via // orchestration.taskList. This prevents the RPC call from blocking the diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts index e795b997930..ef191250a8b 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/ask-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { clampOrchestrationAskTimeoutMs } from '../../../../../../shared/orchestration-ask-timeout' import { isGroupAddress } from '../../../../orchestration/groups' @@ -6,7 +6,7 @@ import { AskParams } from '../schemas' import { rejectFederatedExplicitTarget } from '../routing' import { askRemoteRunHome } from './ask-remote' -export const ORCHESTRATION_ASK_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_ASK_METHODS = [ defineMethod({ name: 'orchestration.ask', params: AskParams, @@ -16,8 +16,9 @@ export const ORCHESTRATION_ASK_METHODS: RpcMethod[] = [ ) => { // Why: group addresses have no unambiguous first-answer authority. if (params.to && isGroupAddress(params.to)) { - throw new Error( - 'ask does not support group addresses; use send for non-blocking fan-out questions' + throw new OrchestrationError( + 'invalid_argument', + 'ask does not support group addresses; ask your owning run:, or use send for a non-blocking fan-out within your Run.' ) } diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/ask.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/ask.test.ts index 72bb588939e..4a46861a1c1 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/ask.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/ask.test.ts @@ -279,17 +279,19 @@ describe('orchestration RPC methods', () => { } }) - it('rejects group addresses with a dedicated error (no message persisted)', async () => { - setup() - await expect( - call('orchestration.ask', { - from: 'term_worker', - to: '@reviewers', - question: 'ok?' + it.each(['@all', '@idle', '@codex', '@reviewers'])( + 'rejects %s with invalid_argument naming the Run mailbox (no message persisted)', + async (to) => { + setup() + await expect( + call('orchestration.ask', { from: 'term_worker', to, question: 'ok?' }) + ).rejects.toMatchObject({ + code: 'invalid_argument', + message: expect.stringMatching(/does not support group addresses.*run:/) }) - ).rejects.toThrow(/does not support group addresses/) - expect(db.getInbox(10)).toHaveLength(0) - }) + expect(db.getInbox(10)).toHaveLength(0) + } + ) it('ignores unrelated wakes until the durable question is answered', async () => { setup() diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts index a12428253c3..20ac25e6512 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/check-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { CheckParams } from '../schemas' import { parseMessageTypes } from '../routing' @@ -12,7 +12,7 @@ import { isSupersededDispatch } from './dispatch-mailbox-fence' -export const ORCHESTRATION_CHECK_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_CHECK_METHODS = [ defineMethod({ name: 'orchestration.check', params: CheckParams, diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts index 58e0b3aa0ca..f0d5e4933a6 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/check-worker-federated-attachment.test.ts @@ -3,7 +3,7 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it, vi } from 'vitest' import { ORCHESTRATION_METHODS } from '../../orchestration' -import type { RpcContext } from '../../../core' +import { eraseRpcMethods, type RpcContext } from '../../../core' import { OrchestrationDb } from '../../../../orchestration/db' import { OrcaRuntimeService } from '../../../../orca-runtime' import { @@ -55,7 +55,9 @@ describe('orchestration.check on a federated attachment across a restart', () => } function check(ctx: RpcContext, params: Record = {}): Promise { - const method = ORCHESTRATION_METHODS.find((entry) => entry.name === 'orchestration.check') + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (entry) => entry.name === 'orchestration.check' + ) if (!method) { throw new Error('orchestration.check is not registered') } diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts index 61808c19aed..51a9d9bc0c5 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/message-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import type { TaskStatus } from '../../../../orchestration/db' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { ORCHESTRATION_LEGACY_RUN_ID } from '../../../../../../shared/orchestration-rpc-contract' @@ -19,7 +19,7 @@ import { TaskUpdateParams } from '../schemas' -export const ORCHESTRATION_MESSAGE_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_MESSAGE_METHODS = [ defineMethod({ name: 'orchestration.reply', params: ReplyParams, diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/recipient-routing.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/recipient-routing.test.ts index 958775c2254..cf78bfa6636 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/recipient-routing.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/recipient-routing.test.ts @@ -284,119 +284,128 @@ describe('orchestration recipient routing oracle', () => { expect(db.getInbox(100)).toEqual([]) }) - it.each(['@all', '@worktree:wt_target'])( - 'partially delivers %s when a listed recipient disappears before routing', - async (address) => { - setup() - vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ - terminals: [terminal('term_coord'), terminal('term_live'), terminal('term_disappeared')], - totalCount: 3, - truncated: false - }) - mockTerminalPaneKeys((handle) => - handle === 'term_coord' - ? harness.coordinatorPaneKey - : handle === 'term_live' - ? 'tab_live:leaf_live' - : null - ) - const adoptionLookup = vi.spyOn(db, 'getLegacyAdoptedRunMailboxOwner') - - const result = (await call({ - from: 'term_coord', - to: address, - subject: 'fan-out' - })) as GroupSendResult - - expect(result.messages).toHaveLength(1) - expect(result.messages[0]).toMatchObject({ to_handle: 'term_live' }) - expect(result.recipients).toBe(1) - expect(result.warnings?.map((warning) => warning.code).sort()).toEqual([ - 'legacy_terminal_recipient', - 'recipient_unreachable' - ]) - expect(db.getInbox(100)).toHaveLength(1) - expect(adoptionLookup).toHaveBeenCalledTimes(1) - } - ) - - it('fans out once when historical handles resolve to the same Run mailbox', async () => { + it('partially delivers @worktree: when a listed recipient disappears before routing', async () => { setup() - const foreignRun = db.createRun({ - objective: 'Foreign Run', - coordinatorHandle: 'term_foreign_first', - coordinatorPaneKey: 'tab_foreign_first:leaf_foreign_first' - }) - db.bindRun({ - runId: foreignRun.id, - coordinatorHandle: 'term_foreign_second', - coordinatorPaneKey: 'tab_foreign_second:leaf_foreign_second' - }) vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ - terminals: [ - terminal('term_coord'), - terminal('term_foreign_first'), - terminal('term_foreign_second') - ], + terminals: [terminal('term_coord'), terminal('term_live'), terminal('term_disappeared')], totalCount: 3, truncated: false }) + mockTerminalPaneKeys((handle) => + handle === 'term_coord' + ? harness.coordinatorPaneKey + : handle === 'term_live' + ? 'tab_live:leaf_live' + : null + ) + const adoptionLookup = vi.spyOn(db, 'getLegacyAdoptedRunMailboxOwner') + + const result = (await call({ + from: 'term_coord', + to: '@worktree:wt_target', + subject: 'fan-out' + })) as GroupSendResult + + expect(result.messages).toHaveLength(1) + expect(result.messages[0]).toMatchObject({ to_handle: 'term_live' }) + expect(result.recipients).toBe(1) + expect(result.warnings?.map((warning) => warning.code).sort()).toEqual([ + 'legacy_terminal_recipient', + 'recipient_unreachable' + ]) + expect(db.getInbox(100)).toHaveLength(1) + expect(adoptionLookup).toHaveBeenCalledTimes(1) + }) + + it('addresses @all recipients by Dispatch, so a vanished worker terminal still gets durable mail', async () => { + setup() + const task = db.createTask({ spec: 'worker whose pane closed' }) + const dispatch = createRootDispatch(db, task.id, 'term_gone') + vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ + terminals: [terminal('term_coord')], + totalCount: 1, + truncated: false + }) mockTerminalPaneKeys((handle) => (handle === 'term_coord' ? harness.coordinatorPaneKey : null)) const result = (await call({ from: 'term_coord', to: '@all', - subject: 'one mailbox' + subject: 'fan-out' })) as GroupSendResult - expect(result.recipients).toBe(1) - expect(result.messages).toHaveLength(1) - expect(result.messages[0]).toMatchObject({ - run_id: foreignRun.id, - to_handle: `run:${foreignRun.id}` + expect(result.messages).toEqual([ + expect.objectContaining({ to_handle: `dispatch:${dispatch.id}`, run_id: senderRunId }) + ]) + expect(result.warnings).toBeUndefined() + }) + + it('does not reach a Dispatch of another Run through @all, whatever terminals the host lists', async () => { + setup() + const foreignRun = db.createRun({ + objective: 'Foreign Run', + coordinatorHandle: 'term_foreign_coord', + coordinatorPaneKey: 'tab_foreign:leaf_foreign' }) + const foreignTask = db.createTask({ spec: 'foreign work', runId: foreignRun.id }) + createRootDispatch(db, foreignTask.id, 'term_foreign_worker') + const ownTask = db.createTask({ spec: 'own work' }) + const own = createRootDispatch(db, ownTask.id, 'term_own_worker') + vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ + terminals: [ + terminal('term_coord'), + terminal('term_foreign_worker'), + terminal('term_own_worker') + ], + totalCount: 3, + truncated: false + }) + + const result = (await call({ + from: 'term_coord', + to: '@all', + subject: 'mine only' + })) as GroupSendResult + + expect(result.messages.map((message) => message.to_handle)).toEqual([`dispatch:${own.id}`]) expect(db.getInbox(100)).toHaveLength(1) }) - it('excludes historical handles that resolve back to the sender mailbox', async () => { + it('skips a federated Dispatch with a warning naming the direct address', async () => { setup() - db.bindRun({ - runId: senderRunId, - coordinatorHandle: 'term_middle', - coordinatorPaneKey: 'tab_middle:leaf_middle' - }) - db.bindRun({ - runId: senderRunId, - coordinatorHandle: 'term_sender', - coordinatorPaneKey: 'tab_sender:leaf_sender' + const local = createRootDispatch(db, db.createTask({ spec: 'local' }).id, 'term_local') + const federated = db.createStartingWorkerDispatch({ + taskSpec: 'remote work', + taskRunId: senderRunId, + startOptions: {}, + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + federation: { + environmentId: 'environment_remote', + environmentName: 'remote', + peerFingerprint: 'remote_peer', + protocolVersion: 3 + } }) vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ - terminals: [ - terminal('term_sender'), - terminal('term_coord'), - terminal('term_middle'), - terminal('term_live') - ], - totalCount: 4, + terminals: [terminal('term_coord'), terminal('term_local')], + totalCount: 2, truncated: false }) - mockTerminalPaneKeys((handle) => - handle === 'term_sender' - ? 'tab_sender:leaf_sender' - : handle === 'term_live' - ? 'tab_live:leaf_live' - : null - ) const result = (await call({ - from: 'term_sender', + from: 'term_coord', to: '@all', - subject: 'exclude self aliases' + subject: 'fan-out' })) as GroupSendResult - expect(result.messages).toHaveLength(1) - expect(result.messages[0].to_handle).toBe('term_live') - expect(result.warnings).toMatchObject([{ code: 'legacy_terminal_recipient' }]) + expect(result.messages.map((message) => message.to_handle)).toEqual([`dispatch:${local.id}`]) + expect(result.warnings).toEqual([ + expect.objectContaining({ + code: 'recipient_unreachable', + recipient: `dispatch:${federated.dispatch.id}` + }) + ]) }) it('replays one honest receipt and discards retry receipts for rejected recipients', async () => { @@ -479,20 +488,13 @@ describe('orchestration recipient routing oracle', () => { it('rolls back a partial group insert before an idempotent retry', async () => { setup() + createRootDispatch(db, db.createTask({ spec: 'first' }).id, 'term_first') + createRootDispatch(db, db.createTask({ spec: 'second' }).id, 'term_second') vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ terminals: [terminal('term_coord'), terminal('term_first'), terminal('term_second')], totalCount: 3, truncated: false }) - mockTerminalPaneKeys((handle) => - handle === 'term_coord' - ? harness.coordinatorPaneKey - : handle === 'term_first' - ? 'tab_first:leaf_first' - : handle === 'term_second' - ? 'tab_second:leaf_second' - : null - ) const insertMessage = db.insertMessage.bind(db) vi.spyOn(db, 'insertMessage') .mockImplementationOnce(insertMessage) @@ -516,18 +518,12 @@ describe('orchestration recipient routing oracle', () => { it('replays a completed group receipt when notification fails after durable insertion', async () => { setup() + createRootDispatch(db, db.createTask({ spec: 'live' }).id, 'term_live') vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ terminals: [terminal('term_coord'), terminal('term_live')], totalCount: 2, truncated: false }) - mockTerminalPaneKeys((handle) => - handle === 'term_coord' - ? harness.coordinatorPaneKey - : handle === 'term_live' - ? 'tab_live:leaf_live' - : null - ) vi.spyOn(runtime, 'notifyMessageArrived').mockImplementationOnce(() => { throw new Error('injected notification failure') }) diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-group.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-group.test.ts new file mode 100644 index 00000000000..b076be06593 --- /dev/null +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-group.test.ts @@ -0,0 +1,631 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RpcContext } from '../../../core' +import { createOrchestrationRpcHarness } from '../rpc-test-harness' +import type { OrchestrationDb } from '../../../../orchestration/db' +import type { OrcaRuntimeService } from '../../../../orca-runtime' +import type { RuntimeTerminalSummary } from '../../../../../../shared/runtime-types' +import { createRootDispatch } from '../../../../orchestration/db/root-dispatch-test-fixture' + +// Group addresses mean the sender's Run. The host-wide meaning they had before let one +// coordinator's `@all` reach every terminal in every open project on the machine. +describe('orchestration.send group addresses', () => { + const h = createOrchestrationRpcHarness() + const { coordinatorPaneKey } = h + let db: OrchestrationDb + let runtime: OrcaRuntimeService + let ctx: RpcContext + let activeRunId: string | undefined + + function setup(withBoundRun = true): void { + ;({ db, runtime, ctx, activeRunId } = h.setup(withBoundRun)) + } + + afterEach(() => { + h.cleanup() + }) + + async function call(name: string, params: Record) { + return h.call(name, params, ctx) + } + + function makeSummary( + handle: string, + opts: Partial = {} + ): RuntimeTerminalSummary { + return { + handle, + ptyId: opts.ptyId ?? handle, + worktreeId: opts.worktreeId ?? 'wt_default', + worktreePath: opts.worktreePath ?? '/tmp/wt', + branch: opts.branch ?? 'main', + tabId: opts.tabId ?? 'tab_1', + leafId: opts.leafId ?? handle, + title: opts.title ?? null, + connected: opts.connected ?? true, + writable: opts.writable ?? true, + lastOutputAt: opts.lastOutputAt ?? null, + preview: opts.preview ?? '', + // Why spread: absent `agentIdentity` means unknown, so the helper must be able to + // produce a summary that genuinely lacks the field. + ...(opts.agentIdentity ? { agentIdentity: opts.agentIdentity } : {}) + } + } + + function setupWithTerminals( + terminals: RuntimeTerminalSummary[], + agentStatuses?: Record + ): void { + setup() + vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ + terminals, + totalCount: terminals.length, + truncated: false + }) + vi.mocked(runtime.getTerminalPaneKey).mockImplementation((handle) => { + if (handle === 'term_coord') { + return coordinatorPaneKey + } + const terminal = terminals.find((candidate) => candidate.handle === handle) + return terminal ? `${terminal.tabId}:${terminal.leafId}` : null + }) + vi.spyOn(runtime, 'getAgentStatusForHandle').mockImplementation( + (handle: string) => agentStatuses?.[handle] ?? null + ) + } + + /** A live worker Dispatch in `runId` whose terminal is `handle`. */ + function dispatchWorker(handle: string, runId = activeRunId!): string { + const task = db.createTask({ spec: `work for ${handle}`, runId }) + return createRootDispatch(db, task.id, handle).id + } + + type GroupReceipt = { + messages: { to_handle: string; run_id: string; thread_id: string }[] + recipients: number + warnings?: { code: string; recipient: string }[] + } + + it('fans out @all to the live Dispatches of the sender Run and nothing else', async () => { + // The defect this pins: a coordinator meaning "my three reviewers" reached 126 agents + // across every open project, because @all enumerated every terminal on the host. + setupWithTerminals([ + makeSummary('term_coord'), + makeSummary('term_a'), + makeSummary('term_b'), + makeSummary('term_other_project'), + makeSummary('term_plain_pane') + ]) + const dispatchA = dispatchWorker('term_a') + const dispatchB = dispatchWorker('term_b') + const otherRun = db.createRun({ + objective: 'Another project', + coordinatorHandle: 'term_other_coord', + coordinatorPaneKey: 'tab_other:leaf_other' + }) + dispatchWorker('term_other_project', otherRun.id) + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'broadcast' + })) as GroupReceipt + + expect(result.recipients).toBe(2) + expect(result.messages.map((m) => m.to_handle).sort()).toEqual( + [`dispatch:${dispatchA}`, `dispatch:${dispatchB}`].sort() + ) + expect(result.messages.every((m) => m.run_id === activeRunId)).toBe(true) + expect(result.warnings).toBeUndefined() + expect(db.getInbox(100)).toHaveLength(2) + }) + + it('leaves settled Dispatches out of @all', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a'), makeSummary('term_b')]) + const live = dispatchWorker('term_a') + db.completeDispatch(dispatchWorker('term_b')) + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'only the living' + })) as GroupReceipt + + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${live}`]) + }) + + it('reaches a Dispatch whose worker terminal is not attached yet', async () => { + // Durable delivery: the mailbox exists before the terminal does. + setupWithTerminals([makeSummary('term_coord')]) + const started = db.createStartingWorkerDispatch({ + taskSpec: 'starting worker', + taskRunId: activeRunId, + startOptions: {}, + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER + }) + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'early guidance' + })) as GroupReceipt + + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${started.dispatch.id}`]) + }) + + it('lets a worker address its Run siblings with @all, excluding itself', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a'), makeSummary('term_b')]) + dispatchWorker('term_a') + const sibling = dispatchWorker('term_b') + + const result = (await call('orchestration.send', { + from: 'term_a', + to: '@all', + subject: 'sibling ping' + })) as GroupReceipt + + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${sibling}`]) + }) + + it.each([false, true])( + 'addresses the nested coordinator child Run (explicit scope: %s)', + async (explicit) => { + // A nested coordinator is both a worker of its parent Run and the coordinator of the Run it + // created. It typed `@all` while coordinating, so it means the workers it started. Reaching + // its siblings instead is the wrong-audience delivery this whole change exists to remove, + // and it reports success, so the sender never learns its sub-workers heard nothing. + const nestedPane = 'tab_nested:11111111-1111-4111-8111-111111111111' + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_nested')]) + vi.mocked(runtime.getTerminalPaneKey).mockImplementation((handle) => + handle === 'term_coord' ? coordinatorPaneKey : handle === 'term_nested' ? nestedPane : null + ) + createRootDispatch( + db, + db.createTask({ spec: 'nested', runId: activeRunId }).id, + 'term_nested', + nestedPane + ) + const sibling = dispatchWorker('term_sibling') + const childRun = db.createRun({ + objective: 'child Run', + coordinatorHandle: 'term_nested', + coordinatorPaneKey: nestedPane + }) + const subWorker = dispatchWorker('term_sub', childRun.id) + + const result = (await call('orchestration.send', { + from: 'term_nested', + to: '@all', + ...(explicit ? { run: childRun.id } : {}), + subject: 'shared context' + })) as GroupReceipt + + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${subWorker}`]) + expect(result.messages.map((m) => m.to_handle)).not.toContain(`dispatch:${sibling}`) + } + ) + + it('names the remote workers it skipped when every live Dispatch is federated', async () => { + setupWithTerminals([makeSummary('term_coord')]) + const federated = db.createStartingWorkerDispatch({ + taskSpec: 'remote work', + taskRunId: activeRunId, + startOptions: {}, + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + federation: { + environmentId: 'environment_remote', + environmentName: 'remote', + peerFingerprint: 'remote_peer', + protocolVersion: 3 + } + }) + + // Without the skip explanation the sender is told "no recipients" while three remote + // workers exist and are each individually addressable. + await expect( + call('orchestration.send', { from: 'term_coord', to: '@all', subject: 'pause' }) + ).rejects.toMatchObject({ + code: 'terminal_not_found', + message: expect.stringContaining(`dispatch:${federated.dispatch.id}`) + }) + }) + + it.each(['@all', '@idle', '@codex'])( + 'rejects %s from a sender in no Run, naming the durable alternatives', + async (to) => { + setup(false) + const listTerminals = vi.spyOn(runtime, 'listTerminals') + vi.mocked(runtime.getTerminalPaneKey).mockImplementation((handle) => + handle === 'term_loner' ? 'tab_loner:leaf_loner' : null + ) + + await expect( + call('orchestration.send', { from: 'term_loner', to, subject: 'anyone?' }) + ).rejects.toMatchObject({ + code: 'invalid_argument', + message: expect.stringMatching(/run: or dispatch:/) + }) + + // No host-wide fallback: the host's terminals are never even enumerated. + expect(listTerminals).not.toHaveBeenCalled() + expect(db.getInbox(100)).toHaveLength(0) + } + ) + + it('rejects @all from a bound coordinator whose Run has no live Dispatch', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_bystander')]) + + await expect( + call('orchestration.send', { from: 'term_coord', to: '@all', subject: 'nobody home' }) + ).rejects.toThrow('No recipients resolved for group address') + expect(db.getInbox(100)).toHaveLength(0) + }) + + it('continues to fan out status messages to groups', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a'), makeSummary('term_b')]) + const dispatchA = dispatchWorker('term_a') + const dispatchB = dispatchWorker('term_b') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'status broadcast', + type: 'status' + })) as { messages: { to_handle: string; type: string }[]; recipients: number } + + expect(result.recipients).toBe(2) + expect(result.messages.map((m) => m.to_handle).sort()).toEqual( + [`dispatch:${dispatchA}`, `dispatch:${dispatchB}`].sort() + ) + expect(result.messages.every((m) => m.type === 'status')).toBe(true) + }) + + it('fans out @idle to only the idle Dispatches of the Run', async () => { + setupWithTerminals( + [ + makeSummary('term_coord'), + makeSummary('term_a'), + makeSummary('term_b'), + makeSummary('term_idle_elsewhere') + ], + { term_a: 'idle', term_b: 'busy', term_idle_elsewhere: 'idle' } + ) + const idle = dispatchWorker('term_a') + dispatchWorker('term_b') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@idle', + subject: 'idle check' + })) as GroupReceipt + + expect(result.recipients).toBe(1) + expect(result.messages[0].to_handle).toBe(`dispatch:${idle}`) + }) + + it('fans out an agent name group by host-resolved identity within the Run', async () => { + setupWithTerminals([ + makeSummary('term_coord', { agentIdentity: 'claude' }), + makeSummary('term_a', { agentIdentity: 'codex' }), + makeSummary('term_b', { agentIdentity: 'claude' }), + makeSummary('term_codex_elsewhere', { agentIdentity: 'codex' }) + ]) + const codex = dispatchWorker('term_a') + dispatchWorker('term_b') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@codex', + subject: 'codex only' + })) as GroupReceipt + + expect(result.recipients).toBe(1) + expect(result.messages[0].to_handle).toBe(`dispatch:${codex}`) + }) + + it('fans out @droid without claiming a pane whose title merely contains the word', async () => { + setupWithTerminals([ + makeSummary('term_coord', { agentIdentity: 'codex' }), + makeSummary('term_b', { agentIdentity: 'droid' }), + // Why kept: "Android build" contains `droid` as a substring. It was excluded before by + // whole-token matching and is excluded now because its identity is not droid. + makeSummary('term_c', { agentIdentity: 'claude', title: 'Android build' }) + ]) + const droid = dispatchWorker('term_b') + dispatchWorker('term_c') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@droid', + subject: 'droid only' + })) as GroupReceipt + + expect(result.recipients).toBe(1) + expect(result.messages[0].to_handle).toBe(`dispatch:${droid}`) + }) + + it('fans out @cursor without claiming a Claude pane discussing a text cursor', async () => { + // The original hazard: `@cursor` matched any pane whose TITLE contained "cursor", so a + // Claude pane titled "Fix the text cursor blink" received Cursor's instructions. + setupWithTerminals([ + makeSummary('term_coord', { agentIdentity: 'codex' }), + makeSummary('term_b', { agentIdentity: 'cursor' }), + makeSummary('term_c', { agentIdentity: 'claude', title: '✳ Fix the text cursor blink' }) + ]) + const cursor = dispatchWorker('term_b') + dispatchWorker('term_c') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@cursor', + subject: 'cursor only' + })) as GroupReceipt + + expect(result.recipients).toBe(1) + expect(result.messages[0].to_handle).toBe(`dispatch:${cursor}`) + }) + + it('fans out @worktree: to matching worktree terminals, unchanged by Run scoping', async () => { + setupWithTerminals([ + makeSummary('term_a', { worktreeId: 'wt_1' }), + makeSummary('term_b', { worktreeId: 'wt_1' }), + makeSummary('term_c', { worktreeId: 'wt_2' }) + ]) + + const result = (await call('orchestration.send', { + from: 'term_a', + to: '@worktree:wt_1', + subject: 'worktree msg' + })) as GroupReceipt + + expect(result.recipients).toBe(1) + expect(result.messages[0].to_handle).toBe('term_b') + }) + + it('shares thread_id across fan-out messages', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a'), makeSummary('term_b')]) + dispatchWorker('term_a') + dispatchWorker('term_b') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'threaded', + threadId: 'my_thread' + })) as GroupReceipt + + expect(result.messages[0].thread_id).toBe('my_thread') + expect(result.messages[1].thread_id).toBe('my_thread') + }) + + it('generates a shared thread_id when none provided', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a'), makeSummary('term_b')]) + dispatchWorker('term_a') + dispatchWorker('term_b') + + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'auto thread' + })) as GroupReceipt + + expect(result.messages[0].thread_id).toMatch(/^thread_/) + expect(result.messages[0].thread_id).toBe(result.messages[1].thread_id) + }) + it.each(['@all', '@idle'])('does not enumerate host terminals for %s', async (to) => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a')], { term_a: 'idle' }) + const worker = dispatchWorker('term_a') + const result = (await call('orchestration.send', { + from: 'term_coord', + to, + subject: 'guidance' + })) as GroupReceipt + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${worker}`]) + expect(runtime.listTerminals).not.toHaveBeenCalled() + }) + + it.each(['run', 'payload'])('does not acquire group membership from %s', async (source) => { + setupWithTerminals([ + makeSummary('term_coord'), + makeSummary('term_a'), + makeSummary('term_loner') + ]) + const worker = dispatchWorker('term_a') + const scope = + source === 'run' ? { run: activeRunId } : { payload: JSON.stringify({ dispatchId: worker }) } + await expect( + call('orchestration.send', { + from: 'term_loner', + to: '@all', + subject: 'outside sender', + ...scope + }) + ).rejects.toMatchObject({ code: 'invalid_argument' }) + expect(db.getInbox(100)).toHaveLength(0) + expect(runtime.listTerminals).not.toHaveBeenCalled() + }) + + it('rejects an explicit Run that conflicts with the group audience', async () => { + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_a')]) + dispatchWorker('term_a') + const other = db.createRun({ + objective: 'other', + coordinatorHandle: 'term_other', + coordinatorPaneKey: 'tab_other:leaf_other' + }) + dispatchWorker('term_b', other.id) + await expect( + call('orchestration.send', { + from: 'term_coord', + to: '@all', + run: other.id, + subject: 'explicit scope' + }) + ).rejects.toMatchObject({ code: 'invalid_argument' }) + expect(db.getInbox(100)).toHaveLength(0) + }) + + it.each(['@codex', '@idle'])( + '%s resolves a reminted worker handle by its stable pane', + async (to) => { + const pane = 'tab_worker:11111111-1111-4111-8111-111111111111' + setupWithTerminals( + [ + makeSummary('term_coord'), + makeSummary('term_new', { + tabId: 'tab_worker', + leafId: '11111111-1111-4111-8111-111111111111', + agentIdentity: 'codex' + }) + ], + { term_new: 'idle' } + ) + vi.spyOn(runtime, 'getTerminalHandleForPaneKey').mockImplementation((key) => + key === pane ? 'term_new' : null + ) + const task = db.createTask({ spec: 'surviving worker', runId: activeRunId }) + const dispatch = createRootDispatch(db, task.id, 'term_old', pane) + const result = (await call('orchestration.send', { + from: 'term_coord', + to, + subject: 'still reachable' + })) as GroupReceipt + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${dispatch.id}`]) + } + ) + + it('delivers a parent broadcast to the mailbox a nested coordinator reads', async () => { + const nestedPane = 'tab_nested:11111111-1111-4111-8111-111111111111' + setupWithTerminals([makeSummary('term_coord'), makeSummary('term_nested')]) + vi.mocked(runtime.getTerminalPaneKey).mockImplementation((handle) => + handle === 'term_coord' ? coordinatorPaneKey : handle === 'term_nested' ? nestedPane : null + ) + const dispatch = createRootDispatch( + db, + db.createTask({ spec: 'nested', runId: activeRunId }).id, + 'term_nested', + nestedPane + ) + const child = db.createRun({ + objective: 'child', + coordinatorHandle: 'term_nested', + coordinatorPaneKey: nestedPane + }) + const sent = (await call('orchestration.send', { + from: 'term_coord', + to: '@all', + subject: 'pause all work' + })) as GroupReceipt + expect(sent.messages.map((m) => m.to_handle)).toEqual([`run:${child.id}`]) + const checked = (await call('orchestration.check', { + terminal: 'term_nested', + peek: true + })) as { messages: { subject: string }[] } + expect(checked.messages.map((m) => m.subject)).toEqual(['pause all work']) + expect(db.getUnreadMessages(`dispatch:${dispatch.id}`)).toHaveLength(0) + }) + it.each(['@codex', '@idle'])('does not claim remote membership for %s', async (to) => { + setupWithTerminals( + [makeSummary('term_coord'), makeSummary('term_codex', { agentIdentity: 'codex' })], + { term_codex: 'idle' } + ) + const local = dispatchWorker('term_codex') + const remote = db.createStartingWorkerDispatch({ + taskSpec: 'remote work', + taskRunId: activeRunId, + startOptions: {}, + creator: { kind: 'system' }, + maxDepth: Number.MAX_SAFE_INTEGER, + federation: { + environmentId: 'remote', + environmentName: 'remote', + peerFingerprint: 'peer', + protocolVersion: 3 + } + }) + const result = (await call('orchestration.send', { + from: 'term_coord', + to, + subject: 'filtered guidance' + })) as GroupReceipt + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${local}`]) + expect(result.warnings).toBeUndefined() + expect(db.getUnreadMessages(`dispatch:${remote.dispatch.id}`)).toHaveLength(0) + + db.completeDispatch(local) + await expect( + call('orchestration.send', { + from: 'term_coord', + to, + subject: 'no known matches' + }) + ).rejects.toMatchObject({ + code: 'terminal_not_found', + message: `No recipients resolved for group address: ${to}` + }) + }) + it.each(['@all', '@idle', '@codex'])( + 'excludes an owning coordinator with a self-Dispatch from %s', + async (to) => { + setupWithTerminals( + [ + makeSummary('term_coord', { agentIdentity: 'codex' }), + makeSummary('term_a', { agentIdentity: 'codex' }), + makeSummary('term_b', { agentIdentity: 'codex' }) + ], + { term_coord: 'idle', term_a: 'idle', term_b: 'idle' } + ) + createRootDispatch( + db, + db.createTask({ spec: 'coordinator context', runId: activeRunId }).id, + 'term_coord', + coordinatorPaneKey + ) + dispatchWorker('term_a') + const sibling = dispatchWorker('term_b') + const result = (await call('orchestration.send', { + from: 'term_a', + to, + subject: 'siblings only' + })) as GroupReceipt + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${sibling}`]) + expect(db.getUnreadMessages(`run:${activeRunId}`)).toHaveLength(0) + } + ) + + it.each(['term_snapshot', 'term_original'])( + 'preserves pane identity across discovery when the recorded handle is %s', + async (recordedHandle) => { + const pane = 'tab_worker:11111111-1111-4111-8111-111111111111' + const snapshot = makeSummary('term_snapshot', { + tabId: 'tab_worker', + leafId: '11111111-1111-4111-8111-111111111111', + agentIdentity: 'codex' + }) + setupWithTerminals([makeSummary('term_coord'), snapshot]) + const dispatch = createRootDispatch( + db, + db.createTask({ spec: 'worker', runId: activeRunId }).id, + recordedHandle, + pane + ) + vi.spyOn(runtime, 'getTerminalHandleForPaneKey').mockImplementation((key) => + key === pane ? 'term_snapshot' : null + ) + vi.mocked(runtime.listTerminals).mockImplementation(async () => { + // The captured identity still belongs to this pane after its handle is reissued. + vi.mocked(runtime.getTerminalHandleForPaneKey).mockImplementation((key) => + key === pane ? 'term_new' : null + ) + return { terminals: [snapshot], totalCount: 1, truncated: false } + }) + const result = (await call('orchestration.send', { + from: 'term_coord', + to: '@codex', + subject: 'codex guidance' + })) as GroupReceipt + expect(result.messages.map((m) => m.to_handle)).toEqual([`dispatch:${dispatch.id}`]) + } + ) +}) diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-group.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-group.ts index 6681ce3236a..d5ee366a40f 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/send-group.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-group.ts @@ -2,18 +2,92 @@ import type { MessagePriority, MessageType, OrchestrationDb } from '../../../../ import type { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { resolveGroupAddress } from '../../../../orchestration/groups' +import { isEquivalentPaneKey } from '../../../../orchestration/db/pane-key-match' import { resolveBareOrchestrationRecipient } from './recipient-routing' -import { listAddressableStructuredWorkers } from '../../../../orchestration/structured-worker-group-addressing' +import { + listAddressableStructuredWorkers, + type OrchestrationAddressableAgent +} from '../../../../orchestration/structured-worker-group-addressing' import { legacyWorkerDeliveryContract } from '../routing' import { exposeMessages } from './mailbox-message-receipt' import { recordReceiptBeforeNudge } from './mutation-replay-nudge' -import type { SendRecipientWarning } from './recipient-routing' +import type { BareRecipientResolution, SendRecipientWarning } from './recipient-routing' import type { SendParams } from '../schemas' import type { z } from 'zod' type SendParamsInput = z.infer type SendReceipt = (receipt: T) => T & { warnings?: SendRecipientWarning[] } +type GroupAgentSnapshot = OrchestrationAddressableAgent & { tabId?: string; leafId?: string } + +/** Run candidates already identify a durable mailbox. */ +type GroupCandidate = OrchestrationAddressableAgent & { mailbox?: { to: string; runId: string } } + +function listRunGroupCandidates(args: { + db: OrchestrationDb + runtime: OrcaRuntimeService + senderRunId: string + groupAddress: string + agents: readonly GroupAgentSnapshot[] + warnings: SendRecipientWarning[] +}): GroupCandidate[] { + const { db, runtime, senderRunId, groupAddress, agents, warnings } = args + const live = db + .listWorkerTerminalResources({ runId: senderRunId }) + .filter((row) => row.dispatchStatus === 'pending' || row.dispatchStatus === 'dispatched') + // A federated worker reads relayed control mail, not this database's Dispatch mailbox. + const federated = new Set( + db.listFederatedDispatchesByIds(live.map((row) => row.dispatchId)).map((row) => row.dispatch_id) + ) + const identityByHandle = new Map(agents.map((agent) => [agent.handle, agent.agentIdentity])) + return live.flatMap((row) => { + const to = `dispatch:${row.dispatchId}` + if (federated.has(row.dispatchId)) { + // Remote identity and status are unknown, so only @all establishes membership. + if (groupAddress.toLowerCase() === '@all') { + warnings.push({ + code: 'recipient_unreachable', + recipient: to, + message: `${to} runs on a remote Orca server; group fan-out does not relay there. Send --to ${to} instead.` + }) + } + return [] + } + const paneKey = + row.paneKey ?? + (row.agentTerminalHandle ? runtime.getLiveTerminalPaneKey(row.agentTerminalHandle) : null) + const handle = + (paneKey ? runtime.getTerminalHandleForPaneKey(paneKey) : null) ?? + row.agentTerminalHandle ?? + to + // Nested coordinators consume their child Run mailbox, not their parent Dispatch mailbox. + const coordinated = paneKey ? db.getCurrentRunForPane(paneKey) : undefined + if (coordinated?.id === row.runId) { + return [] + } + // Discovery can precede a handle remint; the pane still owns the captured identity. + const agentIdentity = + identityByHandle.get(handle) ?? + agents.find( + (agent) => + paneKey && + agent.tabId && + agent.leafId && + isEquivalentPaneKey(`${agent.tabId}:${agent.leafId}`, paneKey) + )?.agentIdentity + return [ + { + handle, + worktreeId: row.worktreeId ?? '', + ...(agentIdentity ? { agentIdentity } : {}), + mailbox: coordinated + ? { to: `run:${coordinated.id}`, runId: coordinated.id } + : { to, runId: row.runId } + } + ] + }) +} + export async function sendGroupMessage(args: { params: SendParamsInput runtime: OrcaRuntimeService @@ -41,39 +115,83 @@ export async function sendGroupMessage(args: { revalidateLegacyCoordinator, recordMutationReceipt } = args - // Why: fan out one message per recipient (independent read-tracking) but share a thread_id for correlation (Section 4.5). - const { terminals } = await runtime.listTerminals(undefined, undefined, { - includeVisualLayouts: false - }) - // Structured workers are on no PTY surface, so `listTerminals` cannot see them and a broadcast - // silently missed every one. Composed here rather than inside `listTerminals`, whose result is - // published to paired clients and to consumers that assume a summary is writable. - const recipients = [...terminals, ...listAddressableStructuredWorkers()] - const handles = resolveGroupAddress(groupAddress, from, recipients, (handle: string) => + // Audience follows the sender's binding, never a caller-supplied message Run or payload. + function resolveAudienceRunId(): string { + const coordinated = senderPaneKey ? db.getCurrentRunForPane(senderPaneKey) : undefined + const runId = + coordinated?.id ?? + db.getActiveDispatchForIdentity(from, senderPaneKey)?.run_id ?? + legacyCoordinatorRunId + if (!runId) { + throw new OrchestrationError( + 'invalid_argument', + `${groupAddress} addresses the sender's Run, and ${from} is not bound to one. Send to run: or dispatch: instead.` + ) + } + if (explicitRunId && explicitRunId !== runId) { + throw new OrchestrationError( + 'invalid_argument', + `${groupAddress} addresses Run ${runId}, not explicitly requested Run ${explicitRunId}.` + ) + } + return runId + } + + // `@worktree:` names one workspace explicitly; every other group means the sender's Run. + const worktreeGroup = groupAddress.toLowerCase().startsWith('@worktree:') + let audienceRunId = worktreeGroup ? undefined : resolveAudienceRunId() + let agents: GroupAgentSnapshot[] = [] + if (worktreeGroup || !['@all', '@idle'].includes(groupAddress.toLowerCase())) { + const { terminals } = await runtime.listTerminals(undefined, undefined, { + includeVisualLayouts: false + }) + agents = [...terminals, ...listAddressableStructuredWorkers()] + } + // Revalidate after discovery before selecting recipients or writing mail. + revalidateLegacyCoordinator?.() + if (!worktreeGroup) { + audienceRunId = resolveAudienceRunId() + } + const groupWarnings: SendRecipientWarning[] = [] + const candidates: GroupCandidate[] = + worktreeGroup || !audienceRunId + ? agents + : listRunGroupCandidates({ + db, + runtime, + senderRunId: audienceRunId, + groupAddress, + agents, + warnings: groupWarnings + }) + const handles = resolveGroupAddress(groupAddress, from, candidates, (handle: string) => runtime.getAgentStatusForHandle(handle) ) if (handles.length === 0) { - throw new Error(`No recipients resolved for group address: ${groupAddress}`) + // Preserve the recovery addresses even when every worker was skipped. + const skipped = groupWarnings.map((warning) => warning.message).join(' ') + throw new OrchestrationError( + 'terminal_not_found', + `No recipients resolved for group address: ${groupAddress}${skipped ? ` ${skipped}` : ''}` + ) } const legacyAdoptedMailboxOwner = db.getLegacyAdoptedRunMailboxOwner() - const resolvedRecipients = handles.map((handle) => ({ - handle, - resolution: resolveBareOrchestrationRecipient({ - runtime, - db, - handle, - senderRunId, - explicitRunId, - legacyAdoptedMailboxOwner - }) - })) + const resolvedRecipients = handles.map((handle): BareRecipientResolution => { + const mailbox = candidates.find((candidate) => candidate.handle === handle)?.mailbox + return mailbox + ? { ok: true, to: mailbox.to, runId: mailbox.runId } + : resolveBareOrchestrationRecipient({ + runtime, + db, + handle, + senderRunId, + explicitRunId, + legacyAdoptedMailboxOwner + }) + }) const deliverableRecipients = resolvedRecipients.filter( - ( - recipient - ): recipient is typeof recipient & { - resolution: { ok: true; to: string; runId?: string; warning?: SendRecipientWarning } - } => recipient.resolution.ok + (recipient): recipient is BareRecipientResolution & { ok: true } => recipient.ok ) const senderRecipient = resolveBareOrchestrationRecipient({ runtime, @@ -86,7 +204,7 @@ export async function sendGroupMessage(args: { ? `${senderRecipient.runId ?? ''}\u0000${senderRecipient.to}` : undefined const seenMailboxes = new Set() - const uniqueRecipients = deliverableRecipients.filter(({ resolution }) => { + const uniqueRecipients = deliverableRecipients.filter((resolution) => { const mailboxKey = `${resolution.runId ?? ''}\u0000${resolution.to}` if (mailboxKey === senderMailboxKey || seenMailboxes.has(mailboxKey)) { return false @@ -101,10 +219,9 @@ export async function sendGroupMessage(args: { ) } - revalidateLegacyCoordinator?.() const threadId = params.threadId ?? `thread_${Date.now()}` const messages = db.insertMessages( - uniqueRecipients.map(({ resolution }) => ({ + uniqueRecipients.map((resolution) => ({ from, to: resolution.to, subject: params.subject, @@ -122,8 +239,10 @@ export async function sendGroupMessage(args: { ) })) ) - const groupWarnings = resolvedRecipients.flatMap(({ resolution }) => - resolution.ok ? (resolution.warning ? [resolution.warning] : []) : [resolution.warning] + groupWarnings.push( + ...resolvedRecipients.flatMap((resolution) => + resolution.ok ? (resolution.warning ? [resolution.warning] : []) : [resolution.warning] + ) ) const receipt = { messages: exposeMessages(messages), diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts index 5be1f7806ab..7d1edb77602 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { isGroupAddress } from '../../../../orchestration/groups' import { orchestrationSkillRecoveryData } from '../../../../../../shared/orchestration-rpc-contract' @@ -20,7 +20,7 @@ import { sendPointToPointMessage } from './send-point-to-point' import { sendGroupMessage } from './send-group' import { sendFederatedControlMail } from './send-control-mail' -export const ORCHESTRATION_SEND_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_SEND_METHODS = [ defineMethod({ name: 'orchestration.send', params: SendParams, @@ -80,12 +80,15 @@ export const ORCHESTRATION_SEND_METHODS: RpcMethod[] = [ }) } + const runGroup = + params.to && isGroupAddress(params.to) && !params.to.toLowerCase().startsWith('@worktree:') + // Run groups validate their own audience; message scope cannot select a parent Dispatch. const routing = resolveMessageRun(runtime, { from, senderPaneKey, to: params.to, - runId: params.run, - payload: params.payload + runId: runGroup ? undefined : params.run, + payload: runGroup ? undefined : params.payload }) if ( params.type === 'worker_done' && diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send-receipt-plumbing.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send-receipt-plumbing.test.ts index 02ad171926a..232a5992a87 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/send-receipt-plumbing.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send-receipt-plumbing.test.ts @@ -4,6 +4,7 @@ import type { OrchestrationDb } from '../../../../orchestration/db' import type { OrcaRuntimeService } from '../../../../orca-runtime' import type { RuntimeTerminalSummary } from '../../../../../../shared/runtime-types' import { createOrchestrationRpcHarness } from '../rpc-test-harness' +import { createRootDispatch } from '../../../../orchestration/db/root-dispatch-test-fixture' // The same delivery plumbing `check` already strips; a send/reply receipt is the same mailbox row. const INTERNAL_COLUMNS = [ @@ -62,20 +63,17 @@ describe('orchestration send and reply receipts', () => { it('keeps delivery plumbing out of a group send receipt', async () => { setup() - const terminals = [terminalSummary('term_a'), terminalSummary('term_b')] + const terminals = [terminalSummary('term_coord'), terminalSummary('term_worker')] vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ terminals, totalCount: terminals.length, truncated: false }) - vi.mocked(runtime.getTerminalPaneKey).mockImplementation((handle) => { - const terminal = terminals.find((candidate) => candidate.handle === handle) - return terminal ? `${terminal.tabId}:${terminal.leafId}` : null - }) + createRootDispatch(db, db.createTask({ spec: 'work' }).id, 'term_worker') const result = (await h.call( 'orchestration.send', - { from: 'term_a', to: '@all', subject: 'group plumbing' }, + { from: 'term_coord', to: '@all', subject: 'group plumbing' }, ctx )) as { messages: Record[] } diff --git a/src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts b/src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts index e30d2824897..bb57df1a6f5 100644 --- a/src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/messaging/send.test.ts @@ -5,7 +5,6 @@ import { RpcDispatcher } from '../../../dispatcher' import { createOrchestrationRpcHarness } from '../rpc-test-harness' import type { OrchestrationDb } from '../../../../orchestration/db' import type { OrcaRuntimeService } from '../../../../orca-runtime' -import type { RuntimeTerminalSummary } from '../../../../../../shared/runtime-types' import { ORCHESTRATION_CONTRACT_VERSION } from '../../../../../../shared/protocol-version' import { createRootDispatch } from '../../../../orchestration/db/root-dispatch-test-fixture' @@ -590,81 +589,6 @@ describe('orchestration RPC methods', () => { expect(db.getInbox(100)).toHaveLength(0) }) - function makeSummary( - handle: string, - opts: Partial = {} - ): RuntimeTerminalSummary { - return { - handle, - ptyId: opts.ptyId ?? handle, - worktreeId: opts.worktreeId ?? 'wt_default', - worktreePath: opts.worktreePath ?? '/tmp/wt', - branch: opts.branch ?? 'main', - tabId: opts.tabId ?? 'tab_1', - leafId: opts.leafId ?? handle, - title: opts.title ?? null, - connected: opts.connected ?? true, - writable: opts.writable ?? true, - lastOutputAt: opts.lastOutputAt ?? null, - preview: opts.preview ?? '', - // Why spread: absent `agentIdentity` means unknown, so the helper must be able to - // produce a summary that genuinely lacks the field. - ...(opts.agentIdentity ? { agentIdentity: opts.agentIdentity } : {}) - } - } - - function setupWithTerminals( - terminals: RuntimeTerminalSummary[], - agentStatuses?: Record - ): void { - setup() - vi.spyOn(runtime, 'listTerminals').mockResolvedValue({ - terminals, - totalCount: terminals.length, - truncated: false - }) - vi.mocked(runtime.getTerminalPaneKey).mockImplementation((handle) => { - if (handle === 'term_coord') { - return coordinatorPaneKey - } - const terminal = terminals.find((candidate) => candidate.handle === handle) - return terminal ? `${terminal.tabId}:${terminal.leafId}` : null - }) - vi.spyOn(runtime, 'getAgentStatusForHandle').mockImplementation( - (handle: string) => agentStatuses?.[handle] ?? null - ) - } - - it('fans out @all to all terminals except sender', async () => { - setupWithTerminals([makeSummary('term_a'), makeSummary('term_b'), makeSummary('term_c')]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@all', - subject: 'broadcast' - })) as { messages: { to_handle: string }[]; recipients: number } - - expect(result.recipients).toBe(2) - expect(result.messages).toHaveLength(2) - const recipients = result.messages.map((m) => m.to_handle).sort() - expect(recipients).toEqual(['term_b', 'term_c']) - }) - - it('continues to fan out status messages to groups', async () => { - setupWithTerminals([makeSummary('term_a'), makeSummary('term_b'), makeSummary('term_c')]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@all', - subject: 'status broadcast', - type: 'status' - })) as { messages: { to_handle: string; type: string }[]; recipients: number } - - expect(result.recipients).toBe(2) - expect(result.messages.map((m) => m.to_handle).sort()).toEqual(['term_b', 'term_c']) - expect(result.messages.every((m) => m.type === 'status')).toBe(true) - }) - it('rejects heartbeat group sends before inserting rows', async () => { setup() const listTerminals = vi.spyOn(runtime, 'listTerminals') @@ -728,132 +652,6 @@ describe('orchestration RPC methods', () => { ) }) - it('fans out @idle to only idle agents', async () => { - setupWithTerminals([makeSummary('term_a'), makeSummary('term_b'), makeSummary('term_c')], { - term_b: 'idle', - term_c: 'busy' - }) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@idle', - subject: 'idle check' - })) as { messages: { to_handle: string }[]; recipients: number } - - expect(result.recipients).toBe(1) - expect(result.messages[0].to_handle).toBe('term_b') - }) - - it('fans out an agent name group by host-resolved identity', async () => { - setupWithTerminals([ - makeSummary('term_a', { agentIdentity: 'claude' }), - makeSummary('term_b', { agentIdentity: 'claude' }), - makeSummary('term_c', { agentIdentity: 'codex' }) - ]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@claude', - subject: 'claude only' - })) as { messages: { to_handle: string }[]; recipients: number } - - expect(result.recipients).toBe(1) - expect(result.messages[0].to_handle).toBe('term_b') - }) - - it('fans out @droid without claiming a pane whose title merely contains the word', async () => { - setupWithTerminals([ - makeSummary('term_a', { agentIdentity: 'codex' }), - makeSummary('term_b', { agentIdentity: 'droid' }), - // Why kept: "Android build" contains `droid` as a substring. It was excluded before by - // whole-token matching and is excluded now because its identity is not droid. - makeSummary('term_c', { agentIdentity: 'claude', title: 'Android build' }) - ]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@droid', - subject: 'droid only' - })) as { messages: { to_handle: string }[]; recipients: number } - - expect(result.recipients).toBe(1) - expect(result.messages[0].to_handle).toBe('term_b') - }) - - it('fans out @cursor without claiming a Claude pane discussing a text cursor', async () => { - setupWithTerminals([ - makeSummary('term_a', { agentIdentity: 'codex' }), - makeSummary('term_b', { agentIdentity: 'cursor' }), - // The original hazard, now excluded structurally rather than by a bespoke predicate. - makeSummary('term_c', { agentIdentity: 'claude', title: '✳ Fix the text cursor blink' }) - ]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@cursor', - subject: 'cursor only' - })) as { messages: { to_handle: string }[]; recipients: number } - - expect(result.recipients).toBe(1) - expect(result.messages[0].to_handle).toBe('term_b') - }) - - it('fans out @worktree: to matching worktree', async () => { - setupWithTerminals([ - makeSummary('term_a', { worktreeId: 'wt_1' }), - makeSummary('term_b', { worktreeId: 'wt_1' }), - makeSummary('term_c', { worktreeId: 'wt_2' }) - ]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@worktree:wt_1', - subject: 'worktree msg' - })) as { messages: { to_handle: string }[]; recipients: number } - - expect(result.recipients).toBe(1) - expect(result.messages[0].to_handle).toBe('term_b') - }) - - it('shares thread_id across fan-out messages', async () => { - setupWithTerminals([makeSummary('term_a'), makeSummary('term_b'), makeSummary('term_c')]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@all', - subject: 'threaded', - threadId: 'my_thread' - })) as { messages: { thread_id: string }[] } - - expect(result.messages[0].thread_id).toBe('my_thread') - expect(result.messages[1].thread_id).toBe('my_thread') - }) - - it('generates a shared thread_id when none provided', async () => { - setupWithTerminals([makeSummary('term_a'), makeSummary('term_b'), makeSummary('term_c')]) - - const result = (await call('orchestration.send', { - from: 'term_a', - to: '@all', - subject: 'auto thread' - })) as { messages: { thread_id: string }[] } - - expect(result.messages[0].thread_id).toMatch(/^thread_/) - expect(result.messages[0].thread_id).toBe(result.messages[1].thread_id) - }) - - it('throws when group resolves to no recipients', async () => { - setupWithTerminals([makeSummary('term_a')]) - - await expect( - call('orchestration.send', { - from: 'term_a', - to: '@all', - subject: 'nobody home' - }) - ).rejects.toThrow('No recipients resolved for group address') - }) - it('releases dispatch lock before waking recipients when worker_done is sent via send', async () => { setup() const task = db.createTask({ spec: 'lock-release work' }) diff --git a/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts b/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts index dfba4bd143f..19e636eca21 100644 --- a/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts +++ b/src/main/runtime/rpc/methods/orchestration/rpc-test-harness.ts @@ -1,6 +1,6 @@ import { vi } from 'vitest' import { ORCHESTRATION_METHODS } from '../orchestration' -import type { RpcContext } from '../../core' +import { eraseRpcMethods, type RpcContext } from '../../core' import { OrchestrationDb } from '../../../orchestration/db' import { OrcaRuntimeService } from '../../../orca-runtime' @@ -66,7 +66,7 @@ export function createOrchestrationRpcHarness() { } function findMethod(name: string) { - const method = ORCHESTRATION_METHODS.find((m) => m.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find((m) => m.name === name) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts b/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts index abc941bf98c..a4e6b427d7d 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/dispatch-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { buildDispatchPreamble } from '../../../../orchestration/preamble' import { resolveDispatchCreator } from './dispatch-creator' @@ -10,7 +10,7 @@ import { import { resolveRunScope } from './run-scope' import { DispatchParams, DispatchShowParams } from '../schemas' -export const ORCHESTRATION_DISPATCH_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_DISPATCH_METHODS = [ defineMethod({ name: 'orchestration.dispatch', params: DispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts b/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts index 5dd72b61c6e..a1cf43ab424 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/mutation-request-show.ts @@ -2,13 +2,10 @@ import { describeMutationRequestState, type OrchestrationMutationRequestShowResult } from '../../../../../../shared/orchestration-mutation-request' -import { defineMethod, type RpcMethod } from '../../../core' -import { requiredString } from '../../../schemas' -import { z } from 'zod' +import { defineMethod } from '../../../core' +import { RequestShowParams } from '../../../../../../shared/rpc-contract/orchestration-runs-mutation-request-show-params' -const RequestShowParams = z.object({ request: requiredString('Missing --request') }) - -export const ORCHESTRATION_MUTATION_REQUEST_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_MUTATION_REQUEST_METHODS = [ defineMethod({ name: 'orchestration.requestShow', params: RequestShowParams, diff --git a/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts b/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts index b4be53ecad5..6946606651f 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/reset-methods.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { ResetParams } from '../schemas' -export const ORCHESTRATION_RESET_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_RESET_METHODS = [ defineMethod({ name: 'orchestration.reset', params: ResetParams, diff --git a/src/main/runtime/rpc/methods/orchestration/runs/runs.ts b/src/main/runtime/rpc/methods/orchestration/runs/runs.ts index 77bcea4924c..eab6eb2913e 100644 --- a/src/main/runtime/rpc/methods/orchestration/runs/runs.ts +++ b/src/main/runtime/rpc/methods/orchestration/runs/runs.ts @@ -1,30 +1,16 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../../../core' -import { OptionalBoolean, OptionalString, requiredString } from '../../../schemas' -import { ORCHESTRATION_RUN_PAGE_LIMIT } from '../../../../../../shared/orchestration-run-pagination' +import { defineMethod } from '../../../core' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import { assertCallerHandleMatchesEvidence, resolveOrchestrationCaller } from './run-scope' import { exposeRun } from './run-receipt' +import { + RunCreateParams, + RunCurrentParams, + RunListParams, + RunShowParams, + RunUseParams +} from '../../../../../../shared/rpc-contract/orchestration-runs-params' -const RunCreateParams = z.object({ - objective: requiredString('Missing --objective'), - from: requiredString('Missing coordinator terminal') -}) - -const RunUseParams = z.object({ - id: requiredString('Missing --id'), - from: requiredString('Missing coordinator terminal'), - takeoverLegacy: OptionalBoolean -}) - -const RunCurrentParams = z.object({ from: requiredString('Missing coordinator terminal') }) -const RunListParams = z.object({ - limit: z.number().int().min(1).max(ORCHESTRATION_RUN_PAGE_LIMIT).optional(), - cursor: z.string().min(1).optional() -}) -const RunShowParams = z.object({ id: requiredString('Missing --id'), from: OptionalString }) - -export const ORCHESTRATION_RUN_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_RUN_METHODS = [ defineMethod({ name: 'orchestration.runCreate', params: RunCreateParams, diff --git a/src/main/runtime/rpc/methods/orchestration/schemas.ts b/src/main/runtime/rpc/methods/orchestration/schemas.ts index 51b51137475..eae80849ffc 100644 --- a/src/main/runtime/rpc/methods/orchestration/schemas.ts +++ b/src/main/runtime/rpc/methods/orchestration/schemas.ts @@ -1,15 +1,26 @@ import { z } from 'zod' import { setImmediate as yieldToEventLoop } from 'node:timers/promises' -import { - OptionalFiniteNumber, - OptionalString, - OptionalBoolean, - requiredString -} from '../../schemas' +import { OptionalString, OptionalBoolean, requiredString } from '../../schemas' import type { TaskStatus } from '../../../orchestration/db' import { isGroupAddress } from '../../../orchestration/groups' import { MESSAGE_TYPES } from '../../../orchestration/types' import { OrchestrationError } from '../../../orchestration/orchestration-error' +import { + getLifecycleGroupRecipientError, + isDispatchMutationMessageType +} from '../../../../../shared/rpc-contract/orchestration-params' +export { + AskParams, + CheckParams, + DispatchParams, + DispatchShowParams, + InboxParams, + ReplyParams, + ResetParams, + TaskCreateParams, + TaskListParams +} from '../../../../../shared/rpc-contract/orchestration-params' +export { getLifecycleGroupRecipientError, isDispatchMutationMessageType } export const TASK_STATUSES: TaskStatus[] = [ 'pending', @@ -44,27 +55,6 @@ const SEND_MESSAGE_TYPE_ERROR = [ 'To answer a worker question, use the same Orca CLI executable with orchestration reply --id --body .' ].join(' ') -export type DispatchMutationMessageType = - | 'worker_done' - | 'heartbeat' - | 'escalation' - | 'decision_gate' - -export function isDispatchMutationMessageType( - type: string | undefined -): type is DispatchMutationMessageType { - return ( - type === 'worker_done' || - type === 'heartbeat' || - type === 'escalation' || - type === 'decision_gate' - ) -} - -export function getLifecycleGroupRecipientError(type: DispatchMutationMessageType): string { - return `${type} messages belong to one exact Dispatch and cannot target a group address.` -} - export function parseRemoteWorkerPayload(payload: string | undefined): Record { if (!payload) { return {} @@ -131,73 +121,6 @@ export const SendParams = z }) }) -export const CheckParams = z - .object({ - terminal: OptionalString, - terminalPaneKey: OptionalString, - unread: OptionalBoolean, - peek: OptionalBoolean, - // Why: `all` surfaces every message and skips mark-read; legacy encoding was the `{unread: false}` trick (design doc §3.2/§3.3). - all: OptionalBoolean, - types: OptionalString, - format: OptionalBoolean, - // Why: one-release RPC compatibility only; the public CLI uses --format because no terminal input is injected. - inject: OptionalBoolean, - ack: OptionalString, - compatibilityAck: OptionalString, - compatibilityQuestionAck: OptionalString, - compatibilityCliCommand: z.enum(['orca', 'orca-ide', 'orca-dev']).optional(), - run: OptionalString, - wait: OptionalBoolean, - timeoutMs: OptionalFiniteNumber - }) - .superRefine((params, ctx) => { - // Why: CLI encodes --peek as {peek:true, unread:false} for pre-peek runtimes, so that pair is one mode, not a conflict. - const modes = [ - params.unread === true, - params.peek === true, - params.all === true || (params.unread === false && params.peek !== true) - ].filter(Boolean) - if (modes.length > 1) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Choose at most one message read mode: --unread, --peek, or --all.' - }) - } - }) - -export const ReplyParams = z.object({ - id: requiredString('Missing --id'), - body: requiredString('Missing --body'), - from: OptionalString, - run: OptionalString -}) - -export const InboxParams = z.object({ - limit: OptionalFiniteNumber, - // Why: filters the inbox to a handle so inbox and check --all give agreeing results (design doc §3.3). - terminal: OptionalString -}) - -export const TaskCreateParams = z.object({ - spec: requiredString('Missing --spec'), - taskTitle: OptionalString, - displayName: OptionalString, - deps: OptionalString, - parent: OptionalString, - callerTerminalHandle: OptionalString, - run: OptionalString -}) - -export const TaskListParams = z.object({ - status: z.enum(['pending', 'ready', 'dispatched', 'completed', 'failed', 'blocked']).optional(), - ready: OptionalBoolean, - // Why: server-side truncation keeps --brief cheap over SSH/relay instead of shipping full specs the CLI throws away. - brief: OptionalBoolean, - run: OptionalString, - callerTerminalHandle: OptionalString -}) - export const TaskUpdateParams = z.object({ id: requiredString('Missing --id'), status: z @@ -217,61 +140,4 @@ export const TaskUpdateParams = z.object({ run: OptionalString, callerTerminalHandle: OptionalString }) - -export const DispatchParams = z.object({ - task: requiredString('Missing --task'), - // Why: --to is optional so --dry-run can preview without a target; the handler enforces presence before any side-effecting work. - to: OptionalString, - from: OptionalString, - inject: OptionalBoolean, - dryRun: OptionalBoolean, - returnPreamble: OptionalBoolean, - devMode: OptionalBoolean, - run: OptionalString -}) - -export const DispatchShowParams = z.object({ - task: OptionalString, - preamble: OptionalBoolean, - from: OptionalString, - devMode: OptionalBoolean -}) - -export const AskParams = z - .object({ - to: OptionalString, - question: OptionalString, - resume: OptionalString, - options: OptionalString, - timeoutMs: OptionalFiniteNumber, - from: OptionalString, - run: OptionalString, - compatibilityCliCommand: z.enum(['orca', 'orca-ide', 'orca-dev']).optional(), - compatibilityWindowsCommand: z.enum(['orca', 'orca-ide']).optional() - }) - .superRefine((params, ctx) => { - if ((params.question ? 1 : 0) + (params.resume ? 1 : 0) !== 1) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Choose exactly one of --question or --resume.' - }) - } - }) - -export const ResetParams = z - .object({ - all: OptionalBoolean, - tasks: OptionalBoolean, - messages: OptionalBoolean - }) - .superRefine((params, ctx) => { - const selectedScopeCount = [params.all, params.tasks, params.messages].filter( - (scope) => scope === true - ).length - if (selectedScopeCount !== 1) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Choose exactly one reset scope: --all, --tasks, or --messages.' - }) - } - }) +export type { DispatchMutationMessageType } from '../../../../../shared/rpc-contract/orchestration-params' diff --git a/src/main/runtime/rpc/methods/orchestration/worker/agent-status-producer-census.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/agent-status-producer-census.test.ts index 48fa39f5a31..61a28f4612c 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/agent-status-producer-census.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/agent-status-producer-census.test.ts @@ -94,6 +94,11 @@ const CENSUS: readonly CensusRow[] = [ kind: 'wiring', role: 'binds the hook server snapshot into the runtime deps' }, + { + path: 'main/orcad/orcad-entry.ts', + kind: 'wiring', + role: 'binds the same snapshot and structured sink into the headless orcad runtime deps' + }, { path: 'main/runtime/orca-runtime-state-fields.ts', kind: 'wiring', diff --git a/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts index aee45e25259..c9e0f077bc1 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/composed-workers.test.ts @@ -476,9 +476,15 @@ describe('orchestration RPC methods', () => { ) }) - it.each(['codex-update-prompt', 'codex-trust-workspace'] as const)( + // Why the second column: an older host still publishes the codex-* token, and this receipt + // reaches the user verbatim -- so it names the neutral spelling the same way the CLI does. + it.each([ + ['codex-update-prompt', 'codex-update-prompt (agent-update-prompt)'], + ['codex-trust-workspace', 'codex-trust-workspace (agent-trust-workspace)'], + ['agent-trust-workspace', 'agent-trust-workspace'] + ] as const)( 'returns a truthful readiness failure for %s', - async (blockedReason) => { + async (blockedReason, expectedReason) => { setup() mockCurrentWorkerStart() vi.mocked(runtime.waitForTerminal).mockResolvedValueOnce({ @@ -500,7 +506,7 @@ describe('orchestration RPC methods', () => { expect(result).toMatchObject({ state: 'failed', failedStage: 'agent_readiness', - lastError: `Agent startup blocked: ${blockedReason}` + lastError: `Agent startup blocked: ${expectedReason}` }) expect(runtime.sendTerminalAgentPrompt).not.toHaveBeenCalled() } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts b/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts index ec188695f5c..f8cd1033c97 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/local-worker-start.ts @@ -1,4 +1,5 @@ import type { OrcaRuntimeService } from '../../../../orca-runtime' +import { describeTerminalWaitBlockedReason } from '../../../../../../shared/terminal-wait-blocked-reason-legacy-alias' import type { OrchestrationDb } from '../../../../orchestration/db' import type { RunRow, TaskRow } from '../../../../orchestration/types' import { resolveDispatchCreator } from '../runs/dispatch-creator' @@ -186,7 +187,7 @@ export async function startLocalWorker(args: { } throw new Error( wait.blockedReason - ? `Agent startup blocked: ${wait.blockedReason}` + ? `Agent startup blocked: ${describeTerminalWaitBlockedReason(wait.blockedReason)}` : structuredSession ? `Setup did not finish before the structured worker started (${wait.status}).` : `Agent did not become ready (${wait.status}).` diff --git a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts index 2890fa08938..25da0b311a3 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-observation.test.ts @@ -3,6 +3,7 @@ import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' import { createRootDispatch } from '../../../../orchestration/db/root-dispatch-test-fixture' +import { eraseRpcMethods } from '../../../core' describe('manual Dispatch observation', () => { let db: OrchestrationDb | undefined @@ -51,7 +52,7 @@ describe('manual Dispatch observation', () => { coordinatorPaneKey }) const task = db.createTask({ spec: 'injected lane', runId: run.id }) - const dispatchMethod = ORCHESTRATION_METHODS.find( + const dispatchMethod = eraseRpcMethods(ORCHESTRATION_METHODS).find( (candidate) => candidate.name === 'orchestration.dispatch' ) if (!dispatchMethod) { @@ -77,7 +78,7 @@ describe('manual Dispatch observation', () => { capability_hash: expect.any(String) }) - const workerShowMethod = ORCHESTRATION_METHODS.find( + const workerShowMethod = eraseRpcMethods(ORCHESTRATION_METHODS).find( (candidate) => candidate.name === 'orchestration.workerShow' ) if (!workerShowMethod) { @@ -132,7 +133,9 @@ describe('manual Dispatch observation', () => { }) const context = { runtime } const call = async (name: string, params: Record) => { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Missing method ${name}`) } @@ -233,7 +236,7 @@ describe('manual Dispatch observation', () => { const task = db.createTask({ spec: 'operator lane', runId: run.id }) const dispatch = createRootDispatch(db, task.id, 'term_worker', 'tab_worker:leaf_worker') - const workerListMethod = ORCHESTRATION_METHODS.find( + const workerListMethod = eraseRpcMethods(ORCHESTRATION_METHODS).find( (candidate) => candidate.name === 'orchestration.workerList' ) if (!workerListMethod) { @@ -280,7 +283,9 @@ describe('manual Dispatch observation', () => { 'launch-hash', 'runtime_test:term_worker:1' ) - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Missing method ${name}`) diff --git a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts index ee1f5a3162a..e340e51b01d 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/manual-dispatch-release.test.ts @@ -3,6 +3,7 @@ import type Database from '../../../../../sqlite/sync-database' import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' const COORDINATOR = 'term_coordinator' const TARGET = 'term_target' @@ -177,7 +178,9 @@ describe('manual Dispatch release', () => { } async function call(name: string, params: Record): Promise { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts index cf08ce31f69..5e8babf3c5e 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-control.ts @@ -1,9 +1,6 @@ -import { z } from 'zod' -import { ORCHESTRATION_WORKER_READ_SOURCES } from '../../../../../../shared/orchestration-worker-output' import { contextOnlyAbandonWarning } from '../../../../orchestration/context-only-dispatch-release' import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' -import { OptionalFiniteNumber, requiredString } from '../../../schemas' +import { defineMethod } from '../../../core' import { exposeDispatchContext, exposeObservation, @@ -20,14 +17,12 @@ import { readExactWorkerOutput } from './worker-output' import { exposeWorkerTerminalResource } from './worker-release-completion' import { readFederatedWorkerOutput } from '../federation/federated-worker-read' import { showFederatedWorker } from '../federation/federated-worker-show' -const WorkerDispatchParams = z.object({ dispatch: requiredString('Missing --dispatch') }) -const WorkerReadParams = WorkerDispatchParams.extend({ - cursor: z.union([z.number().int().nonnegative(), z.string().min(1).max(2_048)]).optional(), - limit: OptionalFiniteNumber, - source: z.enum(ORCHESTRATION_WORKER_READ_SOURCES).optional() -}) +import { + WorkerDispatchParams, + WorkerReadParams +} from '../../../../../../shared/rpc-contract/orchestration-worker-control-params' -export const ORCHESTRATION_WORKER_CONTROL_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_CONTROL_METHODS = [ defineMethod({ name: 'orchestration.workerShow', params: WorkerDispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts index 5c8ae65fa86..c0f8097a690 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-list-method.ts @@ -4,7 +4,7 @@ import type { OrchestrationDb } from '../../../../orchestration/db' import { WORKER_LIST_CURSOR_EXPIRED_MESSAGE } from '../../../../orchestration/db/worker-terminal/worker-terminal-listing' import { OrchestrationError } from '../../../../orchestration/orchestration-error' import type { OrcaRuntimeService } from '../../../../orca-runtime' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { applyFederatedFleetObservations, readFederatedFleetSnapshots @@ -24,7 +24,7 @@ import { projectWorkerFleet, type WorkerListPageParams } from './worker-list-pro import { exposeWorkerTerminalResource } from './worker-release-completion' import { WORKER_TERMINAL_LIST_STATES, WorkerListParams } from './worker-release-schemas' -export const ORCHESTRATION_WORKER_LIST_METHOD: RpcMethod = defineMethod({ +export const ORCHESTRATION_WORKER_LIST_METHOD = defineMethod({ name: 'orchestration.workerList', params: WorkerListParams, handler: async (params, { runtime }) => { diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts index 238ad12fad8..7c324cdf798 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-methods.ts @@ -1,10 +1,9 @@ -import type { RpcMethod } from '../../../core' import { ORCHESTRATION_WORKER_CONTROL_METHODS } from './worker-control' import { ORCHESTRATION_WORKER_RELEASE_METHODS } from './worker-release' import { ORCHESTRATION_WORKER_STOP_METHODS } from './worker-stop' import { ORCHESTRATION_WORKER_START_METHODS } from './workers' -export const ORCHESTRATION_WORKER_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_METHODS = [ ...ORCHESTRATION_WORKER_START_METHODS, ...ORCHESTRATION_WORKER_CONTROL_METHODS, ...ORCHESTRATION_WORKER_STOP_METHODS, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts index 68d40b4a04e..dd4ce2522ea 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-mobile-report.test.ts @@ -2,7 +2,7 @@ import { afterEach, beforeEach, expect, it, vi } from 'vitest' import { createOrchestrationWorkerReleaseHarness } from './worker-release.test-support' import { TERMINAL_SEND_METHODS } from '../../terminal/terminal-send-method' import { sendTerminalStreamInput } from '../../terminal/terminal-input-delivery' -import { isStreamingMethod, type RpcMethod } from '../../../core' +import { eraseRpcMethods, isStreamingMethod, type RpcMethod } from '../../../core' const h = createOrchestrationWorkerReleaseHarness() beforeEach(() => h.setup()) @@ -93,7 +93,7 @@ it.each(['unary', 'stream'])('mobile %s bytes do no orchestration database work' 'delivered' ) } else { - const method = TERMINAL_SEND_METHODS.find( + const method = eraseRpcMethods(TERMINAL_SEND_METHODS).find( (m): m is RpcMethod => m.name === 'terminal.send' && !isStreamingMethod(m) )! await expect( diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts index a7481ea3b68..177eb479d42 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-recovery.test.ts @@ -2,7 +2,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { OrchestrationDb } from '../../../../orchestration/db' import { reconcileRequestedWorkerTerminalReleases } from '../../../../orchestration/worker-terminal-release-reconciliation' import { OrcaRuntimeService } from '../../../../orca-runtime' -import type { RpcContext } from '../../../core' +import { eraseRpcMethods, type RpcContext } from '../../../core' import { ORCHESTRATION_METHODS } from '../../orchestration' function deferred(): { promise: Promise; resolve: (value: T) => void } { @@ -101,7 +101,9 @@ describe('orchestration worker release recovery', () => { }) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-schemas.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-schemas.ts index 52310a2fd9b..66eaf2263f9 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release-schemas.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release-schemas.ts @@ -1,24 +1,6 @@ -import { z } from 'zod' -import { ORCHESTRATION_FLEET_PAGE_MAX } from '../../../../../../shared/orchestration-fleet-projection' -import { requiredString } from '../../../schemas' - -export const WorkerDispatchParams = z.object({ dispatch: requiredString('Missing --dispatch') }) -export const WorkerRetainParams = WorkerDispatchParams.strict() - -export const WORKER_TERMINAL_LIST_STATES = [ - 'active', - 'reclaimable', - 'retained', - 'release_pending', - 'release_unknown', - 'released' -] as const - -export const WorkerListParams = z.object({ - run: z.string().min(1).optional(), - terminalState: z.enum(WORKER_TERMINAL_LIST_STATES).optional(), - cursor: z.string().min(1).max(2_048).optional(), - limit: z.number().int().min(1).max(ORCHESTRATION_FLEET_PAGE_MAX).optional(), - includeRemote: z.boolean().optional(), - paginate: z.boolean().optional() -}) +export { + WORKER_TERMINAL_LIST_STATES, + WorkerDispatchParams, + WorkerListParams, + WorkerRetainParams +} from '../../../../../../shared/rpc-contract/orchestration-worker-release-schemas-params' diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts index ff1ea59a263..a13ea320670 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.test-support.ts @@ -1,6 +1,6 @@ import { expect, vi } from 'vitest' import { ORCHESTRATION_METHODS } from '../../orchestration' -import type { RpcContext } from '../../../core' +import { eraseRpcMethods, type RpcContext } from '../../../core' import { OrchestrationDb } from '../../../../orchestration/db' import { OrcaRuntimeService } from '../../../../orca-runtime' @@ -130,7 +130,7 @@ export function createOrchestrationWorkerReleaseHarness(): OrchestrationWorkerRe } function findMethod(name: string) { - const method = ORCHESTRATION_METHODS.find((m) => m.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find((m) => m.name === name) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts index e121f1b3f25..f641485e757 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-release.ts @@ -1,6 +1,5 @@ -import { z } from 'zod' import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { releaseFederatedWorker } from '../federation/federated-worker-release' import { ORCHESTRATION_WORKER_LIST_METHOD } from './worker-list-method' import { resolvePinnedFederatedServer } from './worker-observation' @@ -10,8 +9,9 @@ import { type WorkerReleaseReceipt } from './worker-release-completion' import { WorkerDispatchParams, WorkerRetainParams } from './worker-release-schemas' +import { OrchestrationWorkerTerminalUserInputParams } from '../../../../../../shared/rpc-contract/orchestration-worker-release-params' -export const ORCHESTRATION_WORKER_RELEASE_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_RELEASE_METHODS = [ defineMethod({ name: 'orchestration.workerRelease', params: WorkerDispatchParams, @@ -135,16 +135,7 @@ export const ORCHESTRATION_WORKER_RELEASE_METHODS: RpcMethod[] = [ // `sessionId` addresses a worker that IS a structured agent session. Its pane key is a random // identity credential that never leaves main, so the caller names the session and the owning // runtime resolves it — a renderer echoing the pane key back would make it learnable. - params: z - .object({ - paneKey: z.string().min(1).optional(), - sessionId: z.string().min(1).optional(), - terminal: z.string().min(1).optional() - }) - .refine( - (value) => Boolean(value.paneKey ?? value.sessionId ?? value.terminal), - 'Missing paneKey, sessionId or terminal' - ), + params: OrchestrationWorkerTerminalUserInputParams, // Real user keystrokes durably relinquish orchestration ownership on the owning runtime, so // restarts, SSH drops, remote viewing, and renderer remounts cannot erase the takeover. handler: (params, { runtime }) => { diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-schema.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-schema.ts index 2f9d9456609..b0f5328801d 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-start-schema.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-start-schema.ts @@ -1,63 +1,6 @@ -import { z } from 'zod' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../../../schemas' - -export const OptionalWorkerLaunchPreference = z - .string() - .min(1) - .max(512) - .refine((value) => value === value.trim(), 'Surrounding whitespace is invalid') - .optional() - -export const WorkerStartParams = z - .object({ - task: OptionalString, - spec: OptionalString, - taskTitle: OptionalString, - deps: OptionalString, - parent: OptionalString, - on: OptionalString, - run: OptionalString, - from: requiredString('Missing --from'), - worktree: OptionalString, - name: OptionalString, - repo: OptionalString, - baseBranch: OptionalString, - displayName: OptionalString, - comment: OptionalString, - setup: z.enum(['run', 'skip', 'inherit']).optional(), - terminal: OptionalString, - agent: OptionalString, - model: OptionalWorkerLaunchPreference, - effort: OptionalWorkerLaunchPreference, - retryOf: OptionalString, - timeoutMs: OptionalFiniteNumber, - devMode: z.boolean().optional() - }) - .superRefine((params, ctx) => { - if (!params.task && !params.spec) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['task'], - message: 'Missing --task or --spec' - }) - } - if (params.task && params.spec) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['spec'], - message: '--task and --spec are mutually exclusive' - }) - } - // Why: --spec creates a new Task, so a retry link to a prior Dispatch could never resolve and - // the refusal named a Task id the caller never supplied. - if (params.retryOf && params.spec) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - path: ['retryOf'], - message: - '--retry-of needs --task naming the failed Task; --spec creates a new one' - }) - } - }) +import type { z } from 'zod' +import { WorkerStartParams } from '../../../../../../shared/rpc-contract/orchestration-worker-start-params' +export { OptionalWorkerLaunchPreference } from '../../../../../../shared/rpc-contract/orchestration-worker-start-params' +export { WorkerStartParams } export type WorkerStartInput = z.infer diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts index f0b65281df1..c8dec854e4c 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop-liveness-verdict.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' // The aggregate terminal inventory only iterates registered providers, so a // dropped relay clears `connected` for every remote PTY at once. That is lost @@ -29,7 +30,9 @@ describe('worker-stop against a terminal we lost contact with', () => { afterEach(() => db.close()) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts index 643323cf62e..98d3c376f61 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/worker-stop.ts @@ -1,7 +1,5 @@ -import { z } from 'zod' import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' -import { requiredString } from '../../../schemas' +import { defineMethod } from '../../../core' import { describeUnconfirmedAgentStop } from '../../../../../../shared/pty-liveness-verdict' import { ORCHESTRATION_WORKER_STOP_VERDICT_RUNTIME_CAPABILITY } from '../../../../../../shared/protocol-version' import type { RuntimeStatus } from '../../../../../../shared/runtime-types' @@ -12,10 +10,9 @@ import { stopStructuredWorker } from '../../orchestration-structured-worker-lifecycle' import { isStructuredWorkerHandle } from '../../../../structured-worker-identity' +import { WorkerDispatchParams } from '../../../../../../shared/rpc-contract/orchestration-worker-stop-params' -const WorkerDispatchParams = z.object({ dispatch: requiredString('Missing --dispatch') }) - -export const ORCHESTRATION_WORKER_STOP_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_STOP_METHODS = [ defineMethod({ name: 'orchestration.workerStop', params: WorkerDispatchParams, diff --git a/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts b/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts index a635a316b23..9e95d7f33e2 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/workers-recovery.test.ts @@ -2,6 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { OrcaRuntimeService } from '../../../../orca-runtime' import { OrchestrationDb } from '../../../../orchestration/db' import { ORCHESTRATION_METHODS } from '../../orchestration' +import { eraseRpcMethods } from '../../../core' function deferred(): { promise: Promise; resolve: (value: T) => void } { let resolve!: (value: T) => void @@ -46,7 +47,9 @@ describe('orchestration worker recovery', () => { afterEach(() => db.close()) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/orchestration/worker/workers.ts b/src/main/runtime/rpc/methods/orchestration/worker/workers.ts index 8b14ec044cf..b1a1f40d45a 100644 --- a/src/main/runtime/rpc/methods/orchestration/worker/workers.ts +++ b/src/main/runtime/rpc/methods/orchestration/worker/workers.ts @@ -1,5 +1,5 @@ import { OrchestrationError } from '../../../../orchestration/orchestration-error' -import { defineMethod, type RpcMethod } from '../../../core' +import { defineMethod } from '../../../core' import { startFederatedWorker } from '../federation/federated-worker-start' import { startLocalWorker } from './local-worker-start' import { @@ -14,7 +14,7 @@ import { } from '../../../../../../shared/orchestration-timing-budgets' import { assertWorkerStartTaskSpecWithinPromptBudget } from './worker-start-prompt-budget' -export const ORCHESTRATION_WORKER_START_METHODS: RpcMethod[] = [ +export const ORCHESTRATION_WORKER_START_METHODS = [ defineMethod({ name: 'orchestration.workerStart', params: WorkerStartParams, diff --git a/src/main/runtime/rpc/methods/pairing.ts b/src/main/runtime/rpc/methods/pairing.ts index 7762881ba37..5a32ddab62f 100644 --- a/src/main/runtime/rpc/methods/pairing.ts +++ b/src/main/runtime/rpc/methods/pairing.ts @@ -1,10 +1,10 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { PairingGetEndpointsParamsSchema, PairingProvisionRelayParamsSchema } from '../../../../shared/mobile-relay-credential-contract' -export const PAIRING_METHODS: readonly RpcAnyMethod[] = [ +export const PAIRING_METHODS = [ defineMethod({ name: 'pairing.getEndpoints', params: PairingGetEndpointsParamsSchema, diff --git a/src/main/runtime/rpc/methods/plugins.test.ts b/src/main/runtime/rpc/methods/plugins.test.ts index bf67d29f19a..e44570bab56 100644 --- a/src/main/runtime/rpc/methods/plugins.test.ts +++ b/src/main/runtime/rpc/methods/plugins.test.ts @@ -1,12 +1,12 @@ import { afterEach, describe, expect, it, vi } from 'vitest' -import type { RpcContext, RpcMethod } from '../core' +import { eraseRpcMethods, type RpcContext, type RpcMethod } from '../core' import type { PluginService } from '../../../plugins/plugin-service' import { PLUGIN_METHODS, setPluginServiceForRpc } from './plugins' const SESSION_TOKEN = 's'.repeat(43) function method(name: string): RpcMethod { - const found = PLUGIN_METHODS.find((entry) => entry.name === name) + const found = eraseRpcMethods(PLUGIN_METHODS).find((entry) => entry.name === name) if (!found) { throw new Error(`missing ${name}`) } diff --git a/src/main/runtime/rpc/methods/plugins.ts b/src/main/runtime/rpc/methods/plugins.ts index bb035b984eb..667aff9179d 100644 --- a/src/main/runtime/rpc/methods/plugins.ts +++ b/src/main/runtime/rpc/methods/plugins.ts @@ -1,5 +1,4 @@ -import { z } from 'zod' -import { defineMethod, type RpcContext, type RpcMethod } from '../core' +import { defineMethod, type RpcContext } from '../core' import type { PluginPanelEntry } from '../../../../shared/plugins/plugin-panel-bridge' import { listPluginsForClients } from '../../../plugins/plugin-client-list' import type { PluginListEntry } from '../../../plugins/plugin-list-projection' @@ -8,7 +7,12 @@ import { pluginConsentRequestSchema, type PluginConsentRequest } from '../../../../shared/plugins/plugin-consent-request' -import { isQualifiedPluginKey } from '../../../../shared/plugins/plugin-manifest' +import { + PluginInvokeCommandParams, + PluginReadPanelEntryParams, + PluginSetEnabledParams, + PluginsPanelActionParams +} from '../../../../shared/rpc-contract/plugins-params' /** * Serve/headless parity surface: the same consent, enablement, panel-action, @@ -45,22 +49,6 @@ function requirePluginService(): PluginService { return pluginServiceForRpc } -const PluginSetEnabledParams = z.object({ - pluginKey: z.string().refine(isQualifiedPluginKey, 'invalid qualified plugin key'), - enabled: z.boolean() -}) - -const PluginReadPanelEntryParams = z.object({ - pluginKey: z.string().min(1), - panelId: z.string().min(1) -}) - -const PluginInvokeCommandParams = z.object({ - pluginKey: z.string().min(1), - commandId: z.string().min(1), - args: z.unknown().optional() -}) - async function listForRpc(): Promise { return listPluginsForClients(requirePluginService()) } @@ -77,7 +65,7 @@ function bindRpcPanelOwner(service: PluginService, context: RpcContext): string return ownerKey } -export const PLUGIN_METHODS: readonly RpcMethod[] = [ +export const PLUGIN_METHODS = [ defineMethod({ name: 'plugins.list', params: null, @@ -118,7 +106,7 @@ export const PLUGIN_METHODS: readonly RpcMethod[] = [ name: 'plugins.panelAction', // Why: raw admission must run before strict schema parsing so malformed // and oversized traffic cannot bypass the panel budget. - params: z.unknown(), + params: PluginsPanelActionParams, handler: async (params, context) => { const service = requirePluginService() await service.whenReady() diff --git a/src/main/runtime/rpc/methods/preflight.ts b/src/main/runtime/rpc/methods/preflight.ts index f863a1941d1..cc1dd5705c3 100644 --- a/src/main/runtime/rpc/methods/preflight.ts +++ b/src/main/runtime/rpc/methods/preflight.ts @@ -1,5 +1,4 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { detectRemoteAgents, detectRemoteWindowsTerminalCapabilities, @@ -7,18 +6,13 @@ import { refreshShellPathAndDetectAgents, runPreflightCheck } from '../../../preflight/agent-detection' +import { + PreflightCheck, + PreflightDetectRemoteAgents, + PreflightDetectRemoteWindowsTerminalCapabilities +} from '../../../../shared/rpc-contract/preflight-params' -const PreflightCheck = z.object({ - force: z.boolean().optional() -}) -const PreflightDetectRemoteAgents = z.object({ - connectionId: z.string().min(1) -}) -const PreflightDetectRemoteWindowsTerminalCapabilities = z.object({ - connectionId: z.string().min(1) -}) - -export const PREFLIGHT_METHODS: RpcMethod[] = [ +export const PREFLIGHT_METHODS = [ defineMethod({ name: 'preflight.check', params: PreflightCheck, diff --git a/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts b/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts index c25671d5bed..f67705f6cdd 100644 --- a/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts +++ b/src/main/runtime/rpc/methods/project-runtime-rpc-methods.ts @@ -1,102 +1,15 @@ -import { z } from 'zod' -import { - LOCAL_EXECUTION_HOST_ID, - normalizeExecutionHostId, - parseExecutionHostId -} from '../../../../shared/execution-host' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' import { projectRepoResultVisibilityForClient } from '../repo-visibility-projection' +import { + ProjectHostSetupClone, + ProjectHostSetupCreate, + ProjectHostSetupDelete, + ProjectHostSetupExistingFolder, + ProjectHostSetupUpdate, + ProjectUpdate +} from '../../../../shared/rpc-contract/project-runtime-params' -const ProjectProviderIdentity = z.object({ - provider: z.literal('github'), - owner: requiredString('Missing project owner'), - repo: requiredString('Missing project repository'), - host: OptionalString -}) - -// Why: `runtime:` ids are minted by the calling client's own pairing store -// (addEnvironmentFromPairingCode -> randomUUID), so they name a machine only relative to that -// client. A client sending one to this runtime is addressing *us*, and runtimes do not proxy -// these calls onward, so the host it names is this machine. Persisting the caller's id verbatim -// makes one machine look like a different host to every other client, hides its rows from them, -// and defeats the (projectId, hostId) duplicate check. Store our own spelling instead: `local`. -// Rows written before this normalization keep their client-minted stamp; readers still project -// `local` back to `runtime:`, so the client-visible model is unchanged. -const RequestedHostId = requiredString('Missing host ID').transform((value, ctx) => { - const hostId = normalizeExecutionHostId(value) - if (!hostId) { - ctx.addIssue({ code: 'custom', message: 'Invalid host ID' }) - return z.NEVER - } - return parseExecutionHostId(hostId)?.kind === 'runtime' ? LOCAL_EXECUTION_HOST_ID : hostId -}) - -const ProjectHostSetupExistingFolder = z.object({ - projectId: requiredString('Missing project ID'), - projectProviderIdentity: ProjectProviderIdentity.optional(), - hostId: RequestedHostId, - path: requiredString('Missing project path'), - kind: z.enum(['git', 'folder']).optional(), - displayName: OptionalString, - setupMethod: z.enum(['imported-existing-folder', 'cloned']).optional() -}) - -const ProjectHostSetupClone = z.object({ - projectId: requiredString('Missing project ID'), - projectProviderIdentity: ProjectProviderIdentity.optional(), - hostId: RequestedHostId, - url: requiredString('Missing clone URL'), - destination: requiredString('Missing clone destination'), - displayName: OptionalString -}) - -const LocalWindowsRuntimePreference = z.discriminatedUnion('kind', [ - z.object({ kind: z.literal('inherit-global') }), - z.object({ kind: z.literal('windows-host') }), - z.object({ kind: z.literal('wsl'), distro: requiredString('Missing WSL distro') }) -]) - -const ProjectUpdate = z.object({ - projectId: requiredString('Missing project ID'), - updates: z.object({ - localWindowsRuntimePreference: LocalWindowsRuntimePreference.optional() - }) -}) - -const ProjectHostSetupCreate = z.object({ - projectId: requiredString('Missing project ID'), - hostId: RequestedHostId, - setupId: OptionalString, - path: OptionalString, - kind: z.enum(['git', 'folder']).optional(), - displayName: OptionalString, - worktreeBasePath: OptionalString, - gitUsername: OptionalString, - setupState: z.enum(['ready', 'not-set-up', 'setting-up', 'error', 'unsupported']).optional(), - setupMethod: z.enum(['imported-existing-folder', 'cloned', 'provisioned']).optional() -}) - -const ProjectHostSetupUpdate = z.object({ - setupId: requiredString('Missing setup ID'), - updates: z.object({ - displayName: OptionalString, - path: OptionalString, - worktreeBasePath: OptionalString, - setupState: z.enum(['ready', 'not-set-up', 'setting-up', 'error', 'unsupported']).optional(), - setupMethod: z - .enum(['legacy-repo', 'imported-existing-folder', 'cloned', 'provisioned']) - .optional(), - gitUsername: OptionalString, - kind: z.enum(['git', 'folder']).optional() - }) -}) - -const ProjectHostSetupDelete = z.object({ - setupId: requiredString('Missing setup ID') -}) - -export const PROJECT_RUNTIME_METHODS: RpcMethod[] = [ +export const PROJECT_RUNTIME_METHODS = [ defineMethod({ name: 'project.list', params: null, diff --git a/src/main/runtime/rpc/methods/repo-update-schema.ts b/src/main/runtime/rpc/methods/repo-update-schema.ts index b613b35d8ba..f4e189f84a7 100644 --- a/src/main/runtime/rpc/methods/repo-update-schema.ts +++ b/src/main/runtime/rpc/methods/repo-update-schema.ts @@ -1,76 +1,4 @@ -import { z } from 'zod' -import { OptionalFiniteNumber, OptionalString } from '../schemas' -import { sanitizeRepoIcon } from '../../../../shared/repo-icon' -import { normalizeRepoBadgeColor } from '../../../../shared/repo-badge-color' -import { normalizeRepoSourceControlAiOverrides } from '../../../../shared/source-control-ai' -import { - normalizeCustomWorktreeVisibilitySources, - normalizeWorktreeVisibilitySourcePreferences -} from '../../../../shared/worktree/visibility-sources' - -export const RepoSourceControlAiOverrides = z - .unknown() - .optional() - .transform((value) => - value === undefined - ? undefined - : value === null - ? null - : normalizeRepoSourceControlAiOverrides(value) - ) - -const RepoBadgeColor = z - .unknown() - .optional() - .transform((value) => - value === undefined ? undefined : (normalizeRepoBadgeColor(value) ?? undefined) - ) - -const RepoUpstream = z - .object({ - owner: z.string().min(1), - repo: z.string().min(1) - }) - .nullable() - .optional() - -export function createRepoUpdateSchema( - selectorShape: T -): z.ZodObject }> { - return z.object({ - ...selectorShape, - updates: z.object({ - displayName: OptionalString, - badgeColor: RepoBadgeColor, - repoIcon: z - .unknown() - .transform((value) => sanitizeRepoIcon(value)) - .optional(), - upstream: RepoUpstream, - hookSettings: z.unknown().optional(), - worktreeBaseRef: OptionalString, - worktreeBasePath: OptionalString, - kind: z.enum(['git', 'folder']).optional(), - symlinkPaths: z.array(z.string()).optional(), - issueSourcePreference: z.enum(['auto', 'upstream', 'origin']).optional(), - forkSyncMode: z.enum(['ask', 'safe-auto', 'off']).optional(), - externalWorktreeVisibility: z.enum(['hide', 'show']).nullable().optional(), - externalWorktreeVisibilityPromptDismissedAt: z.number().finite().optional(), - externalWorktreeInboxBaselinePaths: z.array(z.string()).optional(), - importedExternalWorktreePaths: z.array(z.string()).optional(), - agentWorktreeVisibility: z.enum(['hide', 'show']).nullable().optional(), - customWorktreeVisibilitySources: z - .unknown() - .transform((value) => normalizeCustomWorktreeVisibilitySources(value)) - .optional(), - worktreeVisibilitySourcePreferences: z - .unknown() - .transform((value) => normalizeWorktreeVisibilitySourcePreferences(value)) - .optional(), - externalWorktreeDiscoverySuppressedAt: z.number().finite().nullable().optional(), - projectGroupId: OptionalString.nullable().optional(), - projectGroupOrder: OptionalFiniteNumber, - sourceControlAi: RepoSourceControlAiOverrides - }) - }) as z.ZodObject }> -} +export { + RepoSourceControlAiOverrides, + createRepoUpdateSchema +} from '../../../../shared/rpc-contract/repo-update-params' diff --git a/src/main/runtime/rpc/methods/repo.ts b/src/main/runtime/rpc/methods/repo.ts index 7bf42922db8..6498de8a4f3 100644 --- a/src/main/runtime/rpc/methods/repo.ts +++ b/src/main/runtime/rpc/methods/repo.ts @@ -1,115 +1,30 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' import { PROJECT_RUNTIME_METHODS } from './project-runtime-rpc-methods' import { FOLDER_WORKSPACE_METHODS } from './folder-workspace' -import { createRepoUpdateSchema } from './repo-update-schema' +import { RepoSelector } from './github-repo-target-schemas' import { projectRepoResultVisibilityForClient, projectRepoVisibilityForClient } from '../repo-visibility-projection' +import { + ProjectGroupCreate, + ProjectGroupImportNested, + ProjectGroupMoveProject, + ProjectGroupScanNested, + ProjectGroupSelector, + ProjectGroupUpdate, + RepoClone, + RepoCreate, + RepoIssueCommandWrite, + RepoPath, + RepoReorder, + RepoSearchRefs, + RepoSetBaseRef, + RepoSparsePresetSave, + RepoUpdate +} from '../../../../shared/rpc-contract/repo-params' -const RepoSelector = z.object({ - repo: requiredString('Missing repo selector') -}) - -const RepoPath = z.object({ - path: requiredString('Missing repo path'), - kind: z.enum(['git', 'folder']).optional(), - displayName: OptionalString -}) - -const RepoCreate = z.object({ - parentPath: requiredString('Missing parent path'), - name: requiredString('Missing repo name'), - kind: z.enum(['git', 'folder']).optional() -}) - -const RepoClone = z.object({ - url: requiredString('Missing clone URL'), - destination: requiredString('Missing clone destination') -}) - -const RepoSetBaseRef = z.object({ - repo: requiredString('Missing repo selector'), - ref: requiredString('Missing base ref') -}) - -const RepoUpdate = createRepoUpdateSchema(RepoSelector.shape) - -const RepoSearchRefs = z.object({ - repo: requiredString('Missing repo selector'), - query: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : undefined)) - .pipe(z.string({ message: 'Missing query' })), - limit: OptionalFiniteNumber -}) - -const RepoReorder = z.object({ - orderedIds: z.array(z.string()) -}) - -const ProjectGroupCreate = z.object({ - name: requiredString('Missing group name'), - parentPath: OptionalString, - connectionId: OptionalString.nullable().optional(), - parentGroupId: OptionalString.nullable().optional(), - createdFrom: z.enum(['manual', 'folder-scan', 'migration']).optional() -}) - -const ProjectGroupUpdate = z.object({ - groupId: requiredString('Missing group id'), - updates: z.object({ - name: OptionalString, - isCollapsed: z.boolean().optional(), - tabOrder: OptionalFiniteNumber, - color: OptionalString.nullable().optional() - }) -}) - -const ProjectGroupSelector = z.object({ - groupId: requiredString('Missing group id') -}) - -const ProjectGroupMoveProject = z.object({ - repo: requiredString('Missing repo selector'), - groupId: OptionalString.nullable(), - order: OptionalFiniteNumber -}) - -const ProjectGroupScanNested = z.object({ - path: requiredString('Missing folder path') -}) - -const ProjectGroupImportNested = z.discriminatedUnion('mode', [ - z.object({ - parentPath: requiredString('Missing parent path'), - groupName: z.string().optional().default(''), - projectPaths: z.array(z.string()), - mode: z.literal('group') - }), - z.object({ - parentPath: requiredString('Missing parent path'), - // Why: blank group names fall back to the scanned folder basename; separate - // imports do not create a group but share the same renderer payload shape. - groupName: z.string().optional().default(''), - projectPaths: z.array(z.string()), - mode: z.literal('separate') - }) -]) - -const RepoIssueCommandWrite = RepoSelector.extend({ - content: z.string() -}) - -const RepoSparsePresetSave = RepoSelector.extend({ - id: OptionalString, - name: requiredString('Missing preset name'), - directories: z.array(z.string()) -}) - -export const REPO_METHODS: RpcMethod[] = [ +export const REPO_METHODS = [ defineMethod({ name: 'repo.list', params: null, diff --git a/src/main/runtime/rpc/methods/runtime-client-capabilities.ts b/src/main/runtime/rpc/methods/runtime-client-capabilities.ts index a1ab53267b3..2fa62b53934 100644 --- a/src/main/runtime/rpc/methods/runtime-client-capabilities.ts +++ b/src/main/runtime/rpc/methods/runtime-client-capabilities.ts @@ -1,14 +1,8 @@ -import { z } from 'zod' import type { RuntimeCapability } from '../../../../shared/protocol-version' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' +import { ClientCapabilitiesUpdate } from '../../../../shared/rpc-contract/runtime-client-capabilities-params' -const ClientCapabilitiesUpdate = z - .object({ - clientCapabilities: z.array(z.string().min(1).max(128)).max(64) - }) - .strict() - -export const RUNTIME_CLIENT_CAPABILITY_METHODS: RpcAnyMethod[] = [ +export const RUNTIME_CLIENT_CAPABILITY_METHODS = [ defineMethod({ name: 'runtime.clientCapabilities.update', params: ClientCapabilitiesUpdate, diff --git a/src/main/runtime/rpc/methods/session-tab-close-methods.ts b/src/main/runtime/rpc/methods/session-tab-close-methods.ts index 4800b7d33c1..a7065cf6ba2 100644 --- a/src/main/runtime/rpc/methods/session-tab-close-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-close-methods.ts @@ -1,13 +1,13 @@ import { withSpan } from '../../../observability/tracer' import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { CloseLifecycleTab, CloseTab } from './session-tabs-schemas' import { assertProjectedSessionTabVisible } from './session-tab-browser-placement-projection' import { assertAgentSessionTabDestructiveMutationSupported } from './session-tab-agent-status-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' -export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_CLOSE_METHODS = [ defineMethod({ name: 'session.tabs.close', params: CloseTab, diff --git a/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts b/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts index 6144cd3e546..f2be1d4a61d 100644 --- a/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-markdown-methods.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' -export const SESSION_TAB_MARKDOWN_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_MARKDOWN_METHODS = [ defineMethod({ name: 'markdown.readTab', params: ActivateTab, diff --git a/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts b/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts index 462d00d869d..d62f50be595 100644 --- a/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts @@ -1,6 +1,6 @@ import { resolveRuntimeNavigationTarget } from '../../../../shared/runtime-navigation' import type { OrcaRuntimeService } from '../../orca-runtime' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { assertProjectedSessionTabVisible, translateProjectedSessionTabMove @@ -9,7 +9,7 @@ import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' import { ActivateTab, MoveTab, SetTabProps, UpdatePaneLayout } from './session-tabs-schemas' -export const SESSION_TAB_MUTATION_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_MUTATION_METHODS = [ defineMethod({ name: 'session.tabs.activate', params: ActivateTab, diff --git a/src/main/runtime/rpc/methods/session-tabs-schemas.ts b/src/main/runtime/rpc/methods/session-tabs-schemas.ts index 1f44e17ea0b..ecd434f1e0a 100644 --- a/src/main/runtime/rpc/methods/session-tabs-schemas.ts +++ b/src/main/runtime/rpc/methods/session-tabs-schemas.ts @@ -1,229 +1,14 @@ -import { z } from 'zod' -import { MAX_QUICK_COMMAND_AGENT_PROMPT_LENGTH } from '../../../../shared/terminal-quick-commands' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import type { TuiAgent } from '../../../../shared/tui-agent' -import { sleepingAgentLaunchConfigSchema } from '../../../../shared/workspace-session-sleeping-agents' -import { RUNTIME_NAVIGATION_TARGETS } from '../../../../shared/runtime-navigation' -import { TAB_ACTIVATION_INTENTS } from '../../../../shared/tab-activation-intent' -import { OptionalBoolean } from '../schemas' - -export const WorktreeTabSelector = z.object({ - worktree: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing worktree selector')) -}) - -export const SessionTabsUnsubscribe = WorktreeTabSelector.extend({ - subscriptionId: z.string().min(1).optional() -}) - -export const ActivateTab = WorktreeTabSelector.extend({ - tabId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing tab id')), - leafId: z.string().max(128).optional(), - notifyClients: OptionalBoolean, - navigation: z.enum(RUNTIME_NAVIGATION_TARGETS).optional(), - // Why: absent means user intent, so clients that predate this field keep the - // tab-open wake gesture. Only 'automatic' may be refused for a slept pane. - intent: z.enum(TAB_ACTIVATION_INTENTS).optional() -}) - -export const CloseTab = ActivateTab.extend({ - // Why: optional preserves authenticated legacy user closes; lifecycle intent - // uses the additive evidence-bearing method instead. - reason: z.literal('user').optional() -}) - -export const CloseLifecycleTab = ActivateTab.extend({ - reason: z.enum(['pty-exit', 'cleanup']), - publicationEpoch: z.string().min(1).max(128), - terminal: z.string().min(1).max(256) -}) - -export type TerminalPaneLayoutNodeInput = - | { type: 'leaf'; leafId: string } - | { - type: 'split' - direction: 'horizontal' | 'vertical' - first: TerminalPaneLayoutNodeInput - second: TerminalPaneLayoutNodeInput - ratio?: number - } - -// Why: this schema parses UNTRUSTED remote-client input. A recursive zod parse -// of a deeply-nested tree would overflow the main-process stack, so validate -// iteratively with hard depth + node-count caps before building the typed value. -const MAX_PANE_LAYOUT_DEPTH = 64 -const MAX_PANE_LAYOUT_NODES = 1024 - -function parseTerminalPaneLayoutNode(value: unknown): TerminalPaneLayoutNodeInput | null { - // Iterative validate-then-build: first walk the raw tree with an explicit - // stack (no recursion) enforcing caps, then build bottom-up. - let nodeCount = 0 - const stack: { raw: unknown; depth: number }[] = [{ raw: value, depth: 0 }] - while (stack.length > 0) { - const { raw, depth } = stack.pop()! - if (depth > MAX_PANE_LAYOUT_DEPTH || ++nodeCount > MAX_PANE_LAYOUT_NODES) { - return null - } - if (typeof raw !== 'object' || raw === null) { - return null - } - const node = raw as Record - if (node.type === 'leaf') { - if (typeof node.leafId !== 'string' || node.leafId.length < 1 || node.leafId.length > 128) { - return null - } - continue - } - if (node.type === 'split') { - if (node.direction !== 'horizontal' && node.direction !== 'vertical') { - return null - } - if ( - node.ratio !== undefined && - (typeof node.ratio !== 'number' || - !Number.isFinite(node.ratio) || - node.ratio < 0 || - node.ratio > 1) - ) { - return null - } - stack.push({ raw: node.first, depth: depth + 1 }, { raw: node.second, depth: depth + 1 }) - continue - } - return null - } - return value as TerminalPaneLayoutNodeInput -} - -export const TerminalPaneLayoutNodeSchema = z - .unknown() - .transform((value) => parseTerminalPaneLayoutNode(value)) - .pipe( - z.custom((value) => value !== null, { - message: 'Invalid or too-deep pane layout tree' - }) - ) - -export const UpdatePaneLayout = WorktreeTabSelector.extend({ - tabId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing tab id')), - root: z.union([z.null(), TerminalPaneLayoutNodeSchema]), - expandedLeafId: z.string().max(128).nullable().optional(), - titlesByLeafId: z.record(z.string(), z.string()).optional() -}) - -export const SetTabProps = WorktreeTabSelector.extend({ - tabId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing tab id')), - // undefined = leave unchanged; null = clear color / unset. - color: z.string().max(64).nullable().optional(), - isPinned: z.boolean().optional(), - // undefined = leave unchanged; no "clear" semantic (absence means default 'terminal'). - viewMode: z.enum(['terminal', 'chat']).optional() -}) - -export const CreateTerminalTab = WorktreeTabSelector.extend({ - afterTabId: z.string().optional(), - targetGroupId: z.string().optional(), - command: z.string().optional(), - cwd: z.string().min(1).optional(), - env: z.record(z.string(), z.string()).optional(), - envToDelete: z.array(z.string().min(1).max(256)).max(32).optional(), - startupCommandDelivery: z.enum(['fast', 'shell-ready']).optional(), - launchConfig: sleepingAgentLaunchConfigSchema, - launchToken: z.string().min(1).max(128).optional(), - agent: z - .custom(isTuiAgent, { - message: 'Unknown agent preset' - }) - .optional(), - // Why: agent prompts must be quoted and injected for the host shell (native, - // WSL, or SSH) instead of pasted from the mobile client before the TUI is ready. - agentPrompt: z - .string() - .max(MAX_QUICK_COMMAND_AGENT_PROMPT_LENGTH) - .refine((value) => value.trim().length > 0, { message: 'Agent prompt cannot be empty' }) - .optional(), - // Why: `agent` is the legacy preset field; `launchAgent` is the launch-plan - // identity used when preserving resume config across runtime boundaries. - launchAgent: z - .custom(isTuiAgent, { - message: 'Unknown launch agent' - }) - .optional(), - viewMode: z.enum(['terminal', 'chat']).optional(), - activate: z.boolean().optional(), - select: z.boolean().optional(), - navigation: z.enum(RUNTIME_NAVIGATION_TARGETS).optional(), - // Why: idempotency key so a retried create (double-tap, reconnect replay) - // returns the in-flight operation instead of spawning a duplicate terminal. - clientMutationId: z.string().min(1).max(128).optional() -}).superRefine((value, context) => { - if (value.agentPrompt !== undefined && value.agent === undefined) { - context.addIssue({ - code: 'custom', - path: ['agentPrompt'], - message: 'Agent prompt requires an agent preset' - }) - } - if (value.agentPrompt !== undefined && value.command !== undefined) { - context.addIssue({ - code: 'custom', - path: ['agentPrompt'], - message: 'Agent prompt cannot be combined with a startup command' - }) - } -}) - -const MoveTabBase = { - worktree: WorktreeTabSelector.shape.worktree, - tabId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing tab id')), - targetGroupId: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing target group id')) -} as const - -export const MoveTab = z.discriminatedUnion('kind', [ - z - .object({ - ...MoveTabBase, - kind: z.literal('reorder'), - tabOrder: z.array(z.string().min(1)).min(1, 'Missing tab order') - }) - .strict(), - z - .object({ - ...MoveTabBase, - kind: z.literal('move-to-group'), - index: z.number().int().nonnegative().optional() - }) - .strict(), - z - .object({ - ...MoveTabBase, - kind: z.literal('split'), - splitDirection: z.enum(['left', 'right', 'up', 'down']) - }) - .strict() -]) - -export const SaveMarkdownTab = ActivateTab.extend({ - baseVersion: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing base version')), - content: z.string() -}) +export { + ActivateTab, + CloseLifecycleTab, + CloseTab, + CreateTerminalTab, + MoveTab, + SaveMarkdownTab, + SessionTabsUnsubscribe, + SetTabProps, + TerminalPaneLayoutNodeSchema, + UpdatePaneLayout, + WorktreeTabSelector +} from '../../../../shared/rpc-contract/session-tabs-schemas-params' +export type { TerminalPaneLayoutNodeInput } from '../../../../shared/rpc-contract/session-tabs-schemas-params' diff --git a/src/main/runtime/rpc/methods/session-tabs.ts b/src/main/runtime/rpc/methods/session-tabs.ts index aa0e0b24939..d6441ee84cb 100644 --- a/src/main/runtime/rpc/methods/session-tabs.ts +++ b/src/main/runtime/rpc/methods/session-tabs.ts @@ -1,6 +1,5 @@ -import { z } from 'zod' import { resolveRuntimeNavigationTarget } from '../../../../shared/runtime-navigation' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod } from '../core' +import { defineMethod, defineStreamingMethod } from '../core' import { CreateTerminalTab, SessionTabsUnsubscribe, @@ -18,8 +17,9 @@ import { createSessionTabsRetirementProofDelta } from './session-tabs-retirement import { restoreStructuredTabsIfSupported } from './structured-session-tab-restore' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' import { assertLegacyAiVaultResumeCommandAllowed } from '../../../ai-vault/structured-session-ownership' +import { SessionTabsUnsubscribeAllParams } from '../../../../shared/rpc-contract/session-tabs-params' -export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ +export const SESSION_TAB_METHODS = [ defineMethod({ name: 'session.tabs.list', params: WorktreeTabSelector, @@ -181,11 +181,7 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ }), defineMethod({ name: 'session.tabs.unsubscribeAll', - params: z - .object({ - subscriptionId: z.string().min(1).optional() - }) - .nullish(), + params: SessionTabsUnsubscribeAllParams, handler: async (params, { runtime, connectionId }) => { const cleanupPrefix = `session.tabs:${connectionId ?? 'local'}:*` if (params?.subscriptionId) { diff --git a/src/main/runtime/rpc/methods/skills.test.ts b/src/main/runtime/rpc/methods/skills.test.ts index 0425e5922eb..9bc5a647c82 100644 --- a/src/main/runtime/rpc/methods/skills.test.ts +++ b/src/main/runtime/rpc/methods/skills.test.ts @@ -1,5 +1,5 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' -import type { RpcContext } from '../core' +import { eraseRpcMethods, type RpcContext } from '../core' vi.mock('electron', () => ({ app: { getPath: () => '/orca-state', isPackaged: true } @@ -41,7 +41,7 @@ function makeContext(overrides: { } function discoverMethod() { - const method = SKILL_METHODS.find((entry) => entry.name === 'skills.discover') + const method = eraseRpcMethods(SKILL_METHODS).find((entry) => entry.name === 'skills.discover') if (!method) { throw new Error('skills.discover method not registered') } @@ -49,7 +49,7 @@ function discoverMethod() { } function installMethod() { - const method = SKILL_METHODS.find((entry) => entry.name === 'skills.install') + const method = eraseRpcMethods(SKILL_METHODS).find((entry) => entry.name === 'skills.install') if (!method) { throw new Error('skills.install method not registered') } @@ -57,7 +57,7 @@ function installMethod() { } function method(name: string) { - const value = SKILL_METHODS.find((entry) => entry.name === name) + const value = eraseRpcMethods(SKILL_METHODS).find((entry) => entry.name === name) if (!value) { throw new Error(`${name} method not registered`) } diff --git a/src/main/runtime/rpc/methods/skills.ts b/src/main/runtime/rpc/methods/skills.ts index 13ecbefc3c7..39a3a73eb7c 100644 --- a/src/main/runtime/rpc/methods/skills.ts +++ b/src/main/runtime/rpc/methods/skills.ts @@ -1,5 +1,5 @@ -import { defineMethod, type RpcMethod } from '../core' -import { z } from 'zod' +import { defineMethod } from '../core' +import type { z } from 'zod' import { getAppEnvironment } from '../../../../shared/app-environment' import { SkillDeleteRequestSchema } from '../../../../shared/skill-delete-contract' import { @@ -7,7 +7,7 @@ import { runSkillDeleteRequest, type SkillDeleteRequestDependencies } from '../../../skills/skill-delete/request-service' -import { SkillDiscoveryTargetSchema } from '../../../../shared/skills' +import type { SkillDiscoveryTargetSchema } from '../../../../shared/skills' import { SkillInstallPreviewRequestSchema, SkillInstallRequestSchema, @@ -33,6 +33,11 @@ import { AgentSkillShareRequestSchema, AgentSkillSharingError } from '../../../../shared/agent-skill-sharing-contract' +import { + SkillsCancelInstallParams, + SkillsDiscoverParams, + SkillsGetInstallProgressParams +} from '../../../../shared/rpc-contract/skills-params' /** Exported so the delete plan's root rebuild resolves its target exactly the * way `skills.discover` resolved the scan's — including WSL. */ @@ -59,10 +64,10 @@ function skillDeleteDependencies( } } -export const SKILL_METHODS: RpcMethod[] = [ +export const SKILL_METHODS = [ defineMethod({ name: 'skills.discover', - params: SkillDiscoveryTargetSchema.default({}), + params: SkillsDiscoverParams, handler: async (params, { runtime }) => { // Why: the executing runtime owns WSL project preferences. Remote callers // send worktree identity only; trusting their projectRuntime absence @@ -146,14 +151,14 @@ export const SKILL_METHODS: RpcMethod[] = [ }), defineMethod({ name: 'skills.cancelInstall', - params: z.object({ operationId: z.string().min(1).max(128) }).strict(), + params: SkillsCancelInstallParams, handler: (params, { runtime }) => ({ cancelled: runtime.cancelSharedSkillInstall(params.operationId) }) }), defineMethod({ name: 'skills.getInstallProgress', - params: z.object({ operationId: z.string().min(1).max(128) }).strict(), + params: SkillsGetInstallProgressParams, handler: (params, { runtime }) => { const progress = runtime.getSharedSkillInstallProgress(params.operationId) return progress ? SkillBundleInstallProgressSchema.parse(progress) : null diff --git a/src/main/runtime/rpc/methods/speech.ts b/src/main/runtime/rpc/methods/speech.ts index d686475ab3f..086a528ab0e 100644 --- a/src/main/runtime/rpc/methods/speech.ts +++ b/src/main/runtime/rpc/methods/speech.ts @@ -1,54 +1,13 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' +import { defineMethod } from '../core' +import { + DictationChunk, + DictationHandle, + DictationSetup, + DictationStart, + SpeechModelAction +} from '../../../../shared/rpc-contract/speech-params' -const AUDIO_BASE64_PATTERN = /^[A-Za-z0-9+/]*={0,2}$/ -const DICTATION_SAMPLE_RATE = 16_000 -const PCM_BYTES_PER_SAMPLE = 2 -const MAX_DICTATION_AUDIO_SECONDS = 5 -const MAX_DICTATION_AUDIO_CHUNK_BYTES = - DICTATION_SAMPLE_RATE * PCM_BYTES_PER_SAMPLE * MAX_DICTATION_AUDIO_SECONDS -const MAX_DICTATION_AUDIO_CHUNK_BASE64_LENGTH = Math.ceil(MAX_DICTATION_AUDIO_CHUNK_BYTES / 3) * 4 - -function isValidAudioBase64(value: string): boolean { - return value.length % 4 !== 1 && AUDIO_BASE64_PATTERN.test(value) -} - -const DictationStart = z.object({ - dictationId: requiredString('Missing dictation ID'), - modelId: OptionalString -}) - -const DictationChunk = z.object({ - dictationId: requiredString('Missing dictation ID'), - audioBase64: requiredString('Missing audio chunk') - // Why: feedMobileDictation decodes into Buffer + Float32Array; reject - // oversized chunks before allocation. This mirrors the mobile pending-audio budget. - .refine( - (value) => value.length <= MAX_DICTATION_AUDIO_CHUNK_BASE64_LENGTH, - 'Audio chunk is too large' - ) - // Why: Buffer.from(..., 'base64') silently drops malformed bytes; reject - // bad mobile audio chunks instead of feeding empty/corrupt PCM. - .refine(isValidAudioBase64, 'Audio chunk must be base64'), - sampleRate: z.number().finite().positive() -}) - -const DictationHandle = z.object({ - dictationId: requiredString('Missing dictation ID') -}) - -const SpeechModelAction = z.object({ - modelId: requiredString('Missing model ID') -}) - -const DictationSetup = z.object({ - enabled: z.boolean().optional(), - modelId: OptionalString, - dictationMode: z.enum(['toggle', 'hold']).optional() -}) - -export const SPEECH_METHODS: RpcMethod[] = [ +export const SPEECH_METHODS = [ defineMethod({ name: 'speech.models.list', params: null, diff --git a/src/main/runtime/rpc/methods/ssh.ts b/src/main/runtime/rpc/methods/ssh.ts index e6cb6b47b50..2c8e2520f9b 100644 --- a/src/main/runtime/rpc/methods/ssh.ts +++ b/src/main/runtime/rpc/methods/ssh.ts @@ -1,17 +1,13 @@ -import { z } from 'zod' import { connectRegisteredSshTarget, getRegisteredSshState, listRegisteredRemovedSshTargetLabels, listRegisteredSshTargets } from '../../../ssh/ssh-target-registry' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { getPublicSshError, getPublicSshState } from '../../public-ssh-state' import type { SshTargetSummary } from '../../../../shared/ssh-types' - -const SshTarget = z.object({ - targetId: z.string().min(1) -}) +import { SshTarget } from '../../../../shared/rpc-contract/ssh-params' // Why: `generation` stays optional on the wire — an old server simply omits it and its rows key on target id alone. function listRegisteredSshTargetSummaries(): SshTargetSummary[] { @@ -29,7 +25,7 @@ function listRegisteredSshTargetSummaries(): SshTargetSummary[] { }) } -export const SSH_METHODS: RpcMethod[] = [ +export const SSH_METHODS = [ defineMethod({ name: 'ssh.getState', params: SshTarget, diff --git a/src/main/runtime/rpc/methods/stats.ts b/src/main/runtime/rpc/methods/stats.ts index 59f71701c3a..9cdfe5269d3 100644 --- a/src/main/runtime/rpc/methods/stats.ts +++ b/src/main/runtime/rpc/methods/stats.ts @@ -1,6 +1,6 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' -export const STATS_METHODS: RpcMethod[] = [ +export const STATS_METHODS = [ defineMethod({ name: 'stats.summary', params: null, diff --git a/src/main/runtime/rpc/methods/status.ts b/src/main/runtime/rpc/methods/status.ts index 03d66f84fb1..dac38097b7a 100644 --- a/src/main/runtime/rpc/methods/status.ts +++ b/src/main/runtime/rpc/methods/status.ts @@ -1,7 +1,7 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { getRemoteServerUpdaterSnapshot } from '../../remote-server-updater' -export const STATUS_METHODS: RpcMethod[] = [ +export const STATUS_METHODS = [ defineMethod({ name: 'status.get', params: null, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts index cc5a7282080..2b62b00525e 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.test.ts @@ -1,6 +1,9 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { AgentSessionBackgroundTaskState } from '../../../../shared/agent-session-wire' -import { AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY } from '../../../../shared/protocol-version' +import { + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY +} from '../../../../shared/protocol-version' import { remoteRuntimeClientCapabilities } from '../../../../shared/remote-runtime-client-capabilities' import type { AgentSessionSubscribeInput } from '../../../native-chat/agent-session-wire/structured-agent-session-subscribers' import { @@ -24,6 +27,20 @@ const CURRENT_CLIENT = { ...STRUCTURED_CLIENT, clientCapabilities: remoteRuntimeClientCapabilities(STRUCTURED_CLIENT.clientCapabilities) } +/** Understands a stopless roster, but predates per-row stoppability. */ +const STOP_ONLY_CLIENT = { + ...STRUCTURED_CLIENT, + clientCapabilities: remoteRuntimeClientCapabilities(STRUCTURED_CLIENT.clientCapabilities).filter( + (capability) => capability !== AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY + ) +} +const FOREGROUND_ROW = { id: 'fore', kind: 'agent', stoppable: false } as const +const BACKGROUNDED_ROW = { id: 'back', kind: 'agent' } as const +const MIXED_ROWS: AgentSessionBackgroundTaskState = { + state: 'monitoring', + supportsTaskStop: true, + tasks: [FOREGROUND_ROW, BACKGROUNDED_ROW] +} describe('background-task stop capability at the RPC boundary', () => { it('advertises reader support on remote requests and subscriptions', () => { @@ -93,6 +110,58 @@ describe('background-task stop capability at the RPC boundary', () => { } ) + it('advertises row-stop support separately from stop support', () => { + // A client can advertise the stop capability and still predate `stoppable`, + // so the two must not be conflated. + expect(CURRENT_CLIENT.clientCapabilities).toContain( + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY + ) + expect(STOP_ONLY_CLIENT.clientCapabilities).not.toContain( + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY + ) + }) + + it.each([ + ['row-stop reader', () => CURRENT_CLIENT, MIXED_ROWS], + [ + 'stop-only reader', + () => STOP_ONLY_CLIENT, + { state: 'monitoring', tasks: [BACKGROUNDED_ROW] } + ], + ['in-process reader', () => undefined, MIXED_ROWS] + ] as const)('projects unstoppable rows for a %s', async (_label, client, expected) => { + hostCalls.history.mockReturnValue({ + ok: true, + page: { items: [], backgroundTasks: MIXED_ROWS } + }) + expect( + await call('agentSession.history', { sessionId: SESSION, direction: 'tail' }, client()) + ).toMatchObject({ ok: true, result: { page: { backgroundTasks: expected } } }) + }) + + it('hands a reader that predates the field no strip when every row is unstoppable', async () => { + // Its pre-feature view exactly: the host published no foreground rows at all. + const foregroundOnly = { + state: 'monitoring' as const, + supportsTaskStop: true, + tasks: [{ id: 'fore', kind: 'agent' as const, stoppable: false }] + } + hostCalls.history.mockReturnValue({ + ok: true, + page: { items: [], backgroundTasks: foregroundOnly } + }) + expect( + await call( + 'agentSession.history', + { sessionId: SESSION, direction: 'tail' }, + STOP_ONLY_CLIENT + ) + ).toMatchObject({ ok: true, result: { page: { backgroundTasks: null } } }) + expect( + await call('agentSession.history', { sessionId: SESSION, direction: 'tail' }, CURRENT_CLIENT) + ).toMatchObject({ ok: true, result: { page: { backgroundTasks: foregroundOnly } } }) + }) + it('preserves legacy stoppable state for both readers', async () => { const stoppable = { state: 'monitoring', tasks: TASKS.tasks } hostCalls.history.mockReturnValue({ ok: true, page: { items: [], backgroundTasks: stoppable } }) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts index 02afc5569bf..ae69de4a069 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-background-task-capability.ts @@ -3,7 +3,10 @@ import type { AgentSessionHistoryResult, AgentSessionSubscribeEvent } from '../../../../shared/agent-session-wire' -import { AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY } from '../../../../shared/protocol-version' +import { + AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY, + AGENT_SESSION_BACKGROUND_TASK_STOP_CAPABILITY +} from '../../../../shared/protocol-version' import type { RpcContext } from '../core' type BackgroundTaskReader = Pick @@ -15,14 +18,37 @@ function supportsReadOnlyTasks(ctx: BackgroundTaskReader): boolean { ) } +function honoursRowStop(ctx: BackgroundTaskReader): boolean { + return ( + ctx.clientKind === undefined || + ctx.clientCapabilities?.includes(AGENT_SESSION_BACKGROUND_TASK_ROW_STOP_CAPABILITY) === true + ) +} + +/** A reader that predates `stoppable` draws a per-row stop on every row it is + * handed, and the host cannot honour one on a row marked unstoppable — the + * dead button the field exists to remove. The host publishing such rows at all + * is new, so withholding them hands that reader exactly its pre-feature view; + * a state whose every row is withheld becomes no strip, as it was. */ +function withoutUnstoppableRows( + state: AgentSessionBackgroundTaskState +): AgentSessionBackgroundTaskState | null { + if (!state.tasks?.some((task) => task.stoppable === false)) { + return state + } + const tasks = state.tasks.filter((task) => task.stoppable !== false) + return tasks.length > 0 ? { ...state, tasks } : null +} + function projectState( state: AgentSessionBackgroundTaskState | null | undefined, ctx: BackgroundTaskReader ): AgentSessionBackgroundTaskState | null | undefined { + const rows = !state || honoursRowStop(ctx) ? state : withoutUnstoppableRows(state) // Legacy readers always offer a stop; retain their pre-producer empty strip. - return state?.supportsStopAll === false && !state.supportsTaskStop && !supportsReadOnlyTasks(ctx) + return rows?.supportsStopAll === false && !rows.supportsTaskStop && !supportsReadOnlyTasks(ctx) ? null - : state + : rows } export function projectBackgroundTaskHistory( diff --git a/src/main/runtime/rpc/methods/structured-agent-session-hold.ts b/src/main/runtime/rpc/methods/structured-agent-session-hold.ts index 280804711e6..18fb600a949 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-hold.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-hold.ts @@ -9,7 +9,7 @@ // the hold is deliberate: re-registering an id runs the previous cleanup synchronously, so the // stale release lands before this hold rather than after it. -import { defineMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled, requireStructuredCleanupHost, @@ -28,7 +28,7 @@ function holdCleanupIdFor(sessionId: string, holderKey: string): string { return `${HOLD_CLEANUP_PREFIX}:${holderKey}:${sessionId}` } -export const STRUCTURED_AGENT_SESSION_HOLD_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_HOLD_METHODS = [ defineMethod({ name: 'agentSession.hold', params: HoldParams, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts b/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts index 5f2ab0e8cac..47f30a5030d 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-reveal.ts @@ -12,7 +12,7 @@ import { isAgentSessionWireRefusalCode } from '../../../../shared/agent-session-wire' import type { StructuredAgentSessionReveal } from '../../../native-chat/agent-session-wire/structured-agent-session-host-types' import { refuseAgentSessionMutation } from '../../../native-chat/agent-session-wire/structured-agent-session-mutation-admission' -import { defineMethod, type RpcAnyMethod } from '../core' +import { defineMethod } from '../core' import { ensureStructuredHostInstalled, requireStructuredCapability, @@ -20,7 +20,7 @@ import { } from './structured-agent-session-gate' import { OptionsParams } from './structured-agent-session-schemas' -export const STRUCTURED_AGENT_SESSION_REVEAL_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_REVEAL_METHODS = [ defineMethod({ name: 'agentSession.reveal', params: OptionsParams, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-rpc.test-fixture.ts b/src/main/runtime/rpc/methods/structured-agent-session-rpc.test-fixture.ts index a702bda5afc..b8fe9d5b01b 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-rpc.test-fixture.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-rpc.test-fixture.ts @@ -12,7 +12,10 @@ import { type StructuredAgentSessionStatusSubscriber } from '../../../native-chat/agent-session-wire/structured-agent-session-status-feed' import type { OrcaRuntimeService } from '../../orca-runtime' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import { + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../../../shared/protocol-version' import type { RpcRequest, RpcResponse } from '../core' import { RpcDispatcher } from '../dispatcher' import { STRUCTURED_AGENT_SESSION_METHODS } from './structured-agent-session' @@ -97,6 +100,7 @@ function statusFeed(): StructuredAgentSessionStatusFeed { { journal: { isReadOnly: false, + cursor: () => ({ epoch: 'epoch-status', sequence: 2 }), lastActivityAt: () => 2, snapshot: () => ({ items: STATUS_ITEMS }) } as unknown as AgentSessionJournal, @@ -140,7 +144,26 @@ export function hostStub(): StructuredAgentSessionHost { } })), rewind: vi.fn(async () => ({ ok: true, value: { itemId: 'chosen', epoch: 'next' } })), - send: vi.fn(async () => ({ ok: true, replayed: false })), + send: vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 1 }, + value: { + clientMessageId: OPERATION, + submission: { + clientMessageId: OPERATION, + fence: 1, + payloadFingerprint: FINGERPRINT, + dispatchState: 'accepted', + providerItemId: 'provider-1', + reason: null, + submittedAt: 1, + resolvedAt: 2 + } + } + })), + waitForSendSettlement: vi.fn(), cancel: vi.fn(async () => ({ ok: true, replayed: false })), close: vi.fn(async () => undefined), revealSession: vi.fn(async () => ({ @@ -236,6 +259,7 @@ export async function call( clientId?: string clientKind?: 'mobile' | 'runtime' clientCapabilities?: string[] + signal?: AbortSignal }, runtimeOverrides: Record = {} ): Promise { @@ -254,11 +278,17 @@ export async function call( export const STRUCTURED_CLIENT = { clientKind: 'runtime' as const, - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY + ] } export const STRUCTURED_MOBILE_CLIENT = { clientKind: 'mobile' as const, - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY + ] } /** Every suite wants the same lifecycle: a fresh stub per test, no host left installed. */ diff --git a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts index 5c7f40d7f35..7725e70bded 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts @@ -2,246 +2,25 @@ // // Strict objects throughout: zod drops unknown keys, and a silently dropped key // is how a newer client's field becomes a different effect on an older host. - -import { z } from 'zod' -import { isAgentSessionId } from '../../../../shared/agent-session-record' -import { - AGENT_SESSION_HISTORY_DIRECTIONS, - AGENT_SESSION_HISTORY_MAX_LIMIT -} from '../../../../shared/agent-session-wire' -import { normalizeExecutionHostId } from '../../../../shared/execution-host' - -const MAX_ID_LENGTH = 512 -// Four Claude questions with all four generated choices occupy 610 chars when fully percent-encoded. -const MAX_RESPONSE_OPTION_ID_LENGTH = 1024 -const MAX_PROMPT_BYTES = 256 * 1024 -const MAX_BLOCKS = 64 -const MAX_OPTION_LABEL = 512 - -export const SessionId = z - .string() - .max(MAX_ID_LENGTH) - .refine(isAgentSessionId, 'Invalid agent session id') - -const Identifier = (message: string, maxLength = MAX_ID_LENGTH) => - z - .string() - .min(1, message) - .max(maxLength, message) - .refine((value) => value === value.trim(), message) - -export const JournalCursor = z - .object({ - epoch: Identifier('Invalid journal epoch'), - sequence: z.number().int().nonnegative() - }) - .strict() - -export const MutationEnvelope = z - .object({ - sessionId: SessionId, - clientOperationId: Identifier('Invalid client operation id'), - /** Null is the "must not exist yet" case; every other call fences. */ - expectedRuntimeFence: z.number().int().positive().nullable(), - payloadFingerprint: z - .string() - .regex(/^[0-9a-f]{64}$/, 'Payload fingerprint must be a sha256 hex digest') - }) - .strict() - -const ProviderHandle = z.discriminatedUnion('kind', [ - z.object({ kind: z.literal('codex'), threadId: Identifier('Invalid thread id') }).strict(), - z - .object({ - kind: z.literal('claude'), - sessionId: Identifier('Invalid provider session id'), - leafUuid: Identifier('Invalid leaf uuid').nullable() - }) - .strict() -]) - -const ExecutionHostId = z - .string() - .max(MAX_ID_LENGTH) - .transform((value) => normalizeExecutionHostId(value)) - .refine((value): value is NonNullable => value !== null, { - message: 'Invalid execution host id' - }) - -const ExecutionLocation = z - .object({ - executionHostId: ExecutionHostId, - wslDistro: Identifier('Invalid WSL distro').nullable(), - workspaceId: Identifier('Invalid workspace id'), - workspaceKind: z.enum(['git-worktree', 'folder']) - }) - .strict() - -const AccountHome = z - .object({ - variable: z.enum(['CLAUDE_CONFIG_DIR', 'CODEX_HOME']), - path: z.string().min(1).max(4096) - }) - .strict() - -export const AttachParams = z - .object({ - envelope: MutationEnvelope, - location: ExecutionLocation, - provider: z.enum(['codex', 'claude']), - agent: Identifier('Invalid agent'), - accountHome: AccountHome, - runtimeKind: z.enum(['native', 'tui']), - providerHandle: ProviderHandle - }) - .strict() - -/** An identity, and nothing the host would otherwise read off disk. A transcript path or account - * home here would let a client choose which file this host imports and which credential directory - * the provider child launches against; both are derived host-side from this id instead. */ -const ResumeSource = z - .object({ - providerSessionId: Identifier('Invalid provider session id') - }) - .strict() - -export const CreateIntentParams = z - .object({ - envelope: MutationEnvelope, - worktree: Identifier('Invalid worktree selector'), - agent: z.enum(['claude', 'codex']), - resumeFrom: ResumeSource.optional() - }) - .strict() - -export const CreateParams = z.union([AttachParams, CreateIntentParams]) - -export const CreateSupportParams = z - .object({ - worktree: Identifier('Invalid worktree selector'), - agent: z.enum(['claude', 'codex']) - }) - .strict() - -/** Clients may only author user turns. Accepting an assistant or tool role here - * would let one client write words into the agent's mouth in another's - * timeline, and the provider — not the client — owns those. */ -const SendBlock = z.discriminatedUnion('type', [ - z.object({ type: z.literal('text'), text: z.string() }).strict(), - z - .object({ - type: z.literal('image-ref'), - path: z.string().min(1).max(4096).optional(), - url: z.string().min(1).max(4096).optional(), - alt: z.string().max(MAX_OPTION_LABEL).optional() - }) - .strict() - .refine( - (value) => Boolean(value.path) !== Boolean(value.url), - 'Provide exactly one of path/url' - ) -]) - -export const SendParams = z - .object({ - envelope: MutationEnvelope, - retryUnknown: z.literal(true).optional(), - body: z - .object({ - kind: z.literal('message'), - role: z.literal('user'), - blocks: z.array(SendBlock).min(1).max(MAX_BLOCKS) - }) - .strict() - .refine( - (value) => Buffer.byteLength(JSON.stringify(value.blocks), 'utf8') <= MAX_PROMPT_BYTES, - 'Message is too large' - ) - }) - .strict() - -export const CancelParams = z - .object({ - envelope: MutationEnvelope, - turnId: Identifier('Invalid turn id'), - scope: z.literal('background-tasks').optional(), - taskId: Identifier('Invalid task id').optional() - }) - .strict() - .refine((value) => value.taskId === undefined || value.scope === 'background-tasks', { - message: 'A task id requires background-task scope' - }) - -export const RespondParams = z - .object({ - envelope: MutationEnvelope, - itemId: Identifier('Invalid item id'), - /** Compare-and-set: the revision the client had on screen. */ - expectedRevision: z.number().int().positive(), - optionId: Identifier('Invalid option id', MAX_RESPONSE_OPTION_ID_LENGTH) - }) - .strict() - -export const SetOptionParams = z - .object({ - envelope: MutationEnvelope, - key: Identifier('Invalid option key'), - value: z.string().max(MAX_OPTION_LABEL) - }) - .strict() - -export const HandoffParams = z - .object({ - envelope: MutationEnvelope, - direction: z.enum(['to-tui', 'to-native']), - mode: z.enum(['now', 'after-turn', 'stop-turn']), - action: z.enum(['start', 'cancel-queued', 'retry', 'recover']).optional() - }) - .strict() - -export const OptionsParams = z.object({ sessionId: SessionId }).strict() - -export const ConversationCommandParams = z - .object({ - envelope: MutationEnvelope, - command: z.enum(['clear', 'compact']) - }) - .strict() - -/** One surface's claim on one session. The id names the surface, not the client: two chat views - * looking at the same session are two holders, and either leaving must not release - * the other's. */ -export const HoldParams = z - .object({ sessionId: SessionId, holderId: Identifier('Invalid holder id') }) - .strict() - -export const HistoryParams = z - .object({ - sessionId: SessionId, - direction: z.enum(AGENT_SESSION_HISTORY_DIRECTIONS), - cursor: JournalCursor.optional(), - limit: z.number().int().positive().max(AGENT_SESSION_HISTORY_MAX_LIMIT).optional() - }) - .strict() - -export const SubscribeParams = z - .object({ sessionId: SessionId, cursor: JournalCursor.optional() }) - .strict() - -export const UnsubscribeParams = z - .object({ - sessionId: SessionId, - subscriptionId: Identifier('Invalid subscription id').optional() - }) - .strict() - -/** Read-only owner classification retained for restart safety; mutation handoff is separate. */ -export const HandoffStatusParams = z.object({ sessionId: SessionId }).strict() - -export const RewindParams = z - .object({ - envelope: MutationEnvelope, - itemId: Identifier('Invalid item id', 4096), - expectedEpoch: Identifier('Invalid journal epoch') - }) - .strict() +export { + AttachParams, + CancelParams, + ConversationCommandParams, + CreateIntentParams, + CreateParams, + CreateSupportParams, + HandoffParams, + HandoffStatusParams, + HistoryParams, + HoldParams, + JournalCursor, + MutationEnvelope, + OptionsParams, + RespondParams, + RewindParams, + SendParams, + SessionId, + SetOptionParams, + SubscribeParams, + UnsubscribeParams +} from '../../../../shared/rpc-contract/structured-agent-session-params' diff --git a/src/main/runtime/rpc/methods/structured-agent-session-send-compatibility.ts b/src/main/runtime/rpc/methods/structured-agent-session-send-compatibility.ts new file mode 100644 index 00000000000..8b948869990 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-send-compatibility.ts @@ -0,0 +1,26 @@ +import { AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' +import type { RpcContext } from '../core' +import { requireStructuredHost, structuredCallerFor } from './structured-agent-session-gate' + +export async function sendStructuredAgentSessionForClient( + params: Parameters[1], + context: RpcContext +) { + const host = requireStructuredHost(context) + const result = await host.send(structuredCallerFor(context), params) + if ( + !result.ok || + result.value.submission.dispatchState !== 'pending' || + context.clientKind === undefined || + context.clientCapabilities?.includes(AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY) + ) { + return result + } + const settled = await host.waitForSendSettlement( + params.envelope.sessionId, + result.value.clientMessageId, + context.signal + ) + return settled ? { ...result, ...settled } : result +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts b/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts index 8637089c254..c93401e0e22 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-status-stream.ts @@ -3,7 +3,7 @@ // Session lists read turn state from here instead of replaying transcripts: one stream per client // covers every session, and unlike a transcript subscription it retains none of them. -import { defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { defineStreamingMethod, type RpcContext } from '../core' import { requireStructuredHost as requireHost } from './structured-agent-session-gate' import { structuredAgentSessionStatusSubscriptionId } from './structured-agent-session-subscription-id' @@ -41,7 +41,7 @@ export function bindStructuredAgentSessionStream( return { isClosed: () => closed } } -export const STRUCTURED_AGENT_SESSION_STATUS_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_STATUS_METHODS = [ defineStreamingMethod({ name: 'agentSession.subscribeStatus', params: null, diff --git a/src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.test.ts b/src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.test.ts new file mode 100644 index 00000000000..bf7eb50fb23 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.test.ts @@ -0,0 +1,154 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { + AgentSessionHistoryPage, + AgentSessionHistoryResult, + AgentSessionSubscribeEvent +} from '../../../../shared/agent-session-wire' +import { AGENT_SESSION_TURN_ITEM_CAPABILITY } from '../../../../shared/protocol-version' +import type { AgentSessionSubscribeInput } from '../../../native-chat/agent-session-wire/structured-agent-session-subscribers' +import { + call, + clearStructuredHostStub, + hostCalls, + installStructuredHostStub, + SESSION, + STRUCTURED_CLIENT +} from './structured-agent-session-rpc.test-fixture' +import { + projectTurnItemEvent, + projectTurnItemHistory +} from './structured-agent-session-turn-item-capability' + +beforeEach(installStructuredHostStub) +afterEach(clearStructuredHostStub) + +const TURN = { + turnId: 'turn-1', + state: 'completed' as const, + userItemId: 'user-1', + startedAt: 10, + completedAt: 42, + durationMs: 30 +} +const USER_ITEM: AgentJournalRenderItem = { + itemId: 'user-1', + revision: 1, + sequence: 1, + observedAt: 1, + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] } +} +const TURN_ITEM: AgentJournalRenderItem = { + itemId: 'legacy:claude:session:turn-1', + revision: 2, + sequence: 2, + observedAt: 2, + body: { kind: 'turn', ...TURN } +} +const LEGACY_STATUS_ITEM: AgentJournalRenderItem = { + ...TURN_ITEM, + body: { kind: 'status', text: 'Claude turn completed', turnLifecycle: TURN } +} +const CURRENT_CLIENT = { + ...STRUCTURED_CLIENT, + clientCapabilities: [...STRUCTURED_CLIENT.clientCapabilities, AGENT_SESSION_TURN_ITEM_CAPABILITY] +} + +function page(items: AgentJournalRenderItem[]): AgentSessionHistoryPage { + return { + sessionId: SESSION, + epoch: 'a', + direction: 'tail', + items, + removedItemIds: [], + submissions: [], + window: { oldest: null, newest: null, nextCursor: { epoch: 'a', sequence: 0 } }, + hasOlder: false, + hasNewer: false + } +} + +describe('turn item capability at the RPC boundary', () => { + it.each([ + ['legacy reader', STRUCTURED_CLIENT, LEGACY_STATUS_ITEM], + ['current reader', CURRENT_CLIENT, TURN_ITEM], + ['in-process reader', undefined, TURN_ITEM] + ] as const)('projects history for a %s', async (_label, client, expected) => { + hostCalls.history.mockReturnValue({ ok: true, page: page([USER_ITEM, TURN_ITEM]) }) + expect( + await call('agentSession.history', { sessionId: SESSION, direction: 'tail' }, client) + ).toMatchObject({ ok: true, result: { page: { items: [USER_ITEM, expected] } } }) + }) + + it.each(['snapshot', 'batch', 'reset'] as const)( + 'downgrades the %s stream only for a legacy reader', + async (type) => { + hostCalls.hold = vi.fn(async () => undefined) + hostCalls.subscribe.mockImplementation((input: AgentSessionSubscribeInput) => { + const base = { sessionId: SESSION, fence: 1 } + if (type === 'batch') { + input.emit({ + ...base, + type, + batch: { + cursor: { epoch: 'a', sequence: 2 }, + items: [USER_ITEM, TURN_ITEM], + removedItemIds: [], + submissions: [] + } + }) + } else { + input.emit( + type === 'snapshot' + ? { ...base, type, page: page([USER_ITEM, TURN_ITEM]) } + : { ...base, type, page: page([USER_ITEM, TURN_ITEM]), reset: 'epoch_changed' } + ) + } + return () => {} + }) + for (const [client, expected] of [ + [STRUCTURED_CLIENT, LEGACY_STATUS_ITEM], + [CURRENT_CLIENT, TURN_ITEM] + ] as const) { + const reply = await call('agentSession.subscribe', { sessionId: SESSION }, client) + const event = (reply.ok ? reply.result : null) as AgentSessionSubscribeEvent + const items = + event.type === 'batch' ? event.batch.items : 'page' in event ? event.page.items : [] + expect(items).toEqual([USER_ITEM, expected]) + } + } + ) +}) + +describe('turn item projection', () => { + const history: AgentSessionHistoryResult = { ok: true, page: page([USER_ITEM, TURN_ITEM]) } + const snapshot: AgentSessionSubscribeEvent = { + type: 'snapshot', + sessionId: SESSION, + fence: 1, + page: page([USER_ITEM, TURN_ITEM]) + } + + it('publishes the status form with the lifecycle intact to a legacy reader', () => { + const projected = projectTurnItemHistory(history, STRUCTURED_CLIENT) + expect(projected.page.items).toEqual([USER_ITEM, LEGACY_STATUS_ITEM]) + // Untouched rows keep their identity; the journal's own body is never mutated. + expect(projected.page.items[0]).toBe(USER_ITEM) + expect(TURN_ITEM.body.kind).toBe('turn') + }) + + it.each([ + ['capable client', CURRENT_CLIENT], + ['in-process caller', {}] + ] as const)('hands a %s the same object back', (_label, ctx) => { + expect(projectTurnItemHistory(history, ctx)).toBe(history) + expect(projectTurnItemEvent(snapshot, ctx)).toBe(snapshot) + }) + + it('returns the same object when nothing needs downgrading', () => { + const plain: AgentSessionHistoryResult = { ok: true, page: page([USER_ITEM]) } + expect(projectTurnItemHistory(plain, STRUCTURED_CLIENT)).toBe(plain) + const event: AgentSessionSubscribeEvent = { type: 'end' } + expect(projectTurnItemEvent(event, STRUCTURED_CLIENT)).toBe(event) + }) +}) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.ts b/src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.ts new file mode 100644 index 00000000000..f0694811a9a --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-turn-item-capability.ts @@ -0,0 +1,72 @@ +// Transitional: remove once no supported release lacks AGENT_SESSION_TURN_ITEM_CAPABILITY. +// +// A client that predates the `turn` item renders the unknown kind as a text bubble, so the +// host publishes the legacy status form to it at the RPC boundary only. The journal, the +// status feed, and every in-process reader keep the canonical body. + +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import { legacyAgentJournalTurnStatusBody } from '../../../../shared/agent-session-turn-record' +import type { + AgentSessionHistoryPage, + AgentSessionHistoryResult, + AgentSessionSubscribeEvent +} from '../../../../shared/agent-session-wire' +import { AGENT_SESSION_TURN_ITEM_CAPABILITY } from '../../../../shared/protocol-version' +import type { RpcContext } from '../core' + +type TurnItemReader = Pick + +function readsTurnItems(ctx: TurnItemReader): boolean { + // An in-process caller is this build; only a negotiated client can predate the item. + return ( + ctx.clientKind === undefined || + ctx.clientCapabilities?.includes(AGENT_SESSION_TURN_ITEM_CAPABILITY) === true + ) +} + +function projectItems(items: AgentJournalRenderItem[]): AgentJournalRenderItem[] { + if (!items.some((item) => item.body.kind === 'turn')) { + return items + } + return items.map((item) => { + if (item.body.kind !== 'turn') { + return item + } + const { kind: _kind, ...turn } = item.body + return { ...item, body: legacyAgentJournalTurnStatusBody(turn, item.itemId) } + }) +} + +function projectPage(page: AgentSessionHistoryPage): AgentSessionHistoryPage { + const items = projectItems(page.items) + return items === page.items ? page : { ...page, items } +} + +export function projectTurnItemHistory( + result: AgentSessionHistoryResult, + ctx: TurnItemReader +): AgentSessionHistoryResult { + if (readsTurnItems(ctx)) { + return result + } + const page = projectPage(result.page) + return page === result.page ? result : { ...result, page } +} + +export function projectTurnItemEvent( + event: AgentSessionSubscribeEvent, + ctx: TurnItemReader +): AgentSessionSubscribeEvent { + if (readsTurnItems(ctx)) { + return event + } + if (event.type === 'batch') { + const items = projectItems(event.batch.items) + return items === event.batch.items ? event : { ...event, batch: { ...event.batch, items } } + } + if (event.type === 'snapshot' || event.type === 'reset') { + const page = projectPage(event.page) + return page === event.page ? event : { ...event, page } + } + return event +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index 94a344b6f18..c2d46b09818 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -4,6 +4,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { setStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' import { + AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY, RUNTIME_CAPABILITIES, RUNTIME_PROTOCOL_VERSION, STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, @@ -146,6 +147,7 @@ describe('capability gating', () => { it('advertises the capability without bumping the protocol version', () => { expect(RUNTIME_CAPABILITIES).toContain(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + expect(RUNTIME_CAPABILITIES).toContain(AGENT_SESSION_PENDING_SEND_RESULT_RUNTIME_CAPABILITY) expect(RUNTIME_CAPABILITIES).toContain(STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY) expect(RUNTIME_CAPABILITIES).toContain(STRUCTURED_AGENT_SESSION_REVEAL_RUNTIME_CAPABILITY) // Additive methods do not break an old client; bumping would strand every @@ -207,6 +209,124 @@ describe('capability gating', () => { expect(hostCalls.send).toHaveBeenCalledTimes(1) }) + it('returns a settlement to older structured clients when observed within the window', async () => { + const pendingSubmission = { + clientMessageId: 'client-1', + fence: 1, + payloadFingerprint: 'fingerprint', + dispatchState: 'pending' as const, + providerItemId: null, + reason: null, + submittedAt: 1, + resolvedAt: null + } + hostCalls.send.mockResolvedValueOnce({ + ok: true, + replayed: true, + fence: 7, + cursor: { epoch: 'epoch-a', sequence: 1 }, + value: { clientMessageId: 'client-1', submission: pendingSubmission } + }) + hostCalls.waitForSendSettlement.mockResolvedValueOnce({ + cursor: { epoch: 'epoch-a', sequence: 2 }, + value: { + clientMessageId: 'client-1', + submission: { + ...pendingSubmission, + dispatchState: 'accepted', + providerItemId: 'provider-1', + resolvedAt: 2 + } + } + }) + const controller = new AbortController() + + const response = await call('agentSession.send', sendParams(), { + clientKind: 'runtime', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + signal: controller.signal + }) + + expect(hostCalls.waitForSendSettlement).toHaveBeenCalledWith( + SESSION, + 'client-1', + controller.signal + ) + expect(response).toMatchObject({ + ok: true, + result: { + ok: true, + replayed: true, + fence: 7, + cursor: { sequence: 2 }, + value: { submission: { dispatchState: 'accepted' } } + } + }) + }) + + it('returns durable pending when an older-client settlement observer cannot be retained', async () => { + hostCalls.send.mockResolvedValueOnce({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 1 }, + value: { + clientMessageId: 'client-1', + submission: { + clientMessageId: 'client-1', + fence: 1, + payloadFingerprint: 'fingerprint', + dispatchState: 'pending', + providerItemId: null, + reason: null, + submittedAt: 1, + resolvedAt: null + } + } + }) + hostCalls.waitForSendSettlement.mockResolvedValueOnce(undefined) + + const response = await call('agentSession.send', sendParams(), { + clientKind: 'runtime', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + }) + + expect(response).toMatchObject({ + ok: true, + result: { value: { submission: { dispatchState: 'pending' } } } + }) + }) + + it('returns durable pending immediately to clients that understand admission', async () => { + hostCalls.send.mockResolvedValueOnce({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 1 }, + value: { + clientMessageId: 'client-1', + submission: { + clientMessageId: 'client-1', + fence: 1, + payloadFingerprint: 'fingerprint', + dispatchState: 'pending', + providerItemId: null, + reason: null, + submittedAt: 1, + resolvedAt: null + } + } + }) + + const response = await call('agentSession.send', sendParams(), STRUCTURED_CLIENT) + + expect(hostCalls.waitForSendSettlement).not.toHaveBeenCalled() + expect(response).toMatchObject({ + ok: true, + result: { value: { submission: { dispatchState: 'pending' } } } + }) + }) + it('requires the host structured-chat setting for mobile clients', async () => { const response = await call('agentSession.send', sendParams(), STRUCTURED_MOBILE_CLIENT, { getClientSettings: () => ({ experimentalStructuredNativeChat: false }) diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index d629b1c29a1..f1d0fc59ec5 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -14,7 +14,11 @@ import { projectBackgroundTaskEvent, projectBackgroundTaskHistory } from './structured-agent-session-background-task-capability' -import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' +import { + projectTurnItemEvent, + projectTurnItemHistory +} from './structured-agent-session-turn-item-capability' +import { defineMethod, defineStreamingMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled as ensureHostInstalled, requireStructuredCapability, @@ -56,6 +60,7 @@ import { SubscribeParams, UnsubscribeParams } from './structured-agent-session-schemas' +import { sendStructuredAgentSessionForClient } from './structured-agent-session-send-compatibility' /** * The attach-shaped entries take the location from the client instead of resolving it from a @@ -86,7 +91,7 @@ async function attachClientSuppliedLocation( return host.attach(callerFor(ctx), attachParams) } -export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ +export const STRUCTURED_AGENT_SESSION_METHODS = [ defineMethod({ name: 'agentSession.rewind', params: RewindParams, @@ -190,7 +195,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ defineMethod({ name: 'agentSession.send', params: SendParams, - handler: async (params, ctx) => requireHost(ctx).send(callerFor(ctx), params) + handler: sendStructuredAgentSessionForClient }), defineMethod({ // Stopping a turn, so it stays available after admission is revoked: see the gate's rule. @@ -256,7 +261,10 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ name: 'agentSession.history', params: HistoryParams, handler: async (params, ctx) => - projectBackgroundTaskHistory(requireHost(ctx).history(params), ctx) + projectTurnItemHistory( + projectBackgroundTaskHistory(requireHost(ctx).history(params), ctx), + ctx + ) }), defineStreamingMethod({ name: 'agentSession.subscribe', @@ -283,7 +291,7 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ dispose = host.subscribe({ id: subscriptionId, sessionId: params.sessionId, - emit: (event) => emit(projectBackgroundTaskEvent(event, ctx)), + emit: (event) => emit(projectTurnItemEvent(projectBackgroundTaskEvent(event, ctx), ctx)), ...(params.cursor ? { cursor: params.cursor } : {}) }) if (stream.isClosed()) { diff --git a/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts b/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts index 151285ffb1e..aebb420fa8b 100644 --- a/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts +++ b/src/main/runtime/rpc/methods/structured-worker-stop-receipt.test.ts @@ -16,6 +16,7 @@ import { structuredWorkerProcessIncarnation } from '../../structured-worker-identity' import { ORCHESTRATION_METHODS } from './orchestration' +import { eraseRpcMethods } from '../core' const SESSION = 'session-stop-receipt' const HANDLE = 'structworker_22222222-2222-4222-a222-222222222222' @@ -43,7 +44,9 @@ describe('worker-stop on a structured worker this runtime cannot reach', () => { }) async function call(name: string, params: Record) { - const method = ORCHESTRATION_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(ORCHESTRATION_METHODS).find( + (candidate) => candidate.name === name + ) if (!method) { throw new Error(`Method not found: ${name}`) } diff --git a/src/main/runtime/rpc/methods/task-resume-state-schema.ts b/src/main/runtime/rpc/methods/task-resume-state-schema.ts index fc85cad6270..f923d3993b3 100644 --- a/src/main/runtime/rpc/methods/task-resume-state-schema.ts +++ b/src/main/runtime/rpc/methods/task-resume-state-schema.ts @@ -1,37 +1,8 @@ -import { z } from 'zod' +import type { z } from 'zod' import type { TaskResumeState as TaskResumeStateType } from '../../../../shared/ui-chrome-types' import type { AssertNoMissingKeys } from './ui-state-schema-parity' - -/** - * Tasks page-position state persisted through `ui.set`; mirrors `TaskResumeState`. - * - * This object is `.strict()` and sits behind `ui.set`'s field-level `.catch`, so a key - * a host predates makes that host drop the ENTIRE resume state — github and jira with - * it — and report success. Only add a field here when clients must agree on it across - * versions; per-device view preferences belong in client-local storage instead. - */ -export const TaskResumeState = z - .object({ - githubMode: z.enum(['items', 'project']).optional(), - githubItemsPreset: z.string().nullable().optional(), - githubItemsQuery: z.string().optional(), - githubProjectHiddenFieldIdsByView: z.record(z.string(), z.array(z.string())).optional(), - linearMode: z.enum(['issues', 'projects', 'views', 'in-orca']).optional(), - linearPreset: z.enum(['assigned', 'created', 'all', 'completed']).optional(), - linearQuery: z.string().optional(), - linearContext: z - .object({ - kind: z.enum(['project', 'view']), - id: z.string(), - workspaceId: z.string(), - model: z.enum(['issue', 'project']).optional() - }) - .strict() - .optional(), - jiraPreset: z.enum(['assigned', 'reported', 'all', 'done']).optional(), - jiraQuery: z.string().optional() - }) - .strict() +import { TaskResumeState } from '../../../../shared/rpc-contract/task-resume-state-params' +export { TaskResumeState } const _taskResumeStateParity: AssertNoMissingKeys< TaskResumeStateType, diff --git a/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts b/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts index 51001605be0..a1e221d5a7a 100644 --- a/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts +++ b/src/main/runtime/rpc/methods/terminal-create-idempotency.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it, vi } from 'vitest' -import type { RpcContext } from '../core' +import { eraseRpcMethods, type RpcContext } from '../core' import { TERMINAL_METHODS } from './terminal' describe('terminal.create RPC idempotency', () => { @@ -15,7 +15,9 @@ describe('terminal.create RPC idempotency', () => { run: (worktree: string | undefined, handle: string | undefined) => Promise ) => run('id:worktree-1', 'term_stable') ) - const method = TERMINAL_METHODS.find((candidate) => candidate.name === 'terminal.create') + const method = eraseRpcMethods(TERMINAL_METHODS).find( + (candidate) => candidate.name === 'terminal.create' + ) if (!method) { throw new Error('terminal.create method missing') } @@ -73,7 +75,9 @@ describe('terminal.create RPC idempotency', () => { run: (worktree: string | undefined, handle: string | undefined) => Promise ) => run('id:worktree-1', undefined) ) - const method = TERMINAL_METHODS.find((candidate) => candidate.name === 'terminal.create') + const method = eraseRpcMethods(TERMINAL_METHODS).find( + (candidate) => candidate.name === 'terminal.create' + ) if (!method) { throw new Error('terminal.create method missing') } @@ -114,7 +118,9 @@ describe('terminal.create RPC idempotency', () => { run: (worktree: string | undefined, handle: string | undefined) => Promise ) => run('id:worktree-1', undefined) ) - const method = TERMINAL_METHODS.find((candidate) => candidate.name === 'terminal.create') + const method = eraseRpcMethods(TERMINAL_METHODS).find( + (candidate) => candidate.name === 'terminal.create' + ) if (!method) { throw new Error('terminal.create method missing') } diff --git a/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts b/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts index ccdcf5fb7b1..e26788d5d46 100644 --- a/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts +++ b/src/main/runtime/rpc/methods/terminal-manifest-characterization.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it, vi } from 'vitest' import type { OrcaRuntimeService } from '../../orca-runtime' import { TERMINAL_METHODS } from './terminal' +import { eraseRpcMethods } from '../core' import { TerminalMultiplexLegacyAckFrame, TerminalMultiplexSourceRangeAckFrame, @@ -50,14 +51,14 @@ const METHOD_CASES: readonly (readonly [string, unknown, boolean])[] = [ ] function schemaFor(name: string) { - const method = TERMINAL_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(TERMINAL_METHODS).find((candidate) => candidate.name === name) if (!method?.params) { throw new Error(`Missing terminal schema: ${name}`) } return method.params } async function invoke(name: string, params: unknown, runtime: Partial) { - const method = TERMINAL_METHODS.find((candidate) => candidate.name === name) + const method = eraseRpcMethods(TERMINAL_METHODS).find((candidate) => candidate.name === name) if (!method?.params || 'stream' in method) { throw new Error(`Missing unary terminal method: ${name}`) } diff --git a/src/main/runtime/rpc/methods/terminal-orphan.ts b/src/main/runtime/rpc/methods/terminal-orphan.ts index 97296d0fac5..0b728629dcb 100644 --- a/src/main/runtime/rpc/methods/terminal-orphan.ts +++ b/src/main/runtime/rpc/methods/terminal-orphan.ts @@ -1,110 +1,7 @@ -import { z } from 'zod' -import type { TabGroupLayoutNode } from '../../../../shared/tab-types' -import { isPtyIncarnationId, type PtyIncarnationId } from '../../../../shared/pty-incarnation' -import { defineMethod, type RpcAnyMethod } from '../core' -import { OptionalString, requiredString } from '../schemas' -import { TerminalPaneLayoutNodeSchema } from './session-tabs-schemas' +import { defineMethod } from '../core' +import { TerminalAdoptOrphans } from '../../../../shared/rpc-contract/terminal-orphan-params' -function parseOrphanGroupLayout(value: unknown): TabGroupLayoutNode | null { - const stack: { value: unknown; depth: number }[] = [{ value, depth: 0 }] - let count = 0 - while (stack.length > 0) { - const current = stack.pop()! - if ( - current.depth > 64 || - ++count > 1_024 || - !current.value || - typeof current.value !== 'object' - ) { - return null - } - const node = current.value as Record - if (node.type === 'leaf') { - if ( - typeof node.groupId !== 'string' || - node.groupId.length < 1 || - node.groupId.length > 256 - ) { - return null - } - continue - } - if ( - node.type !== 'split' || - (node.direction !== 'horizontal' && node.direction !== 'vertical') || - (node.ratio !== undefined && - (typeof node.ratio !== 'number' || - !Number.isFinite(node.ratio) || - node.ratio < 0 || - node.ratio > 1)) - ) { - return null - } - stack.push( - { value: node.first, depth: current.depth + 1 }, - { value: node.second, depth: current.depth + 1 } - ) - } - return value as TabGroupLayoutNode -} - -const TerminalOrphanGroupLayout = z - .unknown() - .transform(parseOrphanGroupLayout) - .pipe(z.custom((value) => value !== null, 'Invalid orphan group layout')) - -const TerminalOrphanTopology = z.object({ - tabs: z - .array( - z.object({ - tabId: requiredString('Missing topology tab id').pipe(z.string().max(256)), - root: TerminalPaneLayoutNodeSchema, - activeLeafId: requiredString('Missing active leaf id').pipe(z.string().max(128)), - expandedLeafId: z.string().max(128).nullable() - }) - ) - .min(1) - .max(64), - groups: z - .array( - z.object({ - id: z.string().min(1).max(256), - activeTabId: z.string().min(1).max(256), - tabOrder: z.array(z.string().min(1).max(256)).min(1).max(64), - recentTabIds: z.array(z.string().min(1).max(256)).max(64).optional() - }) - ) - .min(1) - .max(64), - groupLayout: TerminalOrphanGroupLayout.optional() -}) - -const TerminalOrphanIncarnationId = z.custom( - isPtyIncarnationId, - 'Invalid PTY incarnation' -) - -const TerminalAdoptOrphans = z.object({ - worktree: requiredString('Missing worktree selector').pipe(z.string().max(32_768)), - expectedTopologyRevision: z.number().int().nonnegative(), - claims: z - .array( - z.object({ - terminal: requiredString('Missing terminal handle').pipe(z.string().max(256)), - ptyId: requiredString('Missing PTY id').pipe(z.string().max(8_192)), - incarnationId: TerminalOrphanIncarnationId, - tabId: requiredString('Missing tab id').pipe(z.string().max(256)), - leafId: requiredString('Missing leaf id').pipe(z.string().max(128)) - }) - ) - .min(1) - .max(64), - activeTabId: OptionalString.pipe(z.string().max(256).optional()), - activeGroupId: OptionalString.pipe(z.string().max(256).optional()), - topology: TerminalOrphanTopology.optional() -}) - -export const TERMINAL_ORPHAN_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_ORPHAN_METHODS = [ defineMethod({ name: 'terminal.adoptOrphans', params: TerminalAdoptOrphans, diff --git a/src/main/runtime/rpc/methods/terminal-quick-command-rpc-schema.ts b/src/main/runtime/rpc/methods/terminal-quick-command-rpc-schema.ts index 18661f10b57..9d4511bbd4d 100644 --- a/src/main/runtime/rpc/methods/terminal-quick-command-rpc-schema.ts +++ b/src/main/runtime/rpc/methods/terminal-quick-command-rpc-schema.ts @@ -1,73 +1 @@ -import { z } from 'zod' -import type { TerminalQuickCommand } from '../../../../shared/terminal-quick-command-types' -import { - MAX_QUICK_COMMAND_AGENT_PROMPT_LENGTH, - MAX_QUICK_COMMAND_ID_LENGTH, - MAX_QUICK_COMMAND_LABEL_LENGTH, - MAX_QUICK_COMMAND_REPO_ID_LENGTH, - MAX_QUICK_COMMAND_TERMINAL_TEXT_LENGTH, - normalizeTerminalQuickCommands, - supportsTerminalAgentQuickCommand -} from '../../../../shared/terminal-quick-commands' - -const TerminalQuickCommandScopeUpdate = z.discriminatedUnion('type', [ - z.object({ type: z.literal('global') }).strict(), - z - .object({ - type: z.literal('repo'), - repoId: z.string().max(MAX_QUICK_COMMAND_REPO_ID_LENGTH) - }) - .strict() -]) - -const TerminalQuickCommandUpdateItem = z.union([ - z - .object({ - id: z.string().max(MAX_QUICK_COMMAND_ID_LENGTH), - label: z.string().max(MAX_QUICK_COMMAND_LABEL_LENGTH), - action: z.literal('terminal-command').optional(), - command: z.string().max(MAX_QUICK_COMMAND_TERMINAL_TEXT_LENGTH), - appendEnter: z.boolean(), - scope: TerminalQuickCommandScopeUpdate.optional() - }) - .strict(), - z - .object({ - id: z.string().max(MAX_QUICK_COMMAND_ID_LENGTH), - label: z.string().max(MAX_QUICK_COMMAND_LABEL_LENGTH), - action: z.literal('agent-prompt'), - agent: z.custom(supportsTerminalAgentQuickCommand, { - message: 'Agent does not support prompt commands' - }), - prompt: z.string().max(MAX_QUICK_COMMAND_AGENT_PROMPT_LENGTH), - scope: TerminalQuickCommandScopeUpdate.optional() - }) - .strict() -]) - -export const TerminalQuickCommandsUpdate = z - .object({ - // Why: a single host-side mutation preserves unrelated desktop/mobile edits - // and avoids retransmitting the full ~240 KB list for every small change. - mutation: z.union([ - z - .object({ - type: z.literal('upsert'), - command: TerminalQuickCommandUpdateItem.transform( - (value) => normalizeTerminalQuickCommands([value])[0] - ).pipe( - z.custom((value) => value !== undefined, { - message: 'Quick command cannot be normalized' - }) - ) - }) - .strict(), - z - .object({ - type: z.literal('delete'), - id: z.string().min(1).max(MAX_QUICK_COMMAND_ID_LENGTH) - }) - .strict() - ]) - }) - .strict() +export { TerminalQuickCommandsUpdate } from '../../../../shared/rpc-contract/terminal-quick-command-params' diff --git a/src/main/runtime/rpc/methods/terminal.ts b/src/main/runtime/rpc/methods/terminal.ts index e80d07773dc..607a29329cd 100644 --- a/src/main/runtime/rpc/methods/terminal.ts +++ b/src/main/runtime/rpc/methods/terminal.ts @@ -1,4 +1,3 @@ -import type { RpcAnyMethod } from '../core' import { TERMINAL_LIFECYCLE_METHODS } from './terminal/terminal-lifecycle-methods' import { TERMINAL_MULTIPLEX_METHODS } from './terminal/terminal-multiplex-method' import { TERMINAL_QUERY_METHODS } from './terminal/terminal-query-methods' @@ -11,7 +10,7 @@ import { // The manifest order is part of the released RPC contract. Keep composition here so the // public entry point owns registration rather than forwarding an aggregated child export. -export const TERMINAL_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_METHODS = [ ...TERMINAL_QUERY_METHODS, ...TERMINAL_SEND_METHODS, ...TERMINAL_LIFECYCLE_METHODS, diff --git a/src/main/runtime/rpc/methods/terminal/stream-schemas.ts b/src/main/runtime/rpc/methods/terminal/stream-schemas.ts index 04a6990fe51..f3e6e3a7870 100644 --- a/src/main/runtime/rpc/methods/terminal/stream-schemas.ts +++ b/src/main/runtime/rpc/methods/terminal/stream-schemas.ts @@ -1,43 +1,12 @@ import { z } from 'zod' import { requiredString } from '../../schemas' import { TerminalViewport } from './unary-schemas' - -const TerminalHandle = z.object({ terminal: requiredString('Missing terminal handle') }) - -export const TerminalResizeForClient = z.discriminatedUnion('mode', [ - z.object({ - terminal: requiredString('Missing terminal handle'), - mode: z.literal('mobile-fit'), - cols: z.number().finite().positive(), - rows: z.number().finite().positive(), - clientId: requiredString('Missing client ID') - }), - z.object({ - terminal: requiredString('Missing terminal handle'), - mode: z.literal('restore'), - clientId: requiredString('Missing client ID') - }) -]) - -export const TerminalSubscribe = TerminalHandle.extend({ - client: z - .object({ - id: requiredString('Missing client ID'), - type: z.enum(['mobile', 'desktop']).default('desktop') - }) - .optional(), - viewport: TerminalViewport.optional(), - capabilities: z - .object({ - terminalBinaryStream: z.literal(1).optional(), - desktopViewportClaims: z.literal(1).optional(), - mobileInputLeaseOnly: z.literal(1).optional(), - writeUnavailable: z.literal(1).optional() - }) - .optional() -}) - -export const TerminalMultiplex = z.object({}) +import { TerminalHandle } from '../../../../../shared/rpc-contract/terminal-stream-params' +export { + TerminalMultiplex, + TerminalResizeForClient, + TerminalSubscribe +} from '../../../../../shared/rpc-contract/terminal-stream-params' export const TerminalMultiplexSubscribeFrame = TerminalHandle.extend({ streamId: z.number().int().min(1), diff --git a/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts index 48649bc372c..8377f923ecc 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-inspect-process-params.test.ts @@ -6,10 +6,13 @@ import { describe, expect, it, vi } from 'vitest' import type { ZodType } from 'zod' import { TERMINAL_QUERY_METHODS } from './terminal-query-methods' import { TerminalHandle, TerminalInspectProcess } from './unary-schemas' +import { eraseRpcMethods } from '../../core' /** The method as registered, so a schema swap on the definition cannot pass unseen. */ function inspectProcessMethod() { - const method = TERMINAL_QUERY_METHODS.find((entry) => entry.name === 'terminal.inspectProcess') + const method = eraseRpcMethods(TERMINAL_QUERY_METHODS).find( + (entry) => entry.name === 'terminal.inspectProcess' + ) if (!method) { throw new Error('terminal.inspectProcess is not registered') } @@ -25,7 +28,7 @@ async function callRegisteredHandler( foregroundProcess: null, hasChildProcesses: false })) - await method.handler(parsed, { runtime: { inspectTerminalProcess } } as never, undefined as never) + await method.handler(parsed, { runtime: { inspectTerminalProcess } } as never) const [terminal, options] = inspectTerminalProcess.mock.calls[0] as unknown as [string, unknown] return { terminal, options } } diff --git a/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts index 51dde7df4d8..891ed65a13f 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-lifecycle-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../../core' +import { defineMethod } from '../../core' import { navigationTargetsHost, resolveRuntimeNavigationTarget @@ -19,7 +19,7 @@ import { } from './unary-schemas' import { TerminalResizeForClient } from './stream-schemas' -export const TERMINAL_LIFECYCLE_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_LIFECYCLE_METHODS = [ defineMethod({ name: 'terminal.wait', params: TerminalWait, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts index 11fd0c4d039..e811614e400 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-multiplex-method.ts @@ -1,4 +1,4 @@ -import { defineStreamingMethod, type RpcAnyMethod } from '../../core' +import { defineStreamingMethod } from '../../core' import { TerminalStreamOpcode } from '../../../../../shared/terminal-stream-protocol' import { TERMINAL_MULTIPLEX_ACK_TOTAL_INITIAL_WINDOW_BYTES } from '../../../../../shared/terminal-multiplex-flow-control' import { TerminalSourceRangeRegistry } from '../../terminal-source-range-registry' @@ -11,7 +11,7 @@ import { installMultiplexCleanup } from './terminal-multiplex-cleanup' import { installMultiplexSlotFrames } from './terminal-multiplex-slot-frames' import { installMultiplexSubscribeFrame } from './terminal-multiplex-subscribe-frame' -export const TERMINAL_MULTIPLEX_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_MULTIPLEX_METHODS = [ defineStreamingMethod({ name: 'terminal.multiplex', params: TerminalMultiplex, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts index 82edd55cd79..52c1063b0cc 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-query-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcAnyMethod } from '../../core' +import { defineMethod } from '../../core' import { TerminalHandle, TerminalInspectProcess, @@ -10,7 +10,7 @@ import { TerminalResolvePane } from './unary-schemas' -export const TERMINAL_QUERY_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_QUERY_METHODS = [ defineMethod({ name: 'terminal.list', params: TerminalListParams, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts index ad471098e49..c62c70c5905 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-send-method.ts @@ -1,6 +1,6 @@ import { isAgentSessionPtyWriteRefusedError } from '../../../../../shared/agent-session-pty-write-admission' import { assertLegacyAiVaultResumeCommandAllowed } from '../../../../ai-vault/structured-session-ownership' -import { InvalidArgumentError, defineMethod, type RpcAnyMethod } from '../../core' +import { InvalidArgumentError, defineMethod } from '../../core' import { isTerminalQueryReply } from '../../../../../shared/terminal-query-reply' import { assertTerminalAgentSendable } from '../../terminal-agent-send-guard' import { TerminalSend } from './unary-schemas' @@ -20,7 +20,7 @@ import { observeReplayedTerminalPrompt } from './terminal-prompt-receipt' -export const TERMINAL_SEND_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_SEND_METHODS = [ defineMethod({ name: 'terminal.send', params: TerminalSend, diff --git a/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts b/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts index bd16382d747..7572d41496f 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-subscribe-method.ts @@ -1,4 +1,4 @@ -import { defineStreamingMethod, type RpcAnyMethod } from '../../core' +import { defineStreamingMethod } from '../../core' import { TerminalSubscribe } from './stream-schemas' import { isTerminalReadPayloadIncomplete } from './terminal-stream-replay' import { runTerminalBinarySubscription } from './terminal-legacy-subscribe-binary' @@ -8,7 +8,7 @@ import { } from './terminal-legacy-simple-subscriptions' import type { TerminalSubscriptionArgs } from './terminal-legacy-subscription-types' -export const TERMINAL_SUBSCRIBE_METHODS: RpcAnyMethod[] = [ +export const TERMINAL_SUBSCRIBE_METHODS = [ // Streams live terminal output over WebSocket; mobile clients pass client+viewport for server-side auto-fit. defineStreamingMethod({ name: 'terminal.subscribe', diff --git a/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts b/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts index 71feee4f24d..91bdf25d84d 100644 --- a/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts +++ b/src/main/runtime/rpc/methods/terminal/terminal-viewport-methods.ts @@ -1,5 +1,4 @@ -import { z } from 'zod' -import { defineMethod, type RpcAnyMethod } from '../../core' +import { defineMethod } from '../../core' import { TerminalHandle } from './unary-schemas' import { TerminalSetAutoRestoreFit, @@ -8,8 +7,9 @@ import { TerminalUpdateViewport } from './viewport-schemas' import { updateViewportForClient } from './terminal-viewport-update' +import { TerminalGetAutoRestoreFitParams } from '../../../../../shared/rpc-contract/terminal-viewport-methods-params' -export const TERMINAL_VIEWPORT_METHODS_BEFORE_STREAMS: RpcAnyMethod[] = [ +export const TERMINAL_VIEWPORT_METHODS_BEFORE_STREAMS = [ defineMethod({ name: 'terminal.setDisplayMode', params: TerminalSetDisplayMode, @@ -78,7 +78,7 @@ export const TERMINAL_VIEWPORT_METHODS_BEFORE_STREAMS: RpcAnyMethod[] = [ }) ] -export const TERMINAL_VIEWPORT_METHODS_AFTER_STREAMS: RpcAnyMethod[] = [ +export const TERMINAL_VIEWPORT_METHODS_AFTER_STREAMS = [ defineMethod({ name: 'terminal.unsubscribe', params: TerminalUnsubscribe, @@ -105,7 +105,7 @@ export const TERMINAL_VIEWPORT_METHODS_AFTER_STREAMS: RpcAnyMethod[] = [ }), defineMethod({ name: 'terminal.getAutoRestoreFit', - params: z.object({}), + params: TerminalGetAutoRestoreFitParams, handler: async (_params, { runtime }) => ({ ms: runtime.getMobileAutoRestoreFitMs() }) diff --git a/src/main/runtime/rpc/methods/terminal/unary-schemas.ts b/src/main/runtime/rpc/methods/terminal/unary-schemas.ts index 89928128e69..0b7f9d90408 100644 --- a/src/main/runtime/rpc/methods/terminal/unary-schemas.ts +++ b/src/main/runtime/rpc/methods/terminal/unary-schemas.ts @@ -1,222 +1,22 @@ -import { z } from 'zod' -import { OptionalFiniteNumber, OptionalString, requiredString } from '../../schemas' -import { TERMINAL_PANE_SPLIT_SOURCES } from '../../../../../shared/feature-education-telemetry' -import { isTuiAgent } from '../../../../../shared/tui-agent-config' - -export const TerminalHandle = z.object({ - terminal: requiredString('Missing terminal handle'), - // Additive fence understood by newer hosts; legacy hosts safely ignore it. - expectedIncarnationId: requiredString('Missing PTY incarnation').optional() -}) - -export const TerminalFocus = TerminalHandle.extend({ - navigation: z.enum(['caller', 'host']).optional() -}) - -/** - * `terminal.inspectProcess` carries one member the sibling handle methods must not: whether the - * caller's answer decides something once, which is what licenses the host to pay for a process-table - * read. Extended rather than added to `TerminalHandle` so `clearBuffer`/`agentStatus`/`isRunningAgent` - * keep refusing an option they have no use for. - */ -export const TerminalInspectProcess = TerminalHandle.extend({ - // Additive request member understood by newer hosts; legacy hosts safely ignore it. - scanChildProcesses: z.boolean().optional() -}) - -export const TerminalListParams = z.object({ - worktree: OptionalString, - limit: OptionalFiniteNumber, - handles: z - .array(requiredString('Missing terminal handle').pipe(z.string().max(256))) - .max(64) - .optional(), - requireFreshPtyLiveness: z.boolean().optional(), - // Why: layouts are ~31% of a large listing and only the human CLI formatter - // reads them. Absent means "include" so pre-flag clients keep rendering them. - includeVisualLayouts: z.boolean().optional() -}) - -export const TerminalResolveActive = z.object({ - worktree: OptionalString, - /** Refuse instead of guessing when several leaves could be the caller's own terminal. */ - requireUnambiguous: z.boolean().optional() -}) - -export const TerminalResolvePane = z.object({ - paneKey: requiredString('Missing pane key'), - worktreeId: OptionalString -}) - -export const TerminalRecoverPane = z.object({ - paneKey: requiredString('Missing pane key'), - worktreeId: requiredString('Missing worktree ID'), - expectedTerminal: requiredString('Missing expected terminal handle').optional() -}) - -export const TerminalRead = TerminalHandle.extend({ - cursor: z - .unknown() - .transform((value) => { - if (value === undefined) { - return undefined - } - if (typeof value !== 'number' || !Number.isInteger(value) || value < 0) { - return Number.NaN - } - return value - }) - .pipe( - z - .number() - .optional() - .refine((v) => v === undefined || Number.isFinite(v), { - message: 'Cursor must be a non-negative integer' - }) - ) - .optional(), - limit: OptionalFiniteNumber, - // Why: optional so an older host that does not understand it simply drops the key and answers - // with its usual stream read; the response's `source` is what tells the caller which it got. - screen: z.literal(true).optional() -}).refine((params) => !(params.screen === true && params.cursor !== undefined), { - // Why: a cursor pages through accumulated output; a screen is the current frame with nothing - // behind it. Honoring both would answer with rendered lines carrying the stream's pagination - // metadata — two frames of reference in one payload, which is the confusion `source` exists to - // remove. The CLI already refuses the pair, but the RPC is reachable without it. - message: 'Cursor cannot be combined with a screen read' -}) - -// Why: preserve the legacy contract — `title: string | null` only, `undefined` rejected, so the CLI's "reset" signal stays distinct. -export const TerminalRename = TerminalHandle.extend({ - title: z.custom((value) => value === null || typeof value === 'string', { - message: 'Missing --title (pass empty string or null to reset)' - }) -}) - -export const TerminalSend = TerminalHandle.extend({ - text: OptionalString, - enter: z.unknown().optional(), - interrupt: z.unknown().optional(), - // Why: older hosts strip this optional intent and retain their direct-send behavior. - agentPrompt: z.literal(true).optional(), - // Why: waiting observes the same prompt receipt; it never authorizes a second write. - waitSubmitMs: z.number().int().min(0).max(3_600_000).optional(), - resolvedLaunchDraft: z - .object({ - text: z.string(), - createdAt: z.number().finite() - }) - .optional(), - requireAgentStatus: z.enum(['sendable']).optional(), - // Why: terminal-generated replies are valid input but must not transfer the shared terminal floor. - inputKind: z.enum(['query-reply']).optional(), - // Why: identifies the caller for the driver state machine; when absent (older clients) the server falls back to the most recent mobile actor (docs/mobile-presence-lock.md). - client: z - .object({ - id: requiredString('Missing client ID'), - type: z.enum(['mobile', 'desktop']).default('desktop').optional() - }) - .optional(), - viewport: z - .object({ - cols: z.number().int().min(1).max(1000), - rows: z.number().int().min(1).max(500) - }) - .optional(), - claimViewport: z.literal(true).optional() -}) - -export const TerminalViewport = z.object({ - cols: z.number().int().min(1).max(1000), - rows: z.number().int().min(1).max(500) -}) - -export const TerminalWait = TerminalHandle.extend({ - for: z.custom<'exit' | 'tui-idle'>((value) => value === 'exit' || value === 'tui-idle', { - message: 'Invalid --for value. Supported: exit, tui-idle' - }), - timeoutMs: OptionalFiniteNumber -}) - -export const TerminalCreateParams = z.object({ - worktree: OptionalString, - clientMutationId: z.string().min(1).max(128).optional(), - reconcileExisting: z.boolean().optional(), - command: OptionalString, - startupCommandDelivery: z.enum(['fast', 'shell-ready']).optional(), - env: z.record(z.string(), z.string()).optional(), - envToDelete: z.array(z.string().min(1).max(256)).max(32).optional(), - launchConfig: z - .object({ - agentCommand: z.string().optional(), - agentArgs: z.string(), - agentEnv: z.record(z.string(), z.string()), - ompResumeFilePath: z - .string() - .min(1) - .max(32 * 1024) - .optional() - }) - .optional(), - resumeProviderSession: z - .object({ - key: z.enum(['session_id', 'conversation_id']), - id: z.string().min(1).max(512), - transcriptPath: z.string().min(1).max(32_768).optional() - }) - .optional(), - launchToken: OptionalString, - launchAgent: z.string().refine(isTuiAgent).optional(), - terminalColorQueryReplies: z - .object({ - foreground: z.string().max(128).optional(), - background: z.string().max(128).optional() - }) - .optional(), - title: OptionalString, - focus: z.unknown().optional(), - rendererBacked: z.unknown().optional(), - activate: z.unknown().optional(), - presentation: z.enum(['background', 'focused']).optional(), - tabId: OptionalString, - leafId: OptionalString -}) - -export const TerminalSplit = TerminalHandle.extend({ - direction: z - .unknown() - .transform((v) => (v === 'vertical' || v === 'horizontal' ? v : undefined)) - .pipe(z.union([z.enum(['vertical', 'horizontal']), z.undefined()])) - .optional(), - command: OptionalString, - env: z.record(z.string(), z.string()).optional(), - telemetrySource: z.enum(TERMINAL_PANE_SPLIT_SOURCES).optional() -}) - -export const TerminalStop = z.object({ - worktree: requiredString('Missing worktree selector') -}) - -export const TerminalCloseAll = TerminalStop - -export const TerminalSleep = TerminalStop - -export const TerminalStopExact = TerminalStop.extend({ - expectedPtyIds: z.array(requiredString('Missing PTY ID')).min(1), - keepHistory: z.boolean().optional(), - targetOnly: z.boolean().optional() -}) - -export const AgentTeamsTmuxCompat = z.object({ - teamId: requiredString('Missing agent team ID'), - token: requiredString('Missing agent team token'), - envPane: requiredString('Missing tmux pane identity'), - cwd: OptionalString, - argv: z.array(z.string()) -}) - -export const AgentTeamsPrepareLaunch = z.object({ - paneKey: requiredString('Missing pane key'), - env: z.record(z.string(), z.string()).optional() -}) +export { + AgentTeamsPrepareLaunch, + AgentTeamsTmuxCompat, + TerminalCloseAll, + TerminalCreateParams, + TerminalFocus, + TerminalHandle, + TerminalInspectProcess, + TerminalListParams, + TerminalRead, + TerminalRecoverPane, + TerminalRename, + TerminalResolveActive, + TerminalResolvePane, + TerminalSend, + TerminalSleep, + TerminalSplit, + TerminalStop, + TerminalStopExact, + TerminalViewport, + TerminalWait +} from '../../../../../shared/rpc-contract/terminal-unary-params' diff --git a/src/main/runtime/rpc/methods/terminal/viewport-schemas.ts b/src/main/runtime/rpc/methods/terminal/viewport-schemas.ts index d9b76d66ea1..70ecd9037d1 100644 --- a/src/main/runtime/rpc/methods/terminal/viewport-schemas.ts +++ b/src/main/runtime/rpc/methods/terminal/viewport-schemas.ts @@ -1,51 +1,6 @@ -import { z } from 'zod' -import { requiredString } from '../../schemas' - -const TerminalHandle = z.object({ terminal: requiredString('Missing terminal handle') }) - -export const TerminalSetDisplayMode = TerminalHandle.extend({ - // Why: 'auto' = mobile drives dims while subscribed (desktop restores on last-leave); 'desktop' = no resize, mobile scales to fit. - mode: z.enum(['auto', 'desktop']), - // Why: identifies the caller for the driver state machine; optional for older mobile clients. - client: z - .object({ - id: requiredString('Missing client ID'), - type: z.enum(['mobile', 'desktop']).default('desktop').optional() - }) - .optional(), - // Why: carries the measured viewport so an 'auto' toggle on a viewport-less record can phone-fit instead of no-op'ing. - viewport: z - .object({ - cols: z.number().int().positive(), - rows: z.number().int().positive() - }) - .optional() -}) - -export const TerminalUnsubscribe = z.object({ - subscriptionId: requiredString('Missing subscription ID'), - // Why: lets the server rebuild the composite `${terminal}:${clientId}` cleanup key when older clients pass a bare subscriptionId (docs/mobile-presence-lock.md). - client: z - .object({ - id: requiredString('Missing client ID') - }) - .optional() -}) - -// Why: in-place update avoids an unsubscribe→resubscribe that flashed the lock banner and stranded the PTY at phone dims (docs/mobile-presence-lock.md). -export const TerminalUpdateViewport = TerminalHandle.extend({ - client: z.object({ - id: requiredString('Missing client ID'), - type: z.enum(['mobile', 'desktop']).default('mobile').optional() - }), - viewport: z.object({ - cols: z.number().int().min(20).max(240), - rows: z.number().int().min(8).max(120) - }), - claim: z.boolean().optional() -}) - -// Why: phone-fit auto-restore preference (docs/mobile-fit-hold.md); `null` = Indefinite, finite ms clamped to [5_000, 60min] server-side. -export const TerminalSetAutoRestoreFit = z.object({ - ms: z.number().nullable() -}) +export { + TerminalSetAutoRestoreFit, + TerminalSetDisplayMode, + TerminalUnsubscribe, + TerminalUpdateViewport +} from '../../../../../shared/rpc-contract/terminal-viewport-schemas-params' diff --git a/src/main/runtime/rpc/methods/ui-update-value-tolerance.ts b/src/main/runtime/rpc/methods/ui-update-value-tolerance.ts index 1b8ca0c2cd4..e03a9bfad06 100644 --- a/src/main/runtime/rpc/methods/ui-update-value-tolerance.ts +++ b/src/main/runtime/rpc/methods/ui-update-value-tolerance.ts @@ -1,25 +1,4 @@ -import type { z } from 'zod' - -/** - * `UiUpdate` rides App.tsx's debounced writer, so one drifted enum member used - * to fail the WHOLE batch and silently drop sidebar widths, filters and agent - * acks alongside it. Degrade instead: a value the schema cannot express is - * dropped from the payload and the rest of the batch still lands. Unknown KEYS - * stay a hard rejection — the parity assertions exist to catch those. - */ -export function tolerateUnknownValues(shape: TShape): TShape { - return Object.fromEntries( - Object.entries(shape).map(([key, schema]) => [ - key, - (schema as z.ZodType).catch(() => undefined) - ]) - ) as unknown as TShape -} - -/** Drops the `undefined` entries `tolerateUnknownValues` leaves behind, so a - * rejected value reads as absent rather than as an explicit clear. */ -export function omitUndefinedValues>(value: TValue): TValue { - return Object.fromEntries( - Object.entries(value).filter(([, entry]) => entry !== undefined) - ) as TValue -} +export { + omitUndefinedValues, + tolerateUnknownValues +} from '../../../../shared/rpc-contract/ui-update-value-tolerance-params' diff --git a/src/main/runtime/rpc/methods/updater.test.ts b/src/main/runtime/rpc/methods/updater.test.ts index 9c3ce0ef810..1a925cbe767 100644 --- a/src/main/runtime/rpc/methods/updater.test.ts +++ b/src/main/runtime/rpc/methods/updater.test.ts @@ -1,4 +1,5 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' +import { eraseRpcMethods, type RpcMethodDeclaration } from '../core' import { configureRemoteServerUpdater } from '../../remote-server-updater' import { STATUS_METHODS } from './status' import { UPDATER_METHODS } from './updater' @@ -10,8 +11,8 @@ const snapshot = { status: { state: 'available', version: '1.5.1', changelog: null } } as const -function handler(methods: typeof UPDATER_METHODS, name: string) { - const method = methods.find((candidate) => candidate.name === name) +function handler(methods: readonly RpcMethodDeclaration[], name: string) { + const method = eraseRpcMethods(methods).find((candidate) => candidate.name === name) if (!method) { throw new Error(`Missing method ${name}`) } diff --git a/src/main/runtime/rpc/methods/updater.ts b/src/main/runtime/rpc/methods/updater.ts index 1baa2aff53b..a14fb5ab6a4 100644 --- a/src/main/runtime/rpc/methods/updater.ts +++ b/src/main/runtime/rpc/methods/updater.ts @@ -1,13 +1,13 @@ -import { defineMethod, type RpcMethod } from '../core' -import { z } from 'zod' +import { defineMethod } from '../core' import { checkRemoteServerUpdater, downloadRemoteServerUpdater, getRemoteServerUpdaterSnapshot, installRemoteServerUpdater } from '../../remote-server-updater' +import { UpdaterCheckParams } from '../../../../shared/rpc-contract/updater-params' -export const UPDATER_METHODS: RpcMethod[] = [ +export const UPDATER_METHODS = [ defineMethod({ name: 'updater.getStatus', params: null, @@ -15,10 +15,7 @@ export const UPDATER_METHODS: RpcMethod[] = [ }), defineMethod({ name: 'updater.check', - params: z.object({ - includePrerelease: z.boolean().optional(), - includePerfPrerelease: z.boolean().optional() - }), + params: UpdaterCheckParams, handler: (params, { runtime }) => checkRemoteServerUpdater(runtime.getRuntimeId(), params) }), defineMethod({ diff --git a/src/main/runtime/rpc/methods/workspace-cleanup-ui-schema.ts b/src/main/runtime/rpc/methods/workspace-cleanup-ui-schema.ts index 624a63ff44e..11e6cab4ee3 100644 --- a/src/main/runtime/rpc/methods/workspace-cleanup-ui-schema.ts +++ b/src/main/runtime/rpc/methods/workspace-cleanup-ui-schema.ts @@ -1,30 +1 @@ -import { z } from 'zod' -import { - normalizeWorkspaceCleanupBrowseState, - type WorkspaceCleanupBrowseState -} from '../../../../shared/workspace-cleanup-browse-state' - -const WorkspaceCleanupDismissal = z.object({ - worktreeId: z.string(), - dismissedAt: z.number().finite(), - fingerprint: z.string(), - classifierVersion: z.number().finite(), - executionHostId: z.string().min(1).optional() -}) - -/** - * Deliberately unvalidated shape, then normalized: the filter groups must NOT be - * strict or enumerated here. A newer client sends filters this build has never - * heard of, and a per-field zod shape would reject the whole `ui.set` payload - * instead of persisting the parts the host does understand. The shared - * normalizer never throws and degrades field by field, so an older host narrows - * the state rather than refusing it. - */ -const WorkspaceCleanupBrowse = z - .custom() - .transform((value) => normalizeWorkspaceCleanupBrowseState(value)) - -export const WorkspaceCleanup = z.object({ - dismissals: z.record(z.string(), WorkspaceCleanupDismissal), - browse: WorkspaceCleanupBrowse.optional() -}) +export { WorkspaceCleanup } from '../../../../shared/rpc-contract/workspace-cleanup-ui-params' diff --git a/src/main/runtime/rpc/methods/workspace-ports.ts b/src/main/runtime/rpc/methods/workspace-ports.ts index 9a6ff82764b..c96c91b436d 100644 --- a/src/main/runtime/rpc/methods/workspace-ports.ts +++ b/src/main/runtime/rpc/methods/workspace-ports.ts @@ -1,18 +1,10 @@ -import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' -import { OptionalString, requiredNumber } from '../schemas' +import { defineMethod } from '../core' +import { + WorkspacePortKillParams, + WorkspacePortScanParams +} from '../../../../shared/rpc-contract/workspace-ports-params' -const WorkspacePortScanParams = z.object({ - repoId: OptionalString -}) - -const WorkspacePortKillParams = z.object({ - repoId: OptionalString, - pid: requiredNumber('Missing process id'), - port: requiredNumber('Missing port') -}) - -export const WORKSPACE_PORT_METHODS: RpcMethod[] = [ +export const WORKSPACE_PORT_METHODS = [ defineMethod({ name: 'workspacePorts.scan', params: WorkspacePortScanParams, diff --git a/src/main/runtime/rpc/methods/worktree-catalog-methods.ts b/src/main/runtime/rpc/methods/worktree-catalog-methods.ts index 2010a219b7c..8b230c169d8 100644 --- a/src/main/runtime/rpc/methods/worktree-catalog-methods.ts +++ b/src/main/runtime/rpc/methods/worktree-catalog-methods.ts @@ -1,4 +1,4 @@ -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { resolveWorktreeCatalogSnapshot } from '../worktree-catalog-snapshot' import { supportsWorktreeVisibilitySourceDefaults } from '../worktree-visibility-client-capability' import { @@ -7,7 +7,7 @@ import { WorktreePsParams } from './worktree-schemas' -export const WORKTREE_CATALOG_METHODS: RpcMethod[] = [ +export const WORKTREE_CATALOG_METHODS = [ defineMethod({ name: 'worktree.ps', params: WorktreePsParams, diff --git a/src/main/runtime/rpc/methods/worktree-create-schemas.ts b/src/main/runtime/rpc/methods/worktree-create-schemas.ts index 61f6eb65e35..659f0c87fcf 100644 --- a/src/main/runtime/rpc/methods/worktree-create-schemas.ts +++ b/src/main/runtime/rpc/methods/worktree-create-schemas.ts @@ -1,154 +1,4 @@ -import { z } from 'zod' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import { workspaceSourceSchema } from '../../../../shared/telemetry-events' -import { sleepingAgentLaunchConfigSchema } from '../../../../shared/workspace-session-sleeping-agents' -import { RUNTIME_NAVIGATION_TARGETS } from '../../../../shared/runtime-navigation' -import { TaskSourceContextSchema } from '../../../../shared/task-source-context-schema' -import { WorkspaceLinkedItemSchema } from '../../../../shared/workspace-linked-item-schema' -import { - OptionalBoolean, - OptionalFiniteNumber, - OptionalString, - TriStateLinkedIssue -} from '../schemas' -import { - assertLinkedWorkItemSourceContextMatch, - AutomationWorkspaceProvenanceRequest, - CliWorkspaceProvenanceRequest, - OptionalTuiAgent -} from './worktree-schemas' - -export const WorktreeCreate = z - .object({ - repo: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing repo selector')), - name: OptionalString, - /** Set by clients that fell back to a generated creature name. Absent means user-typed, so the - * host neither skips a retired candidate nor retires the name it lands on. */ - nameWasGenerated: z.boolean().optional(), - baseBranch: OptionalString, - compareBaseRef: OptionalString, - branchNameOverride: OptionalString, - linkedIssue: TriStateLinkedIssue, - linkedPR: TriStateLinkedIssue, - linkedLinearIssue: z.string().optional(), - linkedLinearIssueWorkspaceId: z.union([z.string(), z.null()]).optional(), - linkedLinearIssueOrganizationUrlKey: z.union([z.string(), z.null()]).optional(), - linkedGitLabMR: TriStateLinkedIssue, - linkedGitLabIssue: TriStateLinkedIssue, - linkedBitbucketPR: TriStateLinkedIssue, - linkedAzureDevOpsPR: TriStateLinkedIssue, - linkedGiteaPR: TriStateLinkedIssue, - linkedWorkItem: WorkspaceLinkedItemSchema.nullable().optional(), - linkedTaskSourceContext: TaskSourceContextSchema.nullable().optional(), - comment: OptionalString, - displayName: OptionalString, - displayNameKind: z.enum(['generated', 'user']).optional(), - telemetrySource: z - .unknown() - .transform((value) => { - const parsed = workspaceSourceSchema.safeParse(value) - return parsed.success ? parsed.data : undefined - }) - .optional(), - workspaceStatus: OptionalString, - manualOrder: OptionalFiniteNumber, - sparseCheckout: z - .object({ - directories: z.array(z.string()), - presetId: OptionalString - }) - .optional(), - pushTarget: z - .object({ - remoteName: z.string(), - branchName: z.string(), - remoteUrl: OptionalString - }) - .optional(), - runHooks: OptionalBoolean, - activate: OptionalBoolean, - // Why: activation on create is view intent, so it is addressed like worktree.activate. - // Contract: a paired desktop/web caller resolves to 'caller' and therefore receives NO - // activateWorktree event — it must reveal from this call's result, which carries setup, - // startup and defaultTabs. Pass an explicit target to opt into an all-surface reveal. - navigation: z.enum(RUNTIME_NAVIGATION_TARGETS).optional(), - parentWorkspace: OptionalString, - // Why: an app-selected parent is a manual action, not the CLI's `--parent-workspace` flag. - // Absent keeps the CLI provenance older clients rely on. - parentWorkspaceOrigin: z.literal('manual').optional(), - envParentWorkspace: OptionalString, - parentWorktree: OptionalString, - cwdParentWorktree: OptionalString, - noParent: OptionalBoolean, - callerTerminalHandle: OptionalString, - orchestrationContext: z - .object({ - parentWorktreeId: OptionalString, - orchestrationRunId: OptionalString, - taskId: OptionalString, - coordinatorHandle: OptionalString - }) - .optional(), - setupDecision: z - .unknown() - .transform((v) => - typeof v === 'string' && (v === 'run' || v === 'skip' || v === 'inherit') ? v : undefined - ) - .pipe(z.union([z.enum(['run', 'skip', 'inherit']), z.undefined()])) - .optional(), - // Why: some clients (e.g. desktop) pass a pre-built launch command so the - // first terminal pane launches the selected agent instead of an idle shell. - // Clients that can't quote for the host shell send `startupAgent` instead. - startupCommand: OptionalString, - startupEnv: z.record(z.string(), z.string()).optional(), - startupLaunchConfig: sleepingAgentLaunchConfigSchema, - startupCommandDelivery: z.enum(['fast', 'shell-ready']).optional(), - // Why: CLI clients should not hardcode agent launch quoting because SSH - // workspaces execute in a different shell than the client process. - startupAgent: OptionalTuiAgent, - startupPrompt: OptionalString, - // Why: task-driven mobile creates need desktop parity: the host chooses - // the same default/detected agent and drafts the linked issue/PR URL into it. - startupDraft: OptionalString, - createdWithAgent: z - .unknown() - .transform((value) => (isTuiAgent(value) ? value : undefined)) - .optional(), - // Why: mobile retries a create interrupted by a connection migration with the - // same key so the host dedupes instead of spawning a duplicate worktree. - clientMutationId: z.string().min(1).max(128).optional(), - automationProvenanceRequest: AutomationWorkspaceProvenanceRequest.optional(), - cliProvenanceRequest: CliWorkspaceProvenanceRequest.optional() - }) - .superRefine((params, ctx) => { - assertLinkedWorkItemSourceContextMatch(params, ctx) - if ((params.parentWorkspace || params.parentWorktree) && params.noParent === true) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Choose either one parent selector or --no-parent.' - }) - } - if (params.parentWorkspace && params.parentWorktree) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Choose either one parent selector or --no-parent.' - }) - } - if (params.startupPrompt !== undefined && params.startupAgent === undefined) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'startupPrompt requires startupAgent' - }) - } - }) - -export const WorktreePrefetchCreateBase = z.object({ - repo: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing repo selector')), - baseBranch: OptionalString -}) +export { + WorktreeCreate, + WorktreePrefetchCreateBase +} from '../../../../shared/rpc-contract/worktree-create-params' diff --git a/src/main/runtime/rpc/methods/worktree-schemas.ts b/src/main/runtime/rpc/methods/worktree-schemas.ts index b7c3b9493a9..41d2c38fec7 100644 --- a/src/main/runtime/rpc/methods/worktree-schemas.ts +++ b/src/main/runtime/rpc/methods/worktree-schemas.ts @@ -1,215 +1,18 @@ -import { z } from 'zod' -import { isTuiAgent } from '../../../../shared/tui-agent-config' -import type { TuiAgent } from '../../../../shared/tui-agent' -import { RUNTIME_NAVIGATION_TARGETS } from '../../../../shared/runtime-navigation' -import { - OptionalBoolean, - OptionalFiniteNumber, - OptionalPlainString, - OptionalString, - TriStateLinkedIssue -} from '../schemas' -import { TaskSourceContextSchema } from '../../../../shared/task-source-context-schema' -import { WorkspaceLinkedItemSchema } from '../../../../shared/workspace-linked-item-schema' -import { isWorkspaceLinkedItemSourceContextMatch } from '../../../../shared/workspace-linked-item-source-context' -import { normalizeExecutionHostId } from '../../../../shared/execution-host' - -const OptionalExecutionHostId = z - .string() - .transform((value, ctx) => { - const hostId = normalizeExecutionHostId(value) - if (!hostId) { - ctx.addIssue({ code: 'custom', message: 'Invalid host id' }) - return z.NEVER - } - return hostId - }) - .optional() - -export const OptionalTuiAgent = z - .unknown() - .superRefine((value, ctx) => { - if (value !== undefined && !isTuiAgent(value)) { - ctx.addIssue({ code: z.ZodIssueCode.custom, message: 'Unknown TUI agent' }) - } - }) - .transform((value): TuiAgent | undefined => (isTuiAgent(value) ? value : undefined)) - .optional() - -export const AutomationWorkspaceProvenanceRequest = z.object({ - automationId: z.string(), - automationRunId: z.string(), - dispatchToken: z.string(), - createRequestId: z.string() -}) - -// Why no dispatch token (unlike automation provenance): this is a descriptive -// origin marker for sidebar filtering, not an authority grant. The host stamps -// createdAt itself so a client clock can't skew sort order. -export const CliWorkspaceProvenanceRequest = z.object({ - callerTerminalHandle: OptionalString -}) - -export const WorktreeListParams = z.object({ - repo: OptionalString, - limit: OptionalFiniteNumber -}) - -export const WorktreeDetectedListParams = z.object({ - repo: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing repo selector')) -}) - -export const WorktreeTeardownMissingTerminalsParams = WorktreeDetectedListParams.extend({ - worktreeIds: z.array(z.string().min(1)).max(10_000), - connectionId: z.string().nullable().optional() -}) - -export const WorktreePsParams = z.object({ - limit: OptionalFiniteNumber, - afterSnapshotId: z.string().min(1).max(128).nullable().optional(), - supportsWorktreeVisibilitySourceDefaults: z.literal(true).optional() -}) - -export const WorktreeSortOrder = z.object({ - orderedIds: z.array(z.string()) -}) - -export const WorktreeSelector = z.object({ - worktree: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing worktree selector')) -}) - -export const WorktreeActivate = WorktreeSelector.extend({ - notifyClients: OptionalBoolean, - navigation: z.enum(RUNTIME_NAVIGATION_TARGETS).optional() -}) - -/** Shared by WorktreeCreate and WorktreeSet so the two error messages cannot drift. */ -export function assertLinkedWorkItemSourceContextMatch( - params: { - linkedWorkItem?: z.infer | null - linkedTaskSourceContext?: z.infer | null - }, - ctx: z.RefinementCtx -): void { - if ( - params.linkedWorkItem && - params.linkedTaskSourceContext && - !isWorkspaceLinkedItemSourceContextMatch(params.linkedWorkItem, params.linkedTaskSourceContext) - ) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Linked work item and source context identities must match' - }) - } -} - -export const WorktreeSet = WorktreeSelector.extend({ - // Why: '' is the blanking contract — "fall back to the branch/folder name". - // OptionalString coerced it to undefined, so on remote/SSH hosts clearing the - // name was dropped here and the old name came back on the next refresh. - displayName: OptionalPlainString, - // Why: empty comments are meaningful metadata updates, so use the plain - // string parser instead of OptionalString's empty-as-undefined behavior. - comment: OptionalPlainString, - linkedIssue: TriStateLinkedIssue, - linkedPR: TriStateLinkedIssue, - suppressedGitHubPR: z.number().int().positive().nullable().optional(), - linkedLinearIssue: z.union([z.string(), z.null()]).optional(), - linkedLinearIssueWorkspaceId: z.union([z.string(), z.null()]).optional(), - linkedLinearIssueOrganizationUrlKey: z.union([z.string(), z.null()]).optional(), - linkedGitLabMR: TriStateLinkedIssue, - linkedGitLabIssue: TriStateLinkedIssue, - linkedBitbucketPR: TriStateLinkedIssue, - linkedAzureDevOpsPR: TriStateLinkedIssue, - linkedGiteaPR: TriStateLinkedIssue, - linkedWorkItem: WorkspaceLinkedItemSchema.nullable().optional(), - linkedTaskSourceContext: TaskSourceContextSchema.nullable().optional(), - isArchived: OptionalBoolean, - isUnread: OptionalBoolean, - isPinned: OptionalBoolean, - sortOrder: OptionalFiniteNumber, - manualOrder: OptionalFiniteNumber, - lastActivityAt: OptionalFiniteNumber, - createdAt: OptionalFiniteNumber, - sparseDirectories: z.array(z.string()).optional(), - sparseBaseRef: OptionalString, - sparsePresetId: OptionalString, - baseRef: OptionalString, - workspaceStatus: OptionalString, - pushTarget: z - .object({ - remoteName: z.string(), - branchName: z.string(), - remoteUrl: OptionalString - }) - .nullable() - .optional(), - diffComments: z.array(z.unknown()).optional(), - mobileDiffReview: z.unknown().optional(), - parentWorktree: OptionalString, - noParent: OptionalBoolean -}).superRefine((params, ctx) => { - assertLinkedWorkItemSourceContextMatch(params, ctx) - if (params.parentWorktree && params.noParent === true) { - ctx.addIssue({ - code: z.ZodIssueCode.custom, - message: 'Choose either --parent-worktree or --no-parent, not both.' - }) - } -}) - -export const WorktreeRemove = WorktreeSelector.extend({ - hostId: OptionalExecutionHostId, - force: OptionalBoolean, - // Why (#11960): the CLI's --force is an unambiguous force affordance, but the - // desktop sets `force` for an ordinary confirmed delete too, so the PTY-stop - // waiver travels on its own field. - allowUnverifiedPtyStop: OptionalBoolean, - runHooks: OptionalBoolean -}) - -export const WorktreeForceDeleteBranch = WorktreeSelector.extend({ - hostId: OptionalExecutionHostId, - branchName: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing branch name')), - expectedHead: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing expected branch head')) -}) - -export const WorktreeResolvePrBase = z.object({ - repo: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing repo selector')), - prNumber: z - .unknown() - .transform((v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0)) - .pipe(z.number().int().positive('Missing PR number')), - headRefName: OptionalString, - baseRefName: OptionalString, - isCrossRepository: OptionalBoolean -}) - -export const WorktreeResolveMrBase = z.object({ - repo: z - .unknown() - .transform((v) => (typeof v === 'string' ? v : '')) - .pipe(z.string().min(1, 'Missing repo selector')), - mrIid: z - .unknown() - .transform((v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0)) - .pipe(z.number().int().positive('Missing MR number')), - sourceBranch: OptionalString, - targetBranch: OptionalString, - isCrossRepository: OptionalBoolean -}) +export { + AutomationWorkspaceProvenanceRequest, + CliWorkspaceProvenanceRequest, + OptionalTuiAgent, + WorktreeActivate, + WorktreeDetectedListParams, + WorktreeForceDeleteBranch, + WorktreeListParams, + WorktreePsParams, + WorktreeRemove, + WorktreeResolveMrBase, + WorktreeResolvePrBase, + WorktreeSelector, + WorktreeSet, + WorktreeSortOrder, + WorktreeTeardownMissingTerminalsParams, + assertLinkedWorkItemSourceContextMatch +} from '../../../../shared/rpc-contract/worktree-params' diff --git a/src/main/runtime/rpc/methods/worktree-visibility-defaults-schema.ts b/src/main/runtime/rpc/methods/worktree-visibility-defaults-schema.ts index 8bb7e1d081e..19a6dba90be 100644 --- a/src/main/runtime/rpc/methods/worktree-visibility-defaults-schema.ts +++ b/src/main/runtime/rpc/methods/worktree-visibility-defaults-schema.ts @@ -1,19 +1 @@ -import { z } from 'zod' -import { - normalizeCustomWorktreeVisibilitySources, - normalizeWorktreeVisibilitySourcePreferences -} from '../../../../shared/worktree/visibility-sources' - -export const WorktreeVisibilityDefaultsUpdate = z - .object({ - external: z.enum(['hide', 'show']).optional(), - customSources: z - .unknown() - .transform((value) => normalizeCustomWorktreeVisibilitySources(value)) - .optional(), - sourcePreferences: z - .unknown() - .transform((value) => normalizeWorktreeVisibilitySourcePreferences(value)) - .optional() - }) - .strict() +export { WorktreeVisibilityDefaultsUpdate } from '../../../../shared/rpc-contract/worktree-visibility-defaults-params' diff --git a/src/main/runtime/rpc/methods/worktree.ts b/src/main/runtime/rpc/methods/worktree.ts index b3d816496c3..be8a0983036 100644 --- a/src/main/runtime/rpc/methods/worktree.ts +++ b/src/main/runtime/rpc/methods/worktree.ts @@ -5,7 +5,7 @@ import { } from '../../../automations/workspace-provenance' import { buildCliWorkspaceProvenance } from '../../../../shared/cli-workspace-provenance' import { displayNameUpdatePinsLabel } from '../../../../shared/worktree/display-name-provenance' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod } from '../core' import { buildManagedWorktreeCreateArgs } from './worktree-create-args' import { resolvePairedCallerHostId } from './paired-caller-host-id' import { resolveRuntimeNavigationTarget } from '../../../../shared/runtime-navigation' @@ -24,7 +24,7 @@ import { } from './worktree-schemas' import { WORKTREE_CATALOG_METHODS } from './worktree-catalog-methods' -export const WORKTREE_METHODS: RpcMethod[] = [ +export const WORKTREE_METHODS = [ ...WORKTREE_CATALOG_METHODS, defineMethod({ name: 'worktree.teardownMissingTerminals', diff --git a/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts b/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts index 71daf09b0e3..c57f7876402 100644 --- a/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts +++ b/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts @@ -315,13 +315,26 @@ describe('legacy coordinator takeover races', () => { it('partitions a coordinator group send by legacy recipient contract', async () => { const harness = createHarness() + // A second worker on the CURRENT contract in the same adopted Run. Group addresses reach a + // Run's Dispatches, so the partition needs two Dispatches, not a Dispatch and a loose pane. + const currentTask = harness.db.createTask({ + runId: harness.adoptedRunId, + spec: 'current-contract assignment', + createdByTerminalHandle: COORDINATOR_HANDLE + }) + const currentDispatch = createRootDispatch( + harness.db, + currentTask.id, + 'term_current_worker', + 'tab_current_worker:22222222-2222-4222-8222-222222222222' + ) vi.mocked(harness.runtime.getTerminalPaneKey).mockImplementation((handle) => handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : handle === WORKER_HANDLE ? WORKER_PANE : handle === 'term_current_worker' - ? 'tab_current_worker:leaf_current_worker' + ? 'tab_current_worker:22222222-2222-4222-8222-222222222222' : null ) vi.spyOn(harness.runtime, 'listTerminals').mockResolvedValue({ @@ -355,7 +368,7 @@ describe('legacy coordinator takeover races', () => { }), expect.objectContaining({ run_id: harness.adoptedRunId, - to_handle: 'term_current_worker', + to_handle: `dispatch:${currentDispatch.id}`, delivery_contract: 'current_delivery' }) ]) @@ -409,7 +422,7 @@ describe('legacy coordinator takeover races', () => { const pending = harness.dispatcher.dispatch( request( 'orchestration.send', - { from: COORDINATOR_HANDLE, to: '@all', subject: 'must remain unsent' }, + { from: COORDINATOR_HANDLE, to: '@codex', subject: 'must remain unsent' }, 'send-group-takeover' ) ) diff --git a/src/main/runtime/rpc/schemas.ts b/src/main/runtime/rpc/schemas.ts index fb7b09d8ceb..2c469341be5 100644 --- a/src/main/runtime/rpc/schemas.ts +++ b/src/main/runtime/rpc/schemas.ts @@ -3,87 +3,15 @@ // recur across domains (optional worktree selector, bounded limit, browser // target envelope, etc.). Methods compose these to declare their real // contract without repeating the same `typeof` gymnastics 90 times. -import { z } from 'zod' - -// Why: the original handlers treated non-numeric/NaN limit values as "no -// limit" rather than as errors. Preserve that forgiving behavior so CLI -// callers passing stringified numbers or Infinity still reach the runtime. -// The outer optional() is required for omitted keys in Zod v4; an optional -// schema hidden behind pipe() still makes z.object require the property. -export const OptionalFiniteNumber = z - .unknown() - .transform((value) => (typeof value === 'number' && Number.isFinite(value) ? value : undefined)) - .pipe(z.union([z.number(), z.undefined()])) - .optional() - -export const OptionalPositiveInt = z - .unknown() - .transform((value) => - typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined - ) - .pipe(z.union([z.number(), z.undefined()])) - .optional() - -export const OptionalString = z - .unknown() - .transform((value) => (typeof value === 'string' && value.length > 0 ? value : undefined)) - .pipe(z.union([z.string(), z.undefined()])) - .optional() - -export const OptionalPlainString = z - .unknown() - .transform((value) => (typeof value === 'string' ? value : undefined)) - .pipe(z.union([z.string(), z.undefined()])) - .optional() - -export const OptionalBoolean = z - .unknown() - .transform((value) => (typeof value === 'boolean' ? value : undefined)) - .pipe(z.union([z.boolean(), z.undefined()])) - .optional() - -// Why: runtime handlers accept `linkedIssue: number | null | undefined` with -// distinct meanings — undefined means "no update", null means "clear", number -// means "set". The ambient JSON decode produces all three shapes as-is. -export const TriStateLinkedIssue = z - .unknown() - .transform((value) => { - if (value === null) { - return null - } - if (typeof value === 'number' && Number.isFinite(value)) { - return value - } - return undefined - }) - .pipe(z.union([z.number(), z.null(), z.undefined()])) - .optional() - -// Why: the legacy extractBrowserTarget treated worktree as a plain-string -// passthrough (empty string preserved) but `page` as non-empty-string. The -// browser bridge uses worktree-as-empty-string to mean "any worktree", so -// keep that asymmetry intact to avoid widening scope unexpectedly. -export const BrowserTarget = z.object({ - worktree: OptionalPlainString, - page: OptionalString -}) - -export function requiredString(message: string) { - return z - .unknown() - .transform((value) => (typeof value === 'string' ? value : '')) - .pipe(z.string().min(1, message)) -} - -export function requiredStringAllowingEmpty(message: string) { - return z.unknown().refine((value): value is string => typeof value === 'string', { message }) -} - -export function requiredNumber(message: string) { - return z - .unknown() - .transform((value) => - typeof value === 'number' && Number.isFinite(value) ? value : Number.NaN - ) - .pipe(z.number().refine((v) => Number.isFinite(v), { message })) -} +export { + BrowserTarget, + OptionalBoolean, + OptionalFiniteNumber, + OptionalPlainString, + OptionalPositiveInt, + OptionalString, + TriStateLinkedIssue, + requiredNumber, + requiredString, + requiredStringAllowingEmpty +} from '../../../shared/rpc-contract/rpc-param-primitives' diff --git a/src/main/runtime/runtime-desktop-surface.ts b/src/main/runtime/runtime-desktop-surface.ts index ac1086e4f35..a36cc0b71f4 100644 --- a/src/main/runtime/runtime-desktop-surface.ts +++ b/src/main/runtime/runtime-desktop-surface.ts @@ -17,6 +17,7 @@ import type { BrowserWindow, IpcMainEvent } from 'electron' export type RuntimeDesktopSurface = { /** Show a native notification. Returns false when the host cannot, so callers can say so. */ + isAwayForMobileNotifications?(): boolean | undefined showNotification(input: { title: string; body: string }): boolean /** The renderer window with this id, or null when there is no desktop. */ findWindowById(id: number): BrowserWindow | null diff --git a/src/main/runtime/runtime-folder-worktree-create.ts b/src/main/runtime/runtime-folder-worktree-create.ts index efea3c22c72..ef7798c91f9 100644 --- a/src/main/runtime/runtime-folder-worktree-create.ts +++ b/src/main/runtime/runtime-folder-worktree-create.ts @@ -172,7 +172,7 @@ export async function createRuntimeFolderWorktree(args: { undefined, args.startup && !didSpawnStartup ? args.startup : undefined ) - } else if (deps.ptySpawnAvailable && !didSpawnStartup) { + } else if (deps.ptySpawnAvailable && !didSpawnStartup && !args.createdWithAgent) { try { await deps.createTerminal(`id:${worktree.id}`, { surfaceOwner: false }) } catch (error) { diff --git a/src/main/runtime/runtime-local-worktree-terminal-startup.test.ts b/src/main/runtime/runtime-local-worktree-terminal-startup.test.ts new file mode 100644 index 00000000000..1153b173553 --- /dev/null +++ b/src/main/runtime/runtime-local-worktree-terminal-startup.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../shared/repo-types' +import type { Worktree } from '../../shared/worktree/types' +import { startRuntimeLocalWorktreeTerminals } from './runtime-local-worktree-terminal-startup' + +const repo: Repo = { + id: 'repo-1', + path: '/repo', + displayName: 'repo', + badgeColor: 'blue', + addedAt: 1 +} + +const worktree: Worktree = { + id: 'worktree-1', + repoId: repo.id, + path: '/worktree', + head: 'abc', + branch: 'feature', + isBare: false, + isMainWorktree: false, + displayName: 'feature', + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 1 +} + +type StartupArgs = Parameters[0] + +function createPorts() { + const createTerminal = vi.fn().mockResolvedValue({ + handle: 'term-1', + worktreeId: worktree.id, + title: null + }) + const ports: StartupArgs['ports'] = { + canSpawn: true, + markTrusted: vi.fn(), + createTerminal, + pasteDraft: vi.fn(), + sendFollowup: vi.fn(), + provision: vi.fn().mockResolvedValue({ setupSpawned: false, setupTerminalHandle: null }), + activate: vi.fn() + } + return { createTerminal, ports } +} + +describe('startRuntimeLocalWorktreeTerminals default shell seeding', () => { + it.each([ + ['Blank Terminal', undefined, 1], + ['an agent', 'codex' as const, 0] + ])('seeds a background shell for %s selection only', async (_label, agent, expectedCalls) => { + const { createTerminal, ports } = createPorts() + + await startRuntimeLocalWorktreeTerminals({ + request: { repoSelector: `id:${repo.id}`, name: worktree.displayName }, + repo, + worktree, + ...(agent ? { createdWithAgent: agent } : {}), + ports + }) + + expect(createTerminal).toHaveBeenCalledTimes(expectedCalls) + if (expectedCalls > 0) { + expect(createTerminal).toHaveBeenCalledWith(`id:${worktree.id}`, { surfaceOwner: false }) + } + }) +}) diff --git a/src/main/runtime/runtime-local-worktree-terminal-startup.ts b/src/main/runtime/runtime-local-worktree-terminal-startup.ts index 35985b53497..7babcd7da7d 100644 --- a/src/main/runtime/runtime-local-worktree-terminal-startup.ts +++ b/src/main/runtime/runtime-local-worktree-terminal-startup.ts @@ -163,7 +163,7 @@ export async function startRuntimeLocalWorktreeTerminals(args: { didSpawnSetup = true } } - } else if (ports.canSpawn) { + } else if (ports.canSpawn && !args.createdWithAgent) { try { await ports.createTerminal(`id:${worktree.id}`, { surfaceOwner: false }) } catch (error) { diff --git a/src/main/runtime/runtime-mobile-notification-controller.ts b/src/main/runtime/runtime-mobile-notification-controller.ts index a9c1d437f95..73578412618 100644 --- a/src/main/runtime/runtime-mobile-notification-controller.ts +++ b/src/main/runtime/runtime-mobile-notification-controller.ts @@ -1,9 +1,23 @@ +import { reserveNotificationCooldown } from '../../shared/notification-burst-cooldown' +import type { AgentStatusState } from '../../shared/agent-status-types' +import type { + MobilePushRegisterInput, + MobilePushRegisterResult +} from '../../shared/mobile-push-contract' import { MobileNotificationReplayBuffer } from './mobile-notification-replay' import { notifyRuntimeListeners } from './runtime-async-boundaries' import { getRuntimeDesktopSurface } from './runtime-desktop-surface' +import { + MobileNotificationDismissalStore, + type DeliveredNotificationIdentity +} from './mobile-notification-dismissal-store' export type MobileNotificationDispatchEvent = { type: 'notification' + legacySocketAllowed?: boolean + desktopAllowed?: boolean + desktopAway?: boolean + emittedAt?: number source: 'agent-task-complete' | 'terminal-bell' | 'test' | 'plugin' title: string body: string @@ -11,6 +25,9 @@ export type MobileNotificationDispatchEvent = { notificationId?: string notificationSeq?: number notificationEpoch?: string + // Why: background push must tell "needs input" from "finished" without re-deriving + // it from the title. Optional and additive — old clients ignore it. + agentState?: AgentStatusState } export type MobileNotificationDismissEvent = { @@ -24,9 +41,45 @@ export type MobileNotificationEvent = | MobileNotificationDispatchEvent | MobileNotificationDismissEvent +/** The desktop push service, once it exists; absent on hosts that never started one. */ +export type MobilePushRegistrar = { + register(input: MobilePushRegisterInput): Promise + unregister(deviceId: string): Promise<{ unregistered: boolean }> +} + export class RuntimeMobileNotificationController { private readonly listeners = new Set<(event: MobileNotificationEvent) => void>() + private readonly legacyCooldown = new Map() private readonly replay = new MobileNotificationReplayBuffer() + private pushRegistrar: MobilePushRegistrar | null = null + private dismissalStore: MobileNotificationDismissalStore | null = null + + configureDismissalStore(userDataPath: string): void { + this.dismissalStore = new MobileNotificationDismissalStore(userDataPath) + } + + reconcileDismissedPushes( + delivered: readonly DeliveredNotificationIdentity[] + ): DeliveredNotificationIdentity[] { + return this.dismissalStore?.reconcile(delivered) ?? [] + } + + setPushRegistrar(registrar: MobilePushRegistrar | null): void { + this.pushRegistrar = registrar + } + + async registerPushDevice(input: MobilePushRegisterInput): Promise { + return ( + (await this.pushRegistrar?.register(input)) ?? { + registered: false, + reason: 'gateway_unreachable' + } + ) + } + + async unregisterPushDevice(deviceId: string): Promise<{ unregistered: boolean }> { + return (await this.pushRegistrar?.unregister(deviceId)) ?? { unregistered: false } + } onDispatched(listener: (event: MobileNotificationEvent) => void): () => void { this.listeners.add(listener) @@ -38,7 +91,32 @@ export class RuntimeMobileNotificationController { } dispatch(event: MobileNotificationEvent): void { + if (event.type === 'notification') { + // Decide once before recording so reconnect and buffer eviction cannot reset cooldown. + const legacySocketAllowed = + event.desktopAllowed !== false && + (event.emittedAt === undefined || + reserveNotificationCooldown( + this.legacyCooldown, + event.worktreeId ?? 'global', + event.emittedAt + )) + event = { + ...event, + legacySocketAllowed, + desktopAway: getRuntimeDesktopSurface().isAwayForMobileNotifications?.() + } + } const seq = this.replay.record(event) + try { + this.dismissalStore?.record({ + ...event, + notificationSeq: seq, + notificationEpoch: this.replay.epoch + }) + } catch { + console.warn('[notifications] Could not persist dismissal recovery state') + } notifyRuntimeListeners( this.listeners, (listener) => diff --git a/src/main/runtime/runtime-remote-managed-worktree-create.ts b/src/main/runtime/runtime-remote-managed-worktree-create.ts index 01a49142dc4..83de82b0594 100644 --- a/src/main/runtime/runtime-remote-managed-worktree-create.ts +++ b/src/main/runtime/runtime-remote-managed-worktree-create.ts @@ -222,7 +222,7 @@ export async function createRuntimeRemoteManagedWorktree( didSpawnSetup = true } } - } else if (!shouldActivate && deps.canSpawn()) { + } else if (!shouldActivate && deps.canSpawn() && !args.createdWithAgent) { try { await deps.createTerminal(`path:${result.worktree.path}`, { surfaceOwner: false }) } catch (err) { diff --git a/src/main/runtime/runtime-rpc-mobile-method-allowlist-fixtures.ts b/src/main/runtime/runtime-rpc-mobile-method-allowlist-fixtures.ts index 7a97745764d..00bf253fe89 100644 --- a/src/main/runtime/runtime-rpc-mobile-method-allowlist-fixtures.ts +++ b/src/main/runtime/runtime-rpc-mobile-method-allowlist-fixtures.ts @@ -117,6 +117,7 @@ export function createMobileRpcSurfaceRuntime() { .fn() .mockResolvedValue({ ok: true, id: 'comment-1' }) const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'test-runtime', getStatus, pushRuntimeGit, diff --git a/src/main/runtime/runtime-rpc-mobile-terminal-streaming.test.ts b/src/main/runtime/runtime-rpc-mobile-terminal-streaming.test.ts index e558d19b54b..c094f54cee3 100644 --- a/src/main/runtime/runtime-rpc-mobile-terminal-streaming.test.ts +++ b/src/main/runtime/runtime-rpc-mobile-terminal-streaming.test.ts @@ -404,6 +404,7 @@ describe('OrcaRuntimeRpcServer', () => { // activation is a local-host concern, so the proxy legitimately lacks // activateRecentPtyPathCandidateTracking and onReady must not throw. const runtimeProxy = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'proxy-runtime-test', getStartedAt: () => 1, getStatus: () => ({ graphStatus: 'unavailable' }), diff --git a/src/main/runtime/runtime-rpc-request-authorization.test.ts b/src/main/runtime/runtime-rpc-request-authorization.test.ts index 5ff19f94563..d5a083a56bf 100644 --- a/src/main/runtime/runtime-rpc-request-authorization.test.ts +++ b/src/main/runtime/runtime-rpc-request-authorization.test.ts @@ -30,6 +30,7 @@ describe('OrcaRuntimeRpcServer', () => { it('rejects WebSocket requests whose request token differs from the authenticated channel token', async () => { const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-rpc-')) const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'test-runtime', getStatus: vi.fn().mockResolvedValue({ graphStatus: 'ok' }) } as unknown as OrcaRuntimeService @@ -184,6 +185,7 @@ describe('OrcaRuntimeRpcServer', () => { const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-rpc-')) const createMobileSessionTerminal = vi.fn() const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'test-runtime', createMobileSessionTerminal } as unknown as OrcaRuntimeService @@ -225,6 +227,7 @@ describe('OrcaRuntimeRpcServer', () => { const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-rpc-')) const pushRuntimeGit = vi.fn().mockResolvedValue({ ok: true }) const runtime = { + configureNotificationDismissalStore: () => {}, getRuntimeId: () => 'test-runtime', pushRuntimeGit } as unknown as OrcaRuntimeService diff --git a/src/main/runtime/runtime-rpc-websocket-long-poll-caps.test.ts b/src/main/runtime/runtime-rpc-websocket-long-poll-caps.test.ts index 7c2b5bd11d7..ea8837c01c9 100644 --- a/src/main/runtime/runtime-rpc-websocket-long-poll-caps.test.ts +++ b/src/main/runtime/runtime-rpc-websocket-long-poll-caps.test.ts @@ -206,7 +206,10 @@ describe('OrcaRuntimeRpcServer', () => { it('shares one socket close listener across concurrent WebSocket dispatches', async () => { const userDataPath = mkdtempSync(join(tmpdir(), 'orca-runtime-rpc-')) - const runtime = { getRuntimeId: () => 'test-runtime' } as unknown as OrcaRuntimeService + const runtime = { + configureNotificationDismissalStore: () => {}, + getRuntimeId: () => 'test-runtime' + } as unknown as OrcaRuntimeService const server = new OrcaRuntimeRpcServer({ runtime, userDataPath, enableWebSocket: false }) server['deviceRegistry'] = new DeviceRegistry(userDataPath) const entry = server['deviceRegistry']!.addDevice('runtime-test', 'runtime') diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts b/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts index 534cf04b883..ca8d326e7e7 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts @@ -172,7 +172,9 @@ export const MOBILE_RPC_METHOD_ALLOWLIST = new Set([ 'markdown.readTab', 'markdown.saveTab', 'notifications.getMissedSince', + 'notifications.registerPush', 'notifications.subscribe', + 'notifications.unregisterPush', 'notifications.unsubscribe', 'pairing.getEndpoints', 'pairing.provisionRelay', diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts b/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts index 3cd3c1a54fb..4923dec1dec 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-pairing-types.ts @@ -1,5 +1,5 @@ import type { OrcaRuntimeService } from '../orca-runtime' -import type { RpcAnyMethod } from '../rpc/core' +import type { RpcAnyMethodDeclaration } from '../rpc/core' import type { DeviceRegistry } from '../device-registry' import type { E2EEKeypair } from '../e2ee-keypair' import type { MobileSocketTransportMetadata } from '../rpc/mobile-socket-wiring' @@ -56,7 +56,7 @@ export type OrcaRuntimeRpcServerOptions = { // Why: test-only override for the ownership reclaim cadence. metadataOwnershipPollMs?: number // Why: tests may inject inert protocol stages before production authorization registers them. - methods?: readonly RpcAnyMethod[] + methods?: readonly RpcAnyMethodDeclaration[] } export type PairingOfferUnavailableReason = diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-pairing.ts b/src/main/runtime/runtime-rpc/runtime-rpc-pairing.ts index 592131779eb..7d8bba9f958 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-pairing.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-pairing.ts @@ -6,6 +6,7 @@ import type { RelayRevokeOutbox, RelayRevokeOutboxItem } from '../relay/relay-revoke-outbox' +import type { PushUnregisterOutbox } from '../push/push-unregister-outbox' import { encodePairingOffer, PAIRING_OFFER_VERSION } from '../../../shared/pairing' import type { RuntimePairingReach } from '../../../shared/runtime-pairing-reach' import { resolveAdvertisedPairingEndpoint } from '../pairing-endpoint' @@ -20,6 +21,8 @@ import { } from './runtime-rpc-pairing-types' export class RuntimeRpcPairing extends RuntimeRpcNetworkExposure { + private onPushUnregisterQueued?: () => void + getDeviceRegistry(): DeviceRegistry | null { return this.deviceRegistry } @@ -44,6 +47,10 @@ export class RuntimeRpcPairing extends RuntimeRpcNetworkExposure { return this.relayRevokeOutbox } + getPushUnregisterOutbox(): PushUnregisterOutbox { + return this.pushUnregisterOutbox + } + setMobileRelayBinding(deviceId: string, binding: RelayDeviceBinding): boolean { const current = this.deviceRegistry?.getDevice(deviceId) if ( @@ -88,6 +95,9 @@ export class RuntimeRpcPairing extends RuntimeRpcNetworkExposure { return false } } + // Why: unpairing must delete the phone's push token at the gateway too, and the + // registration id is only readable while the device row still exists. + this.queuePushUnregister(deviceId, device.pushRegistration?.registrationId) if (!this.deviceRegistry?.removeDevice(deviceId)) { return false } @@ -182,6 +192,23 @@ export class RuntimeRpcPairing extends RuntimeRpcNetworkExposure { } } + /** Best-effort: a failed enqueue must never block the revoke the user asked for. */ + protected queuePushUnregister(deviceId: string, registrationId: string | undefined): void { + if (!registrationId) { + return + } + try { + this.pushUnregisterOutbox.enqueue({ registrationId, deviceId }) + this.onPushUnregisterQueued?.() + } catch (error) { + console.error('[runtime] Failed to persist a push token cleanup:', error) + } + } + + setOnPushUnregisterQueued(callback: (() => void) | null): void { + this.onPushUnregisterQueued = callback ?? undefined + } + protected queueOrRetainRelayDeviceRevoke(deviceId: string, binding: RelayDeviceBinding): void { if (this.queueRelayDeviceRevoke(binding)) { return diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-state.ts b/src/main/runtime/runtime-rpc/runtime-rpc-state.ts index ca9ab173feb..e7f56ceed13 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-state.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-state.ts @@ -10,6 +10,7 @@ import type { E2EEKeypair } from '../e2ee-keypair' import type { UnpairedDeviceAuthThrottle } from '../rpc/unpaired-device-auth-throttle' import type { MobileSocketWiring } from '../rpc/mobile-socket-wiring' import { RelayRevokeOutbox } from '../relay/relay-revoke-outbox' +import { PushUnregisterOutbox } from '../push/push-unregister-outbox' import { RuntimeBinaryMessageRouter } from '../runtime-binary-message-router' import type { RuntimeMetadataOwnershipWatch } from '../runtime-metadata-ownership-watch' import { RUNTIME_METADATA_OWNERSHIP_POLL_MS } from '../runtime-metadata-ownership-watch' @@ -56,6 +57,7 @@ export class RuntimeRpcState { protected readonly browserHostLongPollCapPerDevice: number protected readonly specializedLongPollCap: number protected readonly relayRevokeOutbox: RelayRevokeOutbox + protected readonly pushUnregisterOutbox: PushUnregisterOutbox protected deviceRegistry: DeviceRegistry | null = null protected e2eeKeypair: E2EEKeypair | null = null protected pairingInitializationFailure: PairingOfferUnavailable | null = null @@ -129,5 +131,7 @@ export class RuntimeRpcState { this.browserHostLongPollCapPerDevice = Math.max(1, Math.floor(this.browserHostLongPollCap / 2)) this.specializedLongPollCap = Math.max(1, Math.floor(longPollCap * SPECIALIZED_LONG_POLL_SHARE)) this.relayRevokeOutbox = new RelayRevokeOutbox(userDataPath) + this.pushUnregisterOutbox = new PushUnregisterOutbox(userDataPath) + this.runtime.configureNotificationDismissalStore(userDataPath) } } diff --git a/src/main/runtime/runtime-service-command-surface.ts b/src/main/runtime/runtime-service-command-surface.ts index 19545cc76e6..b7811dbc073 100644 --- a/src/main/runtime/runtime-service-command-surface.ts +++ b/src/main/runtime/runtime-service-command-surface.ts @@ -27,9 +27,14 @@ export type RuntimeServiceCommandSurface = { getMobileNotificationListenerCount: RuntimeMobileNotificationController['getListenerCount'] dispatchMobileNotification: RuntimeMobileNotificationController['dispatch'] getMissedNotificationsSince: RuntimeMobileNotificationController['getMissedSince'] + configureNotificationDismissalStore: RuntimeMobileNotificationController['configureDismissalStore'] + reconcileDismissedPushes: RuntimeMobileNotificationController['reconcileDismissedPushes'] getMobileNotificationEpoch: RuntimeMobileNotificationController['getEpoch'] dismissMobileNotification: RuntimeMobileNotificationController['dismiss'] dispatchPluginNotification: RuntimeMobileNotificationController['dispatchPlugin'] + setMobilePushRegistrar: RuntimeMobileNotificationController['setPushRegistrar'] + registerMobilePushDevice: RuntimeMobileNotificationController['registerPushDevice'] + unregisterMobilePushDevice: RuntimeMobileNotificationController['unregisterPushDevice'] setAccountServices: RuntimeAccountController['setServices'] setCommitMessageAgentEnvironmentResolvers: RuntimeAccountController['setCommitMessageAgentEnvironment'] getCommitMessageAgentEnvironmentResolvers: RuntimeAccountController['getCommitMessageAgentEnvironment'] @@ -107,9 +112,14 @@ export function installRuntimeServiceCommandSurface( getMobileNotificationListenerCount: notifications.getListenerCount.bind(notifications), dispatchMobileNotification: notifications.dispatch.bind(notifications), getMissedNotificationsSince: notifications.getMissedSince.bind(notifications), + configureNotificationDismissalStore: notifications.configureDismissalStore.bind(notifications), + reconcileDismissedPushes: notifications.reconcileDismissedPushes.bind(notifications), getMobileNotificationEpoch: notifications.getEpoch.bind(notifications), dismissMobileNotification: notifications.dismiss.bind(notifications), dispatchPluginNotification: notifications.dispatchPlugin.bind(notifications), + setMobilePushRegistrar: notifications.setPushRegistrar.bind(notifications), + registerMobilePushDevice: notifications.registerPushDevice.bind(notifications), + unregisterMobilePushDevice: notifications.unregisterPushDevice.bind(notifications), setAccountServices: accounts.setServices.bind(accounts), setCommitMessageAgentEnvironmentResolvers: accounts.setCommitMessageAgentEnvironment.bind(accounts), diff --git a/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts b/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts index 84966846be7..83b3d651642 100644 --- a/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts +++ b/src/main/runtime/runtime-worktree-agent-rows-structured.test.ts @@ -1,5 +1,5 @@ import { collectRuntimeWorktreeAgentSources } from './runtime-worktree-agent-sources' -import { describe, expect, it } from 'vitest' +import { beforeEach, describe, expect, it, vi } from 'vitest' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' import { structuredAgentSessionPaneKey, @@ -7,17 +7,24 @@ import { } from '../../shared/structured-agent-session-projection' import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' import type { RuntimeWorktreePsSummary } from '../../shared/runtime-types' +import { AgentHookServer, _internals } from '../agent-hooks/server' + +vi.mock('../telemetry/client', () => ({ track: vi.fn() })) +vi.mock('../telemetry/cohort-classifier', () => ({ + getCohortAtEmit: vi.fn(() => ({ nth_repo_added: 2 })) +})) /** - * A structured session has no PTY, so it reaches none of the hook or retained snapshots that every - * other row comes from. Before this, `worktree ps` reported a worktree running one as idle while - * the sidebar showed it working — the CLI, which is the agent-facing surface, was the blind one. + * A structured session has no PTY and no hook script, so the host publishes its projection into + * the agent-status store itself. This walks that store into `worktree ps` rows: before it, the CLI + * reported a worktree running one as idle while the sidebar showed it working. */ const WORKTREE_ID = 'repo-1::/workspace/app' +const SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' function summary(over: Partial = {}): AgentSessionStatusSummary { return { - sessionId: 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d', + sessionId: SESSION, workspaceId: WORKTREE_ID, agent: 'claude', status: 'working', @@ -29,6 +36,10 @@ function summary(over: Partial = {}): AgentSessionSta } function attach(summaries: AgentSessionStatusSummary[]): RuntimeWorktreePsSummary { + const store = new AgentHookServer() + for (const entry of summaries) { + store.ingestStructuredStatus(entry) + } const row = { worktreeId: WORKTREE_ID, status: 'inactive', @@ -45,8 +56,7 @@ function attach(summaries: AgentSessionStatusSummary[]): RuntimeWorktreePsSummar mirroredWorktreeIdByTabId: new Map(), connectedPtyEvidence: { tabIds: new Set(), paneKeys: new Set(), ptyIds: new Set() }, retainedSnapshots: [], - hookSnapshots: [], - structuredSummaries: summaries + hookSnapshots: store.getStatusSnapshot() }), orchestrationByPaneKey: null, getSummary: (map, _p, _m, id) => map.get(id) ?? null @@ -54,6 +64,10 @@ function attach(summaries: AgentSessionStatusSummary[]): RuntimeWorktreePsSummar return row } +beforeEach(() => { + _internals.resetCachesForTests() +}) + describe('worktree ps reports structured sessions', () => { it('a busy structured session is not reported idle', () => { const row = attach([summary()]) @@ -61,6 +75,7 @@ describe('worktree ps reports structured sessions', () => { expect(row.agents[0]?.state).toBe('working') expect(row.agents[0]?.agentType).toBe('claude') expect(row.agents[0]?.prompt).toBe('ship the thing') + expect(row.status).toBe('working') }) // The same projection the sidebar applies, so the two surfaces cannot disagree about one session. @@ -77,9 +92,8 @@ describe('worktree ps reports structured sessions', () => { it('reports the DERIVED pane key, never an orchestration credential', () => { const row = attach([summary()]) - const sessionId = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' expect(row.agents[0]?.paneKey).toBe( - structuredAgentSessionPaneKey(structuredAgentSessionTabId(sessionId), sessionId) + structuredAgentSessionPaneKey(structuredAgentSessionTabId(SESSION), SESSION) ) }) @@ -87,6 +101,12 @@ describe('worktree ps reports structured sessions', () => { it('omits a session with no projected status', () => { expect(attach([summary({ status: null })]).agents).toHaveLength(0) }) + + it('keeps the journal clock on the row, so a restart republish is not new activity', () => { + const row = attach([summary()]) + expect(row.agents[0]?.updatedAt).toBe(1_757_030_400_000) + expect(row.agents[0]?.stateStartedAt).toBe(1_757_030_400_000) + }) }) /** @@ -95,11 +115,10 @@ describe('worktree ps reports structured sessions', () => { * refresh check goes permanently true and pins shipped clients to a fast cadence with no exit, and * the plugin projection has no field that can carry `writable: false`. Every SAFE consumer of a * terminal summary checks `ptyId`; the breaking ones key off `connected` or mere row presence, - * which no added field can qualify. A separate change publishes an honest partial-listing count - * there instead. This pins that only `worktree ps` gained the enumerator. + * which no added field can qualify. This pins that the listing never reads the status store. */ describe('terminal listing is deliberately left alone', () => { - it('only worktree ps consumes the structured status summaries', async () => { + it('never reads the agent-status store that now carries structured rows', async () => { const { readFile } = await import('node:fs/promises') // orca-runtime-subscribe-to-terminal-resize.ts owns listTerminals. const listing = await readFile( @@ -108,13 +127,7 @@ describe('terminal listing is deliberately left alone', () => { ) // Guard the guard: an empty read would make every assertion below vacuously true. expect(listing).toContain('async listTerminals(') - expect(listing).not.toContain('liveSessionStatusSummaries') - expect(listing).not.toContain('structuredSummaries') - - const worktreePs = await readFile( - new URL('./orca-runtime-get-worktree-ps.ts', import.meta.url), - 'utf8' - ) - expect(worktreePs).toContain('liveSessionStatusSummaries') + expect(listing).not.toContain('getAgentStatusSnapshotFn') + expect(listing).not.toContain('structuredHost') }) }) diff --git a/src/main/runtime/runtime-worktree-agent-rows.ts b/src/main/runtime/runtime-worktree-agent-rows.ts index 80802cf11a3..da145b67091 100644 --- a/src/main/runtime/runtime-worktree-agent-rows.ts +++ b/src/main/runtime/runtime-worktree-agent-rows.ts @@ -61,7 +61,8 @@ export function attachRuntimeWorktreeAgentRows(args: { toolInput: source.toolInput, interrupted: source.interrupted, stateStartedAt: source.stateStartedAt, - updatedAt: source.updatedAt + updatedAt: source.updatedAt, + ...(source.structuredHost === 'owned' ? { structuredHostOwned: true as const } : {}) } const rows = rowsByWorktree.get(summary.worktreeId) if (rows) { @@ -80,15 +81,13 @@ export function attachRuntimeWorktreeAgentRows(args: { let hasForegroundWorkingAgent = false const monitoringSources: RuntimeWorktreeAgentSource[] = [] for (const row of rows) { - const source = rowSources.get(row.paneKey) - const hostHeldStructuredSession = - source?.authority === 'structured-host' && row.state !== 'done' - if (!hostHeldStructuredSession && !isFreshNonDoneAgentStatus(row, now)) { + if (!isFreshNonDoneAgentStatus(row, now)) { continue } summary.hasHostSidebarActivity = true if (row.state === 'working') { if (row.workingMode === 'monitoring') { + const source = rowSources.get(row.paneKey) if (source) { monitoringSources.push(source) } diff --git a/src/main/runtime/runtime-worktree-agent-source.ts b/src/main/runtime/runtime-worktree-agent-source.ts index 2e8bc833b23..984f20e0955 100644 --- a/src/main/runtime/runtime-worktree-agent-source.ts +++ b/src/main/runtime/runtime-worktree-agent-source.ts @@ -1,3 +1,4 @@ +import type { StructuredHostStatus } from '../../shared/agent-hook-listener/listener-event' import type { ParsedAgentStatusPayload } from '../../shared/agent-status-types' export type RuntimeWorktreeAgentSource = { @@ -16,6 +17,6 @@ export type RuntimeWorktreeAgentSource = { interrupted: boolean stateStartedAt: number updatedAt: number - /** Structured host projections remain authoritative after PTY freshness expiry. */ - authority?: 'structured-host' + /** Projected by the structured session host; `owned` rows stay fresh past the staleness window. */ + structuredHost?: StructuredHostStatus } diff --git a/src/main/runtime/runtime-worktree-agent-sources.ts b/src/main/runtime/runtime-worktree-agent-sources.ts index 44ce7848960..9015b4bb0fb 100644 --- a/src/main/runtime/runtime-worktree-agent-sources.ts +++ b/src/main/runtime/runtime-worktree-agent-sources.ts @@ -1,22 +1,13 @@ -import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' import { collectRuntimeWorktreePtyAgentSources } from './runtime-worktree-pty-agent-sources' -import { structuredRuntimeWorktreeAgentSources } from './runtime-worktree-structured-agent-rows' import type { RuntimeWorktreeAgentSource } from './runtime-worktree-agent-source' /** One admitted roster for row and worktree-status projection. */ export function collectRuntimeWorktreeAgentSources( - args: Parameters[0] & { - structuredSummaries: readonly AgentSessionStatusSummary[] - } + args: Parameters[0] ): ReadonlyMap { const sources = new Map() for (const source of collectRuntimeWorktreePtyAgentSources(args)) { sources.set(source.paneKey, source) } - for (const source of structuredRuntimeWorktreeAgentSources(args.structuredSummaries)) { - if (!sources.has(source.paneKey)) { - sources.set(source.paneKey, source) - } - } return sources } diff --git a/src/main/runtime/runtime-worktree-pty-agent-sources.ts b/src/main/runtime/runtime-worktree-pty-agent-sources.ts index 04e8dffc04e..9f297d7edf7 100644 --- a/src/main/runtime/runtime-worktree-pty-agent-sources.ts +++ b/src/main/runtime/runtime-worktree-pty-agent-sources.ts @@ -94,7 +94,11 @@ export function collectRuntimeWorktreePtyAgentSources(args: { toolInput: entry.toolInput ?? null, interrupted: entry.interrupted ?? false, stateStartedAt: entry.stateStartedAt, - updatedAt: entry.receivedAt + // A structured row's clock is its journal, so a restart's republish does not read as new. + updatedAt: entry.structuredHost + ? (entry.evidenceObservedAt ?? entry.receivedAt) + : entry.receivedAt, + ...(entry.structuredHost ? { structuredHost: entry.structuredHost } : {}) }) } const sources: RuntimeWorktreeAgentSource[] = [] @@ -104,7 +108,10 @@ export function collectRuntimeWorktreePtyAgentSources(args: { parsePaneKey(source.paneKey)?.tabId ?? parseLegacyNumericPaneKey(source.paneKey)?.tabId const mirroredWorktreeId = tabId ? args.mirroredWorktreeIdByTabId.get(tabId) : undefined + // Why a structured row skips the connected-process gate: it has no PTY, and the host that + // holds the session drops the row itself on close, so its presence is the liveness evidence. if ( + source.structuredHost === undefined && tabId !== undefined && mirroredWorktreeId === undefined && (source.connectionId === null || isWslHookRelayConnectionId(source.connectionId)) && diff --git a/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts b/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts index 063ee0b36cf..519b9026f17 100644 --- a/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts +++ b/src/main/runtime/runtime-worktree-structured-agent-rows-liveness.test.ts @@ -2,19 +2,27 @@ import { collectRuntimeWorktreeAgentSources } from './runtime-worktree-agent-sou import { mkdtemp, rm } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { StructuredAgentSessionStatusFeed } from '../native-chat/agent-session-wire/structured-agent-session-status-feed' import { createTrackedJournalOpener } from '../native-chat/agent-session-journal/journal-store-test-open' +import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' import type { RuntimeWorktreePsSummary } from '../../shared/runtime-types' +import { AgentHookServer, _internals } from '../agent-hooks/server' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' +vi.mock('../telemetry/client', () => ({ track: vi.fn() })) +vi.mock('../telemetry/cohort-classifier', () => ({ + getCohortAtEmit: vi.fn(() => ({ nth_repo_added: 2 })) +})) + /** - * The whole chain `worktree ps` walks: journal -> status feed -> agent rows -> worktree status. + * The whole chain `worktree ps` walks: journal -> status feed -> agent-status store -> agent rows + * -> worktree status. * - * The feed's `published` map never retracts, so reading it as a roster reports every session the - * app has ever opened. A closed chat that was waiting on an approval is the sharp edge: deliberate - * close does not settle a pending prompt, so the retained summary stays `attention`, which maps to - * a `blocked` row and merges the worktree to `permission` for the 30-minute freshness window. + * The feed's `published` map never retracts, so it cannot be the roster. A closed chat that was + * waiting on an approval is the sharp edge: deliberate close does not settle a pending prompt, so + * the retained summary stays `attention`, which maps to a `blocked` row and would merge the + * worktree to `permission`. The store is the roster: the host drops the row on close. */ const WORKTREE_ID = 'repo-1::/workspace/app' const SESSION = 'a1b2c3d4-e5f6-4a7b-8c9d-0e1f2a3b4c5d' @@ -29,6 +37,7 @@ let root: string const journals = createTrackedJournalOpener() beforeEach(async () => { + _internals.resetCachesForTests() root = await mkdtemp(join(tmpdir(), 'orca-structured-ps-liveness-')) }) @@ -68,22 +77,32 @@ async function awaitingApproval() { const sessions = new Map([ [ SESSION, - { journal, params: { location: { workspaceId: WORKTREE_ID }, provider: 'codex' as const } } + { + journal, + hasProviderChild: true, + params: { location: { workspaceId: WORKTREE_ID }, provider: 'codex' as const } + } ] ]) + const store = new AgentHookServer() + const published: AgentSessionStatusSummary[] = [] const feed = new StructuredAgentSessionStatusFeed({ sessions, getRecord: () => null, - now: () => Date.now() + now: () => Date.now(), + statusSink: () => ({ + publish: (summary) => { + published.push(summary) + store.ingestStructuredStatus(summary) + }, + forget: (sessionId) => store.dropStructuredStatus(sessionId) + }) }) feed.publish(SESSION, journal) - return { feed, sessions } + return { feed, sessions, store, published } } -function worktreeFor( - feed: StructuredAgentSessionStatusFeed, - summaries = feed.liveSessionSummaries() -): RuntimeWorktreePsSummary { +function worktreeFor(store: AgentHookServer): RuntimeWorktreePsSummary { const row = { worktreeId: WORKTREE_ID, status: 'inactive', @@ -98,8 +117,7 @@ function worktreeFor( mirroredWorktreeIdByTabId: new Map(), connectedPtyEvidence: { tabIds: new Set(), paneKeys: new Set(), ptyIds: new Set() }, retainedSnapshots: [], - hookSnapshots: [], - structuredSummaries: summaries + hookSnapshots: store.getStatusSnapshot() }), orchestrationByPaneKey: null, getSummary: (map, _paths, _missing, id) => map.get(id) ?? null @@ -109,49 +127,62 @@ function worktreeFor( describe('worktree ps and a closed structured chat', () => { it('reports the blocked row while the session is still held', async () => { - const { feed } = await awaitingApproval() - const row = worktreeFor(feed) + const { store } = await awaitingApproval() + const row = worktreeFor(store) expect(row.agents).toHaveLength(1) expect(row.agents[0]?.state).toBe('blocked') expect(row.status).toBe('permission') }) - it('stops reporting it once eviction forgets the session', async () => { - const { feed, sessions } = await awaitingApproval() - // `forget-session`, the last eviction step, does exactly this and nothing to the feed. + it('stops reporting it once close forgets the session', async () => { + const { feed, sessions, store } = await awaitingApproval() + // What the host does after eviction: the feed keeps its projection, the store drops the row. sessions.delete(SESSION) + feed.close(SESSION) - const row = worktreeFor(feed) + const row = worktreeFor(store) expect(row.agents).toHaveLength(0) expect(row.status).toBe('inactive') }) it('keeps an aged host-held working state authoritative', async () => { - const { feed } = await awaitingApproval() - const aged = feed.liveSessionSummaries().map((summary) => ({ - ...summary, + const { store, published } = await awaitingApproval() + const aged = { + ...published.at(-1)!, hostExecutionOwned: true as const, updatedAt: Date.now() - 30 * 60 * 1000 - 1, status: 'working' as const - })) - const row = worktreeFor(feed, aged) + } + store.ingestStructuredStatus(aged) + const row = worktreeFor(store) expect(row.agents).toHaveLength(1) expect(row.agents[0]?.state).toBe('working') expect(row.status).toBe('working') - expect(row.agents[0]?.updatedAt).toBe(aged[0]?.updatedAt) + expect(row.agents[0]?.updatedAt).toBe(aged.updatedAt) }) it('keeps an aged host-held approval state authoritative', async () => { - const { feed } = await awaitingApproval() - const aged = feed.liveSessionSummaries().map((summary) => ({ - ...summary, + const { store, published } = await awaitingApproval() + const aged = { + ...published.at(-1)!, hostExecutionOwned: true as const, updatedAt: Date.now() - 30 * 60 * 1000 - 1 - })) - const row = worktreeFor(feed, aged) + } + store.ingestStructuredStatus(aged) + const row = worktreeFor(store) expect(row.agents).toHaveLength(1) expect(row.agents[0]?.state).toBe('blocked') expect(row.status).toBe('permission') - expect(row.agents[0]?.updatedAt).toBe(aged[0]?.updatedAt) + expect(row.agents[0]?.updatedAt).toBe(aged.updatedAt) + }) + + it('lets an aged approval decay once the host no longer owns the child', async () => { + const { store, published } = await awaitingApproval() + const { hostExecutionOwned: _owned, ...held } = published.at(-1)! + store.ingestStructuredStatus({ ...held, updatedAt: Date.now() - 30 * 60 * 1000 - 1 }) + const row = worktreeFor(store) + expect(row.agents).toHaveLength(1) + expect(row.agents[0]?.state).toBe('blocked') + expect(row.status).toBe('inactive') }) }) diff --git a/src/main/runtime/runtime-worktree-structured-agent-rows.ts b/src/main/runtime/runtime-worktree-structured-agent-rows.ts deleted file mode 100644 index c4d075fbd9c..00000000000 --- a/src/main/runtime/runtime-worktree-structured-agent-rows.ts +++ /dev/null @@ -1,48 +0,0 @@ -import type { AgentSessionStatusSummary } from '../../shared/agent-session-wire' -import { - structuredAgentSessionPaneKey, - structuredAgentSessionStatusState, - structuredAgentSessionTabId -} from '../../shared/structured-agent-session-projection' -import type { RuntimeWorktreeAgentSource } from './runtime-worktree-agent-source' - -/** - * Row sources for the structured (non-PTY) sessions a host still holds. - * - * A structured session reaches none of the hook or retained snapshots every other row comes from, - * so `worktree ps` projects it from the host's status feed instead. The feed's retained - * projections are not a roster — the caller passes only sessions the host still holds. - */ -export function structuredRuntimeWorktreeAgentSources( - summaries: readonly AgentSessionStatusSummary[] -): RuntimeWorktreeAgentSource[] { - const sources: RuntimeWorktreeAgentSource[] = [] - for (const summary of summaries) { - // No turn has been persisted yet, so there is nothing to report - the same read the chat shows. - if (!summary.status) { - continue - } - const tabId = structuredAgentSessionTabId(summary.sessionId) - // The DERIVED pane key the renderer already publishes, never the orchestration bearer handle - // or the minted worker pane key: both of those are credentials. - sources.push({ - paneKey: structuredAgentSessionPaneKey(tabId, summary.sessionId), - tabId, - worktreeId: summary.workspaceId, - connectionId: null, - // The shared mapping the sidebar applies, so the CLI and the GUI cannot disagree about one - // session. No hook payload: nothing reads one off a structured row. - state: structuredAgentSessionStatusState(summary.status), - agentType: summary.agent, - prompt: summary.latestPrompt, - lastAssistantMessage: summary.lastAssistantMessage ?? null, - toolName: summary.toolName ?? null, - toolInput: summary.toolInput ?? null, - interrupted: false, - stateStartedAt: summary.updatedAt, - updatedAt: summary.updatedAt, - ...(summary.hostExecutionOwned ? { authority: 'structured-host' as const } : {}) - }) - } - return sources -} diff --git a/src/main/runtime/structured-agent-session-close.test.ts b/src/main/runtime/structured-agent-session-close.test.ts new file mode 100644 index 00000000000..531ca0aa129 --- /dev/null +++ b/src/main/runtime/structured-agent-session-close.test.ts @@ -0,0 +1,237 @@ +/** + * The chat tab must survive a close that did not land. + * + * `closeStructuredAgentSessionChild` hides the tab BEFORE it issues the close, so every failure + * shape past that point used to leave the user's chat tab pulled out of the durable restore index + * for a session that is still running — a destructive operation that refused, and still took + * something away. + */ + +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' + +const hostRef: { current: unknown } = { current: null } + +vi.mock('../native-chat/agent-session-wire/structured-agent-session-registry', () => ({ + getStructuredAgentSessionHost: () => hostRef.current +})) + +const { closeStructuredAgentSessionChild } = await import('./structured-agent-session-close') + +const SESSION = 'session-1' + +function record(sessionId: string): AgentSessionRecord { + return { + sessionId, + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'repo_1::/tmp/wt-a', + workspaceKind: 'folder' + }, + lease: { + sessionId, + runtimeKind: 'native', + claimStatus: 'live', + handoffStage: null, + runtimeFence: 1, + deathEvidence: null + } + } as unknown as AgentSessionRecord +} + +type HostOptions = { + /** Sessions the host keeps holding through a close, so the post-close observation is `live`. */ + stuck?: boolean + /** Rejects the close, without the child going. */ + closeThrows?: Error + /** The child dies and is recorded dead, but the close then fails past that proof. */ + settledThenThrows?: boolean + /** Rejects the visibility write itself, so the hide never lands. */ + visibilityThrows?: Error + /** Sessions already in the persisted visible-tab index. */ + visible?: string[] + /** Blows up the index read, so the rollback cannot prove the tab was ever visible. */ + indexThrows?: boolean +} + +function installHost(options: HostOptions = {}) { + const entry = record(SESSION) + const held = new Set([SESSION]) + const visible = new Set(options.visible ?? [SESSION]) + const setSessionTabVisibility = vi.fn(async (sessionId: string, isVisible: boolean) => { + if (options.visibilityThrows) { + throw options.visibilityThrows + } + if (isVisible) { + visible.add(sessionId) + } else { + visible.delete(sessionId) + } + }) + const close = vi.fn(async (sessionId: string) => { + if (options.closeThrows) { + throw options.closeThrows + } + if (options.stuck) { + return + } + held.delete(sessionId) + entry.lease.claimStatus = 'released' + entry.lease.deathEvidence = { kind: 'exit-observed', detail: 'closed', observedAt: 1 } + if (options.settledThenThrows) { + throw new Error('the event sink could not be flushed') + } + }) + hostRef.current = { + deps: { store: { getRecord: (id: string) => (id === SESSION ? entry : null) } }, + hasSession: (sessionId: string) => held.has(sessionId), + getPersistedVisibleSessionTabIndex: () => { + if (options.indexThrows) { + throw new Error('visible tab index unreadable') + } + return { present: true, sessionIds: [...visible] } + }, + setSessionTabVisibility, + close + } + return { close, setSessionTabVisibility, visible } +} + +describe('closeStructuredAgentSessionChild tab-visibility rollback', () => { + beforeEach(() => { + hostRef.current = null + vi.restoreAllMocks() + }) + + it('retires the tab and reports the close on the success path', async () => { + const host = installHost() + const retire = vi.fn(() => true) + + const outcome = await closeStructuredAgentSessionChild(SESSION, { + runtime: { + retireStructuredAgentSessionTabFromSnapshot: retire + } as never + }) + + expect(outcome).toEqual({ stopped: true, closeAttempted: true }) + expect(host.visible.has(SESSION)).toBe(false) + expect(retire).toHaveBeenCalledWith(SESSION) + // The hide is the only visibility write a settled close performs. + expect(host.setSessionTabVisibility.mock.calls).toEqual([[SESSION, false]]) + }) + + it('restores the tab when the close throws and the child is still there', async () => { + const host = installHost({ closeThrows: new Error('provider round trip failed') }) + + const outcome = await closeStructuredAgentSessionChild(SESSION) + + expect(outcome.stopped).toBe(false) + expect(outcome.closeAttempted).toBe(true) + expect(outcome.reason).toBe('provider round trip failed') + expect(host.visible.has(SESSION)).toBe(true) + expect(host.setSessionTabVisibility.mock.calls).toEqual([ + [SESSION, false], + [SESSION, true] + ]) + }) + + it('restores the tab when the post-close observation is not `exited`', async () => { + const host = installHost({ stuck: true }) + + const outcome = await closeStructuredAgentSessionChild(SESSION) + + expect(outcome.stopped).toBe(false) + expect(outcome.closeAttempted).toBe(true) + expect(host.visible.has(SESSION)).toBe(true) + expect(host.setSessionTabVisibility.mock.calls).toEqual([ + [SESSION, false], + [SESSION, true] + ]) + }) + + it('leaves the tab retired when a close throws PAST a proven exit', async () => { + // `closeStructuredSessionsForWorktree` re-observes and counts this session closed; republishing + // the tab here would resurrect it at the next launch for a workspace that is gone. + const host = installHost({ settledThenThrows: true }) + + const outcome = await closeStructuredAgentSessionChild(SESSION) + + expect(outcome.stopped).toBe(false) + expect(host.visible.has(SESSION)).toBe(false) + expect(host.setSessionTabVisibility.mock.calls).toEqual([[SESSION, false]]) + }) + + it('does not put the tab back when the caller is discarding the workspace anyway', async () => { + // Worktree teardown passes this off for a removal that cannot refuse — force, and the + // folder-workspace paths. A tab put back there is a durable reference to a workspace that is + // about to be gone, so it republishes the chat at the next launch pointing at it. + const host = installHost({ stuck: true }) + + const outcome = await closeStructuredAgentSessionChild(SESSION, { + restoreTabOnUnprovenClose: false + }) + + expect(outcome.stopped).toBe(false) + expect(host.visible.has(SESSION)).toBe(false) + expect(host.setSessionTabVisibility.mock.calls).toEqual([[SESSION, false]]) + }) + + it('does not publish a tab for a session that was already hidden', async () => { + const host = installHost({ closeThrows: new Error('provider round trip failed'), visible: [] }) + + await closeStructuredAgentSessionChild(SESSION) + + expect(host.visible.has(SESSION)).toBe(false) + expect(host.setSessionTabVisibility.mock.calls).toEqual([[SESSION, false]]) + }) + + it('does not roll back a visibility write that never landed', async () => { + const host = installHost({ visibilityThrows: new Error('visibility write failed') }) + + const outcome = await closeStructuredAgentSessionChild(SESSION) + + expect(outcome).toEqual({ + stopped: false, + closeAttempted: false, + reason: 'visibility write failed' + }) + expect(host.close).not.toHaveBeenCalled() + expect(host.setSessionTabVisibility.mock.calls).toEqual([[SESSION, false]]) + }) + + it('keeps the original failure when the restore itself throws', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const host = installHost({ stuck: true }) + host.setSessionTabVisibility.mockImplementation(async (_sessionId, isVisible) => { + if (isVisible) { + throw new Error('agent_session_identity_required') + } + }) + + const outcome = await closeStructuredAgentSessionChild(SESSION) + + expect(outcome.stopped).toBe(false) + expect(outcome.closeAttempted).toBe(true) + expect(outcome.reason).not.toContain('agent_session_identity_required') + expect(warn).toHaveBeenCalled() + }) + + it('claims nothing when the visible-tab index cannot be read', async () => { + const host = installHost({ indexThrows: true, closeThrows: new Error('boom') }) + + await closeStructuredAgentSessionChild(SESSION) + + expect(host.setSessionTabVisibility.mock.calls).toEqual([[SESSION, false]]) + }) + + it('reports no close attempt when no host is installed', async () => { + hostRef.current = null + + const outcome = await closeStructuredAgentSessionChild(SESSION) + + expect(outcome.stopped).toBe(false) + expect(outcome.closeAttempted).toBe(false) + }) +}) diff --git a/src/main/runtime/structured-agent-session-close.ts b/src/main/runtime/structured-agent-session-close.ts index 756dbdeaef4..f65d1204db9 100644 --- a/src/main/runtime/structured-agent-session-close.ts +++ b/src/main/runtime/structured-agent-session-close.ts @@ -11,6 +11,7 @@ * longer live is proven gone. Anything else is retained rather than settled. */ +import type { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import type { OrcaRuntimeService } from './orca-runtime' import { retireSettledStructuredWorkerTab } from './structured-agent-session-tab-retirement' @@ -35,6 +36,15 @@ export type StructuredAgentSessionCloseOptions = { * keep the child un-evictable for the life of the app. Every settlement has to reach it. */ afterClose?: () => void + /** + * Whether an unproven close may put the chat tab back in the durable restore index. + * + * On by default, which is the retryable case: a stop that refused and still took the user's tab + * away is the loss the rollback exists to undo. A caller that will discard the WORKSPACE + * whatever this close reports passes false — a tab put back there is a durable reference to a + * workspace about to be gone, and it republishes the chat at the next launch pointing at it. + */ + restoreTabOnUnprovenClose?: boolean } export async function closeStructuredAgentSessionChild( @@ -50,6 +60,11 @@ export async function closeStructuredAgentSessionChild( reason: 'The structured agent-session host is not installed; no session was closed.' } } + // Read BEFORE the hide, so a rollback puts the tab back exactly as it was. Restoring + // unconditionally would publish a tab for a session that was already hidden — a worker started + // without a chat tab, or one the user had closed — which is a new side effect, not an undo. + const restoreTabIfCloseFails = + options.restoreTabOnUnprovenClose !== false && readPersistedTabVisibility(host, sessionId) // Set only once the close is actually issued: `setSessionTabVisibility` throwing first leaves a // running child, and a receipt that still said `closed_agent_terminal` for it would be the // close-that-never-happened this flag exists to rule out. @@ -59,6 +74,11 @@ export async function closeStructuredAgentSessionChild( closeAttempted = true await host.close(sessionId) } catch (error) { + // Only `closeAttempted` proves the hide landed: the store transaction restores its own state on + // failure, so a `setSessionTabVisibility` that threw hid nothing and has nothing to undo. + if (closeAttempted) { + await restorePersistedTabVisibility(host, sessionId, restoreTabIfCloseFails) + } return { stopped: false, closeAttempted, @@ -68,6 +88,7 @@ export async function closeStructuredAgentSessionChild( options.afterClose?.() const observation = observeStructuredWorker({ sessionId }) if (observation.status !== 'exited') { + await restorePersistedTabVisibility(host, sessionId, restoreTabIfCloseFails) return { stopped: false, closeAttempted: true, @@ -79,3 +100,52 @@ export async function closeStructuredAgentSessionChild( retireSettledStructuredWorkerTab(sessionId, options.runtime) return { stopped: true, closeAttempted: true } } + +function readPersistedTabVisibility(host: StructuredAgentSessionHost, sessionId: string): boolean { + try { + return host.getPersistedVisibleSessionTabIndex?.().sessionIds.includes(sessionId) ?? false + } catch { + // Unreadable index: claim nothing. A rollback that cannot prove the tab was visible must not + // publish one, for the same reason the read exists at all. + return false + } +} + +/** + * Puts the chat tab back after a close that did not settle. + * + * The hide is the one visible side effect this function performs before the destructive step, so a + * failed close that kept it left the user's chat tab gone from the durable restore index — the + * conversation survived under `userData`, but nothing brought the tab back at the next launch. + * + * Re-observed first rather than restored outright: a close can throw PAST its own proof and still + * have taken the child with it, and `closeStructuredSessionsForWorktree` reads exactly that, + * counting such a session closed and retiring its tab. Republishing there would resurrect a tab for + * a session that is demonstrably gone, at the next launch, pointing at a deleted workspace. + * + * That observation NARROWS the window; it does not close it. This one and the sweep's are taken a + * store write apart, so a child that dies in between is unverifiable here and exited there — which + * is why the sweep re-drops the tab reference when it takes that proof. Do not delete either half + * on the strength of the other. + * + * Never throws: the caller's `reason` is what the user is asked to act on, and a rollback failure + * must not replace it. `agent_session_identity_required` is the expected one — the record can be + * gone by now, which is itself the exit this restore is declining to undo. + */ +async function restorePersistedTabVisibility( + host: StructuredAgentSessionHost, + sessionId: string, + restoreTab: boolean +): Promise { + if (!restoreTab || observeStructuredWorker({ sessionId }).status === 'exited') { + return + } + try { + await host.setSessionTabVisibility?.(sessionId, true) + } catch (error) { + console.warn( + `[structured-session-close] could not restore the chat tab for ${sessionId} after a failed close`, + error + ) + } +} diff --git a/src/main/runtime/structured-agent-session-integration.test.ts b/src/main/runtime/structured-agent-session-integration.test.ts index aa14aaa5639..f5a59033d73 100644 --- a/src/main/runtime/structured-agent-session-integration.test.ts +++ b/src/main/runtime/structured-agent-session-integration.test.ts @@ -661,8 +661,10 @@ describe('a structured codex session over agentSession.*', () => { expect(older.page.hasOlder).toBe(false) // Every step of the conversation, in order, from the durable journal alone — // no page overlaps another, and nothing the live stream showed is missing. + // The turn's lifecycle row outlives the turn: it is revised, never tombstoned. expect([...older.page.items, ...tail.page.items].map((item) => item.body?.kind)).toEqual([ 'message', + 'status', 'message', 'tool-call', 'approval', @@ -671,6 +673,7 @@ describe('a structured codex session over agentSession.*', () => { ]) expect([...older.page.items, ...tail.page.items].map(textOf)).toEqual([ 'list files', + '', 'Two files.', '', '', diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index 0245f15054b..f7798b11db8 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -82,6 +82,8 @@ export type StructuredAgentSessionRuntimeDeps = { /** Every structured-session status projection, for host-side reactions such as the first-work * workspace rename that CLI agents get from their hooks. */ onSessionStatusChanged?: StructuredAgentSessionHostDeps['onSessionStatusChanged'] + /** The agent-status store; see `StructuredAgentSessionHostDeps.statusSink`. */ + statusSink?: StructuredAgentSessionHostDeps['statusSink'] handoffTransport?: StructuredAgentSessionHandoffTransport reapOrphanChildren?: typeof stopOrphanAgentSessionChildren } @@ -300,6 +302,7 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise { await store.transitionHandoff(sessionId, (record) => recordAgentSessionProviderHandle({ record, fence: record.lease.runtimeFence, link, now }) diff --git a/src/main/runtime/structured-conversation-tab-replacement.ts b/src/main/runtime/structured-conversation-tab-replacement.ts index 94d9bdf52d9..7384a374c91 100644 --- a/src/main/runtime/structured-conversation-tab-replacement.ts +++ b/src/main/runtime/structured-conversation-tab-replacement.ts @@ -1,3 +1,4 @@ +import { defaultAgentChatLabel } from '../../shared/agent-session-chat-label' import type { RuntimeMobileSessionTabsSnapshot } from '../../shared/runtime-types' import type { ConversationReplacement } from '../native-chat/agent-session-wire/structured-conversation-command' @@ -30,7 +31,7 @@ export function replaceConversationInSnapshot( id, sessionId: replacement.sessionId, agent: replacement.agent, - title: replacement.agent === 'claude' ? 'Claude Chat' : 'Codex Chat', + title: defaultAgentChatLabel(replacement.agent), replacesSessionId: replacement.sourceSessionId } : tab diff --git a/src/main/runtime/structured-session-worktree-teardown.test.ts b/src/main/runtime/structured-session-worktree-teardown.test.ts index a9bdf6aa45c..84977aad4eb 100644 --- a/src/main/runtime/structured-session-worktree-teardown.test.ts +++ b/src/main/runtime/structured-session-worktree-teardown.test.ts @@ -8,18 +8,31 @@ vi.mock('../native-chat/agent-session-wire/structured-agent-session-registry', ( })) const { killAllProcessesForWorktree } = await import('./worktree-teardown') -const { classifyWorktreeForceDeleteReason } = await import('../../shared/worktree/removal') +const { + classifyWorktreeForceDeleteReason, + isProvenLiveStructuredSessionRemovalError, + isUnstoppedPtyRemovalError +} = await import('../../shared/worktree/removal') const { listLiveStructuredSessionsForWorktree } = await import('./structured-session-worktree-teardown') const WORKTREE = 'repo_1::/tmp/wt-a' const OTHER_WORKTREE = 'repo_1::/tmp/wt-b' -function record(sessionId: string, workspaceId: string): AgentSessionRecord { +function record( + sessionId: string, + workspaceId: string, + options: { provider?: 'claude' | 'codex'; executionHostId?: string } = {} +): AgentSessionRecord { return { sessionId, - provider: 'claude', - location: { executionHostId: 'local', wslDistro: null, workspaceId, workspaceKind: 'folder' }, + provider: options.provider ?? 'claude', + location: { + executionHostId: options.executionHostId ?? 'local', + wslDistro: null, + workspaceId, + workspaceKind: 'folder' + }, lease: { sessionId, runtimeKind: 'native', @@ -33,24 +46,66 @@ function record(sessionId: string, workspaceId: string): AgentSessionRecord { function installHost(options: { records: AgentSessionRecord[] - /** Sessions the host still holds; a close removes one unless it is listed as stuck. */ + /** Sessions the host keeps holding through a close, so the post-close observation is `live`. */ stuck?: Set -}): { closed: string[] } { + /** Sessions the host drops without death evidence, so the observation is `unverifiable`. */ + unverifiable?: Set + /** Sessions whose child dies and is recorded dead, but whose close then fails past that point. */ + settledThenThrows?: Set + /** Blocks every close, to exercise the shared sweep budget without fake timers. */ + closeGate?: Promise + /** Blocks ONE session's close, so the serial loop can be caught part-way through. */ + closeGates?: Record> + /** Sessions in the persisted visible-tab index, so a rollback has something to put back. */ + visible?: string[] + /** + * Sessions whose death evidence lands DURING the close's tab-restore write. + * + * `setSessionTabVisibility` is a store transaction — a real disk write — so the close's own + * observation and the sweep's re-read straddle it and can disagree about the same session. + */ + exitsDuringTabRestore?: Set +}): { closed: string[]; visible: Set } { const held = new Set(options.records.map((entry) => entry.sessionId)) const closed: string[] = [] + const visible = new Set(options.visible ?? []) + const recordExit = (sessionId: string): void => { + const entry = options.records.find((candidate) => candidate.sessionId === sessionId) + if (entry) { + entry.lease.claimStatus = 'released' + entry.lease.deathEvidence = { kind: 'exit-observed', detail: 'closed', observedAt: 1 } + } + } hostRef.current = { deps: { store: { listRecords: () => options.records, getRecord: () => null } }, hasSession: (sessionId: string) => held.has(sessionId), - setSessionTabVisibility: async () => {}, + getPersistedVisibleSessionTabIndex: () => ({ present: true, sessionIds: [...visible] }), + setSessionTabVisibility: async (sessionId: string, isVisible: boolean) => { + if (!isVisible) { + visible.delete(sessionId) + return + } + if (options.exitsDuringTabRestore?.has(sessionId)) { + recordExit(sessionId) + } + visible.add(sessionId) + }, close: async (sessionId: string) => { closed.push(sessionId) - if (!options.stuck?.has(sessionId)) { - held.delete(sessionId) - const record = options.records.find((entry) => entry.sessionId === sessionId) - if (record) { - record.lease.claimStatus = 'released' - record.lease.deathEvidence = { kind: 'exit-observed', detail: 'closed', observedAt: 1 } - } + await options.closeGate + await options.closeGates?.[sessionId] + if (options.stuck?.has(sessionId)) { + return + } + held.delete(sessionId) + if (options.unverifiable?.has(sessionId)) { + return + } + if (!options.exitsDuringTabRestore?.has(sessionId)) { + recordExit(sessionId) + } + if (options.settledThenThrows?.has(sessionId)) { + throw new Error('the event sink could not be flushed') } } } @@ -59,7 +114,7 @@ function installHost(options: { hostRef.current as { deps: { store: { getRecord: (id: string) => unknown } } } ).deps.store.getRecord = (sessionId: string) => options.records.find((entry) => entry.sessionId === sessionId) ?? null - return { closed } + return { closed, visible } } const localProvider = { @@ -67,7 +122,7 @@ const localProvider = { shutdown: async () => {} } as never -function destructiveDeps(extra: { allowUnverifiedStop?: boolean } = {}) { +function destructiveDeps(extra: { allowUnverifiedStop?: boolean; timeoutMs?: number } = {}) { return { localProvider, requirePhysicalStop: true, @@ -77,6 +132,15 @@ function destructiveDeps(extra: { allowUnverifiedStop?: boolean } = {}) { } } +/** The structured sweep's own warn — a forced removal can emit a PTY-sweep one onto the same spy. */ +function structuredSessionWarning(warn: { mock: { calls: unknown[][] } }): string { + return ( + warn.mock.calls + .map((call) => String(call[0])) + .find((message) => message.includes('agent session')) ?? '' + ) +} + describe('worktree teardown and structured agent sessions', () => { beforeEach(() => { hostRef.current = null @@ -84,24 +148,98 @@ describe('worktree teardown and structured agent sessions', () => { it('finds sessions by workspace, and ignores a sibling worktree', () => { installHost({ records: [record('s1', WORKTREE), record('s2', OTHER_WORKTREE)] }) - expect(listLiveStructuredSessionsForWorktree(WORKTREE)).toEqual([ + expect(listLiveStructuredSessionsForWorktree(WORKTREE, {})).toEqual([ { sessionId: 's1', agent: 'claude' } ]) }) - it('refuses a destructive removal rather than deleting the checkout under a live child', async () => { - // The defect this pins: all three PTY sweeps enumerate leaves, provider sessions and the local - // registry, and a structured session is on NONE of them. Every sweep answered zero, nothing - // errored, and removal proceeded — leaving the provider child running with its `cwd` deleted - // and the dispatch still reporting the worker live and exact. - installHost({ records: [record('s1', WORKTREE)] }) + it('closes a live session on an ordinary removal instead of refusing it', async () => { + // The defect this pins, and the reason the guard is not simply deleted: all three PTY sweeps + // enumerate leaves, provider sessions and the local registry, and a structured session is on + // NONE of them, so removal used to proceed leaving the provider child running with its `cwd` + // deleted. The stop belongs on the ordinary path — the same one that kills a terminal running + // the same agent — so an idle chat is no harder to delete than that terminal. + const host = installHost({ records: [record('s1', WORKTREE)] }) + await expect(killAllProcessesForWorktree(WORKTREE, destructiveDeps())).resolves.toMatchObject({ + structuredStopped: 1 + }) + expect(host.closed).toEqual(['s1']) + }) + + it('refuses only when the close does not settle', async () => { + installHost({ records: [record('s1', WORKTREE)], stuck: new Set(['s1']) }) await expect(killAllProcessesForWorktree(WORKTREE, destructiveDeps())).rejects.toThrow( - /1 running agent session/ + /still live: 1 agent session \(claude\)/ ) }) + it('puts the chat tab back when the removal refuses over the session', async () => { + // The workspace survives a refusal, so the tab has to survive it too: a destructive operation + // that refused and still took the user's chat tab away is the loss the rollback exists to undo. + const host = installHost({ + records: [record('s1', WORKTREE)], + stuck: new Set(['s1']), + visible: ['s1'] + }) + await expect(killAllProcessesForWorktree(WORKTREE, destructiveDeps())).rejects.toThrow( + /still live: 1 agent session \(claude\)/ + ) + expect([...host.visible]).toEqual(['s1']) + }) + + it('leaves the chat tab dropped when a forced removal deletes the workspace anyway', async () => { + // The other half of the same rollback. Force does not refuse — it warns and goes on to delete + // the checkout — so putting the tab back leaves a DURABLE reference to a workspace that is + // about to be gone, which republishes the chat at the next launch pointing at a deleted + // worktree: the exact outcome this whole sweep exists to remove. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const host = installHost({ + records: [record('s1', WORKTREE)], + stuck: new Set(['s1']), + visible: ['s1'] + }) + await killAllProcessesForWorktree(WORKTREE, destructiveDeps({ allowUnverifiedStop: true })) + expect([...host.visible]).toEqual([]) + warn.mockRestore() + }) + + it('leaves the chat tab dropped for a folder-workspace removal, which never refuses', async () => { + // Same reasoning without the force waiver: this caller cannot refuse at all, so the workspace + // is forgotten whatever the close reports. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const host = installHost({ + records: [record('s1', WORKTREE)], + stuck: new Set(['s1']), + visible: ['s1'] + }) + await killAllProcessesForWorktree(WORKTREE, { + localProvider, + includeProviderInventory: false as const, + includeLocalRegistry: false as const, + closeStructuredSessions: true + }) + expect([...host.visible]).toEqual([]) + warn.mockRestore() + }) + + it('drops the chat tab for a session the sweep proves exited after the close gave up', async () => { + // `host.close` can return BEFORE the child's exit is recorded, so the close's own observation + // reads unverifiable and puts the tab back — and the sweep's re-read, one store write later, + // proves the exit and counts the session closed. The two observations straddle that write and + // can disagree; the tab must not survive the disagreement, because this removal proceeds. + const host = installHost({ + records: [record('s1', WORKTREE)], + visible: ['s1'], + exitsDuringTabRestore: new Set(['s1']) + }) + await expect(killAllProcessesForWorktree(WORKTREE, destructiveDeps())).resolves.toMatchObject({ + structuredStopped: 1 + }) + expect([...host.visible]).toEqual([]) + }) + it('names the force escape hatch in the refusal, like the unstopped-PTY gate', async () => { - installHost({ records: [record('s1', WORKTREE)] }) + installHost({ records: [record('s1', WORKTREE)], stuck: new Set(['s1']) }) await expect(killAllProcessesForWorktree(WORKTREE, destructiveDeps())).rejects.toThrow(/force/i) }) @@ -109,7 +247,7 @@ describe('worktree teardown and structured agent sessions', () => { // The #11960 dead end, and the shape this file's own comments warn about: the desktop // affordance comes ONLY from the classifier, and an ordinary delete already passes force:true // for the dirty-file skip — so a refusal with no matcher shows raw CLI wording with no button. - installHost({ records: [record('s1', WORKTREE)] }) + installHost({ records: [record('s1', WORKTREE)], stuck: new Set(['s1']) }) const error = await killAllProcessesForWorktree(WORKTREE, destructiveDeps()).catch( (thrown: Error) => thrown.message ) @@ -123,12 +261,12 @@ describe('worktree teardown and structured agent sessions', () => { // A session id is one tab-id hop from the random pane key that gates a worker's mailbox, and // this string reaches CLI output and a desktop toast. A count and the providers are what a // user deciding whether to force actually needs. - installHost({ records: [record('s1', WORKTREE)] }) + installHost({ records: [record('s1', WORKTREE)], stuck: new Set(['s1']) }) const error = await killAllProcessesForWorktree(WORKTREE, destructiveDeps()).catch( (thrown: Error) => thrown.message ) expect(error).not.toContain('s1') - expect(error).toContain('1 running agent session') + expect(error).toContain('1 agent session (claude)') }) it('closes best-effort for a folder-workspace removal, which requires no stop proof', async () => { @@ -166,10 +304,33 @@ describe('worktree teardown and structured agent sessions', () => { destructiveDeps({ allowUnverifiedStop: true }) ) expect(result.structuredStopped).toBeUndefined() - expect(warn).toHaveBeenCalledWith(expect.stringContaining('still attached')) + // The live arm of that record, carrying the verdict the refusal would have shown. + expect(structuredSessionWarning(warn)).toContain('still live: 1 agent session (claude)') warn.mockRestore() }) + it('takes the proof when a failed close is re-observed as exited', async () => { + // `closeStructuredAgentSessionChild` reports `stopped: false` for anything that throws past its + // own observation, and for a record whose death evidence lands after it read. The re-read here + // can still PROVE the exit — refusing a delete over a child that is demonstrably gone is the + // defect this whole sweep exists to remove, so the proof has to win over the close's verdict. + const retired: string[] = [] + const runtime = { + stopTerminalsForWorktree: async () => ({ stopped: 0 }), + retireStructuredAgentSessionTabFromSnapshot: (sessionId: string) => { + retired.push(sessionId) + return true + } + } as never + installHost({ records: [record('s1', WORKTREE)], settledThenThrows: new Set(['s1']) }) + await expect( + killAllProcessesForWorktree(WORKTREE, { ...destructiveDeps(), runtime }) + ).resolves.toMatchObject({ structuredStopped: 1 }) + // Retired here because the close gave up before its own retirement step, and a chat tab left + // behind re-attaches a released session pointing at a workspace that is about to be deleted. + expect(retired).toEqual(['s1']) + }) + it('leaves the best-effort reconciliation paths alone', async () => { // Those callers repair state and delete nothing, so a refusal there would wedge a repair. installHost({ records: [record('s1', WORKTREE)] }) @@ -182,6 +343,262 @@ describe('worktree teardown and structured agent sessions', () => { ).resolves.toMatchObject({ runtimeStopped: 0 }) }) + it('leaves a same-id workspace on another execution host alone', async () => { + // A workspace id is `repoId::path` with no host component, so the local, SSH and paired-runtime + // copies of one id are DIFFERENT workspaces. Unfenced, deleting the local one closed a chat + // running on somebody else's machine — a destructive cross-host act, not a spurious refusal. + const host = installHost({ + records: [record('s1', WORKTREE, { executionHostId: 'ssh:host-a' })] + }) + await expect(killAllProcessesForWorktree(WORKTREE, destructiveDeps())).resolves.toMatchObject({ + runtimeStopped: 0 + }) + expect(host.closed).toEqual([]) + }) + + it('reads an explicit local fence the way the PTY sweeps do', () => { + // This helper reuses the PTY fence's own type, so the two cannot answer `null` differently: + // there it means this machine, and it has to mean this machine here. ABSENT is the one + // deliberate difference — no fence at all for the PTY sweeps, narrowed to local here, because + // a single-host-id comparison cannot express match-all and closing every host's chats is + // destructive. Latent today only because `WorktreeTeardownDeps` cannot yet carry the `null`. + installHost({ + records: [record('s1', WORKTREE, { executionHostId: 'ssh:host-a' }), record('s2', WORKTREE)] + }) + const local = [{ sessionId: 's2', agent: 'claude' }] + expect(listLiveStructuredSessionsForWorktree(WORKTREE, { resolvedConnectionId: null })).toEqual( + local + ) + expect(listLiveStructuredSessionsForWorktree(WORKTREE, {})).toEqual(local) + }) + + it('closes only the session on the host the removal resolved to', async () => { + const host = installHost({ + records: [record('s1', WORKTREE, { executionHostId: 'ssh:host-a' }), record('s2', WORKTREE)] + }) + await expect( + killAllProcessesForWorktree(WORKTREE, { + ...destructiveDeps(), + resolvedConnectionId: 'host-a' + }) + ).resolves.toMatchObject({ structuredStopped: 1 }) + expect(host.closed).toEqual(['s1']) + }) + + it('names only the sessions that stayed, and every provider still there', async () => { + installHost({ + records: [ + record('s1', WORKTREE), + record('s2', WORKTREE, { provider: 'codex' }), + record('s3', WORKTREE) + ], + stuck: new Set(['s2', 's3']) + }) + const error = await killAllProcessesForWorktree(WORKTREE, destructiveDeps()).catch( + (thrown: Error) => thrown.message + ) + expect(error).toContain('still live: 2 agent sessions (claude, codex)') + }) + + it('names the unconfirmed sessions too, instead of counting only the live ones', async () => { + // The PTY sibling may drop everything outside its live list because a fresh inventory PROVED + // those exited. Nothing proves that here: an `unverifiable` session is unclosed as well, so + // naming only the live subset told the user "1 agent session" while two were about to go. + installHost({ + records: [record('s1', WORKTREE), record('s2', WORKTREE, { provider: 'codex' })], + stuck: new Set(['s1']), + unverifiable: new Set(['s2']) + }) + const error = await killAllProcessesForWorktree(WORKTREE, destructiveDeps()).catch( + (thrown: Error) => thrown.message + ) + expect(error).toContain( + 'still live: 1 agent session (claude); could not confirm these closed: 1 agent session (codex)' + ) + // The marker still leads, so the toast keeps showing the stronger of the two warnings. + expect(isProvenLiveStructuredSessionRemovalError(error as string)).toBe(true) + }) + + it('still reports what it closed when a forced removal skips the PTY verdict', async () => { + // A sweep that fails outright short-circuits the per-PTY verdict — but not the structured + // close that already ran, so the count has to survive that return or the removal log claims + // `structured=0` for chats it just ended. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const runtime = { + stopTerminalsForWorktree: async () => { + throw new Error('the terminal sweep died') + } + } as never + const host = installHost({ records: [record('s1', WORKTREE)] }) + const result = await killAllProcessesForWorktree(WORKTREE, { + ...destructiveDeps({ allowUnverifiedStop: true }), + runtime + }) + expect(host.closed).toEqual(['s1']) + expect(result.structuredStopped).toBe(1) + warn.mockRestore() + }) + + it('separates a close it could not confirm from one it watched stay attached', async () => { + // `src/shared/worktree/removal.ts` keeps these two apart on purpose: a user waiving "we could + // not confirm" is making a different decision than one discarding a conversation Orca just saw + // running. The toast branches on this marker, so flattening them makes one of the two a lie. + installHost({ records: [record('s1', WORKTREE)], unverifiable: new Set(['s1']) }) + const unconfirmed = await killAllProcessesForWorktree(WORKTREE, destructiveDeps()).catch( + (thrown: Error) => thrown.message + ) + expect(unconfirmed).toContain('could not confirm these closed: 1 agent session (claude)') + expect(isProvenLiveStructuredSessionRemovalError(unconfirmed as string)).toBe(false) + + installHost({ records: [record('s1', WORKTREE)], stuck: new Set(['s1']) }) + const live = await killAllProcessesForWorktree(WORKTREE, destructiveDeps()).catch( + (thrown: Error) => thrown.message + ) + expect(isProvenLiveStructuredSessionRemovalError(live as string)).toBe(true) + }) + + it('refuses in agent-session wording when the close outlives the sweep budget', async () => { + // A structured close that runs out of time used to reject with the PTY timeout sentinel, which + // the classifier reads FIRST — so the toast blamed terminals, and the Force Delete meant to + // clear the wedge hit the same rejection again (#11960). + installHost({ records: [record('s1', WORKTREE)], closeGate: new Promise(() => {}) }) + const error = await killAllProcessesForWorktree( + WORKTREE, + destructiveDeps({ timeoutMs: 5 }) + ).catch((thrown: Error) => thrown.message) + expect(error).toContain('could not confirm these closed: 1 agent session (claude)') + expect(isUnstoppedPtyRemovalError(error as string)).toBe(false) + expect(classifyWorktreeForceDeleteReason(error as string, true)).toBe('running-agent-session') + }) + + it('never wedges Force Delete on a close that will not settle', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + installHost({ records: [record('s1', WORKTREE)], closeGate: new Promise(() => {}) }) + await expect( + killAllProcessesForWorktree( + WORKTREE, + destructiveDeps({ allowUnverifiedStop: true, timeoutMs: 5 }) + ) + ).resolves.toMatchObject({ runtimeStopped: 0 }) + const message = structuredSessionWarning(warn) + expect(message).toContain('could not confirm these closed: 1 agent session (claude)') + // The pin: a close that ran out of time was never watched stay attached. This warn is the only + // record a forced removal leaves, and the removal.ts split exists precisely so "we could not + // confirm" is never reported as "we saw it running" — including here. + expect(message).not.toContain('still attached') + warn.mockRestore() + }) + + it('names only the sessions still open when the budget expires mid-close', async () => { + // The close loop is serial, so a deadline can land part-way through it. A fallback assembled + // at the deadline could only name the whole list — so a removal that had already closed the + // first chat still told the user both were still there, which is the exact thing this sweep + // exists to stop doing: never report state nobody observed. + installHost({ + records: [record('s1', WORKTREE), record('s2', WORKTREE, { provider: 'codex' })], + closeGates: { s2: new Promise(() => {}) } + }) + const error = await killAllProcessesForWorktree( + WORKTREE, + destructiveDeps({ timeoutMs: 40 }) + ).catch((thrown: Error) => thrown.message) + expect(error).toContain('could not confirm these closed: 1 agent session (codex)') + expect(error).not.toContain('claude') + }) + + it('counts the closes that landed before the budget expired', async () => { + // The other half of the same fallback: it reported zero closes, so the removal log said + // `structured=0` for a chat it had just ended. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const slowClose = new Promise((resolve) => { + setTimeout(resolve, 300) + }) + installHost({ + records: [record('s1', WORKTREE), record('s2', WORKTREE, { provider: 'codex' })], + closeGates: { s2: slowClose } + }) + const result = await killAllProcessesForWorktree( + WORKTREE, + destructiveDeps({ allowUnverifiedStop: true, timeoutMs: 40 }) + ) + expect(result.structuredStopped).toBe(1) + expect(structuredSessionWarning(warn)).toContain( + 'could not confirm these closed: 1 agent session (codex)' + ) + warn.mockRestore() + }) + + it('stops issuing new closes once the budget is spent', async () => { + // One slow provider round trip used to starve every session behind it: the outer race had + // already given up on the loop, and it went on issuing closes whose outcome nobody would read. + // The in-flight one is NOT cancelled — nothing here can cancel a provider round trip — so it + // still has to be reported, which is why both sessions are named below. + let releaseFirstClose: () => void = () => {} + const firstClose = new Promise((resolve) => { + releaseFirstClose = resolve + }) + const host = installHost({ + records: [record('s1', WORKTREE), record('s2', WORKTREE)], + closeGates: { s1: firstClose } + }) + const error = await killAllProcessesForWorktree( + WORKTREE, + destructiveDeps({ timeoutMs: 5 }) + ).catch((thrown: Error) => thrown.message) + expect(error).toContain('could not confirm these closed: 2 agent sessions (claude)') + releaseFirstClose() + await new Promise((resolve) => { + setTimeout(resolve, 25) + }) + expect(host.closed).toEqual(['s1']) + }) + + it('leaves the terminals already stopped when it refuses over a stuck session', async () => { + // Pins a tradeoff that was accepted, not an outcome that is wanted. The PTY sweeps now run + // concurrently with the structured close, so a removal that refuses over a session that will + // not close has ALREADY killed that workspace's terminals — the head-first serial order spared + // them. Serialising it back is worse: it spends the whole shared budget before a single PTY is + // asked, and the alternative — refusing before the PTY sweeps — leaves force-delete removing + // files while PTY handles are open. The PTY gate itself already kills first and refuses only + // on what it could not verify stopped. A later change must not flip this back silently. + let terminalSweeps = 0 + const runtime = { + stopTerminalsForWorktree: async () => { + terminalSweeps += 1 + return { stopped: 2 } + } + } as never + installHost({ records: [record('s1', WORKTREE)], stuck: new Set(['s1']) }) + await expect( + killAllProcessesForWorktree(WORKTREE, { ...destructiveDeps(), runtime }) + ).rejects.toThrow(/still live: 1 agent session \(claude\)/) + expect(terminalSweeps).toBe(1) + }) + + it('starts the terminal sweeps while the structured close is still in flight', async () => { + // The close is serial and each one waits on a provider round trip. Awaiting it before the + // sweeps exist spends the shared budget head-first, and the sweeps then report a timeout for + // a stop they never attempted. + let releaseClose: () => void = () => {} + const closeGate = new Promise((resolve) => { + releaseClose = resolve + }) + installHost({ records: [record('s1', WORKTREE)], closeGate }) + let terminalSweepStarted = false + const runtime = { + stopTerminalsForWorktree: async () => { + terminalSweepStarted = true + return { stopped: 0 } + } + } as never + const removal = killAllProcessesForWorktree(WORKTREE, { ...destructiveDeps(), runtime }) + await vi.waitFor(() => { + expect(terminalSweepStarted).toBe(true) + }) + releaseClose() + await expect(removal).resolves.toMatchObject({ structuredStopped: 1 }) + }) + it('does not block removal when no structured host is installed', async () => { // Not being able to look is not evidence a child is there, and reading the persisted store // directly would force-install the host as a side effect of a teardown. diff --git a/src/main/runtime/structured-session-worktree-teardown.ts b/src/main/runtime/structured-session-worktree-teardown.ts index 226f785f039..8b161e4f5a8 100644 --- a/src/main/runtime/structured-session-worktree-teardown.ts +++ b/src/main/runtime/structured-session-worktree-teardown.ts @@ -8,15 +8,29 @@ * kept running with its `cwd` gone, the durable record and chat tab survived to republish at the * next launch pointing at a deleted worktree, and `worker-show` still reported the worker live. * - * Membership is `location.workspaceId`, which every structured session carries — so this covers a - * plain chat session in the worktree as well as a dispatched worker. Liveness is - * `observeStructuredWorker`, the same `live` / `unverifiable` / `exited` vocabulary the rest of the - * structured surface uses; only a PROVEN live child is worth refusing a removal over. + * Membership is `location.workspaceId` PLUS the host fence below, and every structured session + * carries both — so this covers a plain chat session in the worktree as well as a dispatched + * worker. Liveness is `observeStructuredWorker`, the same `live` / `unverifiable` / `exited` + * vocabulary the rest of the structured surface uses. + * + * `live` here is lease state — a provider child is attached — not work in flight, so it says + * nothing about whether the user would lose anything. It selects what to CLOSE, never what to + * refuse over: a removal refuses only on a close that did not settle, exactly as the PTY sweep + * refuses only on a stop it could not verify. */ +import { + LOCAL_EXECUTION_HOST_ID, + toRuntimeExecutionHostId, + toSshExecutionHostId, + type ExecutionHostId +} from '../../shared/execution-host' +import { STILL_LIVE_DETAIL_PREFIX } from '../../shared/worktree/removal' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' import { observeStructuredWorker } from './structured-worker-authority' import { closeStructuredAgentSessionChild } from './structured-agent-session-close' +import { retireSettledStructuredWorkerTab } from './structured-agent-session-tab-retirement' +import type { WorktreePtyHostFence } from './worktree-pty-host-fence' import type { OrcaRuntimeService } from './orca-runtime' export type LiveStructuredSessionInWorkspace = { @@ -24,13 +38,52 @@ export type LiveStructuredSessionInWorkspace = { agent: 'claude' | 'codex' } +export type UnclosedStructuredSession = LiveStructuredSessionInWorkspace & { + /** Read AFTER the close: `live` is a child watched stay attached, not merely one left unproven. */ + status: 'live' | 'unverifiable' +} + export type StructuredWorktreeSweepRuntime = Pick< OrcaRuntimeService, 'forgetStructuredSessionMail' | 'retireStructuredAgentSessionTabFromSnapshot' > /** - * Structured sessions with a proven-live child in this worktree. + * The two fields every teardown caller already resolves to fence its PTY sweeps to one host. + * + * Deliberately the PTY fence's own type rather than a look-alike: these two helpers are written + * against each other, so a widening on one side must not become a silent disagreement on the + * other. `resolvedConnectionId: null` means this machine on both. + * + * They differ in exactly one reading, and only that one: ABSENT. The PTY fence takes it as no + * fence at all and matches every host, which a single-host-id comparison cannot express — and + * closing every host's chats is destructive, not merely noisy. So this side reads absent as local + * too, the narrower half of that pair. Pinned by test, not left to the next reader to rediscover. + */ +export type StructuredSessionHostFence = WorktreePtyHostFence + +/** + * The one execution host this teardown may touch. + * + * A workspace id is `repoId::path` with no host component, so the local machine, an SSH host and a + * paired runtime can all publish the SAME id and each names a DIFFERENT workspace (STA-4343). The + * PTY sweeps fence on exactly these two fields; a structured session records its host directly, so + * the comparison is on `location.executionHostId` instead of on a pty-id shape. + */ +export function structuredSessionTeardownHostId( + fence: StructuredSessionHostFence +): ExecutionHostId { + if (fence.resolvedRuntimeEnvironmentId !== undefined) { + return toRuntimeExecutionHostId(fence.resolvedRuntimeEnvironmentId) + } + // Both no-connection readings collapse here on purpose — see the fence type. A caller that + // resolved no host, and one that resolved this machine, each close nothing on anyone else's. + const connectionId = fence.resolvedConnectionId ?? null + return connectionId === null ? LOCAL_EXECUTION_HOST_ID : toSshExecutionHostId(connectionId) +} + +/** + * Structured sessions with a proven-live child in this worktree, on the fenced host only. * * An uninstalled host answers empty rather than throwing: no host in this generation means no * provider child was started by this process, and the three PTY sweeps fall through the same way @@ -38,7 +91,8 @@ export type StructuredWorktreeSweepRuntime = Pick< * directly — that would force-install the host, which is itself a side effect on a teardown path. */ export function listLiveStructuredSessionsForWorktree( - worktreeId: string + worktreeId: string, + fence: StructuredSessionHostFence ): LiveStructuredSessionInWorkspace[] { const host = getStructuredAgentSessionHost() if (!host) { @@ -50,58 +104,191 @@ export function listLiveStructuredSessionsForWorktree( } catch { return [] } + const hostId = structuredSessionTeardownHostId(fence) return records .filter( (record) => record.location.workspaceId === worktreeId && + record.location.executionHostId === hostId && observeStructuredWorker({ sessionId: record.sessionId }).status === 'live' ) .map((record) => ({ sessionId: record.sessionId, agent: record.provider })) } /** - * Counts and providers, never session ids. + * A count and its providers — never session ids. * * A session id is one tab-id hop from the random pane key that gates a worker's mailbox, and this * string reaches agent-readable CLI output and a desktop toast. The count and the providers are * what a user deciding whether to force actually needs; the ids identify nothing they can act on. */ -export function describeLiveStructuredSessions( - sessions: readonly LiveStructuredSessionInWorkspace[] -): string { +function countStructuredSessions(sessions: readonly UnclosedStructuredSession[]): string { const noun = sessions.length === 1 ? 'agent session' : 'agent sessions' const providers = [...new Set(sessions.map((session) => session.agent))].sort().join(', ') - return `${sessions.length} running ${noun} (${providers})` + return `${sessions.length} ${noun} (${providers})` } /** - * Closes every live structured session in the worktree, and reports what stayed. + * The two post-close verdicts, each with its own count. * - * Force is the documented escape hatch, so it closes rather than orphaning: a child left running - * against a deleted `cwd` is the exact outcome this whole sweep exists to prevent. + * The split is here for the reason `describeUnstoppedPtys` carries one: "we watched it stay + * attached" and "we could not confirm it went" are different decisions to waive, and the delete + * toast branches on the marker a proven-live session leads with. + * + * Both groups are named, though, which is where this differs from the PTY sibling: there, the + * verdict is a fresh inventory, so anything absent from the live list is PROVEN exited and + * rightly dropped. Here an `unverifiable` session is unclosed too — folding it into the live + * count would overstate what Orca watched, and dropping it said "1 agent session" while three + * were about to be discarded. + */ +export function describeUnclosedStructuredSessions( + sessions: readonly UnclosedStructuredSession[] +): string { + const stillLive = sessions.filter((session) => session.status === 'live') + const unconfirmed = sessions.filter((session) => session.status !== 'live') + if (stillLive.length === 0) { + return `could not confirm these closed: ${countStructuredSessions(unconfirmed)}` + } + const live = `${STILL_LIVE_DETAIL_PREFIX} ${countStructuredSessions(stillLive)}` + return unconfirmed.length === 0 + ? live + : `${live}; could not confirm these closed: ${countStructuredSessions(unconfirmed)}` +} + +/** + * What the close loop has done so far, readable while it is still running. + * + * The loop is serial and every close waits on a provider round trip, so the shared sweep budget can + * expire part-way through it. This is written as it goes rather than returned at the end, because + * the caller's timeout path reads THIS: a fabricated whole-list fallback reported sessions the + * sweep had already closed as unclosed, named them in the refusal the user reads, and logged + * `structured=0` for closes that landed. Saying only what was observed is the point of the sweep. + */ +export type StructuredSweepProgress = { + /** The sessions this sweep closes, in the order the loop reaches them. */ + readonly sessions: readonly LiveStructuredSessionInWorkspace[] + /** Sessions no longer attached after their close — the count this sweep reports. */ + closed: number + /** Attempted closes that did not settle, each carrying the verdict re-read after the attempt. */ + unstopped: UnclosedStructuredSession[] + /** How many of `sessions`, from the front, have an outcome recorded. */ + settled: number +} + +export function createStructuredSweepProgress( + sessions: readonly LiveStructuredSessionInWorkspace[] +): StructuredSweepProgress { + return { sessions, closed: 0, unstopped: [], settled: 0 } +} + +/** + * Everything this sweep did not prove closed. + * + * A session with no recorded outcome — never started, or still in flight — reports `unverifiable`, + * the same verdict as an attempted close that stayed unproven. Chosen, not conflated: the vocabulary is `live` / `unverifiable` / `exited` with no + * synonyms, and "we never asked" and "we asked and could not confirm" are both exactly "not + * observed exited". A fourth bucket would need its own refusal wording and its own toast + * classification for a distinction the user cannot act on any differently — and `live` is the only + * verdict either could be mistaken for, which is the one thing neither is allowed to claim. + */ +export function unclosedStructuredSessions( + progress: StructuredSweepProgress +): UnclosedStructuredSession[] { + return [ + ...progress.unstopped, + ...progress.sessions + .slice(progress.settled) + .map((session) => ({ ...session, status: 'unverifiable' as const })) + ] +} + +/** + * Closes the structured sessions in `progress`, recording what stayed as it goes. + * + * Runs on the ordinary removal too, not just force: a child left running against a deleted `cwd` is + * the outcome this whole sweep exists to prevent, and closing is how you prevent it. What stayed is + * the only thing worth refusing over. + * + * Takes the list rather than re-deriving it, so the refusal can only ever name a session out of + * the set this sweep was handed — re-enumerating would run every liveness observation twice and + * let it name one this call never touched. Not every one of them is a session a close was + * attempted on: the deadline check below can leave the tail of the list unasked, and + * `unclosedStructuredSessions` reports those as `unverifiable` precisely because nobody looked. */ export async function closeStructuredSessionsForWorktree( - worktreeId: string, - runtime?: StructuredWorktreeSweepRuntime -): Promise<{ closed: number; unstopped: LiveStructuredSessionInWorkspace[] }> { + progress: StructuredSweepProgress, + deadline: number, + options: { + runtime?: StructuredWorktreeSweepRuntime + /** + * Whether this removal can still refuse over an unclosed session. + * + * It is the only case where the workspace — and therefore its chat tabs — survives, so it is + * the only case where an unproven close may put a tab back. Force and the folder-workspace + * paths discard the workspace whatever the sweep reports. + */ + mayRefuse?: boolean + } = {} +): Promise { + const { runtime, mayRefuse } = options // No `afterClose` for a dispatched worker: `host.close` drops the holds, so nothing keeps a // provider child un-evictable, but the dispatch's redrive subscription and registry entry do // survive until it settles by another verb. That is a bounded leak, not a hazard — and passing // one here would mean resolving a dispatch id per session on a teardown path that must stay // inside the sweep deadline. - const sessions = listLiveStructuredSessionsForWorktree(worktreeId) - const unstopped: LiveStructuredSessionInWorkspace[] = [] - let closed = 0 - for (const session of sessions) { - const outcome = await closeStructuredAgentSessionChild( - session.sessionId, - runtime ? { runtime } : {} - ) - if (outcome.stopped) { - closed += 1 - } else { - unstopped.push(session) + for (const session of progress.sessions) { + // Stops ISSUING new closes once the budget is spent; an in-flight one is left to finish, since + // nothing here can cancel a provider round trip. Without this, one slow round trip starved + // every session behind it: the caller's race had already given up, and the loop went on + // closing sessions whose outcome nobody would read. + if (Date.now() >= deadline) { + return } + const outcome = await closeStructuredAgentSessionChild(session.sessionId, { + ...(runtime ? { runtime } : {}), + restoreTabOnUnprovenClose: mayRefuse === true + }) + if (outcome.stopped) { + progress.closed += 1 + } else { + // Re-observed rather than reusing the close's own reason string: what the user is asked to + // waive is the state AFTER the attempt, and a close that threw never reached an observation. + const status = observeStructuredWorker({ sessionId: session.sessionId }).status + if (status === 'exited') { + // The re-read can PROVE the exit a failed close could not — it threw past its own + // observation, or the record's death evidence landed after it read. Refusing on a child + // that is demonstrably gone is the defect this sweep exists to remove, so take the proof + // and run the retirement `closeStructuredAgentSessionChild` skipped when it gave up. + // + // Including the hide it UNDID: its rollback ran against an observation taken one store + // write before this one, so a child that died in between left the tab republished for a + // session this sweep is about to count closed. Taking the proof has to take that back. + await dropDurableChatTabReference(session.sessionId) + retireSettledStructuredWorkerTab(session.sessionId, runtime) + progress.closed += 1 + } else { + progress.unstopped.push({ ...session, status }) + } + } + // Advanced only once an outcome is recorded, so a close still in flight when the deadline + // lands stays reported as unclosed instead of falling out of both counts. + progress.settled += 1 + } +} + +/** + * Drops a settled session's durable chat-tab reference, and cannot fail the settlement. + * + * The close's own hide is the ordinary path; this is only for the session whose exit this sweep + * proved after that close had already rolled the hide back. + */ +async function dropDurableChatTabReference(sessionId: string): Promise { + try { + await getStructuredAgentSessionHost()?.setSessionTabVisibility?.(sessionId, false) + } catch (error) { + console.warn( + `[worktree-teardown] could not drop the chat tab reference for ${sessionId}`, + error + ) } - return { closed, unstopped } } diff --git a/src/main/runtime/terminal-ansi-normalization.ts b/src/main/runtime/terminal-ansi-normalization.ts index df76da3c35e..e1bbe33a79d 100644 --- a/src/main/runtime/terminal-ansi-normalization.ts +++ b/src/main/runtime/terminal-ansi-normalization.ts @@ -1,4 +1,5 @@ import { MAX_TAIL_PENDING_ANSI_CHARS } from './terminal-tail-limits' +import { classifyTerminalEscapeIntroducer } from '../../shared/terminal-escape-introducer' import { ownRetainedString } from '../../shared/own-retained-string' export function parseAnsiControlSequence( @@ -11,8 +12,9 @@ export function parseAnsiControlSequence( endIndex: number } | null { - const introducer = value[escapeIndex + 1] - if (introducer === '[') { + // charCodeAt, not value[i]: indexing mints a one-char string on every escape. + const introducer = classifyTerminalEscapeIntroducer(value.charCodeAt(escapeIndex + 1)) + if (introducer === 'csi') { for (let index = escapeIndex + 2; index < value.length; index += 1) { const code = value.charCodeAt(index) if (code < 0x40 || code > 0x7e) { @@ -30,7 +32,7 @@ export function parseAnsiControlSequence( } return null } - if (introducer === ']') { + if (introducer === 'osc') { for (let index = escapeIndex + 2; index < value.length; index += 1) { if (value[index] === '\u0007') { return { kind: 'other', endIndex: index } @@ -41,7 +43,7 @@ export function parseAnsiControlSequence( } return null } - if (isStTerminatedStringControlIntroducer(introducer)) { + if (introducer === 'string') { for (let index = escapeIndex + 2; index < value.length; index += 1) { if (value[index] === '\u001b' && value[index + 1] === '\\') { return { kind: 'other', endIndex: index + 1 } @@ -52,10 +54,6 @@ export function parseAnsiControlSequence( return { kind: 'other', endIndex: escapeIndex + 1 } } -function isStTerminatedStringControlIntroducer(introducer: string | undefined): boolean { - return introducer === 'P' || introducer === 'X' || introducer === '^' || introducer === '_' -} - export function hasCanonicalNumericCsiParams(params: string): boolean { return /^[0-9;]*$/.test(params) } diff --git a/src/main/runtime/terminal-interactive-wait-visibility.test.ts b/src/main/runtime/terminal-interactive-wait-visibility.test.ts index 987bdbb15b7..5e652af41f6 100644 --- a/src/main/runtime/terminal-interactive-wait-visibility.test.ts +++ b/src/main/runtime/terminal-interactive-wait-visibility.test.ts @@ -3,7 +3,10 @@ import { readFileSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it, vi } from 'vitest' -import { OrcaRuntimeService } from './orca-runtime' +import { + createTranscriptPane as createPane, + TRANSCRIPT_PANE_PTY_ID as PTY_ID +} from './agent-transcript-pane-test-harness' import { assertTerminalAgentSendable } from './rpc/terminal-agent-send-guard' vi.mock('electron', () => ({ @@ -13,12 +16,11 @@ vi.mock('electron', () => ({ app: { getPath: vi.fn(() => '/tmp') } })) -const LEAF_ID = '11111111-1111-4111-8111-111111111111' -const TAB_ID = 'tab-1' -const WORKTREE_ID = 'wt-1' -const PTY_ID = 'pty-1' - -// Captured verbatim from cursor-agent 2026.08.11-e8db854 driven through Orca. +// cursor-agent 2026.08.11-e8db854's screens, but NOT raw PTY output: these files contain no +// escape bytes and no carriage returns, so they came through a terminal's renderer and a +// clipboard. They evidence wording, ordering and glyphs — which is all the rules below key on — +// and evidence nothing about the caret, cursor moves, repaints or the alternate screen buffer. +// Record new fixtures with config/scripts/capture-agent-pty-transcript.mjs, which keeps the bytes. function fixture(name: string): string { return readFileSync(join(__dirname, '__fixtures__', `${name}.txt`), 'utf8') } @@ -39,73 +41,6 @@ function agentStatusOsc(state: string): string { return `]9999;${JSON.stringify({ state, prompt: 'ship it', agentType: 'claude' })}` } -async function createPane(options: { - paneTitle: string - foregroundProcess: string | null - data: string - /** Set for a pane whose PTY lives on an SSH host or WSL distro rather than locally. */ - connectionId?: string - /** Simulates a PTY controller whose foreground probe never settles. */ - foregroundProbeHangs?: boolean - onForegroundProbe?: () => void -}): Promise<{ runtime: OrcaRuntimeService; handle: string }> { - const runtime = new OrcaRuntimeService(null) - const internals = runtime as unknown as { - resolveTerminalWorkspaceLaunchScope: (selector: string) => Promise - } - vi.spyOn(internals, 'resolveTerminalWorkspaceLaunchScope').mockResolvedValue({ - id: WORKTREE_ID, - path: '/repo/app', - connectionId: options.connectionId ?? null, - repo: null, - folderWorkspace: null - }) - runtime.setPtyController({ - spawn: vi.fn().mockResolvedValue({ id: PTY_ID, incarnationId: 'inc-1' }), - write: () => true, - kill: () => true, - getForegroundProcess: (): Promise => { - options.onForegroundProbe?.() - return options.foregroundProbeHangs === true - ? new Promise(() => {}) - : Promise.resolve(options.foregroundProcess) - } - }) - const terminal = await runtime.createTerminal(`id:${WORKTREE_ID}`, { - tabId: TAB_ID, - leafId: LEAF_ID, - title: 'Terminal' - }) - runtime.attachWindow(1) - runtime.syncWindowGraph(1, { - tabs: [ - { - tabId: TAB_ID, - worktreeId: WORKTREE_ID, - title: 'Terminal', - activeLeafId: LEAF_ID, - layout: null - } - ], - leaves: [ - { - tabId: TAB_ID, - worktreeId: WORKTREE_ID, - leafId: LEAF_ID, - paneRuntimeId: 1, - ptyId: PTY_ID, - paneTitle: options.paneTitle - } - ] - }) - // Why the guard: a restore seed is only applied to a never-written record, so the restore - // cases must not write an empty chunk first. - if (options.data.length > 0) { - runtime.onPtyData(PTY_ID, options.data, Date.now()) - } - return { runtime, handle: terminal.handle } -} - // cursor-agent renders a braille spinner in its OSC title while it works, and Orca reads // that as `working`; the title is identical whether it is running a command or waiting. const CURSOR_TITLE = '⠇ Cursor Agent' @@ -299,7 +234,7 @@ describe('terminal interactive-wait visibility (STA-4513, STA-3714)', () => { }) await expect(runtime.showTerminal(handle)).resolves.toMatchObject({ - agentWait: { source: 'prompt-text', reason: 'codex-trust-workspace' } + agentWait: { source: 'prompt-text', reason: 'agent-trust-workspace' } }) }) diff --git a/src/main/runtime/terminal-send-stale-leaf-liveness.test.ts b/src/main/runtime/terminal-send-stale-leaf-liveness.test.ts index 7d173055b3a..b7c5ed95068 100644 --- a/src/main/runtime/terminal-send-stale-leaf-liveness.test.ts +++ b/src/main/runtime/terminal-send-stale-leaf-liveness.test.ts @@ -386,6 +386,7 @@ function makeOrchestrationDbStub(toHandle: () => string) { runMailbox, markAsDelivered, markAsUndelivered, + releaseMailboxPointerEnter, stageMailboxPointerEnter, insert(subject: string, type: StoredMessageRow['type'] = 'status'): void { rows.push({ @@ -759,7 +760,7 @@ describe('push-on-idle orchestration delivery absence gate', () => { await vi.advanceTimersByTimeAsync(500) expect(write.mock.calls.filter(([, data]) => data === '\r')).toHaveLength(0) expect(stub.stageMailboxPointerEnter).toHaveBeenCalledOnce() - expect(stub.markAsUndelivered).toHaveBeenCalledOnce() + expect(stub.releaseMailboxPointerEnter).toHaveBeenCalledOnce() expect(stub.rows[0].delivered_at).toBeNull() // The replacement's own delivery starts a fresh flight and completes — @@ -804,7 +805,7 @@ describe('push-on-idle orchestration delivery absence gate', () => { await vi.advanceTimersByTimeAsync(500) expect(write.mock.calls.filter(([, data]) => data === '\r')).toHaveLength(0) expect(stub.stageMailboxPointerEnter).toHaveBeenCalledOnce() - expect(stub.markAsUndelivered).toHaveBeenCalledOnce() + expect(stub.releaseMailboxPointerEnter).toHaveBeenCalledOnce() // No stray settle flushed the parked trigger into the dead pty. expect(write).toHaveBeenCalledTimes(1) expect(stub.rows.every((row) => row.delivered_at === null)).toBe(true) @@ -835,7 +836,7 @@ describe('push-on-idle orchestration delivery absence gate', () => { await vi.advanceTimersByTimeAsync(500) expect(write.mock.calls.filter(([, data]) => data === '\r')).toHaveLength(0) expect(stub.stageMailboxPointerEnter).toHaveBeenCalledOnce() - expect(stub.markAsUndelivered).toHaveBeenCalledOnce() + expect(stub.releaseMailboxPointerEnter).toHaveBeenCalledOnce() expect(stub.rows[0].delivered_at).toBeNull() } finally { vi.useRealTimers() diff --git a/src/main/runtime/terminal-tail-sentinel-index.test.ts b/src/main/runtime/terminal-tail-sentinel-index.test.ts index 2b33b2770ea..cd8bf2e68fb 100644 --- a/src/main/runtime/terminal-tail-sentinel-index.test.ts +++ b/src/main/runtime/terminal-tail-sentinel-index.test.ts @@ -193,7 +193,7 @@ describe('terminal tail sentinel index', () => { expect(tailMayContainBlockedSignal(seeded)).toBe(true) const state = computeTerminalTailWaitState(seeded, '', '') expect(state.fromTail).toBe(true) - expect(state.signal?.reason).toBe('codex-update-prompt') + expect(state.signal?.reason).toBe('agent-update-prompt') const clean = ['boot log', 'no prompt here', 'trailing'] expect(tailMayContainBlockedSignal(clean)).toBe(false) diff --git a/src/main/runtime/terminal-wait-detection.test.ts b/src/main/runtime/terminal-wait-detection.test.ts index eda02e60bbb..e52344c5dc0 100644 --- a/src/main/runtime/terminal-wait-detection.test.ts +++ b/src/main/runtime/terminal-wait-detection.test.ts @@ -98,12 +98,12 @@ const LIVE_CODEX_PROMPTS: { name: string; lines: string[]; reason: string }[] = '2. Trust all and continue', 'Press enter to confirm or esc to go back' ], - reason: 'codex-hooks-review-prompt' + reason: 'agent-hooks-review-prompt' }, { name: 'trust workspace', lines: ['Do you trust this workspace directory?', '1. Yes', '2. No'], - reason: 'codex-trust-workspace' + reason: 'agent-trust-workspace' }, { name: 'update', @@ -113,7 +113,7 @@ const LIVE_CODEX_PROMPTS: { name: string; lines: string[]; reason: string }[] = '2. Skip', 'Press enter to continue' ], - reason: 'codex-update-prompt' + reason: 'agent-update-prompt' }, { name: 'cwd selection', @@ -123,7 +123,7 @@ const LIVE_CODEX_PROMPTS: { name: string; lines: string[]; reason: string }[] = ' Current = your current working directory', ' Press enter to continue' ], - reason: 'codex-cwd-prompt' + reason: 'agent-cwd-prompt' }, { name: 'model migration', @@ -142,7 +142,7 @@ const LIVE_CODEX_PROMPTS: { name: string; lines: string[]; reason: string }[] = '2. No, continue without permissions', 'Press enter to confirm or esc to cancel' ], - reason: 'codex-interactive-prompt' + reason: 'agent-interactive-prompt' }, { name: 'permission required', @@ -153,7 +153,7 @@ const LIVE_CODEX_PROMPTS: { name: string; lines: string[]; reason: string }[] = 'Allow always', 'Reject' ], - reason: 'codex-interactive-prompt' + reason: 'agent-interactive-prompt' } ] @@ -195,6 +195,301 @@ describe('detectTerminalWaitBlockedReason live prompts', () => { 'Press enter to confirm' ]) - expect(detectTerminalWaitBlockedReason(waitText)).toBe('codex-hooks-review-prompt') + expect(detectTerminalWaitBlockedReason(waitText)).toBe('agent-hooks-review-prompt') }) }) + +// Why: these matchers never inspect the pane's agent, so a Codex-named reason on a non-Codex screen +// reaches the user verbatim through the CLI and the worker receipt's "Agent startup blocked:" line. +describe('detectTerminalWaitBlockedReason on non-Codex agents', () => { + const NON_CODEX_PROMPTS: { name: string; lines: string[]; reason: string }[] = [ + { + name: 'an Antigravity workspace trust dialog', + lines: [ + 'Antigravity CLI 1.0.3', + 'Do you trust the files in this folder?', + '1. Yes, I trust this folder', + '2. No, exit' + ], + reason: 'agent-trust-workspace' + }, + { + name: 'a Claude Code trusted-workspace dialog', + lines: [ + 'Claude Code', + 'Trusted workspace?', + 'This directory has not been opened before.', + '1. Yes, proceed', + '2. No, exit' + ], + reason: 'agent-trust-workspace' + }, + { + name: 'a Gemini CLI update banner', + lines: [ + 'Gemini CLI', + 'Update available! 1.4.0 -> 1.5.0', + '1. Update now', + '2. Skip', + 'Press enter to continue' + ], + reason: 'agent-update-prompt' + }, + { + name: 'a Gemini CLI permission dialog', + lines: [ + 'Gemini CLI', + 'Permission required', + 'Running this tool requires permission', + 'Allow once', + 'Allow always', + 'Reject' + ], + reason: 'agent-interactive-prompt' + }, + { + name: 'a Claude Code hooks review dialog', + lines: [ + 'Claude Code', + 'Hooks need review', + 'PreToolUse:Bash .claude/hooks/guard.sh', + 'Press enter to confirm' + ], + reason: 'agent-hooks-review-prompt' + }, + { + name: 'an Antigravity sandbox confirmation', + lines: [ + 'Antigravity CLI 1.0.3', + 'This action runs outside the sandbox.', + 'Press enter to confirm or esc to go back' + ], + reason: 'agent-interactive-prompt' + } + ] + + // Why: the reason was previously picked by looking for 'codex' in 600 chars of scrollback, so any + // agent that merely narrated about Codex handed its user a Codex label. + it('does not borrow a Codex label from scrollback that only mentions Codex', () => { + const waitText = waitTextFor([ + 'Antigravity CLI 1.0.3', + 'I read src/codex-notes.md for you.', + 'This action runs outside the sandbox.', + 'Press enter to confirm or esc to go back' + ]) + + expect(waitText.toLowerCase()).toContain('codex') + expect(detectTerminalWaitBlockedReason(waitText)).toBe('agent-interactive-prompt') + }) + + for (const prompt of NON_CODEX_PROMPTS) { + it(`reports an agent-neutral reason for ${prompt.name}`, () => { + const waitText = waitTextFor(prompt.lines) + const reason = detectTerminalWaitBlockedReason(waitText) + + expect(waitText.toLowerCase()).not.toContain('codex') + expect(reason).toBe(prompt.reason) + expect(reason?.startsWith('codex-')).toBe(false) + }) + } +}) + +// Antigravity readiness, and what this file does NOT claim about it. +// +// The detector recognizes a ready screen by header + a 'gemini'-prefixed model line + a lone '>' +// caret. That is narrow: an Antigravity user on a non-Gemini model never reaches ready and the pane +// wedges. Widening it was attempted and reverted -- every candidate rule was tuned against the +// constructed fixtures below, and the last one let a live sign-in dialog read as ready (the +// orchestrator then types the task prompt into an authentication dialog, which is strictly worse +// than a timeout). No real Antigravity transcript exists in this repo; the cursor-agent rules are +// derived from captures under src/main/runtime/__fixtures__ and Antigravity has no equivalent. +// Widening the model rule needs one first. See the ratchet at the bottom of this block for the +// shapes any replacement has to refuse. +describe('Antigravity readiness does not absorb its own startup dialog', () => { + const TRUST_DIALOG_WITH_CARET = [ + 'Antigravity CLI 1.0.3', + 'Do you trust the files in this folder?', + '1. Yes, I trust this folder', + '2. No, exit', + '>' + ] + + const LIVE_DIALOGS_UNDER_THE_HEADER: { name: string; lines: string[]; reason: string | null }[] = + [ + { + name: 'a bare trust dialog', + lines: TRUST_DIALOG_WITH_CARET, + reason: 'agent-trust-workspace' + }, + { + name: 'a trust dialog with an ordinary sentence in it', + lines: [ + 'Antigravity CLI 1.0.3', + 'This workspace has not been opened before.', + 'Do you trust the files in this folder?', + '1. Yes, I trust this folder', + '2. No, exit', + '>' + ], + reason: 'agent-trust-workspace' + }, + { + name: 'a trust dialog printing the folder on its own line', + lines: [ + 'Antigravity CLI 1.0.3', + 'Do you trust the files in this folder?', + '~/orca/workspaces/orca/agy-dispatch-issue', + '1. Yes', + '2. No', + '>' + ], + reason: 'agent-trust-workspace' + } + ] + + for (const dialog of LIVE_DIALOGS_UNDER_THE_HEADER) { + it(`reports ${dialog.name} drawn under the header and stays unready`, () => { + const waitText = waitTextFor(dialog.lines) + + expect(detectTerminalWaitBlockedReason(waitText)).toBe(dialog.reason) + expect(isKnownReadyPromptPreview(waitText)).toBe(false) + }) + } + + // Discriminating: the Gemini model line and caret satisfy readiness, so only the dialog sitting + // *below* them keeps this unready. Drop the ordering rule and this goes green-to-red. + it('keeps reporting a dialog that opens after a Gemini ready screen', () => { + const waitText = waitTextFor([ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Gemini 3.5 Flash (High)', + '~/orca/workspaces/orca/agy-dispatch-issue', + '>', + 'Permission required', + 'Allow once', + 'Reject' + ]) + + expect(isKnownReadyPromptPreview(waitText)).toBe(false) + expect(detectTerminalWaitBlockedReason(waitText)).toBe('agent-interactive-prompt') + }) + + // Discriminating: a stale dialog above a reprinted Gemini ready screen must stop being reported, + // which is the whole point of the dismissed-modal rule. + it('clears once a Gemini ready screen replaces the dialog', () => { + const waitText = waitTextFor([ + ...TRUST_DIALOG_WITH_CARET, + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Gemini 3.5 Flash (High)', + '>' + ]) + + expect(isKnownReadyPromptPreview(waitText)).toBe(true) + expect(detectTerminalWaitBlockedReason(waitText)).toBeNull() + }) + + // Characterization, not a guard: records the wedge this file has not fixed. An Antigravity user on + // a non-Gemini model has no 'gemini' line, so readiness never resolves and the wait times out. + // Flipping this to true is the goal of the follow-up, and needs a captured transcript first. + it('does not yet recognize a non-Gemini ready screen (known wedge)', () => { + const waitText = waitTextFor([ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Claude Sonnet 4.5 (High)', + '~/orca/workspaces/orca/agy-dispatch-issue', + '>' + ]) + + expect(isKnownReadyPromptPreview(waitText)).toBe(false) + }) + + // Ratchet, not a guard of today's code: these pass now only because none of them prints a 'gemini' + // model line. They exist so the next attempt to widen the model rule has to refuse them -- the + // reverted attempt accepted all five as ready on the strength of the account row alone (and an + // 'x@y.z' anywhere in the dialog body did just as well), and readiness is what gates typing the + // task prompt into the pane. A replacement must rest on positive evidence that the agent's input + // prompt is accepting input, not on absence-of-dialog plus an account row. + const SILENT_STARTUP_DIALOGS: { name: string; lines: string[] }[] = [ + { + name: 'an update banner', + lines: [ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'A new version is available', + '~/orca/workspaces/orca/agy-dispatch-issue', + 'Press enter to continue', + '>' + ] + }, + { + name: 'a sign-in dialog', + lines: [ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Sign in to continue', + '~/orca/workspaces/orca/agy-dispatch-issue', + '1. Open browser', + '2. Paste an API key', + '>' + ] + }, + { + name: 'a model picker', + lines: [ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Select a model', + '~/orca/workspaces/orca/agy-dispatch-issue', + '1. Claude Sonnet 4.5', + '2. GPT-5.1', + '>' + ] + }, + { + name: 'a privacy notice', + lines: [ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'We collect usage data to improve the product', + '~/orca/workspaces/orca/agy-dispatch-issue', + '1. Accept', + '2. Decline', + '>' + ] + }, + { + name: 'an onboarding theme picker', + lines: [ + 'Antigravity CLI 1.0.3', + 'user@example.com (Antigravity Business)', + 'Welcome! Choose a theme', + '~/orca/workspaces/orca/agy-dispatch-issue', + '1. Dark', + '2. Light', + '>' + ] + } + ] + + for (const dialog of SILENT_STARTUP_DIALOGS) { + it(`refuses ${dialog.name} whose wording names no blocked reason, account row and all`, () => { + const waitText = waitTextFor(dialog.lines) + + // No blocked-signal rule matches, so the ordering defense cannot reach these: readiness has to + // refuse them on its own or the orchestrator types into a live dialog. + expect(detectTerminalWaitBlockedReason(waitText)).toBeNull() + expect(isKnownReadyPromptPreview(waitText)).toBe(false) + }) + + it(`refuses ${dialog.name} that merely narrates an email address`, () => { + const waitText = waitTextFor([ + ...dialog.lines.slice(0, -1), + 'contact support@antigravity.dev for help', + '>' + ]) + + expect(isKnownReadyPromptPreview(waitText)).toBe(false) + }) + } +}) diff --git a/src/main/runtime/terminal-wait-detection.ts b/src/main/runtime/terminal-wait-detection.ts index ed957d1fe2d..cae25e8bfa6 100644 --- a/src/main/runtime/terminal-wait-detection.ts +++ b/src/main/runtime/terminal-wait-detection.ts @@ -231,11 +231,11 @@ function findBlockedSignalInLiveWindow( const candidates: { reason: RuntimeTerminalWaitBlockedReason; index: number }[] = [] const updateIndex = normalized.lastIndexOf('update available') if (updateIndex !== -1 && normalized.includes('press enter to continue', updateIndex)) { - candidates.push({ reason: 'codex-update-prompt', index: updateIndex }) + candidates.push({ reason: 'agent-update-prompt', index: updateIndex }) } const cwdIndex = normalized.lastIndexOf('choose working directory to') if (cwdIndex !== -1 && normalized.includes('press enter to continue', cwdIndex)) { - candidates.push({ reason: 'codex-cwd-prompt', index: cwdIndex }) + candidates.push({ reason: 'agent-cwd-prompt', index: cwdIndex }) } const modelMigrationIndex = normalized.lastIndexOf('codex just got an upgrade') if ( @@ -246,7 +246,8 @@ function findBlockedSignalInLiveWindow( } const hooksIndex = normalized.lastIndexOf('hooks need review') if (hooksIndex !== -1 && normalized.includes('press enter to confirm', hooksIndex)) { - candidates.push({ reason: 'codex-hooks-review-prompt', index: hooksIndex }) + // Why neutral: this matcher never inspects the agent -- 'hooks need review' is not Codex-only wording. + candidates.push({ reason: 'agent-hooks-review-prompt', index: hooksIndex }) } const trustIndex = Math.max( normalized.lastIndexOf('do you trust'), @@ -261,7 +262,8 @@ function findBlockedSignalInLiveWindow( trustSegment.includes('directory') || trustSegment.includes('repo')) ) { - candidates.push({ reason: 'codex-trust-workspace', index: trustIndex }) + // Why neutral: this matcher never inspects the agent -- every TUI agent ships a workspace-trust dialog. + candidates.push({ reason: 'agent-trust-workspace', index: trustIndex }) } const interactivePromptIndex = Math.max( normalized.lastIndexOf('press enter to confirm'), @@ -274,19 +276,22 @@ function findBlockedSignalInLiveWindow( interactivePromptIndex === -1 ? '' : normalized.slice(Math.max(0, interactivePromptIndex - 600), interactivePromptIndex + 200) - const hasCodexInteractiveContext = + // Why 'codex' only widens detection and never names the reason: the sole Codex evidence here is + // that word somewhere in 600 chars of scrollback, which an agent narrating about Codex satisfies + // on any pane -- enough to suspect a dialog, not enough to label a non-Codex user's pane. + const hasInteractiveDialogContext = interactivePromptContext.includes('codex') || interactivePromptContext.includes('permission') || interactivePromptContext.includes('sandbox') || interactivePromptContext.includes('trust') || interactivePromptContext.includes('hook') - if (interactivePromptIndex !== -1 && hasCodexInteractiveContext) { + if (interactivePromptIndex !== -1 && hasInteractiveDialogContext) { const contextStart = Math.max(0, interactivePromptIndex - 600) const hasSpecificPromptInContext = candidates.some( (candidate) => candidate.index >= contextStart && candidate.index <= interactivePromptIndex ) if (!hasSpecificPromptInContext) { - candidates.push({ reason: 'codex-interactive-prompt', index: interactivePromptIndex }) + candidates.push({ reason: 'agent-interactive-prompt', index: interactivePromptIndex }) } } const cursorApprovalIndex = findCursorApprovalPromptIndex(normalized) @@ -303,8 +308,13 @@ function findBlockedSignalInLiveWindow( permissionSegment.includes(choice) ).length if (decisionCount >= 2) { - // Why: preserve the existing remote receipt value for mixed-version clients. - candidates.push({ reason: 'codex-interactive-prompt', index: permissionPromptIndex }) + // Why neutral: an approval dialog with named choices identifies no agent; older hosts publish + // 'codex-interactive-prompt' here and clients alias the two. Rule 1 additive member -- + // remote-wire-compatibility.md names RuntimeTerminalWaitBlockedReason as Rule 1 because no + // consumer switches exhaustively on it. + // Why alias rather than drop the old spelling: preserve the existing remote receipt value for + // mixed-version clients -- an older host still publishes codex-* on this path. + candidates.push({ reason: 'agent-interactive-prompt', index: permissionPromptIndex }) } } return candidates.length > 0 diff --git a/src/main/runtime/unstopped-pty-verification.ts b/src/main/runtime/unstopped-pty-verification.ts index a2cd83b8eec..7cfd5907797 100644 --- a/src/main/runtime/unstopped-pty-verification.ts +++ b/src/main/runtime/unstopped-pty-verification.ts @@ -2,7 +2,7 @@ import type { IPtyProvider } from '../providers/types' import type { OrcaRuntimeService } from './orca-runtime' import { UNSTOPPED_PTY_DETAIL_SEPARATOR, - UNSTOPPED_PTY_LIVE_DETAIL_PREFIX, + STILL_LIVE_DETAIL_PREFIX, UNSTOPPED_PTY_REMOVAL_PREFIX } from '../../shared/worktree/removal' import { @@ -105,7 +105,7 @@ export function describeUnstoppedPtys( ): string { const detail = verdict.status === 'live' - ? `${UNSTOPPED_PTY_LIVE_DETAIL_PREFIX} ${verdict.ptyIds.join(', ')}` + ? `${STILL_LIVE_DETAIL_PREFIX} ${verdict.ptyIds.join(', ')}` : `could not verify these exited: ${failedPtyIds.join(', ')} (${verdict.reason})` return `${UNSTOPPED_PTY_REMOVAL_PREFIX} ${worktreeId}${UNSTOPPED_PTY_DETAIL_SEPARATOR}${detail}` } diff --git a/src/main/runtime/worktree-pty-host-fence.ts b/src/main/runtime/worktree-pty-host-fence.ts index 855371c4aef..a2a6099bf11 100644 --- a/src/main/runtime/worktree-pty-host-fence.ts +++ b/src/main/runtime/worktree-pty-host-fence.ts @@ -1,8 +1,14 @@ export type WorktreePtyHostFence = { + /** `null` is this machine; ABSENT is no fence at all, so every host matches. */ resolvedConnectionId?: string | null resolvedRuntimeEnvironmentId?: string } +/** + * Also fences the structured sweep, through `structuredSessionTeardownHostId`, which reuses this + * exact type so the two cannot drift. That helper narrows ABSENT to local — the one deliberate + * difference, documented where it is made. + */ export function worktreePtyBelongsToHost( ptyId: string, connectionId: string | null | undefined, diff --git a/src/main/runtime/worktree-teardown.ts b/src/main/runtime/worktree-teardown.ts index 82fa055cc3a..84275b40eed 100644 --- a/src/main/runtime/worktree-teardown.ts +++ b/src/main/runtime/worktree-teardown.ts @@ -15,10 +15,16 @@ import { } from './worktree-pty-surface-sweeps' import { closeStructuredSessionsForWorktree, - describeLiveStructuredSessions, - listLiveStructuredSessionsForWorktree + createStructuredSweepProgress, + describeUnclosedStructuredSessions, + listLiveStructuredSessionsForWorktree, + unclosedStructuredSessions } from './structured-session-worktree-teardown' -import { createWorktreeSweepTracker, settleSweepsForForcedRemoval } from './forced-sweep-settlement' +import { + createWorktreeSweepTracker, + settleSweepsForForcedRemoval, + type WorktreeSweepTracker +} from './forced-sweep-settlement' import { describeError, describeFailedPtySweep, @@ -57,7 +63,7 @@ export type WorktreeTeardownResult = { runtimeStopped: number providerStopped: number registryStopped: number - /** Structured agent sessions closed by the force path; absent when none were found. */ + /** Structured agent sessions this teardown closed; absent when it closed none. */ structuredStopped?: number } @@ -99,12 +105,18 @@ export async function killAllProcessesForWorktree( const deadlineError = new Error( `${WORKTREE_TEARDOWN_TIMEOUT_PREFIX} ${worktreeId}. ${WORKTREE_TEARDOWN_FORCE_HINT}` ) - // FIRST, and before a single PTY sweep starts: a structured agent session is registered on none - // of the three surfaces below, so all three answered zero and removal deleted the checkout out - // from under a running provider child. Refusing costs nothing when there are none, and the check - // is synchronous, so a destructive removal fails fast instead of after the whole sweep budget. - const structuredStopped = await sweepStructuredSessions(worktreeId, deps, deadline, deadlineError) const sweeps = createWorktreeSweepTracker() + // ISSUED first, before a single PTY is touched: a structured agent session is registered on none + // of the three surfaces below, so all three answered zero and removal deleted the checkout out + // from under a running provider child. Asking the agent plane ahead of the terminal plane also + // keeps an intentional stop from reading as a failed process exit. + // + // Not AWAITED first, though. Its close is serial and each one waits on a provider round trip, so + // awaiting here would spend the shared budget before a single PTY was asked — and the sweeps + // would then report a timeout for a stop they never attempted. It is joined below, ahead of the + // PTY verdict, so a structured refusal still outranks one. + const structuredSweep = sweepStructuredSessions(worktreeId, deps, deadline, sweeps) + void structuredSweep.catch(() => undefined) const stopAttempts = new Map>() const stopPty = ( ptyId: string, @@ -196,6 +208,7 @@ export async function killAllProcessesForWorktree( for (const sweep of [runtimeSweep, providerSweep, registrySweep]) { void sweep.catch(() => undefined) } + const structuredStopped = await structuredSweep let runtimeResult: { stopped: number } let providerStopped: number let registryStopped: number @@ -207,7 +220,10 @@ export async function killAllProcessesForWorktree( deadlineError ) if (forced.incomplete) { - return forced.stopped + // Carries the structured count out too: this early return skips the PTY verdict, not the + // sweep that already closed a user's chats, and dropping it makes the log say `structured=0` + // for a removal that closed some. + return { ...forced.stopped, ...(structuredStopped > 0 ? { structuredStopped } : {}) } } runtimeResult = { stopped: forced.stopped.runtimeStopped } providerStopped = forced.stopped.providerStopped @@ -277,59 +293,84 @@ export async function killAllProcessesForWorktree( } /** - * The fourth sweep: structured agent sessions bound to this worktree. + * The fourth sweep: structured agent sessions bound to this worktree, on this host. * - * Refuses rather than auto-closing on the ordinary destructive path. `worktree rm` is the verb - * that deletes a user's work, and a running agent session is exactly the thing they would want to - * be told about before it goes — the same bargain the unstopped-PTY gate already strikes, using - * the same `--force` escape hatch. Force closes them properly instead of orphaning a child against - * a `cwd` that is about to disappear. + * Stops first and refuses only on unproven stops, which is the bargain the unstopped-PTY gate + * actually strikes: that gate kills every PTY — a terminal running an agent included — and refuses + * only for the ones whose exit it could not then verify. Refusing merely because a session is + * attached made an idle chat, which the user is done with, harder to delete than a terminal running + * the same agent. Attachment is lease state, not work in flight, so it was never the right proxy. * - * Two callers participate, for different reasons. A proof-requiring removal (`requirePhysicalStop`) - * refuses, then closes under force. A folder-workspace removal (`closeStructuredSessions`) closes - * best-effort without refusing: it shares its root so no checkout vanishes under the child, and one - * of those paths is a never-throw forget that a refusal would wedge. Reconciliation sweeps set - * neither — they repair state, delete nothing, and must never close a session. + * Two callers participate. A proof-requiring removal (`requirePhysicalStop`) refuses when a close + * does not settle, so nothing deletes a checkout out from under a child that is still there. A + * folder-workspace removal (`closeStructuredSessions`) never refuses: it shares its root so no + * checkout vanishes under the child, and every one of those call sites discards a rejection, so a + * refusal there would be words nobody reads. Reconciliation sweeps set neither — they repair state, + * delete nothing, and must never close a session. */ async function sweepStructuredSessions( worktreeId: string, deps: WorktreeTeardownDeps, deadline: number, - deadlineError: Error + sweeps: WorktreeSweepTracker ): Promise { if (!deps.requirePhysicalStop && !deps.closeStructuredSessions) { return 0 } - const live = listLiveStructuredSessionsForWorktree(worktreeId) + // `deps` carries the same two host fields the PTY sweeps fence on, and a `repoId::path` id names + // a different workspace on every host — so an unfenced list would close a live chat belonging to + // an SSH or paired-runtime copy of the id being removed here. + const live = listLiveStructuredSessionsForWorktree(worktreeId, deps) if (live.length === 0) { return 0 } + // Raced against the same sweep budget every PTY surface is bounded by, because `host.close` + // awaits a provider round trip whose own eviction steps are each bounded well past this budget. + // + // Deliberately NOT fail-closed, unlike the PTY sweeps: their timeout sentinel carries the PTY + // timeout prefix, which the desktop classifier reads as a TERMINAL failure — so a wedged session + // close would refuse in terminal wording, and refuse identically again under the Force Delete + // that is meant to clear it (#11960). A close that ran out of time is a session this removal + // could not confirm closed, which is exactly what the branch below already words. Tracked so a + // forced removal still waits out the abandoned-sweep grace before it deletes files. + // + // The verdict is read off `progress`, which the serial loop fills as it goes, rather than off + // this call's result: the deadline can land mid-loop, and a fallback assembled here could only + // guess — it named every session, including the ones already closed, and reported zero closes. + const progress = createStructuredSweepProgress(live) + await settleBeforeDeadline( + sweeps.track(() => + closeStructuredSessionsForWorktree(progress, deadline, { + ...(deps.runtime ? { runtime: deps.runtime } : {}), + // The only shape of removal that can leave this workspace — and its chat tabs — in place. + mayRefuse: Boolean(deps.requirePhysicalStop) && !deps.allowUnverifiedStop + }) + ), + undefined, + deadline + ) + const closed = progress.closed + const unstopped = unclosedStructuredSessions(progress) + if (unstopped.length === 0) { + return closed + } // Only a proof-requiring removal may refuse. A folder-workspace removal shares its root, so no // checkout disappears under the child — the harm is a session left pointing at a workspace Orca - // has forgotten — and one of those paths is a never-throw forget, which a refusal would wedge. + // has forgotten — and every one of those callers discards a rejection anyway. if (deps.requirePhysicalStop && !deps.allowUnverifiedStop) { // The prefix is what the desktop classifier matches on; without it the toast shows raw CLI // wording and hides the Force Delete button — the #11960 dead end this file already documents. throw new Error( - `${RUNNING_AGENT_SESSION_REMOVAL_PREFIX} ${worktreeId}${UNSTOPPED_PTY_DETAIL_SEPARATOR}${describeLiveStructuredSessions(live)}. ${WORKTREE_TEARDOWN_FORCE_HINT}` + `${RUNNING_AGENT_SESSION_REMOVAL_PREFIX} ${worktreeId}${UNSTOPPED_PTY_DETAIL_SEPARATOR}${describeUnclosedStructuredSessions(unstopped)}. ${WORKTREE_TEARDOWN_FORCE_HINT}` ) } - // Raced against the same sweep budget every PTY surface is bounded by: `host.close` awaits a - // provider round trip, and a wedged one would otherwise hang `worktree rm --force` forever with - // no timeout error at all. On expiry the force path reports the timeout exactly as the PTY - // sweeps do rather than proceeding as if the sessions had closed. - const { closed, unstopped } = await settleBeforeDeadline( - () => closeStructuredSessionsForWorktree(worktreeId, deps.runtime), - { closed: 0, unstopped: live }, - deadline, - deadlineError + // Force is the documented escape hatch, so removal continues — but say so, because the child + // outliving its `cwd` is the failure this sweep exists to make visible. Carries the verdict + // verbatim, like the unstopped-PTY warn above: this line is the only record a forced removal + // leaves, and appending "still attached" asserted the live verdict over sessions the sweep had + // just said it could not confirm either way. + console.warn( + `[worktree-teardown] forcing removal of ${worktreeId}${UNSTOPPED_PTY_DETAIL_SEPARATOR}${describeUnclosedStructuredSessions(unstopped)}` ) - if (unstopped.length > 0) { - // Force is the documented escape hatch, so removal continues — but say so, because the child - // outliving its `cwd` is the failure this sweep exists to make visible. - console.warn( - `[worktree-teardown] forcing removal of ${worktreeId} with ${describeLiveStructuredSessions(unstopped)} still attached` - ) - } return closed } diff --git a/src/main/skills/skill-root-file-walk.test.ts b/src/main/skills/skill-root-file-walk.test.ts index ad84eced6b6..81e9b167fcc 100644 --- a/src/main/skills/skill-root-file-walk.test.ts +++ b/src/main/skills/skill-root-file-walk.test.ts @@ -66,10 +66,15 @@ describe('findSkillFiles', () => { expect(await findSkillFiles(root, 4)).toEqual([join(edge, 'SKILL.md')]) expect(statPaths).toEqual([]) - expect(await findSkillFiles(root, 5)).toEqual([ - join(edge, 'SKILL.md'), - join(edge, 'link00', 'SKILL.md') - ]) + // Why not a fixed array: `readdir` order is filesystem-dependent, and both + // the result order and which link survives dedup follow it. NTFS enumerates + // its name index alphabetically, so `link00` precedes `SKILL.md` on Windows + // and follows it on APFS/ext4. All 32 links share one realpath, so the + // visited set collapses them to a single entry beside the real file. + const withinDepth = await findSkillFiles(root, 5) + expect(withinDepth).toContain(join(edge, 'SKILL.md')) + expect(withinDepth.filter((path) => /[\\/]link\d{2}[\\/]SKILL\.md$/.test(path))).toHaveLength(1) + expect(withinDepth).toHaveLength(2) expect(statPaths).toHaveLength(32) }) diff --git a/src/main/startup/main-process-push-startup.ts b/src/main/startup/main-process-push-startup.ts new file mode 100644 index 00000000000..6d1b9fda1bd --- /dev/null +++ b/src/main/startup/main-process-push-startup.ts @@ -0,0 +1,32 @@ +import { getOrcaPushGatewayUrl } from '../orca-profiles/profile-cloud-auth-config' +import { DesktopPushService } from '../runtime/push/desktop-push-service' +import type { OrcaRuntimeService } from '../runtime/orca-runtime' +import type { OrcaRuntimeRpcServer } from '../runtime/runtime-rpc' +import { mainProcessState as state } from './main-process-state' + +// Why: deliberately not gated on cloud sign-in like the relay is — the push gateway +// authenticates with the host keypair, so an accountless host registers phones on +// exactly the same path. The runtime is read from shared state because both launch +// modes have already stored it there; threading it as a parameter would push the +// launch module past its line budget for no gain. +export function startDesktopPushService(runtimeRpc: OrcaRuntimeRpcServer): void { + const runtime: OrcaRuntimeService | null = state.runtime + if (!runtime) { + console.warn('[push] Background push startup skipped: runtime not started') + return + } + try { + const pushService = DesktopPushService.create({ + runtime, + runtimeRpc, + gatewayUrl: getOrcaPushGatewayUrl() + }) + pushService?.start() + state.desktopPushService = pushService + } catch (error) { + console.warn( + '[push] Background push startup unavailable:', + error instanceof Error ? error.message : String(error) + ) + } +} diff --git a/src/main/startup/main-process-quit.ts b/src/main/startup/main-process-quit.ts index a4149e13ba7..a136981c873 100644 --- a/src/main/startup/main-process-quit.ts +++ b/src/main/startup/main-process-quit.ts @@ -105,6 +105,8 @@ function installWillQuitHandler(): void { if (!quitTeardownStartGate.tryStart(event)) { return } + // A renderer can veto before-quit; push must survive until quit is committed. + state.desktopPushService?.stop() state.unsubscribeSystemResumeBroadcast?.() state.unsubscribeSystemResumeBroadcast = null // Why: renderer guards can still cancel before this committed phase; `log stream` must survive those vetoes. diff --git a/src/main/startup/main-process-relay-status.ts b/src/main/startup/main-process-relay-status.ts new file mode 100644 index 00000000000..1dd254df8b3 --- /dev/null +++ b/src/main/startup/main-process-relay-status.ts @@ -0,0 +1,18 @@ +import type { MobileRelayStatusDetail } from '../../shared/mobile-relay-status' +import { mainProcessState as state } from './main-process-state' + +export function getDesktopRelayStatus(): MobileRelayStatusDetail { + return { + status: state.desktopRelayStatus, + ...(state.desktopRelayCellUrl === undefined ? {} : { cellUrl: state.desktopRelayCellUrl }) + } +} + +export function publishDesktopRelayStatus( + status: MobileRelayStatusDetail['status'], + cellUrl?: string +): void { + state.desktopRelayStatus = status + state.desktopRelayCellUrl = cellUrl + state.mainWindow?.webContents.send('mobile:relayStatusChanged', getDesktopRelayStatus()) +} diff --git a/src/main/startup/main-process-runtime-launch.ts b/src/main/startup/main-process-runtime-launch.ts index 4ec2bd7bfac..fdedf310cbb 100644 --- a/src/main/startup/main-process-runtime-launch.ts +++ b/src/main/startup/main-process-runtime-launch.ts @@ -13,7 +13,7 @@ import { LocalPtyProvider } from '../providers/local-pty-provider' import { HEADLESS_RUNTIME_WINDOW_ID } from '../../shared/runtime-types' import { OffscreenBrowserBackend } from '../browser/offscreen-browser-backend' import { browserManager } from '../browser/browser-manager' -import type { MobileRelayStatusDetail } from '../../shared/mobile-relay-status' +import { getDesktopRelayStatus, publishDesktopRelayStatus } from './main-process-relay-status' import { DesktopRelayService } from '../runtime/relay/desktop-relay-service' import { getServeOptions, getBundledWebClientRoot, printServeReady } from './main-process-serve' import { @@ -36,6 +36,7 @@ import { CliInstaller } from '../cli/cli-installer' import { installLinuxBareOrcaDispatcher } from '../cli/linux-bare-orca-dispatcher' import { scheduleAllPendingHistoryTreeRemovals } from '../terminal-history-deletion' import { triggerStartupNotificationRegistration } from '../ipc/startup-notification-registration' +import { startDesktopPushService } from './main-process-push-startup' import { mainProcessState as state } from './main-process-state' import { logStartupMilestone } from './startup-diagnostics' @@ -92,10 +93,7 @@ function installRuntimeRpc( }) state.runtimeRpc = runtimeRpc registerMobileHandlers(runtimeRpc, { - getRelayStatus: () => ({ - status: state.desktopRelayStatus, - ...(state.desktopRelayCellUrl === undefined ? {} : { cellUrl: state.desktopRelayCellUrl }) - }), + getRelayStatus: getDesktopRelayStatus, consumePendingUnpairedDeviceAuthFailure: (webContentsId) => { if ( !state.mainWindow || @@ -162,6 +160,9 @@ async function launchServeMode( console.error('[runtime] Failed to start headless RPC transport:', error) throw error }) + // Why: a phone paired to a headless host still registers and unregisters its token; + // it simply never receives a push, because nothing dispatches notifications here. + startDesktopPushService(runtimeRpc) settleDesktopActivation() // Why: every attempt must reach app.quit(); a page beforeunload can veto an earlier signal. registerServeSignalHandlers(process, () => app.quit()) @@ -245,6 +246,9 @@ async function launchDesktopMode( // fetcher until the persisted proxy lands, so this only has to keep the launch phase itself // ordered ahead of the relay — it must not gate the renderer. await state.initialProxyApplicationReady + // Why after the proxy await: the push gateway client is an app-owned fetcher, so it must not + // issue its first request ahead of the persisted proxy. + startDesktopPushService(runtimeRpc) const cloudAuth = getOrcaCloudAuthConfig() if (cloudAuth.configured) { try { @@ -253,14 +257,7 @@ async function launchDesktopMode( userDataPath: getProfileUserDataPath(), appVersion: app.getVersion(), runtimeRpc, - onStatus: (status, cellUrl) => { - state.desktopRelayStatus = status - state.desktopRelayCellUrl = cellUrl - state.mainWindow?.webContents.send('mobile:relayStatusChanged', { - status, - ...(cellUrl === undefined ? {} : { cellUrl }) - } satisfies MobileRelayStatusDetail) - } + onStatus: publishDesktopRelayStatus }) state.desktopRelayService = relayService runtimeRpc.setMobileRelayPairingProvider({ diff --git a/src/main/startup/main-process-runtime-service.ts b/src/main/startup/main-process-runtime-service.ts index 1b7f69213ad..8af5630e02b 100644 --- a/src/main/startup/main-process-runtime-service.ts +++ b/src/main/startup/main-process-runtime-service.ts @@ -87,6 +87,12 @@ export function initializeMainProcessRuntime(): OrcaRuntimeService { // Why: worktree.ps pulls hook-reported agent status (same source as the desktop sidebar) at query time so mobile shows the same agents. getAgentStatusSnapshot: () => agentHookServer.getStatusSnapshot().filter((entry) => entry.providerSessionOnly !== true), + // Why: structured chats have no hooks, so the host writes their projections here itself; the + // snapshot above then lists them for the CLI and mobile without a second store. + structuredAgentStatusSink: { + publish: (summary) => agentHookServer.ingestStructuredStatus(summary), + forget: (sessionId) => agentHookServer.dropStructuredStatus(sessionId) + }, // Why captured rather than resolved at read: the fleet snapshot remints cached rows on every // read, so a row observed under one process otherwise acquires whatever the pane owns now. readObservedAgentStatusPaneIdentity: (paneKey) => observedPaneIdentities.read(paneKey), diff --git a/src/main/startup/main-process-state.ts b/src/main/startup/main-process-state.ts index d69518d710a..17a5eb427c6 100644 --- a/src/main/startup/main-process-state.ts +++ b/src/main/startup/main-process-state.ts @@ -13,6 +13,7 @@ import type { OrcaRuntimeService } from '../runtime/orca-runtime' import type { RateLimitService } from '../rate-limits/service' import type { OrcaRuntimeRpcServer } from '../runtime/runtime-rpc' import type { DesktopRelayService } from '../runtime/relay/desktop-relay-service' +import type { DesktopPushService } from '../runtime/push/desktop-push-service' import type { StarNagService } from '../star-nag/service' import type { AgentAwakeService } from '../agent-awake-service' import type { CrashReportStore } from '../crash-reporting/crash-report-store' @@ -65,6 +66,7 @@ export const mainProcessState = { runtimeRpc: null as OrcaRuntimeRpcServer | null, serveReadinessPublisher: new ServeReadinessPublisher(), desktopRelayService: null as DesktopRelayService | null, + desktopPushService: null as DesktopPushService | null, desktopRelayStatus: 'offline' as RelayBrokerStatus, desktopRelayCellUrl: undefined as string | undefined, pendingUnpairedDeviceAuthFailure: false, diff --git a/src/main/startup/main-window-agent-status.ts b/src/main/startup/main-window-agent-status.ts index 3b2ba9cbd71..0f10a373bf9 100644 --- a/src/main/startup/main-window-agent-status.ts +++ b/src/main/startup/main-window-agent-status.ts @@ -43,11 +43,17 @@ export function installMainWindowAgentStatusListeners(options: MainWindowAgentSt promptInteractionKey, restoredUnconfirmed, observation, - isReplay + isReplay, + structuredHost }) => { if (state.mainWindow?.isDestroyed()) { return } + // Why: the renderer still derives structured rows from its own feed subscription; forwarding + // these too would give one pane key two writers until that bridge is retired. + if (structuredHost) { + return + } if (providerSessionOnly) { // Why: session_start just refreshes durable resume identity while Pi is idle; forward it without titles, telemetry, or status UI. state.mainWindow?.webContents.send('agentStatus:set', { diff --git a/src/main/startup/main-window-structured-status-filter.test.ts b/src/main/startup/main-window-structured-status-filter.test.ts new file mode 100644 index 00000000000..a3cd1bfdf0a --- /dev/null +++ b/src/main/startup/main-window-structured-status-filter.test.ts @@ -0,0 +1,90 @@ +// The renderer half of the half-migration seam. +// +// Until PR 2 retires `StructuredAgentSessionStatusBridge`, the renderer writes structured rows +// itself. Main forwarding them too would give one pane key two writers, so the window listener +// drops them — a filter nothing else asserts, which makes deleting it green everywhere. + +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { EnrichedAgentHookEventPayload } from '../agent-hooks/server' + +const hooks = vi.hoisted(() => ({ + listener: null as ((payload: EnrichedAgentHookEventPayload) => void) | null +})) + +vi.mock('electron', () => ({ + app: { getPath: () => '', on: vi.fn(), isReady: () => true } +})) +vi.mock('../agent-hooks/server', () => ({ + agentHookServer: { + setListener: (listener: ((payload: EnrichedAgentHookEventPayload) => void) | null) => { + hooks.listener = listener + }, + setPaneStatusClearListener: vi.fn() + } +})) +vi.mock('../agent-hooks/migration-unsupported-pty-state', () => ({ + setMigrationUnsupportedPtyListener: vi.fn() +})) +vi.mock('../window/dashboard-popout-window', () => ({ + getDashboardPopoutWindow: () => null +})) +vi.mock('./synthetic-title-runtime', () => ({ + driveSyntheticTitleFromHook: vi.fn(), + shouldSuppressCodexAutoApprovalSyntheticTitleFromHook: () => false, + stopAllSyntheticTitleSpinners: vi.fn() +})) + +import { installMainWindowAgentStatusListeners } from './main-window-agent-status' +import { mainProcessState } from './main-process-state' + +const sent: { channel: string; event: { paneKey: string } }[] = [] + +function statusPayload( + over: Partial +): EnrichedAgentHookEventPayload { + return { + paneKey: 'pane-1', + tabId: 'tab-1', + worktreeId: 'repo::/wt', + connectionId: null, + receivedAt: 1, + stateStartedAt: 1, + payload: { state: 'working', prompt: 'ship it', agentType: 'codex' }, + ...over + } as EnrichedAgentHookEventPayload +} + +beforeEach(() => { + sent.length = 0 + hooks.listener = null + mainProcessState.runtime = null + mainProcessState.mainWindow = { + isDestroyed: () => false, + webContents: { + send: (channel: string, event: { paneKey: string }) => sent.push({ channel, event }) + } + } as unknown as typeof mainProcessState.mainWindow + installMainWindowAgentStatusListeners({ + window: mainProcessState.mainWindow!, + maybeAutoRenameBranchOnFirstWork: vi.fn(), + onRecordAgentState: vi.fn() + }) +}) + +describe('the main-window agent-status listener', () => { + it('forwards a hook row but never a structured one', () => { + expect(hooks.listener).not.toBeNull() + + hooks.listener!(statusPayload({ paneKey: 'hook-pane' })) + hooks.listener!( + statusPayload({ + paneKey: 'structured-agent-session-s1:leaf', + structuredHost: 'owned' + }) + ) + + expect(sent.map((entry) => `${entry.channel}:${entry.event.paneKey}`)).toEqual([ + 'agentStatus:set:hook-pane' + ]) + }) +}) diff --git a/src/preload/api/notifications-bridge.ts b/src/preload/api/notifications-bridge.ts index 70c64d4ce0d..210352ff981 100644 --- a/src/preload/api/notifications-bridge.ts +++ b/src/preload/api/notifications-bridge.ts @@ -37,6 +37,8 @@ function disposeCachedNotificationSound(): void { } export const notificationsApi = { + getDesktopAwayState: (): Promise => + ipcRenderer.invoke('notifications:getDesktopAwayState'), dispatch: (args: Record): Promise => ipcRenderer.invoke('notifications:dispatch', args), dismiss: (ids: string[]): Promise => diff --git a/src/preload/api/os-permission-api.ts b/src/preload/api/os-permission-api.ts index 8718cfc29a5..f2033c2fe1c 100644 --- a/src/preload/api/os-permission-api.ts +++ b/src/preload/api/os-permission-api.ts @@ -20,6 +20,7 @@ import type { } from '../../shared/notification-settings-types' export type NotificationsApi = { + getDesktopAwayState: () => Promise dispatch: (args: NotificationDispatchRequest) => Promise dismiss: (ids: string[]) => Promise openSystemSettings: () => Promise diff --git a/src/preload/api/runtime-api.ts b/src/preload/api/runtime-api.ts index f7f553ec2c0..7fdf229dd88 100644 --- a/src/preload/api/runtime-api.ts +++ b/src/preload/api/runtime-api.ts @@ -1,3 +1,4 @@ +import type { RuntimeHostStatusSnapshot } from '../../shared/runtime-host-status' import type { RuntimeBrowserDriverState, RuntimeRendererSyncWindowGraph, @@ -77,6 +78,8 @@ export type RuntimeApi = { ) => () => void } runtimeEnvironments: { + getStatusSnapshots: () => Promise + onStatusChanged: (callback: (snapshot: RuntimeHostStatusSnapshot) => void) => () => void list: () => Promise addFromPairingCode: (args: { name: string diff --git a/src/preload/api/runtime-environments-bridge.ts b/src/preload/api/runtime-environments-bridge.ts index ddfa498dc74..63c31df52f8 100644 --- a/src/preload/api/runtime-environments-bridge.ts +++ b/src/preload/api/runtime-environments-bridge.ts @@ -1,4 +1,8 @@ import { ipcRenderer } from 'electron' +import { + RUNTIME_HOST_STATUS_CHANNEL, + type RuntimeHostStatusSnapshot +} from '../../shared/runtime-host-status' import type { VerifyAndAddRuntimeEnvironmentResult } from '../../shared/remote-pairing-verification' import type { RuntimeStatus } from '../../shared/runtime-types' import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' @@ -12,6 +16,16 @@ import { import type { PreloadApi } from '../api-types' export const runtimeEnvironmentsApi = { + getStatusSnapshots: (): Promise => + ipcRenderer.invoke('runtimeEnvironments:getStatusSnapshots'), + onStatusChanged: (callback: (snapshot: RuntimeHostStatusSnapshot) => void): (() => void) => { + const listener = ( + _event: Electron.IpcRendererEvent, + snapshot: RuntimeHostStatusSnapshot + ): void => callback(snapshot) + ipcRenderer.on(RUNTIME_HOST_STATUS_CHANNEL, listener) + return () => ipcRenderer.removeListener(RUNTIME_HOST_STATUS_CHANNEL, listener) + }, list: (): Promise => ipcRenderer.invoke('runtimeEnvironments:list'), addFromPairingCode: (args: { diff --git a/src/renderer/src/app-shell/AppRootSurfaces.tsx b/src/renderer/src/app-shell/AppRootSurfaces.tsx index 202c99df4d2..31de9b1ce8d 100644 --- a/src/renderer/src/app-shell/AppRootSurfaces.tsx +++ b/src/renderer/src/app-shell/AppRootSurfaces.tsx @@ -1,3 +1,4 @@ +import { NotificationCardStack } from '../components/NotificationCardStack' import { Suspense } from 'react' import { lazyWithRetry as lazy } from '@/lib/lazy-with-retry' import { translate } from '@/i18n/i18n' @@ -58,6 +59,11 @@ const SshPassphraseDialog = lazy(() => const UpdateCard = lazy(() => import('../components/UpdateCard').then((module) => ({ default: module.UpdateCard })) ) +const UnexpectedSignoutCard = lazy(() => + import('../components/UnexpectedSignoutCard').then((module) => ({ + default: module.UnexpectedSignoutCard + })) +) const RemoteServerUpdateDialog = lazy( () => import('../components/settings/RemoteServerUpdateDialog') ) @@ -273,16 +279,23 @@ export function AppRootSurfaces(props: { ) : null} - {shouldMountUpdateCard ? ( + + {shouldMountUpdateCard ? ( + + + + + + ) : null} - - + + - ) : null} - - - + + + + diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.tsx index d17bb8ba56d..7145bed732c 100644 --- a/src/renderer/src/components/NewWorkspaceComposerCard.tsx +++ b/src/renderer/src/components/NewWorkspaceComposerCard.tsx @@ -242,17 +242,11 @@ export default function NewWorkspaceComposerCard( selector: action.environmentId, timeoutMs: 15_000 }) - const runtimeStatus = unwrapRuntimeRpcResult(response) - useAppStore.getState().setRuntimeEnvironmentStatus(action.environmentId, { - status: runtimeStatus, - checkedAt: Date.now() - }) + unwrapRuntimeRpcResult(response) + await useAppStore.getState().readRuntimeHostStatusSnapshots() } catch (error) { if (action.kind === 'runtime') { - useAppStore.getState().setRuntimeEnvironmentStatus(action.environmentId, { - status: null, - checkedAt: Date.now() - }) + await useAppStore.getState().readRuntimeHostStatusSnapshots() } toast.error( error instanceof Error diff --git a/src/renderer/src/components/NotificationCardStack.tsx b/src/renderer/src/components/NotificationCardStack.tsx new file mode 100644 index 00000000000..9d3d5cddf3c --- /dev/null +++ b/src/renderer/src/components/NotificationCardStack.tsx @@ -0,0 +1,9 @@ +import type { ReactNode } from 'react' + +export function NotificationCardStack({ children }: { children: ReactNode }): React.JSX.Element { + return ( +
+ {children} +
+ ) +} diff --git a/src/renderer/src/components/StarNagCard.tsx b/src/renderer/src/components/StarNagCard.tsx index f0e184143fc..0bf3312463e 100644 --- a/src/renderer/src/components/StarNagCard.tsx +++ b/src/renderer/src/components/StarNagCard.tsx @@ -2,7 +2,6 @@ import { useCallback, useEffect, useState } from 'react' import { ExternalLink, Star, X } from 'lucide-react' import { Card } from './ui/card' import { Button } from './ui/button' -import { useAppStore } from '../store' import { useMountedRef } from '@/hooks/useMountedRef' import { translate } from '@/i18n/i18n' @@ -25,12 +24,6 @@ export function StarNagCard(): React.JSX.Element | null { const [busy, setBusy] = useState(false) const [mode, setMode] = useState('gh') const mountedRef = useMountedRef() - // Why: UpdateCard lives at the same bottom-right slot. When it is visible - // (any non-idle / non-not-available state), stack the star-nag card above - // it instead of overlapping — we must not cover a pending update prompt - // because that's a higher-priority action. - const updateStatus = useAppStore((s) => s.updateStatus) - const updateCardVisible = updateStatus.state !== 'idle' && updateStatus.state !== 'not-available' useEffect(() => { const unsubscribeShow = window.api.starNag.onShow((payload) => { @@ -147,15 +140,7 @@ export function StarNagCard(): React.JSX.Element | null { } return ( -
+
diff --git a/src/renderer/src/components/UnexpectedSignoutCard.tsx b/src/renderer/src/components/UnexpectedSignoutCard.tsx new file mode 100644 index 00000000000..ee0e8949ec3 --- /dev/null +++ b/src/renderer/src/components/UnexpectedSignoutCard.tsx @@ -0,0 +1,269 @@ +import { useEffect, useRef, useState } from 'react' +import { BookOpen, ChevronDown, CircleUserRound, Files, Smartphone, X } from 'lucide-react' +import { useAppStore } from '../store' +import { translate } from '@/i18n/i18n' +import { cn } from '@/lib/utils' +import { Button } from './ui/button' +import { Card } from './ui/card' +import { Collapsible, CollapsibleContent, CollapsibleTrigger } from './ui/collapsible' +import { shouldShowUnexpectedSignoutCard } from './unexpected-signout/unexpected-signout-visibility' + +function readPreviewFlag(): boolean { + if (!import.meta.env.DEV) { + return false + } + try { + if (new URLSearchParams(window.location.search).get('showSignoutCard') === '1') { + return true + } + return window.localStorage.getItem('orca-debug-show-signout-card') === '1' + } catch { + return false + } +} + +function FeatureRow({ + icon: Icon, + title, + description +}: { + icon: typeof Files + title: string + description: string +}): React.JSX.Element { + return ( +
+ +
+

{title}

+

{description}

+
+
+ ) +} + +export function UnexpectedSignoutCard(): React.JSX.Element | null { + const authStatus = useAppStore((s) => s.orcaProfileAuthStatus) + const persistedUIReady = useAppStore((s) => s.persistedUIReady) + const persistedDismissedVersion = useAppStore((s) => s.dismissedUnexpectedSignoutVersion) + const dismissedVersions = useAppStore((s) => s.unexpectedSignoutDismissedVersions) + const dismissForVersion = useAppStore((s) => s.dismissUnexpectedSignoutCard) + const connecting = useAppStore((s) => s.orcaProfileConnecting) + const connect = useAppStore((s) => s.connectCurrentOrcaProfile) + const [appVersion, setAppVersion] = useState(null) + const [authRefreshReady, setAuthRefreshReady] = useState(false) + const [expanded, setExpanded] = useState(false) + const [preview] = useState(readPreviewFlag) + const [previewDismissed, setPreviewDismissed] = useState(false) + const reconnectingProfile = useRef(null) + + useEffect(() => { + let cancelled = false + let attempts = 0 + const refresh = (): void => { + attempts += 1 + void useAppStore + .getState() + .fetchOrcaProfileAuthStatus() + .then((status) => { + if (cancelled) { + return + } + if (status != null) { + setAuthRefreshReady(true) + } else if (attempts < 3) { + window.setTimeout(refresh, 500) + } + }) + } + refresh() + return () => { + cancelled = true + } + }, []) + + useEffect(() => { + let cancelled = false + void window.api.updater + .getVersion() + .then((version) => { + if (!cancelled) { + setAppVersion(version) + } + }) + .catch(() => { + if (!cancelled) { + setAppVersion(null) + } + }) + return () => { + cancelled = true + } + }, []) + + const dismissedVersion = + appVersion && dismissedVersions.includes(appVersion) ? appVersion : persistedDismissedVersion + const eligible = shouldShowUnexpectedSignoutCard({ + authStatus, + persistedUIReady, + appVersion, + dismissedVersion + }) + + const visible = preview ? persistedUIReady && !previewDismissed : authRefreshReady && eligible + + // Observe recovery independently of visibility and asynchronous version/hydration reads. + useEffect(() => { + if (preview || !authRefreshReady) { + return + } + if (authStatus?.state === 'reconnect-required' && authStatus.configured && authStatus.cloud) { + reconnectingProfile.current = authStatus.activeProfileId + } else if (authStatus?.state === 'connected') { + if ( + reconnectingProfile.current === authStatus.activeProfileId && + persistedUIReady && + appVersion + ) { + reconnectingProfile.current = null + if (dismissedVersion !== appVersion) { + dismissForVersion(appVersion) + } + } + } else { + reconnectingProfile.current = null + } + }, [ + preview, + authRefreshReady, + authStatus, + persistedUIReady, + appVersion, + dismissedVersion, + dismissForVersion + ]) + + if (!visible) { + return null + } + + const email = authStatus?.cloud?.email?.trim() || null + const canConnect = authStatus?.configured === true + + const handleDismiss = (): void => { + if (preview) { + setPreviewDismissed(true) + } else if (appVersion) { + dismissForVersion(appVersion) + } + } + + return ( +
+ +
+
+
+ +

+ {translate( + 'auto.components.UnexpectedSignoutCard.9f2c1a4b7d', + "You've been signed out" + )} +

+
+ +
+ +

+ {email + ? translate( + 'auto.components.UnexpectedSignoutCard.7b4d9e1f2a', + 'Sign in again as {{value0}} to restore Artifact sharing, Orca Relay, and skill sharing.', + { value0: email } + ) + : translate( + 'auto.components.UnexpectedSignoutCard.5a1c8d3e6f', + 'Sign in again to restore Artifact sharing, Orca Relay, and skill sharing.' + )} +

+ + + + + + + + + + + + +
+ +
+
+
+
+ ) +} diff --git a/src/renderer/src/components/UpdateCard.error-card.test.tsx b/src/renderer/src/components/UpdateCard.error-card.test.tsx index 1e6da6879f7..186a285b26d 100644 --- a/src/renderer/src/components/UpdateCard.error-card.test.tsx +++ b/src/renderer/src/components/UpdateCard.error-card.test.tsx @@ -5,6 +5,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { LinuxPackageInstallRecovery, UpdateStatus } from '../../../shared/update-status-types' import { useAppStore } from '../store' import { UpdateCard } from './UpdateCard' +import { NotificationCardStack } from './NotificationCardStack' const openUrl = vi.fn() const download = vi.fn() @@ -31,7 +32,11 @@ function renderWithInitialStatus(updateStatus: UpdateStatus): RenderResult { updateCardCollapsed: false, updateReassuranceSeen: true }) - return render() + return render( + + + + ) } function renderAfterAvailableStatus(): RenderResult { diff --git a/src/renderer/src/components/UpdateCard.tsx b/src/renderer/src/components/UpdateCard.tsx index 4ccf95ff252..e2023ae21e1 100644 --- a/src/renderer/src/components/UpdateCard.tsx +++ b/src/renderer/src/components/UpdateCard.tsx @@ -239,10 +239,7 @@ export function UpdateCard(): React.JSX.Element | null { !reassuranceSeen && ((status.state === 'available' && !status.externallyManaged) || status.state === 'downloading') return ( -
+
{showReassurance && (
diff --git a/src/renderer/src/components/cmd-j/palette-host-badge.test.ts b/src/renderer/src/components/cmd-j/palette-host-badge.test.ts index cd6310c8580..8d15ff8d034 100644 --- a/src/renderer/src/components/cmd-j/palette-host-badge.test.ts +++ b/src/renderer/src/components/cmd-j/palette-host-badge.test.ts @@ -83,8 +83,7 @@ describe('getPaletteHostBadge', () => { repos: [{ executionHostId: 'runtime:env-1' }], sshTargetLabels: new Map(), settings: { activeRuntimeEnvironmentId: 'env-2' }, - // A live status makes the runtime 'available'; without it the host reads - // 'disconnected' and the badge is suppressed (covered below). + // Only verified availability enables unfiltered host badges. runtimeStatusByEnvironmentId: new Map([ [ 'env-1', @@ -145,3 +144,19 @@ describe('getPaletteHostBadge', () => { expect(getPaletteHostBadge(null, hosts)).toBeNull() }) }) + +it.each(['connecting', 'blocked', 'disconnected', 'error'] as const)( + 'does not infer reachability from %s health, but preserves explicit filter labels', + (health) => { + const hosts = buildSidebarHostOptions({ + repos: [{ executionHostId: 'runtime:env-1' }], + sshTargetLabels: new Map(), + settings: { activeRuntimeEnvironmentId: null } + }).map((host) => (host.kind === 'runtime' ? { ...host, health } : host)) + expect(getPaletteHostBadge({ connectionId: null }, hosts)).toBeNull() + expect(getPaletteHostBadge({ executionHostId: 'runtime:env-1' }, hosts, true)).toEqual({ + hostId: 'runtime:env-1', + label: 'env-1' + }) + } +) diff --git a/src/renderer/src/components/cmd-j/palette-host-badge.ts b/src/renderer/src/components/cmd-j/palette-host-badge.ts index 8ac94adeae4..01a9e33d3ac 100644 --- a/src/renderer/src/components/cmd-j/palette-host-badge.ts +++ b/src/renderer/src/components/cmd-j/palette-host-badge.ts @@ -17,7 +17,7 @@ export type PaletteHostBadge = { // unlike the sidebar gate, which lists disconnected hosts so users can connect. function hasActiveRemoteHost(hostOptions: readonly SidebarHostOption[]): boolean { return hostOptions.some( - (host) => host.id !== LOCAL_EXECUTION_HOST_ID && host.health !== 'disconnected' + (host) => host.id !== LOCAL_EXECUTION_HOST_ID && host.health === 'available' ) } diff --git a/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.test.ts b/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.test.ts index 1d2de1e3624..62ff4b88568 100644 --- a/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.test.ts +++ b/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.test.ts @@ -99,6 +99,23 @@ describe('patchDashboardSnapshotFromAgentStatus', () => { }) }) + it('keeps an unverifiable child compatible with the pop-out wire vocabulary', () => { + const result = patchDashboardSnapshotFromAgentStatus( + snapshot(), + event({ + subagents: [{ id: 'child-1', state: 'unverifiable', startedAt: 100 }] + }) + ) + + expect(result.snapshot.cards[0].subagents).toEqual([ + { + id: 'tab-1:leaf-1\u0000subagent:child-1', + name: 'unknown', + dotState: 'idle' + } + ]) + }) + it('ignores stale, wrong-workspace, and session-only events', () => { const original = snapshot() expect( diff --git a/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.ts b/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.ts index 33ac5a1f737..b4f67ef3f32 100644 --- a/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.ts +++ b/src/renderer/src/components/dashboard-popout/dashboard-agent-status-patch.ts @@ -6,6 +6,7 @@ import { type DashboardSnapshot } from '../../../../shared/dashboard-snapshot' import { dashboardBucketForDotState } from '../dashboard/dashboard-card-bucket' +import { dashboardCardDotState } from '../dashboard/dashboard-row-bucket' export type DashboardAgentStatusPatchResult = { matched: boolean @@ -22,7 +23,7 @@ function patchedSubagents( return event.subagents.map((subagent) => ({ id: `${card.paneKey}\u0000subagent:${subagent.id}`, name: subagent.description || subagent.agentType || 'unknown', - dotState: subagent.state + dotState: dashboardCardDotState(subagent.state) })) } diff --git a/src/renderer/src/components/dashboard/agent-finished-timestamp.test.ts b/src/renderer/src/components/dashboard/agent-finished-timestamp.test.ts index 84116333a1c..91b445dd78f 100644 --- a/src/renderer/src/components/dashboard/agent-finished-timestamp.test.ts +++ b/src/renderer/src/components/dashboard/agent-finished-timestamp.test.ts @@ -52,3 +52,11 @@ describe('lastEnteredDoneAt shares the Smart Sort completion clock', () => { expect(agentEntryCompletionAt(entry)).toBeNull() }) }) + +describe('lastEnteredDoneAt subagent rows', () => { + it('does not report a synthetic completion when the child is unverifiable', () => { + const entry = doneEntry() + + expect(lastEnteredDoneAt({ rowSource: 'subagent', state: 'unverifiable', entry })).toBeNull() + }) +}) diff --git a/src/renderer/src/components/dashboard/agent-finished-timestamp.ts b/src/renderer/src/components/dashboard/agent-finished-timestamp.ts index 599a07d118a..fe101862957 100644 --- a/src/renderer/src/components/dashboard/agent-finished-timestamp.ts +++ b/src/renderer/src/components/dashboard/agent-finished-timestamp.ts @@ -10,9 +10,8 @@ import type { DashboardAgentRow } from './useDashboardData' export function lastEnteredDoneAt( agent: Pick ): number | null { - // Why: idle subagent child rows are alive-but-idle (teammates persist - // between turns), not finished — fall through to the started-at timestamp. - if (agent.rowSource === 'subagent' && agent.state === 'idle') { + // Why: a subagent's synthetic entry may say done while its row is idle or unverifiable. + if (agent.rowSource === 'subagent' && agent.state !== 'done') { return null } const entry = agent.entry diff --git a/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts b/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts index f99738a5c4a..3a1254004a6 100644 --- a/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts +++ b/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts @@ -3,6 +3,7 @@ import type { AgentStatusEntry } from '../../../../shared/agent-status-types' import { makePaneKey } from '../../../../shared/stable-pane-id' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { Worktree } from '../../../../shared/worktree/types' +import type { RetainedAgentEntry } from '@/store/slices/agent-status' import { buildDashboardSnapshot, type DashboardSnapshotState } from './build-dashboard-snapshot' import { createWorktreeAgentRowsCache } from './worktree-agent-rows-cache' @@ -139,4 +140,52 @@ describe('buildDashboardSnapshot rows cache', () => { buildDashboardSnapshot(state, NOW + 60_000, { rowsCache: cache, rowsGeneration: 2 }) expect(cache.lastComputedWorktreeIds.sort()).toEqual(['w1', 'w2']) }) + + it('refreshes a retained row from a provider title published to its current tab', () => { + const cache = createWorktreeAgentRowsCache() + const retainedTab = { ...tab('tab1', 'w1'), title: 'Claude ready' } + const retained: RetainedAgentEntry = { + entry: { + ...entry(PANE_1, 'tab1', 'w1'), + providerSession: { key: 'session_id', id: 'session-a' } + }, + worktreeId: 'w1', + tab: retainedTab, + agentType: 'claude', + startedAt: NOW - 10_000 + } + const initial: DashboardSnapshotState = { + ...baseState(), + tabsByWorktree: { w1: [retainedTab], w2: [tab('tab2', 'w2')] }, + agentStatusByPaneKey: { [PANE_2]: entry(PANE_2, 'tab2', 'w2') }, + retainedAgentsByPaneKey: { [PANE_1]: retained } + } + expect( + buildDashboardSnapshot(initial, NOW, { rowsCache: cache, rowsGeneration: 1 }).cards.find( + (card) => card.paneKey === PANE_1 + )?.conversationName + ).toBeUndefined() + + const titled: DashboardSnapshotState = { + ...initial, + tabsByWorktree: { + ...initial.tabsByWorktree, + w1: [ + { + ...retainedTab, + aiVaultTitle: { agent: 'claude', sessionId: 'session-a', title: 'Provider title' } + } + ] + } + } + const refreshed = buildDashboardSnapshot(titled, NOW, { + rowsCache: cache, + rowsGeneration: 1 + }) + + expect(cache.lastComputedWorktreeIds).toEqual(['w1']) + expect(refreshed.cards.find((card) => card.paneKey === PANE_1)?.conversationName).toBe( + 'Provider title' + ) + }) }) diff --git a/src/renderer/src/components/dashboard/dashboard-card-labels.test.ts b/src/renderer/src/components/dashboard/dashboard-card-labels.test.ts new file mode 100644 index 00000000000..7d08ee7f9d8 --- /dev/null +++ b/src/renderer/src/components/dashboard/dashboard-card-labels.test.ts @@ -0,0 +1,59 @@ +import { describe, expect, it } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalLayoutSnapshot, TerminalTab } from '../../../../shared/terminal-tab-types' +import { rowConversationName } from './dashboard-card-labels' +import type { DashboardAgentRow } from './useDashboardData' + +const LEAF_A = '11111111-1111-4111-8111-111111111111' +const LEAF_B = '22222222-2222-4222-8222-222222222222' +const TAB_ID = 'tab-1' +const TAB: TerminalTab = { + id: TAB_ID, + ptyId: 'pty-1', + worktreeId: 'wt-1', + title: '\u2733 Linear work log', + customTitle: null, + aiVaultTitle: { agent: 'claude', sessionId: 'session-a', title: 'Provider title' }, + color: null, + sortOrder: 0, + createdAt: 0 +} +const LAYOUT: TerminalLayoutSnapshot = { + root: { + type: 'split', + direction: 'horizontal', + first: { type: 'leaf', leafId: LEAF_A }, + second: { type: 'leaf', leafId: LEAF_B } + }, + activeLeafId: LEAF_A, + expandedLeafId: null +} + +function row(leafId: string, sessionId: string): DashboardAgentRow { + const paneKey = makePaneKey(TAB_ID, leafId) + const entry: AgentStatusEntry = { + state: 'working', + prompt: '', + updatedAt: 0, + stateStartedAt: 0, + stateHistory: [], + agentType: 'claude', + paneKey, + providerSession: { key: 'session_id', id: sessionId } + } + return { paneKey, entry, tab: TAB, agentType: 'claude', state: 'working', startedAt: 0 } +} + +describe('rowConversationName', () => { + it('publishes a provider title only for the split-pane session that owns it', () => { + const paneTitles = { 1: '\u2733 Linear work log', 2: '\u2733 Redis cache strategy' } + + expect(rowConversationName(row(LEAF_A, 'session-a'), false, LAYOUT, paneTitles)).toBe( + 'Provider title' + ) + expect(rowConversationName(row(LEAF_B, 'session-b'), false, LAYOUT, paneTitles)).toBe( + 'Redis cache strategy' + ) + }) +}) diff --git a/src/renderer/src/components/dashboard/dashboard-card-labels.ts b/src/renderer/src/components/dashboard/dashboard-card-labels.ts index c6ea3a52bec..5c2b66b1d50 100644 --- a/src/renderer/src/components/dashboard/dashboard-card-labels.ts +++ b/src/renderer/src/components/dashboard/dashboard-card-labels.ts @@ -49,7 +49,12 @@ export function rowConversationName( parsePaneKey(row.paneKey)?.leafId ) return ( - getAgentRowConversationName(row.tab, row.agentType, generatedTitlesEnabled, paneLiveTitle) ?? - undefined + getAgentRowConversationName( + row.tab, + row.agentType, + generatedTitlesEnabled, + paneLiveTitle, + row.entry.providerSession?.id + ) ?? undefined ) } diff --git a/src/renderer/src/components/dashboard/use-agent-row-conversation-name.test.ts b/src/renderer/src/components/dashboard/use-agent-row-conversation-name.test.ts index f6aca88795b..dbf5cae455b 100644 --- a/src/renderer/src/components/dashboard/use-agent-row-conversation-name.test.ts +++ b/src/renderer/src/components/dashboard/use-agent-row-conversation-name.test.ts @@ -197,6 +197,24 @@ describe('useAgentRowConversationName', () => { ) }) + it('gives a provider session title only to the pane that owns that session', () => { + setSplitStore('\u2733 Linear work log') + storeState.current.tabsByWorktree['wt-1'][0] = { + id: 'tab-1', + worktreeId: 'wt-1', + customTitle: null, + title: '\u2733 Linear work log', + aiVaultTitle: { agent: 'claude', sessionId: 'session-a', title: 'Provider title' } + } + const sessionA = splitRow(LEAF_A, '\u2733 Linear work log') + sessionA.entry.providerSession = { key: 'session_id', id: 'session-a' } + const sessionB = splitRow(LEAF_B, '\u2733 Linear work log') + sessionB.entry.providerSession = { key: 'session_id', id: 'session-b' } + + expect(useAgentRowConversationName(sessionA)).toBe('Provider title') + expect(useAgentRowConversationName(sessionB)).toBe('Redis cache strategy') + }) + it('does not rename the sibling row when the other pane is clicked', () => { // Clicking pane B re-syncs the tab title to B's; both rows must be unmoved. setSplitStore('\u2733 Redis cache strategy') diff --git a/src/renderer/src/components/dashboard/use-agent-row-conversation-name.ts b/src/renderer/src/components/dashboard/use-agent-row-conversation-name.ts index 692d0cc20f1..6cdb38f2305 100644 --- a/src/renderer/src/components/dashboard/use-agent-row-conversation-name.ts +++ b/src/renderer/src/components/dashboard/use-agent-row-conversation-name.ts @@ -64,6 +64,7 @@ export function useAgentRowConversationName(agent: DashboardAgentRow): string | liveTab ?? agent.tab, agent.agentType, generatedTitlesEnabled, - paneLiveTitle + paneLiveTitle, + agent.entry.providerSession?.id ) } diff --git a/src/renderer/src/components/editor/EditorPanelHeaderPath.tsx b/src/renderer/src/components/editor/EditorPanelHeaderPath.tsx index f1d0ae5e7b6..eaa5db71209 100644 --- a/src/renderer/src/components/editor/EditorPanelHeaderPath.tsx +++ b/src/renderer/src/components/editor/EditorPanelHeaderPath.tsx @@ -20,11 +20,19 @@ const isMac = navigator.userAgent.includes('Mac') const isLinux = navigator.userAgent.includes('Linux') /** Platform-appropriate label: macOS -> Finder, Windows -> File Explorer, Linux -> Files */ -const revealLabel = isMac - ? 'Reveal in Finder' - : isLinux - ? 'Open Containing Folder' - : 'Reveal in File Explorer' +function getRevealLabel(): string { + return isMac + ? translate('auto.components.editor.EditorPanelHeader.revealInFinder', 'Reveal in Finder') + : isLinux + ? translate( + 'auto.components.editor.EditorPanelHeader.openContainingFolder', + 'Open Containing Folder' + ) + : translate( + 'auto.components.editor.EditorPanelHeader.revealInFileExplorer', + 'Reveal in File Explorer' + ) +} type EditorPanelHeaderPathProps = { activeFile: OpenFile @@ -196,7 +204,7 @@ export function EditorPanelHeaderPath({ {!isVirtualEditorTab && ( - {revealLabel} + {getRevealLabel()} )} diff --git a/src/renderer/src/components/editor/MonacoEditor.tsx b/src/renderer/src/components/editor/MonacoEditor.tsx index 27b1c240e82..31c473bf29f 100644 --- a/src/renderer/src/components/editor/MonacoEditor.tsx +++ b/src/renderer/src/components/editor/MonacoEditor.tsx @@ -10,6 +10,7 @@ import { computeEditorFontSize, resolveEditorFontFamily } from '@/lib/editor-fon import { useContextualCopySetup } from './useContextualCopySetup' import { MonacoGutterContextMenu } from './MonacoGutterContextMenu' import { isLinuxUserAgent } from '../terminal-pane/pane-helpers' +import { MAX_TOKENIZATION_LINE_LENGTH } from '@/lib/monaco-languages/monarch-embed-entry-budget' import { buildFileEditorWordWrapOptions } from './file-editor-word-wrap-options' import { getMonacoAutoHeightForContent, isMonacoAutoHeightCapped } from './monaco-auto-height' import { monacoFindOptions } from './monaco-find-options' @@ -235,6 +236,11 @@ export default function MonacoEditor({ onChange={contentSync.handleChange} onMount={handleMount} options={{ + // `IGlobalEditorOptions`, not per-editor: setting it here pins it for every + // Monaco surface (diff, Peek) too, so this is the only site that needs it. + // Defense-in-depth only — it does NOT guard the Monarch embed recursion, + // which overflowed at ~17_000 chars, under this cap. See the budget module. + maxTokenizationLineLength: MAX_TOKENIZATION_LINE_LENGTH, // Why: only the file editor honors this; Monaco 0.55 DiffEditor hard-overrides minimap.enabled=false on sub-editors (see diffEditorEditors._adjustOptionsForSubEditor). minimap: { enabled: settings?.editorMinimapEnabled ?? false }, scrollBeyondLastLine: false, diff --git a/src/renderer/src/components/editor/markdown-round-trip.test.ts b/src/renderer/src/components/editor/markdown-round-trip.test.ts index 3f896f80e38..8804cc06125 100644 --- a/src/renderer/src/components/editor/markdown-round-trip.test.ts +++ b/src/renderer/src/components/editor/markdown-round-trip.test.ts @@ -16,6 +16,10 @@ function roundTripMarkdown(content: string): string { }) try { + // Why: markdown serialization walks the document without running + // NodeType.checkContent, so it emits byte-identical output from a + // schema-invalid document that would crash on the user's next keystroke. + editor.state.doc.check() return editor.getMarkdown().trimEnd() } finally { editor.destroy() @@ -52,6 +56,32 @@ function markdownAfterTextReplace(content: string, search: string, replacement: } } +function markdownAfterTypingBesideImage(content: string, typed: string): string { + const codec = createRichMarkdownEditorCodec() + const editor = new Editor({ + element: null, + extensions: createRichMarkdownExtensions({ codec }), + content: encodeRawMarkdownHtmlForRichEditor(content, codec), + contentType: 'markdown' + }) + + try { + let after = -1 + editor.state.doc.descendants((node, pos) => { + if (after === -1 && node.type.name === 'image') { + after = pos + node.nodeSize + } + }) + if (after === -1) { + throw new Error('Missing image node') + } + editor.view.dispatch(editor.state.tr.insertText(typed, after, after)) + return editor.getMarkdown().trimEnd() + } finally { + editor.destroy() + } +} + function slashCommandMarkdown(commandId: SlashCommandId): string { const codec = createRichMarkdownEditorCodec() const editor = new Editor({ @@ -116,6 +146,28 @@ describe('rich markdown round trip', () => { ) }) + it('preserves an image in a details summary across an edit', () => { + expect( + markdownAfterTextReplace( + '
Toggle ![i](x.png)

Body

\n', + 'Toggle', + 'Switch' + ) + ).toBe( + '
\nSwitch ![i](x.png)\n\nBody\n\n
' + ) + }) + + it('preserves inline math in a details summary across an edit', () => { + expect( + markdownAfterTextReplace( + '
Toggle $x^2$

Body

\n', + 'Toggle', + 'Switch' + ) + ).toBe('
\nSwitch $x^2$\n\nBody\n\n
') + }) + it('does not double-escape entities in editable details summaries', () => { expect(roundTripMarkdown('
A & B

Body

\n')).toBe( '
\nA & B\n\nBody\n\n
' @@ -298,6 +350,42 @@ describe('rich markdown round trip', () => { ) }) + it('preserves an image that sits mid-sentence inside a paragraph', () => { + expect(roundTripMarkdown('Install the ![icon](icon.png) extension\n')).toBe( + 'Install the ![icon](icon.png) extension' + ) + }) + + it('preserves a mid-sentence image after an editor transaction', () => { + expect( + markdownAfterTextReplace('Install the ![icon](icon.png) extension\n', 'extension', 'add-on') + ).toBe('Install the ![icon](icon.png) add-on') + }) + + it('preserves a standalone image as its own block', () => { + expect(roundTripMarkdown('Intro\n\n![shot](shot.png)\n\nOutro\n')).toBe( + 'Intro\n\n![shot](shot.png)\n\nOutro' + ) + // Typing beside the image must join its paragraph instead of opening a new block, + // which only holds while the standalone image stays wrapped in a paragraph. + expect(markdownAfterTypingBesideImage('Intro\n\n![shot](shot.png)\n\nOutro\n', 'X')).toBe( + 'Intro\n\n![shot](shot.png)X\n\nOutro' + ) + }) + + it('preserves images nested in list items and table cells', () => { + expect(roundTripMarkdown('- step ![shot](shot.png)\n')).toBe('- step ![shot](shot.png)') + expect(roundTripMarkdown('| a |\n| - |\n| ![shot](shot.png) |\n')).toContain( + '![shot](shot.png)' + ) + expect(markdownAfterTextReplace('- step ![shot](shot.png)\n', 'step', 'stage')).toBe( + '- stage ![shot](shot.png)' + ) + expect( + markdownAfterTextReplace('| a |\n| - |\n| b ![shot](shot.png) |\n', 'b ', 'c ') + ).toContain('![shot](shot.png)') + }) + it('preserves links whose label is inline code', () => { expect(roundTripMarkdown('Link to [`foo.md`](./foo.md) here\n')).toBe( 'Link to [`foo.md`](./foo.md) here' diff --git a/src/renderer/src/components/editor/rich-markdown-details-extension.ts b/src/renderer/src/components/editor/rich-markdown-details-extension.ts index 151fe512330..1bfcfc09a15 100644 --- a/src/renderer/src/components/editor/rich-markdown-details-extension.ts +++ b/src/renderer/src/components/editor/rich-markdown-details-extension.ts @@ -286,6 +286,13 @@ const OrcaDetails = Details.extend({ } }) +const OrcaDetailsSummary = DetailsSummary.extend({ + // Why: the summary parser runs parseInline, which emits image/math nodes that + // upstream's text*-only summary rejects, so the doc is schema-invalid until the + // first edit reassembles the summary and ProseMirror throws. + content: 'inline*' +}) + const OrcaDetailsContent = DetailsContent.extend({ // Why: detailsContent's double-Enter escape must run before StarterKit's // generic paragraph split, otherwise users can get stuck inside a toggle. @@ -314,7 +321,7 @@ export function createOrcaDetailsExtensions(): AnyExtension[] { class: 'orca-details' } }), - DetailsSummary, + OrcaDetailsSummary, OrcaDetailsContent ] } diff --git a/src/renderer/src/components/editor/rich-markdown-extensions.ts b/src/renderer/src/components/editor/rich-markdown-extensions.ts index 42904291a23..8286c34077c 100644 --- a/src/renderer/src/components/editor/rich-markdown-extensions.ts +++ b/src/renderer/src/components/editor/rich-markdown-extensions.ts @@ -37,6 +37,7 @@ import type { RichMarkdownEditorCodec } from './rich-markdown-source-transport' import { createRichMarkdownHtmlSuperscriptLink } from './rich-markdown-html-superscript-link' import type { RichMarkdownHtmlSuperscriptLinkContext } from './rich-markdown-html-superscript-link-context' import { RichMarkdownOrderedList } from './rich-markdown-ordered-list' +import { RichMarkdownParagraph } from './rich-markdown-paragraph' import { RichMarkdownCodeBlockLowlight } from './rich-markdown-lowlight' import { RichMarkdownTaskList } from './rich-markdown-task-list' import { createCachedLowlight } from './rich-markdown-lowlight-cache' @@ -77,8 +78,10 @@ export function createRichMarkdownExtensions({ link: false, code: false, codeBlock: false, - orderedList: false + orderedList: false, + paragraph: false }), + RichMarkdownParagraph, RichMarkdownCode, RichMarkdownCodeBlockLowlight.extend({ addNodeView() { @@ -116,8 +119,12 @@ export function createRichMarkdownExtensions({ // native image drag (which sends image bytes) from conflicting with // ProseMirror's node-level drag (which serializes the schema node // for relocation within the document). - const dom = document.createElement('div') + const dom = document.createElement('span') + // Why: the wrapper sits in inline content, so it must not introduce a + // block box or the surrounding text would break onto its own line. + dom.style.display = 'inline-block' dom.style.lineHeight = '0' + dom.style.maxWidth = '100%' const img = document.createElement('img') img.draggable = false @@ -205,7 +212,11 @@ export function createRichMarkdownExtensions({ } } }).configure({ - allowBase64: true + allowBase64: true, + // Why: the markdown parser nests images inside paragraphs, so a block image + // node yields a schema-invalid document that only throws on the first edit + // that reassembles the paragraph. + inline: true }), RichMarkdownOrderedList, RichMarkdownTaskList, diff --git a/src/renderer/src/components/editor/rich-markdown-image-insert-code-block.test.ts b/src/renderer/src/components/editor/rich-markdown-image-insert-code-block.test.ts new file mode 100644 index 00000000000..a83de8559f8 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-image-insert-code-block.test.ts @@ -0,0 +1,149 @@ +// @vitest-environment happy-dom + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { Editor } from '@tiptap/core' +import { renderHook } from '@testing-library/react' +import { createRichMarkdownExtensions } from './rich-markdown-extensions' +import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' +import { useLocalImagePick } from './useLocalImagePick' +import { handleRichMarkdownImagePaste } from './rich-markdown-paste-image' +import { runSlashCommand, slashCommands } from './rich-markdown-slash-commands' + +vi.mock('@/runtime/runtime-file-client', () => ({ + importExternalPathsToRuntime: vi.fn().mockResolvedValue({ + results: [{ status: 'imported', destPath: '/repo/shot.png' }] + }) +})) + +vi.mock('@/lib/connection-context', () => ({ + getConnectionId: vi.fn(() => null) +})) + +vi.mock('@/store', () => ({ + useAppStore: { + getState: vi.fn(() => ({ + settings: { activeRuntimeEnvironmentId: null }, + folderWorkspaces: [], + worktreesByRepo: { repo1: [{ id: 'wt-1', path: '/repo' }] } + })) + } +})) + +vi.mock('@/runtime/runtime-rpc-client', () => ({ + settingsForRuntimeOwner: vi.fn((settings) => settings) +})) + +vi.mock('sonner', () => ({ + toast: { error: vi.fn() } +})) + +const CODE_BLOCK_SOURCE = '```ts\nconst a = 1\n```\n' +// The image splits the fence; both halves keep their ``` fencing and `ts` language. +const SPLIT_CODE_BLOCK = '```ts\nconst\n```\n\n![](shot.png)\n\n```ts\n a = 1\n```' + +function expectFencedSplit(target: Editor): void { + expect(target.getMarkdown().trimEnd()).toBe(SPLIT_CODE_BLOCK) + expect(() => target.state.doc.check()).not.toThrow() +} + +let editor: Editor + +function mountRichMarkdownEditor(markdown: string): Editor { + const host = document.createElement('div') + document.body.appendChild(host) + return new Editor({ + element: host, + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: markdown, + contentType: 'markdown' + }) +} + +function positionInsideCodeBlock(target: Editor): number { + let pos = -1 + target.state.doc.descendants((node, nodePos) => { + if (pos === -1 && node.isText && node.text?.startsWith('const')) { + pos = nodePos + 'const'.length + } + }) + if (pos === -1) { + throw new Error('Missing code block text') + } + return pos +} + +async function flushPromises(): Promise { + for (let index = 0; index < 8; index += 1) { + await Promise.resolve() + } +} + +describe('inserting an image while the cursor is inside a fenced code block', () => { + beforeEach(() => { + document.body.replaceChildren() + vi.clearAllMocks() + editor = mountRichMarkdownEditor(CODE_BLOCK_SOURCE) + editor.commands.setTextSelection(positionInsideCodeBlock(editor)) + globalThis.window.api = { + ...globalThis.window.api, + shell: { pickImage: vi.fn().mockResolvedValue('/tmp/shot.png') }, + ui: { saveClipboardImageAsTempFile: vi.fn().mockResolvedValue('/tmp/shot.png') } + } as unknown as Window['api'] + }) + + afterEach(() => { + editor.destroy() + vi.restoreAllMocks() + }) + + it('keeps both halves fenced when the toolbar picker inserts the image', async () => { + const { result } = renderHook(() => useLocalImagePick(editor as never, '/repo/note.md', 'wt-1')) + + await result.current() + await flushPromises() + + expectFencedSplit(editor) + }) + + it('keeps both halves fenced when the slash command inserts the image', async () => { + const imageCommand = slashCommands.find((command) => command.id === 'image') + expect(imageCommand).toBeDefined() + const { result } = renderHook(() => useLocalImagePick(editor as never, '/repo/note.md', 'wt-1')) + const from = editor.state.selection.from + + runSlashCommand(editor as never, { from, to: from }, imageCommand!, () => { + void result.current() + }) + await flushPromises() + + expectFencedSplit(editor) + }) + + it('keeps both halves fenced when a clipboard screenshot is pasted', async () => { + const handled = handleRichMarkdownImagePaste({ + editor: editor as never, + event: { + clipboardData: { items: [{ kind: 'file', type: 'image/png' }] }, + preventDefault: vi.fn() + } as unknown as ClipboardEvent, + filePath: '/repo/note.md', + worktreeId: 'wt-1' + }) + await flushPromises() + + expect(handled).toBe(true) + expectFencedSplit(editor) + }) + + it('still inserts the image inline when the cursor is in ordinary prose', async () => { + editor.destroy() + editor = mountRichMarkdownEditor('Install the extension\n') + editor.commands.setTextSelection(13) + const { result } = renderHook(() => useLocalImagePick(editor as never, '/repo/note.md', 'wt-1')) + + await result.current() + await flushPromises() + + expect(editor.getMarkdown().trimEnd()).toBe('Install the ![](shot.png)extension') + }) +}) diff --git a/src/renderer/src/components/editor/rich-markdown-image-insert-content.ts b/src/renderer/src/components/editor/rich-markdown-image-insert-content.ts new file mode 100644 index 00000000000..2d0f256bd41 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-image-insert-content.ts @@ -0,0 +1,26 @@ +import type { Editor, JSONContent } from '@tiptap/react' + +/** + * Why: an inline image cannot be fitted into `codeBlock` (`text*`), so inserting one at a + * position inside a fence makes ProseMirror dissolve the block — the remaining code escapes + * as prose and the language attribute is lost. Wrapping the image in a paragraph makes + * ProseMirror split the fence instead, leaving both halves intact. + */ +export function buildRichMarkdownImageInsertContent( + editor: Editor, + pos: number, + attrs: { src: string } +): JSONContent { + const image: JSONContent = { type: 'image', attrs } + const imageType = editor.schema.nodes.image + const doc = editor.state.doc + if (!imageType || pos < 0 || pos > doc.content.size) { + return image + } + const $pos = doc.resolve(pos) + const index = $pos.index() + if ($pos.parent.canReplaceWith(index, index, imageType)) { + return image + } + return { type: 'paragraph', content: [image] } +} diff --git a/src/renderer/src/components/editor/rich-markdown-image-insert.test.ts b/src/renderer/src/components/editor/rich-markdown-image-insert.test.ts index 6eb68d924ad..d9fba294124 100644 --- a/src/renderer/src/components/editor/rich-markdown-image-insert.test.ts +++ b/src/renderer/src/components/editor/rich-markdown-image-insert.test.ts @@ -1,6 +1,11 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +// @vitest-environment happy-dom + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { Editor } from '@tiptap/core' import { toast } from 'sonner' import { insertRichMarkdownImageFromPath } from './rich-markdown-image-insert' +import { createRichMarkdownExtensions } from './rich-markdown-extensions' +import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' import { importExternalPathsToRuntime } from '@/runtime/runtime-file-client' vi.mock('@/runtime/runtime-file-client', () => ({ @@ -27,12 +32,28 @@ vi.mock('sonner', () => ({ toast: { error: vi.fn() } })) -function editorWithRunResult(runResult: boolean) { +const openEditors: Editor[] = [] + +function createRichMarkdownEditor(markdown: string): Editor { + const editor = new Editor({ + element: null, + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: markdown, + contentType: 'markdown' + }) + openEditors.push(editor) + return editor +} + +function editorWithRunResult(runResult: boolean, markdown = 'hello world') { const run = vi.fn(() => runResult) const insertContentAt = vi.fn(() => ({ run })) const focus = vi.fn(() => ({ insertContentAt })) const chain = vi.fn(() => ({ focus })) - return { editor: { chain }, chain, focus, insertContentAt, run } + // Why: the insert path reads the real schema and document to decide whether an + // inline image fits at the target position, so the stub borrows both. + const { schema, state } = createRichMarkdownEditor(markdown) + return { editor: { chain, schema, state }, chain, focus, insertContentAt, run } } describe('insertRichMarkdownImageFromPath', () => { @@ -51,6 +72,12 @@ describe('insertRichMarkdownImageFromPath', () => { } as never) }) + afterEach(() => { + while (openEditors.length > 0) { + openEditors.pop()?.destroy() + } + }) + it('shows an error when TipTap rejects image insertion without throwing', async () => { const { editor } = editorWithRunResult(false) diff --git a/src/renderer/src/components/editor/rich-markdown-image-insert.ts b/src/renderer/src/components/editor/rich-markdown-image-insert.ts index 7051502f45c..3c1354a06b6 100644 --- a/src/renderer/src/components/editor/rich-markdown-image-insert.ts +++ b/src/renderer/src/components/editor/rich-markdown-image-insert.ts @@ -10,6 +10,7 @@ import { captureDirectSshMutationExpectation } from '@/lib/ssh-mutation-expectat import { translate } from '@/i18n/i18n' import { parseWorkspaceKey } from '../../../../shared/workspace-scope' import { extractIpcErrorMessage } from './rich-markdown-ipc-error-message' +import { buildRichMarkdownImageInsertContent } from './rich-markdown-image-insert-content' export type RichMarkdownImageInsertArgs = { editor: Editor @@ -88,7 +89,10 @@ export async function insertRichMarkdownImageFromPath({ const inserted = editor .chain() .focus() - .insertContentAt(insertPos, { type: 'image', attrs: { src: imageSrc } }) + .insertContentAt( + insertPos, + buildRichMarkdownImageInsertContent(editor, insertPos, { src: imageSrc }) + ) .run() if (!inserted) { toast.error( diff --git a/src/renderer/src/components/editor/rich-markdown-inline-image-paragraph.test.ts b/src/renderer/src/components/editor/rich-markdown-inline-image-paragraph.test.ts new file mode 100644 index 00000000000..99f35343760 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-inline-image-paragraph.test.ts @@ -0,0 +1,95 @@ +import { describe, expect, it } from 'vitest' +import { Editor } from '@tiptap/core' +import { encodeRawMarkdownHtmlForRichEditor } from './raw-markdown-html' +import { createRichMarkdownExtensions } from './rich-markdown-extensions' +import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' +import { createRichMarkdownHtmlSuperscriptLinkContext } from './rich-markdown-html-superscript-link-context' + +// Crash report 0e46c048: Vietnamese prose with a soft line break, an inline code +// span and an inline image, which the markdown parser nests inside one paragraph. +const CRASH_SOURCE = + 'Trình duyệt chỉ cho phép cài từ Web Store.\nnh `.crx` (tham chiếu, KHÔNG chặn) ![ảnh](chrome.png) và tiếp tục\n' + +function createRichMarkdownEditorFromSource(source: string): Editor { + const codec = createRichMarkdownEditorCodec() + return new Editor({ + element: null, + extensions: createRichMarkdownExtensions({ + codec, + htmlSuperscriptLinks: true, + htmlSuperscriptLinkContext: createRichMarkdownHtmlSuperscriptLinkContext({ + sourceFilePath: '', + worktreeId: '', + worktreeRoot: null, + sourceOwner: { kind: 'unknown' } + }) + }), + content: encodeRawMarkdownHtmlForRichEditor(source, codec, { htmlSuperscriptLinks: true }), + contentType: 'markdown' + }) +} + +describe('rich markdown inline images inside a paragraph', () => { + it('parses an inline image into a schema-valid paragraph', () => { + const editor = createRichMarkdownEditorFromSource(CRASH_SOURCE) + + try { + expect(() => editor.state.doc.check()).not.toThrow() + } finally { + editor.destroy() + } + }) + + it('survives an ordinary edit in a paragraph that holds an inline image', () => { + const editor = createRichMarkdownEditorFromSource(CRASH_SOURCE) + + try { + // Any ReplaceStep that rebuilds the paragraph runs NodeType.checkContent on + // the reassembled content — the exact frame the crash report bottoms out in. + expect(() => editor.view.dispatch(editor.state.tr.insertText('X', 5, 8))).not.toThrow() + } finally { + editor.destroy() + } + }) + + it('round-trips the reported document without dropping the inline image', () => { + const editor = createRichMarkdownEditorFromSource(CRASH_SOURCE) + + try { + // Why: serialization never runs NodeType.checkContent, so the markdown + // matches byte-for-byte even when the document is schema-invalid. + expect(() => editor.state.doc.check()).not.toThrow() + expect(editor.getMarkdown().trimEnd()).toBe(CRASH_SOURCE.trimEnd()) + } finally { + editor.destroy() + } + }) + + it('keeps a standalone image inside a paragraph rather than directly under the doc', () => { + // Upstream's paragraph parser hoists a lone image out of its paragraph, which + // leaves an inline node as a direct child of `doc` once images are inline. + const editor = createRichMarkdownEditorFromSource('Intro\n\n![shot](shot.png)\n\nOutro\n') + + try { + expect(() => editor.state.doc.check()).not.toThrow() + expect(editor.state.doc.child(1).type.name).toBe('paragraph') + } finally { + editor.destroy() + } + }) + + it('keeps a markdown inline image as an inline node', () => { + const editor = createRichMarkdownEditorFromSource(CRASH_SOURCE) + + try { + const paragraph = editor.state.doc.child(0) + const imageIndex = [...Array(paragraph.childCount).keys()].find( + (index) => paragraph.child(index).type.name === 'image' + ) + expect(imageIndex).toBeDefined() + expect(paragraph.child(imageIndex!).type.isInline).toBe(true) + } finally { + editor.destroy() + } + }) +}) diff --git a/src/renderer/src/components/editor/rich-markdown-local-image.test.ts b/src/renderer/src/components/editor/rich-markdown-local-image.test.ts index bcd4e64c6f1..056fbdd8d33 100644 --- a/src/renderer/src/components/editor/rich-markdown-local-image.test.ts +++ b/src/renderer/src/components/editor/rich-markdown-local-image.test.ts @@ -88,4 +88,23 @@ describe('rich markdown local images', () => { editor.destroy() } }) + + it('renders a mid-sentence image inside its paragraph without a block box', () => { + const host = document.createElement('div') + document.body.appendChild(host) + const editor = new Editor({ + element: host, + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: 'before ![](diagram.png) after', + contentType: 'markdown' + }) + + try { + const img = host.querySelector('p img') + expect(img).not.toBeNull() + expect((img!.parentElement as HTMLElement).style.display).toBe('inline-block') + } finally { + editor.destroy() + } + }) }) diff --git a/src/renderer/src/components/editor/rich-markdown-paragraph.test.ts b/src/renderer/src/components/editor/rich-markdown-paragraph.test.ts new file mode 100644 index 00000000000..d59cb0c4c04 --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-paragraph.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it, vi } from 'vitest' +import { RichMarkdownParagraph } from './rich-markdown-paragraph' + +vi.mock('@tiptap/extension-paragraph', async () => { + const actual = (await vi.importActual('@tiptap/extension-paragraph')) as { + Paragraph: { extend: (config: object) => { config: Record } } + } + // Simulates a Tiptap upgrade that drops `parseMarkdown` from the upstream paragraph. + const Paragraph = actual.Paragraph.extend({}) + Paragraph.config.parseMarkdown = undefined + return { ...actual, Paragraph } +}) + +describe('RichMarkdownParagraph without an upstream markdown parser', () => { + it('parses paragraphs through parseInline instead of throwing', () => { + const parseInline = vi.fn(() => [{ type: 'text', text: 'Install the extension' }]) + const createNode = vi.fn((type: string, attrs: unknown, content: unknown) => ({ + type, + attrs, + content + })) + const parseMarkdown = RichMarkdownParagraph.config.parseMarkdown as ( + token: unknown, + helpers: unknown + ) => unknown + const token = { type: 'paragraph', tokens: [{ type: 'text' }, { type: 'image' }] } + + expect(() => parseMarkdown(token, { createNode, parseInline })).not.toThrow() + expect(createNode).toHaveBeenCalledWith('paragraph', undefined, [ + { type: 'text', text: 'Install the extension' } + ]) + }) +}) diff --git a/src/renderer/src/components/editor/rich-markdown-paragraph.ts b/src/renderer/src/components/editor/rich-markdown-paragraph.ts new file mode 100644 index 00000000000..ad3fa4f1a1e --- /dev/null +++ b/src/renderer/src/components/editor/rich-markdown-paragraph.ts @@ -0,0 +1,23 @@ +import type { MarkdownParseHelpers, MarkdownParseResult, MarkdownToken } from '@tiptap/core' +import { Paragraph } from '@tiptap/extension-paragraph' + +type ParagraphMarkdownParser = ( + token: MarkdownToken, + helpers: MarkdownParseHelpers +) => MarkdownParseResult + +const baseParseMarkdown = Paragraph.config.parseMarkdown as ParagraphMarkdownParser | undefined + +export const RichMarkdownParagraph = Paragraph.extend({ + parseMarkdown: (token, helpers) => { + const tokens = token.tokens ?? [] + // Why: upstream hoists a lone image out of its paragraph, which produces an + // inline image node directly under `doc` now that images are inline nodes. + // The missing-base fallback keeps a Tiptap upgrade that drops the field from + // turning every paragraph parse into a TypeError. + if (!baseParseMarkdown || (tokens.length === 1 && tokens[0]?.type === 'image')) { + return helpers.createNode('paragraph', undefined, helpers.parseInline(tokens)) + } + return baseParseMarkdown(token, helpers) + } +}) diff --git a/src/renderer/src/components/github/use-image-input.test.ts b/src/renderer/src/components/github/use-image-input.test.ts new file mode 100644 index 00000000000..06a059d569e --- /dev/null +++ b/src/renderer/src/components/github/use-image-input.test.ts @@ -0,0 +1,71 @@ +// @vitest-environment happy-dom + +import { afterEach, describe, expect, it, vi } from 'vitest' +import { Editor } from '@tiptap/core' +import { act, renderHook } from '@testing-library/react' +import { createRichMarkdownExtensions } from '@/components/editor/rich-markdown-extensions' +import { createRichMarkdownEditorCodec } from '@/components/editor/rich-markdown-source-transport' +import { useImageInput } from './use-image-input' + +vi.mock('sonner', () => ({ + toast: { error: vi.fn() } +})) + +let editor: Editor | null = null + +function mountComposerEditor(markdown: string): Editor { + const host = document.createElement('div') + document.body.appendChild(host) + editor = new Editor({ + element: host, + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: markdown, + contentType: 'markdown' + }) + return editor +} + +function renderImageInput(target: Editor) { + return renderHook(() => useImageInput({ current: target } as never, { current: false })) +} + +describe('useImageInput', () => { + afterEach(() => { + editor?.destroy() + editor = null + document.body.replaceChildren() + }) + + it('splits a fenced code block instead of dissolving it when inserting an image URL', () => { + const target = mountComposerEditor('```ts\nconst a = 1\n```\n') + let pos = -1 + target.state.doc.descendants((node, nodePos) => { + if (pos === -1 && node.isText && node.text?.startsWith('const')) { + pos = nodePos + 'const'.length + } + }) + target.commands.setTextSelection(pos) + const { result } = renderImageInput(target) + + act(() => result.current.setImageUrl('https://example.com/shot.png')) + act(() => result.current.insertImageUrl()) + + expect(target.getMarkdown().trimEnd()).toBe( + '```ts\nconst\n```\n\n![](https://example.com/shot.png)\n\n```ts\n a = 1\n```' + ) + expect(() => target.state.doc.check()).not.toThrow() + }) + + it('keeps an image URL inline when the cursor is in ordinary prose', () => { + const target = mountComposerEditor('Install the extension\n') + target.commands.setTextSelection(13) + const { result } = renderImageInput(target) + + act(() => result.current.setImageUrl('https://example.com/shot.png')) + act(() => result.current.insertImageUrl()) + + expect(target.getMarkdown().trimEnd()).toBe( + 'Install the ![](https://example.com/shot.png)extension' + ) + }) +}) diff --git a/src/renderer/src/components/github/use-image-input.ts b/src/renderer/src/components/github/use-image-input.ts index 30569ba084f..8decc6e9557 100644 --- a/src/renderer/src/components/github/use-image-input.ts +++ b/src/renderer/src/components/github/use-image-input.ts @@ -2,6 +2,7 @@ import { useCallback, useEffect, useRef, useState } from 'react' import { toast } from 'sonner' import type { Editor } from '@tiptap/react' import { getGitHubMarkdownImageUrlState } from './github-markdown-image-url' +import { buildRichMarkdownImageInsertContent } from '@/components/editor/rich-markdown-image-insert-content' import { translate } from '@/i18n/i18n' export function useImageInput( @@ -54,7 +55,11 @@ export function useImageInput( editor .chain() .focus() - .insertContent({ type: 'image', attrs: { src: imageUrlState.url } }) + .insertContent( + buildRichMarkdownImageInsertContent(editor, editor.state.selection.from, { + src: imageUrlState.url + }) + ) .run() setImageUrl('') setImageInputOpen(false) diff --git a/src/renderer/src/components/native-chat/NativeChatAutocompleteMenus.test.tsx b/src/renderer/src/components/native-chat/NativeChatAutocompleteMenus.test.tsx index d5dac1efab5..60c70baee66 100644 --- a/src/renderer/src/components/native-chat/NativeChatAutocompleteMenus.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatAutocompleteMenus.test.tsx @@ -3,6 +3,8 @@ import { cleanup, fireEvent, render, screen } from '@testing-library/react' import { afterEach, describe, expect, it, vi } from 'vitest' import { NativeChatPickerMenu } from './NativeChatAutocompleteMenus' +import { buildNativeChatPickerItems } from './native-chat-picker-items' +import { sessionSlashCommandSuggestions } from '../../../../shared/native-chat-slash-commands' import type { ComposerAutocomplete } from './native-chat-composer-state' function autocomplete( @@ -13,6 +15,7 @@ function autocomplete( query: '', triggerKey: '/:0', prefix: '/', + dispatchable: true, grouped: true, commandsEnabled: true, skillsEnabled: true, @@ -21,6 +24,7 @@ function autocomplete( kind: 'command', id: 'command:clear', name: 'clear', + token: '/clear', description: 'Clear history', skillCollision: false }, @@ -28,6 +32,7 @@ function autocomplete( kind: 'skill', id: 'skill:browser', name: 'browser', + token: '/browser', description: 'Use a browser', sources: [{ sourceKind: 'repo', skillFilePath: '/repo/browser/SKILL.md' }] } @@ -125,6 +130,45 @@ describe('NativeChatPickerMenu', () => { expect(screen.getAllByText('No matching commands')).toHaveLength(2) }) + it('shows the argument hint the provider reported beside the command token', () => { + render( + ' + }, + { name: 'clear', kind: 'command' }, + { name: 'wordy', kind: 'command', argumentHint: `<${'a'.repeat(200)}>` } + ]), + [], + '', + '/' + ) + })} + activeIndex={0} + listboxId="picker" + onChoose={vi.fn()} + onRetry={vi.fn()} + /> + ) + const goal = screen.getByRole('option', { name: /goal/i }) + expect(goal.textContent).toContain('') + expect(goal.textContent).toContain('Set a goal and keep working until it is met') + // A command the report left hintless renders its row unchanged. + expect(screen.getByRole('option', { name: /clear/i }).textContent).toBe( + '/clearClear conversation history' + ) + // A hint long enough to swamp the row is capped before it reaches the DOM. + expect(screen.getByRole('option', { name: /wordy/i }).textContent).toBe( + `/wordy<${'a'.repeat(79)}` + ) + }) + it('announces a successful empty skill result distinctly from loading', () => { render( + autocomplete: Extract activeIndex: number listboxId: string onChoose: (item: NativeChatPickerItem) => void @@ -54,7 +54,6 @@ export const NativeChatPickerMenu = memo(function NativeChatPickerMenu({ + autocomplete: Extract ): string { - if (autocomplete.mode === 'skill' || !autocomplete.commandsEnabled) { + if (!autocomplete.commandsEnabled) { return translate('components.native-chat.composer.noSkills', 'No matching skills') } if (autocomplete.skillsEnabled) { @@ -174,7 +172,6 @@ function PickerStatus({ children }: { children: React.ReactNode }): React.JSX.El function PickerOption({ item, - prefix, index, activeIndex, listboxId, @@ -182,7 +179,6 @@ function PickerOption({ onChoose }: { item: NativeChatPickerItem - prefix: '/' | '$' index: number activeIndex: number listboxId: string @@ -213,7 +209,14 @@ function PickerOption({ ) : null} - {prefix + item.name} + + {item.token} + {item.kind === 'command' && item.argumentHint ? ( + + {item.argumentHint} + + ) : null} + {item.description ? ( {item.description} ) : null} diff --git a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx index e137220dd66..365be31cc47 100644 --- a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.test.tsx @@ -2,12 +2,34 @@ import '@testing-library/jest-dom/vitest' -import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { act, cleanup, fireEvent, render, screen, within } from '@testing-library/react' +import { Profiler, useState } from 'react' import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' import { NativeChatBackgroundTasksStatus } from './NativeChatBackgroundTasksStatus' -afterEach(cleanup) +afterEach(() => { + cleanup() + vi.useRealTimers() +}) + +/** The strip's disclosure is parent-owned, because the strip unmounts whenever + * live work momentarily drops to nothing; this stands in for that owner. */ +function DisclosureHost( + props: Omit< + Parameters[0], + 'expanded' | 'onExpandedChange' + > +): React.JSX.Element { + const [expanded, setExpanded] = useState(false) + return ( + + ) +} const TASKS: AgentSessionBackgroundTask[] = [ { id: 'codex-agent:child-1', kind: 'agent', description: 'count_a' }, @@ -19,8 +41,11 @@ function renderStrip(props: { supportsTaskStop: boolean; supportsStopAll: boolea } { const onStop = vi.fn() render( - { expect(screen.getByLabelText('Stop background tasks')).toBeInTheDocument() }) + it('withholds a row stop the host reported it cannot act on', () => { + // Claude publishes in-turn foreground rows with `stoppable: false`: the + // session accepts targeted stops, but `stopTask` has no target for this row, + // so a Stop here resolves to an empty list and reports nothing cancelled. + render( + + ) + fireEvent.click(screen.getByRole('button', { expanded: false })) + + expect(screen.getByText('in-turn subagent')).toBeInTheDocument() + expect(screen.queryByLabelText('Stop in-turn subagent')).not.toBeInTheDocument() + expect(screen.getByLabelText('Stop backgrounded subagent')).toBeInTheDocument() + }) + it('offers no stop at all when the provider exposes none', () => { // Codex: a Stop button here would be a control that cannot act. renderStrip({ supportsTaskStop: false, supportsStopAll: false }) @@ -53,3 +105,233 @@ describe('NativeChatBackgroundTasksStatus stop affordances', () => { expect(screen.getByText('sleep 90')).toBeInTheDocument() }) }) + +describe('background-tasks strip header', () => { + function renderHeader(tasks: AgentSessionBackgroundTask[]): HTMLElement { + render( + {}} + /> + ) + return screen.getByRole('button', { expanded: false }) + } + + it('leads each kind segment with that kind icon and keeps the counts in the accessible name', () => { + const header = renderHeader([ + { id: 'a1', kind: 'agent' }, + { id: 'a2', kind: 'agent' }, + { id: 'a3', kind: 'agent' }, + { id: 'm1', kind: 'monitor' } + ]) + expect(header).toHaveAttribute('aria-label', '3 agents · 1 monitor') + expect(header.querySelector('.lucide-bot')).toBeInTheDocument() + // Heartbeat, the same glyph the agent sidebar shows for monitoring. + expect(header.querySelector('.lucide-activity')).toBeInTheDocument() + // Two kind icons and the chevron: the aggregate state dot is gone. + expect(header.querySelectorAll('svg')).toHaveLength(3) + for (const icon of header.querySelectorAll('svg')) { + expect(icon).toHaveAttribute('aria-hidden', 'true') + } + }) + + it('gives the monitor heartbeat the sidebar amber and leaves other kinds neutral', () => { + const header = renderHeader([ + { id: 'a1', kind: 'agent' }, + { id: 'm1', kind: 'monitor' } + ]) + // Same glyph AND same colour as AgentStateDot/StatusIndicator, or a monitor + // here does not read as the monitor there. + expect(header.querySelector('.lucide-activity')?.classList).toContain('text-yellow-500') + expect(header.querySelector('.lucide-bot')?.classList).toContain('text-muted-foreground') + expect(header.querySelector('.lucide-bot')?.classList).not.toContain('text-yellow-500') + }) + + it('dims the monitor amber while a turn owns the voice', () => { + render( + {}} + /> + ) + const header = screen.getByRole('button', { expanded: false }) + expect(header.querySelector('.lucide-activity')?.classList).toContain('text-yellow-500/40') + }) + + it('carries the monitor amber on the expanded row too', () => { + const header = renderHeader([ + { id: 'm1', kind: 'monitor', description: 'watcher' }, + { id: 'c1', kind: 'command', description: 'sleep 90' } + ]) + fireEvent.click(header) + // Each kind group is its own labelled list, so scope to the monitor one. + const monitors = screen.getByRole('list', { name: 'Monitors' }) + expect(monitors.querySelector('.lucide-activity')?.classList).toContain('text-yellow-500') + const shell = screen.getByRole('list', { name: 'Shell' }) + expect(shell.querySelector('.lucide-square-terminal')?.classList).toContain( + 'text-muted-foreground' + ) + }) + + it('draws the segment separator in a visible text tone, not the divider token', () => { + const header = renderHeader([ + { id: 'a1', kind: 'agent' }, + { id: 'c1', kind: 'command' } + ]) + const separators = [...header.querySelectorAll('span')].filter( + (element) => element.textContent === ' · ' + ) + expect(separators).toHaveLength(1) + // `--border` is a divider line (7% white in dark), an order of magnitude + // fainter than the counts it sits between. + expect(separators[0].classList).not.toContain('text-border') + expect(separators[0].classList).toContain('text-muted-foreground') + // One space either side; the icon's own margin is the icon-to-label gap. + expect(header.textContent).toBe('1 agent · 1 shell') + }) + + it('carries no icon on a collapsed total, which spans kinds', () => { + const header = renderHeader([ + { id: 'a1', kind: 'agent' }, + { id: 'c1', kind: 'command' }, + { id: 'm1', kind: 'monitor' }, + { id: 'w1', kind: 'workflow' } + ]) + expect(header).toHaveAttribute('aria-label', '4 background tasks') + expect(header.querySelectorAll('svg')).toHaveLength(1) + }) +}) + +describe('settled rows beside their live siblings', () => { + // Retention is the PR's headline: a finished child stays visible, keeps the + // usage it ended on, and stops claiming a clock or a stop control. + it('keeps a settled row with its final usage, no clock and no stop', () => { + render( + {}} + /> + ) + fireEvent.click(screen.getByRole('button', { expanded: false })) + const agents = screen.getByRole('list', { name: 'Agents' }) + const rows = within(agents).getAllByRole('listitem') + expect(rows).toHaveLength(2) + // First seen first: the settled sibling started earlier. + expect(rows[0].textContent).toBe('settled child18.1k') + expect(rows[1].textContent).toMatch(/^live child4\.1k · .+Stop$/) + expect(within(rows[1]).getByRole('button', { name: 'Stop live child' })).toBeInTheDocument() + expect(within(rows[0]).queryByRole('button')).toBeNull() + }) +}) + +describe('background-task row reasons', () => { + function expandedRows(tasks: AgentSessionBackgroundTask[]): HTMLElement[] { + render( + {}} + /> + ) + fireEvent.click(screen.getByRole('button', { expanded: false })) + return screen.getAllByRole('listitem') + } + + // `unverifiable` is the SSH verdict for "no contact"; a row that hides it reads + // like a working child. `blocked` is the same class of loss. + it('names the reason on every attention state, not only on waiting', () => { + const rows = expandedRows([ + { id: 'a1', kind: 'agent', description: 'ssh child', state: 'unverifiable' }, + { id: 'a2', kind: 'agent', description: 'flaky child', state: 'blocked' }, + { id: 'a3', kind: 'agent', description: 'approval child', state: 'waiting' }, + { id: 'a4', kind: 'agent', description: 'busy child', state: 'working' } + ]) + expect(rows).toHaveLength(4) + expect(rows[0].textContent).toContain('ssh child · no contact') + expect(rows[1].textContent).toContain('flaky child · failed') + expect(rows[2].textContent).toContain('approval child · needs approval') + // A running row has nothing to explain. + expect(rows[3].textContent).not.toContain('·') + }) +}) + +it('stops elapsed renders in a hidden pane and catches up on reveal', () => { + vi.useFakeTimers() + vi.setSystemTime(100_000) + const committed = vi.fn() + const view = (isVisible: boolean) => ( + + {}} + isVisible={isVisible} + tasks={[{ id: 'shell', kind: 'command', startedAt: 1_000 }]} + settledTasks={[]} + indicatorActive + supportsTaskStop={false} + supportsStopAll={false} + stoppingTaskIds={new Set()} + stoppingAll={false} + onStop={() => {}} + /> + + ) + const { rerender, unmount } = render(view(true)) + committed.mockClear() + act(() => vi.advanceTimersByTime(1_000)) + expect(committed).toHaveBeenCalled() + rerender(view(false)) + committed.mockClear() + act(() => vi.advanceTimersByTime(10_000)) + expect(committed).not.toHaveBeenCalled() + rerender(view(true)) + committed.mockClear() + act(() => vi.advanceTimersByTime(1_000)) + expect(committed).toHaveBeenCalled() + unmount() + expect(vi.getTimerCount()).toBe(0) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx index c52b73842a9..d9bef4e6099 100644 --- a/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatBackgroundTasksStatus.tsx @@ -1,62 +1,223 @@ -import { useId, useState } from 'react' -import { ChevronDown } from 'lucide-react' +import { useEffect, useId, useMemo, useRef, useState } from 'react' +import { Activity, Bot, ChevronDown, CircleHelp, SquareTerminal, Workflow } from 'lucide-react' import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' import { AgentStateDot } from '@/components/AgentStateDot' import { Button } from '@/components/ui/button' +import { useNow } from '@/hooks/use-now' import { translate } from '@/i18n/i18n' +import { backgroundTasksHeaderContent } from './background-task-header-content' +import { + backgroundTaskElapsedLabel, + backgroundTaskGroupLabel, + backgroundTaskStateReason, + buildBackgroundTaskGroups, + formatBackgroundTaskTokens, + type BackgroundRosterTask +} from './background-task-roster' -function backgroundTaskLabel(task: AgentSessionBackgroundTask): string { - if (task.description) { - return task.description - } - switch (task.kind) { - case 'agent': - return translate('components.native-chat.backgroundTasks.agent', 'Background agent') - case 'workflow': - return translate('components.native-chat.backgroundTasks.workflow', 'Background workflow') - case 'command': - return translate('components.native-chat.backgroundTasks.command', 'Background command') - case 'monitor': - return translate('components.native-chat.backgroundTasks.monitor', 'Background monitor') - case 'unknown': - return translate('components.native-chat.backgroundTasks.task', 'Background task') - } +/** Below this strip width (border-box, live root font size) the header drops + * its per-kind breakdown for an honest total. A narrow split pane on a wide + * monitor must behave like a narrow window, so no viewport media query. */ +const NARROW_STRIP_REM = 24 + +function rootFontSizePx(): number { + const parsed = Number.parseFloat(getComputedStyle(document.documentElement).fontSize) + return Number.isFinite(parsed) && parsed > 0 ? parsed : 16 +} + +/** Observe the strip's own border-box width; the viewport is only the + * pre-measurement stand-in before the first observer callback. */ +function useNarrowStrip(ref: React.RefObject): boolean { + const [narrow, setNarrow] = useState(() => window.innerWidth < NARROW_STRIP_REM * 16) + useEffect(() => { + const element = ref.current + if (!element || typeof ResizeObserver === 'undefined') { + return + } + const observer = new ResizeObserver((observerEntries) => { + const width = + observerEntries[0]?.borderBoxSize?.[0]?.inlineSize ?? element.getBoundingClientRect().width + setNarrow(width < NARROW_STRIP_REM * rootFontSizePx()) + }) + observer.observe(element, { box: 'border-box' }) + return () => observer.disconnect() + }, [ref]) + return narrow +} + +const KIND_ICONS = { + agent: Bot, + command: SquareTerminal, + monitor: Activity, + workflow: Workflow, + unknown: CircleHelp +} as const + +/** Monitoring is a STATE the app colours the same on every surface — the agent + * sidebar and `AgentStateDot` both draw an amber heartbeat — so the strip must + * match it or the two stop reading as the same thing. The other four are plain + * kind markers and stay neutral. `dimmed` is the running-turn treatment. */ +function kindIconTone(kind: AgentSessionBackgroundTask['kind'], dimmed: boolean): string { + const tone = kind === 'monitor' ? 'text-yellow-500' : 'text-muted-foreground' + return dimmed ? `${tone}/40` : tone +} + +/** Absent means stoppable: a host predating the field published only rows its + * stop could act on, so reading absence as "not stoppable" would hide a + * working control. Only an explicit `false` withholds the button — Claude + * marks its in-turn foreground rows that way, and a Stop on one of those + * resolves to an empty target list and silently reports nothing cancelled. */ +function backgroundTaskStoppable(task: AgentSessionBackgroundTask): boolean { + return task.stoppable !== false +} + +function BackgroundTaskRow(props: { + entry: BackgroundRosterTask + now: number + supportsTaskStop: boolean + stopping: boolean + onStop: (taskId: string) => void +}): React.JSX.Element { + const { entry, now } = props + const Icon = KIND_ICONS[entry.task.kind] + // Every attention state states its reason on the row, the same ones the collapsed + // header names; `unverifiable` ("no contact") must never be silently dropped. + const reason = backgroundTaskStateReason(entry.state) + // Settled rows keep their final usage but no elapsed — a still-growing clock + // on finished work would lie. + const meta = [ + entry.task.totalTokens !== undefined + ? formatBackgroundTaskTokens(entry.task.totalTokens) + : null, + entry.settled ? null : backgroundTaskElapsedLabel(entry.task, now) + ] + .filter((part): part is string => part !== null) + .join(' · ') + return ( +
  • +
  • + ) } export function NativeChatBackgroundTasksStatus(props: { tasks: readonly AgentSessionBackgroundTask[] + settledTasks: readonly AgentSessionBackgroundTask[] supportsTaskStop: boolean /** False when the provider exposes no honest stop at all; the fallback * control is hidden rather than offering a button that cannot act. */ supportsStopAll: boolean stoppingTaskIds: ReadonlySet stoppingAll: boolean + /** True while the session is idle: only then may the strip speak as the + * animated monitoring indicator. A running turn owns the voice. */ + indicatorActive: boolean + isVisible: boolean + /** Owned by the parent. The strip is mounted on live work, so it disappears + * and comes back whenever the roster momentarily empties between two pieces + * of a sequential fan-out — settled rows are flushed at that same instant + * and hold nothing open — and local state would collapse the list on every + * such gap. */ + expanded: boolean + onExpandedChange: (expanded: boolean) => void onStop: (taskId?: string) => void }): React.JSX.Element { - const [expanded, setExpanded] = useState(false) + const expanded = props.expanded const taskListId = useId() + const stripRef = useRef(null) + const narrow = useNarrowStrip(stripRef) + // The 1 Hz elapsed tick must not re-group, re-sort and re-translate the whole roster. + const groups = useMemo( + () => buildBackgroundTaskGroups(props.tasks, props.settledTasks), + [props.tasks, props.settledTasks] + ) + const singleLiveCommand = + groups.length === 1 && groups[0].kind === 'command' && groups[0].tasks.length === 1 + const hasElapsed = groups.some((group) => + group.tasks.some((entry) => !entry.settled && (entry.task.startedAt ?? 0) > 0) + ) + const now = useNow(1_000, props.isVisible && hasElapsed && (expanded || singleLiveCommand)) + const header = backgroundTasksHeaderContent(groups, { narrow, now }) + const headerText = `${header.segments.map((segment) => segment.text).join(' · ')}${header.detail ? `${header.segments.length > 0 ? ' — ' : ''}${header.detail}` : ''}` return (
    -
    +
    - ) : null} - - ) - })} - + ))} + +
    + )) ) : (

    {translate( @@ -119,7 +266,7 @@ export function NativeChatBackgroundTasksStatus(props: {

    )} {!props.supportsTaskStop && props.supportsStopAll ? ( -
    0 ? 'mt-2 border-t border-border pt-2' : 'mt-2'}> +
    0 ? 'mt-2 border-t border-border pt-2' : 'mt-2'}> +
    +
    + {hasMore ? ( +
    + +
    + ) : null} + + {showTurnStatus && isWorking ? ( + + ) : null} + {!showTurnStatus && showTypingIndicator ? : null}
    - ) : null} - {messages.map((message, index) => { - const turnKey = turnKeys[index] - const isCurrentTurn = currentTurnKey - ? turnKey === currentTurnKey - : turnKey === undefined - const status = - index === latestUserIndex - ? turnStatuses.active - : message.role === 'user' && turnKey - ? turnStatuses.completedByTurn[turnKey] - : undefined - const receipt = receipts.get(message.id) - const turnDiff = - turnKey && turnKeys[index + 1] !== turnKey ? turnDiffs.get(turnKey) : undefined - return ( - - {receipt ? ( - - ) : ( - - )} - {showTurnStatus && - status && - (index !== latestUserIndex || showTypingIndicator || !isWorking) ? ( - toggleExpandedTurn(turnKey) - : undefined - } - /> - ) : null} - {turnDiff ? ( - - ) : null} - - ) - })} - {showTurnStatus && - latestUserIndex === -1 && - turnStatuses.active && - showTypingIndicator ? ( - - ) : null} - {showTurnStatus && isWorking ? ( - - ) : null} - {!showTurnStatus && showTypingIndicator ? : null} +
    + {showJump ? ( + + ) : null}
    - {showJump ? ( - + {taskListState.list && taskListState.list.tasks.length > 0 ? ( +
    +
    + +
    +
    ) : null}
    - {taskListState.list && taskListState.list.tasks.length > 0 ? ( -
    -
    - -
    -
    - ) : null} -
    + ) } diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.turn-history.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.turn-history.test.tsx index 628d78b4c58..6447dfcbec0 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.turn-history.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.turn-history.test.tsx @@ -1,7 +1,7 @@ // @vitest-environment happy-dom import '@testing-library/jest-dom/vitest' import { cleanup, fireEvent, render, screen } from '@testing-library/react' -import { afterEach, describe, expect, it, vi } from 'vitest' +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' import type { AgentJournalItemBody, AgentJournalRenderItem @@ -9,8 +9,14 @@ import type { import { projectStructuredItemsToNativeChat } from '../../../../shared/structured-agent-session-projection' import { NativeChatMessageList } from './NativeChatMessageList' import type { NativeChatLiveSession } from './use-native-chat-live-session' +import { installNativeChatMessageListTestViewport } from './native-chat-message-list-test-viewport' const scrollTo = vi.fn() +let restoreViewport = (): void => {} +beforeAll(() => { + restoreViewport = installNativeChatMessageListTestViewport() +}) +afterAll(() => restoreViewport()) afterEach(() => { cleanup() vi.restoreAllMocks() diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.turn-indicator.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.turn-indicator.test.tsx new file mode 100644 index 00000000000..ab706436f91 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.turn-indicator.test.tsx @@ -0,0 +1,563 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' +import type { NativeChatLiveSession } from './use-native-chat-live-session' +import { NativeChatMessageList } from './NativeChatMessageList' +import { installNativeChatMessageListTestViewport } from './native-chat-message-list-test-viewport' +import type { + AgentJournalItemBody, + AgentJournalRenderItem +} from '../../../../shared/agent-session-journal-types' + +// The turn record this host writes, and the legacy status row an older host sends. +const turnItem: AgentJournalItemBody = { kind: 'turn', turnId: 'turn-1', state: 'running' } +const legacyTurnRow: AgentJournalItemBody = { + kind: 'status', + text: 'Codex is working…', + turnLifecycle: { turnId: 'turn-1', state: 'running' } +} +const reasoningRow: AgentJournalItemBody = { + kind: 'message', + role: 'reasoning', + blocks: [{ type: 'text', text: '' }] +} + +function journalItem(sequence: number, body: AgentJournalItemBody): AgentJournalRenderItem { + return { itemId: `item-${sequence}`, revision: 1, sequence, observedAt: sequence, body } +} + +let restoreViewport = (): void => {} +beforeAll(() => { + restoreViewport = installNativeChatMessageListTestViewport() +}) +afterAll(() => restoreViewport()) +afterEach(cleanup) + +const session: NativeChatLiveSession = { + messages: [ + { + id: 'assistant-1', + role: 'assistant', + blocks: [{ type: 'text', text: 'Selectable agent response.' }], + timestamp: 1, + source: 'transcript' + } + ], + status: 'ready', + sessionId: 'session-1', + agent: 'codex', + hasMore: false, + loadingEarlier: false, + loadEarlier: vi.fn(), + readPhase: 'ready' +} + +// The live turn renders exactly one indicator row; a settled turn keeps its own. +describe('NativeChatMessageList turn indicator', () => { + it('keeps a reduced-motion-safe spinner on the live row of a no-tool Codex turn', () => { + render( + + ) + + const activity = screen.getByText('Working for 0s') + const row = activity.closest('[data-native-chat-turn-activity]') + const spinner = row?.querySelector('svg') + expect(activity).not.toHaveClass('animate-pulse', 'animate-spin') + expect(spinner).toHaveClass('size-4', 'animate-spin', 'motion-reduce:animate-none') + expect(row).toHaveAttribute('aria-live', 'polite') + expect(screen.getByText('The answer is still streaming.').compareDocumentPosition(row!)).toBe( + Node.DOCUMENT_POSITION_FOLLOWING + ) + }) + + it('keeps the live row distinct from the running tool row', () => { + render( + + ) + + const toolLabel = screen.getByText('Running pnpm test') + expect(toolLabel).toHaveClass('animate-pulse') + expect(screen.getAllByText('Running pnpm test')).toHaveLength(1) + const activity = screen.getByText('Working for 0s') + expect(activity.textContent).not.toBe(toolLabel.textContent) + expect(activity).not.toHaveTextContent('shell') + expect(activity).not.toHaveTextContent('pnpm test') + const spinner = activity.closest('[data-native-chat-turn-activity]')?.querySelector('svg') + expect(activity).not.toHaveClass('animate-pulse', 'animate-spin') + expect(spinner).toHaveClass('animate-spin', 'motion-reduce:animate-none') + }) + + it('keeps the live row up after a tool settles', () => { + render( + + ) + + const settledTool = screen.getByText('shell') + const activity = screen.getByText('Working for 0s') + expect(activity.textContent).not.toBe(settledTool.textContent) + expect(activity).not.toHaveTextContent('shell') + expect(activity).not.toHaveTextContent('pnpm test') + expect(activity).not.toHaveClass('animate-pulse', 'animate-spin') + expect(activity.closest('[data-native-chat-turn-activity]')?.querySelector('svg')).toHaveClass( + 'animate-spin' + ) + }) + + it('keeps a completed tool row static while the turn tail spins, then removes the tail', () => { + const workingSession: NativeChatLiveSession = { + ...session, + status: 'working', + messages: [ + { + id: 'assistant-settled-tool', + role: 'assistant', + blocks: [ + { + type: 'tool-call', + name: 'shell', + input: { command: 'pnpm test' }, + state: 'completed' + }, + { type: 'tool-result', output: 'passed' } + ], + timestamp: 1, + source: 'transcript' + } + ] + } + const { container, rerender } = render( + + ) + + const settledTool = screen.getByText('shell') + expect(settledTool).toHaveTextContent('shell pnpm test') + expect(settledTool.closest('button')?.querySelector('.animate-pulse')).toBeNull() + expect(settledTool.closest('button')?.querySelector('.lucide-check')).toBeInTheDocument() + const activity = screen.getByText('Preparing the answer') + expect(activity).not.toHaveClass('animate-pulse', 'animate-spin') + expect(activity.closest('[data-native-chat-turn-activity]')?.querySelector('svg')).toHaveClass( + 'animate-spin' + ) + + rerender( + + ) + + expect(container.querySelector('[data-native-chat-turn-activity]')).toBeNull() + expect(container.querySelector('.animate-pulse')).toBeNull() + expect(container.querySelector('.animate-spin')).toBeNull() + }) + + it('keeps bridge chats on the legacy activity chrome', () => { + render( + + ) + + expect(screen.queryByText('Thinking')).toBeNull() + expect(screen.queryByRole('button', { name: 'Toggle turn details' })).toBeNull() + expect(screen.queryByText('Running sleep 5')).toBeNull() + expect(document.querySelectorAll('.animate-bounce')).toHaveLength(3) + }) + + it('reads "Thinking" on the one live row while the turn is reasoning', () => { + const { container } = render( + + ) + + const user = screen.getByText('Start the task') + const thinking = screen.getByText('Thinking') + expect(user.compareDocumentPosition(thinking)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) + // One indicator, not a "Thinking" row stacked above a spinning "Working…" row. + expect(container.querySelectorAll('[data-native-chat-turn-status]')).toHaveLength(1) + expect(thinking.closest('[data-native-chat-turn-activity]')?.querySelector('svg')).toHaveClass( + 'animate-spin' + ) + expect(container.querySelector('.animate-bounce')).toBeNull() + }) + + it('does not reuse completed-turn reasoning while the next dispatch is pending', () => { + render( + + ) + + expect(screen.queryByText('Thinking')).toBeNull() + expect(screen.getByText('Working for 0s')).toBeInTheDocument() + }) + + it('lets provider activity text beat the reasoning label on the same single row', () => { + const { container } = render( + + ) + + expect(screen.getByText('Exploring the repo layout')).toBeInTheDocument() + expect(screen.queryByText('Thinking')).toBeNull() + expect(container.querySelectorAll('[data-native-chat-turn-status]')).toHaveLength(1) + }) + + it('places the one live row after the newest content in the turn', () => { + render( + + ) + + const status = screen.getByText('Working for 0s') + const assistant = screen.getByText('I am checking now.') + // The live row trails the newest content instead of sitting under the prompt. + expect(assistant.compareDocumentPosition(status)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) + expect(document.querySelectorAll('[data-native-chat-turn-status]')).toHaveLength(1) + }) + + it('shows elapsed working time once tool activity starts', () => { + render( + + ) + + expect(screen.getByText('Working for 3s')).toBeInTheDocument() + }) + + it('keeps the completed duration below the user message', () => { + const startedAt = Date.now() - 3000 + const turnSession: NativeChatLiveSession = { + ...session, + status: 'working', + messages: [ + { + id: 'user-complete', + role: 'user', + blocks: [{ type: 'text', text: 'Complete this task' }], + timestamp: startedAt, + source: 'transcript' + }, + { + id: 'assistant-complete', + role: 'assistant', + blocks: [{ type: 'text', text: 'Task complete.' }], + timestamp: Date.now(), + source: 'transcript' + } + ] + } + const { rerender } = render( + + ) + + rerender( + + ) + + const user = screen.getByText('Complete this task') + const status = screen.getByText('Worked for 3s') + const assistant = screen.getByText('Task complete.') + expect(user.compareDocumentPosition(status)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) + expect(status.compareDocumentPosition(assistant)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) + + rerender( + + ) + + expect(screen.getByText('Worked for 3s')).toBeInTheDocument() + expect(screen.getByText('Working for 0s')).toBeInTheDocument() + }) + + it("uses the completed caret to expand that turn's tool details", () => { + const startedAt = Date.now() - 3000 + render( + + ) + + const status = screen.getByRole('button', { name: 'Toggle turn details' }) + expect(status).toHaveAttribute('aria-expanded', 'false') + expect(screen.queryByRole('button', { name: /1× shell/ })).toBeNull() + fireEvent.click(status) + expect(status).toHaveAttribute('aria-expanded', 'true') + const tool = screen.getByRole('button', { name: /1× shell/ }) + expect(tool).toHaveAttribute('aria-expanded', 'true') + expect(screen.getAllByRole('button', { name: /shell pwd/ })[1]).toHaveAttribute( + 'aria-expanded', + 'false' + ) + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx new file mode 100644 index 00000000000..2eacc13ed07 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.turn-timing.test.tsx @@ -0,0 +1,112 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { cleanup, render, screen } from '@testing-library/react' +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' +import type { NativeChatLiveSession } from './use-native-chat-live-session' +import { NativeChatMessageList } from './NativeChatMessageList' +import { installNativeChatMessageListTestViewport } from './native-chat-message-list-test-viewport' + +afterEach(cleanup) + +let restoreViewport = (): void => {} + +beforeAll(() => { + restoreViewport = installNativeChatMessageListTestViewport() +}) + +afterAll(() => restoreViewport()) + +const session: NativeChatLiveSession = { + messages: [ + { + id: 'user-settled', + role: 'user', + blocks: [{ type: 'text', text: 'Settled on the host' }], + timestamp: 1, + source: 'transcript' + }, + { + id: 'assistant-settled', + role: 'assistant', + blocks: [{ type: 'text', text: 'Done.' }], + timestamp: 2, + source: 'transcript' + } + ], + status: 'ready', + sessionId: 'session-1', + agent: 'codex', + hasMore: false, + loadingEarlier: false, + loadEarlier: vi.fn(), + readPhase: 'ready' +} + +const settledTurns = new Map([['user-settled', { startedAt: 1, workedSeconds: 197 }]]) + +describe('NativeChatMessageList host-settled turn timing', () => { + it('does not render a local completed duration when the host cannot verify the end', () => { + const now = vi.spyOn(Date, 'now').mockReturnValue(1_000) + const unknownTurns = new Map([['user-settled', null]]) + try { + const { rerender } = render( + + ) + now.mockReturnValue(60_000) + rerender( + + ) + expect(screen.queryByText(/Worked for/)).not.toBeInTheDocument() + } finally { + now.mockRestore() + } + }) + + it('renders a host-settled duration without ever clocking the turn locally', () => { + // A local clock nowhere near the host's: the value must still be the host's. + const now = vi.spyOn(Date, 'now').mockReturnValue(1_700_000_000_000) + try { + const { rerender } = render( + + ) + expect(screen.getByText('Worked for 3m 17s')).toBeInTheDocument() + now.mockReturnValue(1_700_000_099_000) + rerender( + + ) + expect(screen.getByText('Worked for 3m 17s')).toBeInTheDocument() + } finally { + now.mockRestore() + } + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.windowing.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.windowing.test.tsx new file mode 100644 index 00000000000..c4de79ceeed --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.windowing.test.tsx @@ -0,0 +1,640 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { act, cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalRenderItem +} from '../../../../shared/agent-session-journal-types' +import { projectStructuredItemsToNativeChat } from '../../../../shared/structured-agent-session-projection' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import type { NativeChatLiveSession } from './use-native-chat-live-session' +import { NativeChatMessageList } from './NativeChatMessageList' +import { NATIVE_CHAT_BOTTOM_THRESHOLD_PX } from './native-chat-autoscroll' +import { + estimateNativeChatRowHeight, + NATIVE_CHAT_ROW_GAP_PX, + nativeChatRowContentMetrics +} from './native-chat-row-height-estimate' + +afterEach(cleanup) + +const VIEWPORT_PX = 600 +const TRANSCRIPT_LENGTH = 200 + +/** Everything the document holds below the last row: the transcript column's + * trailing chrome and the scroll root's bottom padding. Non-zero on purpose — + * the document's bottom sits past the window's last row, which is exactly where + * a pin computed from the virtualizer's totals and one computed from the + * document disagree. */ +const BELOW_TRANSCRIPT_PX = 24 + +/** Heights the stubbed layout reports per row index, when a case wants a row to + * measure as something other than its estimate. Empty means "every row at its + * estimate", which is what every non-growth case wants. */ +let measuredRowHeights: readonly number[] = [] + +function marker(index: number): NativeChatMessage { + return { + id: `message-${index}`, + role: 'assistant', + blocks: [{ type: 'text', text: `marker-${index}` }], + timestamp: index + 1, + source: 'transcript' + } +} + +const ROW_PX = estimateNativeChatRowHeight(nativeChatRowContentMetrics(marker(0)), { + hasReceipt: false, + hasStatus: false, + hasTurnDiff: false +}) +const ROW_PITCH_PX = ROW_PX + NATIVE_CHAT_ROW_GAP_PX + +/** Replace a layout property on every element, and hand back the undo. */ +function overrideLayoutProperty(name: string, descriptor: PropertyDescriptor): () => void { + const original = Object.getOwnPropertyDescriptor(HTMLElement.prototype, name) + Object.defineProperty(HTMLElement.prototype, name, { configurable: true, ...descriptor }) + return () => { + if (original) { + Object.defineProperty(HTMLElement.prototype, name, original) + } else { + Reflect.deleteProperty(HTMLElement.prototype, name) + } + } +} + +/** The spacer's reserved height, which is the transcript's whole rendered height: + * windowed rows are absolutely positioned inside it, so a row growing in place + * reaches the document only through the height the window reserves for it. */ +function reservedTranscriptHeight(root: ParentNode): number { + const spacer = root.querySelector('[data-native-chat-window]') + return spacer ? Number.parseFloat(spacer.style.height) || 0 : 0 +} + +// The virtualizer measures with `offsetHeight` — not `clientHeight`, not a +// bounding rect — so that is the one thing a DOM without layout has to answer +// for windowing to engage at all. Rows report the height their own estimate +// predicted, which keeps the totals exact and independent of which rows happen +// to have been mounted long enough to be measured; `measuredRowHeights` is how a +// case says a row measures as something else. +// +// `scrollGeometry` additionally gives the scroll root a document to scroll: a +// height, a viewport, and a `scrollTop` that clamps the way a real one does. +// Off by default, because a transcript with a real document opens pinned to its +// bottom and the cases above are about where the window sits, not where it lands. +function stubLayout({ + scrollGeometry = false, + viewportHeight = () => VIEWPORT_PX +}: { + scrollGeometry?: boolean + viewportHeight?: () => number +} = {}): () => void { + const scrollTops = new WeakMap() + const restores = [ + overrideLayoutProperty('offsetHeight', { + get(this: HTMLElement): number { + if (this.hasAttribute('data-native-chat-scroll')) { + return viewportHeight() + } + if (this.hasAttribute('data-native-chat-window')) { + return reservedTranscriptHeight(this.parentElement ?? this) + } + const index = this.dataset.index + if (index !== undefined) { + return measuredRowHeights[Number(index)] ?? ROW_PX + } + // The transcript column: as tall as the window it wraps, plus what sits + // under it. This is the element the list observes for streamed growth. + return this.classList.contains('max-w-4xl') + ? reservedTranscriptHeight(this) + BELOW_TRANSCRIPT_PX + : 0 + } + }) + ] + if (scrollGeometry) { + restores.push( + overrideLayoutProperty('clientHeight', { + get(this: HTMLElement): number { + return this.hasAttribute('data-native-chat-scroll') ? viewportHeight() : 0 + } + }), + overrideLayoutProperty('scrollHeight', { + get(this: HTMLElement): number { + return this.hasAttribute('data-native-chat-scroll') + ? reservedTranscriptHeight(this) + BELOW_TRANSCRIPT_PX + : 0 + } + }), + overrideLayoutProperty('scrollTop', { + get(this: HTMLElement): number { + return scrollTops.get(this) ?? 0 + }, + set(this: HTMLElement, value: number): void { + // A browser clamps; without this `scrollTop = scrollHeight` would park + // the view past the end and every distance-from-bottom would read 0. + const max = Math.max(0, this.scrollHeight - this.clientHeight) + scrollTops.set(this, Math.min(Math.max(0, value), max)) + } + }) + ) + } + return () => { + for (const restore of restores.toReversed()) { + restore() + } + } +} + +type FakeResizeObservation = { + callback: ResizeObserverCallback + /** Target -> height last delivered. -1 means "never", so the first flush + * delivers, the way a real observer's initial callback does. */ + observed: Map +} + +const resizeObservations = new Set() + +/** happy-dom's ResizeObserver never fires, so nothing that re-measures ever runs. + * This one records what production observes and delivers only when a target's + * height actually changed — the browser's own rule — and only when a test says + * a frame was painted. Entries carry no `borderBoxSize`, so the virtualizer + * falls back to `offsetHeight`, which is the path being modelled. */ +function stubResizeObserver(): () => void { + const original = window.ResizeObserver + class TestResizeObserver { + private readonly observation: FakeResizeObservation + constructor(callback: ResizeObserverCallback) { + this.observation = { callback, observed: new Map() } + resizeObservations.add(this.observation) + } + observe(target: Element): void { + this.observation.observed.set(target, -1) + } + unobserve(target: Element): void { + this.observation.observed.delete(target) + } + disconnect(): void { + this.observation.observed.clear() + resizeObservations.delete(this.observation) + } + } + window.ResizeObserver = TestResizeObserver as unknown as typeof ResizeObserver + return () => { + resizeObservations.clear() + window.ResizeObserver = original + } +} + +/** Deliver one round of resize callbacks; true when anything was delivered. */ +function deliverResizes(): boolean { + let delivered = false + // A copy: a callback may disconnect its own observer mid-delivery. + for (const observation of Array.from(resizeObservations)) { + const entries: ResizeObserverEntry[] = [] + for (const [target, lastHeight] of observation.observed) { + const height = (target as HTMLElement).offsetHeight + if (height !== lastHeight) { + observation.observed.set(target, height) + entries.push({ target } as unknown as ResizeObserverEntry) + } + } + if (entries.length > 0) { + delivered = true + observation.callback(entries, undefined as unknown as ResizeObserver) + } + } + return delivered +} + +function session(messages: NativeChatMessage[]): NativeChatLiveSession { + return { + messages, + status: 'ready', + sessionId: 'session-1', + agent: 'codex', + hasMore: false, + loadingEarlier: false, + loadEarlier: vi.fn(), + readPhase: 'ready' + } +} + +function list(messages: NativeChatMessage[]): React.JSX.Element { + return ( + + ) +} + +/** Reads the window, and refuses to pass if there is no window to read. + * + * Without this a change to the usability gate would quietly send every case + * below down the whole-transcript path, where "fewer rows than messages" is + * false but every other assertion still holds. */ +function windowState(container: HTMLElement): { totalSize: number; indexes: number[] } { + const spacer = container.querySelector('[data-native-chat-window]') + if (!spacer) { + throw new Error('transcript is not windowed: no spacer, every row is mounted') + } + const totalSize = Number.parseFloat(spacer.style.height) + if (!(totalSize > 0)) { + throw new Error(`transcript reserved no height (${spacer.style.height})`) + } + return { + totalSize, + indexes: Array.from(container.querySelectorAll('[data-index]')) + .map((row) => Number(row.dataset.index)) + .sort((left, right) => left - right) + } +} + +/** happy-dom fires no scroll event for an assignment to `scrollTop`. */ +function scrollTranscript(container: HTMLElement, top: number): void { + const scroller = container.querySelector('[data-native-chat-scroll]') + if (!scroller) { + throw new Error('no transcript scroll root') + } + scroller.scrollTop = top + fireEvent.scroll(scroller) +} + +describe('windowed transcript', () => { + let restoreLayout = (): void => {} + beforeEach(() => { + restoreLayout = stubLayout() + }) + afterEach(() => { + restoreLayout() + }) + + const transcript = Array.from({ length: TRANSCRIPT_LENGTH }, (_, index) => marker(index)) + + it('mounts a window over the transcript rather than all of it', () => { + const { container } = render(list(transcript)) + const { indexes } = windowState(container) + + expect(indexes.length).toBeGreaterThan(0) + expect(indexes.length).toBeLessThan(TRANSCRIPT_LENGTH / 4) + expect(indexes).toContain(0) + expect(screen.getByText('marker-0')).toBeInTheDocument() + expect(screen.queryByText(`marker-${TRANSCRIPT_LENGTH - 2}`)).toBeNull() + }) + + // One gap per pair of rows, and none after the last one. The other half of + // this — that a row's own reservation does not include the gap as well — is + // pinned on the estimate itself, where it can be seen without layout. + it('reserves each row once and one gap between each pair', () => { + const { container } = render(list(transcript)) + + expect(windowState(container).totalSize).toBe( + TRANSCRIPT_LENGTH * ROW_PX + (TRANSCRIPT_LENGTH - 1) * NATIVE_CHAT_ROW_GAP_PX + ) + }) + + it('moves the mounted rows to bracket the offset the reader scrolled to', () => { + const { container } = render(list(transcript)) + const offset = 5000 + scrollTranscript(container, offset) + const { indexes } = windowState(container) + const focused = Math.floor(offset / ROW_PITCH_PX) + + expect(indexes).toContain(focused) + expect(indexes[0]).toBeLessThanOrEqual(focused) + expect(indexes.at(-1)).toBeGreaterThanOrEqual(focused) + expect(indexes).not.toContain(0) + expect(indexes.length).toBeLessThan(TRANSCRIPT_LENGTH / 4) + }) + + // The live row announces a running tool through `aria-live`, which says nothing + // from a row that is not in the document. + it('keeps the newest row mounted after the reader scrolls away from it', () => { + const { container } = render(list(transcript)) + scrollTranscript(container, 5000) + + expect(windowState(container).indexes).toContain(TRANSCRIPT_LENGTH - 1) + }) + + it('gives no slot to a message that draws nothing', () => { + const withBlanks = Array.from({ length: TRANSCRIPT_LENGTH }, (_, index) => + index % 4 === 0 + ? { ...marker(index), blocks: [{ type: 'text' as const, text: '' }] } + : marker(index) + ) + const drawn = TRANSCRIPT_LENGTH - TRANSCRIPT_LENGTH / 4 + const { container } = render(list(withBlanks)) + const { totalSize, indexes } = windowState(container) + + expect(totalSize).toBe(drawn * ROW_PX + (drawn - 1) * NATIVE_CHAT_ROW_GAP_PX) + expect(indexes.at(-1)).toBeLessThanOrEqual(drawn - 1) + }) + + it('still has the tool run open when the row carrying it comes back', () => { + const withTool = [...transcript] + withTool[1] = { + ...marker(1), + blocks: [ + { type: 'text', text: 'marker-1' }, + { type: 'tool-call', name: 'shell', input: { command: 'ls' }, state: 'completed' } + ] + } + const { container } = render(list(withTool)) + + const header = screen.getByRole('button', { name: /1×/ }) + expect(header).toHaveAttribute('aria-expanded', 'false') + fireEvent.click(header) + expect(screen.getByRole('button', { name: /1×/ })).toHaveAttribute('aria-expanded', 'true') + + scrollTranscript(container, 5000) + expect(windowState(container).indexes).not.toContain(1) + expect(screen.queryByRole('button', { name: /1×/ })).toBeNull() + + scrollTranscript(container, 0) + expect(screen.getByRole('button', { name: /1×/ })).toHaveAttribute('aria-expanded', 'true') + }) +}) + +// The reveal chain runs message -> tool run -> diff card and lands on a card in +// a DIFFERENT, earlier message than the rollup that was clicked. Under windowing +// that message may not be mounted to be pointed at, so the reveal names it by id +// and the row is pinned into the window until the card can answer for itself. +describe('revealing a diff from a turn rollup', () => { + let restoreLayout = (): void => {} + beforeEach(() => { + restoreLayout = stubLayout() + }) + afterEach(() => { + restoreLayout() + vi.restoreAllMocks() + }) + + function journalItem(itemId: string, body: AgentJournalItemBody, sequence: number) { + return { itemId, body, sequence, observedAt: sequence * 1000, revision: 1 } + } + + const patch = '@@ -1 +1 @@\n-before\n+after' + const items: AgentJournalRenderItem[] = [ + journalItem( + 'user', + { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'Edit it' }] }, + 1 + ), + journalItem( + 'diff', + { + kind: 'diff', + path: 'src/a.ts', + patch: { head: patch, truncated: false, digest: 'fixture', byteLength: patch.length } + }, + 2 + ), + ...Array.from({ length: TRANSCRIPT_LENGTH }, (_, index) => + journalItem( + `tail-${index}`, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: `marker-${index}` }] }, + index + 3 + ) + ) + ] + + it('mounts the row a reveal names even when the window has left it behind', () => { + const scrollTo = vi.fn() + vi.spyOn(HTMLElement.prototype, 'scrollTo').mockImplementation(scrollTo) + const { container } = render( + + ) + // The rollup rides the turn's last row, which is pinned; the diff it points + // at is near the top and long gone from the window. + scrollTranscript(container, 4000) + expect(screen.queryByText('Edited file')).toBeNull() + const mountedBefore = windowState(container).indexes.length + + fireEvent.click(screen.getByRole('button', { name: /1 changed file/ })) + scrollTo.mockClear() + fireEvent.click(screen.getByRole('button', { name: /src\/a.ts/ })) + + expect(screen.getByText('Edited file')).toBeInTheDocument() + expect(screen.getByText('after')).toBeInTheDocument() + expect(scrollTo).toHaveBeenCalled() + // Pinned, not paged to: the window is still a window. + expect(windowState(container).indexes.length).toBeLessThanOrEqual(mountedBefore + 2) + }) +}) + +describe('transcript with a hidden scroll root', () => { + const transcript = Array.from({ length: 40 }, (_, index) => marker(index)) + + it('keeps the transcript bounded and rehydrates when the viewport becomes measurable', () => { + let viewportHeight = 0 + const restoreLayout = stubLayout({ viewportHeight: () => viewportHeight }) + const restoreResizeObserver = stubResizeObserver() + try { + const { container } = render(list(transcript)) + + expect(container.querySelector('[data-native-chat-window]')).toBeInTheDocument() + expect(container.querySelectorAll('[data-index]')).toHaveLength(0) + expect(screen.queryByText(/^marker-/)).toBeNull() + const column = container.querySelector('.max-w-4xl') + expect(column?.children).toHaveLength(1) + + viewportHeight = VIEWPORT_PX + act(() => { + deliverResizes() + }) + const { indexes } = windowState(container) + expect(indexes.length).toBeGreaterThan(0) + expect(indexes.length).toBeLessThan(transcript.length) + } finally { + restoreResizeObserver() + restoreLayout() + } + }) +}) + +// A row that grows in place: the same message id, more content, a taller measured +// box — what a streaming reply looks like to the window. Whole-message appends +// arrive at their final height and are a different case; this is the one where +// the row the reader is looking at keeps changing size underneath them. +// +// Two mechanisms are supposed to hold the pin, and both are exercised here: the +// list's own resize observer on the transcript column (which re-runs +// `scrollToBottom` against the document) and the virtualizer's end anchor (which +// compensates `scrollTop` by the growth when the view was already at the end). +describe('a row growing in place while the view is pinned to the bottom', () => { + const TAIL_INDEX = TRANSCRIPT_LENGTH - 1 + const GROWTH_STEPS = 24 + const LINES_PER_STEP = 12 + /** One wrapped prose line. Content and measured height grow from this one + * number, so a step that adds lines is a step that adds pixels. */ + const STREAM_LINE_PX = 22 + /** Every row but the growing one measures at its estimate, so the reserved + * total is arithmetic rather than a snapshot. */ + const BASE_TOTAL_PX = + (TRANSCRIPT_LENGTH - 1) * ROW_PX + (TRANSCRIPT_LENGTH - 1) * NATIVE_CHAT_ROW_GAP_PX + + /** Fixed so a re-render never restamps the turn and moves the status row. */ + const TURN_STARTED_AT = Date.now() + + const transcript = Array.from({ length: TRANSCRIPT_LENGTH }, (_, index) => marker(index)) + + function tailHeightAt(step: number): number { + return Math.max(ROW_PX, (1 + step * LINES_PER_STEP) * STREAM_LINE_PX) + } + + function transcriptAt(step: number): NativeChatMessage[] { + const lines = Array.from( + { length: step * LINES_PER_STEP }, + (_, index) => `streamed line ${index}` + ) + const next = [...transcript] + next[TAIL_INDEX] = { + ...marker(TAIL_INDEX), + blocks: [{ type: 'text', text: [`marker-${TAIL_INDEX}`, ...lines].join('\n') }] + } + return next + } + + function streamingList(step: number): React.JSX.Element { + return ( + + ) + } + + function scrollRoot(container: HTMLElement): HTMLElement { + const scroller = container.querySelector('[data-native-chat-scroll]') + if (!scroller) { + throw new Error('no transcript scroll root') + } + return scroller + } + + /** One painted frame, repeated to a fixed point: deliver the resize callbacks + * the growth caused, then fire the scroll event a browser fires for any + * `scrollTop` the code wrote itself. Refusing to settle is a failure in its + * own right — that is the view oscillating. */ + function paint(container: HTMLElement): void { + const scroller = scrollRoot(container) + let lastScrollTop = scroller.scrollTop + for (let pass = 0; pass < 12; pass += 1) { + let changed = false + act(() => { + changed = deliverResizes() + }) + if (scroller.scrollTop !== lastScrollTop) { + lastScrollTop = scroller.scrollTop + fireEvent.scroll(scroller) + changed = true + } + if (!changed) { + return + } + } + throw new Error('the transcript never settled: resize and scroll kept moving it') + } + + function distanceFromBottom(container: HTMLElement): number { + const scroller = scrollRoot(container) + return scroller.scrollHeight - scroller.clientHeight - scroller.scrollTop + } + + function setMeasuredTail(step: number): void { + const heights = Array.from({ length: TRANSCRIPT_LENGTH }, () => ROW_PX) + heights[TAIL_INDEX] = tailHeightAt(step) + measuredRowHeights = heights + } + + let restoreLayout = (): void => {} + let restoreResizeObserver = (): void => {} + beforeEach(() => { + restoreLayout = stubLayout({ scrollGeometry: true }) + restoreResizeObserver = stubResizeObserver() + setMeasuredTail(0) + }) + afterEach(() => { + restoreResizeObserver() + restoreLayout() + measuredRowHeights = [] + }) + + it('holds the pin, the mount and the reserved total at every frame of the growth', () => { + setMeasuredTail(0) + const { container, rerender } = render(streamingList(0)) + paint(container) + + expect(distanceFromBottom(container)).toBeLessThanOrEqual(NATIVE_CHAT_BOTTOM_THRESHOLD_PX) + expect(windowState(container).totalSize).toBe(BASE_TOTAL_PX + tailHeightAt(0)) + + const frames: { step: number; tail: number; total: number; distance: number }[] = [] + for (let step = 1; step <= GROWTH_STEPS; step += 1) { + setMeasuredTail(step) + rerender(streamingList(step)) + paint(container) + + const { totalSize, indexes } = windowState(container) + const distance = distanceFromBottom(container) + frames.push({ step, tail: tailHeightAt(step), total: totalSize, distance }) + + // Pinned: the reader is still looking at the bottom of the row. + expect(distance).toBeLessThanOrEqual(NATIVE_CHAT_BOTTOM_THRESHOLD_PX) + // Mounted: never swapped for reserved space while it is the live row. + expect(indexes).toContain(TAIL_INDEX) + expect(screen.getByText(/streamed line 0/)).toBeInTheDocument() + // Tracking: the reservation follows the measurement, not the estimate. + expect(totalSize).toBe(BASE_TOTAL_PX + tailHeightAt(step)) + // Still a window, not the whole transcript remounted by the growth. + expect(indexes.length).toBeLessThan(TRANSCRIPT_LENGTH / 4) + } + + expect(frames).toHaveLength(GROWTH_STEPS) + expect(frames.at(-1)?.tail).toBeGreaterThan(VIEWPORT_PX * 10) + expect(Math.max(...frames.map((frame) => frame.distance))).toBeLessThanOrEqual( + NATIVE_CHAT_BOTTOM_THRESHOLD_PX + ) + }) + + it('leaves a reader who scrolled up where they were, however far the row grows', () => { + setMeasuredTail(4) + const { container, rerender } = render(streamingList(4)) + paint(container) + + const readingAt = 2000 + scrollTranscript(container, readingAt) + paint(container) + expect(distanceFromBottom(container)).toBeGreaterThan(NATIVE_CHAT_BOTTOM_THRESHOLD_PX) + expect(screen.getByRole('button', { name: /jump to latest/i })).toBeInTheDocument() + + for (let step = 5; step <= GROWTH_STEPS; step += 1) { + setMeasuredTail(step) + rerender(streamingList(step)) + paint(container) + + const { totalSize, indexes } = windowState(container) + // Not yanked: the offset the reader chose is the offset they still have. + expect(scrollRoot(container).scrollTop).toBe(readingAt) + // The row is off screen but still measured, which is what keeps the + // reserved total — and so the scrollbar — honest while it grows. + expect(indexes).toContain(TAIL_INDEX) + expect(totalSize).toBe(BASE_TOTAL_PX + tailHeightAt(step)) + } + + expect(screen.getByRole('button', { name: /jump to latest/i })).toBeInTheDocument() + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx b/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx index 773e6f2c2ff..d73772dfaf9 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageRow.tsx @@ -1,23 +1,17 @@ -import { memo, useCallback, useMemo, useRef } from 'react' +import { memo, useCallback, useRef } from 'react' import CommentMarkdown, { type CommentMarkdownLinkClickHandler } from '@/components/sidebar/CommentMarkdown' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' -import { - isSubagentGroupFallbackText, - subagentGroupBlocks -} from '../../../../shared/native-chat-subagent-summary' -import { - isSubagentGroupBlock, - type NativeChatMessage, - type NativeChatToolCallBlock +import type { + NativeChatMessage, + NativeChatToolCallBlock } from '../../../../shared/native-chat-types' -import { splitNativeChatBlocks } from './native-chat-tool-fold' +import { deriveNativeChatRowContent } from './native-chat-row-content' import { NativeChatToolRun } from './NativeChatToolRun' import { NativeChatNoticeRow } from './NativeChatNoticeRow' import { NativeChatMessageTimestamp } from './NativeChatMessageTimestamp' -import { nativeChatProseToMarkdown } from './native-chat-prose' import { NativeChatAgentControls, NativeChatImageAttachments, @@ -62,32 +56,11 @@ export const MessageRow = memo(function MessageRow({ runtimeContext?: RuntimeFileOperationArgs | null }): React.JSX.Element | null { const rowRef = useRef(null) - // One pass per block set: a streaming turn re-renders this row on every frame, and these - // derivations used to re-run each time even though `message.blocks` had not changed. - const { hasImages, markdown, prose, subagentGroups, tools } = useMemo(() => { - const split = splitNativeChatBlocks(message.blocks) - const groups = subagentGroupBlocks(split.prose) - // A spawn-group row carries a plain-text twin so a client without the block - // type still reads the roster. This one draws the block, so the twin is - // dropped rather than printed beside it — only the twin, never the prose - // beside it: the block is provider-agnostic, so a lane that folds a roster - // into a message with real text must not lose that text here. - const prose = - groups.length === 0 - ? split.prose - : split.prose.filter( - (block) => - !isSubagentGroupBlock(block) && - !(block.type === 'text' && isSubagentGroupFallbackText(block.text)) - ) - return { - tools: split.tools, - prose, - subagentGroups: groups, - markdown: nativeChatProseToMarkdown(prose), - hasImages: prose.some((block) => block.type === 'image-ref') - } - }, [message.blocks]) + // One pass per block set, shared with the list that decides whether this row + // occupies a slot — so "draws nothing" means the same thing to both. + const { hasImages, markdown, prose, subagentGroups, tools } = deriveNativeChatRowContent( + message.blocks + ) const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' const isSystem = message.role === 'system' @@ -220,6 +193,7 @@ export const MessageRow = memo(function MessageRow({ expandOverride={activityExpandOverride} activeTurnIsWorking={activeTurnIsWorking} structuredActivityUi={structuredActivityUi} + disclosureId={message.id} /> ) : null} {showControls ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx b/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx index 5f79c492fc9..89f28683429 100644 --- a/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx +++ b/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx @@ -1,4 +1,5 @@ import { useCallback, useEffect, useMemo, useRef, useState } from 'react' +import { useNativeChatComposerRevealFocus } from './use-native-chat-composer-reveal-focus' import { useAppStore } from '../../store' import { useNativeChatLaunchDraftSignal } from './use-native-chat-launch-draft-adoption' import { useNativeChatRetainedSession } from './use-native-chat-retained-session' @@ -63,6 +64,7 @@ export function NativeChatResolvedView({ sessionId, transcriptPath, isVisible, + isFocusedGroup, targetPtyId, terminalTabId, ownsTabWideLaunchDraft, @@ -131,6 +133,13 @@ export function NativeChatResolvedView({ composerRef, questionAnswerInputRef }) + useNativeChatComposerRevealFocus({ + rootRef, + composerRef, + isVisible, + isFocusedGroup, + composerReady: !questionActive && targetPtyId !== null && canSend + }) const contextMenu = useNativeChatContextMenu({ rootRef, onSwitchToTerminal, diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test-harness.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test-harness.tsx new file mode 100644 index 00000000000..e34cdc0d824 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test-harness.tsx @@ -0,0 +1,208 @@ +import { forwardRef, useImperativeHandle, useRef } from 'react' +import { vi, type Mock } from 'vitest' +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' +import type { NativeChatQuestionCardProps } from './NativeChatQuestionCard' +import type { NativeChatLaunchSeed } from './native-chat-composer-types' + +// Why: a named spy type keeps the harness's inferred return type portable across the test files. +type StructuredSessionSpy = Mock + +/** + * Shared mock state and `vi.mock` factories for the NativeChatStructuredSession test files. + * Load it through `await vi.hoisted(async () => (await import(...)).createStructuredSessionMocks())` + * so the factories can close over `mocks` before the mocked modules resolve. + */ +export function createStructuredSessionMocks() { + const mocks = { + call: vi.fn() as StructuredSessionSpy, + fileLinkClick: vi.fn() as StructuredSessionSpy, + mode: 'static' as 'static' | 'outbox', + status: 'ready' as 'idle' | 'loading' | 'ready' | 'error', + messages: null as null | unknown[], + messageListProps: null as null | { + allowFileUriLinks?: boolean + onLinkClick?: (...args: unknown[]) => void + showTurnStatus?: boolean + runtimeContext?: unknown + }, + composerProps: null as null | { + launchSeed?: NativeChatLaunchSeed + structuredTransport?: Record + isWorking?: boolean + }, + questionCardProps: null as NativeChatQuestionCardProps | null, + promptItems: [] as AgentJournalRenderItem[], + respond: vi.fn() as StructuredSessionSpy, + handlePasteEvent: vi.fn() as StructuredSessionSpy, + pasteFromClipboard: vi.fn() as StructuredSessionSpy, + submissions: [] as unknown[], + monitoringBackgroundTasks: false, + showBackgroundTasks: false, + isWorking: false, + turnId: null as string | null, + supportsBackgroundTaskStop: false, + supportsBackgroundTaskStopAll: true, + backgroundTasks: [] as AgentSessionBackgroundTask[], + settledBackgroundTasks: [] as AgentSessionBackgroundTask[], + stopBackgroundTask: vi.fn() as StructuredSessionSpy + } + + const moduleFactories = { + structuredAgentSessionClient: () => ({ + callStructuredAgentSession: mocks.call + }), + useStructuredAgentSession: async () => { + const { useStructuredAgentSessionOutbox } = + await import('./use-structured-agent-session-outbox') + return { + useStructuredAgentSession: (props: { + sessionId: string + target: { kind: 'local' } | { kind: 'environment'; environmentId: string } + }) => { + const outbox = useStructuredAgentSessionOutbox({ + sessionId: props.sessionId, + target: props.target, + fence: 1, + submissions: mocks.submissions as never + }) + return { + messages: + mocks.messages ?? + (mocks.mode === 'outbox' + ? [] + : [ + { + id: 'message-1', + role: 'assistant', + source: 'transcript', + timestamp: 1, + blocks: [ + { + type: 'text', + text: '[file](file:///repo/src/main.ts)' + } + ] + } + ]), + status: mocks.status, + error: outbox.error, + hasOlder: false, + loadingOlder: false, + loadOlder: vi.fn() as StructuredSessionSpy, + prompts: mocks.promptItems, + outbox: outbox.outbox, + blockedClientMessageId: outbox.blockedClientMessageId, + send: outbox.send, + retry: outbox.retry, + isWorking: mocks.isWorking, + backgroundTasks: { + show: mocks.showBackgroundTasks || mocks.monitoringBackgroundTasks, + isMonitoring: mocks.monitoringBackgroundTasks, + tasks: mocks.backgroundTasks, + settledTasks: mocks.settledBackgroundTasks, + supportsStop: mocks.supportsBackgroundTaskStop, + supportsStopAll: mocks.supportsBackgroundTaskStopAll + }, + turnId: mocks.turnId, + cancel: vi.fn() as StructuredSessionSpy, + stopBackgroundTask: (taskId?: string) => + mocks.stopBackgroundTask(props.sessionId, taskId), + respond: mocks.respond, + optionSnapshot: [ + { + id: 'model', + label: 'Model', + category: 'model', + kind: { + type: 'select', + currentValue: 'gpt-live', + choices: [{ value: 'gpt-live', label: 'GPT Live' }] + }, + valueSource: 'reported', + settable: true + } + ], + optionSurface: { + getSnapshot: () => [], + setOption: vi.fn() as StructuredSessionSpy, + invokeAction: vi.fn() as StructuredSessionSpy, + subscribe: () => () => {} + }, + setStructuredOption: vi.fn() as StructuredSessionSpy + } + } + } + }, + useNativeChatFontScale: () => ({ + useNativeChatFontScale: () => ({ scale: 1 }) + }), + useNativeChatFileLinkContext: () => ({ + useNativeChatFileLinkContext: () => ({ + worktreeId: 'wt-1', + worktreePath: '/repo', + runtimeEnvironmentId: null + }) + }), + useNativeChatFileLinkClick: () => ({ + useNativeChatFileLinkClick: (context: unknown) => (context ? mocks.fileLinkClick : undefined) + }), + nativeChatMessageList: () => ({ + NativeChatMessageList: (props: typeof mocks.messageListProps) => { + mocks.messageListProps = props + return
    + } + }), + nativeChatComposer: () => ({ + NativeChatComposer: forwardRef((props: typeof mocks.composerProps, ref) => { + mocks.composerProps = props + const fieldRef = useRef(null) + useImperativeHandle(ref, () => ({ + // Real DOM focus: the reveal-focus loop retries until focus lands in the pane. + focus: () => { + fieldRef.current?.focus() + return true + }, + insertTypedText: () => true, + handlePasteEvent: mocks.handlePasteEvent, + pasteFromClipboard: mocks.pasteFromClipboard + })) + return