diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index c4eea80933e..d5c8134934b 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -181,11 +181,12 @@ jobs: env: ORCA_RELAY_ADMIN_ID_TOKEN: ${{ steps.deploy-auth.outputs.id_token }} run: | - RETRY_ARGS=() - if test "${WAVE_INDEX}" != 0; then RETRY_ARGS=(--retry-freshness); fi + # Freshness-only failures are publish lag, not health, on every wave + # including the first; the CLI still caps the retry at the wave's + # evidence-age budget, so this cannot mutate on aged evidence. pnpm incident:relay-preflight -- \ --state-file "${OUTPUT_DIRECTORY}/relay-${MONITOR_RUN_ID}-dry-run.state.json" \ - --wave-index "${WAVE_INDEX}" "${RETRY_ARGS[@]}" + --wave-index "${WAVE_INDEX}" --retry-freshness - name: Require durable rehome disabled and exact selector env: diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts index 18ee4495078..412c2905408 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts @@ -6,7 +6,10 @@ import { livePreflightGcloud, runIncidentLivePreflight } from './incident-live-preflight-cli.js' -import type { IncidentSample } from './incident-monitor.js' +import { + INCIDENT_MONITOR_THRESHOLDS, + type IncidentSample +} from './incident-monitor.js' import type { AdmissionSelector } from './incident-selector.js' const directories: string[] = [] @@ -313,7 +316,7 @@ describe('relay incident live preflight', () => { it('retries freshness-only failures when explicitly requested', async () => { const stale = sample() stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const missing = sample() delete missing.sources['relay-logs'] const collect = vi.fn() @@ -331,11 +334,44 @@ describe('relay incident live preflight', () => { expect(wait).toHaveBeenNthCalledWith(2, 15_000) }) + it('retries a first-wave stale sample and passes on the fresh one', async () => { + const stale = sample() + stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() + const collect = vi.fn().mockResolvedValueOnce(stale).mockResolvedValueOnce(sample()) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', '0', '--retry-freshness'], + { now: () => now, collect, wait } + )).resolves.toBeUndefined() + expect(collect).toHaveBeenCalledTimes(2) + expect(wait).toHaveBeenCalledOnce() + }) + + it('stops retrying when the next wait would exceed the evidence-age bound', async () => { + const completedAt = now - 290_000 + const stale = sample() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() + const collect = vi.fn(async () => stale) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('strict', { + startedAt: new Date(completedAt - 17 * 60_000).toISOString(), + windowStartedAt: new Date(completedAt - 16 * 60_000).toISOString(), + lastSampleAt: new Date(completedAt - 30_000).toISOString(), + completedAt: new Date(completedAt).toISOString() + }), '--retry-freshness'], + { now: () => now, collect, wait } + )).rejects.toThrow('cloud-monitoring/source_stale') + expect(collect).toHaveBeenCalledOnce() + expect(wait).not.toHaveBeenCalled() + }) + it('does not retry a threshold failure', async () => { const unhealthy = sample() unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = - new Date(now - 180_001).toISOString() + new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => unhealthy) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( @@ -348,7 +384,7 @@ describe('relay incident live preflight', () => { it('fails closed after the bounded freshness retry window', async () => { const stale = sample() - stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - (INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 1)).toISOString() const collect = vi.fn(async () => stale) const wait = vi.fn(async () => undefined) await expect(runIncidentLivePreflight( diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts index e82627a3e80..2fcce3ed85d 100644 --- a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts @@ -7,6 +7,7 @@ import { suppliedIdentityToken } from './incident-monitor-cli.js' import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js' import { evaluateIncidentSample, + FRESHNESS_FAILURE_CODES, preDrainDryRunPassed, type IncidentSample } from './incident-monitor.js' @@ -18,12 +19,6 @@ const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000 // Matches the same-cap cell job timeout-minutes; bounds each predecessor wave. const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000 const WAVE_INDEX_PATTERN = /^[0-3]$/ -const FRESHNESS_FAILURE_CODES = new Set([ - 'signal_missing', - 'signal_stale', - 'source_missing', - 'source_stale' -]) export function livePreflightGcloud( gcloud: ReturnType, @@ -173,7 +168,11 @@ export async function runIncidentLivePreflight( const freshnessOnly = evaluation.failures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code) ) - if (!freshnessOnly || attempt === attempts) { + // Waiting must never carry the mutation past the same evidence-age bound + // the entry check enforces, so the wave budget also caps the retry window. + const budgetExhausted = + now() + FRESHNESS_RETRY_INTERVAL_MS - completedAt > maxEvidenceAgeMs + if (!freshnessOnly || attempt === attempts || budgetExhausted) { throw new Error( `relay live preflight failed: ${evaluation.failures .map((failure) => `${failure.source}/${failure.code}`) diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.ts index e090be7ea58..adfe6cad480 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-cli.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.ts @@ -50,6 +50,8 @@ const StateSchema = z.object({ continuityEvents: z.array(z.object({ recordedAt: z.string(), windowSequence: z.number().int().nonnegative(), + // Pre-2026-09-05 state files predate tolerated freshness gaps. + tolerated: z.boolean().default(false), failures: z.array(z.object({ code: z.string(), source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']), diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts index 09b7b16fa45..74054b6c0ba 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts @@ -93,7 +93,7 @@ describe('incident monitor sources', () => { }) it('zero-fills an expired sparse lock-wait point', async () => { - let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ points: [{ @@ -141,7 +141,7 @@ describe('incident monitor sources', () => { it('freshens a sparse zero without masking a recent nonzero lock wait', async () => { let value = 0 - const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs const readAt = now + 11_879 const fetchImpl: typeof fetch = async () => Response.json({ timeSeries: [{ diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.ts index 0b78c2c4f6b..a97bfe3df43 100644 --- a/cloud/apps/relay-ops/src/incident-monitor-sources.ts +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.ts @@ -95,7 +95,7 @@ export const GOOGLE_METRICS: GoogleMetricDefinition[] = [ 'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"', aggregation: 'latest-max', emptyIsZero: true, - zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudLockWaitCarryMs }, { signal: 'cloud_sql.deadlocks', diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 4e1da9fab26..076cff3de3b 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import { evaluateIncidentSample, INCIDENT_CHECKPOINT_MINUTES, + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES, INCIDENT_MONITOR_THRESHOLDS, INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS, initialIncidentMonitorState, @@ -182,12 +183,49 @@ describe('incident monitor evaluator', () => { code: 'source_missing', source: 'relay-logs' }) - const stale = healthySample(startedAt - 180_001) + const stale = healthySample( + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) const failures = evaluateIncidentSample(stale, startedAt).failures expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true) expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true) }) + // Why: production run 33944873727 at 2026-09-05T04:46:09Z read + // cloud_sql.lock_waits 189 286 ms old and restarted a 15-minute window on + // Google's publish lag. Cloud SQL documents 60 s sampling plus up to 165 s of + // invisibility, so that age is Google's clock, not our fleet. + it('reads a 189-second cloud signal as fresh and holds the other sources at 180 s', () => { + const lagged = healthySample() + lagged.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, startedAt - 189_286) + expect(evaluateIncidentSample(lagged, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + const laggedDirector = healthySample() + laggedDirector.sources['director-admin']!.observedAt = + new Date(startedAt - 189_286).toISOString() + expect(evaluateIncidentSample(laggedDirector, startedAt).failures).toContainEqual( + expect.objectContaining({ code: 'source_stale', source: 'director-admin' }) + ) + }) + + it('still fails a cloud signal past the documented publish lag', () => { + const dark = healthySample() + dark.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal( + 0, + startedAt - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + expect(evaluateIncidentSample(dark, startedAt).failures).toContainEqual( + expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + }) + ) + }) + it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => { const sample = healthySample() sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81) @@ -592,7 +630,7 @@ describe('incident monitor lifecycle', () => { 'restarts a %i-minute continuous window after stale telemetry', async (durationMinutes) => { let now = startedAt - let staleInjected = false + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const checkpoints: Array<[number, number]> = [] const state = initialIncidentMonitorState({ incidentId: 'incident-1', @@ -612,9 +650,11 @@ describe('incident monitor lifecycle', () => { now += ms }, collect: async () => { - if (!staleInjected && now === startedAt + 5 * 60_000) { - staleInjected = true - return healthySample(now - 180_001) + if (staleSamples > 0 && now >= startedAt + 5 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) } return healthySample(now) }, @@ -623,16 +663,20 @@ describe('incident monitor lifecycle', () => { checkpoints.push([summary.windowSequence, summary.checkpointMinute]) } }) + const restartMinute = 5 + INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 expect(result.windowSequence).toBe(1) expect(result.windowStartedAt).toBe( - new Date(startedAt + 6 * 60_000).toISOString() + new Date(startedAt + restartMinute * 60_000).toISOString() ) expect(result.completedAt).toBe( - new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString() + new Date(startedAt + (durationMinutes + restartMinute) * 60_000).toISOString() ) expect(result.sampleCount).toBe(durationMinutes + 1) - expect(result.continuityEvents).toHaveLength(1) - expect(result.continuityEvents[0]!.failures).toEqual( + expect(result.continuityEvents.map((event) => event.tolerated)).toEqual([ + ...Array(INCIDENT_FRESHNESS_TOLERANCE_SAMPLES).fill(true), + false + ]) + expect(result.continuityEvents.at(-1)!.failures).toEqual( expect.arrayContaining([ expect.objectContaining({ code: 'source_stale' }) ]) @@ -642,6 +686,188 @@ describe('incident monitor lifecycle', () => { } ) + // Why: run 33944873727 on 2026-09-05 restarted at 04:46:09Z on a single + // 189-second cloud reading and then blew the 25-minute lineage cap, so a + // green fleet produced no verdict at all. One unread sample now continues the + // window; the sample is still checked against every threshold it can read. + it('carries a 15-minute window through a single stale cloud sample', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 10 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.windowStartedAt).toBe(new Date(startedAt).toISOString()) + expect(result.completedAt).toBe(new Date(startedAt + 15 * 60_000).toISOString()) + expect(result.sampleCount).toBe(16) + expect(result.frozenAt).toBeNull() + expect(result.continuityEvents).toEqual([{ + recordedAt: new Date(startedAt + 10 * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [expect.objectContaining({ + code: 'signal_stale', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits' + })] + }]) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('gives a signal a fresh budget only after it reads fresh again', async () => { + let now = startedAt + const staleMinutes = new Set([3, 5, 6, 9, 10]) + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (staleMinutes.has((now - startedAt) / 60_000)) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(0) + expect(result.continuityEvents).toHaveLength(staleMinutes.size) + expect(result.continuityEvents.every((event) => event.tolerated)).toBe(true) + expect(preDrainDryRunPassed(result)).toBe(true) + }) + + it('does not hand a resumed monitor a fresh tolerance budget', async () => { + let now = startedAt + 3 * 60_000 + const resumed = { + ...initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }), + windowStartedAt: new Date(startedAt).toISOString(), + lastSampleAt: new Date(startedAt + 2 * 60_000).toISOString(), + sampleCount: 3, + totalSampleCount: 3, + continuityEvents: Array.from( + { length: INCIDENT_FRESHNESS_TOLERANCE_SAMPLES }, + (_, index) => ({ + recordedAt: new Date(startedAt + (index + 1) * 60_000).toISOString(), + windowSequence: 0, + tolerated: true, + failures: [{ + code: 'signal_stale', + source: 'cloud-monitoring' as const, + signal: 'cloud_sql.lock_waits' + }] + }) + ) + } + const stop = new Error('stop after the resumed sample') + await expect(runIncidentMonitor(resumed, { + now: () => now, + wait: async () => { + throw stop + }, + collect: async () => { + const sample = healthySample(now) + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + return sample + }, + persist: async (state) => { + expect(state.windowSequence).toBe(1) + expect(state.windowStartedAt).toBeNull() + expect(state.continuityEvents.at(-1)!.tolerated).toBe(false) + }, + checkpoint: async () => {} + })).rejects.toThrow(stop) + }) + + it('freezes on a threshold breach that arrives with a tolerated stale signal', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const sample = healthySample(now) + if (now === startedAt + 2 * 60_000) { + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = + signal(0, now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1) + sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81, now) + } + return sample + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.frozenAt).toBe(new Date(startedAt + 2 * 60_000).toISOString()) + expect(result.failures).toContainEqual(expect.objectContaining({ + code: 'threshold_max', + signal: 'cloud_sql.cpu' + })) + expect(preDrainDryRunPassed(result)).toBe(false) + }) + it('resets at the next fresh sample after a runner gap', async () => { let now = startedAt + 10 * 60_000 const state = { @@ -690,13 +916,21 @@ describe('incident monitor lifecycle', () => { durationMinutes: 15, intervalMs: 60_000 }) + let staleSamples = INCIDENT_FRESHNESS_TOLERANCE_SAMPLES + 1 const result = await runIncidentMonitor(state, { now: () => now, wait: async (ms) => { now += ms }, - collect: async () => - healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now), + collect: async () => { + if (staleSamples > 0 && now >= startedAt + 10 * 60_000) { + staleSamples-- + return healthySample( + now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs - 1 + ) + } + return healthySample(now) + }, persist: async () => {}, checkpoint: async () => {} }) @@ -706,7 +940,7 @@ describe('incident monitor lifecycle', () => { ) expect(result.frozenAt).not.toBeNull() expect(result.windowSequence).toBe(1) - expect(result.sampleCount).toBe(15) + expect(result.sampleCount).toBe(13) expect(result.failures).toContainEqual({ code: 'continuity_deadline_exceeded', source: 'active-probe', diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index a121568d918..2785eb573af 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -6,7 +6,27 @@ import { export const INCIDENT_MONITOR_THRESHOLDS = { activeProbeMaxAgeMs: 60_000, - cloudDataMaxAgeMs: 180_000, + // Why: Cloud Monitoring publishes on Google's clock, not ours. Per the metric + // list read 2026-09-05, Cloud Run instance_count / cpu / memory / + // max_request_concurrencies / request_count are "Sampled every 60 seconds. + // After sampling, data is not visible for up to 120 seconds" (60+120=180 s), + // and Cloud SQL cpu / memory / num_backends / backends_in_wait / + // deadlock_count say "up to 165 seconds" (60+165=225 s). Window-sum signals + // age differently: observedAt is the newest point in the 5-minute query + // window, so a label series that stops emitting reads as 300 s old while its + // summed value is still complete. 330 s clears the worst of the three (the + // 300 s query window) plus ~30 s of collect-to-evaluate latency. The old + // 180 s bar restarted healthy 15-minute windows at 181 s, 189 s and 255 s on + // 2026-09-04/05, once burning the whole 25-minute lineage with no verdict. + cloudDataMaxAgeMs: 330_000, + // Why: the director admin API answers live on our own request, so hold its + // freshness bar where it sat while it shared cloudDataMaxAgeMs. + directorAdminMaxAgeMs: 180_000, + // Why: how long a nonzero backends-in-wait point is carried before it reads as + // zero. Held at the pre-2026-09-05 cloud bar: carrying it for the full + // cloudDataMaxAgeMs would hand the evaluator a point older than its own + // freshness bar as soon as collection latency is added. + cloudLockWaitCarryMs: 180_000, relayLogMaxAgeMs: 180_000, heartbeatMaxAgeMs: 45_000, endpointLatencyMs: 2_000, @@ -175,6 +195,7 @@ export type IncidentMonitorState = { continuityEvents: { recordedAt: string windowSequence: number + tolerated: boolean failures: IncidentFailure[] }[] frozenAt: string | null @@ -307,7 +328,7 @@ const SOURCE_MAX_AGE: Record = { 'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs, 'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs, 'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs, - 'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + 'director-admin': INCIDENT_MONITOR_THRESHOLDS.directorAdminMaxAgeMs } function ageMs(timestamp: string, nowMs: number): number { @@ -608,14 +629,59 @@ function checkpointMinutes(durationMinutes: number): number[] { return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes) } -const CONTINUITY_FAILURE_CODES = new Set([ - 'collector_failed', - 'monitor_gap', +// Freshness-only failures: we could not read a signal this sample. Distinct from +// collector_failed / monitor_gap, where the whole sample is absent. +export const FRESHNESS_FAILURE_CODES = new Set([ + 'signal_missing', 'signal_stale', 'source_missing', 'source_stale' ]) +const CONTINUITY_FAILURE_CODES = new Set([ + 'collector_failed', + 'monitor_gap', + ...FRESHNESS_FAILURE_CODES +]) + +// Why: Cloud Monitoring overshoots its own publish bar, and one unread sample is +// not evidence of an unhealthy fleet. Under the 25-minute lineage cap a restart +// past minute 10 costs the entire verdict, so a healthy fleet produced none on +// 2026-09-05. A signal may miss this many consecutive samples before the window +// restarts; the sample is still evaluated against every threshold it can read, +// and a threshold breach still freezes the run outright. +export const INCIDENT_FRESHNESS_TOLERANCE_SAMPLES = 2 + +function freshnessKey(failure: IncidentFailure): string { + return `${failure.source}/${failure.signal ?? '*'}` +} + +// Rebuild the per-signal tolerated streak from the trailing continuity events so a +// resumed monitor cannot hand a signal a fresh budget. +function resumeFreshnessStreaks( + state: IncidentMonitorState +): Map { + const events = state.continuityEvents + const streaks = new Map() + const last = events[events.length - 1] + if (!last?.tolerated) return streaks + for (const key of new Set(last.failures.map(freshnessKey))) { + let streak = 0 + let laterAt: number | null = null + for (let index = events.length - 1; index >= 0; index--) { + const event = events[index]! + const recordedAt = Date.parse(event.recordedAt) + if (!event.tolerated) break + if (laterAt !== null && laterAt - recordedAt > state.intervalMs * 1.5) break + if (!event.failures.some((failure) => freshnessKey(failure) === key)) break + streak++ + laterAt = recordedAt + } + streaks.set(key, streak) + } + return streaks +} + function resetContinuousWindow( state: IncidentMonitorState, recordedAt: string, @@ -631,6 +697,7 @@ function resetContinuousWindow( state.continuityEvents.push({ recordedAt, windowSequence: state.windowSequence, + tolerated: false, failures }) } @@ -681,6 +748,7 @@ export async function runIncidentMonitor( await dependencies.persist(state) return state } + const freshnessStreaks = resumeFreshnessStreaks(state) while (state.completedAt === null) { if (dependencies.now() > lineageDeadlineMs) { completeContinuityDeadline(state, dependencies.now(), lineageStartMs) @@ -715,9 +783,34 @@ export async function runIncidentMonitor( const thresholdFailures = evaluation.failures.filter((failure) => !CONTINUITY_FAILURE_CODES.has(failure.code) ) - if (continuityFailures.length > 0) { + const toleratedKeys = new Set( + state.windowStartedAt !== null && + continuityFailures.length > 0 && + continuityFailures.every((failure) => FRESHNESS_FAILURE_CODES.has(failure.code)) + ? continuityFailures.map(freshnessKey) + : [] + ) + for (const key of [...freshnessStreaks.keys()]) { + if (!toleratedKeys.has(key)) freshnessStreaks.delete(key) + } + let tolerated = toleratedKeys.size > 0 + for (const key of toleratedKeys) { + const streak = (freshnessStreaks.get(key) ?? 0) + 1 + freshnessStreaks.set(key, streak) + if (streak > INCIDENT_FRESHNESS_TOLERANCE_SAMPLES) tolerated = false + } + if (continuityFailures.length > 0 && !tolerated) { + freshnessStreaks.clear() resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures) } else { + if (tolerated) { + state.continuityEvents.push({ + recordedAt: evaluation.evaluatedAt, + windowSequence: state.windowSequence, + tolerated: true, + failures: continuityFailures + }) + } if (state.windowStartedAt === null) { state.windowStartedAt = evaluation.evaluatedAt } diff --git a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs index e31403f6dd6..a9e92d6cd57 100644 --- a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs +++ b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs @@ -92,7 +92,11 @@ test('same-cap wrapper is reusable, canary-bound, and sequential', () => { // age checks must scale by wave or cell_2+ can never pass; the bound's // per-wave step is the cell job timeout, so the two must move together. assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/) - assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/) + // Wave 0 must retry freshness-only failures too: one Cloud Monitoring publish + // lag at the sample instant is not health evidence, and single-shot wave 0 + // failed a whole batch on a series that was fresh again a minute later. + assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" --retry-freshness/) + assert.doesNotMatch(job, /RETRY_ARGS/) assert.match(job, /timeout-minutes: 75/) // Both age gates step by the cell job timeout above; the constant is // duplicated across the two languages, so pin each copy to it. diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md index 870c95dd413..8a8dfda1495 100644 --- a/cloud/docs/relay-incident-monitor.md +++ b/cloud/docs/relay-incident-monitor.md @@ -73,6 +73,13 @@ for a committed forward-recovery gate. Durable files default to gap resets the active window at the next fresh sample and preserves the prior window evidence. A threshold freeze never clears automatically. +A signal that reads missing or stale may miss up to two consecutive samples +without restarting the window. The sample still counts and is still checked +against every threshold it can read, and each tolerated gap is recorded in +`continuityEvents` with `tolerated: true`. A third consecutive miss of the same +signal, a failed collector, a runner gap, or any threshold breach restarts or +freezes as before. + A production candidate or multi-target mutation must download the exact dry-run artifact by workflow run ID and attempt. It verifies the artifact hashes and provenance, requires a green completed 15-minute state no older @@ -89,7 +96,8 @@ durably marked consumed before mutation and cannot authorize another run. | Signal | Freeze condition | | --- | ---: | | Active probe age | over 60 seconds | -| Cloud/log data age | over 180 seconds | +| Cloud Monitoring data age | over 330 seconds | +| Relay log and director admin data age | over 180 seconds | | Cell heartbeat age | over 45 seconds | | Endpoint latency | over 2,000 ms | | Cloud SQL CPU | over 80% | @@ -155,6 +163,25 @@ heartbeats, and matching live admission. separate it from today's baseline; the exhausted-retry bar (incident peak 467 vs bar 300), director concurrency, and the pool bars carry that role. Re-tighten after the fleet is on the 500 ms lock wait. +- Raised the Cloud Monitoring freshness bar from 180 s to 330 s and let a + freshness-only failure miss up to two consecutive samples without restarting + the window (2026-09-05). Basis: Google's metric list documents Cloud Run + `request_count`, `container/instance_count`, `container/cpu/utilizations`, + `container/memory/utilizations` and `container/max_request_concurrencies` as + "Sampled every 60 seconds. After sampling, data is not visible for up to 120 + seconds", and Cloud SQL `database/cpu/utilization`, + `database/memory/utilization`, `database/postgresql/num_backends`, + `database/postgresql/backends_in_wait` and `database/postgresql/deadlock_count` + as "up to 165 seconds", so the newest visible point is up to 180 s and 225 s + old respectively. Window-sum signals age further: `observedAt` is the newest + point in the 5-minute query window, so a label series that stops emitting + reads as 300 s old while its summed value is complete. The old bar sat under + all three. Production on 2026-09-04/05 restarted healthy 15-minute windows at + 181 s and 255 s (`auth.errors`, run 33928912676) and at 189 s + (`cloud_sql.lock_waits`, run 33944873727), and the last of those then blew the + 25-minute lineage cap at 1 500 004 ms, so a green fleet produced no verdict. + The director admin bar stays at 180 s and the nonzero lock-wait carry window + stays at 180 s; both publish on our own cadence. - Recalibrated the exhausted-PostgreSQL-retry freeze from 0 to 300 per five minutes (2026-09-04). Basis: #18521 cut the request-path cell-inventory lock wait from the 1 s pool `lock_timeout` to 500 ms, so contended waiters diff --git a/config/docker/cli-launch-contract/Dockerfile b/config/docker/cli-launch-contract/Dockerfile index f6a618a8ece..c90cbcd979c 100644 --- a/config/docker/cli-launch-contract/Dockerfile +++ b/config/docker/cli-launch-contract/Dockerfile @@ -6,8 +6,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive # Install Electron's link-time libraries without adding a display server or FUSE. -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ coreutils \ diff --git a/config/docker/headless-pairing/Dockerfile b/config/docker/headless-pairing/Dockerfile index 03664f68b0d..e4b4cafeefc 100644 --- a/config/docker/headless-pairing/Dockerfile +++ b/config/docker/headless-pairing/Dockerfile @@ -5,8 +5,13 @@ ARG LIBASOUND_PACKAGE=libasound2t64 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/docker/headless-serve-shutdown/Dockerfile b/config/docker/headless-serve-shutdown/Dockerfile index 13b1ed2b69f..8ee7b942499 100644 --- a/config/docker/headless-serve-shutdown/Dockerfile +++ b/config/docker/headless-serve-shutdown/Dockerfile @@ -2,8 +2,13 @@ FROM ubuntu@sha256:678c6550cc43645e08669028bc177f50be4e7c5b8cca677067b1914d4afc7 ENV DEBIAN_FRONTEND=noninteractive -RUN apt-get update \ - && apt-get install -y --no-install-recommends \ +# Why: archive.ubuntu.com mid-sync returns Hash Sum mismatch / wrong-size indexes and stalls per-package fetches; retry with bounded timeouts and drop half-synced lists between attempts. +RUN for attempt in 1 2 3 4 5; do \ + apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && break; \ + if [ "$attempt" = 5 ]; then exit 100; fi; \ + rm -rf /var/lib/apt/lists/*; sleep 20; \ + done \ + && apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y --no-install-recommends \ bash \ ca-certificates \ dbus-x11 \ diff --git a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs index dd26784ee46..1e908754dc6 100644 --- a/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs +++ b/config/relay-assets/node-pty-1.1.0-windows-pty-teardown-patch.cjs @@ -12,14 +12,15 @@ const { join, resolve } = require('node:path') * `_inSocket`, and it wraps a real Windows named-pipe handle from `fs.openSync(term.conin, 'w')`. * Every terminal leaks one File handle for the life of the host process. * - * The obvious fix -- and the one the desktop patch ships -- releases it at the TOP of the branch, - * before `_getConsoleProcessList()` forks and before the native kill. That is measurably worse than - * leaving the leak alone: teardown aborts partway, the forked console-list agent is never reaped, - * and both pipe handles stay alive instead of one. This asset releases it at the END of the branch - * instead, after the fork and the kill have already happened. + * The obvious fix -- and the placement `config/patches/node-pty@1.1.0.patch` uses -- releases it at + * the TOP of the branch, before `_getConsoleProcessList()` forks and before the native kill. That is + * measurably worse than leaving the leak alone: teardown aborts partway, the forked console-list + * agent is never reaped, and both pipe handles stay alive instead of one. This asset releases it at + * the END of the branch instead, after the fork and the kill have already happened. * * Measured on a Windows SSH host, 20 spawn/kill cycles, handles bucketed by NT object type - * (identical numbers standalone and through a real relay): + * (identical numbers standalone and through a real relay). Every row is the NON-DLL branch, which + * is the branch a relay runs -- see the divergence note below for why that matters: * * published node-pty File +1/terminal, Process flat * desktop patch placement File +2/terminal, Process +1/terminal <-- 3x WORSE @@ -34,18 +35,64 @@ const { join, resolve } = require('node:path') * Why this ships as a relay asset rather than only in config/patches/node-pty@1.1.0.patch: pnpm * patches do not cross the SSH boundary -- a relay host runs the tree `npm install` put there. * - * DELIBERATE DIVERGENCE FROM THE DESKTOP: the desktop patch has the early placement and therefore - * the +2 File / +1 Process regression, measured against its exact installed tree. Correcting it - * there is a separate change with its own verification, so the two trees differ on this one hunk on - * purpose, and the test pins that so a future "sync the patches" does not copy the bug back. + * DELIBERATE DIVERGENCE FROM THE DESKTOP, AND WHY IT IS NOT A DESKTOP-TERMINAL BUG: the two hosts + * do not run the same branch of `kill()`. node-pty defaults `_useConptyDll` to false + * (`windowsPtyAgent.js`). Every desktop site that opens a terminal pane sets it true -- + * `local-pty-utils.ts` (two) and `native-pty-spawn.ts` -- as does the `windows-conpty-warmup.ts` + * warm-up, so all of those take the `else` branch, where UPSTREAM ALREADY destroys the input + * socket. The relay passes no such option (`src/relay/pty-handler.ts`), so it takes the + * `!useConptyDll` branch -- the one this asset and the desktop patch both edit. * - * NOT ADDRESSED, AND A SEPARATE DEFECT THAT IS STILL OPEN: a terminal that exits on its own is - * still torn down through `kill()` -- both hosts call `destroy()` on natural exit and - * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, and the - * ordering this patch relies on does not hold. Measured over 20 self-exit cycles with that - * `destroy()` issued: published +3 File/+1 Process per terminal, desktop-patched +2/+1, this tree - * +2/+1. So this patch does not close it and the desktop patch does not either. It is reachable - * for every Windows user, local and relay, on every terminal closed by typing `exit`. + * THE DESKTOP IS NOT ENTIRELY OFF THAT BRANCH. Two desktop sites omit the option and so run it + * too: the hidden rate-limit probes in `src/main/rate-limits/claude-pty.ts` and + * `codex-pty-rate-limit-probe.ts`. Both recur -- their fetchers poll -- and both tear down through + * `kill()`, so this hunk is live on the desktop, just never for a pane a user can see. Do not + * restate this as "the desktop never executes that branch": that sentence stood here for two + * revisions and is false. + * + * What the numbers above therefore do NOT cover: they were measured on relay-style spawn/kill + * cycles. Whether the early placement costs the same +2 File / +1 Process across a probe's + * lifecycle is UNMEASURED -- plausible, not established, and worth measuring before anyone quotes + * a desktop figure. What IS settled is the claim this comment replaced: that the desktop patch made + * every Windows user worse off ON EVERY TERMINAL. Terminals take the DLL branch, and the harness + * that produced that claim defaulted into the branch it was not trying to measure. + * + * The divergence is therefore about which branch each host runs for the workload that matters, not + * about a regression in the terminals users open. The test still pins it, because a future "sync + * the patches" would put the early placement onto the relay's branch, where it does cost +2 File + * and +1 Process per terminal. + * + * If you extend this enumeration, grep for `node-pty` rather than for a static import: those two + * probes were missed three times because they use `await import('node-pty')`. + * + * THE SELF-EXIT LEAK: FIXED FOR THE DESKTOP BY #18635, STILL LIVE ON A RELAY. A terminal that exits + * on its own is also torn down through `kill()` -- both hosts call `destroy()` on natural exit and + * `WindowsTerminal.destroy()` is `kill()` -- but the shell is already gone by then, so the ordering + * this asset relies on does not hold. Measured over 20 self-exit cycles on the NON-DLL branch: + * published +3 File/+1 Process per terminal, desktop patch placement +2/+1, this tree +2/+1. This + * asset does not close it. + * + * #18635 does, in `config/patches/node-pty@1.1.0.patch`: the baton outlives the shell so `PtyKill` + * still reaches `ClosePseudoConsole`, plus an unconditional conout dispose on the DLL branch. That + * fix does not reach a Windows relay, and no hunk in THIS file can carry it, because it is mostly + * NATIVE (`src/win/conpty.cc`) and this asset only rewrites `lib/*.js`. Three delivery paths exist + * and none currently covers Windows: + * + * - the pnpm patch does not cross the SSH boundary -- the remote `npm install` yields upstream's + * unpatched node-pty; + * - the orcad prebuild matrix has no win32 entry (`MATRIX_SLOTS`, + * `config/scripts/build-orcad-prebuilds.mjs`), so no Windows binary is ever compiled from + * patched source to ship; + * - a relay asset CAN patch native source and rebuild on the host -- that is exactly what + * `node-pty-1.1.0-master-cloexec-patch.cjs` does -- but it returns + * `skipped:unsupported-platform` for anything but linux/darwin. Extending it to win32 means + * requiring an MSVC toolchain on the relay host, a far heavier precondition than on Linux, + * where node-gyp already runs at install time. + * + * So a Windows SSH relay still leaks a pseudoconsole per self-exiting terminal, and closing it is a + * DELIVERY problem, not another hunk here. Do not read #18635's flat self-exit relay numbers as + * covering deployed relays: they were measured against a locally rebuilt binary, so they describe + * the relay CODE PATH on a patched tree, not the tree a relay host actually installs. */ const EXPECTED_NODE_PTY_VERSION = '1.1.0' diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index b7905fa0419..9f5885edc8a 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -2855,7 +2855,7 @@ "https://github.com/stablyai/orca/pull/13876" ], "invariant": "Opening one HTML preview from a paired client renders the workspace document in exactly one client-local browser tab, located by that document and served over the orca-preview scheme. The client gains exactly that one browser workspace and it is the document one — blank where a URL page carries a URL, named by the document, with the chip naming the file — while the host gains no browser page at all, neither in its own page registry nor in the tab snapshot its clients publish into. The preview occupies its own split without taking focus from the source editor; an explicit click activates it, and closing it removes only the preview. Following a document from a file link is the other half of that switch and does move the reader to it, tab group included, whether the preview is new or already open, because opening a file is a request to look at it. A document tab quit with the client comes back as the same row on a grant the relaunched client mints afresh. A preview is named by the browser page it is open in, not by a namespace of its own, and the page registry has two halves: a workspace-document guest is registered in its own map and is absent from the browsing one entirely. That absence is the fence. Page, session and profile management, agent tab enumeration and command targeting, download routing and certificate attribution all read the browsing map directly, in more places than a per-channel guard could be remembered in, so none of them can name a document page and none of them carries a guard. Browser tools the reader drives (element grab, hover describe, selection capture, the annotation viewport bridge) are the one operation that legitimately spans the halves, and they go through the single authority that reads both, keyed by the page and its hosting renderer. The halves are disjoint in both directions: browsing registration refuses a page the document half already holds, and minting a grant refuses a page the browsing half already holds, so one id can never name a surface in both. The headless backend acts on that refusal by destroying the window it had already opened rather than leaving a policy-less page behind an id nothing can drive, keeping nothing under that id for its own shutdown to hand back. Registration refuses on the same terms when the guest it was asked about is already gone. The exit door is guarded in both its halves: a preview withdraws by revoking its grant and never through the unregister channel, so a page the document half holds arriving there is refused before either the registration teardown or the grab-state disposal beside it, which would otherwise drop the intent an in-flight preview grab compares by identity and leave that grab answering ok without ever arming its guest. A bridge request whose guest does not resolve is refused without tearing down the page it named, so a misaddressed request cannot cancel a healthy page's in-flight downloads and grabs. The annotation viewport bridge resolves its guest when its serialized op actually runs rather than when the request arrived, so a cross-process navigation while it waited cannot leave the bridge installed in a retired guest while the reader looks at a new one. State main keys by a preview's page is disposed when that page's grant is revoked, which is the only signal a preview's surface is gone. A tool asking for a page whose guest has not attached yet waits for that registration and arms when it arrives, rather than answering not-ready at the reader; that wait resolves only the request already naming this page, never the worktree-wide or any-tab waits the CLI and agents use to ask for a browser tab to drive. Handing the previewed document to the reader's own machine routes on the owners its grant was minted against — the file's own connection owner and the worktree's own runtime owner, neither read from the tab's stored fields. Only a document proven to live on this machine reaches the client OS; one with a resolved remote owner is downloaded first; and one whose owner cannot be resolved at all, workspace root included, is refused with a message naming that, because the download route would otherwise read the same absolute path on the client and hand back a same-named local file under the remote document's name. A runtime-owned path that falls outside its worktree root is refused by that route itself and surfaces as a failure toast rather than a download. Nothing the document does writes a file to this machine either: the preview partition denies downloads outright instead of routing them through the browser download flow, which has no page to attribute a preview's bytes to and would otherwise reserve a name in this desktop's Downloads folder and write them there unprompted. That refusal is visible to the reader and invisible to the document: the preview's shell carries a fixed sentence saying downloads are off, published at most once per preview per interval so a document asking in a loop cannot fill Orca's chrome, while the page itself gets back exactly what it got before, which is nothing. The sentence names no file, because the document chooses the name it offers; and a refusal never takes the document away the way an entry document's own failure does, whatever it names. A preview is a browser tab, not an editor tab in a preview mode: it is named the way a browser tab is named — by the document it shows when that document declares a title, and by the file it shows when it does not — while the chip goes on naming the file and the host whatever the document calls itself. A title is refused on the same terms the url is: a document that declares none has Chromium report the grant URL as its title, and that title is stored, mirrored onto the tab and written to disk, so anything carrying the scheme falls back to the file instead. It is created by the preview action as a page located by its document, it carries the workspace-relative path copy the editor's path header owned, and closing it revokes the grant that made the document readable while a URL tab closing beside it revokes nothing. Chrome persisted by builds that made previews editor tabs is dropped on restore rather than coming back naming a surface no restore can produce, and the ordinary editor tab for the same document is left alone. A document tab is held back at the mobile publish boundary — no client holds its grant, and the wire has no tab kind for it — while an ordinary browser tab beside it still publishes. It is held back from the group projection that publishes tab order, recency and group activity as well as from the tab list itself, so no published group names a tab the phone is never sent. A browser page can be located by a workspace document instead of a URL, and the document is the whole of its stored identity. The grant and the orca-preview URL that document is served over are minted when the page mounts and replaced by a hard reload, so neither is ever written to the page's url, mirrored onto its tab, persisted or published: such a page's url is the blank URL from creation through restore, including when a session written elsewhere carries a grant URL in, and what the session carries is the worktree and path a restored page mints afresh against today's owners. Every door onto a page's url holds that line — creation, the title update, and the navigation commit alike — so a report about a document page cannot give it a URL it never had, and the title fallback and the loading affordance follow the url each door actually wrote. The mirror carries the document too, so a tab entry cannot go on naming a document its active page has left. Every guest in the app is policy-attached through one door: a workspace document takes a restricted profile there rather than a separate installer beside it, so the attachment bookkeeping that door owns — what registration refuses, and what teardown frees — covers a preview on the same terms as a browsing page, and a preview takes none of the browsing machinery that door installs. That authority answers from the moment the embedder hands the guest over rather than only after a later navigation: the guest binds to the grant it is already showing, so the tools reach the document the reader opened and not just one they navigated to. A read the host reports as truncated or over-cap is refused rather than served partially, and a document outside the paired worktree is refused with a message naming that boundary instead of a bare read failure. The rendered document reaches nothing off-machine on its own: every served response carries a self-only content security policy, the preview session cancels any request that is not in-document, subframes cannot navigate outside the grant, a guest no document has yet bound to a grant may not navigate at all, the guest gathers no ICE candidates, and an SSH path that canonicalizes outside the grant root is refused before it is read. The one route out is a link the reader presses: a trusted click on an anchor, reported by the preview's own preload from a guest still bound to a live grant, leaves as an Orca browser tab rather than a native window or a dead click — and only after the reader confirms the exact destination URL, so a document cannot spend a single stray press exfiltrating what it can read into a link it authored. The preview hands its guest that focus itself whenever it is the surface the reader is in — a browsing page gets it from the chrome around it, and a preview has no chrome to get it from — and it does so only then, so a preview mounted behind a terminal or an editor never takes the keyboard from what the reader is actually in. It offers again when the window itself takes focus back and nothing in the embedder has claimed that focus, because another app coming to the front lands focus on the embedder rather than the guest and the route out would otherwise stay shut until something remounted the pane — while the same window focus also arrives when the reader presses a tab, that being the guest's own blur returning, and taking focus back from there would fight the reader for their own click. Nothing else does. A navigation or popup the document starts by itself is swallowed whatever else is happening, including immediately after a genuine press elsewhere in the document, so a page that can read its grant cannot hand it to a browser tab; a middle click opens nothing; and a fragment link is answered inside the document. A preview attach carries the preview preload and no renderer-supplied one, and no other attach path can acquire it. A subresource the workspace will not send degrades the document to a notice naming that file, never to a failure panel over a page that rendered. A grant outlives neither the tab that owns it nor the renderer document that minted it, and only the trusted renderer can mint or revoke one. For the browser creations this gate still owns, owner-pinned creation returns the canonical host page identity before navigation readiness; delayed navigation cannot turn a created page into an unidentifiable failure or a duplicate retry. Capability rejection before host mutation must preserve the original error, issue no RPC, surface a failure toast, and remove only a caller-declared newly-created empty split. Post-create reconciliation failure requires exact rollback; ambiguous rollback rejects without local fallback.", - "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and anti-detection while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", + "oracle": "In paired Electron, write an HTML fixture that declares its own title on the host, invoke the Explorer preview action on the client, and require the document text to be readable out of the orca-preview guest before judging any absence. With that presence established, require the client to hold exactly one browser workspace more than its baseline and that workspace to be the document one: page and tab url blank, the document path mirrored onto both, the tab named by the document's title, the chip naming the file, no editor row of the retired preview species anywhere, and the guest URL carrying the orca-preview scheme. Ask the host through its own page registry as well as through the tab snapshot, and in the same run open an ordinary URL browser tab from the same client and require that one to arrive in both — the presence precondition without which “the host gained nothing” is satisfied just as well by an oracle that cannot see browser pages at all. Require the preview to sit in a group other than the source editor's while the active group and tab remain the source editor's. Then click away to the terminal, click the preview tab, require it to reactivate and still render, close it with its own X, and require the document tab to be gone while the host still holds only the URL tab and the source group, source editor and terminal survive. Quit the client with a document tab open and relaunch it on the same profile: require the same workspace row to come back, blank and named by the document, rendering the document again over a grant URL that differs from the one that was quit, with no preview-scheme or document-named page anywhere in what the host holds. Prove the halves are live by flipping one product property at a time and requiring the run to fail: publish document workspaces to the host like ordinary ones, and stop mirroring the document onto the workspace row. Have the fixture document attempt its own egress on every load — an unattended window.open and location.href to an off-machine URL, plus an inline ICE gathering probe — and require the same baseline counts and a candidate count of zero, so the document's own attempts are measured rather than assumed. Then, as a separate phase after the close oracle has already run, bring the client window to the front, press the document's heading with a real mouse event, and require the document to report that the same press drove it to attempt a second window.open and location.href while both browser counts stay at that phase's baseline and nothing routes — the case a recent-input gate cannot distinguish from the press's own effect. Only then press the target=_blank link with a real mouse event and require both a recorded routing call that returned success and a browser count above that baseline, with the preview tab still open. Drive the preload's click policy as a unit oracle over a real document: a dispatched click, a trusted press on an external anchor, an anchor reached through what it wraps, an SVG animated href, a sibling preview link, fragment and percent-encoded fragment targets, a bare hash, and a middle click. Create a browser page located by a workspace document, handing creation a live grant URL, and require its stored url, its mirrored tab url and the written session payload all to be blank with no orca-preview string anywhere in what was written, while an ordinary page created the same way keeps the URL it was given and asks for the address bar the document page never does. Parse the written page and tab through the session schema and require the document to survive both halves. Hydrate them back and require the document page to return blank and still named — including when its page row was salvaged away and only the tab's own copy remains, and when a foreign session carried a grant URL into both rows. Drive the mirror across a page switch out of the document and back, and across a repair in which the document is the only mirrored field that differs. Name a document page from its document and require the tab to take that name, name it with an empty title and require the file, name it with a live grant URL and require the file again with no preview scheme anywhere in the written session, and require an ordinary blank browser tab beside it to still be called New Tab. Dispatch a title update out of a rendered preview's own guest and require it to reach the page state while the identity chip still reads the document's workspace-relative path. Attach a browsing guest and a workspace-document guest through the same method in one run and require the browsing one to take clicked-link routing, popup handling and auth-identity detach tracking while the document guest takes none of them, stays inside the grant it is showing, denies every window it asks for, and is dropped from the page-keyed document registry by the same teardown that frees its id for a later attach. Register a browsing guest and attach a workspace-document guest in one run against the real manager, require the one door to answer each page with the guest of its own half, and require the document page to be absent from the browsing map and from its enumeration. Drive both browsing registration entry points with a page the document half already holds and require them to register nothing, and drive the mint channel with a page the browsing half already holds and require it to refuse; and drive the offscreen one with a guest that is missing and with one already destroyed, requiring the same refusal. Arm a grab on a live preview target, drive the unregister channel at that same target in the window before the queued operation runs, and require the grab to reach the guest anyway — then drive the same sequence for an ordinary browser page and require its grab state to be disposed after all. Hold one viewport-bridge op open, queue a second behind it, swap the page's guest while that second op waits, and require the injection to land in the guest the page has then. Revoke a grant after a tool has run against its page and require that page's grab state to be cancelled and disposed. Ask a tool for a document page whose guest has not attached, require the request to park in the registration wait, attach the guest, and require the same request to arm on it; require a page nothing ever renders to answer not-ready once that wait elapses. With a document open, ask a tool for a browsing page id and require it to be answered by the browsing half or not at all, with the same channel reaching the document guest under the page it really renders. Drive open-externally for a document whose per-file owner is remote while the workspace-scoped owner is unresolved, for a runtime-owned worktree whose preview tab carries no runtime id of its own, and for a worktree that resolves no runtime owner while the tab still carries one, requiring the download route in each; and for an owner that cannot be resolved at all, and for an unknown workspace root while nothing names another host, requiring a refusal that neither opens nor downloads. Drive the headless backend with a page the document half already holds and require it to reject, destroy the window, and unregister nothing — then shut the backend down and require it still to have unregistered nothing. Navigate a bound preview guest at a second grant through both latch events and require it to stay on the grant it bound to. Mount a preview while a renderer drag is already in flight and require its guest to be click-through at the moment it is appended, not a turn later. Render the editor panel shell in each remaining tab mode and require the path header exactly where the surface does not already name itself. Drive the preview action and require a browser tab located by the document rather than an editor tab, require a second open of the same document to activate the tab it is already in, and require closing that tab to revoke its grant while a URL tab closed beside it revokes none. Hydrate a session carrying preview chrome from a build that made previews editor tabs and require it dropped while the ordinary editor tab for the same document survives. Publish a worktree holding a document tab and a URL tab and require only the URL tab to reach the mobile snapshot. Install the shared partition policies for a preview partition and for an ordinary browsing partition in the same run, fire each one's own will-download listener, and require the preview's to cancel while the browsing one still reaches the download router — then require the preview protocol installer to be what asks for that deny. In the same run, require the cancelled download to raise a reader-facing notice and the routed one to raise none. Drive that notice directly for a guest bound to a live grant, for repeated attempts inside and outside its interval, for two previews at once, and for a contents no preview is bound to; require the guest registry to name the bound grant for a live preview guest and nothing for a contents that is not one, has committed no document, or is gone. Drive the shell with a refusal and require one fixed sentence, still one row after three more refusals, standing beside an asset failure rather than being counted with it, gone behind the failure panel, and ignored when it names another preview's grant. Drive the main-side report gate directly for a sender that is no preview guest, a guest with no bound or a revoked grant, a genuine press Electron's webview focus flag misreports as unfocused, and non-web URLs; drive the reader-facing confirmation for accept, cancel, and a confirmed tab the browser refuses; and drive will-attach-webview in both preload directions. Run the per-owner reader, grant-containment, scheme-admission, guest-policy, and plan-routing contracts as unit oracles, including a host-reported truncation, an over-cap binary, and an out-of-worktree paired path. Drive the reader-facing component with the payloads the reader can actually produce — the entry document fails only as truncated or unreadable, a subresource additionally as a refused format — and require the asset case to leave the guest mounted. Drive the closed-tab cleanup hook, the window installer, and the grant IPC handlers directly, requiring the grant to be released when the preview tab closes, cleared at window creation and on a cross-document main-frame navigation, and refused to any sender that is not the trusted renderer. For the browser creations this gate still owns, run the unchanged contract oracle for direct create and side-preview callers with absent status, unknown capabilities, and a mixed-version host, requiring the original unsupported error or visible toast, zero RPCs, and no retained new split; hold a real navigation response beyond the 15-second client deadline after host creation and require the first RPC to return the exact host inventory page ID, one host page, and no retry; repeat reconciliation faults against headless serve and retain the separate exact-page reconciliation rollback oracle.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/main/runtime/orca-runtime-browser.test.ts src/main/runtime/rpc/methods/browser.test.ts src/renderer/src/lib/file-preview.test.ts src/renderer/src/runtime/web-session-browser-placement.test.ts src/renderer/src/runtime/web-runtime-session.test.ts src/renderer/src/runtime/web-session-tabs-sync.test.ts src/renderer/src/runtime/remote-server-parity.test.ts", "pnpm exec vitest run --config config/vitest.config.ts src/renderer/src/runtime/web-runtime-browser-materialization.test.ts", diff --git a/config/scripts/agent-inspection-cadence-batching-benchmark.mjs b/config/scripts/agent-inspection-cadence-batching-benchmark.mjs new file mode 100644 index 00000000000..128627676c4 --- /dev/null +++ b/config/scripts/agent-inspection-cadence-batching-benchmark.mjs @@ -0,0 +1,139 @@ +#!/usr/bin/env node +// Counts how many whole-host process-table captures the agent-completion cadence costs. +// +// Local panes all resolve out of one TTL-deduped snapshot, and the inspection queue collapses +// every shared-observation task enqueued in the same tick onto a single capture. So the capture +// count is the number of DISTINCT wake instants across panes, not the number of pane wakes. +// +// This drives the production interval picker (`nextCadenceInspectionDelayMs`) against a baseline +// that reproduces the pre-change ±10% jitter, over a simulated wall-clock window. +import { spawnSync } from 'node:child_process' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) { + const candidate = new URL(`${specifier}.ts`, context.parentURL) + if (fs.existsSync(fileURLToPath(candidate))) { + return { url: candidate.href, shortCircuit: true } + } + } + return nextResolve(specifier, context) + } +}) + +const ROOT = path.resolve(import.meta.dirname, '../..') +const WINDOW_MS = Number(process.env.ORCA_INSPECTION_BENCH_WINDOW_MS ?? '60000') +const PANE_COUNTS = (process.env.ORCA_INSPECTION_BENCH_PANES ?? '1,2,4,8') + .split(',') + .map((value) => Number(value.trim())) + +if (!Number.isSafeInteger(WINDOW_MS) || WINDOW_MS <= 0) { + throw new Error(`ORCA_INSPECTION_BENCH_WINDOW_MS must be a positive integer, got ${WINDOW_MS}`) +} +for (const paneCount of PANE_COUNTS) { + if (!Number.isSafeInteger(paneCount) || paneCount <= 0) { + throw new Error(`ORCA_INSPECTION_BENCH_PANES entries must be positive, got ${paneCount}`) + } +} + +const { nextCadenceInspectionDelayMs } = await import( + path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-interval.ts') +) +const { POLL_TIER_INTERVAL_MS } = await import( + path.join(ROOT, 'src/renderer/src/components/terminal-pane/agent-completion-poll-cadence.ts') +) +const { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } = await import( + path.join(ROOT, 'src/shared/process-table-snapshot-reader.ts') +) + +// Pre-change: independent ±10% jitter per pane, re-rolled on every reschedule. +function baselineDelayMs(baseMs) { + return Math.round(baseMs * (1 + (Math.random() * 0.2 - 0.1))) +} + +function simulate(paneCount, baseMs, pickDelay) { + const startedAt = 1_700_000_000_000 + const wakes = [] + for (let pane = 0; pane < paneCount; pane += 1) { + // Panes mount at arbitrary moments, which is what spreads them apart in the first place. + let clock = startedAt + Math.floor(Math.random() * baseMs) + while ((clock += pickDelay(baseMs, clock)) < startedAt + WINDOW_MS) { + wakes.push(clock) + } + } + // A wake is served from the snapshot the previous capture produced until that snapshot's TTL + // lapses, so the TTL window starts at the capture, not on an epoch grid. + let captures = 0 + let snapshotExpiresAt = -Infinity + for (const wakeAt of wakes.sort((left, right) => left - right)) { + if (wakeAt >= snapshotExpiresAt) { + captures += 1 + snapshotExpiresAt = wakeAt + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + } + } + return captures +} + +function medianOf(rounds, run) { + const samples = Array.from({ length: rounds }, run).sort((left, right) => left - right) + return samples[Math.floor(samples.length / 2)] +} + +const baseMs = POLL_TIER_INTERVAL_MS.idle +console.log( + `Agent-completion cadence — whole-host \`ps\` captures over ${WINDOW_MS / 1000}s at the idle tier (${baseMs}ms)\n` +) +console.log('| visible panes | before | after | reduction |') +console.log('| --- | --- | --- | --- |') +for (const paneCount of PANE_COUNTS) { + const before = medianOf(21, () => simulate(paneCount, baseMs, baselineDelayMs)) + const after = medianOf(21, () => + simulate(paneCount, baseMs, (base, now) => + nextCadenceInspectionDelayMs({ + baseMs: base, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + ) + ) + // A window shorter than one cadence tier can leave the baseline at zero; reporting a + // percentage off that divides by zero and prints a meaningless reduction. + const reduction = before > 0 ? `${(((before - after) / before) * 100).toFixed(0)}%` : 'n/a' + console.log(`| ${paneCount} | ${before} | ${after} | ${reduction} |`) +} + +// Detection latency must not regress: the grid deadline is always within one interval. +let worstDelay = 0 +for (let sample = 0; sample < 100_000; sample += 1) { + const now = 1_700_000_000_000 + sample * 7 + worstDelay = Math.max( + worstDelay, + nextCadenceInspectionDelayMs({ + baseMs, + hasConsecutiveErrors: false, + alignToSharedGrid: true, + now + }) + ) +} +if (worstDelay > baseMs) { + throw new Error(`grid alignment delayed a poll to ${worstDelay}ms, above the ${baseMs}ms tier`) +} +console.log( + `\nWorst observed wait: ${worstDelay}ms (tier interval ${baseMs}ms) — no inspection is ever delayed.` +) diff --git a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs index 64fb1b056b8..ba3e64bfa2f 100644 --- a/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs +++ b/config/scripts/node-pty-windows-pty-teardown-patch.test.mjs @@ -1,11 +1,18 @@ -// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with the -// desktop's own node-pty patch. pnpm patches do not cross the SSH boundary, so a relay runs the tree -// `npm install` put there; the desktop had this fix and the relay did not, and every terminal on a -// Windows SSH host leaked one File handle for the life of the relay process. +// The relay's copy of the ConPTY teardown release, and the guard that keeps it in lockstep with +// `config/patches/node-pty@1.1.0.patch`. pnpm patches do not cross the SSH boundary, so a relay runs +// the tree `npm install` put there, and every terminal on a Windows SSH host leaked one File handle +// for the life of the relay process. // -// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- what the -// desktop patch does -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a new -// Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// The ORDER of the conin release is the fix. Releasing it at the top of the branch -- the placement +// the desktop patch uses -- was measured at 3x WORSE than shipping nothing (File +2/terminal and a +// new Process +1/terminal); releasing it after the console-list fork and the native kill is flat. +// +// Those numbers are the `!useConptyDll` branch, which is the branch a RELAY runs. Every desktop +// site that opens a terminal pane sets `useConptyDll: true` and takes the other branch, where +// upstream already destroys the input socket. Two hidden rate-limit probes +// (`src/main/rate-limits/claude-pty.ts`, `codex-pty-rate-limit-probe.ts`) do omit the option and so +// do run this hunk, but no user-visible pane does. The divergence pinned below is about which +// branch each host runs for terminals -- not about a regression in the panes users open. import { createRequire } from 'node:module' import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { join, resolve } from 'node:path' @@ -90,9 +97,11 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => { ) }) - // The one hunk that must NOT match the desktop, and the reason is measured, not stylistic: - // releasing conin before `_getConsoleProcessList()` forks aborts teardown partway. - it('releases conin after the console-list fork, not before it like the desktop patch', () => { + // The one hunk that must NOT match the desktop patch, and the reason is measured, not stylistic: + // on the branch a relay runs, releasing conin before `_getConsoleProcessList()` forks aborts + // teardown partway. Desktop terminal panes take the other branch, so no pane is affected either + // way; what this guards is a patch sync putting the early placement onto the relay's branch. + it('releases conin after the console-list fork, unlike the desktop patch placement', () => { const fixture = writeNodePtyFixture('1.1.0') patchNodePtyWindowsTeardown(fixture.root) const patched = readFileSync(join(fixture.libDir, 'windowsPtyAgent.js'), 'utf8') @@ -108,7 +117,8 @@ describe('Windows SSH relay node-pty ConPTY teardown patch', () => { expect(branch.indexOf('this._inSocket.destroy();')).toBeGreaterThan( branch.indexOf('this._getConsoleProcessList()') ) - // Pinned so a future "sync the relay asset to config/patches" cannot copy the regression back. + // Pinned so a future "sync the relay asset to config/patches" cannot copy the early placement + // onto the relay's branch, where it costs +2 File and +1 Process per terminal. expect(patched).not.toBe(readFileSync(desktopPath('windowsPtyAgent.js'), 'utf8')) }) diff --git a/config/scripts/renderer-quadratic-scan-benchmark.mjs b/config/scripts/renderer-quadratic-scan-benchmark.mjs new file mode 100644 index 00000000000..681b6cf8550 --- /dev/null +++ b/config/scripts/renderer-quadratic-scan-benchmark.mjs @@ -0,0 +1,364 @@ +#!/usr/bin/env node +// Benchmarks four renderer projections that scaled worse than linearly with user data, each on a +// path that reruns per keystroke or per store write. +// +// Scenarios 1, 3 and 4 time the production export against a hand-written reproduction of the +// pre-change shape and assert both agree first. Scenario 2 is MODELLED on both sides: the +// projection lives inside the `useTabGroupItemProjections` React hook and cannot be imported +// without a renderer, so it reproduces the before/after loops rather than driving production. +import { spawnSync } from 'node:child_process' +import { transformSync } from 'esbuild' +import { performance } from 'node:perf_hooks' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath, pathToFileURL } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +const ROOT = path.resolve(import.meta.dirname, '../..') +const RENDERER = path.join(ROOT, 'src/renderer/src') + +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (!context.parentURL) { + return nextResolve(specifier, context) + } + const candidates = specifier.startsWith('@/') + ? ['.ts', '.tsx', '/index.ts', '/index.tsx', ''].map( + (suffix) => path.join(RENDERER, specifier.slice(2)) + suffix + ) + : specifier.startsWith('.') && !/\.[cm]?[jt]sx?$/.test(specifier) + ? ['.ts', '.tsx'].map((suffix) => + fileURLToPath(new URL(specifier + suffix, context.parentURL)) + ) + : [] + const resolved = candidates.find((file) => fs.existsSync(file) && fs.statSync(file).isFile()) + return resolved + ? { url: pathToFileURL(resolved).href, shortCircuit: true } + : nextResolve(specifier, context) + }, + // Node strips types from .ts but not .tsx; the sidebar row model transitively imports icons. + load(url, context, nextLoad) { + if (url.endsWith('.tsx')) { + const source = fs.readFileSync(fileURLToPath(url), 'utf8') + const { code } = transformSync(source, { loader: 'tsx', format: 'esm', jsx: 'automatic' }) + return { format: 'module', source: code, shortCircuit: true } + } + if (url.endsWith('.json') && !url.includes('/node_modules/')) { + const source = fs.readFileSync(fileURLToPath(url), 'utf8') + return { format: 'module', source: `export default ${source}`, shortCircuit: true } + } + return nextLoad(url, context) + } +}) + +const importRenderer = (relativePath) => + import(pathToFileURL(path.join(RENDERER, relativePath)).href) + +function envInt(name, fallback) { + const value = Number(process.env[name] ?? fallback) + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`${name} must be a positive integer, got ${value}`) + } + return value +} + +const KEYSTROKES = envInt('ORCA_QUADRATIC_BENCH_KEYSTROKES', 12) +const WORKTREES = envInt('ORCA_QUADRATIC_BENCH_WORKTREES', 300) +const TABS = envInt('ORCA_QUADRATIC_BENCH_TABS', 60) +const OPEN_FILES = envInt('ORCA_QUADRATIC_BENCH_OPEN_FILES', 120) +const CHANGED_FILES = envInt('ORCA_QUADRATIC_BENCH_CHANGED_FILES', 5000) +const SIDEBAR_ROWS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS', 600) +const SIDEBAR_REPOS = envInt('ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS', 80) +if (SIDEBAR_REPOS > SIDEBAR_ROWS) { + throw new Error( + 'ORCA_QUADRATIC_BENCH_SIDEBAR_REPOS must not exceed ORCA_QUADRATIC_BENCH_SIDEBAR_ROWS' + ) +} + +function timeRounds(run, rounds = 7) { + run() + const samples = Array.from({ length: rounds }, () => { + const start = performance.now() + run() + return performance.now() - start + }).sort((left, right) => left - right) + return samples[Math.floor(rounds / 2)] +} + +function repeat(times, run) { + return () => { + let last + for (let round = 0; round < times; round += 1) { + last = run() + } + return last + } +} + +const results = [] +function compare({ label, scale, drives, before, after }) { + if (JSON.stringify(before()) !== JSON.stringify(after())) { + throw new Error(`${label}: baseline disagreed with the indexed shape`) + } + results.push({ label, scale, drives, beforeMs: timeRounds(before), afterMs: timeRounds(after) }) +} + +// ------------------------------------------------- 1. workspace board search index + +const { buildWorkspaceBoardPaletteDocuments, matchWorkspaceBoardWorktrees } = await importRenderer( + 'components/sidebar/workspace-kanban-search.ts' +) + +const repoMap = new Map([ + ['repo-1', { id: 'repo-1', name: 'orca', path: '/tmp/orca', branch: 'main' }] +]) +const boardWorktrees = Array.from({ length: WORKTREES }, (_, index) => ({ + id: `repo-1::/tmp/worktree-${index}`, + repoId: 'repo-1', + path: `/tmp/worktree-${index}`, + branch: `feature/search-target-${index}`, + title: `Workspace ${index} search target`, + isMain: false +})) +const queries = Array.from({ length: KEYSTROKES }, (_, index) => 'search'.slice(0, (index % 6) + 1)) +const matchAll = (documents) => + queries.map((query) => [ + ...matchWorkspaceBoardWorktrees({ worktrees: boardWorktrees, query, repoMap, documents }) + ]) + +compare({ + label: 'workspace board filter (per keystroke burst)', + scale: `${WORKTREES} worktrees x ${KEYSTROKES} keystrokes`, + drives: 'production', + // Omitting `documents` is the pre-change shape: the index is rebuilt inside every match. + before: () => matchAll(undefined), + // The hook memoizes the index on [worktrees, repoMap]; only the match reruns per keystroke. + after: () => matchAll(buildWorkspaceBoardPaletteDocuments({ worktrees: boardWorktrees, repoMap })) +}) + +// ------------------------------------------------- 2. tab-group projections (modelled) + +const groupTabs = Array.from({ length: TABS }, (_, index) => ({ + id: `tab-${index}`, + entityId: `entity-${index}`, + contentType: index % 3 === 0 ? 'editor' : 'terminal' +})) +const openFiles = Array.from({ length: OPEN_FILES }, (_, index) => ({ + id: `entity-${index}`, + path: `/tmp/file-${index}.ts` +})) +const tabOrder = groupTabs.map((tab) => tab.id) +// Production memoizes each index on its own source list, so a unified-tab write reuses it. +const openFileById = new Map(openFiles.map((item) => [item.id, item])) +const groupTabById = new Map(groupTabs.map((item) => [item.id, item])) + +function tabProjections(findOpenFile, findGroupTab) { + const editorItems = groupTabs + .filter((item) => item.contentType === 'editor') + .map((item) => findOpenFile(item.entityId)) + .filter((file) => file !== undefined) + const order = tabOrder.map((itemId) => findGroupTab(itemId)?.entityId ?? itemId) + return [editorItems, order] +} + +compare({ + label: 'tab-group projections (per unified-tab write)', + scale: `${TABS} tabs x ${OPEN_FILES} open files`, + drives: 'modelled', + before: repeat(200, () => + tabProjections( + (id) => openFiles.find((candidate) => candidate.id === id), + (id) => groupTabs.find((candidate) => candidate.id === id) + ) + ), + after: repeat(200, () => + tabProjections( + (id) => openFileById.get(id), + (id) => groupTabById.get(id) + ) + ) +}) + +// ------------------------------------------------- 3. source-control tree build + +const { buildSourceControlTree } = await importRenderer( + 'components/right-sidebar/source-control-tree.ts' +) +const { normalizeRelativePath } = await importRenderer('lib/path.ts') +const { splitPathSegments } = await importRenderer('components/right-sidebar/path-tree.ts') +const { compareFileNames } = await import( + pathToFileURL(path.join(ROOT, 'src/shared/file-name-sort.ts')).href +) + +const changedEntries = Array.from({ length: CHANGED_FILES }, (_, index) => ({ + path: `src/area-${index % 20}/module-${index % 60}/nested/deep/part-${index % 7}/file-${index}.ts` +})) + +// Pre-change `buildSourceControlTree`: identical except each ancestor path is re-joined. +function buildSourceControlTreeBefore(area, entries) { + const makeDirectory = (dirPath, name, depth) => ({ + type: 'directory', + key: `dir::${area}::${dirPath}`, + name, + path: dirPath, + area, + depth, + fileCount: 0, + children: [], + directoryChildren: new Map() + }) + const root = makeDirectory('', '', -1) + for (const entry of entries) { + const normalizedPath = normalizeRelativePath(entry.path) + const segments = splitPathSegments(normalizedPath) + if (segments.length === 0) { + continue + } + let parent = root + for (let index = 0; index < segments.length - 1; index += 1) { + const name = segments[index] + const dirPath = segments.slice(0, index + 1).join('/') + let dir = parent.directoryChildren.get(name) + if (!dir) { + dir = makeDirectory(dirPath, name, index) + parent.directoryChildren.set(name, dir) + parent.children.push(dir) + } + parent = dir + } + parent.children.push({ + type: 'file', + key: `${area}::${entry.path}`, + name: segments.at(-1), + path: normalizedPath, + entry, + area, + depth: segments.length - 1 + }) + } + const finalize = (node) => { + const directories = node.children.filter((child) => child.type === 'directory').map(finalize) + const files = node.children.filter((child) => child.type === 'file') + directories.sort((a, b) => compareFileNames(a.name, b.name)) + files.sort((a, b) => compareFileNames(a.entry.path, b.entry.path)) + const { directoryChildren: _, ...rest } = node + return { + ...rest, + fileCount: files.length + directories.reduce((count, dir) => count + dir.fileCount, 0), + children: [...directories, ...files] + } + } + return finalize(root).children +} + +compare({ + label: 'source-control tree build (per filter keystroke)', + scale: `${CHANGED_FILES} changed files`, + drives: 'production', + before: () => buildSourceControlTreeBefore('unstaged', changedEntries), + after: () => buildSourceControlTree('unstaged', changedEntries) +}) + +// ------------------------------------------------- 4. sidebar header boundaries + +const { getRepoHeaderSectionEndByRepoId } = await importRenderer( + 'components/sidebar/worktree-header-section-boundaries.ts' +) +const { estimateRenderRowSize } = await importRenderer( + 'components/sidebar/worktree-list/viewport/virtual-rows.ts' +) + +const headerRowIndexes = new Set( + Array.from({ length: SIDEBAR_REPOS }, (_, repo) => + Math.floor((repo * SIDEBAR_ROWS) / SIDEBAR_REPOS) + ) +) +const sidebarRows = Array.from({ length: SIDEBAR_ROWS }, (_, index) => + headerRowIndexes.has(index) + ? { + type: 'header', + key: `repo:${index}`, + label: '', + count: 0, + tone: '', + repo: { id: `repo-${index}` } + } + : { type: 'item', rowKey: `wt:${index}`, sectionKey: '', depth: 0, groupDepth: 0 } +) +const headerRepoIds = sidebarRows.filter((row) => row.type === 'header').map((row) => row.repo.id) +const boundaryArgs = { + rows: sidebarRows, + firstHeaderIndex: 0, + // What `getSidebarOrderedRepoHeaderIdsByBucket` yields for repos outside any project group. + sidebarRepoHeaderIdsByBucket: new Map([['ungrouped', headerRepoIds]]), + repoHeaderBucketByRepoId: new Map(headerRepoIds.map((id) => [id, 'ungrouped'])) +} + +// Pre-change `getRepoHeaderSectionEndByRepoId`: a findIndex and an indexOf per header row. +function getRepoHeaderSectionEndByRepoIdBefore(args) { + const rowStarts = [] + let offset = 0 + for (let index = 0; index < args.rows.length; index += 1) { + rowStarts[index] = offset + offset += estimateRenderRowSize(args.rows, index, args.firstHeaderIndex, null) + } + rowStarts[args.rows.length] = offset + const sectionEndByRepoId = new Map() + for (let index = 0; index < args.rows.length; index += 1) { + const row = args.rows[index] + const repoId = row?.type === 'header' ? row.repo?.id : undefined + if (!repoId) { + continue + } + const bucketKey = args.repoHeaderBucketByRepoId.get(repoId) + const bucketRepoIds = bucketKey ? args.sidebarRepoHeaderIdsByBucket.get(bucketKey) : undefined + const bucketIndex = bucketRepoIds?.indexOf(repoId) ?? -1 + const nextRepoId = bucketIndex >= 0 ? bucketRepoIds?.[bucketIndex + 1] : undefined + let endIndex = -1 + if (nextRepoId) { + endIndex = args.rows.findIndex((r) => r.type === 'header' && r.repo?.id === nextRepoId) + } else { + endIndex = args.rows.length + for (let next = index + 1; next < args.rows.length; next += 1) { + if (args.rows[next]?.type === 'header' || args.rows[next]?.type === 'host-header') { + endIndex = next + break + } + } + } + sectionEndByRepoId.set( + repoId, + rowStarts[endIndex >= 0 ? endIndex : args.rows.length] ?? rowStarts[args.rows.length] ?? 0 + ) + } + return sectionEndByRepoId +} + +compare({ + label: 'sidebar header boundaries (per row-model rebuild)', + scale: `${SIDEBAR_REPOS} repos x ${SIDEBAR_ROWS} rows`, + drives: 'production', + before: repeat(50, () => [...getRepoHeaderSectionEndByRepoIdBefore(boundaryArgs)]), + after: repeat(50, () => [...getRepoHeaderSectionEndByRepoId(boundaryArgs)]) +}) + +// ------------------------------------------------- + +console.log('Renderer quadratic-scan removals\n') +console.log('| projection | drives | scale | before | after | |') +console.log('| --- | --- | --- | --- | --- | --- |') +for (const row of results) { + console.log( + `| ${row.label} | ${row.drives} | ${row.scale} | ${row.beforeMs.toFixed(2)} ms | ${row.afterMs.toFixed(2)} ms | ${(row.beforeMs / row.afterMs).toFixed(1)}x |` + ) +} diff --git a/config/scripts/run-headless-linux-pairing-docker.mjs b/config/scripts/run-headless-linux-pairing-docker.mjs index 635c66348cc..799b8d73ab1 100644 --- a/config/scripts/run-headless-linux-pairing-docker.mjs +++ b/config/scripts/run-headless-linux-pairing-docker.mjs @@ -70,7 +70,7 @@ function valueAfter(flag) { function buildImage(image) { console.log(`Building ${image.name} fixture...`) - docker([ + const buildArgs = [ 'build', '--build-arg', `BASE_IMAGE=${image.base}`, @@ -81,7 +81,16 @@ function buildImage(image) { '-t', image.tag, '.' - ]) + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + try { + docker(buildArgs) + } catch (error) { + console.error( + `${error instanceof Error ? error.message : String(error)}\nRetrying docker build once...` + ) + docker(buildArgs) + } } function extractAppImage(image) { diff --git a/config/scripts/run-headless-serve-shutdown-docker.mjs b/config/scripts/run-headless-serve-shutdown-docker.mjs index 184713c41a0..d8dcdd345ad 100755 --- a/config/scripts/run-headless-serve-shutdown-docker.mjs +++ b/config/scripts/run-headless-serve-shutdown-docker.mjs @@ -43,7 +43,7 @@ const artifactVolume = `orca-headless-serve-shutdown-${suffix}` const sha256 = createHash('sha256').update(readFileSync(appImage)).digest('hex') try { - docker([ + const buildArgs = [ 'build', '--platform', platform, @@ -52,7 +52,15 @@ try { '-t', image, shutdownDockerDirectory - ]) + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + const firstBuild = docker(buildArgs, { allowFailure: true }) + if (firstBuild.status !== 0) { + process.stderr.write( + `${firstBuild.stdout}${firstBuild.stderr}\ndocker build failed with status ${firstBuild.status}; retrying once...\n` + ) + docker(buildArgs) + } docker(['volume', 'create', artifactVolume]) runDesktopStartupOracle({ image, appImage, platform }) docker([ diff --git a/config/scripts/run-linux-cli-launch-contract-docker.mjs b/config/scripts/run-linux-cli-launch-contract-docker.mjs index 901e0877e85..bd414947026 100755 --- a/config/scripts/run-linux-cli-launch-contract-docker.mjs +++ b/config/scripts/run-linux-cli-launch-contract-docker.mjs @@ -173,20 +173,26 @@ function runCase(caseName) { function buildImage() { console.log(`Building ${tag}…`) - docker( - [ - 'build', - ...dockerPlatformArgs, - '--build-arg', - `BASE_IMAGE=${base}`, - '-f', - 'config/docker/cli-launch-contract/Dockerfile', - '-t', - tag, - 'config/docker/cli-launch-contract' - ], - { timeoutMs: BUILD_TIMEOUT_MS } - ) + const buildArgs = [ + 'build', + ...dockerPlatformArgs, + '--build-arg', + `BASE_IMAGE=${base}`, + '-f', + 'config/docker/cli-launch-contract/Dockerfile', + '-t', + tag, + 'config/docker/cli-launch-contract' + ] + // Why: apt fetches from archive.ubuntu.com stall or fail mid-sync; a second build usually lands on a healthy index. + try { + docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS }) + } catch (error) { + console.error( + `${error instanceof Error ? error.message : String(error)}\nRetrying docker build once…` + ) + docker(buildArgs, { timeoutMs: BUILD_TIMEOUT_MS }) + } } // Extract unprivileged so chrome-sandbox is not root-owned setuid. diff --git a/config/scripts/session-write-hot-path-benchmark.mjs b/config/scripts/session-write-hot-path-benchmark.mjs new file mode 100644 index 00000000000..361e7eb7098 --- /dev/null +++ b/config/scripts/session-write-hot-path-benchmark.mjs @@ -0,0 +1,221 @@ +#!/usr/bin/env node +// Benchmarks two CPU costs `setLocalWorkspaceSession` pays on every session write — the write +// that fires on something as ordinary as clicking between two terminal split panes. +// +// 1. capTerminalScrollbackSessionBuffer — UTF-8 budget scan per retained scrollback buffer +// 2. remapPaneKeys — pane-key map rebuild that steady state throws away +// +// The snapshot disk rewrite on the same path is measured separately (#18764). +// +// Each scenario runs the production export against a baseline that reproduces the pre-change +// shape, so the reported speedup cannot drift away from what production actually does. +import { spawnSync } from 'node:child_process' +import { performance } from 'node:perf_hooks' +import fs from 'node:fs' +import nodeModule from 'node:module' +import path from 'node:path' +import process from 'node:process' +import { fileURLToPath } from 'node:url' + +if (!process.execArgv.includes('--experimental-transform-types')) { + const result = spawnSync( + process.execPath, + ['--experimental-transform-types', '--no-warnings', import.meta.filename], + { stdio: 'inherit' } + ) + process.exit(result.status ?? 1) +} + +// The app's TS sources import siblings without an extension; Node's ESM resolver needs it. +nodeModule.registerHooks({ + resolve(specifier, context, nextResolve) { + if (specifier.startsWith('.') && !/\.[cm]?[jt]s$/.test(specifier) && context.parentURL) { + const candidate = new URL(`${specifier}.ts`, context.parentURL) + if (fs.existsSync(fileURLToPath(candidate))) { + return { url: candidate.href, shortCircuit: true } + } + } + return nextResolve(specifier, context) + } +}) + +const ROOT = path.resolve(import.meta.dirname, '../..') +const ROUNDS = Number(process.env.ORCA_SESSION_WRITE_BENCH_ROUNDS ?? '9') +const LEAVES = Number(process.env.ORCA_SESSION_WRITE_BENCH_LEAVES ?? '8') +const PANE_KEYS = Number(process.env.ORCA_SESSION_WRITE_BENCH_PANE_KEYS ?? '2000') + +for (const [name, value] of [ + ['ORCA_SESSION_WRITE_BENCH_ROUNDS', ROUNDS], + ['ORCA_SESSION_WRITE_BENCH_LEAVES', LEAVES], + ['ORCA_SESSION_WRITE_BENCH_PANE_KEYS', PANE_KEYS] +]) { + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`${name} must be a positive integer, got ${value}`) + } +} + +const { capTerminalScrollbackSessionBuffer } = await import( + path.join(ROOT, 'src/shared/workspace-session-terminal-buffers.ts') +) +const { TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT } = await import( + path.join(ROOT, 'src/shared/terminal-scrollback-limits.ts') +) +const { remapAcknowledgedAgentPaneKeys } = await import( + path.join(ROOT, 'src/main/persistence/restoring-sessions/pane-key-remapping.ts') +) +const { clampUtf8TextTail, measureUtf8ByteLength } = await import( + path.join(ROOT, 'src/shared/utf8-byte-limits.ts') +) +const { isTerminalLeafId, makePaneKey, parsePaneKey } = await import( + path.join(ROOT, 'src/shared/stable-pane-id.ts') +) + +function median(samples) { + const sorted = [...samples].sort((left, right) => left - right) + return sorted[Math.floor(sorted.length / 2)] +} + +function timeRounds(run) { + const samples = [] + run() + for (let round = 0; round < ROUNDS; round += 1) { + const start = performance.now() + run() + samples.push(performance.now() - start) + } + return median(samples) +} + +function report(label, baselineMs, currentMs, extra = '') { + const speedup = baselineMs / currentMs + console.log( + `${label}\n before ${baselineMs.toFixed(3)} ms → after ${currentMs.toFixed(3)} ms (${speedup.toFixed(1)}x)${extra}` + ) + return speedup +} + +// ---------------------------------------------------------------- scenario 1 + +// Verbatim pre-change capTerminalScrollbackSessionBuffer; measureUtf8ByteLength itself is unchanged. +function baselineCapScrollbackBuffer(buffer) { + if ( + buffer.length <= TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT && + !measureUtf8ByteLength(buffer, { + stopAfterBytes: TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT + }).exceededLimit + ) { + return buffer + } + return clampUtf8TextTail(buffer, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT).text +} + +// A terminal that has been running a while sits at the cap, which is the case that scanned in full. +const scrollbackLine = `${''}build output line with a path /Users/dev/project/src/index.ts and a status ok\n` +let atCapBuffer = '' +while (atCapBuffer.length < TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) { + atCapBuffer += scrollbackLine +} +atCapBuffer = atCapBuffer.slice(0, TERMINAL_SCROLLBACK_SESSION_BUFFER_BYTE_LIMIT) + +if (capTerminalScrollbackSessionBuffer(atCapBuffer) !== baselineCapScrollbackBuffer(atCapBuffer)) { + throw new Error('scrollback cap disagreed with the baseline implementation') +} + +// The session write runs the prune twice, once per retained leaf. +const CAP_CALLS_PER_WRITE = LEAVES * 2 +const capBaselineMs = timeRounds(() => { + for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) { + baselineCapScrollbackBuffer(atCapBuffer) + } +}) +const capCurrentMs = timeRounds(() => { + for (let call = 0; call < CAP_CALLS_PER_WRITE; call += 1) { + capTerminalScrollbackSessionBuffer(atCapBuffer) + } +}) + +console.log( + `Session-write hot path — ${LEAVES} retained scrollback leaves, ${PANE_KEYS} accumulated pane keys\n` +) +report( + `1. scrollback UTF-8 budget scan (${CAP_CALLS_PER_WRITE} calls/write @ ${(atCapBuffer.length / 1024).toFixed(0)} KB)`, + capBaselineMs, + capCurrentMs +) + +// ---------------------------------------------------------------- scenario 2 + +const paneKeys = {} +const leafIdByInputLeafIdByTabId = new Map() +for (let index = 0; index < PANE_KEYS; index += 1) { + const tabId = `tab-${index % 64}` + const leafId = `${(index % 64).toString(16).padStart(8, '0')}-0000-4000-8000-${index.toString(16).padStart(12, '0')}` + paneKeys[makePaneKey(tabId, leafId)] = index + let leaves = leafIdByInputLeafIdByTabId.get(tabId) + if (!leaves) { + leaves = new Map() + leafIdByInputLeafIdByTabId.set(tabId, leaves) + } + // Steady state: a stable UUID leaf maps to itself. + leaves.set(leafId, leafId) +} + +// Verbatim pre-change remapPaneKeys: parses every key, then rebuilds the object regardless. +function baselineRemapPaneKeys(values, remap) { + if (!values || Object.keys(values).length === 0) { + return { values, changed: false } + } + let changed = false + const next = {} + const setValue = (paneKey, value) => { + const existing = next[paneKey] + next[paneKey] = existing === undefined ? value : Math.max(existing, value) + } + for (const [paneKey, value] of Object.entries(values)) { + if (parsePaneKey(paneKey)) { + setValue(paneKey, value) + continue + } + const delimiter = paneKey.indexOf(':') + if (delimiter <= 0 || delimiter === paneKey.length - 1) { + setValue(paneKey, value) + continue + } + const tabId = paneKey.slice(0, delimiter) + const remappedLeafId = remap.get(tabId)?.get(paneKey.slice(delimiter + 1)) + if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { + setValue(paneKey, value) + continue + } + try { + setValue(makePaneKey(tabId, remappedLeafId), value) + changed = true + } catch { + setValue(paneKey, value) + } + } + return { values: next, changed } +} + +// The write remaps three of these maps: acknowledgements, activity cutoffs, manual unread. +const REMAP_CALLS_PER_WRITE = 3 +const remapBaselineMs = timeRounds(() => { + for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) { + baselineRemapPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) + } +}) +const remapCurrentMs = timeRounds(() => { + for (let call = 0; call < REMAP_CALLS_PER_WRITE; call += 1) { + remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) + } +}) +const remapResult = remapAcknowledgedAgentPaneKeys(paneKeys, leafIdByInputLeafIdByTabId) +if (remapResult.changed || remapResult.acknowledgements !== paneKeys) { + throw new Error('steady-state remap should return the input map untouched') +} +report( + `2. pane-key remap (${REMAP_CALLS_PER_WRITE} maps/write @ ${PANE_KEYS} keys)`, + remapBaselineMs, + remapCurrentMs, + ' — and 3 discarded objects/write become 0' +) diff --git a/config/scripts/terminal-partial-escape-tail-benchmark.mjs b/config/scripts/terminal-partial-escape-tail-benchmark.mjs new file mode 100644 index 00000000000..0337daa4e9d --- /dev/null +++ b/config/scripts/terminal-partial-escape-tail-benchmark.mjs @@ -0,0 +1,88 @@ +#!/usr/bin/env node +// Times the partial-escape-tail fold that runs once per PTY chunk for every terminal against a +// baseline with the pre-change shape (unconditional concat + per-code-unit walk). Equivalence is +// proven over a corpus first, so the reported speedup cannot come from the gate changing the answer. +import { performance } from 'node:perf_hooks' +import { + advancePartialEscapeTail, + extractPartialEscapeTail, + MAX_PARTIAL_ESCAPE_TAIL_LENGTH +} from '../../src/shared/terminal-partial-escape-tail.ts' + +const CHUNK_BYTES = 16 * 1024 +const CHUNKS = 640 +const ROUNDS = 7 + +function baselineAdvance(pendingTail, chunk) { + const tail = extractPartialEscapeTail(pendingTail + chunk) + return tail.length > MAX_PARTIAL_ESCAPE_TAIL_LENGTH ? '' : tail +} + +const chunkOf = (line) => line.repeat(Math.ceil(CHUNK_BYTES / line.length)).slice(0, CHUNK_BYTES) +const escFreeChunk = chunkOf('[build] compiled src/renderer/src/components/thing.tsx in 12ms\n') +const colouredChunk = chunkOf( + '\x1b[32m[build]\x1b[0m compiled src/renderer/src/components/thing.tsx in 12ms\n' +) + +// Every state the scanner can be left in, plus the boundaries the gate must not swallow. +const PIECES = [ + '', + 'plain output\n', + '\x1b[32mgreen\x1b[0m', + '\x1b[3', + '\x1b]0;title\x07', + '\x1b]0;partial', + '\x1bP dcs payload', + '\x1b', + '\x18', + '\x1a', + '\x1b]8;;https://example.com\x1b\\', + '\x1b]8;;https://example.com\x1b', + '\x1b(B', + '\x1b(', + '\x1b[1;2;3', + escFreeChunk +] +let checked = 0 +for (const pending of PIECES.map((piece) => extractPartialEscapeTail(piece))) { + for (const chunk of PIECES) { + const expected = baselineAdvance(pending, chunk) + const actual = advancePartialEscapeTail(pending, chunk) + if (expected !== actual) { + throw new Error( + `gate changed the tracked tail: ${JSON.stringify({ pending, chunk, expected, actual })}` + ) + } + checked += 1 + } +} + +function medianMs(advance, chunk) { + // First sample is the warm-up and is discarded. + const samples = Array.from({ length: ROUNDS + 1 }, () => { + const start = performance.now() + let tail = '' + for (let index = 0; index < CHUNKS; index += 1) { + tail = advance(tail, chunk) + } + return performance.now() - start + }) + return samples.slice(1).sort((left, right) => left - right)[Math.floor(ROUNDS / 2)] +} + +const megabytes = ((CHUNK_BYTES * CHUNKS) / 1024 / 1024).toFixed(1) +console.log( + `Partial-escape-tail fold: ${CHUNKS} x ${CHUNK_BYTES / 1024} KB chunks (${megabytes} MB), ${checked} equivalence cases verified\n` +) +console.log('| stream shape | before | after | |') +console.log('| --- | --- | --- | --- |') +for (const [label, chunk] of [ + ['ESC-free (build logs, `cat`, piped output)', escFreeChunk], + ['SGR-coloured output (gate does not apply)', colouredChunk] +]) { + const before = medianMs(baselineAdvance, chunk) + const after = medianMs(advancePartialEscapeTail, chunk) + console.log( + `| ${label} | ${before.toFixed(2)} ms | ${after.toFixed(2)} ms | ${(before / after).toFixed(1)}x |` + ) +} diff --git a/docs/site/content/docs/browser/profiles.mdx b/docs/site/content/docs/browser/profiles.mdx index 97286a6cb0c..102dd1e443d 100644 --- a/docs/site/content/docs/browser/profiles.mdx +++ b/docs/site/content/docs/browser/profiles.mdx @@ -9,9 +9,7 @@ Browser-use profiles let you run the Orca browser with a specific identity — a 1. Open [Settings → Browser → Profiles](/docs/settings). 1. Click **Add profile**, give it a name. 1. Optionally seed it with cookies, a user-agent, and a viewport size. -1. For sites that reject Orca's default Chrome-shaped UA (some Google sign-in flows), create a profile that keeps the **native Electron user agent** instead of spoofing. Default profiles still use the cleaned Chrome UA for broader Cloudflare compatibility. - -You can also create a no-spoof profile from the CLI with `orca tab profile create --no-ua-spoof` when you script browser setup. +1. Every profile presents Electron's own user agent. Orca no longer rewrites it to look like Chrome, because Cloudflare Turnstile rejects a Chrome-shaped UA that sends no client hints and accepts a declared Electron client. The only exception is Google's sign-in hosts, where Orca presents a Firefox identity so Google issues cookies bound to the embedded browser. A **native user agent** profile (`orca tab profile create --no-ua-spoof`) also skips that Google exception. ## Cookie import and Google sign-in diff --git a/mobile/app.json b/mobile/app.json index d5a420cc74b..131e3899396 100644 --- a/mobile/app.json +++ b/mobile/app.json @@ -2,7 +2,7 @@ "expo": { "name": "Orca", "slug": "orca-mobile", - "version": "0.0.47", + "version": "0.0.48", "orientation": "default", "icon": "./assets/icon.png", "userInterfaceStyle": "automatic", diff --git a/mobile/src/components/MobileMarkdown.tsx b/mobile/src/components/MobileMarkdown.tsx index cc88c01e564..2f5b52cfe18 100644 --- a/mobile/src/components/MobileMarkdown.tsx +++ b/mobile/src/components/MobileMarkdown.tsx @@ -192,6 +192,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return ( {renderInline(block.text, onOpenFile)} @@ -201,7 +202,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile if (block.type === 'quote') { return ( - {renderInline(block.text, onOpenFile)} + + {renderInline(block.text, onOpenFile)} + ) } @@ -223,7 +226,9 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile return ( {block.language ? {block.language} : null} - {block.text} + + {block.text} + ) } @@ -251,7 +256,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleHeaders.map((header, cellIndex) => ( - + {renderInline(header, onOpenFile)} ))} @@ -259,7 +264,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile {visibleRows.map((row, rowIndex) => ( {visibleHeaders.map((_, cellIndex) => ( - + {renderInline(row[cellIndex] ?? '', onOpenFile)} ))} @@ -290,7 +295,7 @@ function MobileMarkdownInner({ content, fallback = '', textScale = 1, onOpenFile ? '[x]' : '[ ]'} - + {renderInline(item.text, onOpenFile)} diff --git a/mobile/src/session/MobileNativeChatMessage.test.ts b/mobile/src/session/MobileNativeChatMessage.test.ts index 10b96cc3d20..677b76d240c 100644 --- a/mobile/src/session/MobileNativeChatMessage.test.ts +++ b/mobile/src/session/MobileNativeChatMessage.test.ts @@ -6,11 +6,21 @@ import type { NativeChatMessage } from '../../../src/shared/native-chat-types' vi.mock('react-native', async () => { const React = await import('react') + const Text = ({ children, ...props }: { children?: unknown }): unknown => + React.createElement('Text', props, children) return { + Animated: { + Text, + Value: class { + setValue(): void {} + }, + loop: (animation: unknown) => animation, + sequence: () => ({ start: vi.fn(), stop: vi.fn() }), + timing: () => ({ start: vi.fn(), stop: vi.fn() }) + }, Image: 'Image', Pressable: 'Pressable', - Text: ({ children, ...props }: { children?: unknown }) => - React.createElement('Text', props, children), + Text, View: ({ children, ...props }: { children?: unknown }) => React.createElement('View', props, children), StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } @@ -21,7 +31,10 @@ vi.mock('lucide-react-native', () => ({ ArrowUp: 'ArrowUp', ChevronDown: 'ChevronDown', Copy: 'Copy', - SquareChevronRight: 'SquareChevronRight' + SquareChevronRight: 'SquareChevronRight', + SquareTerminal: 'SquareTerminal', + Wrench: 'Wrench', + ChevronRight: 'ChevronRight' })) vi.mock('../components/MobileMarkdown', () => ({ MobileMarkdown: 'MobileMarkdown' })) @@ -45,7 +58,18 @@ describe('MobileNativeChatMessage', () => { function render( message: NativeChatMessage, - props: { toolsExpanded?: boolean } = {} + props: { + toolsExpanded?: boolean + structuredActivityUi?: boolean + activeTurnIsWorking?: boolean + turnExpanded?: boolean + turnStatus?: { + startedAt: number | null + thinking: boolean + workedSeconds: number | null + } | null + onToggleTurn?: () => void + } = {} ): ReactTestRenderer { act(() => { renderer = create(createElement(MobileNativeChatMessage, { message, ...props })) @@ -152,4 +176,96 @@ describe('MobileNativeChatMessage', () => { expect(tree.root.findAllByType('ChevronDown' as never)).toHaveLength(1) expect(tree.root.findAllByType('SquareChevronRight' as never)).toHaveLength(1) }) + + describe('structured activity UI', () => { + const runningCall = { + type: 'tool-call' as const, + name: 'Bash', + input: { command: 'npm test' }, + state: 'running' as const + } + const settledCall = { + type: 'tool-call' as const, + name: 'Read', + input: { file_path: 'a/b.ts' }, + state: 'completed' as const + } + + it('shows the live tool label with a terminal glyph while a command runs', () => { + const tree = render(toolMessage([runningCall]), { + structuredActivityUi: true, + activeTurnIsWorking: true + }) + expect(textIn(tree.root)).toContain('Running npm test') + expect(tree.root.findAllByType('SquareTerminal' as never)).toHaveLength(1) + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) + }) + + it('uses the wrench glyph for a non-command tool', () => { + const tree = render( + toolMessage([ + { type: 'tool-call', name: 'Read', input: { file_path: 'a/b.ts' }, state: 'running' } + ]), + { structuredActivityUi: true, activeTurnIsWorking: true } + ) + expect(textIn(tree.root)).toContain('Running Read a/b.ts') + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(1) + }) + + it('falls back to the collapsed count row once the run settles', () => { + const tree = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: true + }) + expect(textIn(tree.root)).not.toContain('Running Read a/b.ts') + expect(textIn(tree.root)).toContain('1×') + }) + + it("hides a completed turn's activity until the turn caret discloses it", () => { + const collapsed = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false + }) + expect(textIn(collapsed.root)).not.toContain('1×') + act(() => collapsed.unmount()) + + const disclosed = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false, + turnExpanded: true + }) + expect(textIn(disclosed.root)).toContain('1×') + }) + + it('lets the global Tools toggle reveal a hidden settled run', () => { + // Otherwise the composer's Tools control is a no-op on every settled turn. + const tree = render(toolMessage([settledCall]), { + structuredActivityUi: true, + activeTurnIsWorking: false, + toolsExpanded: true + }) + expect(textIn(tree.root)).toContain('1\u00d7') + }) + + it('keeps the bridge lane on its always-visible tool run', () => { + const tree = render(toolMessage([settledCall]), { activeTurnIsWorking: false }) + expect(textIn(tree.root)).toContain('1×') + expect(tree.root.findAllByType('Wrench' as never)).toHaveLength(0) + }) + + it('renders the turn status row under a user message', () => { + const tree = render(userMessage([{ type: 'text', text: 'go' }]), { + structuredActivityUi: true, + turnStatus: { startedAt: Date.now(), thinking: true, workedSeconds: null } + }) + expect(textIn(tree.root)).toContain('Thinking') + }) + + it('does not render a turn status row without one', () => { + const tree = render(userMessage([{ type: 'text', text: 'go' }]), { + structuredActivityUi: true + }) + expect(textIn(tree.root)).toEqual(['go']) + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatMessage.tsx b/mobile/src/session/MobileNativeChatMessage.tsx index 0a676061e5b..cc6c086b1f3 100644 --- a/mobile/src/session/MobileNativeChatMessage.tsx +++ b/mobile/src/session/MobileNativeChatMessage.tsx @@ -1,139 +1,20 @@ import { memo, useEffect, useRef, useState } from 'react' import { Image, Pressable, Text, View } from 'react-native' import * as Clipboard from 'expo-clipboard' -import { ArrowUp, ChevronDown, Copy, SquareChevronRight } from 'lucide-react-native' -import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff' -import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff' -import { pairToolBlocks, splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' -import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold' -import { - createToolInputDisplay, - summarizeToolRun, - truncateToolDetail -} from '../../../src/shared/native-chat-tool-summary' +import { ArrowUp, Copy } from 'lucide-react-native' +import { splitNativeChatBlocks } from '../../../src/shared/native-chat-tool-fold' +import { selectActiveToolCall } from '../../../src/shared/native-chat-tool-activity' import { isImageRefBlock, isTextBlock } from '../../../src/shared/native-chat-types' import type { NativeChatBlock, NativeChatMessage } from '../../../src/shared/native-chat-types' import { MobileMarkdown } from '../components/MobileMarkdown' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' +import { ToolRun } from './MobileNativeChatToolRun' +import type { NativeChatTurnStatus } from './use-mobile-native-chat-turn-status' import { colors } from '../theme/mobile-theme' import { isRenderableImageUri } from './mobile-native-chat-image-preview' import { styles, TEXT_SIZE } from './mobile-native-chat-message-styles' import { nativeChatMessageText } from './mobile-native-chat-message-text' -const MAX_VISIBLE_TOOL_PAIRS = 6 -const MAX_TOOL_RUN_DIFF_ROWS = 240 - -function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element { - return ( - - {lines.map((line, i) => ( - - {line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '} - {line.text} - - ))} - - ) -} - -/** A single inline tool line — `▸ ToolName preview` — that expands in place to - * show the call's diff/input or the result's body. Mirrors the reference design - * where tool calls read as flat lines in the conversation, not boxed blocks. */ -function ResultBody({ - output, - isError, - diff -}: { - output: string - isError?: boolean - diff: DiffLine[] | null -}): React.JSX.Element { - if (diff) { - return - } - return ( - - {truncateToolDetail(output)} - - ) -} - -/** One request: a tool call and its result rendered together as a single - * expandable line. `defaultExpanded` lets the group toggle open every line. */ -function ToolLine({ - pair, - defaultExpanded, - diffLineLimit, - onOpenFile -}: { - pair: ToolPair - defaultExpanded: boolean - diffLineLimit: number - onOpenFile?: (relativePath: string) => void -}): React.JSX.Element { - const [expanded, setExpanded] = useState(defaultExpanded) - const { call, result } = pair - const name = call ? call.name : 'Result' - const inputDisplay = call ? createToolInputDisplay(call.input) : null - const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? '' - // Why: collapsed tool rows are the common path; defer bounded diff parsing - // and detail formatting until the user asks to reveal the detail. - const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null - const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null - const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined - const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true - // The group toggle opens every line at once, bypassing the tap guard, so the - // panel has to consult it too — else a detail-less row echoes its own label - // under itself and no tap can dismiss it. - const showDetail = hasDetail && expanded - // A tool that targets a file (Read/Edit/Write…) renders its preview as a - // tappable link that opens the file, independent of the line's expand tap. - const filePath = inputDisplay?.filePath ?? null - const openable = filePath !== null && onOpenFile !== undefined - return ( - - hasDetail && setExpanded((v) => !v)} - hitSlop={6} - > - {showDetail ? ( - - ) : ( - - )} - {name} - {preview ? ( - onOpenFile!(filePath!) : undefined} - suppressHighlighting={!openable} - > - {preview} - - ) : null} - - {showDetail ? ( - - {callDiff ? : null} - {callDetail ? {callDetail} : null} - {result ? ( - - ) : null} - - ) : null} - - ) -} - function Prose({ block, invert, @@ -150,7 +31,9 @@ function Prose({ // markdown renderer's light-on-dark palette. if (invert) { return ( - {block.text} + + {block.text} + ) } return ( @@ -180,67 +63,6 @@ function Prose({ return null } -/** A run of a message's tool calls/results, collapsed to a one-line summary that - * expands to the individual inline tool lines. `defaultExpanded` lets the global - * toolbar toggle drive every run at once while still allowing per-run override. */ -function ToolRun({ - blocks, - defaultExpanded, - trailing, - onOpenFile -}: { - blocks: NativeChatBlock[] - defaultExpanded: boolean - trailing?: React.ReactNode - onOpenFile?: (relativePath: string) => void -}): React.JSX.Element { - const [open, setOpen] = useState(defaultExpanded) - const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS) - const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1))) - let callCount = 0 - for (const block of blocks) { - if (block.type === 'tool-call') { - callCount++ - } - } - callCount ||= pairs.length - const summary = summarizeToolRun(blocks) - return ( - - - setOpen((v) => !v)} hitSlop={6}> - {open ? ( - - ) : ( - - )} - {callCount}× - - {summary || `${callCount} tool ${callCount === 1 ? 'call' : 'calls'}`} - - - {trailing} - - {open ? ( - - {pairs.map((pair, i) => ( - - ))} - {callCount > pairs.length ? ( - … {callCount - pairs.length} more tool calls - ) : null} - - ) : null} - - ) -} - /** Subtle top-right controls for an agent message: copy its prose, or scroll so * this message's top aligns to the top of the viewport. */ function AgentControls({ @@ -280,7 +102,13 @@ function MobileNativeChatMessageImpl({ fontScale = 1, messageIndex, onScrollToMessage, - onOpenFile + onOpenFile, + turnStatus, + turnExpanded, + turnKey, + onToggleTurn, + activeTurnIsWorking, + structuredActivityUi = false }: { message: NativeChatMessage toolsExpanded?: boolean @@ -291,6 +119,18 @@ function MobileNativeChatMessageImpl({ /** Ask the list to align this message's top to the top of the viewport. */ onScrollToMessage?: (index: number) => void onOpenFile?: (relativePath: string) => void + /** This turn's status row, rendered under a user message (desktop parity). */ + turnStatus?: NativeChatTurnStatus | null + /** Whether the turn caret has disclosed this turn's activity. */ + turnExpanded?: boolean + /** Set only when this row's turn has settled and can disclose its activity. */ + turnKey?: string + /** Stable across renders; the row supplies its own key when tapped. */ + onToggleTurn?: (turnKey: string) => void + /** Session-level working state for this message's turn; gates the live tool row. */ + activeTurnIsWorking?: boolean + /** Structured lane only: live tool progress plus the turn-status disclosure. */ + structuredActivityUi?: boolean }): React.JSX.Element { const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' @@ -310,6 +150,20 @@ function MobileNativeChatMessageImpl({ // tool calls fold into a collapsible run beneath. The user's own messages get // an inverted (filled accent) bubble so they stand apart from agent prose. const { prose, tools } = splitNativeChatBlocks(message.blocks) + const activeCall = structuredActivityUi + ? selectActiveToolCall(tools, { activeTurnIsWorking }) + : null + // A completed turn's activity belongs behind the turn-status caret. Leaving the + // grouped row visible made a failed child command read as a failed response. + // The composer's global Tools toggle still overrides this, or it would silently + // do nothing on every settled turn. + const settledToolsHidden = + structuredActivityUi && + activeCall == null && + activeTurnIsWorking === false && + !turnExpanded && + !toolsExpanded + const showToolRun = tools.length > 0 && !settledToolsHidden const handleCopy = (): void => { const text = nativeChatMessageText(message.blocks) @@ -338,39 +192,52 @@ function MobileNativeChatMessageImpl({ ) : null return ( - - - {prose.map((block, index) => ( - - ))} - {tools.length > 0 ? ( - - ) : controls ? ( - {controls} - ) : null} + <> + + + {prose.map((block, index) => ( + + ))} + {showToolRun ? ( + + ) : controls ? ( + {controls} + ) : null} + - + {turnStatus ? ( + onToggleTurn(turnKey) : undefined} + /> + ) : null} + ) } diff --git a/mobile/src/session/MobileNativeChatOverlay.tsx b/mobile/src/session/MobileNativeChatOverlay.tsx index 23724300ddf..357a089466e 100644 --- a/mobile/src/session/MobileNativeChatOverlay.tsx +++ b/mobile/src/session/MobileNativeChatOverlay.tsx @@ -71,6 +71,7 @@ export function MobileNativeChatOverlay({ error={session.error} agent={controller.nativeChatAgent} agentWorking={controller.nativeChatAgentWorking} + structuredActivityUi={controller.nativeChatStructured} streaming={streaming} onStop={controller.handleNativeChatStop} ask={controller.nativeChatAsk} diff --git a/mobile/src/session/MobileNativeChatPromptCard.tsx b/mobile/src/session/MobileNativeChatPromptCard.tsx new file mode 100644 index 00000000000..470ba2ee8b6 --- /dev/null +++ b/mobile/src/session/MobileNativeChatPromptCard.tsx @@ -0,0 +1,74 @@ +import type { AskAnswerSelection, AskPrompt } from '../../../src/shared/native-chat-ask' +import { MobileNativeChatAsk } from './MobileNativeChatAsk' +import { MobileNativeChatPermission } from './MobileNativeChatPermission' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import { MobileNativeChatQuestion } from './MobileNativeChatQuestion' +import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question' + +/** The one pending agent prompt shown above the composer: a structured + * AskUserQuestion wins, then a heuristic permission, then a heuristic question. + * The controller owns dismissal (it must survive this subtree unmounting on a + * view toggle); `ask` arrives already nulled while dismissed. */ +export function MobileNativeChatPromptCard({ + ask, + askKey, + onDismissAsk, + onAnswerAsk, + onCancelAsk, + permission, + onRespondPermission, + question, + onAnswerQuestion +}: { + ask?: AskPrompt | null + askKey?: string | null + onDismissAsk?: () => void + onAnswerAsk?: (prompt: AskPrompt, selections: AskAnswerSelection[]) => Promise + onCancelAsk?: () => Promise + permission?: MobileChatPermission | null + onRespondPermission?: (send: string) => Promise + question?: MobileChatQuestion | null + onAnswerQuestion?: (text: string) => Promise +}): React.JSX.Element | null { + if (ask) { + return ( + { + const accepted = (await onAnswerAsk?.(ask, selections)) ?? false + if (accepted) { + onDismissAsk?.() + } + return accepted + }} + onCancel={async () => { + const accepted = (await onCancelAsk?.()) ?? false + if (accepted) { + onDismissAsk?.() + } + return accepted + }} + /> + ) + } + if (permission) { + return ( + (await onRespondPermission?.(send)) ?? false} + /> + ) + } + if (question) { + return ( + (await onAnswerQuestion?.(text)) ?? false} + /> + ) + } + return null +} diff --git a/mobile/src/session/MobileNativeChatToolRun.tsx b/mobile/src/session/MobileNativeChatToolRun.tsx new file mode 100644 index 00000000000..db3ddbea063 --- /dev/null +++ b/mobile/src/session/MobileNativeChatToolRun.tsx @@ -0,0 +1,250 @@ +import { useEffect, useRef, useState } from 'react' +import { Animated, Pressable, Text, View } from 'react-native' +import { ChevronDown, SquareChevronRight, SquareTerminal, Wrench } from 'lucide-react-native' +import { diffFromText, diffFromToolCall } from '../../../src/shared/native-chat-diff' +import type { NativeChatDiffLine as DiffLine } from '../../../src/shared/native-chat-diff' +import { pairToolBlocks } from '../../../src/shared/native-chat-tool-fold' +import type { NativeChatToolPair as ToolPair } from '../../../src/shared/native-chat-tool-fold' +import { + createToolInputDisplay, + summarizeToolRun, + truncateToolDetail +} from '../../../src/shared/native-chat-tool-summary' +import { + describeActiveToolCall, + formatActiveToolLabel, + formatToolCallCount, + isCommandToolName, + selectActiveToolCall +} from '../../../src/shared/native-chat-tool-activity' +import type { NativeChatBlock } from '../../../src/shared/native-chat-types' +import { colors } from '../theme/mobile-theme' +import { styles } from './mobile-native-chat-message-styles' + +const MAX_VISIBLE_TOOL_PAIRS = 6 +const MAX_TOOL_RUN_DIFF_ROWS = 240 + +function DiffView({ lines }: { lines: DiffLine[] }): React.JSX.Element { + return ( + + {lines.map((line, i) => ( + + {line.kind === 'add' ? '+' : line.kind === 'del' ? '-' : ' '} + {line.text} + + ))} + + ) +} + +/** A single inline tool line — `▸ ToolName preview` — that expands in place to + * show the call's diff/input or the result's body. Mirrors the reference design + * where tool calls read as flat lines in the conversation, not boxed blocks. */ +function ResultBody({ + output, + isError, + diff +}: { + output: string + isError?: boolean + diff: DiffLine[] | null +}): React.JSX.Element { + if (diff) { + return + } + return ( + + {truncateToolDetail(output)} + + ) +} + +/** One request: a tool call and its result rendered together as a single + * expandable line. `defaultExpanded` lets the group toggle open every line. */ +function ToolLine({ + pair, + defaultExpanded, + diffLineLimit, + onOpenFile +}: { + pair: ToolPair + defaultExpanded: boolean + diffLineLimit: number + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const [expanded, setExpanded] = useState(defaultExpanded) + const { call, result } = pair + const name = call ? call.name : 'Result' + const inputDisplay = call ? createToolInputDisplay(call.input) : null + const preview = inputDisplay?.label ?? result?.output.split('\n')[0]?.slice(0, 80) ?? '' + // Why: collapsed tool rows are the common path; defer bounded diff parsing + // and detail formatting until the user asks to reveal the detail. + const callDiff = expanded && call ? diffFromToolCall(call.name, call.input, diffLineLimit) : null + const resultDiff = expanded && result ? diffFromText(result.output, diffLineLimit) : null + const callDetail = expanded && inputDisplay && !callDiff ? inputDisplay.formatDetail() : undefined + const hasDetail = callDiff !== null || result !== undefined || inputDisplay?.hasDetail === true + // The group toggle opens every line at once, bypassing the tap guard, so the + // panel has to consult it too — else a detail-less row echoes its own label + // under itself and no tap can dismiss it. + const showDetail = hasDetail && expanded + // A tool that targets a file (Read/Edit/Write…) renders its preview as a + // tappable link that opens the file, independent of the line's expand tap. + const filePath = inputDisplay?.filePath ?? null + const openable = filePath !== null && onOpenFile !== undefined + return ( + + hasDetail && setExpanded((v) => !v)} + hitSlop={6} + > + {showDetail ? ( + + ) : ( + + )} + {name} + {preview ? ( + onOpenFile!(filePath!) : undefined} + suppressHighlighting={!openable} + > + {preview} + + ) : null} + + {showDetail ? ( + + {callDiff ? : null} + {callDetail ? {callDetail} : null} + {result ? ( + + ) : null} + + ) : null} + + ) +} + +/** Breathing label for a still-running tool, matching desktop's `animate-pulse`. */ +function PulsingText({ + style, + numberOfLines, + children +}: { + style?: React.ComponentProps['style'] + numberOfLines?: number + children: React.ReactNode +}): React.JSX.Element { + const pulse = useRef(new Animated.Value(1)).current + useEffect(() => { + const animation = Animated.loop( + Animated.sequence([ + Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), + Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) + ]) + ) + animation.start() + return () => animation.stop() + }, [pulse]) + return ( + + {children} + + ) +} + +/** A run of a message's tool calls/results, collapsed to a one-line summary that + * expands to the individual inline tool lines. `defaultExpanded` lets the global + * toolbar toggle drive every run at once while still allowing per-run override. */ +export function ToolRun({ + blocks, + defaultExpanded, + expandChildren, + activeCall, + trailing, + onOpenFile +}: { + blocks: NativeChatBlock[] + defaultExpanded: boolean + /** Child tool lines stay collapsed when the turn caret drove the run open. */ + expandChildren: boolean + /** The still-running call, when the turn is live (desktop parity). */ + activeCall: ReturnType + trailing?: React.ReactNode + onOpenFile?: (relativePath: string) => void +}): React.JSX.Element { + const [open, setOpen] = useState(defaultExpanded) + const pairs = pairToolBlocks(blocks, MAX_VISIBLE_TOOL_PAIRS) + const diffLineLimit = Math.max(1, Math.floor(MAX_TOOL_RUN_DIFF_ROWS / (pairs.length * 2 || 1))) + let callCount = 0 + for (const block of blocks) { + if (block.type === 'tool-call') { + callCount++ + } + } + callCount ||= pairs.length + const summary = summarizeToolRun(blocks) + const ActiveToolIcon = activeCall && isCommandToolName(activeCall.name) ? SquareTerminal : Wrench + return ( + + + {activeCall ? ( + setOpen((v) => !v)} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded: open }} + accessibilityLiveRegion="polite" + > + + + {formatActiveToolLabel(describeActiveToolCall(activeCall))} + + {open ? : null} + + ) : ( + setOpen((v) => !v)} hitSlop={6}> + {open ? ( + + ) : ( + + )} + {callCount}× + + {summary || formatToolCallCount(callCount)} + + + )} + {trailing} + + {open ? ( + + {pairs.map((pair, i) => ( + + ))} + {callCount > pairs.length ? ( + … {callCount - pairs.length} more tool calls + ) : null} + + ) : null} + + ) +} diff --git a/mobile/src/session/MobileNativeChatTurnStatus.test.ts b/mobile/src/session/MobileNativeChatTurnStatus.test.ts new file mode 100644 index 00000000000..78ac01e0d37 --- /dev/null +++ b/mobile/src/session/MobileNativeChatTurnStatus.test.ts @@ -0,0 +1,112 @@ +import { createElement } from 'react' +import { act, create, type ReactTestInstance, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('react-native', async () => { + const React = await import('react') + const Text = ({ children, ...props }: { children?: unknown }): unknown => + React.createElement('Text', props, children) + return { + Animated: { + Text, + Value: class { + constructor(private value: number) {} + setValue(next: number): void { + this.value = next + } + }, + loop: (animation: unknown) => animation, + sequence: () => ({ start: vi.fn(), stop: vi.fn() }), + timing: () => ({ start: vi.fn(), stop: vi.fn() }) + }, + Pressable: ({ children, ...props }: { children?: unknown }) => + React.createElement('Pressable', props, children), + Text, + View: ({ children, ...props }: { children?: unknown }) => + React.createElement('View', props, children), + StyleSheet: { create: (styles: unknown) => styles, hairlineWidth: 1 } + } +}) +vi.mock('lucide-react-native', () => ({ ChevronRight: 'ChevronRight' })) + +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' + +describe('MobileNativeChatTurnStatus', () => { + let renderer: ReactTestRenderer | null = null + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-09-04T00:00:00Z')) + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + vi.useRealTimers() + }) + + function render(props: { + startedAt: number | null + thinking: boolean + workedSeconds?: number | null + expanded?: boolean + onToggleExpanded?: () => void + }): ReactTestRenderer { + act(() => { + renderer = create(createElement(MobileNativeChatTurnStatus, props)) + }) + return renderer! + } + + const labels = (node: ReactTestInstance): string[] => + node.findAllByType('Text' as never).map((text) => String(text.children.join(''))) + + it('reads "Thinking" before the turn produces output', () => { + const tree = render({ startedAt: Date.now(), thinking: true }) + expect(labels(tree.root)).toEqual(['Thinking']) + }) + + it('counts up once the turn is producing output', () => { + const startedAt = Date.now() + const tree = render({ startedAt, thinking: false }) + expect(labels(tree.root)).toEqual(['Working for 0s']) + act(() => { + vi.advanceTimersByTime(12_000) + }) + expect(labels(tree.root)).toEqual(['Working for 12s']) + }) + + it('settles to a tappable "Worked for" row that toggles the turn', () => { + const onToggleExpanded = vi.fn() + const tree = render({ + startedAt: Date.now(), + thinking: false, + workedSeconds: 184, + onToggleExpanded + }) + expect(labels(tree.root)).toEqual(['Worked for 3m 4s']) + const button = tree.root.findByType('Pressable' as never) + expect(button.props.accessibilityLabel).toBe('Toggle turn details') + expect(button.props.accessibilityState).toEqual({ expanded: false }) + act(() => button.props.onPress()) + expect(onToggleExpanded).toHaveBeenCalledOnce() + }) + + it('stays a plain row when the settled turn has nothing to disclose', () => { + const tree = render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + expect(tree.root.findAllByType('Pressable' as never)).toHaveLength(0) + expect(labels(tree.root)).toEqual(['Worked for 5s']) + }) + + it('holds no interval once the turn has settled', () => { + render({ startedAt: Date.now(), thinking: false, workedSeconds: 5 }) + expect(vi.getTimerCount()).toBe(0) + }) + + it('announces the live row to assistive tech', () => { + const tree = render({ startedAt: Date.now(), thinking: true }) + const row = tree.root.findByType('View' as never) + expect(row.props.accessibilityLiveRegion).toBe('polite') + expect(row.props.accessibilityLabel).toBe('Agent is responding') + }) +}) diff --git a/mobile/src/session/MobileNativeChatTurnStatus.tsx b/mobile/src/session/MobileNativeChatTurnStatus.tsx new file mode 100644 index 00000000000..4ce73cdcd38 --- /dev/null +++ b/mobile/src/session/MobileNativeChatTurnStatus.tsx @@ -0,0 +1,117 @@ +import { useEffect, useRef, useState } from 'react' +import { Animated, Pressable, StyleSheet, Text, View } from 'react-native' +import { ChevronRight } from 'lucide-react-native' +import { + formatNativeChatTurnStatusLabel, + NATIVE_CHAT_TURN_STATUS_COPY, + nativeChatElapsedSeconds +} from '../../../src/shared/native-chat-turn-status' +import { colors, spacing, typography } from '../theme/mobile-theme' + +/** Seconds tick only while a turn is actually counting, so a settled transcript + * holds no timers. */ +function useElapsedSeconds(startedAt: number | null, counting: boolean): number { + // Preserves the pre-stamp epoch for the frame before the turn's startedAt lands. + const [mountedAt] = useState(() => Date.now()) + const [now, setNow] = useState(() => Date.now()) + useEffect(() => { + if (!counting) { + return + } + setNow(Date.now()) + const timer = setInterval(() => setNow(Date.now()), 1_000) + return () => clearInterval(timer) + }, [counting]) + return counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 +} + +/** The per-turn status row — "Thinking", then "Working for 12s" while the turn + * runs, settling to a tappable "Worked for 3m 4s" that discloses the turn's + * tool activity. Desktop parity: `NativeChatWorkingStatus`. */ +export function MobileNativeChatTurnStatus({ + startedAt, + thinking, + workedSeconds, + expanded = false, + onToggleExpanded +}: { + startedAt: number | null + thinking: boolean + workedSeconds?: number | null + expanded?: boolean + onToggleExpanded?: () => void +}): React.JSX.Element { + const counting = !thinking && workedSeconds == null + const elapsedSeconds = useElapsedSeconds(startedAt, counting) + const label = formatNativeChatTurnStatusLabel({ thinking, workedSeconds, elapsedSeconds }) + + const pulse = useRef(new Animated.Value(1)).current + useEffect(() => { + if (!thinking) { + pulse.setValue(1) + return + } + const animation = Animated.loop( + Animated.sequence([ + Animated.timing(pulse, { toValue: 0.45, duration: 700, useNativeDriver: true }), + Animated.timing(pulse, { toValue: 1, duration: 700, useNativeDriver: true }) + ]) + ) + animation.start() + return () => animation.stop() + }, [pulse, thinking]) + + const rowStyle = [styles.row, thinking ? null : styles.rowSettled] + + if (workedSeconds != null && onToggleExpanded) { + return ( + [...rowStyle, pressed && styles.pressed]} + onPress={onToggleExpanded} + hitSlop={6} + accessibilityRole="button" + accessibilityState={{ expanded }} + accessibilityLabel={NATIVE_CHAT_TURN_STATUS_COPY.toggleDetails} + > + {label} + + + + + ) + } + + return ( + + {label} + + ) +} + +const styles = StyleSheet.create({ + row: { + flexDirection: 'row', + alignItems: 'center', + gap: spacing.xs, + minHeight: 28, + paddingHorizontal: spacing.md + }, + rowSettled: { + borderBottomWidth: StyleSheet.hairlineWidth, + borderBottomColor: colors.borderSubtle + }, + pressed: { + opacity: 0.6 + }, + label: { + color: colors.textMuted, + fontSize: typography.bodySize + }, + caretOpen: { + transform: [{ rotate: '90deg' }] + } +}) diff --git a/mobile/src/session/MobileNativeChatView.test.ts b/mobile/src/session/MobileNativeChatView.test.ts index 9d171e3ac8a..d101c3f0ef6 100644 --- a/mobile/src/session/MobileNativeChatView.test.ts +++ b/mobile/src/session/MobileNativeChatView.test.ts @@ -72,6 +72,9 @@ type Overrides = { inputLockReason?: 'disconnected' | 'waiting' | null onSend?: (text: string) => Promise pending?: Parameters[0]['pending'] + structuredActivityUi?: boolean + agentWorking?: boolean + sendSurfaceId?: string } function assistantTurn(id: string, text: string): NativeChatMessage { @@ -243,4 +246,111 @@ describe('MobileNativeChatView', () => { vi.useRealTimers() } }) + + describe('structured turn status wiring', () => { + const userTurn = (id: string, text: string): NativeChatMessage => ({ + id, + role: 'user', + blocks: [{ type: 'text', text }], + timestamp: 0, + source: 'transcript' + }) + + function rowProps(id: string): Record { + return (renderedRow(id) as { props: Record }).props + } + + function workingIndicators(): ReactTestInstance[] { + return renderer!.root.findAll((node) => node.type === 'WorkingIndicator') + } + + it('gives the live user turn a status row and drops the three-dot indicator', async () => { + const folded = [userTurn('u1', 'go')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + const props = rowProps('u1') + expect(props.structuredActivityUi).toBe(true) + expect(props.turnStatus).toMatchObject({ thinking: true, workedSeconds: null }) + expect(props.activeTurnIsWorking).toBe(true) + expect(workingIndicators()).toHaveLength(0) + }) + + it('keeps the bridge lane on the three-dot indicator with no turn status', async () => { + const folded = [userTurn('u1', 'go')] + await render({ messages: folded, folded, agentWorking: true }) + const props = rowProps('u1') + expect(props.structuredActivityUi).toBe(false) + expect(props.turnStatus).toBeNull() + expect(props.activeTurnIsWorking).toBe(false) + expect(workingIndicators()).toHaveLength(1) + }) + + it('settles the finished turn to a tappable duration', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + expect(rowProps('u1').turnStatus).toMatchObject({ thinking: false, workedSeconds: null }) + await update({ messages: folded, folded, structuredActivityUi: true, agentWorking: false }) + const settled = rowProps('u1') + expect(settled.turnStatus).toMatchObject({ thinking: false }) + expect((settled.turnStatus as { workedSeconds: number | null }).workedSeconds).toBeTypeOf( + 'number' + ) + expect(settled.onToggleTurn).toBeTypeOf('function') + expect(settled.activeTurnIsWorking).toBe(false) + }) + + it('hangs no status row on an assistant row', async () => { + const folded = [userTurn('u1', 'go'), assistantTurn('a1', 'done')] + await render({ messages: folded, folded, structuredActivityUi: true, agentWorking: true }) + expect(rowProps('a1').turnStatus).toBeNull() + // The assistant row still belongs to the live turn, so its tool row stays visible. + expect(rowProps('a1').activeTurnIsWorking).toBe(true) + }) + + it('does not carry a running turn clock across chat surfaces', async () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const firstTab = [userTurn('u1', 'first')] + await render({ + messages: firstTab, + folded: firstTab, + structuredActivityUi: true, + agentWorking: true, + sendSurfaceId: 'host\0worktree\0tab-a' + }) + expect(rowProps('u1').turnStatus).toMatchObject({ startedAt: 1_000 }) + + vi.setSystemTime(12_000) + const secondTab = [userTurn('u2', 'second')] + await update({ + messages: secondTab, + folded: secondTab, + structuredActivityUi: true, + agentWorking: true, + sendSurfaceId: 'host\0worktree\0tab-b' + }) + + expect(rowProps('u2').turnStatus).toMatchObject({ startedAt: 12_000 }) + } finally { + vi.useRealTimers() + } + }) + + it('does not treat pre-user history as part of the live turn', async () => { + const history = [ + assistantTurn('a0', 'before the first prompt'), + userTurn('u1', 'go'), + assistantTurn('a1', 'working') + ] + await render({ + messages: history, + folded: history, + structuredActivityUi: true, + agentWorking: true + }) + + expect(rowProps('a0').activeTurnIsWorking).toBe(false) + expect(rowProps('a1').activeTurnIsWorking).toBe(true) + }) + }) }) diff --git a/mobile/src/session/MobileNativeChatView.tsx b/mobile/src/session/MobileNativeChatView.tsx index 4db4e437b43..59c024435b6 100644 --- a/mobile/src/session/MobileNativeChatView.tsx +++ b/mobile/src/session/MobileNativeChatView.tsx @@ -21,16 +21,16 @@ import { type MobileNativeChatPendingItem } from './mobile-native-chat-render-data' import { useMobileNativeChatPinchGesture } from './use-mobile-native-chat-pinch-gesture' +import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' +import { MobileNativeChatTurnStatus } from './MobileNativeChatTurnStatus' import { MobileAgentWorkingIndicator } from './MobileAgentWorkingIndicator' import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment' import { MobileNativeChatComposer } from './MobileNativeChatComposer' +import { MobileNativeChatPromptCard } from './MobileNativeChatPromptCard' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import type { MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatSessionOptionPickersProps } from './MobileNativeChatSessionOptionPickers' import { MobileNativeChatMessage } from './MobileNativeChatMessage' -import { MobileNativeChatAsk } from './MobileNativeChatAsk' -import { MobileNativeChatPermission } from './MobileNativeChatPermission' -import type { MobileChatPermission } from './mobile-native-chat-permission' -import { MobileNativeChatQuestion } from './MobileNativeChatQuestion' -import { mobileChatQuestionKey, type MobileChatQuestion } from './mobile-native-chat-question' import type { MobileNativeChatStatus } from './use-mobile-native-chat-session' const INPUT_LOCK_SETTLE_MS = 600 @@ -49,6 +49,9 @@ type Props = { /** Resolved agent for this chat; names the empty-state copy (desktop parity). */ agent?: string | null agentWorking?: boolean + /** Structured lane: per-turn "Working for N" status plus live tool progress, + * replacing the bridge lane's static three-dot working row (desktop parity). */ + structuredActivityUi?: boolean /** Interrupt the agent mid-turn (shown as a Stop button on the working bar). */ onStop?: () => void /** Live partial assistant text to show as an in-progress bubble, already gated @@ -126,6 +129,7 @@ export function MobileNativeChatView({ error, agent, agentWorking, + structuredActivityUi = false, onStop, streaming, hasMore, @@ -252,6 +256,15 @@ export function MobileNativeChatView({ listRef.current?.scrollToIndex({ index, viewPosition: 0, animated: true }) }, []) + // Per-turn "Thinking / Working for N / Worked for N" rows. The structured lane + // owns them; the bridge lane keeps its three-dot indicator. + const turns = useMobileNativeChatTurnDisclosure({ + messages: data, + enabled: structuredActivityUi, + isWorking: agentWorking === true, + scopeKey: sendSurfaceId + }) + const renderItem = useCallback( ({ item, index }: { item: NativeChatMessage; index: number }) => ( ), - [toolsExpanded, fontScale, onScrollToMessage, onOpenFile] + [toolsExpanded, fontScale, onScrollToMessage, onOpenFile, structuredActivityUi, turns] ) const emptyState = mobileNativeChatEmptyState(status, agent ?? null, error) @@ -337,6 +353,15 @@ export function MobileNativeChatView({ ) : null } + ListFooterComponent={ + turns.activeTurnIsUnanchored && turns.active ? ( + + ) : null + } ListEmptyComponent={ emptyState ? ( @@ -360,47 +385,22 @@ export function MobileNativeChatView({ ) : null} )} - {/* Pending agent prompt: a structured AskUserQuestion wins, then a - heuristic permission, then a heuristic question. The controller owns - dismissal (it must survive this subtree unmounting on a view toggle); - `ask` arrives already nulled while dismissed. */} - {ask ? ( - { - const accepted = (await onAnswerAsk?.(ask, selections)) ?? false - if (accepted) { - onDismissAsk?.() - } - return accepted - }} - onCancel={async () => { - const accepted = (await onCancelAsk?.()) ?? false - if (accepted) { - onDismissAsk?.() - } - return accepted - }} - /> - ) : permission ? ( - (await onRespondPermission?.(send)) ?? false} - /> - ) : question ? ( - (await onAnswerQuestion?.(text)) ?? false} - /> - ) : null} + {/* Chrome row above the composer: the working indicator and the global tool-calls expand/collapse toggle on the left, Stop in the far corner. */} - {agentWorking ? : null} + {agentWorking && !structuredActivityUi ? : null} [styles.chromeToggle, pressed && styles.pressed]} onPress={() => setToolsExpanded((v) => !v)} diff --git a/mobile/src/session/mobile-native-chat-controller-contract.ts b/mobile/src/session/mobile-native-chat-controller-contract.ts index 890a3a1562e..53187e0d6db 100644 --- a/mobile/src/session/mobile-native-chat-controller-contract.ts +++ b/mobile/src/session/mobile-native-chat-controller-contract.ts @@ -25,6 +25,8 @@ export type MobileNativeChatController = { chatPending: MobileNativeChatPendingMessage[] chatImagePreviewsByMessageId: Record nativeChatSession: ReturnType + /** Structured lane: drives the per-turn status row and live tool progress. */ + nativeChatStructured: boolean nativeChatAgentWorking: boolean nativeChatStreamingText?: string /** Agent mid-turn, regardless of whether chat is the visible view. */ diff --git a/mobile/src/session/mobile-native-chat-message-styles.ts b/mobile/src/session/mobile-native-chat-message-styles.ts index ad7cf4b4009..7ae1128445a 100644 --- a/mobile/src/session/mobile-native-chat-message-styles.ts +++ b/mobile/src/session/mobile-native-chat-message-styles.ts @@ -80,6 +80,18 @@ export const styles = StyleSheet.create({ fontFamily: typography.monoFamily, fontSize: MONO_SIZE }, + toolRunActive: { + flex: 1, + flexDirection: 'row', + alignItems: 'center', + gap: spacing.sm, + paddingVertical: 3 + }, + toolRunActiveLabel: { + flex: 1, + color: colors.textSecondary, + fontSize: typography.bodySize + }, toolRunBody: { paddingLeft: spacing.sm, borderLeftWidth: 2, diff --git a/mobile/src/session/mobile-structured-agent-session-launch.test.ts b/mobile/src/session/mobile-structured-agent-session-launch.test.ts index 54f9b5cbe88..f575d5ac6d4 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.test.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.test.ts @@ -139,4 +139,76 @@ describe('mobile structured Codex launch', () => { kind: 'unknown' }) }) + + it.each(['structured_agent_session_unsupported', 'method_not_found'])( + 'treats a top-level %s as a definitive refusal', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'structured create unavailable' } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'failed', + message: 'structured create unavailable' + }) + } + ) + + it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])( + 'keeps a top-level %s outcome unknown', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'create outcome ambiguous' } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unknown', + message: 'create outcome ambiguous' + }) + } + ) + + it('treats an envelope unsupported refusal as definitive', async () => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { + code: 'structured_agent_session_unsupported', + message: 'structured create unavailable' + } + } + } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'failed', + message: 'structured create unavailable' + }) + }) + + it.each(['agent_session_operation_unknown', 'agent_session_ownership_unknown', 'future_code'])( + 'keeps an envelope %s refusal unknown', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { code, message: 'create outcome ambiguous' } + } + } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unknown', + message: 'create outcome ambiguous' + }) + } + ) }) diff --git a/mobile/src/session/mobile-structured-agent-session-launch.ts b/mobile/src/session/mobile-structured-agent-session-launch.ts index ecad0410dfd..b7eb8289e84 100644 --- a/mobile/src/session/mobile-structured-agent-session-launch.ts +++ b/mobile/src/session/mobile-structured-agent-session-launch.ts @@ -2,6 +2,7 @@ import type { AgentSessionAttachResult, AgentSessionMutationResult } from '../../../src/shared/agent-session-wire' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../src/shared/agent-session-definitive-refusal' import { structuredAgentSessionPayloadFingerprint } from '../../../src/shared/structured-agent-session-mutation' import type { RpcClient } from '../transport/rpc-client' import { structuredSessionOperationId } from './mobile-structured-agent-session-rpc' @@ -66,6 +67,13 @@ function unknownCreateResult(error: unknown): MobileStructuredCodexLaunchResult } } +function classifyCreateRefusal(code: string, message: string): MobileStructuredCodexLaunchResult { + if (!isDefinitiveAgentSessionCreateRefusal(code)) { + return unknownCreateResult(new Error(message)) + } + return { kind: 'failed', message: message || 'Could not open Codex chat.' } +} + export async function createMobileStructuredCodexSession( client: RpcClient, worktreeId: string @@ -125,10 +133,7 @@ export async function createMobileStructuredCodexSession( ) { return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) } - if (response.error.code === 'agent_session_operation_unknown') { - return unknownCreateResult(new Error(response.error.message)) - } - return { kind: 'failed', message: response.error.message || 'Could not open Codex chat.' } + return classifyCreateRefusal(response.error.code, response.error.message) } const result = response.result as AgentSessionMutationResult if (!result || typeof result !== 'object' || typeof result.ok !== 'boolean') { @@ -142,10 +147,7 @@ export async function createMobileStructuredCodexSession( ) { return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) } - if (result.refusal.code === 'agent_session_operation_unknown') { - return unknownCreateResult(new Error(result.refusal.message)) - } - return { kind: 'failed', message: result.refusal.message || 'Could not open Codex chat.' } + return classifyCreateRefusal(result.refusal.code, result.refusal.message) } if ( !result.value || diff --git a/mobile/src/session/use-mobile-native-chat-controller.ts b/mobile/src/session/use-mobile-native-chat-controller.ts index bf944398e87..a946956f8d6 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.ts @@ -10,9 +10,8 @@ import { useMobileNativeChatDrafts } from './use-mobile-native-chat-drafts' import { useMobileNativeChatFileSearch } from './use-mobile-native-chat-file-search' import { useMobileNativeChatMessageSend } from './use-mobile-native-chat-message-send' import { mobileNativeChatStreamPreview } from './mobile-native-chat-streaming-gate' -import { useMobileNativeChatSession } from './use-mobile-native-chat-session' import { useMobileNativeChatSessionOptionController } from './use-mobile-native-chat-session-option-controller' -import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' +import { useMobileNativeChatSessionLane } from './use-mobile-native-chat-session-lane' import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge' import { useMobileNativeChatPrompts } from './use-mobile-native-chat-prompts' import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' @@ -82,27 +81,19 @@ export function useMobileNativeChatController(args: { nativeChatTranscriptIsLocalReadable }) - const legacyNativeChatSession = useMobileNativeChatSession({ - client, - sourceIdentity, - agent: activeChatStructured ? null : (activeChatResolution?.agent ?? null), - sessionId: activeChatStructured ? null : activeChatSessionId, - transcriptPath: activeChatStructured ? null : (activeChatResolution?.transcriptPath ?? null) - }) - const structuredNativeChat = useMobileStructuredAgentSession({ - client, - sessionId: activeChatStructured ? activeChatSessionId : null, - sourceIdentity, - enabled: showNativeChat, - // Holds are connection-scoped; dropping this on transport loss lets the hook - // reacquire the provider without clearing the cached transcript. - connected: connState === 'connected', - agent: activeChatStructured ? activeChatAgent : null, - onSendError - }) - const nativeChatSession = activeChatStructured - ? structuredNativeChat.session - : legacyNativeChatSession + const { structuredSession: structuredNativeChat, session: nativeChatSession } = + useMobileNativeChatSessionLane({ + client, + structured: activeChatStructured, + agent: activeChatAgent, + resolvedAgent: activeChatResolution?.agent ?? null, + transcriptPath: activeChatResolution?.transcriptPath ?? null, + sessionId: activeChatSessionId, + sourceIdentity, + enabled: showNativeChat, + connState, + onSendError + }) const { composerText: chatComposerText, setComposerText: setChatComposerText, @@ -303,6 +294,8 @@ export function useMobileNativeChatController(args: { chatPending, chatImagePreviewsByMessageId, nativeChatSession, + /** Structured lane: drives the per-turn status row and live tool progress. */ + nativeChatStructured: activeChatStructured, nativeChatAgentWorking, nativeChatStreamingText, nativeChatStreamLive, diff --git a/mobile/src/session/use-mobile-native-chat-session-lane.ts b/mobile/src/session/use-mobile-native-chat-session-lane.ts new file mode 100644 index 00000000000..fc465de902b --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-session-lane.ts @@ -0,0 +1,59 @@ +import type { RpcClient } from '../transport/rpc-client' +import type { ConnectionState } from '../transport/types' +import { useMobileNativeChatSession } from './use-mobile-native-chat-session' +import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' + +/** Mounts both transcript sources and hands back the one this tab's lane owns. + * Both hooks always run (hook order is fixed); the inactive lane is starved of + * its identity inputs rather than unmounted, so a lane flip keeps its cache. */ +export function useMobileNativeChatSessionLane({ + client, + structured, + agent, + resolvedAgent, + transcriptPath, + sessionId, + sourceIdentity, + enabled, + connState, + onSendError +}: { + client: RpcClient | null + structured: boolean + /** Agent id for the structured provider session. */ + agent: string | null + /** Agent resolved from the terminal, for the bridge transcript reader. */ + resolvedAgent: string | null + transcriptPath: string | null + sessionId: string | null + sourceIdentity: Parameters[0]['sourceIdentity'] + enabled: boolean + connState: ConnectionState + onSendError: (message: string) => void +}): { + structuredSession: ReturnType + session: ReturnType +} { + const bridgeSession = useMobileNativeChatSession({ + client, + sourceIdentity, + agent: structured ? null : resolvedAgent, + sessionId: structured ? null : sessionId, + transcriptPath: structured ? null : transcriptPath + }) + const structuredSession = useMobileStructuredAgentSession({ + client, + sessionId: structured ? sessionId : null, + sourceIdentity, + enabled, + // Holds are connection-scoped; dropping this on transport loss lets the hook + // reacquire the provider without clearing the cached transcript. + connected: connState === 'connected', + agent: structured ? agent : null, + onSendError + }) + return { + structuredSession, + session: structured ? structuredSession.session : bridgeSession + } +} diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx new file mode 100644 index 00000000000..8467684ce16 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.test.tsx @@ -0,0 +1,153 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { useMobileNativeChatTurnDisclosure } from './use-mobile-native-chat-turn-disclosure' + +function userMessage(id: string): NativeChatMessage { + return { + id, + role: 'user', + blocks: [{ type: 'text', text: id }], + timestamp: null, + source: 'transcript' + } +} + +function Harness({ + messages, + enabled, + isWorking = true, + scopeKey = 'host\0worktree\0tab-a' +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking?: boolean + scopeKey?: string +}): React.JSX.Element { + const disclosure = useMobileNativeChatTurnDisclosure({ + messages, + enabled, + isWorking, + scopeKey + }) + return createElement('result', { disclosure }) +} + +describe('useMobileNativeChatTurnDisclosure', () => { + let renderer: ReactTestRenderer | null = null + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + }) + + it('does not scan bridge-lane transcripts', () => { + const messages: NativeChatMessage[] = [ + { + id: 'u1', + role: 'user', + blocks: [{ type: 'text', text: 'go' }], + timestamp: null, + source: 'transcript' + } + ] + const findLastIndex = vi.spyOn(messages, 'findLastIndex') + const slice = vi.spyOn(messages, 'slice') + const filter = vi.spyOn(messages, 'filter') + const map = vi.spyOn(messages, 'map') + + act(() => { + renderer = create(createElement(Harness, { messages, enabled: false })) + }) + + expect(findLastIndex).not.toHaveBeenCalled() + expect(slice).not.toHaveBeenCalled() + expect(filter).not.toHaveBeenCalled() + expect(map).not.toHaveBeenCalled() + }) + + it('keeps a settled turn handler stable for NUL-delimited scope keys', () => { + vi.useFakeTimers() + try { + vi.setSystemTime(1_000) + const messages: NativeChatMessage[] = [ + { + id: 'u1', + role: 'user', + blocks: [{ type: 'text', text: 'go' }], + timestamp: null, + source: 'transcript' + } + ] + act(() => { + renderer = create(createElement(Harness, { messages, enabled: true })) + }) + vi.setSystemTime(6_000) + act(() => { + renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false })) + }) + const first = renderer!.root.findByType('result').props.disclosure.resolveRow(0, messages[0]) + + const refreshed = [...messages] + act(() => { + renderer?.update( + createElement(Harness, { messages: refreshed, enabled: true, isWorking: false }) + ) + }) + const second = renderer!.root + .findByType('result') + .props.disclosure.resolveRow(0, refreshed[0]) + + // The row carries the key; the handler itself lives on the hook and stays + // stable for the scope, so a re-render never disturbs a row's memo. + expect(first.turnKey).toBe('u1') + expect(second.turnKey).toBe('u1') + const firstHandler = renderer!.root.findByType('result').props.disclosure.onToggleTurn + expect(firstHandler).toBeTypeOf('function') + act(() => { + renderer?.update( + createElement(Harness, { messages: [...refreshed], enabled: true, isWorking: false }) + ) + }) + expect(renderer!.root.findByType('result').props.disclosure.onToggleTurn).toBe(firstHandler) + } finally { + vi.useRealTimers() + } + }) + + it('keeps at most the latest 128 turns expanded', () => { + vi.useFakeTimers() + try { + let messages: NativeChatMessage[] = [] + for (let index = 0; index < 129; index++) { + messages = messages.concat(userMessage(`u${index}`)) + vi.setSystemTime(index * 2_000) + act(() => { + if (renderer) { + renderer.update(createElement(Harness, { messages, enabled: true })) + } else { + renderer = create(createElement(Harness, { messages, enabled: true })) + } + }) + vi.setSystemTime(index * 2_000 + 1_000) + act(() => { + renderer?.update(createElement(Harness, { messages, enabled: true, isWorking: false })) + }) + const disclosureNow = renderer!.root.findByType('result').props.disclosure + const row = disclosureNow.resolveRow(index, messages[index]) + act(() => disclosureNow.onToggleTurn(row.turnKey)) + } + + const disclosure = renderer!.root.findByType('result').props.disclosure + const expanded = messages.filter( + (message, index) => disclosure.resolveRow(index, message).turnExpanded + ) + expect(expanded).toHaveLength(128) + expect(disclosure.resolveRow(0, messages[0]).turnExpanded).toBe(false) + expect(disclosure.resolveRow(128, messages[128]).turnExpanded).toBe(true) + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts new file mode 100644 index 00000000000..46b58f29cba --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-disclosure.ts @@ -0,0 +1,126 @@ +import { useCallback, useMemo, useState } from 'react' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { + MOBILE_UNANCHORED_TURN_KEY, + useMobileNativeChatTurnStatus, + type NativeChatTurnStatus +} from './use-mobile-native-chat-turn-status' + +const EMPTY_TURN_IDS: ReadonlySet = new Set() +const EMPTY_TURN_KEYS: readonly undefined[] = [] +const MAX_EXPANDED_TURNS = 128 + +export type MobileNativeChatTurnRow = { + turnStatus: NativeChatTurnStatus | null + turnExpanded: boolean + /** Set only on a settled turn — the one row that has activity to disclose. */ + turnKey?: string + activeTurnIsWorking: boolean +} + +/** Owns the transcript's per-turn status rows and their disclosure state, and + * resolves what one list row needs. Bridge-lane chats pass `enabled: false` and + * keep their single three-dot working indicator instead. */ +export function useMobileNativeChatTurnDisclosure({ + messages, + enabled, + isWorking, + scopeKey +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking: boolean + /** Host/worktree/tab identity for timing and disclosure isolation. */ + scopeKey: string +}): { + active: NativeChatTurnStatus | null + /** True when the live turn has no user message to hang its status row under. */ + activeTurnIsUnanchored: boolean + onToggleTurn: (turnKey: string) => void + resolveRow: (index: number, message: NativeChatMessage) => MobileNativeChatTurnRow +} { + const turnStatuses = useMobileNativeChatTurnStatus({ + messages, + enabled, + isWorking, + scopeKey + }) + const [expandedTurns, setExpandedTurns] = useState<{ + scopeKey: string + turnIds: ReadonlySet + }>(() => ({ scopeKey, turnIds: new Set() })) + const expandedTurnIds = + expandedTurns.scopeKey === scopeKey ? expandedTurns.turnIds : EMPTY_TURN_IDS + const toggleExpandedTurn = useCallback( + (turnKey: string) => { + setExpandedTurns((current) => { + const next = new Set(current.scopeKey === scopeKey ? current.turnIds : []) + if (!next.delete(turnKey)) { + if (next.size >= MAX_EXPANDED_TURNS) { + const oldest = next.values().next().value + if (oldest) { + next.delete(oldest) + } + } + next.add(turnKey) + } + return { scopeKey, turnIds: next } + }) + }, + [scopeKey] + ) + // Resolve each row's turn boundary once — a findLast per row is quadratic on a + // long transcript. + const turnKeys = useMemo(() => { + if (!enabled) { + return EMPTY_TURN_KEYS + } + let turnKey: string | undefined + return messages.map((message) => { + if (message.role === 'user') { + turnKey = message.id + } + return turnKey + }) + }, [enabled, messages]) + + const { active, activeTurnKey, completedByTurn } = turnStatuses + const resolveRow = useCallback( + (index: number, message: NativeChatMessage): MobileNativeChatTurnRow => { + const turnKey = turnKeys[index] + const turnStatus = + !enabled || message.role !== 'user' + ? null + : turnKey === activeTurnKey + ? active + : turnKey + ? (completedByTurn[turnKey] ?? null) + : null + return { + turnStatus, + turnExpanded: turnKey ? expandedTurnIds.has(turnKey) : false, + // Why: the key travels and the row calls one stable handler with it. A + // closure per row would be a new identity every render of a streaming + // transcript, defeating the row's memo; caching one per turn would mean + // writing a ref during render, which react-freeze can discard. + turnKey: turnKey && turnStatus?.workedSeconds != null ? turnKey : undefined, + // With no user boundary at all, the session's working state stays authoritative. + activeTurnIsWorking: + enabled && + isWorking && + (turnKey === activeTurnKey || + (turnKey === undefined && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY)) + } + }, + [turnKeys, enabled, activeTurnKey, active, completedByTurn, expandedTurnIds, isWorking] + ) + + return { + active, + /** Stable for a given chat scope, so it never disturbs a row's memo. */ + onToggleTurn: toggleExpandedTurn, + activeTurnIsUnanchored: + enabled && active != null && activeTurnKey === MOBILE_UNANCHORED_TURN_KEY, + resolveRow + } +} diff --git a/mobile/src/session/use-mobile-native-chat-turn-status.ts b/mobile/src/session/use-mobile-native-chat-turn-status.ts new file mode 100644 index 00000000000..13afbe70c09 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-turn-status.ts @@ -0,0 +1,105 @@ +import { useEffect, useMemo, useRef, useState } from 'react' +import type { NativeChatMessage } from '../../../src/shared/native-chat-types' +import { + nativeChatTurnHasResponse, + reduceNativeChatTurnTiming, + selectNativeChatTurnStatuses, + type NativeChatTurnStatus, + type NativeChatTurnTimingByTurn +} from '../../../src/shared/native-chat-turn-status' + +export type { NativeChatTurnStatus } + +export const MOBILE_UNANCHORED_TURN_KEY = '__unanchored__' +const EMPTY_TURN_TIMING_BY_TURN: NativeChatTurnTimingByTurn = Object.freeze({}) + +type ScopedTurnTiming = { + scopeKey: string + timingByTurn: NativeChatTurnTimingByTurn +} + +/** Per-turn "Thinking / Working for N / Worked for N" timing, on the same shared + * state machine the desktop renderer uses so the two surfaces stamp turns alike. */ +export function useMobileNativeChatTurnStatus({ + messages, + enabled, + isWorking, + workingStartedAt, + scopeKey +}: { + messages: readonly NativeChatMessage[] + enabled: boolean + isWorking: boolean + workingStartedAt?: number | null + /** Host/worktree/tab identity. Timings never carry across chat surfaces. */ + scopeKey: string +}): { + active: NativeChatTurnStatus | null + completedByTurn: Readonly> + activeTurnKey: string +} { + const latestUserIndex = enabled + ? messages.findLastIndex((message) => message.role === 'user') + : -1 + const hasCurrentTurnResponse = enabled && nativeChatTurnHasResponse(messages, latestUserIndex) + const latestUserId = latestUserIndex !== -1 ? (messages[latestUserIndex]?.id ?? null) : null + const activeTurnKey = latestUserId ?? MOBILE_UNANCHORED_TURN_KEY + const [scopedTiming, setScopedTiming] = useState(() => ({ + scopeKey, + timingByTurn: {} + })) + // Do not expose the previous surface's state during the render before the + // timing effect adopts the new scope, or scan it while this UI is disabled. + const timingByTurn = + enabled && scopedTiming.scopeKey === scopeKey + ? scopedTiming.timingByTurn + : EMPTY_TURN_TIMING_BY_TURN + // An accepted send renders as `pending-N` until the transcript echo lands under + // its real id. That is one turn under two keys, so the clock must survive the swap. + const previousActiveTurn = useRef<{ scopeKey: string; turnKey: string } | null>(null) + + useEffect(() => { + if (!enabled) { + return + } + const validTurnKeys = new Set( + messages.filter((message) => message.role === 'user').map((message) => message.id) + ) + const previousActiveTurnKey = + previousActiveTurn.current?.scopeKey === scopeKey + ? previousActiveTurn.current.turnKey + : undefined + previousActiveTurn.current = { scopeKey, turnKey: activeTurnKey } + setScopedTiming((current) => { + const currentTiming = + current.scopeKey === scopeKey ? current.timingByTurn : EMPTY_TURN_TIMING_BY_TURN + const nextTiming = reduceNativeChatTurnTiming(currentTiming, { + activeTurnKey, + previousActiveTurnKey, + validTurnKeys, + isWorking, + workingStartedAt, + now: Date.now() + }) + return current.scopeKey === scopeKey && nextTiming === currentTiming + ? current + : { scopeKey, timingByTurn: nextTiming } + }) + }, [activeTurnKey, enabled, isWorking, messages, scopeKey, workingStartedAt]) + + // Why: the selection rebuilds its status objects on every call, and a streaming + // turn re-renders ~20x/s. Without this, every settled turn's row gets fresh + // props each tick and the memoized message rows all re-render. + const turnIsWorking = enabled && isWorking + const statuses = useMemo( + () => + selectNativeChatTurnStatuses(timingByTurn, { + activeTurnKey, + isWorking: turnIsWorking, + workingStartedAt, + hasCurrentTurnResponse + }), + [timingByTurn, activeTurnKey, turnIsWorking, workingStartedAt, hasCurrentTurnResponse] + ) + return { ...statuses, activeTurnKey } +} diff --git a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts index c0a4e8368c5..e46bf087f38 100644 --- a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts +++ b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts @@ -137,14 +137,17 @@ describe('mobile + Codex tab creation routing', () => { expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('terminal-tab-1') }) - it('falls back to a terminal when structured creation is refused', async () => { + it('falls back to a terminal when structured creation is definitively refused', async () => { const client = clientReturning( { ok: true, result: { supported: true } }, { ok: true, result: { ok: false, - refusal: { code: 'agent_session_ownership_unknown', message: 'provider unavailable' } + refusal: { + code: 'structured_agent_session_unsupported', + message: 'provider unavailable' + } } }, terminalCreateResponse() @@ -228,4 +231,34 @@ describe('mobile + Codex tab creation routing', () => { expect(scope.setCreateError).toHaveBeenCalledWith('still unknown') expect(scope.showToast).toHaveBeenCalledWith('still unknown', 1800) }) + + it.each(['agent_session_operation_unknown', 'runtime_error', 'future_unknown_code'])( + 'does not create a legacy sibling after a top-level %s response', + async (code) => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: false, error: { code, message: 'create outcome ambiguous' } } + ) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + const sendRequest = client.sendRequest as unknown as ReturnType + expect(sendRequest.mock.calls.map(([method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create' + ]) + expect(scope.setCreateError).toHaveBeenCalledWith('create outcome ambiguous') + expect(scope.showToast).toHaveBeenCalledWith('create outcome ambiguous', 1800) + } + ) }) diff --git a/package.json b/package.json index c519d7ea6b1..d1d930be708 100644 --- a/package.json +++ b/package.json @@ -141,6 +141,10 @@ "bench:main-thread-jank": "pnpm run ensure:electron-runtime && node tests/tools/benchmarks/main-thread-jank-bench.mjs", "bench:worktree-deletion": "node tests/tools/benchmarks/worktree-deletion-dev-bench.mjs", "bench:zustand-selector-fanout": "node config/scripts/zustand-selector-fanout-benchmark.mjs", + "bench:agent-inspection-cadence": "node config/scripts/agent-inspection-cadence-batching-benchmark.mjs", + "bench:renderer-quadratic-scans": "node config/scripts/renderer-quadratic-scan-benchmark.mjs", + "bench:session-write-hot-path": "node config/scripts/session-write-hot-path-benchmark.mjs", + "bench:terminal-partial-escape-tail": "node config/scripts/terminal-partial-escape-tail-benchmark.mjs", "bench:worktree-refresh-churn": "node --disable-warning=MODULE_TYPELESS_PACKAGE_JSON config/scripts/worktree-refresh-churn-benchmark.mjs", "bench:multi-workspace-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-multi-workspace-typing-bench.mjs", "bench:ai-vault-typing": "pnpm run ensure:electron-runtime && node config/scripts/run-ai-vault-typing-bench.mjs", diff --git a/src/main/browser/anti-detection-permission-status.test.ts b/src/main/browser/anti-detection-permission-status.test.ts deleted file mode 100644 index d686e32c975..00000000000 --- a/src/main/browser/anti-detection-permission-status.test.ts +++ /dev/null @@ -1,210 +0,0 @@ -import { runInNewContext } from 'node:vm' -import { describe, expect, it } from 'vitest' - -import { ANTI_DETECTION_SCRIPT } from './anti-detection' - -type PermissionQueryResult = EventTarget & { - state: string - onchange: EventListener | null - marker: string -} - -type PermissionStatusConstructor = { - new (): PermissionQueryResult - prototype: PermissionQueryResult -} - -type AntiDetectionContext = { - Notification: { - permission: string - requestPermission: (callback?: (permission: string) => void) => Promise - } - PermissionStatus: PermissionStatusConstructor - dispatchPermissionChange: (name: string) => void - navigator: { - permissions: { - query: (descriptor: { name: string }) => Promise - } - } -} - -function createContext(args: { - nativeNotificationPermission: string - requestedNotificationPermission: string - rejectedPermissions?: string[] -}): AntiDetectionContext & Record { - class PermissionStatus extends EventTarget { - #state = 'denied' - #onchange: EventListener | null = null - marker = 'real-status' - - get state(): string { - return this.#state - } - - get onchange(): EventListener | null { - return this.#onchange - } - - set onchange(listener: EventListener | null) { - if (this.#onchange) { - super.removeEventListener('change', this.#onchange) - } - this.#onchange = typeof listener === 'function' ? listener : null - if (this.#onchange) { - super.addEventListener('change', this.#onchange) - } - } - } - - const statuses = new Map() - const rejectedPermissions = new Set(args.rejectedPermissions) - - class Permissions { - query(descriptor: { name: string }): Promise { - if (rejectedPermissions.has(descriptor.name)) { - return Promise.reject(new Error('Unsupported permission')) - } - const status = new PermissionStatus() - const permissionStatuses = statuses.get(descriptor.name) ?? [] - permissionStatuses.push(status) - statuses.set(descriptor.name, permissionStatuses) - return Promise.resolve(status) - } - } - - const Notification = { - permission: args.nativeNotificationPermission, - requestPermission(callback?: (permission: string) => void): Promise { - callback?.(args.requestedNotificationPermission) - return Promise.resolve(args.requestedNotificationPermission) - } - } - Object.defineProperty(Notification, 'permission', { - configurable: true, - get: () => args.nativeNotificationPermission - }) - - return { - Date, - Event, - EventTarget, - Object, - Promise, - Set, - performance: { now: () => 0 }, - // Why: the script's Firefox gate reads navigator.userAgent, so a non-Firefox UA keeps these - // tests on the ordinary-page path where the PermissionStatus override applies. - window: { chrome: {} }, - navigator: { - userAgent: - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/150.0.0.0 Safari/537.36', - plugins: [], - languages: [], - permissions: new Permissions() - }, - Permissions, - PermissionStatus, - Notification, - dispatchPermissionChange(name: string): void { - for (const status of statuses.get(name) ?? []) { - status.dispatchEvent(new Event('change')) - } - } - } as AntiDetectionContext & Record -} - -describe('ANTI_DETECTION_SCRIPT — PermissionStatus', () => { - it('keeps an existing notification status current after permission changes', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'notifications' }) - - expect(context.Notification.permission).toBe('default') - expect(status.state).toBe('prompt') - - await expect(context.Notification.requestPermission()).resolves.toBe('granted') - - expect(context.Notification.permission).toBe('granted') - expect(status.state).toBe('granted') - }) - - it('preserves native PermissionStatus identity and methods', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'camera' }) - const expectedSource = Function.prototype.toString.call( - context.PermissionStatus.prototype.addEventListener - ) - - expect(status).toBeInstanceOf(context.PermissionStatus) - expect(status.state).toBe('prompt') - expect(status.constructor.name).toBe('PermissionStatus') - expect(status.marker).toBe('real-status') - expect(status.addEventListener.name).toBe('addEventListener') - expect(status.addEventListener).toBe(status.addEventListener) - expect(Function.prototype.toString.call(status.addEventListener)).toBe(expectedSource) - expect(expectedSource).toContain('addEventListener') - }) - - it('delivers change events through the returned status with the overridden state', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'notifications' }) - const events: { receiver: EventTarget; target: EventTarget | null; state: string }[] = [] - const recordEvent = function (this: EventTarget, event: Event): void { - events.push({ - receiver: this, - target: event.target, - state: (event.target as PermissionQueryResult).state - }) - } - - status.addEventListener('change', recordEvent) - expect(() => { - status.onchange = function (this: EventTarget, event): void { - recordEvent.call(this, event) - } - }).not.toThrow() - - await context.Notification.requestPermission() - context.dispatchPermissionChange('notifications') - - expect(events).toHaveLength(2) - expect(events).toEqual([ - { receiver: status, target: status, state: 'granted' }, - { receiver: status, target: status, state: 'granted' } - ]) - }) - - // Why: 'camera' rather than 'storage-access' — #14685 narrowed the intercepted set to - // camera/microphone, so a name outside it falls through to the real query and never reaches - // the fallback at all. - it('uses a non-enumerable EventTarget fallback when the native query rejects', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted', - rejectedPermissions: ['camera'] - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - const status = await context.navigator.permissions.query({ name: 'camera' }) - - expect(status).toBeInstanceOf(EventTarget) - expect(status).not.toBeInstanceOf(context.PermissionStatus) - expect(status.state).toBe('prompt') - expect(Object.keys(status)).toEqual([]) - }) -}) diff --git a/src/main/browser/anti-detection.test.ts b/src/main/browser/anti-detection.test.ts deleted file mode 100644 index ddb904cb436..00000000000 --- a/src/main/browser/anti-detection.test.ts +++ /dev/null @@ -1,157 +0,0 @@ -import { runInNewContext } from 'node:vm' -import { describe, expect, it } from 'vitest' - -import { ANTI_DETECTION_SCRIPT } from './anti-detection' -import { googleAuthUserAgent } from './browser-google-auth-ua' - -type PermissionQueryResult = { - state: string - onchange: null -} - -type AntiDetectionContext = { - Notification: { - permission: string - requestPermission: (callback?: (permission: string) => void) => Promise - } - navigator: { - userAgent: string - permissions: { - query: (descriptor: { name: string }) => Promise - } - } - window: { - chrome?: { - runtime?: unknown - csi?: () => unknown - loadTimes?: () => unknown - } - } -} - -function createContext(args: { - nativeNotificationPermission: string - requestedNotificationPermission: string - userAgent?: string -}): AntiDetectionContext & Record { - class Permissions { - query(): Promise { - return Promise.resolve({ state: 'denied', onchange: null }) - } - } - - const Notification = { - permission: args.nativeNotificationPermission, - requestPermission(callback?: (permission: string) => void): Promise { - callback?.(args.requestedNotificationPermission) - return Promise.resolve(args.requestedNotificationPermission) - } - } - Object.defineProperty(Notification, 'permission', { - configurable: true, - get: () => args.nativeNotificationPermission - }) - - return { - Date, - Object, - Promise, - Set, - performance: { now: () => 0 }, - // Electron 43 exposes this native object before the anti-detection script runs. - window: { chrome: {} }, - navigator: { - userAgent: - args.userAgent ?? - 'Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko) Chrome/151.0.0.0 Safari/537.36', - plugins: [], - languages: [], - permissions: new Permissions() - }, - Permissions, - Notification - } as AntiDetectionContext & Record -} - -describe('ANTI_DETECTION_SCRIPT', () => { - it('does not expose Chrome globals under a Firefox identity', () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied', - userAgent: googleAuthUserAgent() - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.window.chrome).toBeUndefined() - expect('chrome' in context.window).toBe(false) - }) - - it('keeps Chrome API stubs aligned with an ordinary Chrome page', () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.window.chrome?.runtime).toBeUndefined() - expect(context.window.chrome?.csi).toBeTypeOf('function') - expect(context.window.chrome?.loadTimes).toBeTypeOf('function') - }) - - it.each(['geolocation', 'idle-detection', 'midi', 'storage-access'])( - 'passes non-intercepted permission queries through to the native state for %s', - async (name) => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'denied' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - await expect(context.navigator.permissions.query({ name })).resolves.toEqual({ - state: 'denied', - onchange: null - }) - } - ) - - it('reports notification permission as granted after a site permission request succeeds', async () => { - const context = createContext({ - nativeNotificationPermission: 'denied', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.Notification.permission).toBe('default') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'prompt', - onchange: null - }) - - await expect(context.Notification.requestPermission()).resolves.toBe('granted') - - expect(context.Notification.permission).toBe('granted') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'granted', - onchange: null - }) - }) - - it('preserves notification permission when Electron already reports a grant', async () => { - const context = createContext({ - nativeNotificationPermission: 'granted', - requestedNotificationPermission: 'granted' - }) - - runInNewContext(ANTI_DETECTION_SCRIPT, context) - - expect(context.Notification.permission).toBe('granted') - await expect(context.navigator.permissions.query({ name: 'notifications' })).resolves.toEqual({ - state: 'granted', - onchange: null - }) - }) -}) diff --git a/src/main/browser/anti-detection.ts b/src/main/browser/anti-detection.ts deleted file mode 100644 index d725fbf965f..00000000000 --- a/src/main/browser/anti-detection.ts +++ /dev/null @@ -1,161 +0,0 @@ -// Why: Cloudflare Turnstile and similar bot detectors probe multiple browser -// APIs beyond navigator.webdriver. This script runs via -// Page.addScriptToEvaluateOnNewDocument before any page JS to mask automation -// signals that CDP debugger attachment and Electron's webview expose. -export const ANTI_DETECTION_SCRIPT = `(function() { - Object.defineProperty(navigator, 'webdriver', { get: () => false }); - // Why: Electron webviews expose an empty plugins array. Real Chrome always - // has at least a few default plugins (PDF Viewer, etc.). An empty array is - // a strong automation signal. - if (navigator.plugins.length === 0) { - Object.defineProperty(navigator, 'plugins', { - get: () => [ - { name: 'Chrome PDF Plugin', filename: 'internal-pdf-viewer' }, - { name: 'Chrome PDF Viewer', filename: 'mhjfbmdgcfjbbpaeojofohoefgiehjai' }, - { name: 'Native Client', filename: 'internal-nacl-plugin' } - ] - }); - } - // Why: auth hosts present Firefox, where Electron's native window.chrome is an identity mismatch. - if (navigator.userAgent.includes('Firefox/')) { - try { - delete window.chrome; - if ('chrome' in window) { - window.chrome = undefined; - } - } catch {} - } else { - // Why: Electron webviews may not have the window.chrome object that real - // Chrome exposes. Turnstile checks for its presence. The csi() and - // loadTimes() stubs satisfy deeper probes of Chrome-specific APIs. - if (!window.chrome) { - window.chrome = {}; - } - if (!window.chrome.csi) { - window.chrome.csi = function() { - return { - startE: Date.now(), - onloadT: Date.now(), - pageT: performance.now(), - tran: 15 - }; - }; - } - if (!window.chrome.loadTimes) { - window.chrome.loadTimes = function() { - return { - commitLoadTime: Date.now() / 1000, - connectionInfo: 'h2', - finishDocumentLoadTime: Date.now() / 1000, - finishLoadTime: Date.now() / 1000, - firstPaintAfterLoadTime: 0, - firstPaintTime: Date.now() / 1000, - navigationType: 'Other', - npnNegotiatedProtocol: 'h2', - requestTime: Date.now() / 1000 - 0.16, - startLoadTime: Date.now() / 1000 - 0.3, - wasAlternateProtocolAvailable: false, - wasFetchedViaSpdy: true, - wasNpnNegotiated: true - }; - }; - } - } - // Why: Electron's Permission API defaults to 'denied' for most permissions, - // but real Chrome returns 'prompt' for ungranted permissions. Returning - // 'denied' is a strong bot signal. Override the query result for common - // permissions that Turnstile and similar detectors probe. - var notificationPermission = 'default'; - var setNotificationPermission = function(permission) { - if (permission === 'granted' || permission === 'denied') { - notificationPermission = permission; - return permission; - } - notificationPermission = 'default'; - return 'default'; - }; - var notificationPermissionState = function() { - return notificationPermission === 'default' ? 'prompt' : notificationPermission; - }; - try { - if (Notification.permission === 'granted') { - notificationPermission = 'granted'; - } - } catch {} - const promptPerms = new Set([ - 'camera', 'microphone' - ]); - const origQuery = Permissions.prototype.query; - // Why: sites must receive the genuine PermissionStatus so native events, brand checks and method - // identity survive. Shadow only state, and resolve it lazily so existing statuses stay current. - function withOverriddenState(realStatus, stateProvider) { - Object.defineProperty(realStatus, 'state', { - configurable: true, - get: stateProvider - }); - return realStatus; - } - // Why: some names the real implementation rejects outright; fall back to an EventTarget so - // listener registration still works instead of throwing. - function fallbackStatus(stateProvider) { - const status = new EventTarget(); - Object.defineProperties(status, { - state: { configurable: true, get: stateProvider }, - onchange: { configurable: true, value: null, writable: true } - }); - return status; - } - function queryWithState(permissions, desc, stateProvider) { - let real; - try { - real = origQuery.call(permissions, desc); - } catch { - return Promise.resolve(fallbackStatus(stateProvider)); - } - return Promise.resolve(real).then( - (status) => withOverriddenState(status, stateProvider), - () => fallbackStatus(stateProvider) - ); - } - Permissions.prototype.query = function(desc) { - if (desc.name === 'notifications') { - return queryWithState(this, desc, notificationPermissionState); - } - if (promptPerms.has(desc.name)) { - return queryWithState(this, desc, () => 'prompt'); - } - return origQuery.call(this, desc); - }; - // Why: Electron may report Notification.permission as 'denied' by default - // whereas real Chrome reports 'default' for sites that haven't been granted - // or blocked. Turnstile cross-references this with the Permissions API. - try { - Object.defineProperty(Notification, 'permission', { - get: () => notificationPermission - }); - const origRequestPermission = Notification.requestPermission; - if (typeof origRequestPermission === 'function') { - Notification.requestPermission = function(callback) { - var wrappedCallback = typeof callback === 'function' - ? function(permission) { - callback(setNotificationPermission(permission)); - } - : undefined; - var result = origRequestPermission.call(Notification, wrappedCallback); - if (result && typeof result.then === 'function') { - return result.then(function(permission) { - return setNotificationPermission(permission); - }); - } - return result; - }; - } - } catch {} - // Why: Electron webviews may have an empty languages array. Real Chrome - // always has at least one entry. An empty array is an automation signal. - if (!navigator.languages || navigator.languages.length === 0) { - Object.defineProperty(navigator, 'languages', { - get: () => ['en-US', 'en'] - }); - } -})()` diff --git a/src/main/browser/browser-google-auth-ua.ts b/src/main/browser/browser-google-auth-ua.ts index 16ed14eb80f..e9b802f6d70 100644 --- a/src/main/browser/browser-google-auth-ua.ts +++ b/src/main/browser/browser-google-auth-ua.ts @@ -1,12 +1,12 @@ // Why: Google binds a signed-in session to the browser identity that created it. -// Cookies copied in from another browser (or sent under an Electron/Chrome-shaped -// UA that doesn't match a real first-party browser) get flagged by anti-fraud on -// accounts.google.com and expire within ~1h. Presenting a Firefox identity scoped +// Cookies copied in from another browser (or sent under a UA that doesn't match a +// real first-party browser) get flagged by anti-fraud on accounts.google.com and +// expire within ~1h. Presenting a Firefox identity scoped // to Google's auth hosts lets the user sign in *inside* the embedded browser, so // Google issues cookies bound to THIS browser that self-refresh — instead of us // transplanting cookies that go stale. Scope is deliberately the auth hosts only: // post-auth app surfaces (mail.google.com, myaccount.google.com, drive, etc.) keep -// the profile's real Chrome-shaped identity so nothing else about the session shifts. +// the profile's real identity so nothing else about the session shifts. // Why: exact hostname match — subdomains such as myaccount.google.com are post-auth // app surfaces, not the sign-in flow, and must retain the profile's real identity. diff --git a/src/main/browser/browser-manager-auth-user-agent.test.ts b/src/main/browser/browser-manager-auth-user-agent.test.ts index 405b689e850..eb0bd75b354 100644 --- a/src/main/browser/browser-manager-auth-user-agent.test.ts +++ b/src/main/browser/browser-manager-auth-user-agent.test.ts @@ -51,7 +51,7 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_CLEAN_UA + GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' const { @@ -197,8 +197,9 @@ describe('browserManager', () => { // Why: popup child windows get attachGuestPolicies but are never entered into tabIdByWebContentsId, // so a direct lookup of the UA mode misses the native opt-out. That is worse than doing nothing — - // native sessions skip setupClientHintsOverride, so the popup would send the raw Electron UA on the - // wire while navigator.userAgent claimed Firefox. Google sign-in popups are a first-class surface. + // native sessions never install the header-level Firefox switch, so the popup would send the + // Electron UA on the wire while navigator.userAgent claimed Firefox. Google sign-in popups are a + // first-class surface. it('leaves the UA untouched on auth hosts for a popup owned by a native-UA profile', () => { const ownerGuest = { id: 415, @@ -543,7 +544,7 @@ describe('browserManager', () => { ) expect(uaWrites.length).toBeGreaterThan(0) for (const [, params] of uaWrites) { - expect((params as { userAgent: string }).userAgent).toBe(GUEST_CLEAN_UA) + expect((params as { userAgent: string }).userAgent).toBe(GUEST_ELECTRON_UA) } }) }) diff --git a/src/main/browser/browser-manager-guest-lifecycle.test.ts b/src/main/browser/browser-manager-guest-lifecycle.test.ts index c8550083dfc..e136f532474 100644 --- a/src/main/browser/browser-manager-guest-lifecycle.test.ts +++ b/src/main/browser/browser-manager-guest-lifecycle.test.ts @@ -637,9 +637,9 @@ describe('browserManager', () => { ).toHaveLength(2) }) - it('cancels pending anti-detection reattach timers when unregistering a guest', () => { - vi.useFakeTimers() - + // Why: a plain browsing tab must never attach a debugger (Cloudflare treats CDP as a bot signal); + // the only debugger wiring it keeps is the detach listener that invalidates the auth-host UA override. + it('never attaches a debugger to a browsing guest and drops its detach listener on unregister', () => { const debuggerHandlers = new Map void>() const debuggerAttachMock = vi.fn() const guest = { @@ -670,18 +670,17 @@ describe('browserManager', () => { browserManager.attachGuestPolicies(guest as never) browserManager.registerGuest({ - browserPageId: 'browser-reattach', + browserPageId: 'browser-no-debugger', webContentsId: 809, rendererWebContentsId }) - debuggerHandlers.get('detach')?.() - expect(vi.getTimerCount()).toBe(1) + expect(debuggerAttachMock).not.toHaveBeenCalled() + expect(guest.debugger.sendCommand).not.toHaveBeenCalled() + expect(debuggerHandlers.has('detach')).toBe(true) - browserManager.unregisterGuest('browser-reattach') - expect(vi.getTimerCount()).toBe(0) - - vi.advanceTimersByTime(500) - expect(debuggerAttachMock).toHaveBeenCalledTimes(1) + browserManager.unregisterGuest('browser-no-debugger') + expect(debuggerHandlers.has('detach')).toBe(false) + expect(debuggerAttachMock).not.toHaveBeenCalled() }) }) diff --git a/src/main/browser/browser-manager-guest-policy-profile.test.ts b/src/main/browser/browser-manager-guest-policy-profile.test.ts index 47871f64ea9..c5c4c2521fe 100644 --- a/src/main/browser/browser-manager-guest-policy-profile.test.ts +++ b/src/main/browser/browser-manager-guest-policy-profile.test.ts @@ -52,6 +52,8 @@ type GuestFake = { isAttached: () => boolean attach: ReturnType sendCommand: ReturnType + on: ReturnType + off: ReturnType } on: (event: string, listener: (...args: never[]) => void) => void once: (event: string, listener: (...args: never[]) => void) => void @@ -81,7 +83,9 @@ function createGuest(id: number, url: string): GuestFake { debugger: { isAttached: () => true, attach: vi.fn(), - sendCommand: vi.fn(async () => undefined) + sendCommand: vi.fn(async () => undefined), + on: vi.fn(), + off: vi.fn() }, on: (event, listener) => { listeners.set(event, [...(listeners.get(event) ?? []), listener]) @@ -143,7 +147,7 @@ describe('guest policy profiles', () => { // The presence half of every absence below: a browsing guest observably takes all of it through // the same method, so a profile that fenced nothing — or an attach path that stopped installing // anything at all — cannot pass these by being uniformly empty. - it('gives a browsing guest link routing, popups and anti-detection', () => { + it('gives a browsing guest link routing, popups and auth-identity detach tracking', () => { const guest = createGuest(300, 'https://example.com/') browserManager.attachGuestPolicies(guest as never) @@ -151,7 +155,10 @@ describe('guest policy profiles', () => { expect(listenerCount(guest, 'dom-ready')).toBe(1) expect(listenerCount(guest, 'frame-created')).toBe(1) expect(listenerCount(guest, 'did-create-window')).toBe(1) - expect(guest.debugger.sendCommand).toHaveBeenCalled() + expect(guest.debugger.on).toHaveBeenCalledWith('detach', expect.any(Function)) + // Why: a plain browsing tab must never attach a debugger; Cloudflare treats CDP as a bot signal. + expect(guest.debugger.attach).not.toHaveBeenCalled() + expect(guest.debugger.sendCommand).not.toHaveBeenCalled() expect(navigateTo(guest, 'https://elsewhere.example/')).toBe(false) }) @@ -161,6 +168,7 @@ describe('guest policy profiles', () => { expect(listenerCount(guest, 'dom-ready')).toBe(0) expect(listenerCount(guest, 'frame-created')).toBe(0) expect(listenerCount(guest, 'did-create-window')).toBe(0) + expect(guest.debugger.on).not.toHaveBeenCalled() expect(guest.debugger.sendCommand).not.toHaveBeenCalled() expect(guest.executeJavaScriptInIsolatedWorld).not.toHaveBeenCalled() }) diff --git a/src/main/browser/browser-manager-guest-policy.ts b/src/main/browser/browser-manager-guest-policy.ts index c0d522235c8..3952a31c08b 100644 --- a/src/main/browser/browser-manager-guest-policy.ts +++ b/src/main/browser/browser-manager-guest-policy.ts @@ -35,8 +35,7 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean this.clickedLinkFrameNameByGuestId.set(guest.id, clickedLinkFrameName) } - // Why: bot detectors probe APIs that differ in Electron webviews; inject overrides each load so manual browsing passes. - const disposeAntiDetection = this.injectAntiDetection(guest) + const disposeAuthDetachTracking = this.trackDebuggerDetachForAuthUserAgent(guest) // Why: disable throttling so background screenshots still get frames; else the compositor stalls and capture returns empty. guest.setBackgroundThrottling(false) const disposePopupPolicy = this.installGuestPopupPolicy(guest, clickedLinkFrameName) @@ -44,14 +43,14 @@ export abstract class BrowserManagerGuestPolicy extends BrowserManagerGuestClean // Why: store cleanup so unregisterGuest can drop these listeners on teardown and let the WebContents wrapper GC. this.policyCleanupByGuestId.set(guest.id, () => { - disposeAntiDetection() + disposeAuthDetachTracking() disposePopupPolicy() disposeNavigationPolicy() }) } /** - * A workspace document is not the web: no popups, no link routing, no anti-detection, and no + * A workspace document is not the web: no popups, no link routing, no auth-identity tracking, and no * navigation bookkeeping for chrome it does not have. What it does share with a browsing guest is * this method's teardown, so a retired preview drops its listeners on the same path. */ diff --git a/src/main/browser/browser-manager-navigation.ts b/src/main/browser/browser-manager-navigation.ts index 4e061d288aa..5cb46ccb682 100644 --- a/src/main/browser/browser-manager-navigation.ts +++ b/src/main/browser/browser-manager-navigation.ts @@ -1,5 +1,4 @@ import { openPopupWithOriginBar, type PopupChildWindowOptions } from './popup-origin-bar-window' -import { cleanElectronUserAgent } from './browser-session-ua' import { getBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { googleAuthUserAgent, isGoogleAuthUrl } from './browser-google-auth-ua' import { buildViewportUserAgentOverride } from './browser-viewport-user-agent' @@ -12,10 +11,10 @@ import { BrowserManagerVisibility } from './browser-manager-visibility' export abstract class BrowserManagerNavigation extends BrowserManagerVisibility { // Why: navigator.userAgent (read by Google's auth JS) reflects the WebContents UA, - // not the request header, so the header-level Firefox switch in setupClientHintsOverride + // not the request header, so the header-level Firefox switch in setupGoogleAuthUserAgentOverride // must be matched here per navigation or the two layers disagree — itself a bot tell. // Restores the session's base identity off the auth hosts. Native-UA profiles opt out - // of the whole clean-UA path, so they keep their untouched identity everywhere. + // of the Firefox switch, so they keep their untouched identity everywhere. protected applyGoogleAuthUserAgent( guest: Electron.WebContents, url: string, @@ -24,8 +23,8 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility const browserPageId = this.tabIdByWebContentsId.get(guest.id) // Why: popup child windows get these policies but are never in tabIdByWebContentsId, so a direct // lookup misses the native-UA opt-out and would hand a native profile's popup the Firefox UA. - // That is worse than doing nothing: native sessions skip setupClientHintsOverride entirely, so - // the popup would send the raw Electron UA on the wire while navigator.userAgent claims Firefox. + // That is worse than doing nothing: native sessions never install the header-level Firefox + // switch, so the popup would send the Electron UA on the wire while navigator.userAgent claims Firefox. const ownerTabId = this.resolveBrowserTabIdForGuestWebContentsId(guest.id) // Session state is authoritative before renderer registration and after a native profile imports a source UA. const mode = @@ -56,15 +55,14 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // navigation (ERR_ABORTED) and replay the original request, which a POST-started OAuth chain // cannot survive — the sign-in lands on a blank tab. CDP retargets navigator.userAgent without // touching the navigation, and it outranks the WebContents UA from then on, so a guest that - // switches to it stays on it. The wire UA never depended on this write: setupClientHintsOverride - // rewrites User-Agent per request for auth-host URLs on its own. + // switches to it stays on it. The wire UA never depended on this write: + // setupGoogleAuthUserAgentOverride rewrites User-Agent per request for auth-host URLs on its own. if (options.duringRedirect === true || overrideState !== undefined) { if (this.canOverrideUserAgentOverCdp(guest)) { authOverrideIssuedOverCdp = true // Why: go through the viewport builder rather than writing nextUa raw, so both CDP writers - // resolve one identity for this URL — Firefox on auth hosts, the profile's clean base off - // them, any mobile preset preserved. Writing the session UA directly would put the - // unlaundered Electron token back on the wire. + // resolve one identity for this URL — Firefox on auth hosts, the session's base identity + // off them, any mobile preset preserved. void this.applyAuthUserAgentOverrideOverCdp( guest, (browserPageId ? this.viewportUaOverrideMobileByTabId.get(browserPageId) : undefined) ?? @@ -188,7 +186,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: Emulation.setUserAgentOverride is set once and stands across every later navigation, // outranking setUserAgent for navigator.userAgent. A viewport preset applied before reaching an - // auth host would otherwise pin navigator.userAgent to the Chrome-shaped preset UA while the + // auth host would otherwise pin navigator.userAgent to the session's preset UA while the // request header says Firefox — the two-layer disagreement this scope exists to remove. protected reapplyViewportUserAgentOverride( guest: Electron.WebContents, @@ -222,7 +220,7 @@ export abstract class BrowserManagerNavigation extends BrowserManagerVisibility // Why: the session UA is the profile's stable base identity. guest.getUserAgent() is not: // applyGoogleAuthUserAgent leaves it pinned to the Firefox auth UA once a guest switches to // the CDP override, so reading it back here would republish that identity on ordinary hosts. - baseUserAgent: cleanElectronUserAgent(baseUserAgent ?? guest.session.getUserAgent()) + baseUserAgent: baseUserAgent ?? guest.session.getUserAgent() }) ) } diff --git a/src/main/browser/browser-manager-state.ts b/src/main/browser/browser-manager-state.ts index bc65cc3d2dc..54b0f99b3c1 100644 --- a/src/main/browser/browser-manager-state.ts +++ b/src/main/browser/browser-manager-state.ts @@ -1,4 +1,3 @@ -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { BrowserGrabSessionController } from './browser-grab-session-controller' import type { BrowserCertificateTrustController } from './browser-certificate-trust-controller' import { @@ -190,56 +189,19 @@ export abstract class BrowserManagerState extends BrowserManagerViewportScrollSt this.settingsResolver = resolver } - // Why: addScriptToEvaluateOnNewDocument (CDP) is the only reliable pre-page-script hook per nav; executeJavaScript ran on the old page context. - protected injectAntiDetection(guest: Electron.WebContents): () => void { - let disposed = false - let reattachTimer: ReturnType | null = null - - const attach = (): void => { - if (disposed || guest.isDestroyed()) { - return - } - try { - if (!guest.debugger.isAttached()) { - guest.debugger.attach('1.3') - } - void guest.debugger - .sendCommand('Page.enable', {}) - .then(() => - guest.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) - ) - .catch(() => {}) - } catch { - /* best-effort — debugger may be unavailable */ - } - } - - // Why: proxy/bridge stop detaches the debugger and drops injections; re-attach (500ms delay to avoid racing a mid-restart) to keep overrides. + // Why: a debugger detach clears every CDP override Chromium holds, including the Google auth-host + // UA override, so the confirmed-override record must be dropped or the next auth navigation + // believes the identity is still installed and skips the write. + protected trackDebuggerDetachForAuthUserAgent(guest: Electron.WebContents): () => void { const onDetach = (): void => { this.authUserAgentOverrideStateByGuestId.delete(guest.id) - if (!disposed && !guest.isDestroyed() && reattachTimer === null) { - reattachTimer = setTimeout(() => { - reattachTimer = null - attach() - }, 500) - } } - try { - attach() guest.debugger.on('detach', onDetach) } catch { - /* best-effort */ + /* debugger may be unavailable */ } - return () => { - disposed = true - if (reattachTimer !== null) { - clearTimeout(reattachTimer) - reattachTimer = null - } try { guest.debugger.off('detach', onDetach) } catch { diff --git a/src/main/browser/browser-manager-types.ts b/src/main/browser/browser-manager-types.ts index a1b832a65bc..b91e2b741fe 100644 --- a/src/main/browser/browser-manager-types.ts +++ b/src/main/browser/browser-manager-types.ts @@ -117,7 +117,7 @@ export type PopupOwnerContext = { /** * What a guest is allowed to be. A browsing guest is the web — popups, clicked-link routing and - * anti-detection all apply. A workspace-document guest renders one granted document and gets none + * auth-identity tracking all apply. A workspace-document guest renders one granted document and gets none * of that; `host` is the renderer that minted its grant, and the only sink for what it reports. */ export type BrowserGuestPolicy = diff --git a/src/main/browser/browser-manager-viewport-override.test.ts b/src/main/browser/browser-manager-viewport-override.test.ts index b7d3bbabe0a..0ffc3c2a6e1 100644 --- a/src/main/browser/browser-manager-viewport-override.test.ts +++ b/src/main/browser/browser-manager-viewport-override.test.ts @@ -49,7 +49,6 @@ import { import { createViewportGuestFactory, flushViewportOps, - GUEST_CLEAN_UA, GUEST_ELECTRON_UA } from './browser-manager-viewport-test-fixtures' @@ -207,7 +206,7 @@ describe('browserManager', () => { mobile: false }) expect(debuggerSendCommand).toHaveBeenLastCalledWith('Emulation.setUserAgentOverride', { - userAgent: GUEST_CLEAN_UA + userAgent: GUEST_ELECTRON_UA }) // Navigating to the auth host must move the standing override to the Firefox identity. @@ -218,11 +217,11 @@ describe('browserManager', () => { userAgent: googleAuthUserAgent() }) - // Leaving the auth host restores the clean Chrome-shaped preset UA. + // Leaving the auth host restores the session's own preset UA. debuggerSendCommand.mockClear() willRedirect({ preventDefault: vi.fn() }, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) // Why: not an ordering race — debugger.sendCommand dispatches in call order over one channel, so @@ -241,9 +240,9 @@ describe('browserManager', () => { } // Why mobile: on the desktop branch the break is masked by coincidence — applyGoogleAuthUserAgent - // has already switched the WebContents UA to Firefox, and cleanElectronUserAgent passes a Firefox - // UA through untouched, so the stale-URL desktop path happens to emit Firefox anyway. The mobile - // branch derives a Chrome-shaped iPhone UA from that same base and exposes the real defect. + // has already switched the WebContents UA to Firefox, so the stale-URL desktop path happens to + // emit Firefox anyway. The mobile branch derives a Chrome-shaped iPhone UA from the session base + // and exposes the real defect. it('does not leave the Chrome preset UA standing when a mobile preset lands mid-navigation onto an auth host', async () => { const { guest, debuggerSendCommand } = makeGuest(4251, 'https://example.com/') // Hold the preset's first CDP command open so the navigation lands inside its await window. @@ -332,7 +331,7 @@ describe('browserManager', () => { // Without the fix the resuming preset re-reads getURL() as the auth host and clobbers the // navigation's correct write, stranding the Firefox UA on a non-auth page. - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('falls back to the committed URL once a navigation commits or fails', async () => { @@ -378,7 +377,7 @@ describe('browserManager', () => { await flushViewportOps() expect(guest.setUserAgent).toHaveBeenLastCalledWith(GUEST_ELECTRON_UA) - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) // A later preset must also resolve the committed, non-auth URL. debuggerSendCommand.mockClear() @@ -457,7 +456,7 @@ describe('browserManager', () => { expect(guest.setUserAgent).not.toHaveBeenCalled() expect(debuggerSendCommand).not.toHaveBeenCalledWith( 'Emulation.setUserAgentOverride', - expect.objectContaining({ userAgent: GUEST_CLEAN_UA }) + expect.objectContaining({ userAgent: GUEST_ELECTRON_UA }) ) }) @@ -517,7 +516,7 @@ describe('browserManager', () => { didFailLoad(null, -3, 'Aborted', 'https://accounts.google.com/redirected', true) await flushViewportOps() expect(guest.setUserAgent).not.toHaveBeenCalled() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('preserves the auth identity when a viewport preset is cleared after a redirect', async () => { @@ -592,7 +591,7 @@ describe('browserManager', () => { didStartNavigation(null, 'https://example.com/', false, true) await flushViewportOps() - expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_CLEAN_UA }) + expect(lastUserAgentOverride(debuggerSendCommand)).toEqual({ userAgent: GUEST_ELECTRON_UA }) }) it('reapplies a preset when navigation starts during its final UA write', async () => { @@ -849,8 +848,7 @@ describe('browserManager', () => { expect(debuggerAttach).toHaveBeenCalledWith('1.3') expect(debuggerSendCommand).toHaveBeenCalled() - // Why: detaching would clear Page.addScriptToEvaluateOnNewDocument - // (anti-detection). Guard regression. + // Why: detaching would clear every standing CDP override (viewport, auth UA). Guard regression. expect((guest.debugger as { detach?: unknown }).detach ?? undefined).toBeUndefined() }) diff --git a/src/main/browser/browser-manager-viewport-test-fixtures.ts b/src/main/browser/browser-manager-viewport-test-fixtures.ts index 8d68977f62e..076524ce5d0 100644 --- a/src/main/browser/browser-manager-viewport-test-fixtures.ts +++ b/src/main/browser/browser-manager-viewport-test-fixtures.ts @@ -3,8 +3,6 @@ import type { BrowserManagerMocks } from './browser-manager-test-harness' export const GUEST_ELECTRON_UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/134.0.0.0 Electron/30.0.0 Safari/537.36' -export const GUEST_CLEAN_UA = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/134.0.0.0 Safari/537.36' // Why: viewport UA writes are queued on the per-tab chain, so draining it takes more than one // microtask hop; loop until the chain is empty rather than guessing a tick count. @@ -53,7 +51,9 @@ export function createViewportGuestFactory( debugger: { isAttached: debuggerIsAttached, attach: debuggerAttach, - sendCommand: debuggerSendCommand + sendCommand: debuggerSendCommand, + on: vi.fn(), + off: vi.fn() } } return { diff --git a/src/main/browser/browser-manager-viewport.ts b/src/main/browser/browser-manager-viewport.ts index 1e79e760942..ce31dbe37e1 100644 --- a/src/main/browser/browser-manager-viewport.ts +++ b/src/main/browser/browser-manager-viewport.ts @@ -30,7 +30,7 @@ export abstract class BrowserManagerViewport extends BrowserManagerDownloadLifec return true } - // Why: emulate viewport via CDP; never detach the debugger here or per-guest overrides (addScriptToEvaluateOnNewDocument) are cleared. + // Why: emulate viewport via CDP; never detach the debugger here or the agent bridge's per-guest state is cleared. async setViewportOverride( browserTabId: string, override: BrowserViewportOverride | null diff --git a/src/main/browser/browser-session-partition-policies.test.ts b/src/main/browser/browser-session-partition-policies.test.ts index 78ce34d95fd..952c199536c 100644 --- a/src/main/browser/browser-session-partition-policies.test.ts +++ b/src/main/browser/browser-session-partition-policies.test.ts @@ -82,8 +82,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: async () => false })) vi.mock('./browser-session-ua', () => ({ - cleanElectronUserAgent: (userAgent: string) => userAgent, - setupClientHintsOverride: vi.fn() + setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ setBrowserSessionUserAgentMode: vi.fn() diff --git a/src/main/browser/browser-session-partition-policies.ts b/src/main/browser/browser-session-partition-policies.ts index 9f25d8840a2..ec3c68fb45e 100644 --- a/src/main/browser/browser-session-partition-policies.ts +++ b/src/main/browser/browser-session-partition-policies.ts @@ -9,7 +9,7 @@ import { } from './browser-session-proxy' import { hasSystemMediaAccess, requestSystemMediaAccess } from './browser-media-access' import { isAutoGrantedBrowserSessionPermission } from './browser-session-permission-policy' -import { cleanElectronUserAgent, setupClientHintsOverride } from './browser-session-ua' +import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserSessionUserAgentMode } from './browser-session-user-agent-mode' import { allowsBrowserWebAuthnPermission, @@ -92,10 +92,8 @@ export function installBrowserSessionPartitionPolicies( } browserManager.installCertificateRequestGuard(sess) - if (profile.userAgentMode !== 'native' && typeof sess.getUserAgent === 'function') { - const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) - sess.setUserAgent(cleanUA) - setupClientHintsOverride(sess, cleanUA) + if (profile.userAgentMode !== 'native') { + setupGoogleAuthUserAgentOverride(sess) } if (options?.permissions === 'deny') { sess.setPermissionRequestHandler((_webContents, _permission, callback) => callback(false)) @@ -191,11 +189,7 @@ export function applyBrowserSessionUserAgentModes(profiles: BrowserSessionProfil if (profile.userAgentMode === 'native') { continue } - - // Why: the default Electron UA leaks "Electron/X.X.X" + app name, which trips Cloudflare Turnstile. - const cleanUA = cleanElectronUserAgent(sess.getUserAgent()) - sess.setUserAgent(cleanUA) - setupClientHintsOverride(sess, cleanUA) + setupGoogleAuthUserAgentOverride(sess) } catch { /* session not available yet (e.g. unit tests or pre-ready) */ } diff --git a/src/main/browser/browser-session-partition-proxy-install.test.ts b/src/main/browser/browser-session-partition-proxy-install.test.ts index 0841ed94785..4beeaab04c9 100644 --- a/src/main/browser/browser-session-partition-proxy-install.test.ts +++ b/src/main/browser/browser-session-partition-proxy-install.test.ts @@ -44,8 +44,7 @@ vi.mock('./browser-media-access', () => ({ requestSystemMediaAccess: vi.fn(async () => false) })) vi.mock('./browser-session-ua', () => ({ - cleanElectronUserAgent: vi.fn((ua: string) => ua), - setupClientHintsOverride: vi.fn() + setupGoogleAuthUserAgentOverride: vi.fn() })) vi.mock('./browser-session-user-agent-mode', () => ({ setBrowserSessionUserAgentMode: vi.fn(), diff --git a/src/main/browser/browser-session-registry.persistence.test.ts b/src/main/browser/browser-session-registry.persistence.test.ts index fdba71e16d6..67653c0c76c 100644 --- a/src/main/browser/browser-session-registry.persistence.test.ts +++ b/src/main/browser/browser-session-registry.persistence.test.ts @@ -27,7 +27,7 @@ function installModuleMocks( copyFailures = new Set() ): { sessionFromPartitionMock: ReturnType - setupClientHintsOverrideMock: ReturnType + setupGoogleAuthUserAgentOverrideMock: ReturnType browserManagerHandleGuestWillDownloadMock: ReturnType browserManagerNotifyPermissionDeniedMock: ReturnType requestSystemMediaAccessMock: ReturnType @@ -36,6 +36,7 @@ function installModuleMocks( partition, setUserAgent: vi.fn(), getUserAgent: vi.fn(() => 'Mozilla/5.0 Electron/31 Orca'), + webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -45,7 +46,7 @@ function installModuleMocks( clearStorageData: vi.fn().mockResolvedValue(undefined), clearCache: vi.fn().mockResolvedValue(undefined) })) - const setupClientHintsOverrideMock = vi.fn() + const setupGoogleAuthUserAgentOverrideMock = vi.fn() const browserManagerHandleGuestWillDownloadMock = vi.fn() const browserManagerNotifyPermissionDeniedMock = vi.fn() const requestSystemMediaAccessMock = vi.fn().mockResolvedValue(true) @@ -119,8 +120,7 @@ function installModuleMocks( requestSystemMediaAccess: requestSystemMediaAccessMock })) vi.doMock('./browser-session-ua', () => ({ - cleanElectronUserAgent: vi.fn((ua: string) => ua.replace(/\s*Electron\/\S+/, '')), - setupClientHintsOverride: setupClientHintsOverrideMock + setupGoogleAuthUserAgentOverride: setupGoogleAuthUserAgentOverrideMock })) // This suite models replay with an in-memory filesystem. The real file-backed SQLite merge has // dedicated coverage; these fixtures are legacy unmarked images and keep the copy path. @@ -149,7 +149,7 @@ function installModuleMocks( return { sessionFromPartitionMock, - setupClientHintsOverrideMock, + setupGoogleAuthUserAgentOverrideMock, browserManagerHandleGuestWillDownloadMock, browserManagerNotifyPermissionDeniedMock, requestSystemMediaAccessMock @@ -234,21 +234,24 @@ describe('BrowserSessionRegistry persistence', () => { }) }) - it('keeps UA cleaning as the fallback for profiles without an override', async () => { + // Why: the stock Electron UA is what clears Cloudflare; only the Google auth switch installs. + it('keeps the stock UA and installs the Google auth switch for profiles without an override', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Default identity') const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value - expect(profileSession.setUserAgent).toHaveBeenCalledWith('Mozilla/5.0 Orca') - expect(setupClientHintsOverrideMock).toHaveBeenCalledWith(profileSession, 'Mozilla/5.0 Orca') + expect(profileSession.setUserAgent).not.toHaveBeenCalled() + expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalledWith(profileSession) }) it('leaves UA and client hints untouched for native-mode profiles', async () => { const fsState = createFsState() - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') await browserSessionRegistry.createProfile('isolated', 'Google', { userAgentMode: 'native' }) @@ -256,7 +259,7 @@ describe('BrowserSessionRegistry persistence', () => { const profileSession = sessionFromPartitionMock.mock.results.at(-1)?.value const { getBrowserSessionUserAgentMode } = await import('./browser-session-user-agent-mode') expect(profileSession.setUserAgent).not.toHaveBeenCalled() - expect(setupClientHintsOverrideMock).not.toHaveBeenCalled() + expect(setupGoogleAuthUserAgentOverrideMock).not.toHaveBeenCalled() expect(getBrowserSessionUserAgentMode(profileSession as never)).toBe('native') }) @@ -379,7 +382,7 @@ describe('BrowserSessionRegistry persistence', () => { // Why: imports before Aug 2026 persisted a synthesized source-browser UA // (fork imports as a broken Chrome/1.x, Chrome imports as a valid version). // Neither may ever be applied again — the engine-derived UA is the only one. - it('ignores legacy persisted UAs, valid or broken, and applies the engine UA', async () => { + it('ignores legacy persisted UAs, valid or broken, and keeps the engine UA', async () => { const importedPartition = 'persist:orca-browser-session-11111111-1111-4111-8111-111111111111' const brokenUa = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/1.158.1 Safari/537.36' @@ -405,7 +408,8 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -413,16 +417,9 @@ describe('BrowserSessionRegistry persistence', () => { const appliedUas = sessionFromPartitionMock.mock.results.flatMap((r) => r.value.setUserAgent.mock.calls.map((c: unknown[]) => c[0]) ) - expect(appliedUas).not.toContain(brokenUa) - expect(appliedUas).not.toContain(validUa) - // Why: every non-native profile falls to Orca's own cleaned engine UA. - expect(appliedUas.length).toBeGreaterThan(0) - expect(appliedUas.every((ua) => ua === 'Mozilla/5.0 Orca')).toBe(true) - expect( - setupClientHintsOverrideMock.mock.calls.every( - (c: unknown[]) => c[1] !== brokenUa && c[1] !== validUa - ) - ).toBe(true) + // Why: no persisted UA is ever written back; every profile keeps the engine's stock UA. + expect(appliedUas).toEqual([]) + expect(setupGoogleAuthUserAgentOverrideMock).toHaveBeenCalled() }) it('never applies a legacy persisted UA to a native-mode profile', async () => { @@ -487,7 +484,8 @@ describe('BrowserSessionRegistry persistence', () => { ] }) - const { sessionFromPartitionMock, setupClientHintsOverrideMock } = installModuleMocks(fsState) + const { sessionFromPartitionMock, setupGoogleAuthUserAgentOverrideMock } = + installModuleMocks(fsState) const { browserSessionRegistry } = await import('./browser-session-registry') browserSessionRegistry.initializeBrowserSessionsFromPersistedState() @@ -498,7 +496,7 @@ describe('BrowserSessionRegistry persistence', () => { expect(importedSessions.length).toBeGreaterThan(0) expect(importedSessions.every((sess) => sess.setUserAgent.mock.calls.length === 0)).toBe(true) expect( - setupClientHintsOverrideMock.mock.calls.some( + setupGoogleAuthUserAgentOverrideMock.mock.calls.some( ([sess]) => (sess as { partition?: string }).partition === importedPartition ) ).toBe(false) diff --git a/src/main/browser/browser-session-registry.test.ts b/src/main/browser/browser-session-registry.test.ts index d5111483fcf..ae81ffda4e2 100644 --- a/src/main/browser/browser-session-registry.test.ts +++ b/src/main/browser/browser-session-registry.test.ts @@ -33,7 +33,7 @@ vi.mock('./browser-manager', () => ({ import { browserSessionRegistry } from './browser-session-registry' import { googleAuthUserAgent } from './browser-google-auth-ua' -import { setupClientHintsOverride } from './browser-session-ua' +import { setupGoogleAuthUserAgentOverride } from './browser-session-ua' import { setBrowserNetworkProxySettingsResolver } from './browser-session-proxy' import { handleElectronProxyLogin } from '../network/electron-proxy-credentials' import { applyProxySettingsToSession } from '../network/proxy-settings' @@ -54,6 +54,7 @@ describe('BrowserSessionRegistry', () => { askForMediaAccessMock.mockResolvedValue(true) getMediaAccessStatusMock.mockReturnValue('granted') sessionFromPartitionMock.mockReturnValue({ + webRequest: { onBeforeSendHeaders: vi.fn() }, setPermissionRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), setDevicePermissionHandler: vi.fn(), @@ -528,80 +529,46 @@ describe('BrowserSessionRegistry', () => { }) }) - describe('setupClientHintsOverride', () => { - it('overrides sec-ch-ua headers for Edge UA', () => { + describe('setupGoogleAuthUserAgentOverride', () => { + const STOCK_UA = + 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) orca/1.0.0 Chrome/147.0.6890.3 Electron/43.0.0 Safari/537.36' + + function install(): (details: unknown, callback: ReturnType) => void { const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const edgeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36 Edg/147.0.3210.5' - - setupClientHintsOverride(mockSess, edgeUa) - + setupGoogleAuthUserAgentOverride({ webRequest: { onBeforeSendHeaders } } as never) expect(onBeforeSendHeaders).toHaveBeenCalledWith( { urls: ['https://*/*'] }, expect.any(Function) ) + return onBeforeSendHeaders.mock.calls[0][1] + } + // Why: the Electron token is what clears Cloudflare Turnstile; a Chrome-shaped UA with no + // client hints is what it rejects, so ordinary hosts must see the session's UA untouched. + it('leaves the stock Electron UA and its client hints alone off the auth hosts', () => { + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] listener( - { requestHeaders: { 'sec-ch-ua': 'old', 'sec-ch-ua-full-version-list': 'old' } }, + { + url: 'https://example.com/api', + requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', Cookie: 'abc=123' } + }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['sec-ch-ua']).toContain('Microsoft Edge') - expect(modified['sec-ch-ua']).toContain('"147"') - expect(modified['sec-ch-ua-full-version-list']).toContain('147.0.3210.5') - }) - - it('overrides sec-ch-ua headers for Chrome UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - - setupClientHintsOverride(mockSess, chromeUa) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener({ requestHeaders: { 'sec-ch-ua': 'old' } }, callback) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['sec-ch-ua']).toContain('Google Chrome') - expect(modified['sec-ch-ua']).not.toContain('Microsoft Edge') - }) - - it('registers handler even for non-Chrome UA but leaves sec-ch-ua untouched off auth hosts', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - - // Why: the Google-auth Firefox switch must install regardless of the base UA. - setupClientHintsOverride(mockSess, 'Mozilla/5.0 (compatible; MSIE 10.0)') - - expect(onBeforeSendHeaders).toHaveBeenCalledWith( - { urls: ['https://*/*'] }, - expect.any(Function) - ) - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener({ url: 'https://example.com/', requestHeaders: { 'sec-ch-ua': 'old' } }, callback) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toBe('old') + expect(modified['User-Agent']).toBe(STOCK_UA) + expect(modified['sec-ch-ua']).toBe('old') + expect(modified.Cookie).toBe('abc=123') }) it('presents a Firefox UA and strips client hints on Google auth hosts', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] listener( { url: 'https://accounts.google.com/v3/signin/identifier', requestHeaders: { - 'User-Agent': 'Chrome/147', + 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old', 'sec-ch-ua-full-version-list': 'old', 'sec-ch-ua-platform': '"macOS"' @@ -610,7 +577,7 @@ describe('BrowserSessionRegistry', () => { callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toMatch(/Firefox\/\d/) + expect(modified['User-Agent']).toBe(googleAuthUserAgent()) expect(modified['User-Agent']).not.toContain('Chrome') expect(modified['sec-ch-ua']).toBeUndefined() expect(modified['sec-ch-ua-full-version-list']).toBeUndefined() @@ -618,15 +585,8 @@ describe('BrowserSessionRegistry', () => { }) it('strips client hints on a cross-host request that carries the Firefox auth UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] // Subresource/XHR to a non-auth Google host while the auth document is on // screen: the WebContents Firefox UA leaks onto the request header. listener( @@ -651,99 +611,19 @@ describe('BrowserSessionRegistry', () => { expect(modified['sec-ch-ua-mobile']).toBeUndefined() }) - it('keeps the clean Chrome identity on cross-host requests that carry the Chrome UA', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, chromeUa) - + it('keeps the session identity on Google app subdomains (not auth hosts)', () => { + const listener = install() const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - // Regression guard: non-Google sites (Cloudflare) must keep Chrome hints. listener( { - url: 'https://example.com/api', - requestHeaders: { 'User-Agent': chromeUa, 'sec-ch-ua': 'old' } - }, - callback - ) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('does not strip hints for the Firefox UA when googleAuthOverride is disabled', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const chromeUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, chromeUa, { googleAuthOverride: false }) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { - url: 'https://play.google.com/log', - requestHeaders: { 'User-Agent': googleAuthUserAgent(), 'sec-ch-ua': 'old' } - }, - callback - ) - // Imported-native profiles never install the Firefox switch, so the strip - // branch stays inert and hints are aligned to Chrome instead. - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('keeps Chrome client hints on Google app subdomains (not auth hosts)', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride( - mockSess, - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - ) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { url: 'https://myaccount.google.com/', requestHeaders: { 'sec-ch-ua': 'old' } }, - callback - ) - expect(callback.mock.calls[0][0].requestHeaders['sec-ch-ua']).toContain('Google Chrome') - }) - - it('keeps an imported native UA on auth hosts while aligning its Chrome hints', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - const importedUa = - 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.6890.3 Safari/537.36' - setupClientHintsOverride(mockSess, importedUa, { googleAuthOverride: false }) - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { - url: 'https://accounts.google.com/v3/signin/identifier', - requestHeaders: { 'User-Agent': importedUa, 'sec-ch-ua': 'old' } + url: 'https://myaccount.google.com/', + requestHeaders: { 'User-Agent': STOCK_UA, 'sec-ch-ua': 'old' } }, callback ) const modified = callback.mock.calls[0][0].requestHeaders - expect(modified['User-Agent']).toBe(importedUa) - expect(modified['sec-ch-ua']).toContain('Google Chrome') - }) - - it('leaves non-Client-Hints headers unchanged', () => { - const onBeforeSendHeaders = vi.fn() - const mockSess = { webRequest: { onBeforeSendHeaders } } as never - setupClientHintsOverride(mockSess, 'Mozilla/5.0 Chrome/147.0.0.0 Safari/537.36') - - const callback = vi.fn() - const listener = onBeforeSendHeaders.mock.calls[0][1] - listener( - { requestHeaders: { Cookie: 'abc=123', 'sec-ch-ua': 'old', Accept: 'text/html' } }, - callback - ) - const modified = callback.mock.calls[0][0].requestHeaders - expect(modified.Cookie).toBe('abc=123') - expect(modified.Accept).toBe('text/html') + expect(modified['User-Agent']).toBe(STOCK_UA) + expect(modified['sec-ch-ua']).toBe('old') }) }) }) diff --git a/src/main/browser/browser-session-ua-wire-identity.electron.test.ts b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts new file mode 100644 index 00000000000..4e0719b2745 --- /dev/null +++ b/src/main/browser/browser-session-ua-wire-identity.electron.test.ts @@ -0,0 +1,180 @@ +import { spawnSync } from 'node:child_process' +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { build as buildVite } from 'vite' + +// Why this runs a real Electron: Cloudflare Turnstile rejects a Chrome-shaped UA that ships no +// client hints (error 600010) and clears a declared Electron client. The header layer is the +// only place that identity can be proven, and the vm-based unit tests cannot see Chromium's +// header emission at all. Every partition must therefore keep the stock Electron UA on the wire +// for ordinary hosts and present the Firefox identity on Google's sign-in hosts only. + +const electronBinary = createRequire(import.meta.url)('electron') as string +const fixtureRoots: string[] = [] + +afterAll(() => { + for (const root of fixtureRoots) { + rmSync(root, { recursive: true, force: true, maxRetries: 5, retryDelay: 100 }) + } +}) + +// Retry once when Electron startup times out before `ready`; keep later failures fatal. +const FIXTURE_LAUNCH_ATTEMPTS = 2 + +type CapturedRequest = { + url: string + userAgent: string | null + clientHints: string[] +} + +type FixtureResult = { + sessionUserAgent: string + navigatorUserAgent: string + requests: CapturedRequest[] +} + +function neverReachedElectronReady(fixtureResult: string): boolean { + try { + return (JSON.parse(fixtureResult) as { step?: string }).step === 'timed out after starting' + } catch { + return false + } +} + +function buildFixtureMain(modulePath: string, resultPath: string): string { + return ` +const { app, BrowserWindow, session } = require('electron') +const { writeFileSync } = require('node:fs') +const { setupGoogleAuthUserAgentOverride } = require(${JSON.stringify(modulePath)}) +const resultPath = ${JSON.stringify(resultPath)} +let currentStep = 'starting' +const mark = (step) => { + currentStep = step + writeFileSync(resultPath, JSON.stringify({ step })) +} + +async function run() { + const timeout = setTimeout(() => { + writeFileSync(resultPath, JSON.stringify({ step: 'timed out after ' + currentStep })) + app.exit(1) + }, 15000) + await app.whenReady() + mark('ready') + const partition = 'persist:wire-identity-test' + const sess = session.fromPartition(partition) + setupGoogleAuthUserAgentOverride(sess) + mark('auth switch installed') + + // Why: onSendHeaders reports the headers exactly as they leave the network stack, after the + // product's onBeforeSendHeaders listener has rewritten them. The requests must actually be + // dispatched for it to fire, so the session is pointed at a proxy that refuses every + // connection: nothing reaches the real hosts and every load fails fast. + await sess.setProxy({ proxyRules: 'http://127.0.0.1:9', proxyBypassRules: '<-loopback>' }) + const requests = [] + sess.webRequest.onSendHeaders({ urls: ['https://*/*'] }, (details) => { + const headers = details.requestHeaders || {} + const uaKey = Object.keys(headers).find((key) => key.toLowerCase() === 'user-agent') + requests.push({ + url: details.url, + userAgent: uaKey ? headers[uaKey] : null, + clientHints: Object.keys(headers) + .filter((key) => key.toLowerCase().startsWith('sec-ch-ua')) + .sort() + }) + }) + + const window = new BrowserWindow({ show: false, webPreferences: { partition } }) + mark('window created') + for (const url of ['https://example.com/', 'https://accounts.google.com/v3/signin/identifier']) { + await window.loadURL(url).catch(() => {}) + } + mark('navigations attempted') + const navigatorUserAgent = await window.webContents.executeJavaScript('navigator.userAgent') + clearTimeout(timeout) + writeFileSync(resultPath, JSON.stringify({ + sessionUserAgent: sess.getUserAgent(), + navigatorUserAgent, + requests + })) + window.destroy() + app.exit(0) +} + +run().catch((error) => { + writeFileSync(resultPath, JSON.stringify({ step: currentStep, error: String(error?.stack || error) })) + app.exit(1) +}) +` +} + +async function runFixture(): Promise { + const root = mkdtempSync(join(tmpdir(), 'orca-wire-identity-')) + fixtureRoots.push(root) + const modulePath = join(root, 'browser-session-ua.cjs') + const resultPath = join(root, 'result.json') + const fixturePath = join(root, 'main.cjs') + await buildVite({ + configFile: false, + logLevel: 'silent', + build: { + emptyOutDir: false, + lib: { + entry: join(process.cwd(), 'src/main/browser/browser-session-ua.ts'), + formats: ['cjs'], + fileName: () => 'browser-session-ua.cjs' + }, + outDir: root, + target: 'node20', + rollupOptions: { external: ['electron', /^node:/] } + } + }) + writeFileSync(fixturePath, buildFixtureMain(modulePath, resultPath)) + const { ELECTRON_RUN_AS_NODE: _electronRunAsNode, ...env } = process.env + const executable = process.platform === 'linux' ? 'xvfb-run' : electronBinary + for (let attempt = 1; ; attempt += 1) { + rmSync(resultPath, { force: true }) + // Why a fresh profile per attempt: a launch that never reached `ready` may have left the + // Chromium profile mid-initialization, and reusing it would bias the retry. + const electronArgs = [fixturePath, `--user-data-dir=${join(root, `profile-${attempt}`)}`] + const run = spawnSync( + executable, + process.platform === 'linux' + ? ['--auto-servernum', electronBinary, ...electronArgs, '--no-sandbox'] + : electronArgs, + { encoding: 'utf8', env, timeout: 60_000 } + ) + const fixtureResult = existsSync(resultPath) ? readFileSync(resultPath, 'utf8') : 'no result' + if (attempt < FIXTURE_LAUNCH_ATTEMPTS && neverReachedElectronReady(fixtureResult)) { + continue + } + expect(run.error).toBeUndefined() + expect(run.status, `${fixtureResult}\n${run.stdout}\n${run.stderr}`).toBe(0) + return JSON.parse(fixtureResult) as FixtureResult + } +} + +describe('browser session wire identity under Electron', () => { + it('sends the stock Electron UA to ordinary hosts and Firefox to Google auth hosts', async () => { + const result = await runFixture() + + // Presence precondition: the stock identity still carries the Electron token that the old + // Chrome-shaped rewrite stripped, so an identity check below cannot pass on an empty UA. + expect(result.sessionUserAgent).toMatch(/ Electron\/\d/) + + const ordinary = result.requests.find((request) => request.url === 'https://example.com/') + expect(ordinary, JSON.stringify(result.requests)).toBeDefined() + expect(ordinary?.userAgent).toBe(result.sessionUserAgent) + expect(result.navigatorUserAgent).toBe(result.sessionUserAgent) + + const auth = result.requests.find((request) => + request.url.startsWith('https://accounts.google.com/') + ) + expect(auth, JSON.stringify(result.requests)).toBeDefined() + expect(auth?.userAgent).toMatch(/Firefox\/\d/) + expect(auth?.userAgent).not.toContain('Chrome') + expect(auth?.clientHints).toEqual([]) + }) +}) diff --git a/src/main/browser/browser-session-ua.ts b/src/main/browser/browser-session-ua.ts index 96c55cf5ac9..1375ebc66f5 100644 --- a/src/main/browser/browser-session-ua.ts +++ b/src/main/browser/browser-session-ua.ts @@ -8,93 +8,28 @@ import { stripClientHints } from './browser-google-auth-ua' -// Why: Electron's default UA includes "Electron/X.X.X" and the app name -// (e.g. "orca/1.2.3"), which Cloudflare Turnstile and other bot detectors -// flag as non-human traffic. Strip those tokens so the webview's UA and -// sec-ch-ua Client Hints look like standard Chrome. -export function cleanElectronUserAgent(ua: string): string { - return ( - ua - .replace(/\s+Electron\/\S+/, '') - // Why: \S+ matches any non-whitespace token (e.g. "orca/1.3.8-rc.0") - // including pre-release semver strings that [\d.]+ would miss. - .replace(/(\)\s+)\S+\s+(Chrome\/)/, '$1$2') - ) -} - -// Why: Electron emits sec-ch-ua brands like "Not A(Brand" without a -// "Google Chrome" entry, which disagrees with the Chrome-shaped UA the session -// presents. Rewrite the hint headers to the brand set Chrome ships for the same -// engine version so the two surfaces tell one story. Also owns the Google -// auth-host Firefox switch, which must install even for a non-Chrome-shaped UA. -export function setupClientHintsOverride( - sess: Session, - ua: string, - options: { googleAuthOverride?: boolean } = {} -): void { - // Why: only Chrome-shaped base UAs carry sec-ch-ua hints to rewrite, but the - // Google-auth Firefox switch below must install regardless, so keep the hints - // optional rather than bailing out of the whole handler. - const chromeHints = buildChromeClientHints(ua) +// Why: the session keeps Electron's stock UA. Stripping the Electron/app tokens to look like +// plain Chrome is what Cloudflare Turnstile rejects (error 600010): a Chrome UA that ships no +// client hints reads as a spoof, while a declared Electron client clears the same challenge. +// This handler only owns the Google auth-host Firefox switch, which is a proven, host-scoped +// exception that must stay consistent across the header and every cross-host subresource. +export function setupGoogleAuthUserAgentOverride(sess: Session): void { const firefoxUa = googleAuthUserAgent() sess.webRequest.onBeforeSendHeaders({ urls: ['https://*/*'] }, (details, callback) => { const headers = details.requestHeaders - if (options.googleAuthOverride !== false && isGoogleAuthUrl(details.url)) { + if (isGoogleAuthUrl(details.url)) { // Why: present a Firefox identity on Google's sign-in hosts so the user logs // in inside the app and Google issues self-refreshing bound cookies. Strip // sec-ch-ua* because real Firefox sends none. setUserAgentHeader(headers, firefoxUa) stripClientHints(headers) - callback({ requestHeaders: headers }) - return - } - if (options.googleAuthOverride !== false && currentUserAgent(headers) === firefoxUa) { - // Why: while the auth document is on screen the WebContents UA is Firefox, - // so its cross-host subresource/XHR requests (gstatic, play.google.com, the - // sign-in challenge endpoints) reach here carrying the Firefox UA yet still - // bearing Chromium client hints. Rewriting those to Chrome pairs a Firefox - // UA with Chrome hints — a sharper cross-host identity tell than either - // alone, which can stall Google's password-submit challenge. Real Firefox - // sends no client hints, so strip them to keep one identity for the flow. + } else if (currentUserAgent(headers) === firefoxUa) { + // Why: while the auth document is on screen the WebContents UA is Firefox, so its + // cross-host subresource/XHR requests carry the Firefox UA yet still bear Chromium + // client hints — a sharper cross-host identity tell than either alone. stripClientHints(headers) - callback({ requestHeaders: headers }) - return - } - if (chromeHints) { - for (const key of Object.keys(headers)) { - const lower = key.toLowerCase() - if (lower === 'sec-ch-ua') { - headers[key] = chromeHints.secChUa - } else if (lower === 'sec-ch-ua-full-version-list') { - headers[key] = chromeHints.secChUaFull - } - } } callback({ requestHeaders: headers }) }) } - -function buildChromeClientHints(ua: string): { secChUa: string; secChUaFull: string } | null { - const chromeMatch = ua.match(/Chrome\/([\d.]+)/) - if (!chromeMatch) { - return null - } - const fullChromeVersion = chromeMatch[1] - const majorVersion = fullChromeVersion.split('.')[0] - - let brand = 'Google Chrome' - let brandFullVersion = fullChromeVersion - - const edgeMatch = ua.match(/Edg\/([\d.]+)/) - if (edgeMatch) { - brand = 'Microsoft Edge' - brandFullVersion = edgeMatch[1] - } - const brandMajor = brandFullVersion.split('.')[0] - - return { - secChUa: `"${brand}";v="${brandMajor}", "Chromium";v="${majorVersion}", "Not/A)Brand";v="24"`, - secChUaFull: `"${brand}";v="${brandFullVersion}", "Chromium";v="${fullChromeVersion}", "Not/A)Brand";v="24.0.0.0"` - } -} diff --git a/src/main/browser/browser-viewport-user-agent.ts b/src/main/browser/browser-viewport-user-agent.ts index 7dedf8d8a6e..b8159a44c2a 100644 --- a/src/main/browser/browser-viewport-user-agent.ts +++ b/src/main/browser/browser-viewport-user-agent.ts @@ -23,7 +23,7 @@ export type ViewportUserAgentOverride = { } // Why: responsive sites UA-sniff; this is Chrome DevTools' default iPhone UA template with the real -// Chrome major spliced in to keep sec-ch-ua consistent (see setupClientHintsOverride). +// Chrome major spliced in so the userAgentMetadata brands below agree with it. function buildMobileUserAgent(chromeMajor: string): string { return `Mozilla/5.0 (iPhone; CPU iPhone OS 17_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) CriOS/${chromeMajor}.0.0.0 Mobile/15E148 Safari/604.1` } @@ -44,7 +44,7 @@ export function buildViewportUserAgentOverride(args: { return { userAgent: googleAuthUserAgent() } } if (!args.mobile) { - // Why: desktop presets still need the clean (non-Electron) UA so Cloudflare/Turnstile don't flag the session. + // Why: desktop presets republish the session's own identity unchanged. return { userAgent: args.baseUserAgent } } const chromeMajor = extractChromeMajor(args.baseUserAgent) diff --git a/src/main/browser/browser-webauthn-profile-delete.test.ts b/src/main/browser/browser-webauthn-profile-delete.test.ts index 9a5e129885f..3c471fe2dc2 100644 --- a/src/main/browser/browser-webauthn-profile-delete.test.ts +++ b/src/main/browser/browser-webauthn-profile-delete.test.ts @@ -48,7 +48,8 @@ function mockSession(): MockSession { setDevicePermissionHandler: vi.fn(), setDisplayMediaRequestHandler: vi.fn(), setPermissionCheckHandler: vi.fn(), - setPermissionRequestHandler: vi.fn() + setPermissionRequestHandler: vi.fn(), + webRequest: { onBeforeSendHeaders: vi.fn() } }) as unknown as MockSession } diff --git a/src/main/browser/cdp-debugger-channel.ts b/src/main/browser/cdp-debugger-channel.ts index 352bc741ef0..18b819a1ce9 100644 --- a/src/main/browser/cdp-debugger-channel.ts +++ b/src/main/browser/cdp-debugger-channel.ts @@ -1,6 +1,5 @@ import { WebSocket } from 'ws' import type { WebContents } from 'electron' -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { acquireElectronDebugger, type ElectronDebuggerLease } from './electron-debugger-lease' import type { CdpClientResponseWriter } from './cdp-client-response-writer' import type { CdpSyntheticSessionRegistry } from './cdp-synthetic-session-registry' @@ -34,14 +33,8 @@ export class CdpDebuggerChannel { } this.attached = true - // Why: attaching the CDP debugger sets navigator.webdriver = true and - // exposes other automation signals that Cloudflare Turnstile checks. - // Inject before any page loads so challenges succeed. try { await this.webContents.debugger.sendCommand('Page.enable', {}) - await this.webContents.debugger.sendCommand('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) } catch { /* best-effort — page domain may not be ready yet */ } diff --git a/src/main/browser/cdp-debugger-events.ts b/src/main/browser/cdp-debugger-events.ts index 023d648a8ca..59d30105910 100644 --- a/src/main/browser/cdp-debugger-events.ts +++ b/src/main/browser/cdp-debugger-events.ts @@ -32,9 +32,11 @@ export function createCdpDebuggerMessageListener( | undefined if (p?.sessionId && p.targetInfo?.type === 'iframe' && p.targetInfo.targetId) { state.iframeSessions.set(p.targetInfo.targetId, p.sessionId) + // Why: no Runtime.enable here. Cross-origin iframes include challenge widgets + // (Cloudflare Turnstile), and the Runtime domain's console/Error.stack serialization + // is the CDP tell they detect; nothing reads iframe Runtime events anyway. guest.debugger.sendCommand('DOM.enable', {}, p.sessionId).catch(() => {}) guest.debugger.sendCommand('Accessibility.enable', {}, p.sessionId).catch(() => {}) - guest.debugger.sendCommand('Runtime.enable', {}, p.sessionId).catch(() => {}) } } if (method === 'Target.detachedFromTarget') { diff --git a/src/main/browser/cdp-debugger-lifecycle.ts b/src/main/browser/cdp-debugger-lifecycle.ts index f969f113d59..225eb689dba 100644 --- a/src/main/browser/cdp-debugger-lifecycle.ts +++ b/src/main/browser/cdp-debugger-lifecycle.ts @@ -1,5 +1,4 @@ import type { WebContents } from 'electron' -import { ANTI_DETECTION_SCRIPT } from './anti-detection' import { BrowserError } from './browser-error' import type { CdpTabState } from './cdp-auxiliary-commands' import type { CdpCommandSender } from './snapshot-engine' @@ -62,11 +61,6 @@ export class CdpDebuggerLifecycle { flatten: true }) - // Why: CDP attach exposes automation signals (navigator.webdriver) that Cloudflare checks; override per new document. - await sender('Page.addScriptToEvaluateOnNewDocument', { - source: ANTI_DETECTION_SCRIPT - }) - // Why: only remove this bridge's listeners; screencast/proxy sessions share the debugger and own their teardown. this.removeDebuggerListeners(guest, state) diff --git a/src/main/browser/cdp-ws-proxy-focus-replay.test.ts b/src/main/browser/cdp-ws-proxy-focus-replay.test.ts index 71b00cd9ff6..8404614ef8d 100644 --- a/src/main/browser/cdp-ws-proxy-focus-replay.test.ts +++ b/src/main/browser/cdp-ws-proxy-focus-replay.test.ts @@ -52,7 +52,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 99 }], ['DOM.focus', { backendNodeId: 99 }], ['Input.insertText', { text: 'hello' }] @@ -80,7 +79,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.result).toEqual({}) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 123 }, 'oopif-session-123'], ['DOM.focus', { backendNodeId: 123 }, 'oopif-session-123'], ['Input.insertText', { text: 'frame text' }, 'oopif-session-123'] @@ -112,7 +110,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 44 }], ['Runtime.callFunctionOn', { functionDeclaration: '() => document.activeElement?.id' }], ['Input.insertText', { text: 'after eval' }] @@ -155,7 +152,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 55 }], ['Input.insertText', { text: 'fallback' }] ]) @@ -197,7 +193,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(mock.webContents.focus).toHaveBeenCalledTimes(1) expect(getSendCommandCalls(mock)).toEqual([ ['Page.enable', {}], - ['Page.addScriptToEvaluateOnNewDocument', expect.any(Object)], ['DOM.focus', { backendNodeId: 77 }], ['DOM.focus', { backendNodeId: 77 }] ]) @@ -242,7 +237,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse?.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'DOM.focus', 'DOM.focus', 'Input.insertText' @@ -270,12 +264,7 @@ describe('CdpWsProxy DOM.focus replay', () => { // Why: both Page.bringToFront and Input.insertText natively call focus(), // independent of the (now-cleared) DOM.focus replay. expect(mock.webContents.focus).toHaveBeenCalledTimes(2) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'DOM.focus', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'DOM.focus', 'Input.insertText']) client.close() }) @@ -298,7 +287,6 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'DOM.focus', 'Page.captureScreenshot', 'Input.insertText' @@ -322,12 +310,7 @@ describe('CdpWsProxy DOM.focus replay', () => { expect(insertResponse.id).toBe(34) expect(insertResponse.result).toEqual({}) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'DOM.focus', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'DOM.focus', 'Input.insertText']) second.close() }) diff --git a/src/main/browser/cdp-ws-proxy.test.ts b/src/main/browser/cdp-ws-proxy.test.ts index c5f098ffa91..d5a2d14a05b 100644 --- a/src/main/browser/cdp-ws-proxy.test.ts +++ b/src/main/browser/cdp-ws-proxy.test.ts @@ -400,11 +400,7 @@ describe('CdpWsProxy', () => { }) expect(mock.webContents.focus).toHaveBeenCalledTimes(1) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'Input.insertText' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'Input.insertText']) client.close() }) @@ -421,7 +417,6 @@ describe('CdpWsProxy', () => { expect(response.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'Network.enable', 'Page.enable', 'Page.setLifecycleEventsEnabled', @@ -442,7 +437,6 @@ describe('CdpWsProxy', () => { expect(response.result).toEqual({}) expect(getSendCommandMethods(mock)).toEqual([ 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', 'Network.enable', 'Page.enable', 'Page.setLifecycleEventsEnabled' @@ -462,7 +456,7 @@ describe('CdpWsProxy', () => { sessionId: 'iframe-session-123' }) - expect(getSendCommandCalls(mock).slice(2)).toEqual([ + expect(getSendCommandCalls(mock).slice(1)).toEqual([ ['Network.enable', {}, 'iframe-session-123'], ['Page.enable', {}, 'iframe-session-123'], ['Page.setLifecycleEventsEnabled', { enabled: true }, 'iframe-session-123'], @@ -481,7 +475,7 @@ describe('CdpWsProxy', () => { sessionId: 'iframe-session-123' }) - expect(getSendCommandCalls(mock).slice(2)).toEqual([ + expect(getSendCommandCalls(mock).slice(1)).toEqual([ ['Network.enable', {}, 'iframe-session-123'], ['Page.enable', {}, 'iframe-session-123'], ['Page.setLifecycleEventsEnabled', { enabled: true }, 'iframe-session-123'], @@ -562,11 +556,7 @@ describe('CdpWsProxy', () => { expect(response.id).toBe(13) expect(response.result).toEqual({}) - expect(getSendCommandMethods(mock)).toEqual([ - 'Page.enable', - 'Page.addScriptToEvaluateOnNewDocument', - 'Runtime.evaluate' - ]) + expect(getSendCommandMethods(mock)).toEqual(['Page.enable', 'Runtime.evaluate']) client.close() }) diff --git a/src/main/claude/claude-agent-sdk-control-requests.test.ts b/src/main/claude/claude-agent-sdk-control-requests.test.ts new file mode 100644 index 00000000000..f76650d8151 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-control-requests.test.ts @@ -0,0 +1,26 @@ +import type { Query } from '@anthropic-ai/claude-agent-sdk' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { createClaudeControlSurface } from './claude-agent-sdk-control-requests' + +afterEach(() => { + vi.useRealTimers() +}) + +describe('createClaudeControlSurface stopTask', () => { + it('bounds a lost reply and permits a later stop request', async () => { + vi.useFakeTimers() + const stopTask = vi + .fn<() => Promise>() + .mockImplementationOnce(() => new Promise(() => {})) + .mockResolvedValueOnce() + const controls = createClaudeControlSurface({ stopTask } as unknown as Query) + const timedOut = expect(controls.stopTask('task-1', { timeoutMs: 25 })).rejects.toThrow( + 'claude stop_task request timed out' + ) + + await vi.advanceTimersByTimeAsync(25) + await timedOut + await expect(controls.stopTask('task-2', { timeoutMs: 25 })).resolves.toBeUndefined() + expect(stopTask).toHaveBeenCalledTimes(2) + }) +}) diff --git a/src/main/claude/claude-agent-sdk-control-requests.ts b/src/main/claude/claude-agent-sdk-control-requests.ts index 6bd396413fa..71498f28421 100644 --- a/src/main/claude/claude-agent-sdk-control-requests.ts +++ b/src/main/claude/claude-agent-sdk-control-requests.ts @@ -91,6 +91,7 @@ export type ClaudeControlSurface = { settings: Parameters[0], options?: ClaudeControlOptions ) => Promise + stopTask: (taskId: string, options?: ClaudeControlOptions) => Promise supportedModels: (options?: ClaudeControlOptions) => Promise initializationResult: (options?: ClaudeControlOptions) => Promise getSettings: (options?: ClaudeControlOptions) => Promise @@ -135,6 +136,10 @@ export function createClaudeControlSurface(query: Query): ClaudeControlSurface { () => query.applyFlagSettings(settings), options?.timeoutMs ).then(() => {}), + stopTask: (taskId, options) => + runClaudeControl('stop_task', () => query.stopTask(taskId), options?.timeoutMs).then( + () => {} + ), supportedModels: (options) => runClaudeControl('list_models', () => query.supportedModels(), options?.timeoutMs), initializationResult: (options) => diff --git a/src/main/claude/claude-background-task-tracker.test.ts b/src/main/claude/claude-background-task-tracker.test.ts new file mode 100644 index 00000000000..d8f316d7dcd --- /dev/null +++ b/src/main/claude/claude-background-task-tracker.test.ts @@ -0,0 +1,340 @@ +import { describe, expect, it } from 'vitest' +import { + ClaudeBackgroundTaskTracker, + classifyClaudeBackgroundTaskKind +} from './claude-background-task-tracker' + +function system(subtype: string, fields: Record): Record { + return { type: 'system', subtype, session_id: 'provider-1', uuid: crypto.randomUUID(), ...fields } +} + +function result(): Record { + return { type: 'result', subtype: 'success', session_id: 'provider-1', uuid: crypto.randomUUID() } +} + +function aggregate(tasks: unknown[]): Record { + return system('background_tasks_changed', { tasks }) +} + +describe('ClaudeBackgroundTaskTracker', () => { + it('classifies SDK task types without inferring them from descriptions', () => { + expect(classifyClaudeBackgroundTaskKind('local_agent')).toBe('agent') + expect(classifyClaudeBackgroundTaskKind('local_workflow')).toBe('workflow') + expect(classifyClaudeBackgroundTaskKind('local_bash')).toBe('command') + expect(classifyClaudeBackgroundTaskKind('monitor')).toBe('monitor') + expect(classifyClaudeBackgroundTaskKind('future_task')).toBe('unknown') + }) + + it('waits for the foreground turn to settle before monitoring a background task', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(tracker.state).toBeNull() + + expect(tracker.observe(result())).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'agent' }] + }) + }) + + it('uses an explicit background update for a foreground task and ignores progress alone', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_bash', + is_backgrounded: false + }) + ) + expect( + tracker.observe(system('task_progress', { task_id: 'task-1', description: 'still working' })) + ).toBe(false) + tracker.observe(result()) + expect(tracker.state).toBeNull() + + tracker.observe(system('task_updated', { task_id: 'task-1', patch: { is_backgrounded: true } })) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command' }] + }) + }) + + it('publishes bounded display details when a running task description changes', () => { + const tracker = new ClaudeBackgroundTaskTracker() + expect( + tracker.observe( + system('task_started', { + task_id: 'task-1', + task_type: 'local_bash', + is_backgrounded: true, + description: ' run\n the build ' + }) + ) + ).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-1', kind: 'command', description: 'run the build' }] + }) + + expect( + tracker.observe( + system('task_updated', { + task_id: 'task-1', + patch: { description: 'x'.repeat(600) } + }) + ) + ).toBe(true) + expect(tracker.state?.tasks?.[0]?.description).toHaveLength(512) + expect( + tracker.observe( + system('task_updated', { + task_id: 'task-1', + patch: { description: 'x'.repeat(600) } + }) + ) + ).toBe(false) + }) + + it('replaces its roster from aggregate lifecycle frames and preserves stoppable provider ids', () => { + const tracker = new ClaudeBackgroundTaskTracker() + expect( + tracker.observe( + aggregate([ + { task_id: 'task-agent', task_type: 'local_agent', description: 'agent' }, + { task_id: 'task-bash', task_type: 'local_bash', description: 'bash' } + ]) + ) + ).toBe(true) + expect(tracker.stoppableTaskIds).toEqual(['task-agent', 'task-bash']) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [ + { id: 'task-agent', kind: 'agent', description: 'agent' }, + { id: 'task-bash', kind: 'command', description: 'bash' } + ] + }) + + expect( + tracker.observe( + aggregate([{ task_id: 'task-next', task_type: 'local_workflow', description: 'workflow' }]) + ) + ).toBe(true) + expect(tracker.stoppableTaskIds).toEqual(['task-next']) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-next', kind: 'workflow', description: 'workflow' }] + }) + + expect(tracker.observe(aggregate([]))).toBe(true) + expect(tracker.stoppableTaskIds).toEqual([]) + expect(tracker.state).toBeNull() + }) + + it('excludes ambient aggregate tasks', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + aggregate([ + { task_id: 'ambient', task_type: 'monitor', description: 'watcher', ambient: true }, + { task_id: 'visible', task_type: 'local_bash', description: 'command' } + ]) + ) + + expect(tracker.stoppableTaskIds).toEqual(['visible']) + }) + + it('does not let late edge frames revive tasks cleared by an aggregate roster', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + aggregate([{ task_id: 'task-late', task_type: 'local_agent', description: 'agent' }]) + ) + tracker.observe(aggregate([])) + + tracker.observe( + system('task_started', { + task_id: 'task-late', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + tracker.observe( + system('task_updated', { task_id: 'task-late', patch: { is_backgrounded: true } }) + ) + + expect(tracker.stoppableTaskIds).toEqual([]) + expect(tracker.state).toBeNull() + }) + + it('lets an authoritative aggregate roster replace earlier terminal-edge evidence', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe(system('task_notification', { task_id: 'task-live', status: 'completed' })) + + tracker.observe( + aggregate([{ task_id: 'task-live', task_type: 'local_agent', description: 'agent' }]) + ) + + expect(tracker.stoppableTaskIds).toEqual(['task-live']) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-live', kind: 'agent', description: 'agent' }] + }) + }) + + it('keeps terminal edges authoritative on either side of aggregate replacement', () => { + const terminalFirst = new ClaudeBackgroundTaskTracker() + terminalFirst.observe( + system('task_notification', { task_id: 'task-first', status: 'completed' }) + ) + terminalFirst.observe(aggregate([])) + terminalFirst.observe( + system('task_started', { + task_id: 'task-first', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(terminalFirst.state).toBeNull() + + const terminalLast = new ClaudeBackgroundTaskTracker() + terminalLast.observe( + aggregate([{ task_id: 'task-last', task_type: 'local_agent', description: 'agent' }]) + ) + terminalLast.observe(system('task_notification', { task_id: 'task-last', status: 'completed' })) + terminalLast.observe( + system('task_started', { + task_id: 'task-last', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(terminalLast.state).toBeNull() + }) + + it('keeps terminal evidence authoritative across duplicates and out-of-order starts', () => { + const tracker = new ClaudeBackgroundTaskTracker() + const terminal = system('task_notification', { task_id: 'task-late', status: 'completed' }) + tracker.observe(terminal) + tracker.observe(terminal) + tracker.observe( + system('task_started', { + task_id: 'task-late', + task_type: 'local_workflow', + is_backgrounded: true + }) + ) + expect(tracker.state).toBeNull() + + tracker.observe( + system('task_started', { + task_id: 'task-live', + task_type: 'monitor' + }) + ) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-live', kind: 'monitor' }] + }) + expect( + tracker.observe(system('task_updated', { task_id: 'task-live', patch: { status: 'killed' } })) + ).toBe(true) + expect(tracker.state).toBeNull() + }) + + it('recognizes task types that are registered only as background work', () => { + for (const taskType of ['local_workflow', 'monitor']) { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe(system('task_started', { task_id: taskType, task_type: taskType })) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: taskType, kind: taskType === 'local_workflow' ? 'workflow' : 'monitor' }] + }) + } + }) + + it('admits unknown background updates conservatively and bounds edge-only fallback ids', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + system('task_updated', { task_id: 'unknown', patch: { is_backgrounded: true } }) + ) + expect(tracker.stoppableTaskIds).toEqual(['unknown']) + + for (let index = 0; index < 400; index += 1) { + tracker.observe( + system('task_started', { + task_id: `task-${index}`, + task_type: 'local_agent', + is_backgrounded: true + }) + ) + } + expect(tracker.stoppableTaskIds.length).toBeLessThanOrEqual(256) + }) + + it('bounds aggregate rosters and resets to the edge-only fallback on clear', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + aggregate( + Array.from({ length: 400 }, (_, index) => ({ + task_id: `aggregate-${index}`, + task_type: 'local_bash', + description: 'command' + })) + ) + ) + expect(tracker.stoppableTaskIds).toHaveLength(256) + + tracker.clear() + tracker.observe( + system('task_started', { + task_id: 'edge-after-reset', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(tracker.stoppableTaskIds).toEqual(['edge-after-reset']) + }) + + it('gates aggregate monitoring behind foreground turn completion', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe({ type: 'user' }, true) + tracker.observe( + aggregate([{ task_id: 'task-live', task_type: 'local_bash', description: 'command' }]) + ) + expect(tracker.state).toBeNull() + + expect(tracker.observe(result())).toBe(true) + expect(tracker.state).toEqual({ + state: 'monitoring', + tasks: [{ id: 'task-live', kind: 'command', description: 'command' }] + }) + }) + + it('ignores ambient SDK tasks and clears all liveness when the session ends', () => { + const tracker = new ClaudeBackgroundTaskTracker() + tracker.observe( + system('task_started', { + task_id: 'ambient', + task_type: 'monitor', + is_backgrounded: true, + ambient: true + }) + ) + expect(tracker.state).toBeNull() + tracker.observe( + system('task_started', { + task_id: 'task-live', + task_type: 'local_agent', + is_backgrounded: true + }) + ) + expect(tracker.clear()).toBe(true) + expect(tracker.state).toBeNull() + }) +}) diff --git a/src/main/claude/claude-background-task-tracker.ts b/src/main/claude/claude-background-task-tracker.ts new file mode 100644 index 00000000000..a1504a65dec --- /dev/null +++ b/src/main/claude/claude-background-task-tracker.ts @@ -0,0 +1,251 @@ +import type { + AgentSessionBackgroundTask, + AgentSessionBackgroundTaskState +} from '../../shared/agent-session-wire' + +const MAX_TRACKED_TASKS = 256 +const MAX_TASK_ID_LENGTH = 512 +const MAX_TASK_DESCRIPTION_LENGTH = 512 +const TERMINAL_TASK_STATES = new Set(['completed', 'failed', 'killed', 'stopped']) + +export type ClaudeBackgroundTaskKind = AgentSessionBackgroundTask['kind'] + +type TrackedTask = { + backgrounded: boolean + kind: ClaudeBackgroundTaskKind + description?: string +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null ? (value as Record) : null +} + +function taskId(message: Record): string | null { + const value = message.task_id + return typeof value === 'string' && value.length > 0 && value.length <= MAX_TASK_ID_LENGTH + ? value + : null +} + +function taskDescription(value: unknown): string | undefined { + if (typeof value !== 'string') { + return undefined + } + const trimmed = value.trim().replace(/\s+/g, ' ') + return trimmed.length > 0 ? trimmed.slice(0, MAX_TASK_DESCRIPTION_LENGTH) : undefined +} + +export function classifyClaudeBackgroundTaskKind(taskType: unknown): ClaudeBackgroundTaskKind { + switch (taskType) { + case 'local_agent': + return 'agent' + case 'local_workflow': + return 'workflow' + case 'local_bash': + return 'command' + case 'monitor': + return 'monitor' + default: + return 'unknown' + } +} + +export class ClaudeBackgroundTaskTracker { + private readonly tasks = new Map() + private readonly terminalTaskIds = new Set() + private aggregateRosterObserved = false + private foregroundTurnActive = false + private monitoring = false + private publishedTasksFingerprint = '' + + get state(): AgentSessionBackgroundTaskState | null { + if (!this.monitoring) { + return null + } + return { + state: 'monitoring', + tasks: this.backgroundTaskDetails() + } + } + + get stoppableTaskIds(): string[] { + const ids: string[] = [] + for (const [id, task] of this.tasks) { + if (task.backgrounded) { + ids.push(id) + } + } + return ids + } + + observe(message: Record, startsTurn = false): boolean { + if (startsTurn) { + this.foregroundTurnActive = true + } + if (message.type === 'result') { + this.foregroundTurnActive = false + } else if (message.type === 'system') { + if (!this.observeSystemFrame(message) && !startsTurn) { + return false + } + } else if (!startsTurn) { + return false + } + return this.refreshMonitoring() + } + + clear(): boolean { + this.tasks.clear() + this.terminalTaskIds.clear() + this.aggregateRosterObserved = false + this.foregroundTurnActive = false + return this.refreshMonitoring() + } + + private observeSystemFrame(message: Record): boolean { + if (message.subtype === 'background_tasks_changed') { + this.replaceAggregateRoster(message.tasks) + return true + } + const id = taskId(message) + if (!id) { + return false + } + if (message.subtype === 'task_notification') { + this.finish(id) + return true + } + if (message.subtype === 'task_updated') { + const patch = record(message.patch) + if (!patch) { + return false + } + if (TERMINAL_TASK_STATES.has(String(patch.status))) { + this.finish(id) + return true + } + const existing = this.tasks.get(id) + if ( + (patch.is_backgrounded === true || taskDescription(patch.description)) && + (!this.aggregateRosterObserved || existing) + ) { + this.upsert(id, { + backgrounded: patch.is_backgrounded === true || existing?.backgrounded === true, + kind: existing?.kind ?? 'unknown', + description: taskDescription(patch.description) ?? existing?.description + }) + return true + } + return false + } + if (message.subtype !== 'task_started' || this.terminalTaskIds.has(id)) { + return false + } + if (message.ambient === true || message.skip_transcript === true) { + this.finish(id) + return true + } + if (this.aggregateRosterObserved && !this.tasks.has(id)) { + return false + } + const kind = classifyClaudeBackgroundTaskKind(message.task_type) + this.upsert(id, { + backgrounded: message.is_backgrounded === true || kind === 'workflow' || kind === 'monitor', + kind, + description: taskDescription(message.description) + }) + return true + } + + private replaceAggregateRoster(value: unknown): void { + if (!Array.isArray(value)) { + return + } + this.aggregateRosterObserved = true + this.tasks.clear() + this.terminalTaskIds.clear() + for (const valueTask of value) { + if (this.tasks.size >= MAX_TRACKED_TASKS) { + break + } + const task = record(valueTask) + if (!task || task.ambient === true) { + continue + } + const id = taskId(task) + if (!id) { + continue + } + this.tasks.set(id, { + backgrounded: true, + kind: classifyClaudeBackgroundTaskKind(task.task_type), + description: taskDescription(task.description) + }) + } + } + + private upsert(id: string, task: TrackedTask): void { + const existing = this.tasks.get(id) + if (existing) { + this.tasks.set(id, { + backgrounded: existing.backgrounded || task.backgrounded, + kind: existing.kind === 'unknown' ? task.kind : existing.kind, + description: task.description ?? existing.description + }) + return + } + if (this.tasks.size >= MAX_TRACKED_TASKS) { + let foregroundId: string | undefined + for (const [candidateId, candidate] of this.tasks) { + if (!candidate.backgrounded) { + foregroundId = candidateId + break + } + } + if (!foregroundId) { + return + } + this.tasks.delete(foregroundId) + } + this.tasks.set(id, task) + } + + private finish(id: string): void { + this.tasks.delete(id) + this.terminalTaskIds.delete(id) + this.terminalTaskIds.add(id) + if (this.terminalTaskIds.size > MAX_TRACKED_TASKS) { + const oldest = this.terminalTaskIds.values().next() + if (!oldest.done) { + this.terminalTaskIds.delete(oldest.value) + } + } + } + + private refreshMonitoring(): boolean { + const details = this.foregroundTurnActive ? [] : this.backgroundTaskDetails() + const next = details.length > 0 + const fingerprint = next ? JSON.stringify(details) : '' + if (next === this.monitoring && fingerprint === this.publishedTasksFingerprint) { + return false + } + this.monitoring = next + this.publishedTasksFingerprint = fingerprint + return true + } + + private backgroundTaskDetails(): AgentSessionBackgroundTask[] { + const details: AgentSessionBackgroundTask[] = [] + for (const [id, task] of this.tasks) { + if (!task.backgrounded) { + continue + } + details.push({ + id, + kind: task.kind, + ...(task.description ? { description: task.description } : {}) + }) + } + return details + } +} diff --git a/src/main/claude/claude-structured-acquisition-release.ts b/src/main/claude/claude-structured-acquisition-release.ts index 1b633553e89..6be06c86e93 100644 --- a/src/main/claude/claude-structured-acquisition-release.ts +++ b/src/main/claude/claude-structured-acquisition-release.ts @@ -23,6 +23,7 @@ export async function releaseClaudeAcquisition(input: { onExitProven?: (sessionId: string, exit: ClaudeSessionExit) => Promise persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] onEvent?: ClaudeStructuredSessionAdapterDeps['onEvent'] + onBackgroundTasksChanged?: ClaudeStructuredSessionAdapterDeps['onBackgroundTasksChanged'] }): Promise { const exit = input.exits.get(input.sessionId) if (!exit || input.sessions.has(input.sessionId) || input.acquisitions.get(input.sessionId)) { diff --git a/src/main/claude/claude-structured-control-actions.test.ts b/src/main/claude/claude-structured-control-actions.test.ts index c471a8aca80..a90cb7908ba 100644 --- a/src/main/claude/claude-structured-control-actions.test.ts +++ b/src/main/claude/claude-structured-control-actions.test.ts @@ -1,8 +1,13 @@ import { describe, expect, it, vi } from 'vitest' -import { cancelClaudeTurn, answerClaudePrompt } from './claude-structured-control-actions' +import { + cancelClaudeTurn, + answerClaudePrompt, + stopClaudeBackgroundTasks +} from './claude-structured-control-actions' import { ClaudeControlRequestError } from './claude-stream-json-connection' import { ClaudePromptRegistry } from './claude-structured-prompt-replies' import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' type InterruptResult = Awaited> @@ -111,3 +116,46 @@ describe('answerClaudePrompt', () => { ).rejects.toThrow(/no longer waiting/) }) }) + +describe('stopClaudeBackgroundTasks', () => { + it('stops each live SDK task id and never depends on an active turn id', async () => { + const backgroundTasks = new ClaudeBackgroundTaskTracker() + backgroundTasks.observe({ + type: 'system', + subtype: 'background_tasks_changed', + tasks: [ + { task_id: 'task-agent', task_type: 'local_agent', description: 'agent' }, + { task_id: 'task-bash', task_type: 'local_bash', description: 'bash' } + ] + }) + const stopTask = vi.fn(async (_taskId: string, _options?: { timeoutMs?: number }) => {}) + const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession + + await expect(stopClaudeBackgroundTasks(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(stopTask.mock.calls).toEqual([ + ['task-agent', { timeoutMs: 5_000 }], + ['task-bash', { timeoutMs: 5_000 }] + ]) + }) + + it('stops issuing requests when ownership changes between tasks', async () => { + const backgroundTasks = new ClaudeBackgroundTaskTracker() + for (const taskId of ['task-1', 'task-2']) { + backgroundTasks.observe({ + type: 'system', + subtype: 'task_started', + task_id: taskId, + task_type: 'local_agent', + is_backgrounded: true + }) + } + let current = true + const stopTask = vi.fn(async (_taskId: string) => { + current = false + }) + const session = { backgroundTasks, connection: { stopTask } } as unknown as ClaudeSession + + await stopClaudeBackgroundTasks(session, undefined, () => current) + expect(stopTask).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.ts b/src/main/claude/claude-structured-control-actions.ts index d4484963aae..d216304c311 100644 --- a/src/main/claude/claude-structured-control-actions.ts +++ b/src/main/claude/claude-structured-control-actions.ts @@ -43,6 +43,29 @@ export async function cancelClaudeTurn( } } +export async function stopClaudeBackgroundTasks( + session: ClaudeSession, + timeoutMs: number | undefined, + isCurrent: ClaudeTurnCancellationGuard = () => true +): Promise<{ cancelled: boolean }> { + const taskIds = session.backgroundTasks.stoppableTaskIds + let cancelled = false + for (const taskId of taskIds) { + if (!isCurrent()) { + break + } + try { + await session.connection.stopTask(taskId, { timeoutMs }) + cancelled = true + } catch (error) { + if (!(error instanceof ClaudeControlRequestError)) { + throw error + } + } + } + return { cancelled } +} + export async function answerClaudePrompt( session: ClaudeSession, input: { itemId: string; kind: 'approval' | 'question'; optionId: string } diff --git a/src/main/claude/claude-structured-dispatch.test.ts b/src/main/claude/claude-structured-dispatch.test.ts index 4e8289a89e3..d66a64f82eb 100644 --- a/src/main/claude/claude-structured-dispatch.test.ts +++ b/src/main/claude/claude-structured-dispatch.test.ts @@ -6,6 +6,7 @@ import type { AgentJournalMessageItem } from '../../shared/agent-session-journal import { dispatchClaudeTurn, resolveClaudeReplayWaiter } from './claude-structured-dispatch' import { readClaudeImage } from './claude-structured-dispatch-content' import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' function sessionFor(send = vi.fn().mockResolvedValue(undefined)): ClaudeSession { return { @@ -19,6 +20,7 @@ function sessionFor(send = vi.fn().mockResolvedValue(undefined)): ClaudeSession dispatchWaiters: [], retiredDispatchWaiters: [], replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), dispatchSequence: 0, optionMutationSequence: 0, options: new Map(), diff --git a/src/main/claude/claude-structured-options.test.ts b/src/main/claude/claude-structured-options.test.ts index bc18a589e10..0738095f45d 100644 --- a/src/main/claude/claude-structured-options.test.ts +++ b/src/main/claude/claude-structured-options.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it, vi } from 'vitest' import { setClaudeStructuredOption } from './claude-structured-options' import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' function sessionFor(setModel: ClaudeSession['connection']['setModel']): ClaudeSession { return { @@ -14,6 +15,7 @@ function sessionFor(setModel: ClaudeSession['connection']['setModel']): ClaudeSe dispatchWaiters: [], retiredDispatchWaiters: [], replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), dispatchSequence: 0, optionMutationSequence: 0, options: new Map(), diff --git a/src/main/claude/claude-structured-session-acquisition-processless.test.ts b/src/main/claude/claude-structured-session-acquisition-processless.test.ts index e0996477e84..2c83dce1875 100644 --- a/src/main/claude/claude-structured-session-acquisition-processless.test.ts +++ b/src/main/claude/claude-structured-session-acquisition-processless.test.ts @@ -40,6 +40,7 @@ describe('Claude structured processless acquisition', () => { setPermissionMode: async () => {}, applyFlagSettings: async () => {}, send: async () => {}, + stopTask: async () => {}, close } return connection diff --git a/src/main/claude/claude-structured-session-adapter.ts b/src/main/claude/claude-structured-session-adapter.ts index f28b6e37f8f..40b12ecf74d 100644 --- a/src/main/claude/claude-structured-session-adapter.ts +++ b/src/main/claude/claude-structured-session-adapter.ts @@ -4,7 +4,11 @@ import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter' import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' -import { answerClaudePrompt, cancelClaudeTurn } from './claude-structured-control-actions' +import { + answerClaudePrompt, + cancelClaudeTurn, + stopClaudeBackgroundTasks +} from './claude-structured-control-actions' import { dispatchClaudeTurn } from './claude-structured-dispatch' import { releaseClaudeAcquisition } from './claude-structured-acquisition-release' import { acquireClaudeSession } from './claude-structured-session-acquisition' @@ -149,12 +153,21 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda } private emit( - _session: ClaudeSession | null, + session: ClaudeSession | null, _events: StructuredAgentSessionEventSink | undefined, event: ClaudeStructuredSessionEvent ): void { - _session?.translator?.handle(event) + const backgroundTasksChanged = + event.type === 'ended' + ? (session?.backgroundTasks.clear() ?? false) + : event.type === 'message' + ? (session?.backgroundTasks.observe(event.message, event.startsTurn === true) ?? false) + : false + session?.translator?.handle(event) this.deps.onEvent?.(event) + if (backgroundTasksChanged) { + this.deps.onBackgroundTasksChanged?.(event.sessionId, session?.backgroundTasks.state ?? null) + } } bindPromptItemId( @@ -191,6 +204,21 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda ) }) } + stopBackgroundTasks: StructuredAgentSessionAdapter['stopBackgroundTasks'] = (input) => { + const session = this.session(input.sessionId) + const acquisitionGeneration = session.acquisitionGeneration + return stopClaudeBackgroundTasks(session, this.deps.requestTimeoutMs, () => + Boolean( + this.sessions.get(input.sessionId) === session && + session.fence === input.fence && + session.acquisitionGeneration === acquisitionGeneration && + session.backgroundTasks.state + ) + ) + } + backgroundTaskState: NonNullable = ( + sessionId + ) => this.sessions.get(sessionId)?.backgroundTasks.state answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => answerClaudePrompt(this.session(input.sessionId), input) setOption: StructuredAgentSessionAdapter['setOption'] = (input) => @@ -210,6 +238,9 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda exits: this.exits, onExitProven: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit), ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: this.deps.onBackgroundTasksChanged } + : {}), ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) }) @@ -223,6 +254,9 @@ export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAda acquisitions: this.acquisitions, ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), ...(this.deps.readTranscriptLeaf ? { readTranscriptLeaf: this.deps.readTranscriptLeaf } : {}), + ...(this.deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: this.deps.onBackgroundTasksChanged } + : {}), ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) }) } diff --git a/src/main/claude/claude-structured-session-close.test.ts b/src/main/claude/claude-structured-session-close.test.ts index f92975e9ef4..0df0e913e50 100644 --- a/src/main/claude/claude-structured-session-close.test.ts +++ b/src/main/claude/claude-structured-session-close.test.ts @@ -4,7 +4,13 @@ import type { ClaudeStructuredSessionAdapterDeps, ClaudeStructuredSessionEvent } from './claude-structured-session-adapter' -import { adapterFor, fakeClaude, identityFor } from './claude-structured-session-test-support' +import { + PROVIDER_SESSION_ID, + adapterFor, + fakeClaude, + identityFor +} from './claude-structured-session-test-support' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' describe('Claude published session close lifecycle', () => { it('ends the session even when the durable handle write rejects', async () => { @@ -15,7 +21,17 @@ describe('Claude published session close lifecycle', () => { .fn>() .mockRejectedValueOnce(persistenceError) .mockResolvedValueOnce(undefined) - const adapter = adapterFor(claude, {}, events, [], undefined, undefined, persistHandle) + const backgroundStates: (AgentSessionBackgroundTaskState | null)[] = [] + const adapter = adapterFor( + claude, + {}, + events, + [], + undefined, + undefined, + persistHandle, + (_sessionId, state) => backgroundStates.push(state) + ) const journalSink: StructuredAgentSessionEventSink = { appendItem: () => {}, appendTombstone: () => {}, @@ -27,6 +43,18 @@ describe('Claude published session close lifecycle', () => { spawnToken: 'spawn-9', events: journalSink }) + claude.connections[0]!.handlers.onMessage?.({ + type: 'system', + subtype: 'task_started', + session_id: PROVIDER_SESSION_ID, + uuid: 'task-start', + task_id: 'background-1', + task_type: 'local_agent', + is_backgrounded: true + }) + expect(backgroundStates).toEqual([ + { state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] } + ]) const session = ( adapter as unknown as { sessions: Map void } | null }> @@ -39,6 +67,10 @@ describe('Claude published session close lifecycle', () => { expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) expect(events.filter((event) => event.type === 'handle')).toHaveLength(0) expect(disposeTranslator).toHaveBeenCalledOnce() + expect(backgroundStates).toEqual([ + { state: 'monitoring', tasks: [{ id: 'background-1', kind: 'agent' }] }, + null + ]) await expect(adapter.closeSession('session-1')).resolves.toBe(true) expect(persistHandle).toHaveBeenCalledTimes(2) diff --git a/src/main/claude/claude-structured-session-close.ts b/src/main/claude/claude-structured-session-close.ts index d37d3917796..de07439919d 100644 --- a/src/main/claude/claude-structured-session-close.ts +++ b/src/main/claude/claude-structured-session-close.ts @@ -11,6 +11,7 @@ import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' import { closeProcessRegistry } from '../../shared/child-process/close-process-registry' import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' @@ -52,6 +53,10 @@ type CloseClaudePublishedSessionInput = { fence: number }) => Promise onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void readTranscriptLeaf?: (input: { providerSessionId: string previousLeafUuid: string | null @@ -72,6 +77,9 @@ async function finalizeClaudePublishedSession( if ((await session.connection.close()) !== true) { return false } + if (session.backgroundTasks.clear()) { + input.onBackgroundTasksChanged?.(input.sessionId, null) + } try { const transcriptLeaf = input.readTranscriptLeaf ? await readClaudeTranscriptLeafWithReproof({ @@ -194,6 +202,10 @@ export function closeClaudePublishedSessionForDeps( fence: number }) => Promise onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void readTranscriptLeaf?: (input: { providerSessionId: string previousLeafUuid: string | null @@ -215,6 +227,10 @@ export async function closeClaudeSession(input: { fence: number }) => Promise onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void readTranscriptLeaf?: (input: { providerSessionId: string previousLeafUuid: string | null diff --git a/src/main/claude/claude-structured-session-publication.ts b/src/main/claude/claude-structured-session-publication.ts index 29d1113c814..7f8cc4b5692 100644 --- a/src/main/claude/claude-structured-session-publication.ts +++ b/src/main/claude/claude-structured-session-publication.ts @@ -4,6 +4,7 @@ import { claudeProviderHandleLink } from './claude-structured-owner-identity' import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' import type { ClaudeSession } from './claude-structured-session-state' +import { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' export function createClaudeSessionPublication(input: { connection: ClaudeSession['connection'] @@ -50,6 +51,7 @@ export function createClaudeSessionPublication(input: { dispatchWaiters: [], retiredDispatchWaiters: [], replayContentFallbackBlocked: false, + backgroundTasks: new ClaudeBackgroundTaskTracker(), dispatchSequence: 0, optionMutationSequence: 0, options: new Map(input.options), diff --git a/src/main/claude/claude-structured-session-state.ts b/src/main/claude/claude-structured-session-state.ts index 346ff686f76..30d3af84e7f 100644 --- a/src/main/claude/claude-structured-session-state.ts +++ b/src/main/claude/claude-structured-session-state.ts @@ -9,6 +9,8 @@ import type { ClaudeJournalTranslator } from './claude-structured-journal-transl import type { ClaudePendingPrompt, ClaudePromptRegistry } from './claude-structured-prompt-replies' import { cancelProcessAcquisition } from '../../shared/child-process/cancel-process-acquisition' import { randomUUID } from 'node:crypto' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' +import type { ClaudeBackgroundTaskTracker } from './claude-background-task-tracker' export type ClaudeAuthDiagnostic = { apiKeySourceConfigured: boolean @@ -54,6 +56,10 @@ export type ClaudeStructuredSessionAdapterDeps = { identity: AgentSessionJournalIdentity }) => Promise onEvent?: (event: ClaudeStructuredSessionEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void openConnection?: typeof openClaudeStreamJsonConnection readProcessStartTime?: (pid: number) => Promise mintLinkId?: () => string @@ -119,6 +125,7 @@ export type ClaudeSession = { capabilities: readonly string[] /** Provider uuid of the most recently admitted turn, if one is active. */ activeTurnId?: string + backgroundTasks: ClaudeBackgroundTaskTracker /** Monotonic fence advanced when a dispatch starts, including unresolved dispatches. */ dispatchSequence: number /** Dispatch sequence that admitted activeTurnId. */ diff --git a/src/main/claude/claude-structured-session-test-support.ts b/src/main/claude/claude-structured-session-test-support.ts index 6b0768b5134..903cafae416 100644 --- a/src/main/claude/claude-structured-session-test-support.ts +++ b/src/main/claude/claude-structured-session-test-support.ts @@ -155,6 +155,10 @@ export function fakeClaude( connection.calls.push({ subtype: 'cancel_async_message', params: { uuid } }) routed('cancel_async_message', { uuid }) }, + stopTask: async (taskId) => { + connection.calls.push({ subtype: 'stop_task', params: { taskId } }) + routed('stop_task', { taskId }) + }, send: async (message) => { connection.sent.push(message) if (message.type === 'user' && options.replayUuid !== null) { @@ -191,7 +195,8 @@ export function adapterFor( persistedHandles: unknown[] = [], initTimeoutMs?: number, readTranscriptLeaf?: ClaudeStructuredSessionAdapterDeps['readTranscriptLeaf'], - persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'], + onBackgroundTasksChanged?: ClaudeStructuredSessionAdapterDeps['onBackgroundTasksChanged'] ): ClaudeStructuredSessionAdapter { return new ClaudeStructuredSessionAdapter({ resolveLaunch: async () => ({ @@ -215,6 +220,7 @@ export function adapterFor( (async (handle) => { persistedHandles.push(handle) }), + ...(onBackgroundTasksChanged ? { onBackgroundTasksChanged } : {}), ...(readTranscriptLeaf ? { readTranscriptLeaf } : {}) }) } diff --git a/src/main/crash-reporting/expected-teardown-state.ts b/src/main/crash-reporting/expected-teardown-state.ts index 1480ecbbfe2..c2993769633 100644 --- a/src/main/crash-reporting/expected-teardown-state.ts +++ b/src/main/crash-reporting/expected-teardown-state.ts @@ -10,9 +10,17 @@ type Clock = () => number const monotonicNow = (): number => performance.now() let now: Clock = monotonicNow let systemSessionEndedAt: number | null = null +let systemSessionEnded = false export function markSystemSessionEnding(): void { systemSessionEndedAt = now() + systemSessionEnded = true +} + +// Why latched, unlike the 5s crash-suppression window below: a native dialog or a recovery verdict is never +// right once the OS is tearing the session down, however long the process outlives the signal. +export function isSystemSessionEnding(): boolean { + return systemSessionEnded } function isRecentSystemSessionEnd(): boolean { @@ -55,4 +63,5 @@ export function resolveExpectedTeardownScope({ export function resetExpectedTeardownStateForTest(clock: Clock = monotonicNow): void { now = clock systemSessionEndedAt = null + systemSessionEnded = false } diff --git a/src/main/daemon/daemon-foreground-process-protocol.ts b/src/main/daemon/daemon-foreground-process-protocol.ts index 25c6113057c..5c29e8a9e31 100644 --- a/src/main/daemon/daemon-foreground-process-protocol.ts +++ b/src/main/daemon/daemon-foreground-process-protocol.ts @@ -18,5 +18,7 @@ export type InspectProcessRequest = Omit & type: 'inspectProcess' payload: GetForegroundProcessRequest['payload'] & { expectedIncarnationId?: string + /** Optional; a daemon that predates it answers with the full capture as it always did. */ + steadyState?: boolean } } diff --git a/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts b/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts new file mode 100644 index 00000000000..ae7c29847c9 --- /dev/null +++ b/src/main/daemon/daemon-pty-adapter-steady-state-compat.test.ts @@ -0,0 +1,78 @@ +import { describe, expect, it, vi } from 'vitest' +import { DaemonPtyAdapter } from './daemon-pty-adapter' +import { COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION, PROTOCOL_VERSION } from './types' + +type ClientInternals = { + client: { request: ReturnType; disconnect: ReturnType } +} + +function createAdapter( + protocolVersion: number, + request: ReturnType +): DaemonPtyAdapter { + const adapter = new DaemonPtyAdapter({ + socketPath: '/tmp/orca-steady-state-compat.sock', + tokenPath: '/tmp/orca-steady-state-compat.token', + protocolVersion + }) + ;(adapter as unknown as ClientInternals).client = { request, disconnect: vi.fn() } + return adapter +} + +describe('steadyState across daemon versions', () => { + it('sends steadyState as an additive optional field on the existing inspectProcess request', async () => { + const request = vi.fn(async () => ({ foregroundProcess: 'claude', hasChildProcesses: true })) + const adapter = createAdapter(PROTOCOL_VERSION, request) + await adapter.inspectProcess('sess-a', { steadyState: true }) + expect(request).toHaveBeenCalledWith('inspectProcess', { + sessionId: 'sess-a', + steadyState: true + }) + adapter.dispose() + }) + + it('omits the field entirely when not requested, so the wire is byte-identical to before', async () => { + const request = vi.fn(async () => ({ foregroundProcess: null, hasChildProcesses: false })) + const adapter = createAdapter(PROTOCOL_VERSION, request) + await adapter.inspectProcess('sess-a', { expectedIncarnationId: 'inc-1', steadyState: false }) + expect(request).toHaveBeenCalledWith('inspectProcess', { + sessionId: 'sess-a', + expectedIncarnationId: 'inc-1' + }) + adapter.dispose() + }) + + it('an old daemon that ignores steadyState still answers with the full-capture shape, and the client accepts it', async () => { + // A pre-field daemon returns exactly what it always did: name + evidence, never a cheap answer. + const oldDaemonAnswer = { + foregroundProcess: 'claude', + hasChildProcesses: true, + foregroundProcessEvidence: { + verdict: 'live', + processName: 'claude', + authorityGeneration: 'gen', + observationEpoch: 1, + capturedAgeMs: 0, + ptyId: 'sess-a', + ptyIncarnationId: 'inc-1' + } + } + const request = vi.fn(async () => oldDaemonAnswer) + const adapter = createAdapter(COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION, request) + await expect(adapter.inspectProcess('sess-a', { steadyState: true })).resolves.toEqual( + oldDaemonAnswer + ) + adapter.dispose() + }) + + it('a pre-inspection daemon never sees the field: the client composes from getForegroundProcess as before', async () => { + const request = vi.fn(async () => ({ foregroundProcess: 'codex' })) + const adapter = createAdapter(COMPLETION_PROCESS_INSPECTION_PROTOCOL_VERSION - 1, request) + await expect(adapter.inspectProcess('sess-a', { steadyState: true })).resolves.toEqual({ + foregroundProcess: 'codex', + hasChildProcesses: true + }) + expect(request).toHaveBeenCalledWith('getForegroundProcess', { sessionId: 'sess-a' }) + adapter.dispose() + }) +}) diff --git a/src/main/daemon/daemon-pty-process-inspection.ts b/src/main/daemon/daemon-pty-process-inspection.ts index b05a8a6c6b6..335c641da62 100644 --- a/src/main/daemon/daemon-pty-process-inspection.ts +++ b/src/main/daemon/daemon-pty-process-inspection.ts @@ -25,7 +25,7 @@ export abstract class DaemonPtyProcessInspection extends DaemonPtyBufferSnapshot async inspectProcess( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { if (this.protocolVersion < GET_FOREGROUND_PROCESS_PROTOCOL_VERSION) { return clientOnlyUnverifiableInspection('old_host') @@ -47,7 +47,9 @@ export abstract class DaemonPtyProcessInspection extends DaemonPtyBufferSnapshot sessionId: id, ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } - : {}) + : {}), + // Additive: an older daemon ignores it and pays for the full capture. + ...(options?.steadyState === true ? { steadyState: true } : {}) }) } diff --git a/src/main/daemon/daemon-pty-router.ts b/src/main/daemon/daemon-pty-router.ts index 78e12215504..962cde6760e 100644 --- a/src/main/daemon/daemon-pty-router.ts +++ b/src/main/daemon/daemon-pty-router.ts @@ -179,7 +179,7 @@ export class DaemonPtyRouter implements IPtyProvider { async inspectProcess( id: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { return this.adapterForInspection(id).inspectProcess(id, options) } diff --git a/src/main/daemon/daemon-pty-session-spawn.ts b/src/main/daemon/daemon-pty-session-spawn.ts index 70519439919..bbb899b64a3 100644 --- a/src/main/daemon/daemon-pty-session-spawn.ts +++ b/src/main/daemon/daemon-pty-session-spawn.ts @@ -12,11 +12,8 @@ import { DaemonPtySpawnResult } from './daemon-pty-spawn-result' import type { DaemonPtySpawnContext } from './daemon-pty-spawn-request' import type { ColdRestoreInfo } from './history-reader' import { mintPtySessionId } from './pty-session-id' -import { - shellPathSupportsPtyStartupBarrier, - shellReadyMarkerComesFromLineEditor, - resolvePtyShellPath -} from './shell-ready' +import { shellPathSupportsPtyStartupBarrier, resolvePtyShellPath } from './shell-ready' +import { shellReadyMarkerComesFromLineEditor } from '../../shared/shell-ready-marker-timing' import { getRecoveredHistorySeedSegments } from './terminal-history-seed-segments' import { AGENT_SESSION_CLAIM_DAEMON_PROTOCOL_VERSION, type CreateOrAttachResult } from './types' import { normalizeWslColdRestoreCwd } from './wsl-cold-restore-cwd' diff --git a/src/main/daemon/daemon-request-router.ts b/src/main/daemon/daemon-request-router.ts index e081281fa8a..bb7d0d1a256 100644 --- a/src/main/daemon/daemon-request-router.ts +++ b/src/main/daemon/daemon-request-router.ts @@ -105,12 +105,17 @@ export class DaemonRequestRouter { return { foregroundProcess: this.options.host.getForegroundProcess(request.payload.sessionId) } - case 'inspectProcess': - return request.payload.expectedIncarnationId - ? this.options.host.inspectProcess(request.payload.sessionId, { - expectedIncarnationId: request.payload.expectedIncarnationId - }) + case 'inspectProcess': { + const options = { + ...(request.payload.expectedIncarnationId + ? { expectedIncarnationId: request.payload.expectedIncarnationId } + : {}), + ...(request.payload.steadyState === true ? { steadyState: true } : {}) + } + return Object.keys(options).length > 0 + ? this.options.host.inspectProcess(request.payload.sessionId, options) : this.options.host.inspectProcess(request.payload.sessionId) + } case 'confirmForegroundProcess': return { foregroundProcess: await this.options.host.confirmForegroundProcess( diff --git a/src/main/daemon/pty-subprocess/foreground-process-tracker.ts b/src/main/daemon/pty-subprocess/foreground-process-tracker.ts index 115b795202d..8726dc87281 100644 --- a/src/main/daemon/pty-subprocess/foreground-process-tracker.ts +++ b/src/main/daemon/pty-subprocess/foreground-process-tracker.ts @@ -36,7 +36,9 @@ type CachedAgentForeground = { processName: string; pid: number | null; refreshe export type PtyForegroundProcessTracker = { recordOutput(data: string): void markDead(): void - getForegroundProcess(): string | null + /** `rawFallback`: node-pty's own name only, with no identity cache and no background + * process-table refresh -- the cheap-tier tick must not fork a full `ps` as a side effect. */ + getForegroundProcess(options?: { rawFallback?: boolean }): string | null confirmForegroundProcess(): Promise confirmShellForeground(): Promise } @@ -213,10 +215,13 @@ export function createPtyForegroundProcessTracker(args: { cachedAgentForeground = null startupAgentForeground = null }, - getForegroundProcess: () => { + getForegroundProcess: (options) => { if (args.isDead()) { return null } + if (options?.rawFallback === true) { + return getFallbackProcess() + } try { const fallbackProcess = getFallbackProcess() const fallbackRecognition = recognizeAgentProcess(fallbackProcess) diff --git a/src/main/daemon/pty-subprocess/shell-launch-plan.ts b/src/main/daemon/pty-subprocess/shell-launch-plan.ts index 12e049d9f72..ba60e592743 100644 --- a/src/main/daemon/pty-subprocess/shell-launch-plan.ts +++ b/src/main/daemon/pty-subprocess/shell-launch-plan.ts @@ -35,11 +35,7 @@ import { } from '../../../shared/agent-process-recognition' import { ORCA_HERMES_STARTUP_QUERY_ENV } from '../../../shared/hermes-startup-query' import { WINDOWS_GIT_BASH_SHELL } from '../../../shared/windows-terminal-shell' -import { - getShellLaunchConfig, - resolvePtyShellPath, - shellReadyMarkerComesFromLineEditor -} from '../shell-ready' +import { getShellLaunchConfig, resolvePtyShellPath } from '../shell-ready' import { resolveWslSessionContext } from '../wsl-session-context' import { finalizeDaemonPtyEnvironment, rescrubDaemonPtyEnvironment } from './spawn-environment' import type { PtySubprocessOptions } from '../pty-subprocess' @@ -196,10 +192,10 @@ export function createPtyShellLaunchPlan( const waitsForShellReady = Boolean(opts.command) && (startupAgentRecognition?.agent !== 'codex' || - shellReadyMarkerComesFromLineEditor(shellPath) || shouldUseShellReadyStartupDelivery({ command: opts.command, - startupCommandDelivery: opts.startupCommandDelivery + startupCommandDelivery: opts.startupCommandDelivery, + shellPath })) delete env.ORCA_SHELL_FEATURES const shellLaunch = getShellLaunchConfig( diff --git a/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts b/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts index 52462e3a2eb..a1db9461070 100644 --- a/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts +++ b/src/main/daemon/repro-13767-shell-ready-marker-lost-to-exec.test.ts @@ -1,7 +1,7 @@ import { spawnSync } from 'node:child_process' -import { existsSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { existsSync, mkdtempSync, realpathSync, rmSync, writeFileSync } from 'node:fs' import { tmpdir } from 'node:os' -import { join } from 'node:path' +import { dirname, join } from 'node:path' import { describe, expect, it, vi } from 'vitest' import { createPtySubprocess } from './pty-subprocess' import { Session } from './session' @@ -10,6 +10,20 @@ const describePosix = process.platform === 'win32' ? describe.skip : describe const hasZsh = process.platform !== 'win32' && spawnSync('/bin/zsh', ['--version']).status === 0 const hasBash = process.platform !== 'win32' && spawnSync('/bin/bash', ['--version']).status === 0 const COMMAND_OUTPUT = 'ORCA_STARTUP_COMMAND_RAN' +// A second Bash install with its own canonical path -- the shape a login profile +// switches to (`exec /opt/homebrew/bin/bash`) and the one #18768 stalled on. A +// symlink cannot stand in: both sides are realpath'd before they are compared. +const alternateBashPath = ['/opt/homebrew/bin/bash', '/usr/local/bin/bash', '/usr/bin/bash'].find( + (candidate) => + hasBash && existsSync(candidate) && realpathSync(candidate) !== realpathSync('/bin/bash') +) +if (process.platform !== 'win32' && !alternateBashPath) { + // Why announced: usrmerge hosts resolve /usr/bin/bash back to /bin/bash, so these + // two skip on most Linux CI. A silent skip reads as coverage that does not exist. + console.warn( + '[repro-13767] no second Bash install with a distinct realpath; skipping the alternate-install recovery tests' + ) +} const READ_STARTED_FILE = '.orca-read-started' type ShellFixture = { @@ -122,7 +136,8 @@ type RunningFixture = { async function startFixture( fixture: ShellFixture, startupContent: string, - extraFiles: Record = {} + extraFiles: Record = {}, + pathEnv: string = process.env.PATH ?? '/usr/bin:/bin' ): Promise { const tempHome = mkdtempSync(join(tmpdir(), 'orca-shell-ready-exec-')) const previousHome = process.env.HOME @@ -150,7 +165,7 @@ async function startFixture( shellOverride: fixture.shellPath, env: { HOME: tempHome, - PATH: process.env.PATH ?? '/usr/bin:/bin', + PATH: pathEnv, SHELL: fixture.shellPath, TERM: 'xterm-256color' }, @@ -416,4 +431,49 @@ fi }, 10_000 ) + + const bashFixture = FIXTURES[2] as ShellFixture + const alternateBashTest = alternateBashPath ? it : it.skip + const alternateBashProfile = `if [[ -z "\${ORCA_EXEC_REPRO_DONE:-}" ]]; then + export ORCA_EXEC_REPRO_DONE=1 + exec ${alternateBashPath ?? '/bin/bash'} --noprofile --norc -l -i +fi +` + + alternateBashTest( + 'releases at the prompt of a second Bash install the pane PATH resolves', + async () => { + const running = await startFixture( + bashFixture, + alternateBashProfile, + {}, + `${dirname(alternateBashPath ?? '/bin/bash')}:/usr/bin:/bin` + ) + try { + await waitForOutput(running.subscribe, () => running.output().includes(COMMAND_OUTPUT)) + expect(running.session.shellState).toBe('ready') + expect(count(running.output(), COMMAND_OUTPUT)).toBe(1) + expect(running.output()).not.toContain('orca-shell-start') + } finally { + await running.cleanup() + } + }, + 10_000 + ) + + alternateBashTest( + 'does not trust a Bash install that the pane PATH cannot reach', + async () => { + const running = await startFixture(bashFixture, alternateBashProfile, {}, '/usr/bin:/bin') + try { + await waitForOutput(running.subscribe, () => running.output().includes('$')) + await new Promise((resolve) => setTimeout(resolve, 500)) + expect(running.session.shellState).toBe('pending') + expect(running.output()).not.toContain(COMMAND_OUTPUT) + } finally { + await running.cleanup() + } + }, + 10_000 + ) }) diff --git a/src/main/daemon/session-shell-ready-barrier.ts b/src/main/daemon/session-shell-ready-barrier.ts index 538fd57a49a..6fce89af89e 100644 --- a/src/main/daemon/session-shell-ready-barrier.ts +++ b/src/main/daemon/session-shell-ready-barrier.ts @@ -1,4 +1,4 @@ -import { shellReadyMarkerComesFromLineEditor } from './shell-ready' +import { shellReadyMarkerComesFromLineEditor } from '../../shared/shell-ready-marker-timing' import { installDeviceAttributesResponder, STARTUP_DA1_RESPONSE diff --git a/src/main/daemon/session-subprocess-handle.ts b/src/main/daemon/session-subprocess-handle.ts index f14469afbb4..9268686d78e 100644 --- a/src/main/daemon/session-subprocess-handle.ts +++ b/src/main/daemon/session-subprocess-handle.ts @@ -6,7 +6,7 @@ export type SubprocessHandle = { pid: number /** Live foreground process name of the PTY (node-pty's `.process`), e.g. * 'claude' / 'codex' / 'zsh'. Null once the child has exited. */ - getForegroundProcess(): string | null + getForegroundProcess(options?: { rawFallback?: boolean }): string | null /** Await process-table evidence captured after this confirmation request. */ confirmForegroundProcess?(): Promise /** Proves a fresh post-boundary PTY process tree contains only the shell. */ diff --git a/src/main/daemon/session.ts b/src/main/daemon/session.ts index b0f1dfa538d..9265b2acfef 100644 --- a/src/main/daemon/session.ts +++ b/src/main/daemon/session.ts @@ -252,8 +252,8 @@ export class Session { return this.output.getCwd() } - getForegroundProcess(): string | null { - return this.subprocess.getForegroundProcess() + getForegroundProcess(options?: { rawFallback?: boolean }): string | null { + return this.subprocess.getForegroundProcess(options) } async confirmForegroundProcess(): Promise { diff --git a/src/main/daemon/shell-ready.ts b/src/main/daemon/shell-ready.ts index 47acc567f6f..5208f9ced6b 100644 --- a/src/main/daemon/shell-ready.ts +++ b/src/main/daemon/shell-ready.ts @@ -101,11 +101,6 @@ export function resolvePtyShellPath(env: Record): string { return env.SHELL || process.env.SHELL || '/bin/zsh' } -export function shellReadyMarkerComesFromLineEditor(shellPath: string): boolean { - const shellName = pathWin32.basename(basename(shellPath)).toLowerCase() - return shellName === 'bash' || shellName === 'zsh' -} - export function shellPathSupportsPtyStartupBarrier(shellPath: string): boolean { const shellName = pathWin32.basename(basename(shellPath)).toLowerCase() // Why fish: markerless, its startup command is written before fish's reader owns diff --git a/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts b/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts new file mode 100644 index 00000000000..b02eec33773 --- /dev/null +++ b/src/main/daemon/terminal-host-cheap-tier-ps-scan-volume.test.ts @@ -0,0 +1,217 @@ +// Measurement for the cheap-tier process inspection. Drives the REAL daemon inspection +// entrypoint (`inspectTerminalHostProcess`) for 8 idle agent panes over a simulated 60s idle +// cadence (POLL_TIER_INTERVAL_MS.idle = 2,000ms) and counts `ps` forks BY COLUMN SET: a fork +// asking for `command=` is the full capture (measured 0.34-0.50s on a 1,900-process Mac, 1.15s +// on Linux), one without it is the cheap capture (0.03s on both). CI cannot time a real `ps` +// portably, so fork counts by column set are what this test measures; the per-fork costs above +// are the numbers measured by hand on the reference hosts. +// +// The second test is the zero-trade-off proof: the same tick sequence, including an agent exit +// and a restart, produces the identical foregroundProcess series with the cheap tier on and off. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Two seams because the two tiers spawn differently: the full evidence reader still forks +// through node:child_process, the cheap reader through Orca's runProcess entry point. +const { execFileMock, runProcessMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + runProcessMock: vi.fn() +})) +vi.mock('node:child_process', () => ({ execFile: execFileMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { resetCheapProcessTableSnapshotForTests } from '../../shared/cheap-process-table-snapshot-reader' +import { resetProcessTableSnapshotForTests } from '../../shared/process-table-snapshot-reader' +import { createPtyForegroundProcessTracker } from './pty-subprocess/foreground-process-tracker' +import { inspectTerminalHostProcess } from './terminal-host-process-inspection' +import type { Session } from './session' + +const PANE_COUNT = 8 +const IDLE_POLL_INTERVAL_MS = 2_000 // POLL_TIER_INTERVAL_MS.idle +const WINDOW_SECONDS = 60 +const TICKS = Math.floor((WINDOW_SECONDS * 1000) / IDLE_POLL_INTERVAL_MS) + +const shellPid = (pane: number): number => 1000 + pane * 100 +const agentPid = (pane: number): number => shellPid(pane) + 1 + +type PaneState = { agent: boolean; agentStart: string } +const panes: PaneState[] = Array.from({ length: PANE_COUNT }, () => ({ + agent: true, + agentStart: 'Thu Sep 3 16:02:05 2026' +})) + +const forks = { full: 0, cheap: 0 } + +function renderRows(): { full: string; cheap: string } { + const full: string[] = [] + const cheap: string[] = [] + panes.forEach((pane, i) => { + const s = shellPid(i) + const a = agentPid(i) + const tpgid = pane.agent ? a : s + const shellStat = pane.agent ? 'Ss' : 'Ss+' + cheap.push(`${s} 1 ${s} ${tpgid} ${shellStat} Thu Sep 3 16:02:01 2026`) + full.push(`${s} 1 ${s} ${tpgid} ${shellStat} ttys00${i} Thu Sep 3 16:02:01 2026 -zsh`) + if (pane.agent) { + cheap.push(`${a} ${s} ${a} ${a} S+ ${pane.agentStart}`) + full.push(`${a} ${s} ${a} ${a} S+ ttys00${i} ${pane.agentStart} node /usr/local/bin/claude`) + } + }) + return { full: `${full.join('\n')}\n`, cheap: `${cheap.join('\n')}\n` } +} + +function installCountingPs(): void { + execFileMock.mockImplementation((_cmd: string, args: string[], _opts: unknown, cb: unknown) => { + expect(args[1]).toContain('command=') + forks.full += 1 + ;(cb as (err: unknown, r: { stdout: string; stderr: string }) => void)(null, { + stdout: renderRows().full, + stderr: '' + }) + }) + runProcessMock.mockImplementation(async (spec: { args: readonly string[] }) => { + expect(spec.args[1]).not.toContain('command=') + forks.cheap += 1 + return { code: 0, signal: null, stdout: renderRows().cheap, stderr: '', timedOut: false } + }) +} + +function createSession(pane: number): Session { + const tracker = createPtyForegroundProcessTracker({ + process: { + pid: shellPid(pane), + get process() { + return panes[pane].agent ? 'node' : 'zsh' + } + } as never, + shellPath: '/bin/zsh', + sessionId: `wt:pane-${pane}`, + startupAgentRecognition: null, + isDead: () => false + }) + return { + pid: shellPid(pane), + incarnationId: `inc-${pane}`, + isAlive: true, + getForegroundProcess: (options?: { rawFallback?: boolean }) => + tracker.getForegroundProcess(options) + } as unknown as Session +} + +async function settle(): Promise { + for (let i = 0; i < 8; i += 1) { + await Promise.resolve() + } +} + +async function runTick(sessions: Session[], steadyState: boolean): Promise<(string | null)[]> { + const results = await Promise.all( + sessions.map((session, pane) => + inspectTerminalHostProcess({ + sessionId: `wt:pane-${pane}`, + session, + ...(steadyState ? { steadyState: true } : {}), + authorityGeneration: 'gen', + nextObservationEpoch: () => 1 + }) + ) + ) + await settle() + return results.map((r) => r.foregroundProcess) +} + +describe('cheap-tier ps scan volume at 8 idle agent panes over 60s', () => { + let platform: PropertyDescriptor | undefined + + beforeEach(() => { + execFileMock.mockReset() + runProcessMock.mockReset() + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + forks.full = 0 + forks.cheap = 0 + panes.forEach((pane) => { + pane.agent = true + pane.agentStart = 'Thu Sep 3 16:02:05 2026' + }) + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(1_000_000) + installCountingPs() + }) + + afterEach(() => { + vi.useRealTimers() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + it('replaces ~all full captures with cheap ones once every pane holds an anchor', async () => { + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + const names = await runTick(sessions, true) + expect(names.every((name) => name === 'claude')).toBe(true) + } + // Baseline today: one full capture per tick (TTL-shared across the 8 panes) = TICKS. + // Now: the first tick establishes every anchor from one full capture; every later tick is + // one TTL-shared cheap capture. Published numbers, from this run: + // before: 30 full (~0.34-0.50s each on macOS, 1.15s Linux) + 0 cheap + // after: 1 full + 29 cheap (~0.03s each) + expect(forks.full).toBe(1) + expect(forks.cheap).toBe(TICKS - 1) + expect(forks.full + forks.cheap).toBe(TICKS) + }) + + it('keeps today’s cost when the caller does not opt in (old client / remote / restore)', async () => { + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + await runTick(sessions, false) + } + expect(forks.cheap).toBe(0) + expect(forks.full).toBe(TICKS) + }) + + it('completion detection is byte-for-byte unchanged: exit, idle, and restart resolve identically with and without the cheap tier', async () => { + const script = async (steadyState: boolean): Promise<(string | null)[][]> => { + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + panes.forEach((pane) => { + pane.agent = true + pane.agentStart = 'Thu Sep 3 16:02:05 2026' + }) + const sessions = Array.from({ length: PANE_COUNT }, (_, pane) => createSession(pane)) + const series: (string | null)[][] = [] + for (let tick = 0; tick < TICKS; tick += 1) { + vi.setSystemTime(1_000_000 + tick * IDLE_POLL_INTERVAL_MS) + if (tick === 5) { + panes[2].agent = false // pane 2's agent exits + } + if (tick === 12) { + panes[2].agent = true // ...and is restarted with a new start time + panes[2].agentStart = 'Thu Sep 3 16:30:00 2026' + } + if (tick === 20) { + panes[6].agent = false + } + series.push(await runTick(sessions, steadyState)) + } + return series + } + const withCheapTier = await script(true) + const cheapForks = forks.cheap + forks.cheap = 0 + forks.full = 0 + const fullOnly = await script(false) + expect(withCheapTier).toEqual(fullOnly) + // And the exit was seen on the very tick it happened, in both modes. + expect(withCheapTier[4][2]).toBe('claude') + expect(withCheapTier[5][2]).not.toBe('claude') + expect(withCheapTier[12][2]).toBe('claude') + expect(withCheapTier[19][6]).toBe('claude') + expect(withCheapTier[20][6]).not.toBe('claude') + expect(cheapForks).toBeGreaterThan(0) + }) +}) diff --git a/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts b/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts new file mode 100644 index 00000000000..abf086e6e46 --- /dev/null +++ b/src/main/daemon/terminal-host-process-inspection-cheap-tier.test.ts @@ -0,0 +1,297 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Two seams because the two tiers spawn differently: the full evidence reader still forks +// through node:child_process, the cheap reader through Orca's runProcess entry point. +const { execFileMock, runProcessMock } = vi.hoisted(() => ({ + execFileMock: vi.fn(), + runProcessMock: vi.fn() +})) +vi.mock('node:child_process', () => ({ execFile: execFileMock })) +vi.mock('../../shared/child-process/run-process', () => ({ runProcess: runProcessMock })) + +import { resetCheapProcessTableSnapshotForTests } from '../../shared/cheap-process-table-snapshot-reader' +import { resetProcessTableSnapshotForTests } from '../../shared/process-table-snapshot-reader' +import { createPtyForegroundProcessTracker } from './pty-subprocess/foreground-process-tracker' +import { + inspectTerminalHostProcess, + type TerminalHostInspectionTier +} from './terminal-host-process-inspection' +import { getSteadyStateAnchor } from './terminal-host-steady-state-anchor' +import type { Session } from './session' + +const SHELL_PID = 4242 +const AGENT_PID = 4300 +const START_SHELL = 'Thu Sep 3 16:02:01 2026' +const START_AGENT = 'Thu Sep 3 16:02:05 2026' + +type Table = { agent: 'claude' | 'stopped' | 'gone' | 'replaced'; children?: number } + +/** One host table rendered in both column sets, so each fork answers by the args it asked for. */ +function renderTable(table: Table): { full: string; cheap: string } { + const shellTpgid = table.agent === 'claude' || table.agent === 'replaced' ? AGENT_PID : SHELL_PID + const shellStat = shellTpgid === SHELL_PID ? 'Ss+' : 'Ss' + const rows: { cheap: string; full: string }[] = [ + { + cheap: `${SHELL_PID} 1 ${SHELL_PID} ${shellTpgid} ${shellStat} ${START_SHELL}`, + full: `${SHELL_PID} 1 ${SHELL_PID} ${shellTpgid} ${shellStat} ttys004 ${START_SHELL} -zsh` + }, + { + cheap: `9000 1 9000 9000 Ss+ Thu Sep 3 12:00:00 2026`, + full: `9000 1 9000 9000 Ss+ ttys009 Thu Sep 3 12:00:00 2026 -zsh` + } + ] + if (table.agent !== 'gone') { + const stat = table.agent === 'stopped' ? 'T' : 'S+' + const start = table.agent === 'replaced' ? 'Thu Sep 3 16:30:00 2026' : START_AGENT + rows.push({ + cheap: `${AGENT_PID} ${SHELL_PID} ${AGENT_PID} ${shellTpgid} ${stat} ${start}`, + full: `${AGENT_PID} ${SHELL_PID} ${AGENT_PID} ${shellTpgid} ${stat} ttys004 ${start} node /usr/local/bin/claude` + }) + for (let i = 0; i < (table.children ?? 0); i += 1) { + const pid = AGENT_PID + 10 + i + rows.push({ + cheap: `${pid} ${AGENT_PID} ${AGENT_PID} ${shellTpgid} S+ Thu Sep 3 16:05:0${i} 2026`, + full: `${pid} ${AGENT_PID} ${AGENT_PID} ${shellTpgid} S+ ttys004 Thu Sep 3 16:05:0${i} 2026 rg --files` + }) + } + } + return { + full: `${rows.map((r) => r.full).join('\n')}\n`, + cheap: `${rows.map((r) => r.cheap).join('\n')}\n` + } +} + +const forks = { full: 0, cheap: 0 } +let table: Table = { agent: 'claude' } + +function installPs(): void { + execFileMock.mockImplementation((_cmd: string, args: string[], _opts: unknown, cb: unknown) => { + expect(args[1]).toContain('command=') + forks.full += 1 + ;(cb as (err: unknown, r: { stdout: string; stderr: string }) => void)(null, { + stdout: renderTable(table).full, + stderr: '' + }) + }) + runProcessMock.mockImplementation(async (spec: { args: readonly string[] }) => { + expect(spec.args[1]).not.toContain('command=') + forks.cheap += 1 + return { + code: 0, + signal: null, + stdout: renderTable(table).cheap, + stderr: '', + timedOut: false + } + }) +} + +function createSession(processName: () => string): Session { + let dead = false + const tracker = createPtyForegroundProcessTracker({ + process: { + pid: SHELL_PID, + get process() { + return processName() + } + } as never, + shellPath: '/bin/zsh', + sessionId: 'wt-1:pane-1', + startupAgentRecognition: null, + isDead: () => dead + }) + return { + pid: SHELL_PID, + incarnationId: 'inc-1', + get isAlive() { + return !dead + }, + getForegroundProcess: (options?: { rawFallback?: boolean }) => + tracker.getForegroundProcess(options), + markDead: () => { + dead = true + tracker.markDead() + } + } as unknown as Session & { markDead(): void } +} + +async function inspect( + session: Session, + options: { steadyState?: boolean; expectedIncarnationId?: string } = {} +): Promise<{ + tier: TerminalHostInspectionTier + result: Awaited> +}> { + let tier: TerminalHostInspectionTier = 'full' + const result = await inspectTerminalHostProcess({ + sessionId: 'wt-1:pane-1', + session, + ...options, + authorityGeneration: 'gen-1', + nextObservationEpoch: () => 1, + onTier: (t) => { + tier = t + } + }) + return { tier, result } +} + +async function settle(): Promise { + // The tracker's recognizing refresh runs off the same TTL-shared capture; let it land. + for (let i = 0; i < 8; i += 1) { + await Promise.resolve() + } +} + +async function advance(ms: number): Promise { + vi.setSystemTime(Date.now() + ms) +} + +describe('daemon cheap-tier process inspection', () => { + let platform: PropertyDescriptor | undefined + + beforeEach(() => { + execFileMock.mockReset() + runProcessMock.mockReset() + resetProcessTableSnapshotForTests() + resetCheapProcessTableSnapshotForTests() + forks.full = 0 + forks.cheap = 0 + table = { agent: 'claude' } + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(1_000_000) + installPs() + }) + + afterEach(() => { + vi.useRealTimers() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + /** Bring a session to a recognized anchor the way production does: one full cadence tick. */ + async function anchoredSession(): Promise { + const session = createSession(() => 'node') + const first = await inspect(session, { steadyState: true }) + await settle() + expect(first.tier).toBe('full') + expect(first.result.foregroundProcess).toBe('claude') + expect(getSteadyStateAnchor(session)?.agentName).toBe('claude') + return session + } + + it('a pane with NO recognized anchor never takes the cheap path, even when asked', async () => { + table = { agent: 'gone' } + const session = createSession(() => 'zsh') + for (let tick = 0; tick < 5; tick += 1) { + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toBeDefined() + } + expect(forks.cheap).toBe(0) + expect(forks.full).toBe(5) + }) + + it('serves an unchanged anchored pane from the cheap tier and OMITS evidence rather than faking it', async () => { + const session = await anchoredSession() + const fullBefore = forks.full + for (let tick = 0; tick < 4; tick += 1) { + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('cheap') + expect(result.foregroundProcess).toBe('claude') + expect(result.hasChildProcesses).toBe(true) + expect(result).not.toHaveProperty('foregroundProcessEvidence') + } + expect(forks.cheap).toBe(4) + expect(forks.full).toBe(fullBefore) + }) + + it('a request without steadyState (old client, remote client, restore path) always gets the full capture with evidence', async () => { + const session = await anchoredSession() + await advance(2_000) + const { tier, result } = await inspect(session) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toMatchObject({ + verdict: 'live', + processName: 'claude' + }) + expect(forks.cheap).toBe(0) + }) + + it('escalates to the full capture the moment the agent exits, and reports the exit', async () => { + const session = await anchoredSession() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + table = { agent: 'gone' } + await advance(2_000) + const { tier, result } = await inspect(session, { steadyState: true }) + expect(tier).toBe('full') + expect(result.foregroundProcessEvidence).toMatchObject({ verdict: 'live', processName: null }) + }) + + it.each<[string, Table]>([ + ['Ctrl-Z stops the agent', { agent: 'stopped' }], + ['exit-and-replace reuses the pid', { agent: 'replaced' }], + ['a child spawns under the agent', { agent: 'claude', children: 1 }] + ])('escalates when %s', async (_name, next) => { + const session = await anchoredSession() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + table = next + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + }) + + it('escalates when node-pty reports a different foreground name, without waiting on ps', async () => { + let name = 'node' + const session = createSession(() => name) + await inspect(session, { steadyState: true }) + await settle() + await advance(2_000) + expect((await inspect(session, { steadyState: true })).tier).toBe('cheap') + name = 'zsh' + await advance(2_000) + const cheapBefore = forks.cheap + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + expect(forks.cheap).toBe(cheapBefore) + }) + + it('falls through to the full capture when the cheap fork fails, and after an incarnation mismatch', async () => { + const session = await anchoredSession() + await advance(2_000) + runProcessMock.mockRejectedValueOnce(new Error('ps died')) + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + await advance(2_000) + const mismatched = await inspect(session, { steadyState: true, expectedIncarnationId: 'other' }) + expect(mismatched.tier).toBe('full') + expect(mismatched.result.foregroundProcessEvidence).toMatchObject({ + reason: 'incarnation_mismatch' + }) + }) + + it('a dead session is never served from its anchor', async () => { + const session = (await anchoredSession()) as Session & { markDead(): void } + session.markDead() + await expect(inspect(session, { steadyState: true })).rejects.toThrow('not found') + expect(forks.cheap).toBe(0) + }) + + it('an anchor is dropped when a full capture stops naming a recognized agent', async () => { + const session = await anchoredSession() + table = { agent: 'gone' } + await advance(2_000) + await inspect(session, { steadyState: true }) + expect(getSteadyStateAnchor(session)).toBeNull() + // Back with a new agent, but the pane must re-anchor via a FULL capture first. + table = { agent: 'claude' } + await advance(2_000) + const cheapBefore = forks.cheap + expect((await inspect(session, { steadyState: true })).tier).toBe('full') + expect(forks.cheap).toBe(cheapBefore) + }) +}) diff --git a/src/main/daemon/terminal-host-process-inspection.ts b/src/main/daemon/terminal-host-process-inspection.ts index 181ca7266d7..6810687631e 100644 --- a/src/main/daemon/terminal-host-process-inspection.ts +++ b/src/main/daemon/terminal-host-process-inspection.ts @@ -1,8 +1,15 @@ import { isShellProcess } from '../../shared/agent-detection' import type { RemoteForegroundEvidence } from '../../shared/foreground-process-evidence' +import { getCheapProcessTableSnapshot } from '../../shared/cheap-process-table-snapshot-reader' import { getStrictProcessTableSnapshotWithAge } from '../../shared/process-table-snapshot-reader' import { resolveRemoteForegroundEvidence } from '../providers/agent-foreground-process' +import { buildPaneProcessFingerprint } from '../providers/posix-pane-foreground-fingerprint' import type { Session } from './session' +import { + clearSteadyStateAnchor, + getSteadyStateAnchor, + rememberSteadyStateAnchor +} from './terminal-host-steady-state-anchor' import { SessionNotFoundError } from './types' export type TerminalHostProcessInspection = { @@ -13,13 +20,23 @@ export type TerminalHostProcessInspection = { type RetiredIncarnation = { incarnationId: string; code: number; expiresAt: number } +/** + * Tick tiers for a POSIX pane. `cheap` forks `ps` without `tty=`/`command=` (11-38x cheaper) + * and answers from the anchored identity when the pane fingerprint is unchanged; anything it + * cannot prove escalates to `full`, today's evidence capture. + */ +export type TerminalHostInspectionTier = 'full' | 'cheap' + export async function inspectTerminalHostProcess(args: { sessionId: string session: Session | null expectedIncarnationId?: string + /** The caller is a self-correcting poll that only reads the process name, never evidence. */ + steadyState?: boolean retiredIncarnation?: RetiredIncarnation authorityGeneration: string nextObservationEpoch: () => number + onTier?: (tier: TerminalHostInspectionTier) => void }): Promise { const { sessionId, session, expectedIncarnationId, retiredIncarnation } = args if (!session || !session.isAlive) { @@ -45,9 +62,22 @@ export async function inspectTerminalHostProcess(args: { throw new SessionNotFoundError(sessionId) } + const incarnationMatches = + !expectedIncarnationId || expectedIncarnationId === session.incarnationId + if (args.steadyState === true && incarnationMatches) { + const anchored = await readAnchoredForeground(session) + if (anchored !== null) { + args.onTier?.('cheap') + // No evidence member on purpose: a tty-less capture cannot fence anything, and a + // fabricated fence would be read by remote/restore consumers as an observation. + return { foregroundProcess: anchored, hasChildProcesses: true } + } + } + args.onTier?.('full') + const foregroundProcess = session.getForegroundProcess() let evidence: RemoteForegroundEvidence - if (expectedIncarnationId && expectedIncarnationId !== session.incarnationId) { + if (!incarnationMatches) { evidence = unverifiableEvidence(args, session, 'incarnation_mismatch') } else { try { @@ -64,8 +94,10 @@ export async function inspectTerminalHostProcess(args: { }, snapshot.rows ) + await rememberSteadyStateAnchor(session, evidence, snapshot.rows) } catch { evidence = unverifiableEvidence(args, session, 'process_table_unreadable') + clearSteadyStateAnchor(session) } } return { @@ -75,6 +107,32 @@ export async function inspectTerminalHostProcess(args: { } } +/** + * Cheap tier, gated on an anchor the last full capture established. Start discovery therefore + * keeps today's exact behaviour: a pane with no anchor never gets here. A recognized agent's + * exit is a pid vanishing from the subtree, which the fingerprint always sees, so completion + * detection is unaffected. Any mismatch, unreadable capture, changed node-pty name, or non-POSIX + * host answers null -> full tier. + */ +async function readAnchoredForeground(session: Session): Promise { + const anchor = getSteadyStateAnchor(session) + if (process.platform === 'win32' || !anchor) { + return null + } + if (session.getForegroundProcess({ rawFallback: true }) !== anchor.rawFallback) { + return null + } + try { + const observed = await buildPaneProcessFingerprint( + await getCheapProcessTableSnapshot(), + session.pid + ) + return observed !== null && observed === anchor.fingerprint ? anchor.agentName : null + } catch { + return null + } +} + function unverifiableEvidence( args: { sessionId: string diff --git a/src/main/daemon/terminal-host-steady-state-anchor.ts b/src/main/daemon/terminal-host-steady-state-anchor.ts new file mode 100644 index 00000000000..562224f856f --- /dev/null +++ b/src/main/daemon/terminal-host-steady-state-anchor.ts @@ -0,0 +1,58 @@ +import { recognizeAgentProcess } from '../../shared/agent-process-recognition' +import type { RemoteForegroundEvidence } from '../../shared/foreground-process-evidence' +import { buildPaneProcessFingerprint } from '../providers/posix-pane-foreground-fingerprint' +import type { Session } from './session' + +/** + * What the last FULL capture proved about a pane: a recognized agent name, the pane subtree + * fingerprint at that moment, and node-pty's raw foreground name at that moment. A later cheap + * tick may re-serve `agentName` only while both of the latter still match. + */ +export type SteadyStateAnchor = { + agentName: string + fingerprint: string + rawFallback: string | null +} + +// Weakly keyed: an anchor dies with its Session, and a recycled pid under a new Session can +// never inherit one. Retired sessions fail `isAlive` before any read gets here regardless. +const anchors = new WeakMap() + +export function getSteadyStateAnchor(session: Session): SteadyStateAnchor | null { + return anchors.get(session) ?? null +} + +export function clearSteadyStateAnchor(session: Session): void { + anchors.delete(session) +} + +/** + * Record (or drop) the anchor after a full capture. Only a `live` verdict naming a recognized + * agent establishes one: the cheap tier is licensed by proven identity, never by a fallback name + * or an unverifiable read, so a pane without one always pays for the full capture. + */ +export async function rememberSteadyStateAnchor( + session: Session, + evidence: RemoteForegroundEvidence, + rows: Parameters[0] +): Promise { + if (evidence.verdict !== 'live' || !recognizeAgentProcess(evidence.processName)) { + anchors.delete(session) + return + } + let fingerprint: string | null + try { + fingerprint = await buildPaneProcessFingerprint(rows, session.pid) + } catch { + fingerprint = null + } + if (fingerprint === null || evidence.processName === null) { + anchors.delete(session) + return + } + anchors.set(session, { + agentName: evidence.processName, + fingerprint, + rawFallback: session.getForegroundProcess({ rawFallback: true }) + }) +} diff --git a/src/main/daemon/terminal-host.ts b/src/main/daemon/terminal-host.ts index 18f7b82a90f..95bedd1a7fd 100644 --- a/src/main/daemon/terminal-host.ts +++ b/src/main/daemon/terminal-host.ts @@ -233,7 +233,7 @@ export class TerminalHost { inspectProcess( sessionId: string, - options?: { expectedIncarnationId?: string } + options?: { expectedIncarnationId?: string; steadyState?: boolean } ): Promise { pruneRetiredPtyIncarnations(this.retiredIncarnations) const session = this.sessions.get(sessionId) @@ -253,6 +253,7 @@ export class TerminalHost { ...(options?.expectedIncarnationId ? { expectedIncarnationId: options.expectedIncarnationId } : {}), + ...(options?.steadyState === true ? { steadyState: true } : {}), retiredIncarnation: this.retiredIncarnations.get(sessionId), authorityGeneration: this.authorityGeneration, nextObservationEpoch: () => ++this.observationEpoch diff --git a/src/main/ipc/pty/ipc/inspect.ts b/src/main/ipc/pty/ipc/inspect.ts index 041abaa7854..5a6d03d8855 100644 --- a/src/main/ipc/pty/ipc/inspect.ts +++ b/src/main/ipc/pty/ipc/inspect.ts @@ -172,7 +172,12 @@ export function installPtyInspectIpcHandlers(deps: { 'pty:inspectProcess', async ( _event, - args: { id: string; expectedIncarnationId?: string; scanChildProcesses?: boolean } + args: { + id: string + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ) => { // Why: same routing hazard as pty:hasPty — an unroutable id must read as client-only unverifiable, not as a local-provider answer or a raised IPC error. if (typeof args?.id !== 'string' || !args.id || args.id.startsWith('remote:')) { @@ -189,7 +194,8 @@ export function installPtyInspectIpcHandlers(deps: { ...(args.expectedIncarnationId ? { expectedIncarnationId: args.expectedIncarnationId } : {}), - ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}) + ...(args.scanChildProcesses === true ? { scanChildProcesses: true } : {}), + ...(args.steadyState === true ? { steadyState: true } : {}) } return Object.keys(options).length > 0 ? inspectPtyProviderProcessForRenderer(getProviderForPty(args.id), args.id, options) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts index 6ac0c8e0fbf..226b9c1aab5 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts @@ -49,6 +49,17 @@ export class StructuredAgentSessionAdapterRouter implements StructuredAgentSessi cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => this.owner(input.sessionId).cancelTurn(input) + stopBackgroundTasks: NonNullable = ( + input + ) => { + const stop = this.owner(input.sessionId).stopBackgroundTasks + return stop ? stop(input) : Promise.resolve({ cancelled: false }) + } + + backgroundTaskState: NonNullable = ( + sessionId + ) => this.owners.get(sessionId)?.backgroundTaskState?.(sessionId) + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => this.owner(input.sessionId).answerPrompt(input) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts index 01c16a60e55..e44e8c39152 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts @@ -17,6 +17,7 @@ import type { AgentSessionProcessIdentity } from '../../../shared/agent-session-record' import type { + AgentSessionBackgroundTaskState, AgentSessionOptionsResult, AgentSessionWireRefusalCode } from '../../../shared/agent-session-wire' @@ -136,6 +137,8 @@ export type StructuredAgentSessionAdapter = { turnId: string fence: number }): Promise<{ cancelled: boolean }> + stopBackgroundTasks?(input: { sessionId: string; fence: number }): Promise<{ cancelled: boolean }> + backgroundTaskState?(sessionId: string): AgentSessionBackgroundTaskState | null | undefined /** Fires the provider callback for an approval or a question. The wire calls * this only after the durable compare-and-set won, so it runs exactly once. */ answerPrompt(input: { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts new file mode 100644 index 00000000000..4592dea26e9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-background-task-channel.ts @@ -0,0 +1,62 @@ +import type { + AgentSessionBackgroundTaskState, + AgentSessionHistoryRequest, + AgentSessionHistoryResult +} from '../../../shared/agent-session-wire' +import { readStructuredAgentSessionHistoryResult } from './structured-agent-session-history-result' +import type { + AgentSessionSubscribers, + AgentSessionSubscribeInput +} from './structured-agent-session-subscribers' +import type { + StructuredAgentSessionHostDeps, + StructuredAgentSessionHostSession +} from './structured-agent-session-host-types' + +export class StructuredAgentSessionBackgroundTaskChannel { + constructor( + private readonly deps: StructuredAgentSessionHostDeps, + private readonly sessions: Map, + private readonly subscribers: AgentSessionSubscribers, + private readonly requireSession: (sessionId: string) => StructuredAgentSessionHostSession, + private readonly handoffStatus: ( + sessionId: string + ) => Parameters[0]['handoff'] + ) {} + + history(request: AgentSessionHistoryRequest): AgentSessionHistoryResult { + const result = readStructuredAgentSessionHistoryResult({ + journal: this.requireSession(request.sessionId).journal, + record: this.deps.store.getRecord(request.sessionId), + request + }) + const backgroundTasks = this.state(request.sessionId) + return backgroundTasks === undefined + ? result + : { ...result, page: { ...result.page, backgroundTasks } } + } + + subscribe(input: AgentSessionSubscribeInput): () => void { + const session = this.requireSession(input.sessionId) + const backgroundTasks = this.state(input.sessionId) + return this.subscribers.open({ + ...input, + journal: session.journal, + fence: this.deps.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 0, + handoff: this.handoffStatus(input.sessionId), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + }) + } + + publish(sessionId: string, publishedState?: AgentSessionBackgroundTaskState | null): void { + const session = this.sessions.get(sessionId) + const state = publishedState !== undefined ? publishedState : this.state(sessionId) + if (session && state !== undefined) { + this.subscribers.backgroundTasks(sessionId, state, session.fence) + } + } + + private state(sessionId: string): AgentSessionBackgroundTaskState | null | undefined { + return this.deps.adapter.backgroundTaskState?.(sessionId) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts index c9afa06e525..7d2648930c4 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host-mutations.ts @@ -74,7 +74,11 @@ export function sendStructuredAgentSessionTurn( export function cancelStructuredAgentSessionTurn( context: StructuredAgentSessionMutationContext, caller: StructuredAgentSessionCaller, - params: { envelope: AgentSessionMutationEnvelope; turnId: string } + params: { + envelope: AgentSessionMutationEnvelope + turnId: string + scope?: 'background-tasks' + } ): Promise> { return mutate(context, caller, params.envelope, cancelPlan(params)) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index c38f64c3318..08ff464e8fc 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -57,8 +57,8 @@ import type { StructuredAgentSessionHostDeps, StructuredAgentSessionHostSession } from './structured-agent-session-host-types' -import { readStructuredAgentSessionHistoryResult } from './structured-agent-session-history-result' import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery' +import { StructuredAgentSessionBackgroundTaskChannel } from './structured-agent-session-background-task-channel' export type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' export class StructuredAgentSessionHost { private readonly sessions = new Map() @@ -71,8 +71,16 @@ export class StructuredAgentSessionHost { private readonly restartRestore = new StructuredAgentSessionRestartRestoreGate() private readonly holds: StructuredAgentSessionHolds private readonly eventRecovery: StructuredAgentSessionEventRecovery + private readonly backgroundTasks: StructuredAgentSessionBackgroundTaskChannel constructor(readonly deps: StructuredAgentSessionHostDeps) { + this.backgroundTasks = new StructuredAgentSessionBackgroundTaskChannel( + deps, + this.sessions, + this.subscribers, + (sessionId) => this.requireSession(sessionId), + (sessionId) => this.handoffs.status(sessionId) + ) this.runtimeState = new StructuredAgentSessionHostRuntimeState( deps, (record) => this.restoreRenewedHandoff(record.sessionId), @@ -308,24 +316,16 @@ export class StructuredAgentSessionHost { ) } - history(request: AgentSessionHistoryRequest): AgentSessionHistoryResult { - return readStructuredAgentSessionHistoryResult({ - journal: this.requireSession(request.sessionId).journal, - record: this.deps.store.getRecord(request.sessionId), - request - }) - } + history = (request: AgentSessionHistoryRequest): AgentSessionHistoryResult => + this.backgroundTasks.history(request) - subscribe(input: AgentSessionSubscribeInput): () => void { - const session = this.requireSession(input.sessionId) - const fence = this.deps.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 0 - return this.subscribers.open({ - ...input, - journal: session.journal, - fence, - handoff: this.handoffs.status(input.sessionId) - }) - } + subscribe = (input: AgentSessionSubscribeInput): (() => void) => + this.backgroundTasks.subscribe(input) + + publishBackgroundTaskState: StructuredAgentSessionBackgroundTaskChannel['publish'] = ( + sessionId, + state + ) => this.backgroundTasks.publish(sessionId, state) unsubscribe = (sessionId: string, id: string): void => this.subscribers.close(sessionId, id) private requireSession(sessionId: string): StructuredAgentSessionHostSession { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts index 99835da7cea..d0eb905443c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-mutation-plans.ts @@ -75,14 +75,16 @@ export function sendPlan(params: { export function cancelPlan(params: { envelope: AgentSessionMutationEnvelope turnId: string + scope?: 'background-tasks' }): MutationPlan { return { method: 'agentSession.cancel', - fields: { turnId: params.turnId }, + fields: { turnId: params.turnId, ...(params.scope ? { scope: params.scope } : {}) }, run: (ctx) => performCancel(ctx, { clientOperationId: params.envelope.clientOperationId, - turnId: params.turnId + turnId: params.turnId, + ...(params.scope ? { scope: params.scope } : {}) }), // Interrupting twice would kill a turn the client never asked to stop, so a // replay reports the turn as already handled instead. diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts index 6b627726f98..6e156d91532 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.test.ts @@ -114,6 +114,54 @@ describe('AgentSessionSubscribers', () => { }) }) + it('publishes background lifecycle without advancing the journal and carries its fence forward', async () => { + const journal = await journals.open({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'provider-1', leafUuid: null } + }, + journalDir: join(root, 'background-journal') + }) + const subscribers = new AgentSessionSubscribers() + const events: AgentSessionSubscribeEvent[] = [] + subscribers.open({ + id: 'subscriber-1', + sessionId: SESSION, + journal, + fence: 1, + backgroundTasks: null, + emit: (event) => events.push(event) + }) + const cursor = journal.cursor() + + const backgroundTasks = { + state: 'monitoring' as const, + tasks: [{ id: 'task-1', kind: 'command' as const, description: 'run the build' }] + } + subscribers.backgroundTasks(SESSION, backgroundTasks, 2) + + expect(journal.cursor()).toEqual(cursor) + expect(events.at(-1)).toEqual({ + type: 'batch', + sessionId: SESSION, + batch: { cursor, items: [], removedItemIds: [], submissions: [] }, + fence: 2, + backgroundTasks + }) + + await journal.appendItem( + { provider: 'orca', clientMessageId: 'after-background-fence' }, + { kind: 'status', text: 'After background state' }, + { fence: 2 } + ) + subscribers.publish(SESSION, journal) + + expect(events.at(-1)).toMatchObject({ type: 'batch', fence: 2 }) + }) + it('catches a subscriber up past a pre-existing unsendable removal with a bounded reset', async () => { const journalDir = join(root, 'oversized-removal-journal') const seeded = await journals.open({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts index 3fa80b28d85..d3131505384 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-subscribers.ts @@ -10,6 +10,7 @@ import type { } from '../../../shared/agent-session-journal-types' import { AGENT_SESSION_HISTORY_MAX_LIMIT, + type AgentSessionBackgroundTaskState, type AgentSessionHandoffStatus, type AgentSessionSubscribeEvent } from '../../../shared/agent-session-wire' @@ -48,6 +49,7 @@ export class AgentSessionSubscribers { emit: AgentSessionSubscriberEmit cursor?: AgentJournalCursor handoff?: AgentSessionHandoffStatus + backgroundTasks?: AgentSessionBackgroundTaskState | null }): () => void { const liveCursor = input.journal.cursor() const subscriber: Subscriber = { @@ -62,7 +64,7 @@ export class AgentSessionSubscribers { this.bySession.set(input.sessionId, session) if (input.cursor) { - this.deliver(subscriber, input.journal, input.handoff, true) + this.deliver(subscriber, input.journal, input.handoff, true, input.backgroundTasks) } else { const page = readAgentSessionHydrationPage(input.journal, input.fence) this.emit(subscriber, { @@ -70,7 +72,8 @@ export class AgentSessionSubscribers { sessionId: input.sessionId, page, fence: input.fence, - ...(input.handoff ? { handoff: input.handoff } : {}) + ...(input.handoff ? { handoff: input.handoff } : {}), + ...(input.backgroundTasks !== undefined ? { backgroundTasks: input.backgroundTasks } : {}) }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor } @@ -104,20 +107,39 @@ export class AgentSessionSubscribers { sessionId: string, journal: AgentSessionJournal, reason: AgentJournalResetReason, - fence: number + fence: number, + backgroundTasks?: AgentSessionBackgroundTaskState | null ): void { const page = readAgentSessionHydrationPage(journal, fence) for (const subscriber of this.subscribers(sessionId)) { - this.emit(subscriber, { type: 'reset', sessionId, reset: reason, page, fence }) + this.emit(subscriber, { + type: 'reset', + sessionId, + reset: reason, + page, + fence, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor subscriber.fence = fence } } - snapshot(sessionId: string, journal: AgentSessionJournal, fence: number): void { + snapshot( + sessionId: string, + journal: AgentSessionJournal, + fence: number, + backgroundTasks?: AgentSessionBackgroundTaskState | null + ): void { const page = readAgentSessionHydrationPage(journal, fence) for (const subscriber of this.subscribers(sessionId)) { - this.emit(subscriber, { type: 'snapshot', sessionId, page, fence }) + this.emit(subscriber, { + type: 'snapshot', + sessionId, + page, + fence, + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) + }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor subscriber.fence = fence } @@ -141,6 +163,28 @@ export class AgentSessionSubscribers { } } + backgroundTasks( + sessionId: string, + state: AgentSessionBackgroundTaskState | null, + fence: number + ): void { + for (const subscriber of this.subscribers(sessionId)) { + this.emit(subscriber, { + type: 'batch', + sessionId, + batch: { + cursor: subscriber.cursor, + items: [], + removedItemIds: [], + submissions: [] + }, + fence, + backgroundTasks: state + }) + subscriber.fence = fence + } + } + private subscribers(sessionId: string): Subscriber[] { return [...(this.bySession.get(sessionId)?.values() ?? [])] } @@ -149,7 +193,8 @@ export class AgentSessionSubscribers { subscriber: Subscriber, journal: AgentSessionJournal, handoff?: AgentSessionHandoffStatus, - emitCheckpoint = false + emitCheckpoint = false, + backgroundTasks?: AgentSessionBackgroundTaskState | null ): void { while (true) { const result = readAgentSessionHistory(journal, { @@ -166,7 +211,8 @@ export class AgentSessionSubscribers { reset: result.reset, page, fence: subscriber.fence, - ...(handoff ? { handoff } : {}) + ...(handoff ? { handoff } : {}), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) }) subscriber.cursor = page.liveCursor ?? page.window.nextCursor return @@ -185,7 +231,8 @@ export class AgentSessionSubscribers { submissions: [] }, fence: subscriber.fence, - ...(handoff ? { handoff } : {}) + ...(handoff ? { handoff } : {}), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) }) } return @@ -200,7 +247,8 @@ export class AgentSessionSubscribers { submissions: page.submissions }, fence: subscriber.fence, - ...(handoff ? { handoff } : {}) + ...(handoff ? { handoff } : {}), + ...(backgroundTasks !== undefined ? { backgroundTasks } : {}) }) subscriber.cursor = page.window.nextCursor if (!page.hasNewer || !this.isActive(subscriber)) { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts index 5222f9557f9..e8e6f998bdd 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.test.ts @@ -73,4 +73,35 @@ describe('performCancel', () => { { kind: 'status', text: 'Cancellation requested.' } ]) }) + + it('stops background tasks without interrupting the foreground turn or writing a row', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-background-task-cancel-')) + const journal = await journals.open({ identity: IDENTITY, journalDir: root }) + const cancelTurn = vi.fn(async () => ({ cancelled: true })) + const stopBackgroundTasks = vi.fn(async () => ({ cancelled: true })) + const ctx: AgentSessionTurnContext = { + sessionId: 'session-1', + journal, + fence: 1, + adapter: { cancelTurn, stopBackgroundTasks } as unknown as StructuredAgentSessionAdapter, + persistOptions: async () => undefined, + resolvedBy: 'client-1', + publish: vi.fn(), + now: () => 1 + } + + const result = await performCancel(ctx, { + clientOperationId: 'cancel-background-tasks', + turnId: 'background-tasks', + scope: 'background-tasks' + }) + + expect(result).toEqual({ + ok: true, + value: { turnId: 'background-tasks', cancelled: true } + }) + expect(stopBackgroundTasks).toHaveBeenCalledWith({ sessionId: 'session-1', fence: 1 }) + expect(cancelTurn).not.toHaveBeenCalled() + expect(journal.snapshot().items).toEqual([]) + }) }) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts index f1717027b8b..76c4e8e89d6 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns.ts @@ -146,18 +146,29 @@ export async function performSend( export async function performCancel( ctx: AgentSessionTurnContext, - input: { clientOperationId: string; turnId: string } + input: { + clientOperationId: string + turnId: string + scope?: 'background-tasks' + } ): Promise> { let cancelled = false let note = 'Cancellation requested.' try { - cancelled = ( - await ctx.adapter.cancelTurn({ - sessionId: ctx.sessionId, - turnId: input.turnId, - fence: ctx.fence - }) - ).cancelled + cancelled = input.scope + ? ( + await ctx.adapter.stopBackgroundTasks?.({ + sessionId: ctx.sessionId, + fence: ctx.fence + }) + )?.cancelled === true + : ( + await ctx.adapter.cancelTurn({ + sessionId: ctx.sessionId, + turnId: input.turnId, + fence: ctx.fence + }) + ).cancelled if (!cancelled) { note = 'The provider had already finished this turn.' } @@ -166,6 +177,9 @@ export async function performCancel( error instanceof Error ? error.message : String(error) }` } + if (input.scope) { + return { ok: true, value: { turnId: input.turnId, cancelled } } + } // Keyed by the operation id so a replayed cancel upserts one item, not two. await appendStatus(ctx, input.clientOperationId, note) return { ok: true, value: { turnId: input.turnId, cancelled } } diff --git a/src/main/native-chat/transcript-line-decoders-claude.ts b/src/main/native-chat/transcript-line-decoders-claude.ts index f819656bdb2..5202035bbc6 100644 --- a/src/main/native-chat/transcript-line-decoders-claude.ts +++ b/src/main/native-chat/transcript-line-decoders-claude.ts @@ -3,6 +3,8 @@ import { NATIVE_CHAT_INTERRUPTED_STATUS_TEXT, type NativeChatBlock, + type NativeChatEditPatch, + type NativeChatEditPatchHunk, type NativeChatMessage } from '../../shared/native-chat-types' import { @@ -15,6 +17,59 @@ import { imageSourcePathFromText } from '../../shared/native-chat-image-transcri import { claudeContentBlocks } from './transcript-record-blocks' import { claudeInterruptedMessageId } from './transcript-turn-markers' +const MAX_EDIT_PATCH_HUNKS = 40 +const MAX_EDIT_PATCH_HUNK_LINES = 400 + +/** Claude reports an edit as a snippet pair on the call, which cannot locate the + * change in the file. The result record carries the hunks it resolved against + * the real file, so keep them for the renderer's line-number gutter. */ +function claudeEditPatch(record: Record): NativeChatEditPatch | null { + const result = asRecord(record.toolUseResult) + const raw = result?.structuredPatch + if (!Array.isArray(raw) || raw.length === 0) { + return null + } + const hunks: NativeChatEditPatchHunk[] = [] + for (const entry of raw.slice(0, MAX_EDIT_PATCH_HUNKS)) { + const hunk = asRecord(entry) + const lines = hunk?.lines + if ( + typeof hunk?.oldStart !== 'number' || + typeof hunk.newStart !== 'number' || + !Array.isArray(lines) + ) { + continue + } + hunks.push({ + oldStart: hunk.oldStart, + oldLines: typeof hunk.oldLines === 'number' ? hunk.oldLines : 0, + newStart: hunk.newStart, + newLines: typeof hunk.newLines === 'number' ? hunk.newLines : 0, + lines: lines + .slice(0, MAX_EDIT_PATCH_HUNK_LINES) + .flatMap((line) => (typeof line === 'string' ? [line] : [])) + }) + } + if (hunks.length === 0) { + return null + } + const filePath = extractString(result?.filePath) + return { ...(filePath ? { filePath } : {}), hunks } +} + +/** Attaches the resolved hunks to the record's tool result, which is the only + * block in a Claude result turn. */ +function withEditPatch(blocks: NativeChatBlock[], patch: NativeChatEditPatch): NativeChatBlock[] { + let attached = false + return blocks.map((block) => { + if (attached || block.type !== 'tool-result') { + return block + } + attached = true + return { ...block, editPatch: patch } + }) +} + export function decodeClaudeTranscriptLine( line: string, fallbackId: string @@ -41,7 +96,9 @@ export function decodeClaudeTranscriptLine( } } const message = asRecord(record.message) - const decodedBlocks = claudeContentBlocks(message?.content) + const editPatch = claudeEditPatch(record) + const contentBlocks = claudeContentBlocks(message?.content) + const decodedBlocks = editPatch ? withEditPatch(contentBlocks, editPatch) : contentBlocks if (decodedBlocks.length === 0) { return null } diff --git a/src/main/native-chat/transcript-line-decoders-codex.ts b/src/main/native-chat/transcript-line-decoders-codex.ts index c4b0ddbf5ba..d70bd7a7c80 100644 --- a/src/main/native-chat/transcript-line-decoders-codex.ts +++ b/src/main/native-chat/transcript-line-decoders-codex.ts @@ -202,6 +202,10 @@ function codexTurnItemBlocks(content: unknown): NativeChatBlock[] { return blocks } +/** The argument payload is passed through exactly as it arrived. Decoding it + * here would change the shape every `.input` consumer sees — including the ask + * surface, which reads a question shape out of any tool's input — so the one + * consumer that needs structure decodes it for itself. */ function codexCallInput(payload: Record): unknown { if (payload.arguments !== undefined) { return payload.arguments diff --git a/src/main/native-chat/transcript-reader-codex-history-mode.test.ts b/src/main/native-chat/transcript-reader-codex-history-mode.test.ts index 5b0f3dc014f..5d18bec8c85 100644 --- a/src/main/native-chat/transcript-reader-codex-history-mode.test.ts +++ b/src/main/native-chat/transcript-reader-codex-history-mode.test.ts @@ -2,6 +2,7 @@ import { mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { join } from 'node:path' import { afterEach, describe, expect, it } from 'vitest' +import { extractPendingAsk } from '../../shared/native-chat-ask' import { decodeCodexTranscriptLine } from './transcript-line-decoders-codex' import { readNativeChatTranscript } from './transcript-reader' import { readNativeChatTranscriptTail } from './transcript-tail-reader' @@ -247,4 +248,43 @@ describe('Codex transcript history modes', () => { blocks: [{ type: 'tool-result', output: 'ok' }] }) }) + + it('passes an argument payload through untouched, so no consumer changes shape', () => { + const call = decodeCodexTranscriptLine( + JSON.stringify({ + type: 'response_item', + payload: { + type: 'function_call', + id: 'call-2', + name: 'shell', + arguments: '{"command":["bash","-lc","echo hi"]}' + } + }), + 'fallback-args' + ) + + expect(call?.blocks[0]).toMatchObject({ + type: 'tool-call', + name: 'shell', + input: '{"command":["bash","-lc","echo hi"]}' + }) + }) + + it('does not raise a question card from an unrelated tool that carries a questions payload', () => { + const call = decodeCodexTranscriptLine( + JSON.stringify({ + type: 'response_item', + payload: { + type: 'function_call', + id: 'call-3', + name: 'some_mcp_tool', + arguments: '{"questions":[{"question":"Which branch?","options":["main","dev"]}]}' + } + }), + 'fallback-questions' + ) + + expect(call).not.toBeNull() + expect(extractPendingAsk(call ? [call] : [])).toBeNull() + }) }) diff --git a/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts b/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts index c0359efd612..2c73045d901 100644 --- a/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts +++ b/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from 'vitest' import { makePaneKey } from '../../../shared/stable-pane-id' import { + remapAcknowledgedAgentPaneKeys, remapActivityClearedAtPaneKeys, remapManuallyUnreadTurnPaneKeys } from './pane-key-remapping' @@ -30,4 +31,16 @@ describe('remapManuallyUnreadTurnPaneKeys', () => { changed: false }) }) + + it('returns the caller map untouched when no key needs remapping', () => { + const remap = new Map([['tab-1', new Map([[STABLE_LEAF_ID, STABLE_LEAF_ID]])]]) + const stable = { [makePaneKey('tab-1', STABLE_LEAF_ID)]: 1 } + + const result = remapAcknowledgedAgentPaneKeys(stable, remap) + + // Why identity and not just equality: this runs on every session write against a map that + // grows with every pane ever opened, so a rebuilt-then-discarded copy is pure garbage. + expect(result.acknowledgements).toBe(stable) + expect(result.changed).toBe(false) + }) }) diff --git a/src/main/persistence/restoring-sessions/pane-key-remapping.ts b/src/main/persistence/restoring-sessions/pane-key-remapping.ts index 4f3440f087e..8e43184b0c5 100644 --- a/src/main/persistence/restoring-sessions/pane-key-remapping.ts +++ b/src/main/persistence/restoring-sessions/pane-key-remapping.ts @@ -3,51 +3,50 @@ import { isTerminalLeafId, makePaneKey, parsePaneKey } from '../../../shared/sta type PaneLeafRemap = Map> +/** Resolves the pane key a legacy entry should move to, or `null` when it stays put. */ +function resolveRemappedPaneKey( + paneKey: string, + leafIdByInputLeafIdByTabId: PaneLeafRemap +): string | null { + if (parsePaneKey(paneKey)) { + return null + } + const delimiter = paneKey.indexOf(':') + if (delimiter <= 0 || delimiter === paneKey.length - 1) { + return null + } + const tabId = paneKey.slice(0, delimiter) + const remappedLeafId = leafIdByInputLeafIdByTabId.get(tabId)?.get(paneKey.slice(delimiter + 1)) + // makePaneKey cannot throw here: tabId is non-empty and colon-free by construction. + return remappedLeafId && isTerminalLeafId(remappedLeafId) + ? makePaneKey(tabId, remappedLeafId) + : null +} + function remapPaneKeys( values: Record | undefined, leafIdByInputLeafIdByTabId: PaneLeafRemap ): { values: Record | undefined; changed: boolean } { - if (!values || Object.keys(values).length === 0) { + // Why the classify-first pass: these maps grow with every pane ever opened and this runs on + // every session write, but post-migration no key is ever rewritten. Rebuilding the whole + // object only to discard it was pure garbage; the rewrite below is unchanged. + if ( + !values || + !Object.keys(values).some( + (paneKey) => resolveRemappedPaneKey(paneKey, leafIdByInputLeafIdByTabId) !== null + ) + ) { return { values, changed: false } } - let changed = false const next: Record = {} - const setValue = (paneKey: string, value: T): void => { - const existing = next[paneKey] - next[paneKey] = existing === undefined ? value : (Math.max(existing, value) as T) - } for (const [paneKey, value] of Object.entries(values)) { - const parsed = parsePaneKey(paneKey) - if (parsed) { - setValue(paneKey, value) - continue - } - - const delimiter = paneKey.indexOf(':') - if (delimiter <= 0 || delimiter === paneKey.length - 1) { - setValue(paneKey, value) - continue - } - - const tabId = paneKey.slice(0, delimiter) - const legacyLeafId = paneKey.slice(delimiter + 1) - const remappedLeafId = leafIdByInputLeafIdByTabId.get(tabId)?.get(legacyLeafId) - if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { - setValue(paneKey, value) - continue - } - - try { - // Carry values over when a legacy leaf is promoted to a UUID. - setValue(makePaneKey(tabId, remappedLeafId), value) - changed = true - } catch { - setValue(paneKey, value) - } + // Carry values over when a legacy leaf is promoted to a UUID; keep the max on collision. + const target = resolveRemappedPaneKey(paneKey, leafIdByInputLeafIdByTabId) ?? paneKey + const existing = next[target] + next[target] = existing === undefined ? value : (Math.max(existing, value) as T) } - - return { values: next, changed } + return { values: next, changed: true } } export function remapAcknowledgedAgentPaneKeys( diff --git a/src/main/providers/local-pty-finalize-environment.ts b/src/main/providers/local-pty-finalize-environment.ts index 0f7d3454a09..1b385639e94 100644 --- a/src/main/providers/local-pty-finalize-environment.ts +++ b/src/main/providers/local-pty-finalize-environment.ts @@ -109,6 +109,9 @@ export function finalizeLocalPtySpawnEnvironment(args: { codexStartupCommand !== undefined && supportsPosixShellStartupCommand(shell) ? codexStartupCommand : undefined + // Why no line-editor widening here (unlike the daemon and relay): a Codex + // startup command this provider wraps is run by the wrapper's own prompt + // hook, never written into the PTY, so there is no early write to double-echo. const waitsForShellReady = Boolean(spawn.command) && (!isCodexStartupCommand || codexRequiresShellReady) return getShellLaunchConfig( diff --git a/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts b/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts new file mode 100644 index 00000000000..37d4cf3c4d0 --- /dev/null +++ b/src/main/providers/local-pty-foreground-inspection-cheap-tier.test.ts @@ -0,0 +1,138 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as ProcessTableSnapshotReader from '../../shared/process-table-snapshot-reader' + +const { cheapSnapshotMock, fullSnapshotMock, resolveMock } = vi.hoisted(() => ({ + cheapSnapshotMock: vi.fn(), + fullSnapshotMock: vi.fn(), + resolveMock: vi.fn() +})) + +vi.mock('../../shared/cheap-process-table-snapshot-reader', () => ({ + getCheapProcessTableSnapshot: cheapSnapshotMock +})) +vi.mock('../../shared/process-table-snapshot-reader', async (importOriginal) => ({ + ...(await importOriginal()), + getProcessTableSnapshot: fullSnapshotMock +})) +vi.mock('./agent-foreground-process', () => ({ + resolveAgentForegroundProcessWithAvailability: resolveMock, + confirmShellForegroundProcess: vi.fn() +})) + +import { getLocalPtyForegroundProcess } from './local-pty-foreground-inspection' +import { ptyLastRecognizedForeground, ptyProcesses, ptyShellName } from './local-pty-provider-state' + +const SHELL_PID = 4242 +const AGENT_PID = 4300 +const ID = 'pty-1' + +type Table = 'agent' | 'shell-only' +let table: Table = 'agent' + +function rows(): Record[] { + const tpgid = table === 'agent' ? AGENT_PID : SHELL_PID + const out: Record[] = [ + { + pid: SHELL_PID, + ppid: 1, + pgid: SHELL_PID, + tpgid, + stat: table === 'agent' ? 'Ss' : 'Ss+', + tty: 'ttys004', + startTime: 'Thu Sep 3 16:02:01 2026', + command: '-zsh' + } + ] + if (table === 'agent') { + out.push({ + pid: AGENT_PID, + ppid: SHELL_PID, + pgid: AGENT_PID, + tpgid, + stat: 'S+', + tty: 'ttys004', + startTime: 'Thu Sep 3 16:02:05 2026', + command: 'node /usr/local/bin/claude' + }) + } + return out +} + +describe('local POSIX provider cheap-tier revalidation', () => { + let platform: PropertyDescriptor | undefined + const proc = { pid: SHELL_PID, process: 'node' } + + beforeEach(() => { + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'darwin' }) + table = 'agent' + proc.process = 'node' + cheapSnapshotMock.mockReset() + cheapSnapshotMock.mockImplementation(async () => rows()) + fullSnapshotMock.mockReset() + fullSnapshotMock.mockImplementation(async () => rows()) + resolveMock.mockReset() + resolveMock.mockImplementation(async () => ({ + available: true, + processName: table === 'agent' ? 'claude' : 'zsh' + })) + ptyProcesses.set(ID, proc as never) + ptyShellName.set(ID, 'zsh') + ptyLastRecognizedForeground.delete(ID) + }) + + afterEach(() => { + ptyProcesses.delete(ID) + ptyShellName.delete(ID) + ptyLastRecognizedForeground.delete(ID) + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + it('a pane with NO recognized anchor never consults the cheap tier', async () => { + table = 'shell-only' + proc.process = 'zsh' + for (let i = 0; i < 3; i += 1) { + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + } + expect(cheapSnapshotMock).not.toHaveBeenCalled() + expect(resolveMock).toHaveBeenCalledTimes(3) + expect(ptyLastRecognizedForeground.get(ID)).toBeUndefined() + }) + + it('once recognized, an unchanged pane re-proves the agent from the cheap tier without a full scan', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + expect(resolveMock).toHaveBeenCalledTimes(1) + expect(ptyLastRecognizedForeground.get(ID)?.steady?.fingerprint).toEqual(expect.any(String)) + for (let i = 0; i < 3; i += 1) { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + } + expect(cheapSnapshotMock).toHaveBeenCalledTimes(3) + expect(resolveMock).toHaveBeenCalledTimes(1) + }) + + it('an agent exit changes the fingerprint, escalates to the full scan, and clears the anchor', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + table = 'shell-only' + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + expect(resolveMock).toHaveBeenCalledTimes(2) + expect(ptyLastRecognizedForeground.get(ID)).toBeUndefined() + }) + + it('a changed node-pty foreground name escalates without consulting the cheap tier', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + proc.process = 'zsh' + table = 'shell-only' + expect(await getLocalPtyForegroundProcess(ID)).toBe('zsh') + expect(cheapSnapshotMock).not.toHaveBeenCalled() + expect(resolveMock).toHaveBeenCalledTimes(2) + }) + + it('a cheap capture failure falls through to the full scan', async () => { + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + cheapSnapshotMock.mockRejectedValueOnce(new Error('ps died')) + expect(await getLocalPtyForegroundProcess(ID)).toBe('claude') + expect(resolveMock).toHaveBeenCalledTimes(2) + }) +}) diff --git a/src/main/providers/local-pty-foreground-inspection.ts b/src/main/providers/local-pty-foreground-inspection.ts index c42c06d9121..d4a717a9de2 100644 --- a/src/main/providers/local-pty-foreground-inspection.ts +++ b/src/main/providers/local-pty-foreground-inspection.ts @@ -1,8 +1,11 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' +import { getCheapProcessTableSnapshot } from '../../shared/cheap-process-table-snapshot-reader' +import { getProcessTableSnapshot } from '../../shared/process-table-snapshot-reader' import { confirmShellForegroundProcess, resolveAgentForegroundProcessWithAvailability } from './agent-foreground-process' +import { buildPaneProcessFingerprint } from './posix-pane-foreground-fingerprint' import { resolveForegroundFallbackProcess } from './local-pty-launch-helpers' import { ptyAgentForegroundContextPaths, @@ -35,6 +38,34 @@ export async function hasLocalPtyChildProcesses(id: string): Promise { } } +/** + * POSIX twin of the Windows job-membership short-circuit below: a pane that already holds a + * recognized agent re-proves it from the cheap `ps` tier when the subtree fingerprint is + * unchanged. Panes with no anchor never get here, so start discovery is untouched. + */ +async function revalidateCachedPosixAgent( + proc: { pid: number }, + cachedEntry: { + name: string + steady?: { fingerprint: string; fallbackProcess: string | null } | null + }, + fallbackProcess: string | null +): Promise { + const steady = cachedEntry.steady + if (!steady || steady.fallbackProcess !== fallbackProcess) { + return false + } + try { + const observed = await buildPaneProcessFingerprint( + await getCheapProcessTableSnapshot(), + proc.pid + ) + return observed !== null && observed === steady.fingerprint + } catch { + return false + } +} + export async function getLocalPtyForegroundProcess(id: string): Promise { const proc = ptyProcesses.get(id) if (!proc) { @@ -89,6 +120,18 @@ export async function getLocalPtyForegroundProcess(id: string): Promise { + if (process.platform === 'win32') { + return null + } + try { + const fingerprint = await buildPaneProcessFingerprint(await getProcessTableSnapshot(), shellPid) + return fingerprint === null ? null : { fingerprint, fallbackProcess } + } catch { + return null + } +} + export async function confirmLocalPtyForegroundProcess(id: string): Promise { const proc = ptyProcesses.get(id) if (!proc) { diff --git a/src/main/providers/local-pty-provider-state.ts b/src/main/providers/local-pty-provider-state.ts index 5e87502b1f6..9e54ac859b0 100644 --- a/src/main/providers/local-pty-provider-state.ts +++ b/src/main/providers/local-pty-provider-state.ts @@ -43,9 +43,16 @@ export const ptyAgentForegroundContextPaths = new Map() // Why: remember the last recognized agent foreground so a degraded scan doesn't report the shell and look like an exit. // `pid` anchors the identity to the row that proved it (null when ambiguous); // `at` is the last confirmation, so unanchored job evidence -- only a superset -- cannot hold it forever. +// `steady` (POSIX) is the pane fingerprint the recognizing capture proved plus node-pty's name at +// that moment; a cheap capture matching it re-proves the identity without the full table. export const ptyLastRecognizedForeground = new Map< string, - { name: string; pid: number | null; at: number } + { + name: string + pid: number | null + at: number + steady?: { fingerprint: string; fallbackProcess: string | null } | null + } >() export const ptyTerminalHandle = new Map() export const ptyWorktreeId = new Map() diff --git a/src/main/providers/posix-pane-foreground-fingerprint.test.ts b/src/main/providers/posix-pane-foreground-fingerprint.test.ts new file mode 100644 index 00000000000..76cb68f79dc --- /dev/null +++ b/src/main/providers/posix-pane-foreground-fingerprint.test.ts @@ -0,0 +1,187 @@ +import { describe, expect, it } from 'vitest' +import { + buildPaneProcessFingerprint, + type PaneFingerprintRow +} from './posix-pane-foreground-fingerprint' + +const SHELL = 4242 +const AGENT = 4300 +const OTHER_PANE = 9000 + +type Row = PaneFingerprintRow + +const shell = (over: Partial = {}): Row => ({ + pid: SHELL, + ppid: 1, + pgid: SHELL, + tpgid: AGENT, + stat: 'Ss', + startTime: 'Thu Sep 3 16:02:01 2026', + ...over +}) +const agent = (over: Partial = {}): Row => ({ + pid: AGENT, + ppid: SHELL, + pgid: AGENT, + tpgid: AGENT, + stat: 'S+', + startTime: 'Thu Sep 3 16:02:05 2026', + ...over +}) +const child = (pid: number, ppid: number, over: Partial = {}): Row => ({ + pid, + ppid, + pgid: AGENT, + tpgid: AGENT, + stat: 'S+', + startTime: `Thu Sep 3 16:03:${String(pid % 60).padStart(2, '0')} 2026`, + ...over +}) +const foreign = (): Row => ({ + pid: OTHER_PANE, + ppid: 1, + pgid: OTHER_PANE, + tpgid: OTHER_PANE, + stat: 'Ss+', + startTime: 'Thu Sep 3 12:00:00 2026' +}) + +const fp = (rows: Row[]): Promise => + buildPaneProcessFingerprint(rows, SHELL, { platform: 'darwin' }) + +describe('buildPaneProcessFingerprint', () => { + const baseline = [foreign(), shell(), agent()] + + it('is stable across captures that differ only in scheduler state, row order, and foreign panes', async () => { + const a = await fp(baseline) + expect(a).not.toBeNull() + // R vs S: a working agent flips this every tick and it says nothing about the pane. + expect(await fp([agent({ stat: 'R+' }), shell({ stat: 'Ss' }), foreign()])).toBe(a) + // The shell going idle-vs-runnable, or a foreign pane starting/exiting, is not our business. + expect(await fp([shell({ stat: 'Rs' }), agent()])).toBe(a) + // lstart padding differs between column sets; both must stamp identically. + expect(await fp([shell({ startTime: 'Thu Sep 3 16:02:01 2026' }), agent()])).toBe(a) + }) + + describe('escalates (fingerprint changes) on every completion-relevant transition', () => { + it('agent exit: the recognized pid vanishes from the subtree', async () => { + const before = await fp(baseline) + expect(await fp([foreign(), shell({ tpgid: SHELL, stat: 'Ss+' })])).not.toBe(before) + }) + + it('exit-and-replace: the same pid is reused by a new process with a new start time', async () => { + const before = await fp(baseline) + expect(await fp([shell(), agent({ startTime: 'Thu Sep 3 16:09:00 2026' })])).not.toBe(before) + }) + + it('Ctrl-Z: the agent stops and the shell takes the terminal back', async () => { + const before = await fp(baseline) + expect(await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'T' })])).not.toBe( + before + ) + }) + + it('bg: the stopped job resumes in the background, foreground stays with the shell', async () => { + const stopped = await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'T' })]) + const backgrounded = await fp([shell({ tpgid: SHELL, stat: 'Ss+' }), agent({ stat: 'S' })]) + expect(backgrounded).not.toBe(stopped) + expect(backgrounded).not.toBe(await fp(baseline)) + }) + + it('child churn: a subprocess appearing or disappearing under the agent', async () => { + const before = await fp(baseline) + const withChild = await fp([shell(), agent(), child(4310, AGENT)]) + expect(withChild).not.toBe(before) + expect(await fp([shell(), agent(), child(4310, AGENT), child(4311, 4310)])).not.toBe( + withChild + ) + // A child exec'ing away from the group (setsid / disown) is also a change. + expect(await fp([shell(), agent(), child(4310, AGENT, { pgid: 4310 })])).not.toBe(withChild) + }) + + it('shell replaced: same pid, different start time', async () => { + const before = await fp(baseline) + expect(await fp([shell({ startTime: 'Thu Sep 3 17:00:00 2026' }), agent()])).not.toBe(before) + }) + }) + + describe('refuses to fingerprint an unfenced pane (caller must take the full capture)', () => { + it('root shell missing from the capture', async () => { + expect(await fp([foreign(), agent()])).toBeNull() + }) + + it('root shell has no start marker', async () => { + expect(await fp([shell({ startTime: undefined }), agent()])).toBeNull() + }) + + it('root shell has no job-control columns', async () => { + expect(await fp([shell({ pgid: undefined, tpgid: undefined }), agent()])).toBeNull() + }) + }) + + describe('Linux', () => { + it('reads /proc start times for the pane subtree only and ignores ps start markers', async () => { + const asked: number[] = [] + const read = async (pid: number): Promise => { + asked.push(pid) + return pid === SHELL ? '1000' : pid === AGENT ? '2000' : null + } + const rows = [foreign(), shell({ startTime: undefined }), agent({ startTime: undefined })] + const a = await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: read + }) + expect(a).toContain(`${SHELL}@1000`) + expect(a).toContain(`${AGENT}@2000`) + expect(asked.sort()).toEqual([SHELL, AGENT].sort()) + // An exit-and-replace changes only the /proc start ticks. + const replaced = await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: async (pid) => (pid === AGENT ? '2500' : read(pid)) + }) + expect(replaced).not.toBe(a) + }) + + it('refuses when the root /proc entry cannot be read', async () => { + expect( + await buildPaneProcessFingerprint([shell(), agent()], SHELL, { + platform: 'linux', + readLinuxStartTime: async () => null + }) + ).toBeNull() + }) + it('refuses when a DESCENDANT start marker cannot be read', async () => { + // Why: the start marker is what makes a pid comparison recycle-safe. Stamping a missing + // one as a placeholder let two captures that both failed to read it compare equal across + // a recycled pid, so a vanished agent looked unchanged and the cheap tier kept serving + // its name. Refusing sends the caller to the full capture. + const rows = [shell(), agent()] + + expect( + await buildPaneProcessFingerprint(rows, SHELL, { + platform: 'linux', + readLinuxStartTime: async (pid) => (pid === SHELL ? '2400' : null) + }) + ).toBeNull() + }) + + it('does not let a recycled descendant pid reuse a fingerprint', async () => { + // Both captures fail to read the descendant marker; the pid is reused by a different + // process in between. Equal fingerprints here would mask the agent's exit. + const readNoDescendant = async (pid: number): Promise => + pid === SHELL ? '2400' : null + const before = await buildPaneProcessFingerprint([shell(), agent()], SHELL, { + platform: 'linux', + readLinuxStartTime: readNoDescendant + }) + const after = await buildPaneProcessFingerprint( + [shell(), agent({ stat: 'S+', pgid: AGENT })], + SHELL, + { platform: 'linux', readLinuxStartTime: readNoDescendant } + ) + + expect(before).toBeNull() + expect(after).toBeNull() + }) + }) +}) diff --git a/src/main/providers/posix-pane-foreground-fingerprint.ts b/src/main/providers/posix-pane-foreground-fingerprint.ts new file mode 100644 index 00000000000..ac1539300e2 --- /dev/null +++ b/src/main/providers/posix-pane-foreground-fingerprint.ts @@ -0,0 +1,93 @@ +import { readFile } from 'node:fs/promises' +import { collectDescendantsFromIndex, getProcessTableIndex } from '../../shared/process-table-index' +import { parseLinuxProcStatStartTime } from '../../shared/process-table-snapshot-reader' + +/** The job-control columns both `ps` tiers carry; `command`/`tty` are deliberately absent. */ +export type PaneFingerprintRow = { + pid: number + ppid: number + pgid?: number + tpgid?: number + stat: string + startTime?: string +} + +export type PaneFingerprintDeps = { + platform?: NodeJS.Platform + /** Linux: `/proc//stat` field 22, read for the pane subtree only. */ + readLinuxStartTime?: (pid: number) => Promise +} + +/** + * Only the job-control bits of `stat`. The scheduler letter (R/S/D/I/U) flips every tick + * on a working agent and says nothing about whether the pane changed hands; stopped, + * zombie, and foreground-group membership do. + */ +function jobControlState(stat: string): string { + const head = stat[0] ?? '' + const lifecycle = head === 'T' || head === 't' ? 'T' : head === 'Z' ? 'Z' : '' + return lifecycle + (stat.includes('+') ? '+' : '') +} + +async function readLinuxProcStartTime(pid: number): Promise { + try { + return parseLinuxProcStatStartTime(await readFile(`/proc/${pid}/stat`, 'utf8')) + } catch { + return null + } +} + +/** + * A per-pane summary of everything the cheap `ps` tier can see: the root shell's identity + * (pid + start marker) and terminal foreground group, and every descendant's identity, group, + * and job-control state. Two captures with equal fingerprints describe the same pane + * subtree, so the name resolved from the last full capture still holds. + * + * Null when the root is missing or unfenced (no start marker, no group columns): callers + * must then take the full capture rather than trust a comparison that could not be made. + */ +export async function buildPaneProcessFingerprint( + rows: readonly PaneFingerprintRow[], + rootPid: number, + deps: PaneFingerprintDeps = {} +): Promise { + const platform = deps.platform ?? process.platform + const index = getProcessTableIndex(rows) + const root = index.byPid.get(rootPid) + if (!root || root.pgid === undefined || root.tpgid === undefined) { + return null + } + const descendants = collectDescendantsFromIndex(index, rootPid) + const subtree = [root, ...descendants] + let startTimes: ReadonlyMap + if (platform === 'linux') { + const read = deps.readLinuxStartTime ?? readLinuxProcStartTime + const entries = await Promise.all( + subtree.map(async (row) => [row.pid, await read(row.pid)] as const) + ) + startTimes = new Map(entries) + } else { + // Collapse `lstart` padding (`Sep 3`) so both column sets stamp identically. + startTimes = new Map( + subtree.map((row) => [row.pid, row.startTime?.replace(/\s+/g, ' ') ?? null] as const) + ) + } + const rootStart = startTimes.get(rootPid) + if (!rootStart) { + return null + } + // Why every member, not just the root: a start marker is what makes a pid comparison + // recycle-safe. Stamping a missing one as a placeholder would let two captures that both + // failed to read it compare equal across a recycled pid, so a vanished agent could look + // unchanged. Refusing the fingerprint sends the caller to the full capture instead. + const members: string[] = [] + for (const row of descendants) { + const startTime = startTimes.get(row.pid) + if (!startTime) { + return null + } + members.push(`${row.pid}@${startTime}:${row.pgid ?? '?'}:${jobControlState(row.stat)}`) + } + members.sort() + return `${rootPid}@${rootStart}#${root.tpgid}:${jobControlState(root.stat)}|${members.join(',')}` +} diff --git a/src/main/providers/pty-process-inspection.ts b/src/main/providers/pty-process-inspection.ts index 59d910b2238..2c316ae05f9 100644 --- a/src/main/providers/pty-process-inspection.ts +++ b/src/main/providers/pty-process-inspection.ts @@ -23,6 +23,9 @@ type CompletionSensitivePtyProvider = IPtyProvider & { export type PtyProcessInspectionOptions = { expectedIncarnationId?: PtyIncarnationId scanChildProcesses?: boolean + /** A self-correcting cadence poll that reads only the process name: licenses a host to answer + * from a cheap capture and OMIT evidence. Never set by a caller that consumes evidence. */ + steadyState?: boolean } export async function inspectPtyProviderProcess( diff --git a/src/main/runtime/claude-structured-session-integration.test.ts b/src/main/runtime/claude-structured-session-integration.test.ts index 712543d0428..d94118f7eb8 100644 --- a/src/main/runtime/claude-structured-session-integration.test.ts +++ b/src/main/runtime/claude-structured-session-integration.test.ts @@ -119,6 +119,9 @@ function fakeClaude() { return undefined }, cancelAsyncMessage: async () => {}, + stopTask: async (taskId) => { + connection.calls.push({ subtype: 'stop_task', params: { taskId } }) + }, send: async (message) => { connection.sent.push(message) if (message.type === 'user') { diff --git a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts index a2a93533b6f..183f981ccee 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts @@ -50,6 +50,8 @@ const METHODS = [ } ] as const +const DESTRUCTIVE_METHOD_NAMES = new Set(['session.tabs.close', 'session.tabs.closeLifecycle']) + describe('session tab structured capability mutations', () => { for (const method of METHODS) { it(`rejects ${method.name} when the structured row is hidden`, async () => { @@ -88,6 +90,38 @@ describe('session tab structured capability mutations', () => { }) } + for (const method of METHODS) { + const expectedToAllowPromptedRow = !DESTRUCTIVE_METHOD_NAMES.has(method.name) + + it(`${expectedToAllowPromptedRow ? 'allows' : 'rejects'} ${method.name} on a row an old mobile client was prompted to update`, async () => { + const { calls, dispatch } = createFixture([], { + clientKind: 'mobile', + structuredNativeChatEnabled: true + }) + + const response = await dispatch(method.name, method.params('codex-session')) + + expect(response.ok).toBe(expectedToAllowPromptedRow) + expect(calls[method.runtimeMethod as keyof typeof calls]).toHaveBeenCalledTimes( + expectedToAllowPromptedRow ? 1 : 0 + ) + }) + + it(`${expectedToAllowPromptedRow ? 'allows' : 'rejects'} ${method.name} on a prompted Claude row for a mobile client without the Claude capability`, async () => { + const { calls, dispatch } = createFixture([STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], { + clientKind: 'mobile', + structuredNativeChatEnabled: true + }) + + const response = await dispatch(method.name, method.params('claude-session')) + + expect(response.ok).toBe(expectedToAllowPromptedRow) + expect(calls[method.runtimeMethod as keyof typeof calls]).toHaveBeenCalledTimes( + expectedToAllowPromptedRow ? 1 : 0 + ) + }) + } + it.each(['session.tabs.close', 'session.tabs.closeLifecycle'] as const)( 'allows capable mobile clients to close structured tabs when the experiment is enabled (%s)', async (method) => { @@ -131,7 +165,10 @@ describe('session tab structured capability mutations', () => { ) }) -function createFixture(capabilities: RuntimeCapability[]) { +function createFixture( + capabilities: RuntimeCapability[], + options: { clientKind?: 'mobile' | 'runtime'; structuredNativeChatEnabled?: boolean } = {} +) { const snapshot = agentSnapshot() const calls = { closeMobileSessionTab: vi.fn().mockResolvedValue({ closed: true }), @@ -142,11 +179,14 @@ function createFixture(capabilities: RuntimeCapability[]) { const runtime = { getRuntimeId: () => 'test-runtime', listMobileSessionTabs: vi.fn().mockResolvedValue(snapshot), + getClientSettings: () => ({ + experimentalStructuredNativeChat: options.structuredNativeChatEnabled === true + }), ...calls } as unknown as OrcaRuntimeService const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) const context: RpcDispatchStreamingOptions = { - clientKind: 'runtime', + clientKind: options.clientKind ?? 'runtime', pairedDeviceId: 'paired-client', clientCapabilities: capabilities } diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts index 7128f756d6e..cf68f7739c0 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts @@ -5,7 +5,11 @@ import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { RuntimeMobileSessionTabsSnapshot } from '../../../../shared/runtime-types' -import { projectSessionTabAgentStatus } from './session-tab-agent-status-projection' +import { + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE, + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + projectSessionTabAgentStatus +} from './session-tab-agent-status-projection' function makeSnapshot(sessionBoundary: boolean): RuntimeMobileSessionTabsSnapshot { return { @@ -160,23 +164,96 @@ describe('projectSessionTabAgentStatus', () => { CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ] - it.each([ - ['mobile', 'mobile' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]], - ['runtime', 'runtime' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]] - ])( - 'withholds Claude rows from a paired %s client that never negotiated them', - (_name, clientKind, capabilities) => { - const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + it('withholds Claude rows from a paired runtime client that never negotiated them', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'runtime', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) - expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) - // A row pruned from `tabs` but left in the layout is its own dead tab. - expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) - expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) - expect(projected.activeGroupId).toBe('group-a') - expect(projected.activeTabId).toBe('agent-session:codex') - expect(projected.activeTabType).toBe('agent-session') + expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) + // A row pruned from `tabs` but left in the layout is its own dead tab. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) + expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) + expect(projected.activeGroupId).toBe('group-a') + expect(projected.activeTabId).toBe('agent-session:codex') + expect(projected.activeTabType).toBe('agent-session') + }) + + it('uses a desktop fallback for an unsupported Claude row instead of withholding it', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'mobile', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + // The row survives so the chat the desktop shows is not simply absent on the phone. + expect(projected.tabs.map((tab) => tab.id)).toEqual([ + 'agent-session:codex', + 'agent-session:claude' + ]) + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + 'Codex Chat', + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + // Nothing is removed, so the layout it belonged to is untouched. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a', 'group-b']) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + expect(projected.activeTabId).toBe('agent-session:codex') + }) + + it('projects agent-specific fallback titles for a mobile client with no capabilities', () => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', [], true) + + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + }) + + it('does not treat the Claude capability as a substitute for the base structured capability', () => { + const projected = projectSessionTabAgentStatus( + claudeSnapshot, + 'mobile', + [CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + + expect(projected.tabs.map((tab) => tab.title)).toEqual([ + STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE, + CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + ]) + }) + + it('shows both real titles once mobile negotiates Claude', () => { + expect(projectSessionTabAgentStatus(claudeSnapshot, 'mobile', structuredMobile, true)).toBe( + claudeSnapshot + ) + }) + + // Why: updating cannot reveal a chat the desktop is not serving, so the prompt would lie. + it('withholds rather than prompts when the desktop experiment is off', () => { + for (const capabilities of [ + [], + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + structuredMobile + ]) { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', capabilities, false) + expect(projected.tabs).toEqual([]) } - ) + }) + + it('never emits an empty structured tab title', () => { + for (const capabilities of [[], [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]]) { + const projected = projectSessionTabAgentStatus(claudeSnapshot, 'mobile', capabilities, true) + for (const tab of projected.tabs) { + expect(tab.title.length).toBeGreaterThan(0) + } + } + }) it.each([ ['mobile', 'mobile' as const, structuredMobile], diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts index 0e0d9c716a5..4496fdc5435 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts @@ -1,6 +1,7 @@ import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' import type { @@ -13,6 +14,43 @@ import { structuredNativeChatProjectionEnabled } from './structured-agent-sessio type SessionTabsPayload = RuntimeMobileSessionTabsResult | RuntimeMobileSessionTabsSnapshot +/** Capped at 128px / one line in every shipped mobile build, so ~15-18 characters render. */ +export const STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE = 'Update to view' +export const CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE = 'Open on desktop' + +function clientCanRenderStructuredAgentSessionTab( + tab: RuntimeMobileSessionAgentTab, + clientCapabilities: readonly RuntimeCapability[] | undefined +): boolean { + if (!clientCapabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY)) { + return false + } + return ( + tab.agent === 'codex' || + clientCapabilities.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) +} + +function resolveMobileStructuredChatFallbackTitle( + tab: RuntimeMobileSessionAgentTab, + args: { + clientKind: 'mobile' | 'runtime' | undefined + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled?: boolean + } +): string | null { + if ( + args.clientKind !== 'mobile' || + args.structuredNativeChatEnabled !== true || + clientCanRenderStructuredAgentSessionTab(tab, args.clientCapabilities) + ) { + return null + } + return tab.agent === 'claude' + ? CLAUDE_STRUCTURED_CHAT_DESKTOP_ONLY_TAB_TITLE + : STRUCTURED_CHAT_UPDATE_REQUIRED_TAB_TITLE +} + export function projectSessionTabAgentStatus( payload: TPayload, clientKind: 'mobile' | 'runtime' | undefined, @@ -24,16 +62,27 @@ export function projectSessionTabAgentStatus true) - // Why: a paired client renders only codex structured tabs unless it says otherwise - // (mobile's resolveMobileNativeChat returns null for every other agent), so an - // ungated row would list and select into a pane that shows neither chat nor terminal. - if ( - structuredVisible && - clientKind !== undefined && - !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) - ) { - projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') + let projected: TPayload + if (clientKind === 'mobile' && structuredNativeChatEnabled === true) { + // Why: deleting the row left the user hunting for a chat the desktop says exists; the row + // survives with a title naming the fix. Nothing is removed, so no group/layout repair applies. + projected = projectUnsupportedAgentSessionTabTitles(payload, { + clientKind, + clientCapabilities, + structuredNativeChatEnabled + }) + } else { + projected = structuredVisible ? payload : projectAgentSessionTabsOut(payload, () => true) + // Why: a paired client renders only codex structured tabs unless it says otherwise + // (mobile's resolveMobileNativeChat returns null for every other agent), so an + // ungated row would list and select into a pane that shows neither chat nor terminal. + if ( + structuredVisible && + clientKind !== undefined && + !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) { + projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') + } } // Why: only paired runtimes have legacy `done` completion side effects; mobile must keep its row without changing the exact v2 auth shape. if ( @@ -55,6 +104,47 @@ export function projectSessionTabAgentStatus( + payload: TPayload, + args: { + clientKind: 'mobile' + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled: true + } +): TPayload { + let changed = false + const tabs = payload.tabs.map((tab) => { + if (tab.type !== 'agent-session') { + return tab + } + const title = resolveMobileStructuredChatFallbackTitle(tab, args) + if (title === null) { + return tab + } + changed = true + return { ...tab, title } + }) + return changed ? ({ ...payload, tabs } as TPayload) : payload +} + +export function assertAgentSessionTabDestructiveMutationSupported( + payload: SessionTabsPayload, + tabId: string, + clientKind: 'mobile' | 'runtime' | undefined, + clientCapabilities: readonly RuntimeCapability[] | undefined +): void { + if (clientKind === undefined) { + return + } + const tab = payload.tabs.find((candidate) => candidate.id === tabId) + if ( + tab?.type === 'agent-session' && + !clientCanRenderStructuredAgentSessionTab(tab, clientCapabilities) + ) { + throw new Error('structured_agent_session_unsupported') + } +} + function projectAgentSessionTabsOut( payload: TPayload, shouldHide: (tab: RuntimeMobileSessionAgentTab) => boolean diff --git a/src/main/runtime/rpc/methods/session-tab-close-methods.ts b/src/main/runtime/rpc/methods/session-tab-close-methods.ts index bd60ecd6ddf..361ba8e4c51 100644 --- a/src/main/runtime/rpc/methods/session-tab-close-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-close-methods.ts @@ -3,6 +3,7 @@ import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/ import { defineMethod, type RpcAnyMethod } from '../core' import { CloseLifecycleTab, CloseTab } from './session-tabs-schemas' import { assertProjectedSessionTabVisible } from './session-tab-browser-placement-projection' +import { assertAgentSessionTabDestructiveMutationSupported } from './session-tab-agent-status-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' @@ -12,8 +13,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ params: CloseTab, handler: async (params, context) => { if (context.clientKind) { + const raw = await context.runtime.listMobileSessionTabs( + params.worktree, + context.pairedDeviceId + ) const visible = projectSessionTabsForClient( - await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), + raw, context.clientKind, context.clientCapabilities, context.clientKind === 'mobile' @@ -21,6 +26,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) + assertAgentSessionTabDestructiveMutationSupported( + raw, + params.tabId, + context.clientKind, + context.clientCapabilities + ) } const requiresIntent = context.clientKind === undefined || @@ -81,8 +92,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ params: CloseLifecycleTab, handler: async (params, context) => { if (context.clientKind) { + const raw = await context.runtime.listMobileSessionTabs( + params.worktree, + context.pairedDeviceId + ) const visible = projectSessionTabsForClient( - await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), + raw, context.clientKind, context.clientCapabilities, context.clientKind === 'mobile' @@ -90,6 +105,12 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) + assertAgentSessionTabDestructiveMutationSupported( + raw, + params.tabId, + context.clientKind, + context.clientCapabilities + ) } return withSpan( 'runtime.session-tabs.close-lifecycle', diff --git a/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts b/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts new file mode 100644 index 00000000000..083f334285e --- /dev/null +++ b/src/main/runtime/rpc/methods/session-tabs-structured-restore.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import { RpcDispatcher } from '../dispatcher' +import type { RpcRequest } from '../core' +import type { OrcaRuntimeService } from '../../orca-runtime' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import { SESSION_TAB_METHODS } from './session-tabs' + +function makeRequest(method: string, params?: unknown): RpcRequest { + return { id: 'req-1', authToken: 'tok', method, params } +} + +describe('session tab structured restore gating', () => { + it('does not restore structured tabs for mobile while the host setting is off', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() + }) + + // Why: an old build has no capability to advertise, and skipping the restore left it with + // nothing to project after a desktop restart — neither the chat nor its fallback row. + it('restores structured tabs for a mobile client that advertises no capability', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { clientKind: 'mobile', clientCapabilities: [] } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) + + it('restores structured tabs for mobile once the setting is present', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) +}) + +function visibleSnapshot() { + return { + worktree: 'wt-1', + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: 'group-1', + activeTabId: 'tab-1::leaf-1', + activeTabType: 'terminal' as const, + tabGroups: [{ id: 'group-1', activeTabId: 'tab-1', tabOrder: ['tab-1'] }], + tabs: [ + { + type: 'terminal' as const, + id: 'tab-1::leaf-1', + parentTabId: 'tab-1', + leafId: 'leaf-1', + title: 'Terminal', + status: 'ready' as const, + terminal: 'pty-1', + isActive: true + } + ] + } +} diff --git a/src/main/runtime/rpc/methods/session-tabs.test.ts b/src/main/runtime/rpc/methods/session-tabs.test.ts index 29131869fa0..be61fc55edf 100644 --- a/src/main/runtime/rpc/methods/session-tabs.test.ts +++ b/src/main/runtime/rpc/methods/session-tabs.test.ts @@ -2,10 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' import type { RpcRequest } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' -import { - SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY, - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY -} from '../../../../shared/protocol-version' +import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { SESSION_TAB_METHODS } from './session-tabs' function makeRequest(method: string, params?: unknown): RpcRequest { @@ -13,48 +10,6 @@ function makeRequest(method: string, params?: unknown): RpcRequest { } describe('session tab RPC methods', () => { - it('does not restore structured tabs for mobile while the host setting is off', async () => { - const runtime = { - getRuntimeId: () => 'test-runtime', - getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), - restoreStructuredAgentSessionTabs: vi.fn(), - listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) - } as unknown as OrcaRuntimeService - const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) - - const response = await dispatcher.dispatch( - makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), - { - clientKind: 'mobile', - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] - } - ) - - expect(response.ok).toBe(true) - expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() - }) - - it('restores structured tabs for mobile only after capability and setting are present', async () => { - const runtime = { - getRuntimeId: () => 'test-runtime', - getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), - restoreStructuredAgentSessionTabs: vi.fn(), - listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) - } as unknown as OrcaRuntimeService - const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) - - const response = await dispatcher.dispatch( - makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), - { - clientKind: 'mobile', - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] - } - ) - - expect(response.ok).toBe(true) - expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) - }) - it('routes mobile-only activation without notifying desktop clients', async () => { const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts index 33018c21f4d..e15544b2d24 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts @@ -2,8 +2,11 @@ // // Shared by every structured method file so one gate governs the whole surface: a client that does // not advertise `agent-session.structured.v1` is told the surface does not exist rather than being -// handed a session it cannot render or drive — and, just as importantly, cannot make the host EXIST -// by calling into it, which is an observable side effect. +// handed the session journal or mutation surface. +// +// This gate no longer implies such a client cannot make the host exist: session-tab restore runs +// for old mobile clients while structured chat is enabled so they receive a fallback row, and that +// path constructs the host. `agentSession.*` stays refused either way, which is what this gate is for. import { getStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' diff --git a/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts new file mode 100644 index 00000000000..1a62045c85b --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.test.ts @@ -0,0 +1,239 @@ +// The create route's pre-commit boundary: a failure before `attach` must reach the client as a +// refusal it can classify, and a failure at or after `attach` must not. + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' +import { setStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' +import { computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' +import { isDefinitiveAgentSessionCreateRefusal } from '../../../../shared/agent-session-definitive-refusal' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { OrcaRuntimeService } from '../../orca-runtime' +import type { RpcResponse } from '../core' +import { RpcDispatcher } from '../dispatcher' +import { STRUCTURED_AGENT_SESSION_METHODS } from './structured-agent-session' + +const SESSION = 'session-alpha' +const OPERATION = '1800000000000-00000000000000000000000000000001' +const WORKTREE = 'id:workspace-1' + +const STRUCTURED_CLIENT = { + clientKind: 'runtime' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} + +function createParams(overrides: Record = {}) { + return { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree: WORKTREE, agent: 'codex' } + }), + ...(overrides.envelope as Record | undefined) + }, + worktree: WORKTREE, + agent: 'codex' + } +} + +let attach: ReturnType + +function hostStub(): StructuredAgentSessionHost { + attach = vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 0 }, + value: { sessionId: SESSION, fence: 1, page: {}, unconfirmedClientMessageIds: [] } + })) + return { attach } as unknown as StructuredAgentSessionHost +} + +const resolvedIntent = { + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'codex', + agent: 'codex', + accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + runtimeKind: 'native' +} + +async function create( + runtimeOverrides: Record = {}, + params: unknown = createParams() +): Promise { + const runtime = { + getRuntimeId: () => 'runtime-1', + registerSubscriptionCleanup: vi.fn(), + cleanupSubscription: vi.fn(), + cleanupSubscriptionsByPrefix: vi.fn(), + ensureStructuredAgentSessionHost: vi.fn(async () => undefined), + resolveStructuredAgentSessionCreateIntent: vi.fn(async (input: { envelope: unknown }) => ({ + envelope: input.envelope, + ...resolvedIntent + })), + publishStructuredAgentSessionTab: vi.fn(async () => undefined), + ...runtimeOverrides + } + const replies: RpcResponse[] = [] + await new RpcDispatcher({ + runtime: runtime as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }).dispatchStreaming( + { id: 'request-1', authToken: 'token', method: 'agentSession.create', params }, + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + STRUCTURED_CLIENT + ) + const first = replies[0] + if (!first) { + throw new Error('no reply for agentSession.create') + } + return first +} + +/** The refusal a client can act on, or null when the reply was not one. */ +function refusalOf(response: RpcResponse): { code: string; message: string } | null { + if (!response.ok) { + return null + } + const result = response.result as { ok: boolean; refusal?: { code: string; message: string } } + return result.ok ? null : (result.refusal ?? null) +} + +beforeEach(() => { + setStructuredAgentSessionHost(hostStub()) + vi.spyOn(console, 'warn').mockImplementation(() => undefined) +}) + +afterEach(() => { + setStructuredAgentSessionHost(null) + vi.restoreAllMocks() +}) + +describe('a create refused before it commits', () => { + it('answers a code-carrying refusal as a definitive envelope', async () => { + const response = await create({ + resolveStructuredAgentSessionCreateIntent: vi.fn(async () => { + throw new Error('structured_agent_session_unsupported') + }) + }) + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('structured_agent_session_unsupported') + expect(refusal?.message).toContain('structured agent chat') + expect(refusal?.message).not.toContain('Codex') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(attach).not.toHaveBeenCalled() + }) + + it('answers a code-less failure as a definitive envelope too, keeping the cause in the message', async () => { + // The class no per-site conversion catches: an unresolvable worktree throws prose, not a code. + const response = await create({ + resolveStructuredAgentSessionCreateIntent: vi.fn(async () => { + throw new Error('No worktree matches id:workspace-1') + }) + }) + + const refusal = refusalOf(response) + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(refusal?.message).toContain('structured agent chat') + expect(refusal?.message).not.toContain('Codex') + expect(refusal?.message).toContain('No worktree matches id:workspace-1') + expect(attach).not.toHaveBeenCalled() + }) + + it('answers a host that will not install as a definitive envelope', async () => { + setStructuredAgentSessionHost(null) + + const response = await create({ + ensureStructuredAgentSessionHost: vi.fn(async () => { + throw new Error('EACCES: could not open the session store') + }) + }) + + const refusal = refusalOf(response) + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + expect(refusal?.message).toContain('could not open the session store') + }) + + it('answers a missing host as a definitive envelope rather than a thrown code', async () => { + setStructuredAgentSessionHost(null) + + const response = await create() + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('structured_agent_session_unsupported') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(true) + }) + + it('still refuses a fingerprint conflict with its own code, not a pre-commit one', async () => { + const response = await create( + {}, + createParams({ envelope: { payloadFingerprint: 'a'.repeat(64) } }) + ) + + expect(refusalOf(response)?.code).toBe('agent_session_operation_conflict') + expect(attach).not.toHaveBeenCalled() + }) +}) + +describe('the boundary the envelope stops at', () => { + it('leaves a failure at attach unknown, because it may have committed', async () => { + attach.mockRejectedValueOnce(new Error('attach exploded')) + + const response = await create() + + expect(response).toMatchObject({ ok: false, error: { code: 'runtime_error' } }) + expect(refusalOf(response)).toBeNull() + }) + + it('leaves a committed create whose tab could not be published unknown', async () => { + const response = await create({ + publishStructuredAgentSessionTab: vi.fn(async () => { + throw new Error('publish failed') + }) + }) + + const refusal = refusalOf(response) + expect(refusal?.code).toBe('agent_session_operation_unknown') + expect(isDefinitiveAgentSessionCreateRefusal(refusal?.code)).toBe(false) + }) + + it('keeps hiding the surface from a client that never advertised it', async () => { + const replies: RpcResponse[] = [] + await new RpcDispatcher({ + runtime: { getRuntimeId: () => 'runtime-1' } as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }).dispatchStreaming( + { + id: 'request-1', + authToken: 'token', + method: 'agentSession.create', + params: createParams() + }, + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + { clientKind: 'runtime', clientCapabilities: [] } + ) + + expect(replies[0]).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + }) + + it('keeps a client-declared fence a programming error, not a refusal', async () => { + const response = await create({}, createParams({ envelope: { expectedRuntimeFence: 1 } })) + + expect(response).toMatchObject({ + ok: false, + error: { code: 'agent_session_operation_invalid' } + }) + }) +}) diff --git a/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts new file mode 100644 index 00000000000..24d9fa2aec4 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-precommit-refusal.ts @@ -0,0 +1,71 @@ +// Nothing before `attach` commits a session, so every failure in that span definitively created +// nothing. Thrown, it reaches a remote client as a generic transport error, indistinguishable from +// an answer that was lost on the way back — and a client that cannot tell those apart either +// strands the user with no chat and no terminal, or spawns a sibling beside a session that may +// already exist. So the whole span answers with a refusal envelope carrying a code, whatever it +// failed on. +// +// Converting the span rather than each throw site is deliberate: alongside the throws that carry a +// code there is a code-less class — an unresolvable worktree, a store that will not open, a host +// that will not install — that no per-site list catches, and it is exactly the class that reaches +// the user as nothing at all. + +import { + AGENT_SESSION_WIRE_REFUSAL_CODES, + type AgentSessionWireRefusal, + type AgentSessionWireRefusalCode +} from '../../../../shared/agent-session-wire' + +export type StructuredCreateRefused = { refusal: AgentSessionWireRefusal } + +/** A pre-commit failure with no code of its own still proves the host could not serve a structured + * session for this request and did not create one, which is what `unsupported` says on the wire. + * A new code would say it more precisely, but only to clients new enough to know it. */ +const UNCODED_PRECOMMIT_REFUSAL_CODE: AgentSessionWireRefusalCode = + 'structured_agent_session_unsupported' + +function wireRefusalCode(error: unknown): AgentSessionWireRefusalCode | null { + const candidates = [ + error instanceof Error && 'code' in error ? (error as { code: unknown }).code : undefined, + error instanceof Error ? error.message : String(error) + ] + for (const candidate of candidates) { + if ( + typeof candidate === 'string' && + (AGENT_SESSION_WIRE_REFUSAL_CODES as readonly string[]).includes(candidate) + ) { + return candidate as AgentSessionWireRefusalCode + } + } + return null +} + +function precommitRefusal(error: unknown): AgentSessionWireRefusal { + const code = wireRefusalCode(error) + if (code) { + return { code, message: 'Orca cannot open a structured agent chat for this workspace.' } + } + const message = error instanceof Error ? error.message : String(error) + // A code-less failure here is often a defect, not a policy answer; the refusal keeps the user + // moving, the log keeps the cause findable. + console.warn('[agent-session] create refused before it committed anything', error) + return { + code: UNCODED_PRECOMMIT_REFUSAL_CODE, + message: `Orca could not prepare a structured agent chat for this workspace: ${message}` + } +} + +/** + * Runs the pre-commit half of a create. Anything it throws becomes a refusal; a refusal it returns + * itself passes through. Must not wrap `attach` or anything after it — past that point a failure no + * longer proves the session does not exist. + */ +export async function resolveUncommittedStructuredCreate( + prepare: () => Promise +): Promise { + try { + return await prepare() + } catch (error) { + return { refusal: precommitRefusal(error) } + } +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts index 4f7c1b20cc3..da66923c731 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts @@ -149,7 +149,11 @@ export const SendParams = z .strict() export const CancelParams = z - .object({ envelope: MutationEnvelope, turnId: Identifier('Invalid turn id') }) + .object({ + envelope: MutationEnvelope, + turnId: Identifier('Invalid turn id'), + scope: z.literal('background-tasks').optional() + }) .strict() export const RespondParams = z diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index 155d0aa6768..e4888e4a9df 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -436,21 +436,34 @@ describe('method routing', () => { /** A client-supplied location skips the worktree-resolving support check, so both attach-shaped * entries must ask the executing host directly or a host that cannot fence a provider child * would create one anyway. */ - it.each(['agentSession.create', 'agentSession.ensure'])( - 'refuses %s for a client-supplied location the executing host does not support', - async (method) => { - hostCalls.supportsCreate.mockReturnValue(false) + it('returns a refusal envelope when create cannot support a client-supplied location', async () => { + hostCalls.supportsCreate.mockReturnValue(false) - const refused = await call(method, attachParams()) + const refused = await call('agentSession.create', attachParams()) - expect(refused).toMatchObject({ + expect(refused).toMatchObject({ + ok: true, + result: { ok: false, - error: { message: expect.stringContaining('structured_agent_session_unsupported') } - }) - expect(hostCalls.attach).not.toHaveBeenCalled() - expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') - } - ) + refusal: { code: 'structured_agent_session_unsupported' } + } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + }) + + it('keeps ensure failures as top-level errors for an unsupported client location', async () => { + hostCalls.supportsCreate.mockReturnValue(false) + + const refused = await call('agentSession.ensure', attachParams()) + + expect(refused).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + }) it('tags the prompt kind from the method name, not from the client', async () => { const params = { diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index 3b18f6b0ef1..b69ff6fd628 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -2,9 +2,8 @@ // // Every method here is gated on the client advertising // `agent-session.structured.v1`. A client that does not is told the surface does -// not exist rather than being handed a session it cannot render or drive; that -// is the whole visibility rule, because nothing else on the runtime publishes a -// structured session. +// not exist rather than receiving the journal or mutation surface. Session-tab +// inventory may expose only a metadata placeholder for an incapable mobile client. import { agentSessionFingerprintConflict, @@ -21,6 +20,7 @@ import { } from './structured-agent-session-gate' import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold' +import { resolveUncommittedStructuredCreate } from './structured-agent-session-precommit-refusal' import { AttachParams, CancelParams, @@ -52,21 +52,27 @@ function subscriptionIdFor(ctx: RpcContext, sessionId: string): string { * host the same question directly: the answer includes host-measured facts the client cannot see * or forge, such as whether this machine can read a provider child's process start time. */ -async function attachClientSuppliedLocation( - params: z.infer, - ctx: RpcContext -): Promise { +async function resolveClientSuppliedAttach(params: z.infer, ctx: RpcContext) { await ensureHostInstalled(ctx) const host = requireHost(ctx) if (!host.supportsCreate(params.location, params.agent)) { throw new Error('structured_agent_session_unsupported') } const { agent: _attachAgent, provider: _attachProvider, ...attachWithoutAgent } = params - return host.attach(callerFor(ctx), { + const attachParams = { ...attachWithoutAgent, provider: params.provider as 'claude' | 'codex', agent: params.agent as 'claude' | 'codex' - } as AgentSessionAttachParams) + } as AgentSessionAttachParams + return { host, attachParams } +} + +async function attachClientSuppliedLocation( + params: z.infer, + ctx: RpcContext +): Promise { + const { host, attachParams } = await resolveClientSuppliedAttach(params, ctx) + return host.attach(callerFor(ctx), attachParams) } export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ @@ -88,60 +94,76 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ if (params.envelope.expectedRuntimeFence !== null) { throw new Error('agent_session_operation_invalid') } - if ('worktree' in params) { - const intentFingerprint = computeAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: params.envelope.sessionId, - fields: { worktree: params.worktree, agent: params.agent } - }) - const conflict = agentSessionFingerprintConflict(params.envelope, intentFingerprint) - if (conflict) { - return { ok: false, refusal: conflict } - } - const resolved = await ctx.runtime.resolveStructuredAgentSessionCreateIntent(params) - const hostFingerprint = computeAgentSessionPayloadFingerprint({ - method: 'agentSession.attach', - sessionId: params.envelope.sessionId, - fields: { - location: resolved.location, - provider: resolved.provider, - agent: resolved.agent, - accountHome: resolved.accountHome, - runtimeKind: resolved.runtimeKind, - expectedRuntimeFence: null + // Everything up to `attach` is pre-commit, and answers with a refusal rather than a throw so + // a client can tell "nothing was created" from "the outcome is unknown". + const prepared = await resolveUncommittedStructuredCreate(async () => { + if ('worktree' in params) { + const intentFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: params.envelope.sessionId, + fields: { worktree: params.worktree, agent: params.agent } + }) + const conflict = agentSessionFingerprintConflict(params.envelope, intentFingerprint) + if (conflict) { + return { refusal: conflict } } - }) - await ensureHostInstalled(ctx) - const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved - const attachParams: AgentSessionAttachParams = { - ...resolvedAttach, - provider: resolved.provider as 'claude' | 'codex', - agent: resolved.agent as 'claude' | 'codex', - envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } - } - const result = await requireHost(ctx).attach(callerFor(ctx), attachParams) - if (result.ok) { - try { - await ctx.runtime.publishStructuredAgentSessionTab({ + const resolved = await ctx.runtime.resolveStructuredAgentSessionCreateIntent(params) + const hostFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.attach', + sessionId: params.envelope.sessionId, + fields: { + location: resolved.location, + provider: resolved.provider, + agent: resolved.agent, + accountHome: resolved.accountHome, + runtimeKind: resolved.runtimeKind, + expectedRuntimeFence: null + } + }) + await ensureHostInstalled(ctx) + const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved + const attachParams: AgentSessionAttachParams = { + ...resolvedAttach, + provider: resolved.provider as 'claude' | 'codex', + agent: resolved.agent as 'claude' | 'codex', + envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } + } + return { + host: requireHost(ctx), + attachParams, + tab: { workspaceId: resolved.location.workspaceId, - sessionId: result.value.sessionId, - agent: resolved.agent as 'claude' | 'codex', - activate: true - }) - } catch (error) { - console.warn('[agent-session] create committed before tab publication failed', error) - return { - ok: false, - refusal: { - code: 'agent_session_operation_unknown', - message: 'The chat may have been created, but its tab could not be confirmed.' - } + agent: resolved.agent as 'claude' | 'codex' } } } - return result + const { host, attachParams } = await resolveClientSuppliedAttach(params, ctx) + return { host, attachParams, tab: null } + }) + if ('refusal' in prepared) { + return { ok: false, refusal: prepared.refusal } } - return attachClientSuppliedLocation(params, ctx) + const result = await prepared.host.attach(callerFor(ctx), prepared.attachParams) + if (result.ok && prepared.tab) { + try { + await ctx.runtime.publishStructuredAgentSessionTab({ + workspaceId: prepared.tab.workspaceId, + sessionId: result.value.sessionId, + agent: prepared.tab.agent, + activate: true + }) + } catch (error) { + console.warn('[agent-session] create committed before tab publication failed', error) + return { + ok: false, + refusal: { + code: 'agent_session_operation_unknown', + message: 'The chat may have been created, but its tab could not be confirmed.' + } + } + } + } + return result } }), defineMethod({ diff --git a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts index 4333713a445..f9efa46d16d 100644 --- a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts +++ b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts @@ -1,13 +1,24 @@ import type { RpcContext } from '../core' -import { supportsStructuredAgentSessions } from './structured-agent-session-policy' +import { + isStructuredNativeChatEnabled, + supportsStructuredAgentSessions +} from './structured-agent-session-policy' +/** Republishes structured tabs into the host's own snapshot map. + * + * Mobile is gated on the host setting alone, NOT on the client's capability: an old build is + * shown a fallback prompt in place of each chat, and gating on capability left it with nothing to + * project after a desktop restart — no chat and no prompt. The setting still gates it, because + * with structured chat off there is nothing for any mobile client to reach. Restoring spawns no + * provider child for a cleanly closed session. */ export async function restoreStructuredTabsIfSupported( context: Pick ): Promise { - if ( - supportsStructuredAgentSessions(context) && - typeof context.runtime.restoreStructuredAgentSessionTabs === 'function' - ) { + const shouldRestore = + context.clientKind === 'mobile' + ? isStructuredNativeChatEnabled(context.runtime) + : supportsStructuredAgentSessions(context) + if (shouldRestore && typeof context.runtime.restoreStructuredAgentSessionTabs === 'function') { await context.runtime.restoreStructuredAgentSessionTabs() } } diff --git a/src/main/runtime/rpc/mobile-socket-wiring.test.ts b/src/main/runtime/rpc/mobile-socket-wiring.test.ts index 14d5beb8cd0..5ce3325083c 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.test.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.test.ts @@ -190,6 +190,58 @@ describe('MobileSocketWiring', () => { expect(wiring.connectionCount).toBe(0) }) + it('lets the capability RPC write capabilities back onto the socket', () => { + // `runtime.clientCapabilities.update` stores the advertised set by assigning + // `authenticatedSocket.clientCapabilities`. A read-only socket makes that a + // TypeError, the RPC answers `runtime_error`, and every structured + // agent-session tab is then projected away from a capable phone. + const desktop = generateKeyPair() + const phone = generateKeyPair() + const ws = new FakeSocket() + const transport = new FakeTransport() + const onText = vi.fn() + const wiring = new MobileSocketWiring({ + deviceRegistry: registryFor('device-1', 'valid-token', 'mobile'), + e2eeKeypair: { + publicKey: desktop.publicKey, + secretKey: desktop.secretKey, + publicKeyB64: Buffer.from(desktop.publicKey).toString('base64') + }, + onText, + onBinary: vi.fn(), + onClose: vi.fn() + }) + wiring.attachTransport(transport) + + transport.receive( + ws, + JSON.stringify({ + type: 'e2ee_hello', + publicKeyB64: Buffer.from(phone.publicKey).toString('base64') + }) + ) + const sharedKey = deriveSharedKey(phone.secretKey, desktop.publicKey) + transport.receive( + ws, + encrypt(JSON.stringify({ type: 'e2ee_auth', deviceToken: 'valid-token' }), sharedKey) + ) + transport.receive(ws, encrypt('{"id":"rpc-1","method":"status.get"}', sharedKey)) + + const socket = onText.mock.calls[0]?.[0] + expect(socket).toBeDefined() + expect(socket.clientCapabilities).toEqual([]) + + expect(() => { + socket.clientCapabilities = ['agent-session.structured.v1'] + }).not.toThrow() + expect(socket.clientCapabilities).toEqual(['agent-session.structured.v1']) + + // Later requests on the same connection must see the updated set, so the + // channel is the single source of truth rather than a detached copy. + transport.receive(ws, encrypt('{"id":"rpc-2","method":"status.get"}', sharedKey)) + expect(onText.mock.calls[1]?.[0].clientCapabilities).toEqual(['agent-session.structured.v1']) + }) + it('closes an unknown-token socket even when reporting the failure throws', () => { const desktop = generateKeyPair() const phone = generateKeyPair() diff --git a/src/main/runtime/rpc/mobile-socket-wiring.ts b/src/main/runtime/rpc/mobile-socket-wiring.ts index 384f403e9de..2d536b15038 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.ts @@ -165,9 +165,16 @@ export class MobileSocketWiring { ws, connectionId, device, + // Why: the channel owns the set for the whole connection, so this reads + // through rather than snapshotting. It must also WRITE through — the + // capability RPC updates the socket, and a getter-only property makes + // that a TypeError, which strands a capable phone with no capabilities. get clientCapabilities() { return channel.clientCapabilities }, + set clientCapabilities(next: readonly RuntimeCapability[]) { + channel.clientCapabilities = next + }, transport: metadata } this.authenticatedSockets.set(ws, socket) diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index 923d5627f72..761f04a4663 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -246,6 +246,8 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise + host?.publishBackgroundTaskState(sessionId, state), ...(deps.openClaudeConnection ? { openClaudeConnection: deps.openClaudeConnection } : {}), ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) }) diff --git a/src/main/runtime/structured-claude-runtime-adapter.ts b/src/main/runtime/structured-claude-runtime-adapter.ts index 398288562b9..95151f1dae9 100644 --- a/src/main/runtime/structured-claude-runtime-adapter.ts +++ b/src/main/runtime/structured-claude-runtime-adapter.ts @@ -1,4 +1,5 @@ import type { AgentSessionRecord } from '../../shared/agent-session-record' +import type { AgentSessionBackgroundTaskState } from '../../shared/agent-session-wire' import { join } from 'node:path' import { resolveClaudeCommand } from '../codex-cli/command' import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' @@ -29,6 +30,10 @@ export type StructuredClaudeRuntimeAdapterDeps = { openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] readProcessStartTime?: ClaudeStructuredSessionAdapterDeps['readProcessStartTime'] onUnexpectedExit: (event: StructuredAgentSessionLifecycleEvent) => void + onBackgroundTasksChanged?: ( + sessionId: string, + state: AgentSessionBackgroundTaskState | null + ) => void } export function createStructuredClaudeRuntimeAdapter( @@ -92,6 +97,9 @@ export function createStructuredClaudeRuntimeAdapter( }) } }, + ...(deps.onBackgroundTasksChanged + ? { onBackgroundTasksChanged: deps.onBackgroundTasksChanged } + : {}), ...(deps.openClaudeConnection ? { openConnection: deps.openClaudeConnection } : {}), ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) }) diff --git a/src/main/shell-prompt-readiness-probe.test.ts b/src/main/shell-prompt-readiness-probe.test.ts index c2c783953ac..56d4b35a06c 100644 --- a/src/main/shell-prompt-readiness-probe.test.ts +++ b/src/main/shell-prompt-readiness-probe.test.ts @@ -3,12 +3,16 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const lineEditorProbe = vi.hoisted(() => vi.fn()) const processReadinessProbe = vi.hoisted(() => vi.fn()) const resolveExecutablePath = vi.hoisted(() => vi.fn((value: string) => Promise.resolve(value))) +const resolveInstalledExecutablePaths = vi.hoisted(() => + vi.fn((): Promise => Promise.resolve([])) +) vi.mock('../shared/pty-slave-line-discipline-echo', () => ({ createPtySlaveLineEditorProbe: () => lineEditorProbe })) vi.mock('../shared/shell-process-readiness', () => ({ readShellProcessReadiness: processReadinessProbe, - resolveShellExecutablePath: resolveExecutablePath + resolveShellExecutablePath: resolveExecutablePath, + resolveInstalledShellExecutablePaths: resolveInstalledExecutablePaths })) import { createShellPromptReadinessProbe } from './shell-prompt-readiness-probe' @@ -19,6 +23,8 @@ describe('shell prompt readiness probe', () => { lineEditorProbe.mockReset() processReadinessProbe.mockReset() resolveExecutablePath.mockClear() + resolveInstalledExecutablePaths.mockClear() + resolveInstalledExecutablePaths.mockResolvedValue([]) }) afterEach(() => { @@ -95,6 +101,99 @@ describe('shell prompt readiness probe', () => { } }) + it('accepts a second installation of the same shell that the pane PATH resolves', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ + executablePath: '/opt/homebrew/bin/bash', + foreground: true + }) + resolveInstalledExecutablePaths.mockResolvedValue(['/bin/bash', '/opt/homebrew/bin/bash']) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + shellCwd: '/work', + shellPathEnv: '/opt/homebrew/bin:/usr/bin:/bin', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(resolveInstalledExecutablePaths).toHaveBeenCalledWith( + 'bash', + '/work', + '/opt/homebrew/bin:/usr/bin:/bin' + ) + expect(onPromptReady).toHaveBeenCalledOnce() + }) + + it('rejects a replacement with the shell basename that the pane PATH cannot reach', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ executablePath: '/tmp/bash', foreground: true }) + resolveInstalledExecutablePaths.mockResolvedValue(['/bin/bash', '/opt/homebrew/bin/bash']) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(onPromptReady).not.toHaveBeenCalled() + }) + + it('does not widen identity when the launched shell path resolves exactly', async () => { + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ executablePath: '/bin/zsh', foreground: true }) + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/zsh', + getShellPid: () => 42, + onPromptReady: vi.fn(), + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + + expect(resolveInstalledExecutablePaths).not.toHaveBeenCalled() + }) + + it('invalidates an alternate-installation result that resolves after disposal', async () => { + const pending: { resolve?: (value: string[]) => void } = {} + lineEditorProbe.mockResolvedValue('line-editor') + processReadinessProbe.mockResolvedValue({ + executablePath: '/opt/homebrew/bin/bash', + foreground: true + }) + resolveInstalledExecutablePaths.mockImplementation( + () => new Promise((resolve) => (pending.resolve = resolve)) + ) + const onPromptReady = vi.fn() + const probe = createShellPromptReadinessProbe({ + slavePath: '/dev/ttys048', + shellPath: '/bin/bash', + getShellPid: () => 42, + onPromptReady, + settleMs: 10 + }) + + probe?.notifyOutput('\x1b[?2004h') + await vi.advanceTimersByTimeAsync(10) + probe?.dispose() + pending.resolve?.(['/opt/homebrew/bin/bash']) + await vi.advanceTimersByTimeAsync(0) + + expect(onPromptReady).not.toHaveBeenCalled() + }) + it('does no external work when the ready marker cancels the settle window', async () => { const probe = createShellPromptReadinessProbe({ slavePath: '/dev/ttys048', diff --git a/src/main/shell-prompt-readiness-probe.ts b/src/main/shell-prompt-readiness-probe.ts index 11583fd2de7..3309ac91108 100644 --- a/src/main/shell-prompt-readiness-probe.ts +++ b/src/main/shell-prompt-readiness-probe.ts @@ -1,6 +1,7 @@ import { createPtySlaveLineEditorProbe } from '../shared/pty-slave-line-discipline-echo' import { readShellProcessReadiness, + resolveInstalledShellExecutablePaths, resolveShellExecutablePath } from '../shared/shell-process-readiness' import { @@ -32,6 +33,7 @@ export function createShellPromptReadinessProbe(options: { } const settleMs = options.settleMs ?? SHELL_PROMPT_PROBE_SETTLE_MS const expectedShellName = options.shellPath ? basename(options.shellPath).toLowerCase() : null + const shellCwd = options.shellCwd ?? process.cwd() const outputScanState = createLineEditorReadyOutputScanState() let disposed = false let timer: ReturnType | null = null @@ -52,11 +54,7 @@ export function createShellPromptReadinessProbe(options: { const [shell, expectedPath] = await Promise.all([ readShellProcessReadiness(shellPid), options.shellPath - ? resolveShellExecutablePath( - options.shellPath, - options.shellCwd ?? process.cwd(), - options.shellPathEnv - ) + ? resolveShellExecutablePath(options.shellPath, shellCwd, options.shellPathEnv) : Promise.resolve(null) ]) if (disposed || scheduledGeneration !== generation) { @@ -66,11 +64,28 @@ export function createShellPromptReadinessProbe(options: { !shell?.foreground || !expectedShellName || !expectedPath || - basename(shell.executablePath).toLowerCase() !== expectedShellName || - shell.executablePath !== expectedPath + basename(shell.executablePath).toLowerCase() !== expectedShellName ) { return } + if (shell.executablePath !== expectedPath) { + // Why widen past the launched path: a startup profile that `exec`s a second + // install of the same shell (Homebrew Bash over /bin/bash) keeps the pid but + // loses the wrapper's marker. Only installs this pane's own PATH resolves + // count, so a binary merely *named* bash/zsh outside it stays rejected. + const installedPaths = await resolveInstalledShellExecutablePaths( + expectedShellName, + shellCwd, + options.shellPathEnv + ) + if ( + disposed || + scheduledGeneration !== generation || + !installedPaths.includes(shell.executablePath) + ) { + return + } + } disposed = true options.onPromptReady() } diff --git a/src/main/ssh/build-toolchain-diagnosis.ts b/src/main/ssh/build-toolchain-diagnosis.ts index c64a41ead51..77d20ce475a 100644 --- a/src/main/ssh/build-toolchain-diagnosis.ts +++ b/src/main/ssh/build-toolchain-diagnosis.ts @@ -163,3 +163,66 @@ export function formatMissingToolchainError( ] return lines.join('\n') } + +const NODE_HEADERS_TARBALL_RE = /node-v[0-9.]+-headers\.tar\.gz/i + +/** + * Whether a native-deps failure is node-gyp failing to download Node headers from nodejs.org. + * + * Why it needs naming: the raw output is forty lines of `gyp http` and stack frames around one + * `ECONNREFUSED`, and it reads as a broken host or a broken Orca. Which of two things it is + * depends on what the local-headers export found first, so the formatter takes that answer. + */ +export function isNodeHeadersDownloadFailure(message: string): boolean { + // Why `configure error` is required: node-gyp's fetch client logs `attempt N failed with ` + // on retries it then recovers from, so a network token alone also matches a build that got its + // headers and died later for an unrelated reason. Only the configure step downloads headers. + return ( + /gyp ERR! configure error/i.test(message) && + NODE_HEADERS_TARBALL_RE.test(message) && + /\b(ECONNREFUSED|ENOTFOUND|ETIMEDOUT|EHOSTUNREACH|ENETUNREACH|EAI_AGAIN|ECONNRESET)\b/.test( + message + ) + ) +} + +const NODE_HEADERS_CONTEXT = + 'node-pty has no prebuilt binary for Linux, so it must be compiled on the remote host, and ' + + 'node-gyp fetches the Node.js headers from nodejs.org unless the Node install provides them ' + + 'at /include/node.' + +/** + * @param localHeadersDir what the local-headers export found: a dir it exported, `null` when + * the host's Node ships no matching headers, `undefined` when the answer never came back. + * + * Why the exported-dir case is its own message: the export is the fix, so node-gyp downloading + * anyway means its `nodedir` env keys were not honoured (a future npm dropping the passthrough, + * a wrapper scrubbing the env). That is an Orca defect, not a host problem, and must not be + * reported as one -- it names the dir so the report is checkable. + */ +export function formatNodeHeadersDownloadError( + underlyingError: string, + localHeadersDir: string | null | undefined +): string { + const lines = localHeadersDir + ? [ + `The remote host could not download the Node.js headers needed to compile node-pty, even ` + + `though its Node install ships matching headers at ${localHeadersDir}/include/node and ` + + `Orca pointed node-gyp at them. node-gyp ignored that setting; this is an Orca defect, ` + + `please report it with the log below.`, + '', + 'Workaround on the remote host until then: allow outbound HTTPS to nodejs.org, or point ' + + 'npm at a mirror: npm config set disturl https:///dist' + ] + : [ + 'The remote host could not download the Node.js headers needed to compile node-pty, and ' + + `its Node install has no local headers matching its own version. ${NODE_HEADERS_CONTEXT}`, + '', + 'Fix one of the following on the remote host, then reconnect:', + ' - Install Node.js from an official build or a version manager (nvm, fnm, volta, n), ' + + 'which ship headers for exactly the Node they run; or', + ' - Allow outbound HTTPS to nodejs.org, or point npm at a mirror: ' + + 'npm config set disturl https:///dist' + ] + return [...lines, '', `Underlying install error: ${underlyingError}`].join('\n') +} diff --git a/src/main/ssh/ssh-relay-build-toolchain.test.ts b/src/main/ssh/ssh-relay-build-toolchain.test.ts index b216b599f5a..6fcc4f5321d 100644 --- a/src/main/ssh/ssh-relay-build-toolchain.test.ts +++ b/src/main/ssh/ssh-relay-build-toolchain.test.ts @@ -3,7 +3,9 @@ import { buildToolchainProbeCommand, parseBuildToolchainProbe, formatMissingToolchainError, + formatNodeHeadersDownloadError, formatSkippedNodePtyWarning, + isNodeHeadersDownloadFailure, shouldProbeBuildToolchainAfterNativeDepsFailure } from './ssh-relay-build-toolchain' @@ -125,3 +127,71 @@ describe('formatSkippedNodePtyWarning', () => { expect(warning).toContain('install a C/C++ toolchain') }) }) + +// Verbatim shape of the STA-6674 failure: node-gyp on a host whose nodejs.org is refused. +const HEADERS_REFUSED = + 'npm error gyp http GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz\n' + + 'npm error gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNREFUSED\n' + + 'npm error gyp ERR! configure error\n' + + 'npm error gyp ERR! stack FetchError: request to https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz failed, reason: connect ECONNREFUSED 127.0.0.1:443' + +describe('isNodeHeadersDownloadFailure', () => { + it('matches node-gyp failing to fetch the Node headers tarball', () => { + expect(isNodeHeadersDownloadFailure(HEADERS_REFUSED)).toBe(true) + expect( + isNodeHeadersDownloadFailure( + 'gyp http fetch GET https://nodejs.org/download/release/v20.19.0/node-v20.19.0-headers.tar.gz attempt 1 failed with ENOTFOUND\ngyp ERR! configure error' + ) + ).toBe(true) + }) + + it('is not the toolchain diagnosis, and does not fire on other network failures', () => { + expect(shouldProbeBuildToolchainAfterNativeDepsFailure(HEADERS_REFUSED)).toBe(false) + // The registry, not nodejs.org: a different remedy. + expect( + isNodeHeadersDownloadFailure( + 'npm error network request to https://registry.npmjs.org/node-pty failed, reason: connect ECONNREFUSED' + ) + ).toBe(false) + // Headers named but the build failed for another reason. + expect( + isNodeHeadersDownloadFailure( + 'gyp info using node-v24.12.0-headers.tar.gz\ngyp ERR! build error make failed with exit code: 2' + ) + ).toBe(false) + // A retried attempt that recovered, then a compile failure: not a download failure. + expect( + isNodeHeadersDownloadFailure( + 'gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNRESET\n' + + 'gyp http 200 https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz\n' + + 'gyp ERR! build error\ngyp ERR! stack Error: `make` failed with exit code: 2' + ) + ).toBe(false) + // A mirror answering non-2xx is a FetchError without a network code: a different remedy. + expect( + isNodeHeadersDownloadFailure( + 'gyp ERR! configure error\ngyp ERR! stack FetchError: 404 Not Found https://mirror/dist/v24.12.0/node-v24.12.0-headers.tar.gz' + ) + ).toBe(false) + }) +}) + +describe('formatNodeHeadersDownloadError', () => { + it('names both host remedies when the host ships no headers', () => { + const msg = formatNodeHeadersDownloadError(HEADERS_REFUSED, null) + expect(msg).toContain('no local headers matching its own version') + expect(msg).toContain('/include/node') + expect(msg).toContain('nvm, fnm, volta, n') + expect(msg).toContain('disturl') + expect(msg).toContain('ECONNREFUSED') + }) + + it('reports an Orca defect, not a host problem, when headers were exported and ignored', () => { + const msg = formatNodeHeadersDownloadError(HEADERS_REFUSED, '/usr/local') + expect(msg).toContain('/usr/local/include/node') + expect(msg).toContain('Orca defect') + expect(msg).not.toContain('no local headers matching its own version') + expect(msg).not.toContain('nvm, fnm, volta, n') + expect(msg).toContain('ECONNREFUSED') + }) +}) diff --git a/src/main/ssh/ssh-relay-build-toolchain.ts b/src/main/ssh/ssh-relay-build-toolchain.ts index bcb64d3bdf9..db7b6353697 100644 --- a/src/main/ssh/ssh-relay-build-toolchain.ts +++ b/src/main/ssh/ssh-relay-build-toolchain.ts @@ -16,7 +16,9 @@ export { shouldProbeBuildToolchainAfterNativeDepsFailure, toolchainInstallHintLines, formatSkippedNodePtyWarning, - formatMissingToolchainError + formatMissingToolchainError, + formatNodeHeadersDownloadError, + isNodeHeadersDownloadFailure } from './build-toolchain-diagnosis' export type { BuildToolchainStatus } from './build-toolchain-diagnosis' diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index e7450478d92..5d8101361c6 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -53,11 +53,14 @@ import { } from './ssh-relay-deploy-timing' import { createSshOperationAbortError, shellEscape } from './ssh-connection-utils' import { isWindowsRelayPlatform } from '../../shared/relay-artifacts' +import { exportLocalNodeHeadersPrefix, localNodeHeadersFromOutput } from './ssh-relay-node-headers' import { probeBuildToolchain, formatMissingToolchainError, formatSkippedNodePtyWarning, - shouldProbeBuildToolchainAfterNativeDepsFailure + shouldProbeBuildToolchainAfterNativeDepsFailure, + formatNodeHeadersDownloadError, + isNodeHeadersDownloadFailure } from './ssh-relay-build-toolchain' import { commandWithNodePath, @@ -1174,7 +1177,7 @@ async function installNativeDeps( hostPlatform, nodePath, remoteDir, - `${resetPrefix}npm install --ignore-scripts=false --omit=dev --no-audit --no-fund ${installArgs} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}${resetPrefix}npm install --ignore-scripts=false --omit=dev --no-audit --no-fund ${installArgs} 2>&1` ) await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, @@ -1236,6 +1239,14 @@ async function installNativeDeps( return } } + // Why: either the local-headers export found nothing (a host both header-less and offline) or + // it did and node-gyp downloaded anyway (the export is broken) -- name which, or the log reads + // as a broken relay either way. + if (platform.startsWith('linux') && isNodeHeadersDownloadFailure(msg)) { + throw new Error(formatNodeHeadersDownloadError(msg, localNodeHeadersFromOutput(msg)), { + cause: err + }) + } throw err } @@ -1254,8 +1265,15 @@ async function installNativeDeps( throw err } signal?.throwIfAborted() + // Same diagnosis as the install catch: this fallback is non-fatal, so the log is the only + // place the offline-headers cause can reach anyone. + const rebuildMsg = (err as Error).message console.warn( - `[ssh-relay][NATIVE-DEPS-REBUILD-FAIL] npm rebuild native deps failed at ${remoteDir} (${platform}): ${(err as Error).message}` + `[ssh-relay][NATIVE-DEPS-REBUILD-FAIL] npm rebuild native deps failed at ${remoteDir} (${platform}): ${ + platform.startsWith('linux') && isNodeHeadersDownloadFailure(rebuildMsg) + ? formatNodeHeadersDownloadError(rebuildMsg, localNodeHeadersFromOutput(rebuildMsg)) + : rebuildMsg + }` ) } signal?.throwIfAborted() @@ -1347,7 +1365,7 @@ async function applyNodePtyMasterCloexecPatch( hostPlatform, nodePath, remoteDir, - `${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` ) const output = await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, @@ -1529,7 +1547,7 @@ async function rebuildNativeDeps( hostPlatform, nodePath, remoteDir, - `npm rebuild --ignore-scripts=false ${depNames.map(shellEscape).join(' ')} 2>&1` + `${exportLocalNodeHeadersPrefix(nodePath)}npm rebuild --ignore-scripts=false ${depNames.map(shellEscape).join(' ')} 2>&1` ) await execHostCommand(conn, hostPlatform, command, { timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, diff --git a/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts b/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts index 4d90a7c8289..e8616e7c230 100644 --- a/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-install-staged-upload.test.ts @@ -154,6 +154,69 @@ describe('installNativeDeps staged uploads', () => { expect(writeObservedAt).toBeLessThanOrEqual(npmInstallIdx) }) + it('exports the host Node headers dir to node-gyp on every command that can compile node-pty (STA-6674)', async () => { + const conn = makeMockConnection(sftpCapture) + // Install succeeds, the probe fails, the rebuild repairs it, then the cloexec patch rebuilds again. + feed(makeExecResponses({ npmInstall: 'ok', probe: 'missing', repairProbe: 'ok' })) + + await deployAndLaunchRelay(conn) + + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + const compiling = ['npm install', 'npm rebuild', 'node-pty-1.1.0-master-cloexec-patch.cjs'] + for (const compileStep of compiling) { + const command = commands.find((candidate) => candidate.includes(compileStep)) + expect(command, compileStep).toBeDefined() + // Both spellings: node-gyp 10 (Node 20) reads only npm_config_, node-gyp >= 11.4 prefers the other. + expect(command).toContain('export npm_config_nodedir=') + expect(command).toContain('npm_package_config_node_gyp_nodedir=') + // The export precedes the compile on the same command line, and only when the probe found headers. + expect(command!.indexOf('npm_config_nodedir')).toBeLessThan(command!.indexOf(compileStep)) + expect(command).toContain('node_version.h') + // The marker lands in the captured output, so a failure after it can say what was exported. + expect(command).toContain('echo "ORCA-NODE-HEADERS:${ORCA_NODE_HEADERS_DIR:-none}"') + } + }) + + // What execCommand actually rejects with: the whole command line (marker echo included) quoted + // ahead of the host's output. A fixture that omits the command hides the marker-parsing bug. + function rejectNpmInstallLikeExecCommand(hostOutput: string): void { + vi.mocked(execCommand).mockImplementationOnce(async (_conn, command) => { + throw new Error(`Command "${command}" failed (exit 1): ${hostOutput}`) + }) + } + const HEADERS_REFUSED = + 'npm error gyp http fetch GET https://nodejs.org/download/release/v24.12.0/node-v24.12.0-headers.tar.gz attempt 1 failed with ECONNREFUSED\nnpm error gyp ERR! configure error' + + it('names the fix when node-gyp cannot download headers and the host ships none (STA-6674)', async () => { + const conn = makeMockConnection(sftpCapture) + feed(makeStagedFirstInstallExecPrefix()) + rejectNpmInstallLikeExecCommand(`ORCA-NODE-HEADERS:none\n${HEADERS_REFUSED}`) + feed(['']) // clean stage root + + const error = await deployAndLaunchRelay(conn).catch((e: Error) => e) + expect((error as Error).message).toContain('could not download the Node.js headers') + expect((error as Error).message).toContain('no local headers matching its own version') + expect((error as Error).message).not.toContain('Orca defect') + expect((error as Error).message).toContain('ECONNREFUSED') + // A full toolchain: the toolchain probe must not run, and this is not a "build tools" error. + expect((error as Error).message).not.toContain('build tools') + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + expect(commands.some((command) => command.includes('command -v "$t"'))).toBe(false) + }) + + it('reports an Orca defect when headers were exported but node-gyp downloaded anyway', async () => { + // The marker says the export happened; a download after it means node-gyp never read the env. + const conn = makeMockConnection(sftpCapture) + feed(makeStagedFirstInstallExecPrefix()) + rejectNpmInstallLikeExecCommand(`ORCA-NODE-HEADERS:/usr/local\n${HEADERS_REFUSED}`) + feed(['']) // clean stage root + + const error = await deployAndLaunchRelay(conn).catch((e: Error) => e) + expect((error as Error).message).toContain('/usr/local/include/node') + expect((error as Error).message).toContain('Orca defect') + expect((error as Error).message).not.toContain('no local headers matching its own version') + }) + it('promotes only after the first-install lock is acquired', async () => { const conn = makeMockConnection(sftpCapture) feed(makeExecResponses({ npmInstall: 'ok', probe: 'ok' })) diff --git a/src/main/ssh/ssh-relay-node-headers.test.ts b/src/main/ssh/ssh-relay-node-headers.test.ts new file mode 100644 index 00000000000..84e9016738b --- /dev/null +++ b/src/main/ssh/ssh-relay-node-headers.test.ts @@ -0,0 +1,164 @@ +import { spawnSync } from 'node:child_process' +import { + chmodSync, + copyFileSync, + mkdtempSync, + mkdirSync, + rmSync, + symlinkSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import process from 'node:process' +import { afterEach, describe, expect, it } from 'vitest' +import { exportLocalNodeHeadersPrefix, localNodeHeadersFromOutput } from './ssh-relay-node-headers' + +const POSIX = process.platform !== 'win32' + +/** Runs the prefix under /bin/sh exactly as the relay does, then prints what node-gyp would see. */ +function runPrefix(nodePath: string): { + nodedir: string + pkgNodedir: string + marker: string | null | undefined +} { + const script = `${exportLocalNodeHeadersPrefix(nodePath)}printf '%s\\n%s\\n' "$npm_config_nodedir" "$npm_package_config_node_gyp_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { encoding: 'utf8' }) + expect(result.status).toBe(0) + const marker = localNodeHeadersFromOutput(result.stdout) + const [nodedir = '', pkgNodedir = ''] = result.stdout + .split('\n') + .filter((line) => !line.startsWith('ORCA-NODE-HEADERS:')) + return { nodedir, pkgNodedir, marker } +} + +/** A fake `/bin/node` whose `include/node/node_version.h` claims `version`. */ +function fakeNodePrefix(root: string, version: string): string { + const prefix = join(root, 'prefix') + mkdirSync(join(prefix, 'bin'), { recursive: true }) + mkdirSync(join(prefix, 'include', 'node'), { recursive: true }) + const [major, minor, patch] = version.split('.') + writeFileSync( + join(prefix, 'include', 'node', 'node_version.h'), + `#define NODE_MAJOR_VERSION ${major}\n#define NODE_MINOR_VERSION ${minor}\n#define NODE_PATCH_VERSION ${patch}\n` + ) + // Why a symlink to the real binary: the probe reads process.execPath, which Node resolves + // through symlinks -- so this stands in for `/usr/bin/node -> /opt/node/bin/node` shims too. + symlinkSync(process.execPath, join(prefix, 'bin', 'node')) + return join(prefix, 'bin', 'node') +} + +describe.skipIf(!POSIX)('exportLocalNodeHeadersPrefix', () => { + const roots: string[] = [] + afterEach(() => { + for (const root of roots.splice(0)) { + rmSync(root, { recursive: true, force: true }) + } + }) + + it('exports nodedir when the running Node ships headers for its own version', () => { + // The test runner's Node is an official build, so its prefix has include/node. + const prefix = dirname(dirname(process.execPath)) + const { nodedir, pkgNodedir, marker } = runPrefix(process.execPath) + expect(nodedir).toBe(prefix) + expect(pkgNodedir).toBe(prefix) + expect(marker).toBe(prefix) + }) + + it('leaves nodedir unset when the shipped headers are for another Node version', () => { + // A symlinked node resolves execPath to the real binary, whose prefix is the real one; so + // to stage a mismatch the probe must run a node whose execPath lands in the fake prefix. + // A copy does that. + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const prefix = join(root, 'prefix') + mkdirSync(join(prefix, 'bin'), { recursive: true }) + mkdirSync(join(prefix, 'include', 'node'), { recursive: true }) + writeFileSync( + join(prefix, 'include', 'node', 'node_version.h'), + '#define NODE_MAJOR_VERSION 1\n#define NODE_MINOR_VERSION 0\n#define NODE_PATCH_VERSION 0\n' + ) + const copied = join(prefix, 'bin', 'node') + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const { nodedir, pkgNodedir, marker } = runPrefix(copied) + expect(nodedir).toBe('') + expect(pkgNodedir).toBe('') + expect(marker).toBeNull() + }) + + it('leaves nodedir unset when the prefix has no headers at all', () => { + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const copied = join(root, 'bin', 'node') + mkdirSync(dirname(copied), { recursive: true }) + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const { nodedir } = runPrefix(copied) + expect(nodedir).toBe('') + }) + + it('follows a symlinked node to the install that owns the headers', () => { + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const shim = fakeNodePrefix(root, '0.0.0') + // The shim's own fake headers are ignored: execPath resolves to the real binary, and the + // real prefix's headers are the ones that match. + const { nodedir } = runPrefix(shim) + expect(nodedir).toBe(dirname(dirname(process.execPath))) + }) + + it('clears an inherited nodedir when the probe finds no matching headers', () => { + // A remote profile's stale nodedir must not survive past the version check. + const root = mkdtempSync(join(tmpdir(), 'orca-node-headers-')) + roots.push(root) + const copied = join(root, 'bin', 'node') + mkdirSync(dirname(copied), { recursive: true }) + copyFileSync(process.execPath, copied) + chmodSync(copied, 0o755) + const script = `${exportLocalNodeHeadersPrefix(copied)}printf '%s|%s|%s' "$npm_config_nodedir" "$NPM_CONFIG_NODEDIR" "$npm_package_config_node_gyp_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { + encoding: 'utf8', + env: { + ...process.env, + npm_config_nodedir: '/usr/stale-headers', + NPM_CONFIG_NODEDIR: '/usr/stale-headers', + npm_package_config_node_gyp_nodedir: '/usr/stale-headers' + } + }) + expect(result.status).toBe(0) + expect(result.stdout.split('\n').at(-1)).toBe('||') + }) + + it('does not fail the command line when node itself cannot run', () => { + const script = `${exportLocalNodeHeadersPrefix('/nonexistent/node')}echo "after:$npm_config_nodedir"` + const result = spawnSync('/bin/sh', ['-c', script], { encoding: 'utf8' }) + expect(result.status).toBe(0) + expect(result.stdout.trim()).toBe('ORCA-NODE-HEADERS:none\nafter:') + }) +}) + +describe('localNodeHeadersFromOutput', () => { + it('reads the host answer, not the copy of the marker echo quoted in an exec-failure head', () => { + // The real shape: execCommand quotes the whole command line, prefix included, before the output. + const command = `export PATH='/usr/local/bin':$PATH && cd '/root/.orca-remote/relay-x' && ${exportLocalNodeHeadersPrefix('/usr/local/bin/node')}npm install node-pty 2>&1` + const failed = (hostOutput: string): string => + `Command "${command}" failed (exit 1): ${hostOutput}` + expect( + localNodeHeadersFromOutput(failed('ORCA-NODE-HEADERS:none\ngyp ERR! configure error')) + ).toBeNull() + expect( + localNodeHeadersFromOutput(failed('ORCA-NODE-HEADERS:/usr/local\ngyp ERR! configure error')) + ).toBe('/usr/local') + // No host output at all after the head: the command copy alone must not count as a marker. + expect(localNodeHeadersFromOutput(failed(''))).toBeUndefined() + }) + + it('distinguishes an exported dir, an explicit none, and no marker at all', () => { + expect(localNodeHeadersFromOutput('x\nORCA-NODE-HEADERS:/usr/local\ngyp ERR!')).toBe( + '/usr/local' + ) + expect(localNodeHeadersFromOutput('ORCA-NODE-HEADERS:none\ngyp ERR!')).toBeNull() + expect(localNodeHeadersFromOutput('gyp ERR! only')).toBeUndefined() + }) +}) diff --git a/src/main/ssh/ssh-relay-node-headers.ts b/src/main/ssh/ssh-relay-node-headers.ts new file mode 100644 index 00000000000..a590bd40fca --- /dev/null +++ b/src/main/ssh/ssh-relay-node-headers.ts @@ -0,0 +1,97 @@ +/** + * Point node-gyp at the headers the host's Node install already ships, so compiling node-pty + * needs nothing from nodejs.org. + * + * Why: node-pty has no Linux prebuild, so every Linux relay compiles it, and node-gyp's default + * is to download `node-v-headers.tar.gz` before configuring. Every official Node build, and + * every version manager that unpacks one (nvm, fnm, volta, mise, n), already has those exact + * headers at `/include/node`. The download was the only step that needed the internet, + * so a firewalled host failed with ECONNREFUSED on work that never had to happen (STA-6674). + * + * Why both variables: node-gyp >= 11.4 prefers `npm_package_config_node_gyp_` and npm 11+ + * warns that arbitrary `npm_config_` is deprecated, but node-gyp 10 (bundled with Node 20) + * reads only `npm_config_`. Both together cover every Node the relay runs on. + * + * Why the version check: node-gyp trusts `nodedir` blindly, so a distro `/usr/include/node` left + * by an older headers package would be compiled against as-is. Whether that binding then misbehaves + * is not established (one measured run loaded a node-20-header build under node 24); refusing is + * the conservative default. A mismatch leaves the variables unset, which is today's path. + */ +import { shellEscape } from './ssh-connection-utils' + +/** Shell variable the probe answers into; namespaced so it cannot collide with npm's own. */ +const NODEDIR_SHELL_VAR = 'ORCA_NODE_HEADERS_DIR' + +/** + * Prints the running Node's install prefix when `/include/node/node_version.h` matches + * `process.versions.node`, and nothing otherwise. `process.execPath` is symlink-resolved, so a + * `/usr/bin/node` -> `/opt/node/bin/node` shim still finds `/opt/node/include`. + */ +export const LOCAL_NODE_HEADERS_PROBE_JS = [ + 'const p=require("path"),f=require("fs");', + 'const d=p.dirname(p.dirname(process.execPath));', + 'try{', + 'const h=f.readFileSync(p.join(d,"include","node","node_version.h"),"utf8");', + 'const v=["MAJOR","MINOR","PATCH"].map(k=>(h.match(new RegExp("#define NODE_"+k+"_VERSION ([0-9]+)"))||[])[1]).join(".");', + 'if(v===process.versions.node)process.stdout.write(d)', + '}catch{}' +].join('') + +/** + * Stdout marker naming what the probe found, printed before the compile so the answer is in the + * captured output of any failure that follows. `none` means no matching local headers. + */ +export const LOCAL_NODE_HEADERS_MARKER_PREFIX = 'ORCA-NODE-HEADERS:' + +/** + * POSIX-sh prefix (`...; `) that exports node-gyp's `nodedir` for the rest of the command line + * when the host's Node ships matching headers. Prepend to any command that may compile node-pty: + * `npm install`, `npm rebuild`, and the cloexec patch (its `npm rebuild` inherits the env). + */ +export function exportLocalNodeHeadersPrefix(nodePath: string): string { + const probe = `${shellEscape(nodePath)} -e ${shellEscape(LOCAL_NODE_HEADERS_PROBE_JS)} 2>/dev/null` + // Why the unset: a remote profile can already export a nodedir (a stale distro header dir), in + // either case npm accepts. Left alone it would bypass the version check above and compile + // against those headers. Deliberately env only: a `nodedir=` in ~/.npmrc is not reachable from here + // -- npm ignores an empty env override, and a CLI `--nodedir=` would also override the good + // export -- so an npmrc setting stays the operator's, as it was before this prefix existed. + return ( + `${NODEDIR_SHELL_VAR}=$(${probe}); ` + + `unset npm_config_nodedir NPM_CONFIG_NODEDIR npm_package_config_node_gyp_nodedir; ` + + `if [ -n "$${NODEDIR_SHELL_VAR}" ]; then ` + + `export npm_config_nodedir="$${NODEDIR_SHELL_VAR}" npm_package_config_node_gyp_nodedir="$${NODEDIR_SHELL_VAR}"; ` + + `fi; ` + + `echo "${LOCAL_NODE_HEADERS_MARKER_PREFIX}\${${NODEDIR_SHELL_VAR}:-none}"; ` + ) +} + +/** + * The headers dir the prefix exported, `null` when it found none, or `undefined` when the + * marker is absent (output truncated, or the command never reached the prefix). + */ +export function localNodeHeadersFromOutput(output: string): string | null | undefined { + // Why the head is stripped first: a failed exec's message is `Command "" failed + // (exit N): `, and quotes this prefix verbatim -- including the marker's + // `echo`. Scanning from the start would match that copy and return `${ORCA_NODE_HEADERS_DIR:- + // none}"...` as a "dir". Only what follows the head is the host's answer. + const head = output.match(EXEC_FAILURE_HEAD_RE) + const hostOutput = head ? output.slice(head[0].length) : output + // First match, not last: the host's own line comes first, and later lines are npm/gyp output + // that must not be able to spoof it. + for (const line of hostOutput.split(/\r?\n/)) { + const at = line.indexOf(LOCAL_NODE_HEADERS_MARKER_PREFIX) + if (at === -1) { + continue + } + const dir = line.slice(at + LOCAL_NODE_HEADERS_MARKER_PREFIX.length).trim() + return dir === 'none' || dir === '' ? null : dir + } + return undefined +} + +/** + * `Command "" failed (exit N): ` -- see ssh-relay-exec-command.ts. + * Lazy `[\s\S]*?` is safe: it stops at the first `" failed (exit N): `, and no command this + * module builds contains that literal, so the match cannot end early inside the command. + */ +const EXEC_FAILURE_HEAD_RE = /^Command "[\s\S]*?" failed \(exit -?\d+\): / diff --git a/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts b/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts new file mode 100644 index 00000000000..c5700c991c1 --- /dev/null +++ b/src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts @@ -0,0 +1,204 @@ +// Why this exists (STA-6674): a Linux host whose only unreachable endpoint is nodejs.org could +// not run a relay. node-pty ships no Linux prebuild, so npm hands it to node-gyp, and node-gyp +// downloads `node-v-headers.tar.gz` unless told the host already has the headers -- which +// every official Node install does, at `/include/node`. This drives the real deploy at a +// Docker sshd whose nodejs.org resolves to 127.0.0.1 (ECONNREFUSED, exactly what the user saw). +// +// Run: ORCA_REVIEW_SSH_OFFLINE_HEADERS=1 pnpm test src/main/ssh/ssh-relay-offline-node-headers.docker.test.ts +// Needs Docker and `pnpm build:relay`. ORCA_REVIEW_SSH_NODE_IMAGE picks the Node image +// (default node:24.12.0-bookworm, the user's version); ORCA_REVIEW_SSH_TARGET_HOST overrides +// the address the app connects to (default 127.0.0.1). +import { execFileSync, spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtempSync, readFileSync, rmSync } from 'node:fs' +import { connect } from 'node:net' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ app: { getAppPath: () => process.cwd() } })) + +import { SshConnection } from './ssh-connection' +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import type { SshTarget } from '../../shared/ssh-types' + +const RUN_REVIEW_ORACLE = process.env.ORCA_REVIEW_SSH_OFFLINE_HEADERS === '1' +const NODE_IMAGE = process.env.ORCA_REVIEW_SSH_NODE_IMAGE ?? 'node:24.12.0-bookworm' +const TARGET_HOST = process.env.ORCA_REVIEW_SSH_TARGET_HOST ?? '127.0.0.1' + +type TargetFixture = { + containerName: string + identityFile: string + port: number + tempDir: string +} + +function run(command: string, args: string[], timeout = 30_000, input?: string): string { + return execFileSync(command, args, { + encoding: 'utf8', + stdio: [input === undefined ? 'ignore' : 'pipe', 'pipe', 'pipe'], + timeout, + input + }).trim() +} + +function dockerExec(fixture: TargetFixture, command: string): string { + return run('docker', ['exec', fixture.containerName, 'bash', '-lc', command], 60_000) +} + +async function startTarget(): Promise { + const image = `orca-review-offline-headers:${NODE_IMAGE.replace(/[^A-Za-z0-9_.-]/g, '-')}` + run( + 'docker', + ['build', '-q', '-t', image, '-'], + 600_000, + [ + `FROM ${NODE_IMAGE}`, + 'RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends openssh-server git && rm -rf /var/lib/apt/lists/* && mkdir -p /run/sshd /root/.ssh && chmod 700 /root/.ssh', + '' + ].join('\n') + ) + const tempDir = mkdtempSync(join(tmpdir(), 'orca-offline-headers-ssh-')) + const identityFile = join(tempDir, 'id_ed25519') + run('ssh-keygen', ['-t', 'ed25519', '-N', '', '-f', identityFile, '-q']) + const publicKey = readFileSync(`${identityFile}.pub`, 'utf8').trim() + const containerName = `orca-offline-headers-${randomUUID().slice(0, 12)}` + // Why a refused connection and not a dropped one: a timeout takes node-gyp's retry path and + // burns the deploy budget; the user's host refused, and that is the path under test. + run( + 'docker', + [ + 'run', + '-d', + '--name', + containerName, + '--add-host', + 'nodejs.org:127.0.0.1', + '-p', + '0.0.0.0::22', + '-e', + `AUTHORIZED_KEY=${publicKey}`, + image, + 'bash', + '-lc', + 'printf "%s\\n" "$AUTHORIZED_KEY" > /root/.ssh/authorized_keys && chmod 600 /root/.ssh/authorized_keys && exec /usr/sbin/sshd -D -e' + ], + 120_000 + ) + const port = Number(run('docker', ['port', containerName, '22/tcp']).split(':').at(-1)) + // `docker run -d` returns before sshd binds; connect() against a closed port is a flake. + await waitForSshBanner(port) + return { containerName, identityFile, port, tempDir } +} + +/** Resolves once sshd answers with its banner on the mapped port, or throws after the deadline. */ +async function waitForSshBanner(port: number, deadlineMs = 60_000): Promise { + const deadline = Date.now() + deadlineMs + for (;;) { + const gotBanner = await new Promise((resolve) => { + const socket = connect({ host: TARGET_HOST, port }) + const done = (value: boolean): void => { + socket.destroy() + resolve(value) + } + socket.setTimeout(2_000, () => done(false)) + socket.once('data', (chunk) => done(chunk.toString('utf8').startsWith('SSH-'))) + socket.once('error', () => done(false)) + }) + if (gotBanner) { + return + } + if (Date.now() > deadline) { + throw new Error(`sshd on port ${port} did not answer within ${deadlineMs / 1000}s`) + } + await new Promise((resolve) => setTimeout(resolve, 500)) + } +} + +function stopTarget(fixture: TargetFixture | null): void { + if (!fixture) { + return + } + spawnSync('docker', ['rm', '-f', fixture.containerName], { stdio: 'ignore', timeout: 30_000 }) + rmSync(fixture.tempDir, { recursive: true, force: true }) +} + +function createConnection(fixture: TargetFixture): SshConnection { + const target: SshTarget = { + id: `offline-headers-${randomUUID()}`, + label: 'Offline node headers Docker SSH target', + source: 'manual', + host: TARGET_HOST, + port: fixture.port, + username: 'root', + identityFile: fixture.identityFile, + identitiesOnly: true + } + return new SshConnection(target, { onStateChange: vi.fn() }) +} + +describe.skipIf(!RUN_REVIEW_ORACLE)( + 'SSH relay deploy on a host that cannot reach nodejs.org', + () => { + let fixture: TargetFixture | null = null + + beforeAll(async () => { + fixture = await startTarget() + }, 900_000) + + afterAll(() => { + stopTarget(fixture) + }) + + it('compiles node-pty from the host Node install headers instead of downloading them', async () => { + const activeFixture = fixture as TargetFixture + expect(dockerExec(activeFixture, 'getent hosts nodejs.org')).toContain('127.0.0.1') + const connection = createConnection(activeFixture) + await connection.connect() + try { + const result = await deployAndLaunchRelay(connection, undefined, 60) + expect(result.remoteRelayDir).toBeTruthy() + + const evidence = dockerExec( + activeFixture, + [ + `cd '${result.remoteRelayDir}'`, + 'test -f node_modules/node-pty/build/Release/pty.node && echo PTY_NODE=built', + 'test -d /root/.cache/node-gyp && echo HEADERS=downloaded || echo HEADERS=local', + `node -e "require('node-pty'); require('@parcel/watcher'); console.log('NATIVE=loadable')"` + ].join('; ') + ) + console.log(`[offline-node-headers] ${NODE_IMAGE}: ${evidence.replace(/\n/g, ' ')}`) + expect(evidence).toContain('PTY_NODE=built') + expect(evidence).toContain('HEADERS=local') + expect(evidence).toContain('NATIVE=loadable') + } finally { + await connection.disconnect() + } + }, 600_000) + + it('names the missing-local-headers cause, not an Orca defect, when the host ships no headers', async () => { + // Same offline host, headers removed and the relay uninstalled so the deploy compiles again. + // This is the shape a review found misreported: the exec-failure message quotes the whole + // command (marker echo included) ahead of the output, and the parser must not read that copy. + const activeFixture = fixture as TargetFixture + dockerExec( + activeFixture, + 'rm -rf /usr/local/include/node /root/.orca-remote /root/.cache/node-gyp' + ) + const connection = createConnection(activeFixture) + await connection.connect() + try { + const error = await deployAndLaunchRelay(connection, undefined, 60).catch((e: Error) => e) + expect(error).toBeInstanceOf(Error) + const message = (error as Error).message + console.log(`[offline-node-headers] ${NODE_IMAGE} no-headers: ${message.split('\n')[0]}`) + expect(message).toContain('no local headers matching its own version') + expect(message).not.toContain('Orca defect') + expect(message).toContain('ECONNREFUSED') + } finally { + await connection.disconnect() + } + }, 600_000) + } +) diff --git a/src/main/startup/main-window-actions.ts b/src/main/startup/main-window-actions.ts index 585fa0a754e..0acc8d1a074 100644 --- a/src/main/startup/main-window-actions.ts +++ b/src/main/startup/main-window-actions.ts @@ -16,7 +16,10 @@ import { describeInstallDirAclPoison, isBlockingInstallDirAclRepairInFlight } from './windows-install-dir-acl-recovery' -import { presentRendererRecoveryPrompt } from '../window/renderer-recovery-prompt' +import { + presentRendererRecoveryPrompt, + type RendererRecoveryPromptFailure +} from '../window/renderer-recovery-prompt' // The window module injects this callback to avoid a cycle between actions and lifecycle code. let openWindow: (options?: { revealOnDidFinishLoad?: boolean }) => BrowserWindow @@ -147,9 +150,14 @@ export function sendOpenCrashReport(targetWindow?: BrowserWindow | null): void { } // Why: on renderer crash-loop the breaker stops auto-reloading and the window goes blank, so a main-process dialog is the only retry/quit surface. -export async function showRendererRecoveryPrompt(recentRecoveryCount: number): Promise { +export async function showRendererRecoveryPrompt( + recentRecoveryCount: number, + failure?: RendererRecoveryPromptFailure, + retry?: () => void +): Promise { await presentRendererRecoveryPrompt({ recentRecoveryCount, + ...(failure ? { failure } : {}), isQuitting: () => state.isQuitting, diagnose: describeInstallDirAclPoison, showMessageBox: (options) => { @@ -164,6 +172,12 @@ export async function showRendererRecoveryPrompt(recentRecoveryCount: number): P } recordDurableCrashBreadcrumb('renderer_recovery_manual_retry') // Why: leave the breaker open so a re-crash re-raises this prompt instead of resuming the auto-reload loop. + // Why watched: Reload is the dialog's default button, and an unwatched retry that stalls returns the user to + // the same silent hang with no further prompt — the watchdog re-raises this dialog instead. + if (retry) { + retry() + return + } loadMainWindow(state.mainWindow) }, quit: () => { diff --git a/src/main/startup/main-window-controller.ts b/src/main/startup/main-window-controller.ts index 63d84df763c..8ea245fc256 100644 --- a/src/main/startup/main-window-controller.ts +++ b/src/main/startup/main-window-controller.ts @@ -113,13 +113,19 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} reason: details.reason, expectedTeardown: getExpectedTeardownScope(webContentsId, false) }), - onRendererRecoveryExhausted: ({ details, recentRecoveryCount }) => { - recordDurableCrashBreadcrumb('renderer_recovery_circuit_breaker_open', { - reason: details.reason, - exitCode: details.exitCode ?? null, - recentRecoveryCount - }) - void showRendererRecoveryPrompt(recentRecoveryCount) + onRendererRecoveryExhausted: ({ details, recentRecoveryCount, cause, retry }) => { + // Why two names: a stalled reload never opened the breaker, and a bundle that says it did misreads the failure. + recordDurableCrashBreadcrumb( + cause === 'reload-stalled' + ? 'renderer_recovery_reload_exhausted' + : 'renderer_recovery_circuit_breaker_open', + { + reason: details.reason, + exitCode: details.exitCode ?? null, + recentRecoveryCount + } + ) + void showRendererRecoveryPrompt(recentRecoveryCount, cause, retry) }, deferLoad: true, ...(options.revealOnDidFinishLoad === true ? { revealOnDidFinishLoad: true } : {}), @@ -131,9 +137,16 @@ export function openMainWindow(options: { revealOnDidFinishLoad?: boolean } = {} } recordCrashBreadcrumb('manual_reload_requested', { ignoreCache }) }, - onBeforeRecoveryReload: (webContentsId) => { + // Manual retries also preserve PTYs, but have their own intent breadcrumb. + onBeforeRecoveryReload: (webContentsId, trigger) => { markRecoveryReloadInFlight(webContentsId) - recordDurableCrashBreadcrumb('renderer_recovery_reload') + if (trigger === 'automatic') { + recordDurableCrashBreadcrumb('renderer_recovery_reload') + } + }, + // Pair the intent breadcrumb with its path-free outcome. + onRecoveryReloadOutcome: ({ status, ...outcome }) => { + recordDurableCrashBreadcrumb(`renderer_recovery_reload_${status}`, outcome) } }) recordCrashBreadcrumb('main_window_created') diff --git a/src/main/window/createMainWindow-close-confirmation.test.ts b/src/main/window/createMainWindow-close-confirmation.test.ts index a8c9f70c503..6f91dd4add8 100644 --- a/src/main/window/createMainWindow-close-confirmation.test.ts +++ b/src/main/window/createMainWindow-close-confirmation.test.ts @@ -73,8 +73,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const onQuitAborted = vi.fn() browserWindowMock.mockImplementation(function () { @@ -122,8 +122,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -178,8 +178,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn(), + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()), close: vi.fn(() => { windowHandlers.close({} as never) }) @@ -238,8 +238,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const updateUI = vi.fn() const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) @@ -296,8 +296,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -353,8 +353,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -401,8 +401,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -451,8 +451,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), destroy, - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) createMainWindow(null, { getIsQuitting: () => true }) @@ -500,8 +500,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), destroy, - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) createMainWindow(null, { getIsQuitting: () => true }) diff --git a/src/main/window/createMainWindow-markdown-editor-focus.test.ts b/src/main/window/createMainWindow-markdown-editor-focus.test.ts index 3189019f837..ef6fca229d3 100644 --- a/src/main/window/createMainWindow-markdown-editor-focus.test.ts +++ b/src/main/window/createMainWindow-markdown-editor-focus.test.ts @@ -54,8 +54,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -104,8 +104,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -157,8 +157,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -216,8 +216,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -275,8 +275,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -355,8 +355,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -432,8 +432,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -496,8 +496,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts b/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts new file mode 100644 index 00000000000..e551579a7ab --- /dev/null +++ b/src/main/window/createMainWindow-recovery-reload-watchdog.test.ts @@ -0,0 +1,671 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as DurableCrashBreadcrumbModule from '../crash-reporting/durable-crash-breadcrumb' + +const { recordDurableCrashBreadcrumbMock } = vi.hoisted(() => ({ + recordDurableCrashBreadcrumbMock: vi.fn() +})) +vi.mock('../crash-reporting/durable-crash-breadcrumb', async (importOriginal) => ({ + ...(await importOriginal()), + recordDurableCrashBreadcrumb: recordDurableCrashBreadcrumbMock +})) + +vi.mock('electron', async () => + (await import('./createMainWindow-test-harness')).electronModuleMock() +) +vi.mock('@electron-toolkit/utils', async () => + (await import('./createMainWindow-test-harness')).electronToolkitUtilsMock() +) +vi.mock('./macos-tahoe-release', async () => + (await import('./createMainWindow-test-harness')).macosTahoeReleaseMock() +) +vi.mock('../app-icon', async () => (await import('./createMainWindow-test-harness')).appIconMock()) +vi.mock('../browser/browser-manager', async () => + (await import('./createMainWindow-test-harness')).browserManagerMock() +) +vi.mock('../browser/browser-client-page-renderer-runtime', async () => { + const harness = await import('./createMainWindow-test-harness') + return { + attachBrowserClientPageRenderer: harness.attachClientPageRendererMock, + retireBrowserClientPageRenderer: harness.retireClientPageRendererMock + } +}) + +import { createMainWindow } from './createMainWindow' +import { + browserWindowMock, + isMock, + powerMonitorOnMock, + resetMainWindowMocks +} from './createMainWindow-test-harness' +import { + RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS, + RENDERER_RECOVERY_LOAD_TIMEOUT_MS +} from './renderer-recovery-reload-watchdog' + +const DOCUMENT_URL = 'file:///opt/orca/renderer/index.html' +// A real macOS install URL: the crash-report redactor's PATH_PATTERNS provably leave this one intact. +const INSTALL_PATH_LOAD_ERROR = + "ERR_FILE_NOT_FOUND (-6) loading 'file:///Users/jane.doe/Applications/Orca.app/Contents/Resources/app.asar/out/renderer/index.html'" +const CRASH = { reason: 'crashed', exitCode: 5 } as Electron.RenderProcessGoneDetails + +/** + * Regression cover for the field failure: the recovery reload is issued, never produces a document, and nothing + * notices — no did-fail-load, no breaker (it counts renderer deaths only), no retry, no prompt. + */ +describe('renderer recovery reload watchdog', () => { + beforeEach(() => { + resetMainWindowMocks() + recordDurableCrashBreadcrumbMock.mockClear() + vi.useFakeTimers() + }) + + const createHarness = () => { + // Why fan-out: dom-ready and did-finish-load have several real registrants on this one webContents, so + // last-writer-wins would silently drop the watchdog's listener if registration order ever changed. + const registered: Record void)[]> = {} + const windowHandlers: Record void> = {} + const register = (event: string, handler: (...args: any[]) => void): void => { + const handlers = (registered[event] ??= []) + handlers.push(handler) + windowHandlers[event] ??= (...args: any[]) => { + for (const listener of handlers.slice()) { + listener(...args) + } + } + } + // Loads stay pending unless a test settles one: that is exactly the stall being reproduced. + const settleLoad: { resolve: () => void; reject: (error: Error) => void }[] = [] + const pendingLoad = (): Promise => + new Promise((resolve, reject) => settleLoad.push({ resolve, reject })) + const webContents = { + id: 143, + getURL: vi.fn(() => DOCUMENT_URL), + isDestroyed: vi.fn(() => false), + on: vi.fn(register), + setZoomLevel: vi.fn(), + setBackgroundThrottling: vi.fn(), + invalidate: vi.fn(), + setWindowOpenHandler: vi.fn(), + send: vi.fn() + } + const browserWindowInstance = { + webContents, + on: vi.fn(register), + isDestroyed: vi.fn(() => false), + isMaximized: vi.fn(() => true), + isFullScreen: vi.fn(() => false), + getSize: vi.fn(() => [1200, 800]), + setSize: vi.fn(), + maximize: vi.fn(), + show: vi.fn(), + setWindowButtonPosition: vi.fn(), + loadFile: vi.fn(pendingLoad), + loadURL: vi.fn(pendingLoad) + } + browserWindowMock.mockImplementation(function () { + return browserWindowInstance + }) + const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) + const crashRenderer = (): void => { + windowHandlers['render-process-gone']?.({} as never, CRASH) + vi.advanceTimersByTime(250) + } + const reachMilestone = (milestone: 'committed' | 'dom-ready'): void => + windowHandlers[milestone === 'committed' ? 'did-navigate' : 'dom-ready']?.() + return { + browserWindowInstance, + consoleError, + crashRenderer, + reachMilestone, + settleLoad, + windowHandlers + } + } + + it('retries once when the recovery reload never produces a document, then hands the user the prompt', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + // 1 initial load + 1 recovery reload, which now stalls forever. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(1) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS, + progress: 'none' + }) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 2 }) + ) + // Retry budget spent: stop reloading and surface the only retry/quit surface the user has. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith({ + details: CRASH, + webContentsId: 143, + recentRecoveryCount: 1, + cause: 'reload-stalled', + retry: expect.any(Function) + }) + + consoleError.mockRestore() + }) + + it('clears the watchdog when the recovery reload finishes loading', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(2_000) + settleLoad[1]?.resolve() + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'loaded', + attempt: 1, + elapsedMs: 2_000 + }) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 3) + expect(onRecoveryReloadOutcome).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + consoleError.mockRestore() + }) + + it('keeps watching the retry when a stale did-finish-load arrives after it was issued', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // did-finish-load carries no attempt token: this one belongs to the load the timer just abandoned. Crediting + // the retry with it disarms the watchdog over a load still in flight — the exact hole this watchdog closes. + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 2 }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('does not take an error page as the retry landing', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_FILE_NOT_FOUND (-6)')) + await vi.advanceTimersByTimeAsync(0) + // Chromium commits an error document for the failed load, and that document emits did-finish-load too. + windowHandlers['did-finish-load']?.() + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('raises one prompt, however many times recovery gives up underneath it', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // The renderer dies again while the box is up; the breaker never counted stalls, so it lets the reload go. + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + + // Nothing dismisses a native message box: a retry the user never asked for, or a second box, stacks on it. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + // The stall is still on the record, so the bundle does not read as a recovery that quietly worked. + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + + // Answering the box with Reload hands the next verdict back to the user. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + + it('still reloads from a crash-loop prompt raised after an earlier recovery had landed', async () => { + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + // Every recovery reload lands, and every landed document then dies with its renderer. + for (let attempt = 1; attempt <= 3; attempt += 1) { + crashRenderer() + settleLoad[attempt]?.resolve() + await vi.advanceTimersByTimeAsync(0) + } + crashRenderer() + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + const loads = browserWindowInstance.loadFile.mock.calls.length + + // The last document landed, but the renderer took it down: declining Reload here strands the user. + + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(loads + 1) + + consoleError.mockRestore() + }) + + it('does not stack a crash-loop prompt on one that is already up', () => { + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + for (let attempt = 0; attempt < 5; attempt += 1) { + crashRenderer() + } + + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('escalates a rejected recovery load immediately instead of waiting out the watchdog', async () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + settleLoad[1]?.reject(new Error("ERR_FILE_NOT_FOUND (-6) loading 'file:///opt/orca'")) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'failed', + attempt: 1, + errorCode: 'ERR_FILE_NOT_FOUND' + }) + ) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('ignores a superseded load rejection so ERR_ABORTED never escalates', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + // A second renderer death supersedes the first reload; Chromium rejects the abandoned load with ERR_ABORTED. + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + settleLoad[1]?.reject(new Error('ERR_ABORTED (-3)')) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('does not escalate when another navigation aborts the live recovery load', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad, windowHandlers } = + createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + // Chromium aborts the recovery load because something else replaced it — a user navigation, a close race, + // another loadURL caller. The attempt token still says this reload is live, so nothing else filters it. + settleLoad[1]?.reject(new Error(`ERR_ABORTED (-3) loading '${DOCUMENT_URL}'`)) + await vi.advanceTimersByTimeAsync(0) + + // A cold retry here would stomp the load that superseded this one. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + // The replacement load lands, and the window the user sees was never worth a Reload/Quit prompt. The crumb + // says so: elapsedMs measures the replacement, and the budget analysis has to be able to leave it out. + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', attempt: 1, superseded: true }) + ) + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + windowHandlers['did-finish-load']?.() + expect(onRecoveryReloadOutcome).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('still escalates on silence when an aborted recovery load has nothing behind it', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_ABORTED (-3)')) + await vi.advanceTimersByTimeAsync(0) + + // Ignoring the abort must not disarm the watchdog: the cap still bounds a load that goes nowhere. + await vi.advanceTimersByTimeAsync(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'reload-stalled' }) + ) + + consoleError.mockRestore() + }) + + it('gives the dev server a longer budget than a packaged load', () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + isMock.dev = true + vi.stubEnv('ELECTRON_RENDERER_URL', 'http://localhost:5173/') + + try { + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + expect(browserWindowInstance.loadURL).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + vi.advanceTimersByTime(1) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'timeout', attempt: 1 }) + ) + } finally { + vi.unstubAllEnvs() + consoleError.mockRestore() + } + }) + + it('stays silent when the stalled window is already closing', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, windowHandlers } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + windowHandlers.close?.({ preventDefault: vi.fn() } as never) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + it('keeps the install path out of the outcome breadcrumb', async () => { + const onRecoveryReloadOutcome = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + settleLoad[1]?.reject(new Error(INSTALL_PATH_LOAD_ERROR)) + await vi.advanceTimersByTimeAsync(0) + + const outcome = onRecoveryReloadOutcome.mock.calls[0]?.[0] + expect(outcome).toEqual({ + status: 'failed', + attempt: 1, + elapsedMs: 0, + progress: 'none', + errorCode: 'ERR_FILE_NOT_FOUND' + }) + // sanitizeCrashReportString cannot redact a file:///Users/... URL, so nothing path-shaped may reach the crumb. + expect(JSON.stringify(outcome)).not.toContain('/') + + consoleError.mockRestore() + }) + + it('records a durable breadcrumb for a rejected load, since console output never reaches the bundle', async () => { + const { consoleError, settleLoad } = createHarness() + + createMainWindow(null, {}) + settleLoad[0]?.reject(new Error(INSTALL_PATH_LOAD_ERROR)) + await vi.advanceTimersByTimeAsync(0) + + // Catching the rejection retired the main_unhandled_rejection crumb this used to produce. + expect(recordDurableCrashBreadcrumbMock).toHaveBeenCalledWith('main_window_load_failed', { + errorCode: 'ERR_FILE_NOT_FOUND' + }) + + consoleError.mockRestore() + }) + + it('escalates to the prompt when both attempts are rejected outright', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + settleLoad[1]?.reject(new Error('ERR_CONNECTION_REFUSED (-102)')) + await vi.advanceTimersByTimeAsync(0) + settleLoad[2]?.reject(new Error('ERR_CONNECTION_REFUSED (-102)')) + await vi.advanceTimersByTimeAsync(0) + + expect(onRecoveryReloadOutcome).toHaveBeenLastCalledWith( + expect.objectContaining({ status: 'failed', attempt: 2, errorCode: 'ERR_CONNECTION_REFUSED' }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'reload-stalled', recentRecoveryCount: 1 }) + ) + + consoleError.mockRestore() + }) + + it('hands the prompt a watched retry so a stalled manual reload re-raises it', () => { + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + // Reload is the dialog's default button; unwatched it returned the user to the same unbounded silent hang. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(5) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(2) + + consoleError.mockRestore() + }) + + it('names the crash-loop cause and gives that prompt a watched retry too', () => { + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRendererRecoveryExhausted }) + for (let attempt = 0; attempt < 4; attempt += 1) { + crashRenderer() + } + + expect(onRendererRecoveryExhausted).toHaveBeenCalledWith( + expect.objectContaining({ cause: 'crash-loop' }) + ) + expect(typeof onRendererRecoveryExhausted.mock.calls[0]?.[0].retry).toBe('function') + + consoleError.mockRestore() + }) + + it('restarts the stall budget when the machine resumes mid-load', () => { + const onRecoveryReloadOutcome = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome }) + crashRenderer() + // Sleep freezes the timer; on wake it would otherwise fire against a load that never got its budget. + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + const resume = powerMonitorOnMock.mock.calls.find(([event]) => event === 'resume')?.[1] as ( + ...args: unknown[] + ) => void + resume() + + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS - 1) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + + vi.advanceTimersByTime(1) + // Why the full span: rewriting the issue time on resume publishes time-since-wake into the bundle, which is + // silently wrong on any laptop — the outcome crumb exists to be honest about how long the load actually ran. + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2 - 1 + }) + ) + + consoleError.mockRestore() + }) + + it('never restarts a load that reached a document, and gives it the rest of the cap', () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, reachMilestone } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(10_000) + reachMilestone('committed') + + // 'no did-finish-load yet' is not a stall: a cold restart here throws away a load that already committed, and + // a machine that would have landed at ~60s misses the budget entirely. + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + + reachMilestone('dom-ready') + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS) + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'timeout', + attempt: 1, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2, + progress: 'dom-ready' + }) + // Still never restarted, and the cap keeps the ~90s worst case the no-document path already had. + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + consoleError.mockRestore() + }) + + it('records a reload that lands after the prompt, and leaves the recovered window alone', async () => { + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { browserWindowInstance, consoleError, crashRenderer, settleLoad } = createHarness() + + createMainWindow(null, { onRecoveryReloadOutcome, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledTimes(1) + + onRecoveryReloadOutcome.mockClear() + vi.advanceTimersByTime(30_000) + settleLoad[2]?.resolve() + await vi.advanceTimersByTimeAsync(0) + + // Nothing cancels a pending Chromium load, so escalation must keep watching: a bundle that reads + // `exhausted` for a recovery that actually worked misleads the next triage round. + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith({ + status: 'loaded', + attempt: 2, + elapsedMs: RENDERER_RECOVERY_LOAD_TIMEOUT_MS + 30_000, + afterPrompt: true + }) + + // No API dismisses a native message box, so Reload is still aimed at a window that came back; taking it + // would destroy the session the recovery just restored. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(3) + + consoleError.mockRestore() + }) + + it('separates the automatic recovery reload from the prompt-driven retry', () => { + const onBeforeRecoveryReload = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const { consoleError, crashRenderer } = createHarness() + + createMainWindow(null, { onBeforeRecoveryReload, onRendererRecoveryExhausted }) + crashRenderer() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + + // The field counts keyed on renderer_recovery_reload mean 'automatic recovery'; a manual retry recorded + // under the same name silently redefines them. + expect(onBeforeRecoveryReload.mock.calls.map(([, trigger]) => trigger)).toEqual([ + 'automatic', + 'automatic', + 'manual-retry' + ]) + + consoleError.mockRestore() + }) + + it('keeps a shutdown-aborted load out of the crash breadcrumb stream', async () => { + const { consoleError, settleLoad } = createHarness() + + createMainWindow(null, {}) + settleLoad[0]?.reject(new Error(`ERR_ABORTED (-3) loading '${DOCUMENT_URL}'`)) + await vi.advanceTimersByTimeAsync(0) + + // A quit or close aborts the in-flight startup load; a healthy shutdown must not look like a launch failure. + expect(recordDurableCrashBreadcrumbMock).not.toHaveBeenCalledWith( + 'main_window_load_failed', + expect.anything() + ) + + consoleError.mockRestore() + }) +}) diff --git a/src/main/window/createMainWindow-renderer-crash-recovery.test.ts b/src/main/window/createMainWindow-renderer-crash-recovery.test.ts index 43dac231130..b3fc210be2c 100644 --- a/src/main/window/createMainWindow-renderer-crash-recovery.test.ts +++ b/src/main/window/createMainWindow-renderer-crash-recovery.test.ts @@ -21,7 +21,7 @@ vi.mock('../browser/browser-client-page-renderer-runtime', async () => { } }) -import { createMainWindow, loadMainWindow } from './createMainWindow' +import { createMainWindow } from './createMainWindow' import { ipcMain } from 'electron' import { shouldRecoverRendererAfterProcessGone } from '../crash-reporting/process-gone-classification' import { @@ -74,8 +74,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -120,8 +120,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -231,8 +231,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -272,8 +272,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -322,8 +322,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) browserWindowMock.mockImplementation(function () { @@ -353,6 +353,8 @@ describe('createMainWindow', () => { const windowHandlers: Record void> = {} const webContents = { id: 143, + getURL: vi.fn(() => 'file:///opt/orca/renderer/index.html'), + isDestroyed: vi.fn(() => false), on: vi.fn((event, handler) => { windowHandlers[event] = handler }), @@ -374,8 +376,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -415,10 +417,12 @@ describe('createMainWindow', () => { const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {}) const { browserWindowInstance, windowHandlers } = createRendererRecoveryWindowHarness() const onBeforeRecoveryReload = vi.fn() + const onRendererRecoveryExhausted = vi.fn() withPlatform('win32', () => { createMainWindow(null, { onBeforeRecoveryReload, + onRendererRecoveryExhausted, shouldRecoverRenderer: (details) => shouldRecoverRendererAfterProcessGone({ reason: details.reason, @@ -436,10 +440,14 @@ describe('createMainWindow', () => { {} as never, { reason: 'killed', exitCode: 1 } as Electron.RenderProcessGoneDetails ) - vi.runAllTimers() + vi.advanceTimersByTime(250) - expect(onBeforeRecoveryReload).toHaveBeenCalledWith(143) + expect(onBeforeRecoveryReload).toHaveBeenCalledWith(143, 'automatic') expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(2) + // Why the watchdog must stay quiet here: this reload is deliberate during logoff, and a process that + // outlives the session-end signal must not put a native modal on screen mid-teardown. + vi.runAllTimers() + expect(onRendererRecoveryExhausted).not.toHaveBeenCalled() consoleError.mockRestore() }) @@ -615,8 +623,8 @@ describe('createMainWindow', () => { // 1 initial load + 3 recoveries; the 4th crash was refused. expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(4) - // The recovery prompt's Reload button goes straight to loadMainWindow, which the breaker never gates. - loadMainWindow(browserWindowInstance as unknown as Electron.BrowserWindow) + // The recovery prompt's Reload button takes the watched retry, which the breaker never gates. + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() expect(browserWindowInstance.loadFile).toHaveBeenCalledTimes(5) // Still-poisoned machine: the next crash re-raises the prompt immediately instead of re-arming auto-reloads. diff --git a/src/main/window/createMainWindow-startup-reveal.test.ts b/src/main/window/createMainWindow-startup-reveal.test.ts index 1a623ab20e3..1200132d319 100644 --- a/src/main/window/createMainWindow-startup-reveal.test.ts +++ b/src/main/window/createMainWindow-startup-reveal.test.ts @@ -57,8 +57,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-system-resume-relay.test.ts b/src/main/window/createMainWindow-system-resume-relay.test.ts index e51f1be0f07..71bf0cd7381 100644 --- a/src/main/window/createMainWindow-system-resume-relay.test.ts +++ b/src/main/window/createMainWindow-system-resume-relay.test.ts @@ -56,8 +56,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return instance diff --git a/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts b/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts index 9b88c60800c..8287b9bcecb 100644 --- a/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts +++ b/src/main/window/createMainWindow-terminal-focus-shortcuts.test.ts @@ -51,8 +51,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -143,8 +143,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -237,8 +237,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -302,8 +302,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -366,8 +366,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -424,8 +424,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -486,8 +486,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -565,8 +565,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -630,8 +630,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -707,8 +707,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -777,8 +777,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow-tray-minimize-close.test.ts b/src/main/window/createMainWindow-tray-minimize-close.test.ts index 14f829b63d9..469d8313830 100644 --- a/src/main/window/createMainWindow-tray-minimize-close.test.ts +++ b/src/main/window/createMainWindow-tray-minimize-close.test.ts @@ -81,8 +81,8 @@ describe('createMainWindow', () => { maximize: vi.fn(), show: vi.fn(), hide: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return instance diff --git a/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts b/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts index d80349a52df..9d0c3cc9048 100644 --- a/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts +++ b/src/main/window/createMainWindow-zoom-and-tab-switch-shortcuts.test.ts @@ -51,8 +51,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -116,8 +116,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -158,8 +158,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -206,8 +206,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -252,8 +252,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -300,8 +300,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -375,8 +375,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -423,8 +423,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -484,8 +484,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow.test.ts b/src/main/window/createMainWindow.test.ts index 18f4dbf5592..79b9a742533 100644 --- a/src/main/window/createMainWindow.test.ts +++ b/src/main/window/createMainWindow.test.ts @@ -78,8 +78,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -140,8 +140,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -320,8 +320,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } }) @@ -379,8 +379,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -424,8 +424,8 @@ describe('createMainWindow', () => { setWindowButtonPosition: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -479,8 +479,8 @@ describe('createMainWindow', () => { }), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -554,8 +554,8 @@ describe('createMainWindow', () => { }), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -630,8 +630,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -678,8 +678,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance @@ -726,8 +726,8 @@ describe('createMainWindow', () => { setSize: vi.fn(), maximize: vi.fn(), show: vi.fn(), - loadFile: vi.fn(), - loadURL: vi.fn() + loadFile: vi.fn(() => Promise.resolve()), + loadURL: vi.fn(() => Promise.resolve()) } browserWindowMock.mockImplementation(function () { return browserWindowInstance diff --git a/src/main/window/createMainWindow.ts b/src/main/window/createMainWindow.ts index 085cc475d49..41d84d1ab93 100644 --- a/src/main/window/createMainWindow.ts +++ b/src/main/window/createMainWindow.ts @@ -14,7 +14,8 @@ import { installMainWindowCloseLifecycle, WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS } from './main-window-close-lifecycle' -import type { CreateMainWindowOptions } from './main-window-contracts' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' +import { mainWindowLoadErrorCode } from './main-window-load-error-code' import { installMainWindowFocusLifecycle } from './main-window-focus-lifecycle' import { installMainWindowShortcutRouting } from './main-window-shortcut-routing' import { installMainWindowStateLifecycle } from './main-window-state-lifecycle' @@ -33,12 +34,25 @@ import { installWindowsPathRegistryChangeListener } from '../pty/windows-path-re export { WINDOW_QUIT_RENDERER_ACK_TIMEOUT_MS } -export function loadMainWindow(mainWindow: BrowserWindow): void { - if (is.dev && process.env.ELECTRON_RENDERER_URL) { - void mainWindow.loadURL(process.env.ELECTRON_RENDERER_URL) - } else { - void mainWindow.loadFile(join(__dirname, '../renderer/index.html')) - } +export function loadMainWindow(mainWindow: BrowserWindow, observer?: MainWindowLoadObserver): void { + const load = + is.dev && process.env.ELECTRON_RENDERER_URL + ? mainWindow.loadURL(process.env.ELECTRON_RENDERER_URL) + : mainWindow.loadFile(join(__dirname, '../renderer/index.html')) + // Observe each load promise so failures cannot leave recovery waiting silently. + load.then( + () => observer?.onLoaded?.(), + (cause: unknown) => { + const error = cause instanceof Error ? cause : new Error(String(cause)) + const errorCode = mainWindowLoadErrorCode(error) + // Keep durable diagnostics path-free and exclude shutdown/navigation aborts. + if (!mainWindow.isDestroyed() && errorCode !== 'ERR_ABORTED') { + recordDurableCrashBreadcrumb('main_window_load_failed', { errorCode }) + } + console.error('[window] Main window load failed', error) + observer?.onError?.(error) + } + ) } export function createMainWindow( @@ -158,8 +172,9 @@ export function createMainWindow( } forceRepaint(mainWindow) mainWindow.webContents.send('system:resumed') + // Give a suspended recovery load its full budget on wake. + focus.notifySystemResume() } - powerMonitor.on('resume', onSystemResume) const state = installMainWindowStateLifecycle({ mainWindow, @@ -172,9 +187,11 @@ export function createMainWindow( isWindowClosing: state.isWindowClosing, mainWindow, opts, - reloadMainWindow: () => loadMainWindow(mainWindow), + reloadMainWindow: (observer) => loadMainWindow(mainWindow, observer), rendererWebContentsId }) + // Register after focus is initialized because the resume callback uses it. + powerMonitor.on('resume', onSystemResume) installMainWindowShortcutRouting({ focus, mainWindow, opts, store }) const closeLifecycle = installMainWindowCloseLifecycle({ focus, diff --git a/src/main/window/main-window-contracts.ts b/src/main/window/main-window-contracts.ts index ce5c6cfe0b2..5135be0fbe6 100644 --- a/src/main/window/main-window-contracts.ts +++ b/src/main/window/main-window-contracts.ts @@ -1,4 +1,15 @@ import type { KeybindingOverrides } from '../../shared/keybindings' +import type { + RecoveryExhaustionCause, + RecoveryReloadMilestone, + RecoveryReloadTrigger +} from './renderer-recovery-reload-watchdog' + +/** Per-load outcome from Electron's load promise, which is scoped to that one load unlike `did-finish-load`. */ +export type MainWindowLoadObserver = { + onLoaded?: () => void + onError?: (error: Error) => void +} export type CreateMainWindowOptions = { /** Returns true when a manual app.quit() (Cmd+Q) is in progress, so the renderer skips the running-process confirm dialog. */ @@ -14,11 +25,14 @@ export type CreateMainWindowOptions = { details: Electron.RenderProcessGoneDetails, webContentsId: number ) => boolean - /** Called when consecutive auto-recoveries hit the circuit-breaker limit so the host can prompt instead of crash-looping. */ + /** Called when auto-recovery gives up — the breaker opened, or the recovery reload never produced a document. */ onRendererRecoveryExhausted?: (info: { details: Electron.RenderProcessGoneDetails webContentsId: number recentRecoveryCount: number + cause?: RecoveryExhaustionCause + /** Watched manual retry for the recovery prompt; an unwatched one cannot re-raise the prompt when it stalls too. */ + retry?: () => void }) => void /** Defer renderer load until IPC handlers are registered, or eager renderer calls race into missing channels. */ deferLoad?: boolean @@ -27,6 +41,24 @@ export type CreateMainWindowOptions = { title?: string getKeybindings?: () => KeybindingOverrides | undefined onBeforeReload?: (options: { ignoreCache: boolean; webContentsId: number }) => void - /** Marks the in-place recovery reload so did-finish-load's PTY orphan sweep spares live sessions until restore re-attaches (#5787). */ - onBeforeRecoveryReload?: (webContentsId: number) => void + /** + * Marks the in-place recovery reload so did-finish-load's PTY orphan sweep spares live sessions until restore + * re-attaches (#5787). The prompt's manual Reload is one too, so `trigger` keeps the automatic-recovery + * breadcrumb counting only automatic recoveries. + */ + onBeforeRecoveryReload?: (webContentsId: number, trigger: RecoveryReloadTrigger) => void + /** Pairs an outcome with the recovery-reload intent crumb: bundles could not tell a landed reload from a stalled one. */ + onRecoveryReloadOutcome?: (outcome: { + status: 'loaded' | 'timeout' | 'failed' + attempt: number + elapsedMs: number + /** How far the load got: 'none' is the blank-window field failure, anything else a document that then hung. */ + progress?: RecoveryReloadMilestone + /** True when the load landed after the recovery prompt was already raised — the recovery worked. */ + afterPrompt?: boolean + /** True when a later navigation replaced this load: elapsedMs then measures the replacement, not the reload. */ + superseded?: boolean + /** `ERR_*` code only, for the same reason — Electron's load-error message embeds the URL. */ + errorCode?: string + }) => void } diff --git a/src/main/window/main-window-focus-lifecycle.ts b/src/main/window/main-window-focus-lifecycle.ts index a021d464d15..d494992e692 100644 --- a/src/main/window/main-window-focus-lifecycle.ts +++ b/src/main/window/main-window-focus-lifecycle.ts @@ -14,13 +14,14 @@ import { matchingRichMarkdownContextMenuTableTarget, parseRichMarkdownContextMenuTableTarget } from './editable-context-menu' -import type { CreateMainWindowOptions } from './main-window-contracts' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' import { browserRouteWebContentsRegistry } from '../browser/browser-route-session-runtime' import { attachBrowserClientPageRenderer, retireBrowserClientPageRenderer } from '../browser/browser-client-page-renderer-runtime' import { registerRendererDocumentNavigation } from './renderer-document-navigation' +import { createRendererRecoveryReloadWatchdog } from './renderer-recovery-reload-watchdog' export type MainWindowFocusLifecycle = { dispose: () => void @@ -30,13 +31,15 @@ export type MainWindowFocusLifecycle = { isRendererProcessGone: () => boolean isShortcutRecorderFocused: () => boolean isTerminalInputFocused: () => boolean + /** Relays powerMonitor 'resume' so a suspend-frozen recovery-reload timer does not fire against an unbudgeted load. */ + notifySystemResume: () => void } export function installMainWindowFocusLifecycle(args: { isWindowClosing: () => boolean mainWindow: BrowserWindow opts?: CreateMainWindowOptions - reloadMainWindow: () => void + reloadMainWindow: (observer: MainWindowLoadObserver) => void rendererWebContentsId: number }): MainWindowFocusLifecycle { const { isWindowClosing, mainWindow, opts, reloadMainWindow, rendererWebContentsId } = args @@ -162,6 +165,16 @@ export function installMainWindowFocusLifecycle(args: { rendererRecoveryTimer = null } } + // Why: the reload can stall with a live window and no document — no did-fail-load fires, and the breaker counts + // renderer deaths, so a load that never lands is invisible to every other observer on this path. + const recoveryReloadWatchdog = createRendererRecoveryReloadWatchdog({ + isRecoveryPending: () => rendererRecoveryTimer !== null, + isWindowClosing, + mainWindow, + opts, + reloadMainWindow, + rendererWebContentsId + }) const scheduleRendererRecovery = (details: Electron.RenderProcessGoneDetails): void => { if ( rendererRecoveryTimer || @@ -187,17 +200,16 @@ export function installMainWindowFocusLifecycle(args: { const recovery = rendererRecoveryCircuitBreaker.registerRecoveryAttempt(Date.now()) if (!recovery.allowed) { // Why: too many reloads means it will just crash again; stop and let the host surface a recovery prompt. - opts?.onRendererRecoveryExhausted?.({ - details, - webContentsId: rendererWebContentsId, - recentRecoveryCount: recovery.recentRecoveryCount - }) + // Why through the watchdog: it owns the one-prompt-at-a-time guard, and the prompt's manual retry is a + // recovery reload too — unwatched, one that stalls leaves a blank window and no further prompt. + recoveryReloadWatchdog.escalate( + { details, recentRecoveryCount: recovery.recentRecoveryCount }, + 'crash-loop' + ) return } // Why: a transient renderer/Network Service loss can blank Chromium; reload the app document once to recover. - // Why: mark this in-place reload so the did-finish-load orphan sweep spares live PTYs until session restore (#5787). - opts?.onBeforeRecoveryReload?.(mainWindow.webContents.id) - reloadMainWindow() + recoveryReloadWatchdog.issue(details, recovery.recentRecoveryCount) }, 250) } mainWindow.webContents.on('render-process-gone', (_event, details) => { @@ -229,6 +241,7 @@ export function installMainWindowFocusLifecycle(args: { rendererProcessGone = false attachBrowserClientPageRenderer(rendererWebContents) clearRendererRecoveryTimer() + recoveryReloadWatchdog.notifyDocumentLoaded() }) const dispose = (): void => { @@ -237,6 +250,7 @@ export function installMainWindowFocusLifecycle(args: { resetFloatingTerminalInputFocus() resetShortcutRecorderFocus() clearRendererRecoveryTimer() + recoveryReloadWatchdog.clear() ipcMain.removeListener(markdownFocusChannel, onMarkdownEditorFocused) ipcMain.removeListener(terminalInputFocusChannel, onTerminalInputFocused) ipcMain.removeListener(floatingFocusChannel, onFloatingFocus) @@ -250,6 +264,7 @@ export function installMainWindowFocusLifecycle(args: { isMarkdownEditorFocused: () => markdownEditorFocused, isRendererProcessGone: () => rendererProcessGone, isShortcutRecorderFocused: () => shortcutRecorderFocused, - isTerminalInputFocused: () => terminalInputFocused + isTerminalInputFocused: () => terminalInputFocused, + notifySystemResume: recoveryReloadWatchdog.notifySystemResume } } diff --git a/src/main/window/main-window-load-error-code.ts b/src/main/window/main-window-load-error-code.ts new file mode 100644 index 00000000000..6dd9a79e6d8 --- /dev/null +++ b/src/main/window/main-window-load-error-code.ts @@ -0,0 +1,12 @@ +// Record only the ERR_* code: Electron error messages embed private install URLs. +export function mainWindowLoadErrorCode(error: unknown): string { + const code = + typeof error === 'object' && error !== null && 'code' in error && typeof error.code === 'string' + ? error.code + : undefined + if (code && /^ERR_[A-Z0-9_]+$/.test(code)) { + return code + } + const message = error instanceof Error ? error.message : String(error) + return /\bERR_[A-Z0-9_]+/.exec(message)?.[0] ?? 'unknown' +} diff --git a/src/main/window/main-window-webview-security.ts b/src/main/window/main-window-webview-security.ts index da6e4159a58..a662449eb58 100644 --- a/src/main/window/main-window-webview-security.ts +++ b/src/main/window/main-window-webview-security.ts @@ -111,7 +111,7 @@ export function installMainWindowWebviewSecurity(mainWindow: BrowserWindow): voi mainWindow.webContents.on('did-attach-webview', (_event, guest) => { if (isDocPreviewSession(guest.session)) { - // Why: preview guests never join browser-tab routing, popups or anti-detection; the + // Why: preview guests never join browser-tab routing, popups or auth-identity tracking; the // workspace-doc profile is what refuses all three. The attach is also the point a live window // exists to receive read failures for that guest. setDocPreviewFailureSink(mainWindow.webContents) diff --git a/src/main/window/renderer-recovery-prompt.test.ts b/src/main/window/renderer-recovery-prompt.test.ts index d5bd7ce11c1..a26700720eb 100644 --- a/src/main/window/renderer-recovery-prompt.test.ts +++ b/src/main/window/renderer-recovery-prompt.test.ts @@ -1,11 +1,14 @@ import type { MessageBoxOptions, MessageBoxReturnValue } from 'electron' -import { describe, expect, it, vi } from 'vitest' +import { beforeEach, afterEach, describe, expect, it, vi } from 'vitest' +import { ensureMainI18n, mainI18n } from '../i18n/main-i18n' import type { InstallDirAclPoisonDiagnosis } from '../startup/windows-install-dir-acl-recovery' import { presentRendererRecoveryPrompt, type RendererRecoveryPromptDeps } from './renderer-recovery-prompt' +vi.mock('electron', () => ({ app: { getLocale: () => 'en-US' } })) + const POISON: InstallDirAclPoisonDiagnosis = { detail: "Windows permissions on Orca's install folder are blocking its own sandboxed processes.", commands: ['icacls "C:\\Orca" /grant "*S-1-15-2-2:(OI)(CI)(RX)"', 'icacls "C:\\Orca" /grant b'] @@ -43,17 +46,60 @@ function harness(overrides: Partial & { responses?: } describe('presentRendererRecoveryPrompt', () => { + beforeEach(async () => { + await ensureMainI18n() + await mainI18n.changeLanguage('en') + }) + + afterEach(() => { + mainI18n.removeResourceBundle('en', 'translation') + }) + + it('interpolates the recovery count', async () => { + const { run, shown } = harness({ recentRecoveryCount: 7 }) + await run() + expect(shown[0].detail).toContain('Orca tried to recover 7 times in a row') + expect(shown[0].detail).not.toContain('{{') + }) + + it.each([ + { responses: [1, 0], reloads: 1, quits: 0 }, + { responses: [1, 2], reloads: 0, quits: 1 } + ])( + 'dispatches translated buttons by response index: $responses', + async ({ responses, reloads, quits }) => { + mainI18n.addResourceBundle('en', 'translation', { + rendererRecovery: { reload: 'Recharger', copyCommands: 'Copier', quit: 'Quitter' } + }) + const { run, shown, copied, reload, quit } = harness({ diagnose: () => POISON, responses }) + await run() + expect(shown[0].buttons).toEqual(['Recharger', 'Copier', 'Quitter']) + expect(copied).toEqual([POISON.commands.join('\r\n')]) + expect(reload).toHaveBeenCalledTimes(reloads) + expect(quit).toHaveBeenCalledTimes(quits) + } + ) + it('offers reload and quit with the generic cause when nothing is diagnosed', async () => { const { run, shown, reload, quit } = harness({ responses: [0] }) await run() expect(shown).toHaveLength(1) expect(shown[0].buttons).toEqual(['Reload', 'Quit']) - expect(shown[0].cancelId).toBe(1) + // Escape lands on cancelId, and this box is window-modal over the window it is about: it must not quit. + expect(shown[0].cancelId).toBe(0) expect(shown[0].detail).toContain('graphics-driver or installation problem') expect(reload).toHaveBeenCalledOnce() expect(quit).not.toHaveBeenCalled() }) + it('names the stalled reload instead of claiming a repeated crash', async () => { + const { run, shown } = harness({ failure: 'reload-stalled', responses: [1] }) + await run() + expect(shown[0].message).toContain('stopped responding while reloading') + expect(shown[0].detail).toContain('never finished loading') + expect(shown[0].detail).not.toContain('times in a row') + }) + it('quits on the last button', async () => { const { run, reload, quit } = harness({ responses: [1] }) await run() @@ -65,7 +111,7 @@ describe('presentRendererRecoveryPrompt', () => { const { run, shown } = harness({ diagnose: () => POISON, responses: [0] }) await run() expect(shown[0].buttons).toEqual(['Reload', 'Copy Commands', 'Quit']) - expect(shown[0].cancelId).toBe(2) + expect(shown[0].cancelId).toBe(0) expect(shown[0].detail).toContain(POISON.detail) expect(shown[0].detail).toContain('graphics driver') }) diff --git a/src/main/window/renderer-recovery-prompt.ts b/src/main/window/renderer-recovery-prompt.ts index 2026d1f10b5..18ab02a8eca 100644 --- a/src/main/window/renderer-recovery-prompt.ts +++ b/src/main/window/renderer-recovery-prompt.ts @@ -1,19 +1,13 @@ import type { MessageBoxOptions, MessageBoxReturnValue } from 'electron' +import { translateMain } from '../i18n/main-i18n' import type { InstallDirAclPoisonDiagnosis } from '../startup/windows-install-dir-acl-recovery' +import type { RecoveryExhaustionCause } from './renderer-recovery-reload-watchdog' -/** - * The dialog shown when the renderer crash-loop breaker opens: the window is - * blank by then, so this is the only retry/quit surface the user has. - */ - -const GENERIC_DETAIL = - 'This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.' -// Why keep it alongside the ACL diagnosis: the probe cannot name-check every -// locale, so a driver crash on a healthy install must not lose its only hint. -const DRIVER_FALLBACK = 'If that does not help, the cause is usually a graphics driver.' +export type RendererRecoveryPromptFailure = RecoveryExhaustionCause export type RendererRecoveryPromptDeps = { recentRecoveryCount: number + failure?: RendererRecoveryPromptFailure isQuitting: () => boolean diagnose: () => InstallDirAclPoisonDiagnosis | null showMessageBox: (options: MessageBoxOptions) => Promise @@ -25,29 +19,59 @@ export type RendererRecoveryPromptDeps = { export async function presentRendererRecoveryPrompt( deps: RendererRecoveryPromptDeps ): Promise { - // Why a loop: copying the commands must not dismiss the only surface offering them. + const stalled = deps.failure === 'reload-stalled' + // Copying must preserve the only available recovery surface. while (!deps.isQuitting()) { const diagnosis = deps.diagnose() - const buttons = diagnosis ? ['Reload', 'Copy Commands', 'Quit'] : ['Reload', 'Quit'] + const buttons = [translateMain('rendererRecovery.reload', 'Reload')] + if (diagnosis) { + buttons.push(translateMain('rendererRecovery.copyCommands', 'Copy Commands')) + } + buttons.push(translateMain('rendererRecovery.quit', 'Quit')) + const recoveryDetail = stalled + ? translateMain( + 'rendererRecovery.stalledDetail', + 'Orca reloaded the window after a crash, but it never finished loading.' + ) + : translateMain( + 'rendererRecovery.crashLoopDetail', + 'Orca tried to recover {{recoveryCount}} times in a row without success.', + { recoveryCount: deps.recentRecoveryCount } + ) + const causeDetail = diagnosis + ? `${diagnosis.detail}\n\n${translateMain( + 'rendererRecovery.driverFallback', + 'If that does not help, the cause is usually a graphics driver.' + )}` + : translateMain( + 'rendererRecovery.genericDetail', + 'This is often a graphics-driver or installation problem. Reload to try again, or quit and relaunch Orca.' + ) const { response } = await deps.showMessageBox({ type: 'error', buttons, defaultId: 0, - cancelId: buttons.length - 1, - title: 'Orca keeps failing to load', - message: 'The app window crashed repeatedly and stopped reloading automatically.', - detail: `Orca tried to recover ${deps.recentRecoveryCount} times in a row without success.\n\n${ - diagnosis ? `${diagnosis.detail}\n\n${DRIVER_FALLBACK}` : GENERIC_DETAIL - }` + // Escape retries instead of destroying the session. + cancelId: 0, + title: translateMain('rendererRecovery.title', 'Orca keeps failing to load'), + message: stalled + ? translateMain( + 'rendererRecovery.stalledMessage', + 'The app window stopped responding while reloading after a crash.' + ) + : translateMain( + 'rendererRecovery.crashLoopMessage', + 'The app window crashed repeatedly and stopped reloading automatically.' + ), + detail: `${recoveryDetail}\n\n${causeDetail}` }) - const choice = buttons[response] - if (choice === 'Copy Commands' && diagnosis) { + if (response === 1 && diagnosis) { deps.copyToClipboard(diagnosis.commands.join('\r\n')) continue } - if (choice === 'Reload') { + if (response === 0) { deps.reload() - } else if (choice === 'Quit') { + } else if (response === buttons.length - 1) { deps.quit() } return diff --git a/src/main/window/renderer-recovery-reload-watchdog.test.ts b/src/main/window/renderer-recovery-reload-watchdog.test.ts new file mode 100644 index 00000000000..b6e854bbbf1 --- /dev/null +++ b/src/main/window/renderer-recovery-reload-watchdog.test.ts @@ -0,0 +1,136 @@ +import { EventEmitter } from 'node:events' +import type { BrowserWindow } from 'electron' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { MainWindowLoadObserver } from './main-window-contracts' +import { + createRendererRecoveryReloadWatchdog, + RENDERER_RECOVERY_LOAD_TIMEOUT_MS +} from './renderer-recovery-reload-watchdog' + +vi.mock('@electron-toolkit/utils', () => ({ is: { dev: false } })) + +function createHarness() { + const webContents = Object.assign(new EventEmitter(), { id: 143 }) + const mainWindow = { webContents, isDestroyed: () => false } as unknown as BrowserWindow + const loads: MainWindowLoadObserver[] = [] + const onRecoveryReloadOutcome = vi.fn() + const onRendererRecoveryExhausted = vi.fn() + const watchdog = createRendererRecoveryReloadWatchdog({ + mainWindow, + rendererWebContentsId: webContents.id, + isRecoveryPending: () => false, + isWindowClosing: () => false, + reloadMainWindow: (observer) => loads.push(observer), + opts: { onRecoveryReloadOutcome, onRendererRecoveryExhausted } + }) + const abortLatestLoad = () => loads.at(-1)?.onError?.(new Error('ERR_ABORTED (-3)')) + watchdog.issue({ reason: 'crashed', exitCode: 5 }, 1) + return { + watchdog, + webContents, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } +} + +describe('superseding recovery navigations', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => vi.useRealTimers()) + + it('removes listeners and the pending stall timer during teardown', () => { + const { watchdog, webContents, loads, onRecoveryReloadOutcome } = createHarness() + expect(webContents.eventNames().sort()).toEqual(['did-fail-load', 'did-navigate', 'dom-ready']) + expect(vi.getTimerCount()).toBe(1) + watchdog.clear() + expect(webContents.eventNames()).toEqual([]) + expect(vi.getTimerCount()).toBe(0) + loads[0]?.onLoaded?.() + loads[0]?.onError?.(new Error('ERR_FILE_NOT_FOUND')) + watchdog.notifySystemResume() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalled() + expect(vi.getTimerCount()).toBe(0) + }) + + it('does not mistake a replacement error page for recovery', () => { + const { + watchdog, + webContents, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + abortLatestLoad() + webContents.emit('did-navigate') + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', true) + watchdog.notifyDocumentLoaded() + + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'failed', errorCode: 'ERR_FILE_NOT_FOUND' }) + ) + watchdog.clear() + }) + + it('ignores subframe failures and aborted replacement navigations', () => { + const { watchdog, webContents, abortLatestLoad, onRecoveryReloadOutcome } = createHarness() + abortLatestLoad() + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', false) + webContents.emit('did-fail-load', {}, -3, 'ERR_ABORTED', 'file:///previous', true) + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', superseded: true }) + ) + watchdog.clear() + }) + + it('keeps Reload available if the replacement fails beneath an existing prompt', () => { + const { + watchdog, + webContents, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + abortLatestLoad() + webContents.emit('did-navigate') + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + webContents.emit('did-fail-load', {}, -6, 'ERR_FILE_NOT_FOUND', 'file:///missing', true) + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).not.toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded' }) + ) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(loads).toHaveLength(2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + watchdog.clear() + }) + + it('recognizes a successful replacement started after the stall prompt', () => { + const { + watchdog, + loads, + abortLatestLoad, + onRecoveryReloadOutcome, + onRendererRecoveryExhausted + } = createHarness() + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + abortLatestLoad() + watchdog.notifyDocumentLoaded() + expect(onRecoveryReloadOutcome).toHaveBeenCalledWith( + expect.objectContaining({ status: 'loaded', superseded: true, afterPrompt: true }) + ) + onRendererRecoveryExhausted.mock.calls[0]?.[0].retry() + expect(loads).toHaveLength(2) + vi.advanceTimersByTime(RENDERER_RECOVERY_LOAD_TIMEOUT_MS * 2) + expect(onRendererRecoveryExhausted).toHaveBeenCalledOnce() + watchdog.clear() + }) +}) diff --git a/src/main/window/renderer-recovery-reload-watchdog.ts b/src/main/window/renderer-recovery-reload-watchdog.ts new file mode 100644 index 00000000000..1295ca13ac4 --- /dev/null +++ b/src/main/window/renderer-recovery-reload-watchdog.ts @@ -0,0 +1,310 @@ +import { is } from '@electron-toolkit/utils' +import type { BrowserWindow } from 'electron' +import { isSystemSessionEnding } from '../crash-reporting/expected-teardown-state' +import type { CreateMainWindowOptions, MainWindowLoadObserver } from './main-window-contracts' +import { mainWindowLoadErrorCode } from './main-window-load-error-code' + +// Field recoveries took up to 30.4s; allow 45s before retrying a load with no document. +export const RENDERER_RECOVERY_LOAD_TIMEOUT_MS = 45_000 +// Vite cold starts need a longer budget than packaged files. +export const RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS = 180_000 +// Retry once before handing recovery back to the user. +const RENDERER_RECOVERY_LOAD_ATTEMPTS = 2 +// Milestones may extend the budget, but cannot postpone the prompt indefinitely. +const RENDERER_RECOVERY_LOAD_CAP_FACTOR = 2 + +/** Automatic recovery vs the prompt's manual Reload; they must not share one breadcrumb name. */ +export type RecoveryReloadTrigger = 'automatic' | 'manual-retry' + +/** How far a load got. Ranked, so an attempt's milestone only ever moves forward. */ +export type RecoveryReloadMilestone = 'none' | 'committed' | 'dom-ready' +const MILESTONE_RANK: Record = { + none: 0, + committed: 1, + 'dom-ready': 2 +} + +export type RecoveryExhaustionCause = 'crash-loop' | 'reload-stalled' + +export type RendererRecoveryReloadWatchdog = { + /** Issues a recovery reload and arms the stall watchdog. */ + issue: ( + details: Electron.RenderProcessGoneDetails, + recentRecoveryCount: number, + trigger?: RecoveryReloadTrigger + ) => void + /** Raises the recovery prompt at most once: a native message box cannot be dismissed, so a second one stacks. */ + escalate: (subject: RecoveryPromptSubject, cause: RecoveryExhaustionCause) => void + /** + * A main-frame document finished loading. Only an attempt whose load was superseded takes this as its outcome; + * every other attempt settles through its own load promise, which an error page or a later navigation cannot fool. + */ + notifyDocumentLoaded: () => void + /** Restarts the stall budget after a suspend froze the timer mid-load. */ + notifySystemResume: () => void + clear: () => void +} + +type RecoveryReload = { + attempt: number + details: Electron.RenderProcessGoneDetails + recentRecoveryCount: number + /** Never rewritten: the elapsedMs a crash bundle reads has to stay time-since-issue. */ + issuedAt: number + /** Absolute deadline. A suspend pushes it out; a milestone cannot. */ + capAt: number + milestone: RecoveryReloadMilestone + progressedSinceArm: boolean + /** Chromium aborted this load for a later navigation, which now owns the outcome. */ + superseded: boolean +} + +type RecoveryReloadSeed = Pick +/** What a raised prompt is about; the crash-loop breaker has no attempt to hand over, only the crash. */ +export type RecoveryPromptSubject = Pick + +/** Bounds stalled recovery reloads while still observing success after escalation. */ +export function createRendererRecoveryReloadWatchdog(args: { + /** True when a renderer death has already queued its own recovery, which then owns the next load. */ + isRecoveryPending: () => boolean + isWindowClosing: () => boolean + mainWindow: BrowserWindow + opts?: CreateMainWindowOptions + reloadMainWindow: (observer: MainWindowLoadObserver) => void + rendererWebContentsId: number +}): RendererRecoveryReloadWatchdog { + const { + isRecoveryPending, + isWindowClosing, + mainWindow, + opts, + reloadMainWindow, + rendererWebContentsId + } = args + // Cache before teardown: accessing a destroyed window's webContents throws. + const rendererWebContents = mainWindow.webContents + let inFlight: RecoveryReload | null = null + // Retain timed-out loads so a late success can disarm the prompt's Reload. + let latest: RecoveryReload | null = null + // Keep one prompt until answered; native message boxes cannot be dismissed programmatically. + let prompt: RecoveryPromptSubject | null = null + let documentLanded = false + let timer: ReturnType | null = null + + const clearTimer = (): void => { + if (timer) { + clearTimeout(timer) + timer = null + } + } + // Match loadMainWindow's dev/prod branch. + const timeoutMs = (): number => + is.dev && process.env.ELECTRON_RENDERER_URL + ? RENDERER_RECOVERY_DEV_LOAD_TIMEOUT_MS + : RENDERER_RECOVERY_LOAD_TIMEOUT_MS + + const armTimer = (reload: RecoveryReload): void => { + clearTimer() + reload.progressedSinceArm = false + timer = setTimeout( + () => onBudgetExpired(reload), + Math.max(0, Math.min(timeoutMs(), reload.capAt - Date.now())) + ) + timer.unref?.() + } + + const onBudgetExpired = (reload: RecoveryReload): void => { + if (inFlight !== reload) { + return + } + // Give a progressing load the remaining budget instead of restarting it cold. + if (reload.progressedSinceArm && Date.now() < reload.capAt) { + armTimer(reload) + return + } + fail(reload) + } + + const start = (seed: RecoveryReloadSeed, trigger: RecoveryReloadTrigger): void => { + const issuedAt = Date.now() + const reload: RecoveryReload = { + ...seed, + issuedAt, + capAt: issuedAt + timeoutMs() * RENDERER_RECOVERY_LOAD_CAP_FACTOR, + milestone: 'none', + progressedSinceArm: false, + superseded: false + } + inFlight = reload + latest = reload + documentLanded = false + // Preserve live PTYs until renderer session restore (#5787). + opts?.onBeforeRecoveryReload?.(mainWindow.webContents.id, trigger) + // Only this load's promise distinguishes success from stale events and error pages. + reloadMainWindow({ + onLoaded: () => settleLoaded(reload), + onError: (error) => onLoadRejected(reload, mainWindowLoadErrorCode(error)) + }) + armTimer(reload) + } + + const settleLoaded = (reload: RecoveryReload): void => { + // A replaced attempt's promise may resolve on the replacement document. + if (reload !== latest) { + return + } + latest = null + documentLanded = true + if (reload === inFlight) { + inFlight = null + clearTimer() + } + opts?.onRecoveryReloadOutcome?.({ + status: 'loaded', + attempt: reload.attempt, + elapsedMs: Math.max(0, Date.now() - reload.issuedAt), + // Record late recovery even if the prompt has already appeared. + ...(prompt ? { afterPrompt: true } : {}), + // Replacement timings must be excluded from recovery-load budget analysis. + ...(reload.superseded ? { superseded: true } : {}) + }) + } + + // ERR_ABORTED transfers ownership to a replacement; the cap still bounds a silent replacement. + const onLoadRejected = (reload: RecoveryReload, errorCode: string): void => { + if (errorCode !== 'ERR_ABORTED') { + fail(reload, errorCode) + return + } + if (latest !== reload) { + return + } + reload.superseded = true + if (inFlight === reload) { + armTimer(reload) + } + } + + const retryFrom = (subject: RecoveryPromptSubject): void => { + prompt = null + // A late recovery makes the prompt's Reload unnecessary. + if (documentLanded) { + return + } + start( + { attempt: 1, details: subject.details, recentRecoveryCount: subject.recentRecoveryCount }, + 'manual-retry' + ) + } + + const escalate = (subject: RecoveryPromptSubject, cause: RecoveryExhaustionCause): void => { + // A new crash invalidates any document that landed while the prompt was open. + documentLanded = false + if (prompt) { + return + } + prompt = subject + opts?.onRendererRecoveryExhausted?.({ + details: subject.details, + webContentsId: rendererWebContentsId, + recentRecoveryCount: subject.recentRecoveryCount, + cause, + // Watch manual retries too, so another stall can offer recovery again. + retry: () => retryFrom(subject) + }) + } + + const fail = (reload: RecoveryReload, errorCode?: string): void => { + // Only the live attempt owns a failure verdict. + if (inFlight !== reload) { + return + } + // Suppress shutdown verdicts; resume may re-arm the retained attempt. + if ( + isWindowClosing() || + opts?.getIsQuitting?.() || + mainWindow.isDestroyed() || + isSystemSessionEnding() + ) { + return + } + inFlight = null + clearTimer() + opts?.onRecoveryReloadOutcome?.({ + status: errorCode === undefined ? 'timeout' : 'failed', + attempt: reload.attempt, + // Wall-clock changes must not produce negative diagnostic durations. + elapsedMs: Math.max(0, Date.now() - reload.issuedAt), + progress: reload.milestone, + ...(errorCode === undefined ? {} : { errorCode }) + }) + // A pending prompt or crash recovery owns the next reload. + if (prompt || isRecoveryPending()) { + return + } + // Restart only loads with no document; preserve progress until the user chooses Reload. + if (reload.attempt < RENDERER_RECOVERY_LOAD_ATTEMPTS && reload.milestone === 'none') { + start({ ...reload, attempt: reload.attempt + 1 }, 'automatic') + return + } + escalate(reload, 'reload-stalled') + } + + // Commit and DOM-ready distinguish a blank load from a document still loading. + const observeMilestone = (milestone: RecoveryReloadMilestone) => (): void => { + if (!inFlight || MILESTONE_RANK[milestone] <= MILESTONE_RANK[inFlight.milestone]) { + return + } + inFlight.milestone = milestone + inFlight.progressedSinceArm = true + } + const onDidNavigate = observeMilestone('committed') + const onDomReady = observeMilestone('dom-ready') + const onDidFailLoad = ( + _event: Electron.Event, + errorCode: number, + errorDescription: string, + _validatedURL: string, + isMainFrame: boolean + ): void => { + if (!isMainFrame || errorCode === -3 || !latest?.superseded) { + return + } + // Error documents also finish loading; only a successful replacement may settle an aborted attempt. + latest.superseded = false + documentLanded = false + fail(latest, mainWindowLoadErrorCode(new Error(errorDescription))) + } + rendererWebContents.on('did-navigate', onDidNavigate) + rendererWebContents.on('dom-ready', onDomReady) + rendererWebContents.on('did-fail-load', onDidFailLoad) + + return { + issue: (details, recentRecoveryCount, trigger = 'automatic') => + start({ attempt: 1, details, recentRecoveryCount }, trigger), + escalate, + notifyDocumentLoaded: () => { + // Timed-out replacements can still recover beneath the prompt. + if (latest?.superseded) { + settleLoaded(latest) + } + }, + // Restore the budget after sleep without rewriting the diagnostic issue time. + notifySystemResume: () => { + if (!inFlight) { + return + } + inFlight.capAt = Date.now() + timeoutMs() * RENDERER_RECOVERY_LOAD_CAP_FACTOR + armTimer(inFlight) + }, + clear: () => { + inFlight = null + latest = null + prompt = null + clearTimer() + rendererWebContents.off?.('did-navigate', onDidNavigate) + rendererWebContents.off?.('dom-ready', onDomReady) + rendererWebContents.off?.('did-fail-load', onDidFailLoad) + } + } +} diff --git a/src/preload/api/pty-api.ts b/src/preload/api/pty-api.ts index a1850398357..dff2b0b5185 100644 --- a/src/preload/api/pty-api.ts +++ b/src/preload/api/pty-api.ts @@ -112,7 +112,11 @@ export type PtyApi = { getForegroundProcess: (id: string) => Promise inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ) => Promise confirmForegroundProcess: (id: string) => Promise getCwd: (id: string) => Promise diff --git a/src/preload/api/pty-bridge-stream-and-serialization.ts b/src/preload/api/pty-bridge-stream-and-serialization.ts index 414a5514bfa..0847291ba7e 100644 --- a/src/preload/api/pty-bridge-stream-and-serialization.ts +++ b/src/preload/api/pty-bridge-stream-and-serialization.ts @@ -7,7 +7,11 @@ import type { TerminalProcessInspection } from '../../shared/terminal-process-in export const ptyStreamAndSerializationApi = { inspectProcess: ( id: string, - options?: { expectedIncarnationId?: string; scanChildProcesses?: boolean } + options?: { + expectedIncarnationId?: string + scanChildProcesses?: boolean + steadyState?: boolean + } ): Promise => ipcRenderer.invoke('pty:inspectProcess', { id, ...options }), confirmForegroundProcess: (id: string): Promise => diff --git a/src/relay/fs-path-metadata-requests.ts b/src/relay/fs-path-metadata-requests.ts index 2a9717a6484..b0fa2347d95 100644 --- a/src/relay/fs-path-metadata-requests.ts +++ b/src/relay/fs-path-metadata-requests.ts @@ -1,6 +1,8 @@ import { readdir, stat, lstat, realpath } from 'node:fs/promises' +import type { Dirent } from 'node:fs' import { join } from 'node:path' import { sortDirEntries } from '../shared/file-name-sort' +import { forEachWithConcurrency } from '../shared/map-with-concurrency' import { expandTilde } from './context' async function resolveSymlinkDirectoryEntry( @@ -34,11 +36,18 @@ function fileStatFromLstat(stats: Awaited>) { } } +// Why bounded: a pnpm `node_modules` is hundreds-to-thousands of package symlinks, and one +// unbounded `Promise.all` of stats from a single readDir saturates libuv's four-thread pool — +// delaying every other relay filesystem operation, including the interactive reads the +// list-files scan coordinator exists to protect. Matches the cap every other bounded probe in +// this codebase uses. +const SYMLINK_DIRECTORY_PROBE_CONCURRENCY = 8 + export async function readRelayDir(params: Record) { const dirPath = expandTilde(params.dirPath as string) const entries = await readdir(dirPath, { withFileTypes: true }) const mapped: { name: string; isDirectory: boolean; isSymlink: boolean }[] = [] - const symlinkProbes: Promise[] = [] + const symlinkEntries: { entry: Dirent; mappedEntry: (typeof mapped)[number] }[] = [] for (const entry of entries) { const mappedEntry = { name: entry.name, @@ -47,15 +56,17 @@ export async function readRelayDir(params: Record) { } mapped.push(mappedEntry) if (!mappedEntry.isDirectory && mappedEntry.isSymlink) { - symlinkProbes.push( - resolveSymlinkDirectoryEntry(dirPath, entry).then((isDirectory) => { - mappedEntry.isDirectory = isDirectory - }) - ) + symlinkEntries.push({ entry, mappedEntry }) } } - if (symlinkProbes.length > 0) { - await Promise.all(symlinkProbes) + if (symlinkEntries.length > 0) { + await forEachWithConcurrency( + symlinkEntries, + SYMLINK_DIRECTORY_PROBE_CONCURRENCY, + async ({ entry, mappedEntry }) => { + mappedEntry.isDirectory = await resolveSymlinkDirectoryEntry(dirPath, entry) + } + ) } return sortDirEntries(mapped) } diff --git a/src/relay/fs-path-metadata-symlink-concurrency.test.ts b/src/relay/fs-path-metadata-symlink-concurrency.test.ts new file mode 100644 index 00000000000..3a342a79e7c --- /dev/null +++ b/src/relay/fs-path-metadata-symlink-concurrency.test.ts @@ -0,0 +1,70 @@ +import { mkdirSync, mkdtempSync, rmSync, symlinkSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as FsPromisesModule from 'node:fs/promises' + +const statCalls = vi.hoisted(() => ({ inFlight: 0, peak: 0, total: 0 })) + +vi.mock('node:fs/promises', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + stat: async (...args: Parameters) => { + statCalls.inFlight += 1 + statCalls.total += 1 + statCalls.peak = Math.max(statCalls.peak, statCalls.inFlight) + try { + return await actual.stat(...args) + } finally { + statCalls.inFlight -= 1 + } + } + } +}) + +const { readRelayDir } = await import('./fs-path-metadata-requests') + +describe('relay readDir symlink probes', () => { + let root: string + let targetRoot: string + + beforeEach(() => { + statCalls.inFlight = 0 + statCalls.peak = 0 + statCalls.total = 0 + root = mkdtempSync(join(tmpdir(), 'orca-relay-readdir-')) + // Kept outside `root` so the listing contains only the symlinks under test. + targetRoot = mkdtempSync(join(tmpdir(), 'orca-relay-readdir-target-')) + const target = join(targetRoot, 'target') + mkdirSync(target) + writeFileSync(join(target, 'index.js'), '') + // A pnpm-shaped node_modules: many package symlinks in one directory. Junctions on + // Windows: plain symlinks need Developer Mode there. + for (let index = 0; index < 60; index += 1) { + symlinkSync( + target, + join(root, `pkg-${index}`), + process.platform === 'win32' ? 'junction' : 'dir' + ) + } + }) + + afterEach(() => { + rmSync(root, { recursive: true, force: true }) + rmSync(targetRoot, { recursive: true, force: true }) + }) + + it('bounds concurrent symlink stats instead of issuing one per entry at once', async () => { + const entries = await readRelayDir({ dirPath: root }) + + expect(statCalls.total).toBe(60) + // Exactly the cap: every worker enters `stat` before any resolves, so the peak proves the + // probes overlap and that no more than 8 ever do. Unbounded, all 60 would be in flight, + // saturating libuv's four-thread pool and stalling every other relay filesystem read. + expect(statCalls.peak).toBe(8) + // Behaviour is unchanged: every symlink still resolves to its target's kind. + expect(entries).toHaveLength(60) + expect(entries.every((entry) => entry.isDirectory && entry.isSymlink)).toBe(true) + }) +}) diff --git a/src/relay/pty-handler-startup-command-delivery.test.ts b/src/relay/pty-handler-startup-command-delivery.test.ts index 817ca756385..af2227a9e92 100644 --- a/src/relay/pty-handler-startup-command-delivery.test.ts +++ b/src/relay/pty-handler-startup-command-delivery.test.ts @@ -133,6 +133,73 @@ describe('PtyHandler', () => { } ) + it.skipIf(process.platform === 'win32')( + 'emits shell-ready markers for plain Codex on a line-editor shell', + async () => { + const oldShell = process.env.SHELL + const oldHome = process.env.HOME + const homeDir = mkdtempSync(join(tmpdir(), 'relay-plain-codex-spawn-')) + + process.env.SHELL = '/bin/bash' + process.env.HOME = homeDir + try { + // No prefill flag and no shell-ready hint: the host decides from its own + // shell, because the client cannot see it (#18767). + await dispatcher.callRequest('pty.spawn', { + env: { HOME: homeDir }, + command: 'codex' + }) + } finally { + if (oldShell === undefined) { + delete process.env.SHELL + } else { + process.env.SHELL = oldShell + } + if (oldHome === undefined) { + delete process.env.HOME + } else { + process.env.HOME = oldHome + } + rmSync(homeDir, { recursive: true, force: true }) + } + + const spawnOptions = mockPtySpawn.mock.calls[0]?.[2] as + | { env?: Record } + | undefined + expect(spawnOptions?.env?.ORCA_SHELL_FEATURES).toContain('ready') + vi.advanceTimersByTime(15_000) + expect(handler.retainedStartupCommandCount).toBe(0) + } + ) + + it.skipIf(process.platform === 'win32')( + 'leaves plain Codex unwaited on a shell that emits the marker before its reader', + async () => { + const oldHome = process.env.HOME + const homeDir = mkdtempSync(join(tmpdir(), 'relay-plain-codex-fish-spawn-')) + + process.env.HOME = homeDir + try { + await dispatcher.callRequest('pty.spawn', { + env: { HOME: homeDir, SHELL: '/usr/bin/fish' }, + command: 'codex' + }) + } finally { + if (oldHome === undefined) { + delete process.env.HOME + } else { + process.env.HOME = oldHome + } + rmSync(homeDir, { recursive: true, force: true }) + } + + const spawnOptions = mockPtySpawn.mock.calls[0]?.[2] as + | { env?: Record } + | undefined + expect(spawnOptions?.env?.ORCA_SHELL_FEATURES ?? '').not.toContain('ready') + } + ) + it.skipIf(process.platform === 'win32')( 'emits shell-ready markers for renderer-delivered Codex native prefill commands', async () => { diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 4b7c6dac2d6..d3a79b8a7c8 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -1896,12 +1896,16 @@ export class PtyHandler { isUnattended: launchAgent !== undefined, platform: process.platform }) + // Why the shell is part of the decision here and not on the client: the client + // cannot see which shell this host runs, and plain Codex must still wait where + // the marker rides the line editor rather than double-echoing an early write. const shouldEmitShellReadyMarker = launchCommandHint !== undefined && shouldUseShellReadyStartupDelivery({ command: launchCommandHint, startupCommandDelivery: - params.startupCommandDelivery === 'shell-ready' ? 'shell-ready' : undefined + params.startupCommandDelivery === 'shell-ready' ? 'shell-ready' : undefined, + shellPath: shell }) const managedStartupCommand = shouldProviderDeliverCommand ? command : launchCommandHint // Why: both renderer- and provider-delivered startup commands use this marker; the delivering side strips it from output. diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index e3d267cb353..3527f11c44e 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -229,6 +229,10 @@ --git-decoration-untracked: #007100; --git-decoration-copied: #007acc; --git-decoration-ignored: #8c8c8c; + --diff-added-ground: color-mix(in srgb, var(--git-decoration-added) 13%, transparent); + --diff-added-gutter: color-mix(in srgb, var(--git-decoration-added) 26%, transparent); + --diff-removed-ground: color-mix(in srgb, var(--git-decoration-deleted) 11%, transparent); + --diff-removed-gutter: color-mix(in srgb, var(--git-decoration-deleted) 22%, transparent); --git-graph-ref: #007acc; --git-graph-remote-ref: #b66dff; --git-graph-base-ref: #ea5c00; @@ -341,6 +345,10 @@ --git-decoration-untracked: #73c991; --git-decoration-copied: #73c991; --git-decoration-ignored: #6e6e6e; + --diff-added-ground: color-mix(in srgb, var(--git-decoration-added) 16%, transparent); + --diff-added-gutter: color-mix(in srgb, var(--git-decoration-added) 30%, transparent); + --diff-removed-ground: color-mix(in srgb, var(--git-decoration-deleted) 18%, transparent); + --diff-removed-gutter: color-mix(in srgb, var(--git-decoration-deleted) 32%, transparent); --git-graph-ref: #3794ff; --git-graph-remote-ref: #b66dff; --git-graph-base-ref: #ea5c00; diff --git a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts index 5e6dc77f67d..aff65e4eb3e 100644 --- a/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts +++ b/src/renderer/src/components/browser-pane/assemble-chrome/BrowserPane.webview-preferences.test.ts @@ -56,7 +56,7 @@ describe('BrowserPane webview preferences', () => { 'persist:orca-browser-session-profile-1' ) expect(ensuredWebview?.webview.getAttribute('webpreferences')).toBe( - ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE + `${ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE},transparent=false` ) expect(registryMocks.registerPersistentWebview).toHaveBeenCalledWith( 'browser-page-1', diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts new file mode 100644 index 00000000000..2d156ed926c --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview-surface.test.ts @@ -0,0 +1,93 @@ +// @vitest-environment happy-dom +import { afterEach, describe, expect, it, vi } from 'vitest' +import { ORCA_BROWSER_BLANK_URL } from '../../../../../shared/constants' +import { ensureBrowserPageWebview } from './browser-page-webview' +import { webviewRegistry } from './webview-registry' + +vi.mock('./webview-registry', () => { + const webviewRegistry = new Map() + return { + webviewRegistry, + registerPersistentWebview: vi.fn((id, guest) => webviewRegistry.set(id, guest)), + replacePersistentWebview: vi.fn(), + destroyPersistentWebview: vi.fn() + } +}) + +afterEach(() => { + document.body.replaceChildren() + webviewRegistry.clear() +}) + +function createGuest(): Electron.WebviewTag { + const container = document.createElement('div') + document.body.appendChild(container) + return ensureBrowserPageWebview({ + browserTabId: 'surface-test', + container, + inputLocked: false, + webviewPartition: 'persist:browser-test', + resolveContainer: () => container + })!.webview +} + +function commit(guest: Electron.WebviewTag, url: string, isMainFrame = true): void { + guest.dispatchEvent(Object.assign(new Event('load-commit'), { url, isMainFrame })) +} + +describe('browser page surface ownership', () => { + it('themes the host before attach and uses an opaque native canvas for real pages', () => { + const guest = createGuest() + expect(guest.style.background).toBe('var(--background)') + expect(guest.getAttribute('webpreferences')).toContain('transparent=false') + expect(guest.getAttribute('webpreferences')).toContain('disableHtmlFullscreenWindowResize=true') + }) + + it.each(['about:blank', ORCA_BROWSER_BLANK_URL])( + 'keeps %s unavailable through first navigation, then reveals the committed page', + (url) => { + const guest = createGuest() + commit(guest, url) + expect(guest.style.visibility).toBe('hidden') + guest.dispatchEvent(new Event('did-start-loading')) + expect(guest.style.visibility).toBe('hidden') + commit(guest, 'https://example.test') + expect(guest.style.visibility).toBe('visible') + guest.dispatchEvent(new Event('did-start-loading')) + expect(guest.style.visibility).toBe('visible') + commit(guest, 'about:blank', false) + expect(guest.style.visibility).toBe('visible') + } + ) + + it('preserves a reused guest and initializes the same surface after a container remount', () => { + const guest = createGuest() + commit(guest, 'https://example.test') + const container = guest.parentElement as HTMLDivElement + const reused = ensureBrowserPageWebview({ + browserTabId: 'surface-test', + container, + inputLocked: false, + webviewPartition: 'persist:browser-test', + resolveContainer: () => container + })! + expect(reused.created).toBe(false) + expect(reused.webview).toBe(guest) + expect(reused.webview.style.visibility).toBe('visible') + const replacement = createGuest() + expect(replacement).not.toBe(guest) + expect(replacement.style.background).toBe('var(--background)') + expect(replacement.getAttribute('webpreferences')).toContain('transparent=false') + commit(replacement, ORCA_BROWSER_BLANK_URL) + expect(replacement.style.visibility).toBe('hidden') + }) + + it('exposes the themed host after renderer loss until a recovered document commits', () => { + const guest = createGuest() + commit(guest, 'https://example.test') + guest.dispatchEvent(new Event('render-process-gone')) + expect(guest.style.visibility).toBe('hidden') + commit(guest, 'https://example.test') + expect(guest.style.visibility).toBe('visible') + }) +}) diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts index 30e1cc18442..3c959751051 100644 --- a/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts +++ b/src/renderer/src/components/browser-pane/host-guest/browser-page-webview.ts @@ -1,3 +1,4 @@ +import { ORCA_BROWSER_BLANK_URL } from '../../../../../shared/constants' import { ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE } from '../../../../../shared/browser-guest-web-preferences' import { destroyPersistentWebview, @@ -59,16 +60,29 @@ export function ensureBrowserPageWebview({ webview.setAttribute('allowpopups', '') // Why: Electron spreads the webpreferences keys verbatim, so the shared // camelCase attribute must stay intact for fullscreen containment to work. - webview.setAttribute('webpreferences', ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE) + // Keep Chromium's normal page canvas opaque while the host underneath follows Orca's theme. + webview.setAttribute( + 'webpreferences', + `${ORCA_BROWSER_GUEST_WEB_PREFERENCES_ATTRIBUTE},transparent=false` + ) webview.style.display = 'flex' webview.style.flex = '1' webview.style.width = '100%' webview.style.height = '100%' webview.style.border = 'none' setBrowserPageWebviewInputLock(webview, inputLocked) - // Why: some pages never paint a background, and a white viewport matches - // normal browser behavior instead of leaking Orca chrome through the guest. - webview.style.background = '#ffffff' + webview.style.background = 'var(--background)' + const guest = webview + // A committed synthetic blank document belongs to New Tab, including while its first URL waits. + guest.addEventListener('load-commit', (event) => { + if (event.isMainFrame) { + guest.style.visibility = + event.url === 'about:blank' || event.url === ORCA_BROWSER_BLANK_URL ? 'hidden' : 'visible' + } + }) + guest.addEventListener('render-process-gone', () => { + guest.style.visibility = 'hidden' + }) registerPersistentWebview(browserTabId, webview) activeContainer.appendChild(webview) created = true diff --git a/src/renderer/src/components/confirmation-dialog-context.ts b/src/renderer/src/components/confirmation-dialog-context.ts index 4675c181fb7..b112191a4e6 100644 --- a/src/renderer/src/components/confirmation-dialog-context.ts +++ b/src/renderer/src/components/confirmation-dialog-context.ts @@ -1,4 +1,5 @@ import { createContext, useContext } from 'react' +import type { LucideIcon } from 'lucide-react' // Keep the context component-free so Fast Refresh preserves its identity. @@ -9,6 +10,9 @@ export type ConfirmationDialogOptions = { confirmLabel?: string cancelLabel?: string confirmVariant?: 'default' | 'destructive' + icon?: LucideIcon + cancelVariant?: 'outline' | 'ghost' + initialFocus?: 'confirm' /** Renders a "Don't ask again" checkbox. `onConfirmed` runs only when the user confirms with it checked. */ dontAskAgain?: { label?: string; onConfirmed: () => void } } diff --git a/src/renderer/src/components/confirmation-dialog.test.tsx b/src/renderer/src/components/confirmation-dialog.test.tsx index 25738ee4c17..5097a5d7786 100644 --- a/src/renderer/src/components/confirmation-dialog.test.tsx +++ b/src/renderer/src/components/confirmation-dialog.test.tsx @@ -44,6 +44,37 @@ function renderDialog(options: ConfirmationDialogOptions): { onSettled: ReturnTy describe('ConfirmationDialogProvider', () => { afterEach(cleanup) + it('focuses the primary action when requested and confirms with Enter', async () => { + const { onSettled } = renderDialog({ + title: 'Reveal hidden workspace?', + confirmLabel: 'Clear filters and reveal', + initialFocus: 'confirm' + }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => + expect(screen.getByRole('button', { name: 'Clear filters and reveal' })).toHaveFocus() + ) + await userEvent.keyboard('{Enter}') + await waitFor(() => expect(onSettled).toHaveBeenCalledWith(true)) + }) + + it('still cancels with Escape when the primary action has focus', async () => { + const { onSettled } = renderDialog({ + title: 'Reveal hidden workspace?', + initialFocus: 'confirm' + }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => expect(screen.getByRole('button', { name: 'Confirm' })).toHaveFocus()) + await userEvent.keyboard('{Escape}') + await waitFor(() => expect(onSettled).toHaveBeenCalledWith(false)) + }) + + it('keeps the default cancel focus for callers that do not opt in', async () => { + renderDialog({ title: 'Delete artifact?', confirmVariant: 'destructive' }) + await userEvent.click(screen.getByRole('button', { name: 'ask' })) + await waitFor(() => expect(screen.getByRole('button', { name: 'Cancel' })).toHaveFocus()) + }) + it('omits the checkbox unless the caller opts in', async () => { renderDialog({ title: 'Delete artifact?' }) diff --git a/src/renderer/src/components/confirmation-dialog.tsx b/src/renderer/src/components/confirmation-dialog.tsx index a1670b9bb22..615a2400bc6 100644 --- a/src/renderer/src/components/confirmation-dialog.tsx +++ b/src/renderer/src/components/confirmation-dialog.tsx @@ -32,6 +32,7 @@ export function ConfirmationDialogProvider({ children: React.ReactNode }): React.JSX.Element { const nextIdRef = useRef(0) + const confirmButtonRef = useRef(null) const [queue, setQueue] = useState([]) const [dontAskAgain, setDontAskAgain] = useState(false) const activeRequest = queue[0] ?? null @@ -46,6 +47,7 @@ export function ConfirmationDialogProvider({ } // Why: Radix keeps dialog content mounted while closing; keep labels stable without a post-render Effect. const displayedRequest = activeRequest ?? lastDisplayedRequestRef.current + const Icon = displayedRequest?.options.icon useEffect(() => { // Why: this provider's dialog is not represented by activeModal. Block @@ -96,18 +98,37 @@ export function ConfirmationDialogProvider({ open={activeRequest !== null} onOpenChange={(open) => !open && settleActiveRequest(false)} > - - - {displayedRequest?.options.title} - {displayedRequest?.options.description ? ( - // Callers pass multi-line descriptions (e.g. one path per line). - - {displayedRequest.options.description} - - ) : null} - + { + if (activeRequest?.options.initialFocus === 'confirm') { + event.preventDefault() + confirmButtonRef.current?.focus() + } + }} + > +
+ {Icon && ( +
+
+ )} + + {displayedRequest?.options.title} + {displayedRequest?.options.description ? ( + // Callers pass multi-line descriptions (e.g. one path per line). + + {displayedRequest.options.description} + + ) : null} + +
{displayedRequest?.options.dontAskAgain ? (
) : null} - - + + + {expanded ? ( +
+ {props.tasks.length > 0 ? ( +
    + {props.tasks.map((task) => ( +
  • +
  • + ))} +
+ ) : ( +

+ {translate( + 'components.native-chat.backgroundTasks.detailsUnavailable', + 'Task details are unavailable for this session.' + )} +

+ )} +
+ ) : null} + + + ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx b/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx index 36fa791f57d..0e933584d41 100644 --- a/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx +++ b/src/renderer/src/components/native-chat/NativeChatCopyButton.tsx @@ -12,9 +12,12 @@ import { translate } from '@/i18n/i18n' */ export function NativeChatCopyButton({ text, + label: copyLabel, className }: { text: string + /** What this button copies, when it is not the whole message. */ + label?: string className?: string }): React.JSX.Element { const [copied, setCopied] = useState(false) @@ -46,7 +49,7 @@ export function NativeChatCopyButton({ const label = copied ? translate('components.native-chat.copyMessage.copied', 'Copied') - : translate('components.native-chat.copyMessage.copy', 'Copy message') + : (copyLabel ?? translate('components.native-chat.copyMessage.copy', 'Copy message')) return ( +
+ {file.oldPath ? ( + <> + + {baseName(file.oldPath)} + + → + + ) : null} + + {baseName(file.path)} + + + {file.truncated ? ( + // Beside the counts rather than under the rows: a collapsed card, and + // one clipped down to no rows at all, would otherwise say nothing. + + {translate('components.native-chat.tool.diffTruncated', 'Diff truncated')} + + ) : null} + +
+ {hasBody && expanded ? ( + // Focusable so the rows can be scrolled from the keyboard. +
+ {(() => { + const seen = new Map() + return file.lines.map((line) => { + const signature = `${line.kind}:${line.oldLineNumber}:${line.newLineNumber}:${line.text}` + const occurrence = seen.get(signature) ?? 0 + seen.set(signature, occurrence + 1) + return ( + + ) + }) + })()} +
+ ) : null} + + ) +} diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.stream-render.perf.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.stream-render.perf.test.tsx new file mode 100644 index 00000000000..361045b4642 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.stream-render.perf.test.tsx @@ -0,0 +1,91 @@ +// @vitest-environment happy-dom + +import '@testing-library/jest-dom/vitest' + +import { cleanup, render } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type * as NativeChatProseModule from './native-chat-prose' +import type { NativeChatMessage } from '../../../../shared/native-chat-types' +import type { NativeChatLiveSession } from './use-native-chat-live-session' + +// Counting real per-row work rather than a render counter: a future refactor could keep the +// render count low while still re-deriving every row's markdown. +const proseCalls = vi.hoisted(() => ({ count: 0 })) +vi.mock('./native-chat-prose', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + nativeChatProseToMarkdown: (prose: Parameters[0]) => { + proseCalls.count += 1 + return actual.nativeChatProseToMarkdown(prose) + } + } +}) + +const { NativeChatMessageList } = await import('./NativeChatMessageList') + +afterEach(cleanup) + +const TRANSCRIPT_LENGTH = 120 + +function settledMessages(): NativeChatMessage[] { + return Array.from({ length: TRANSCRIPT_LENGTH }, (_, index) => ({ + id: `message-${index}`, + role: index % 2 === 0 ? ('user' as const) : ('assistant' as const), + blocks: [{ type: 'text' as const, text: `settled line ${index}` }], + timestamp: index + 1, + source: 'transcript' as const + })) +} + +function sessionWith(messages: NativeChatMessage[]): NativeChatLiveSession { + return { + messages, + status: 'ready', + sessionId: 'session-1', + agent: 'codex', + hasMore: false, + loadingEarlier: false, + loadEarlier: vi.fn(), + readPhase: 'ready' + } +} + +describe('native chat transcript re-render cost during a streaming turn', () => { + it('rebuilds only the rows whose blocks changed, not the whole transcript per frame', () => { + const messages = settledMessages() + const { rerender } = render( + + ) + + const afterFirstPaint = proseCalls.count + expect(afterFirstPaint).toBeGreaterThanOrEqual(TRANSCRIPT_LENGTH) + + // A streaming turn publishes a frame per SDK event; only the tail message's blocks change. + const STREAM_FRAMES = 20 + for (let frame = 1; frame <= STREAM_FRAMES; frame += 1) { + const streaming = messages.slice(0, -1).concat({ + ...messages.at(-1)!, + blocks: [{ type: 'text' as const, text: `streaming token ${frame}` }] + }) + rerender( + + ) + } + + const perFrame = (proseCalls.count - afterFirstPaint) / STREAM_FRAMES + // Without row memoization every settled row rebuilt its markdown on every frame. Settled + // rows keep their block identity, so only the streaming tail should rebuild. + expect(perFrame).toBeLessThan(TRANSCRIPT_LENGTH / 10) + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx index 7d8f1cf4049..debcfb1c94b 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx @@ -7,7 +7,7 @@ import { orderNativeChatMessages } from './native-chat-message-grouping' import { stripNoiseMessages } from './native-chat-noise' import { foldToolMessages } from './native-chat-tool-fold' import { isNearBottom, shouldShowJumpToLatest, type ScrollGeometry } from './native-chat-autoscroll' -import { NativeChatMessageRow } from './NativeChatMessageRow' +import { MessageRow } from './NativeChatMessageRow' import { shouldShowNativeChatTypingIndicator } from './native-chat-typing-indicator' import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' import { useNativeChatTurnStatus } from './use-native-chat-turn-status' @@ -225,7 +225,7 @@ export function NativeChatMessageList({ : undefined return ( - (null) - const split = useMemo(() => splitNativeChatBlocks(message.blocks), [message.blocks]) - const tools = split.tools - const subagentGroups = useMemo(() => subagentGroupBlocks(split.prose), [split.prose]) - // A spawn-group row carries a plain-text twin so a client without the block - // type still reads the roster. This one draws the block, so the twin is - // dropped rather than printed beside it. - const prose = useMemo( - () => - subagentGroups.length === 0 + // One pass per block set: a streaming turn re-renders this row on every frame, and these + // derivations used to re-run each time even though `message.blocks` had not changed. + const { hasImages, markdown, prose, subagentGroups, tools } = useMemo(() => { + const split = splitNativeChatBlocks(message.blocks) + const groups = subagentGroupBlocks(split.prose) + // A spawn-group row carries a plain-text twin so a client without the block + // type still reads the roster. This one draws the block, so the twin is + // dropped rather than printed beside it. + const prose = + groups.length === 0 ? split.prose - : split.prose.filter((block) => block.type !== 'text' && !isSubagentGroupBlock(block)), - [split.prose, subagentGroups.length] - ) - const markdown = nativeChatProseToMarkdown(prose) - const hasImages = prose.some((block) => block.type === 'image-ref') + : split.prose.filter((block) => block.type !== 'text' && !isSubagentGroupBlock(block)) + return { + tools: split.tools, + prose, + subagentGroups: groups, + markdown: nativeChatProseToMarkdown(prose), + hasImages: prose.some((block) => block.type === 'image-ref') + } + }, [message.blocks]) const isUser = message.role === 'user' const isReasoning = message.role === 'reasoning' const isSystem = message.role === 'system' @@ -152,6 +159,7 @@ export function NativeChatMessageRow({ className="text-sm" onLinkClick={onLinkClick} allowFileUriLinks={allowFileUriLinks} + linkifyFilePaths={onLinkClick !== undefined} /> ) : null} {tools.length > 0 || subagentGroups.length > 0 ? ( @@ -173,4 +181,4 @@ export function NativeChatMessageRow({ ) : null} ) -} +}) diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx index 70afd6758aa..afdf5ace7c5 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx @@ -4,6 +4,7 @@ import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-libra import React, { forwardRef, useImperativeHandle } from 'react' import { afterEach, describe, expect, it, vi } from 'vitest' import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import type { AgentSessionBackgroundTask } from '../../../../shared/agent-session-wire' import { decodeAgentSessionQuestionAnswers } from '../../../../shared/agent-session-question-answer' import type { NativeChatQuestionCardProps } from './NativeChatQuestionCard' @@ -17,13 +18,19 @@ const mocks = vi.hoisted(() => ({ showTurnStatus?: boolean runtimeContext?: unknown }, - composerProps: null as null | { structuredTransport?: Record }, + composerProps: null as null | { + structuredTransport?: Record + isWorking?: boolean + }, questionCardProps: null as NativeChatQuestionCardProps | null, promptItems: [] as AgentJournalRenderItem[], respond: vi.fn(), handlePasteEvent: vi.fn(), pasteFromClipboard: vi.fn(), - submissions: [] as unknown[] + submissions: [] as unknown[], + monitoringBackgroundTasks: false, + backgroundTasks: [] as AgentSessionBackgroundTask[], + stopBackgroundTasks: vi.fn() })) vi.mock('@/runtime/structured-agent-session-client', () => ({ @@ -67,8 +74,11 @@ vi.mock('./use-structured-agent-session', async () => { send: outbox.send, retry: outbox.retry, isWorking: false, + isMonitoringBackgroundTasks: mocks.monitoringBackgroundTasks, + backgroundTasks: mocks.backgroundTasks, turnId: null, cancel: vi.fn(), + stopBackgroundTasks: mocks.stopBackgroundTasks, respond: mocks.respond, optionSnapshot: [ { @@ -155,6 +165,9 @@ describe('NativeChatStructuredSession', () => { mocks.handlePasteEvent.mockReset() mocks.pasteFromClipboard.mockReset() mocks.submissions = [] + mocks.monitoringBackgroundTasks = false + mocks.stopBackgroundTasks.mockReset() + mocks.backgroundTasks = [] }) it('routes app-menu paste into the structured composer', () => { @@ -165,7 +178,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-paste" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -176,15 +188,14 @@ describe('NativeChatStructuredSession', () => { expect(mocks.pasteFromClipboard).toHaveBeenCalledOnce() }) - it('wires local structured file links through the native chat opener', () => { + it('wires remote structured file links through the host-aware native chat opener', () => { render( ) @@ -204,7 +215,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-parity" target={{ kind: 'local' }} agent={agent} - allowFileUriLinks /> ) @@ -213,6 +223,47 @@ describe('NativeChatStructuredSession', () => { } ) + it('places background monitoring above the usable composer and stops without an active turn', async () => { + mocks.monitoringBackgroundTasks = true + mocks.backgroundTasks = [ + { id: 'task-command', kind: 'command', description: 'sleep 180' }, + { id: 'task-agent', kind: 'agent' } + ] + mocks.stopBackgroundTasks.mockResolvedValue({ cancelled: true }) + + render( + + ) + + const status = screen + .getByText('Monitoring background tasks') + .closest('[data-native-chat-background-tasks="true"]') + const composer = screen.getByTestId('structured-composer') + if (!status) { + throw new Error('background task status was not rendered') + } + expect(status.compareDocumentPosition(composer) & Node.DOCUMENT_POSITION_FOLLOWING).toBeTruthy() + expect(mocks.composerProps?.isWorking).toBe(false) + expect(screen.queryByRole('list', { name: 'Running background tasks' })).toBeNull() + + const disclosure = screen.getByRole('button', { name: 'Monitoring background tasks' }) + expect(disclosure.getAttribute('aria-expanded')).toBe('false') + fireEvent.click(disclosure) + expect(disclosure.getAttribute('aria-expanded')).toBe('true') + expect(screen.getByRole('list', { name: 'Running background tasks' })).toBeTruthy() + expect(screen.getByText('sleep 180')).toBeTruthy() + expect(screen.getByText('Background agent')).toBeTruthy() + + fireEvent.click(screen.getByRole('button', { name: 'Stop' })) + await waitFor(() => expect(mocks.stopBackgroundTasks).toHaveBeenCalledOnce()) + }) + it('routes a bare model command to the native option picker', async () => { render( { sessionId="session-1" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) const dispatchCommand = mocks.composerProps?.structuredTransport?.dispatchCommand as @@ -257,7 +307,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-1" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -289,7 +338,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-wedge" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -321,7 +369,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-probe-flag" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -354,7 +401,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-parked" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -404,7 +450,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-churn" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) const { rerender } = render(makeView()) @@ -458,7 +503,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-target-switch" target={target} agent="codex" - allowFileUriLinks /> ) const { rerender } = render(makeView({ kind: 'local' })) @@ -493,7 +537,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-forced" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -532,7 +575,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-pending" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -562,7 +604,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-budget" target={{ kind: 'local' }} agent="codex" - allowFileUriLinks /> ) @@ -633,7 +674,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-questions" target={{ kind: 'local' }} agent="claude" - allowFileUriLinks={false} /> ) @@ -692,7 +732,6 @@ describe('NativeChatStructuredSession', () => { sessionId="session-legacy-question" target={{ kind: 'local' }} agent="claude" - allowFileUriLinks={false} /> ) diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index 34ee980f0b3..9ac354f8a73 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -20,6 +20,7 @@ import { NativeChatOrchestrationPausedNotice } from './NativeChatOrchestrationPa import { useNativeChatImageRuntimeContext } from './native-chat-image-runtime-context' import { useStructuredNativeChatPaneCommands } from './use-structured-native-chat-pane-commands' import type { NativeChatStructuredViewProps } from './native-chat-view-types' +import { NativeChatBackgroundTasksStatus } from './NativeChatBackgroundTasksStatus' function encodeQuestionAnswer(questionId: string, answer: string): string { return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` @@ -30,6 +31,7 @@ export function NativeChatStructuredSession( ): React.JSX.Element { const controller = useStructuredAgentSession(props) const [composerError, setComposerError] = useState(null) + const [stoppingBackgroundTasks, setStoppingBackgroundTasks] = useState(false) const [optionPickerRequest, setOptionPickerRequest] = useState<{ id: string sequence: number @@ -80,7 +82,7 @@ export function NativeChatStructuredSession( const fontScale = useNativeChatFontScale(viewState.kind === 'ready') const fileLinkContext = useNativeChatFileLinkContext(props.tabId) const imageRuntimeContext = useNativeChatImageRuntimeContext(props.tabId) - const fileLinkClick = useNativeChatFileLinkClick(props.allowFileUriLinks ? fileLinkContext : null) + const fileLinkClick = useNativeChatFileLinkClick(fileLinkContext) const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null const questions = @@ -272,6 +274,16 @@ export function NativeChatStructuredSession( {controller.error ?? composerError}

) : null} + {controller.isMonitoringBackgroundTasks ? ( + { + setStoppingBackgroundTasks(true) + void controller.stopBackgroundTasks().finally(() => setStoppingBackgroundTasks(false)) + }} + /> + ) : null} {prompt ? null : ( { { type: 'tool-call', name: 'apply_patch', + // The patch lives on the call in this lane, so the provider's own + // completion is what says the edit landed. + state: 'completed', input: { changes: [ { @@ -46,8 +49,9 @@ describe('NativeChatToolRun', () => { const { container } = render() - expect(screen.getByText('+after')).toBeInTheDocument() - expect(screen.getByText('-before')).toBeInTheDocument() + expect(screen.getByText('after')).toBeInTheDocument() + expect(screen.getByText('before')).toBeInTheDocument() + expect(screen.getByText('Edited file')).toBeInTheDocument() expect(container.querySelector('pre')).toBeNull() }) @@ -78,18 +82,156 @@ describe('NativeChatToolRun', () => { ) - expect(screen.getByText('+after')).toHaveClass( - 'bg-emerald-500/10', - 'text-[var(--git-decoration-added)]' - ) - expect(screen.getByText('-before')).toHaveClass( - 'bg-rose-500/10', - 'text-[var(--git-decoration-deleted)]' - ) + // Row grounds come from the diff tokens, not a hardcoded palette value. + expect(screen.getByText('after').closest('div')).toHaveClass('bg-[var(--diff-added-ground)]') + expect(screen.getByText('before').closest('div')).toHaveClass('bg-[var(--diff-removed-ground)]') expect(container).not.toHaveTextContent('"changes"') expect(container.querySelector('pre')).toBeNull() }) + it('keeps the provider error visible for an edit the agent could not apply', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'missing', new_string: 'now' } + }, + { type: 'tool-result', output: 'String to replace not found in file.', isError: true } + ] + + const { container } = render() + + expect(screen.queryByText('Edited file')).toBeNull() + const body = container.querySelector('pre') + expect(body).toHaveTextContent('String to replace not found in file.') + expect(body).toHaveClass('text-destructive') + }) + + it('leaves a `git diff` command as a command row rather than an edit card', () => { + const blocks: NativeChatBlock[] = [ + { type: 'tool-call', name: 'exec', input: { command: 'git diff' }, state: 'completed' }, + { + type: 'tool-result', + output: 'diff --git a/a.ts b/a.ts\n--- a/a.ts\n+++ b/a.ts\n@@ -1 +1 @@\n-was\n+now' + } + ] + + const { container } = render() + + expect(screen.queryByText('Edited file')).toBeNull() + expect(container).toHaveTextContent('git diff') + }) + + it('shows no gutter number for a snippet edit, which cannot locate itself', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts', old_string: 'was', new_string: 'now' }, + state: 'completed' + }, + { type: 'tool-result', output: 'ok' } + ] + + render() + + // Exact, because a snippet-relative number would sit ahead of the marker. + expect(screen.getByText('now').closest('div')?.textContent).toBe('+now') + expect(screen.getByText('was').closest('div')?.textContent).toBe('-was') + }) + + it('separates two regions of a file so the gutter jump is accounted for', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + state: 'completed' + }, + { + type: 'tool-result', + output: 'ok', + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 42, oldLines: 1, newStart: 42, newLines: 1, lines: ['-was', '+now'] }, + { oldStart: 310, oldLines: 1, newStart: 310, newLines: 1, lines: ['-old', '+new'] } + ] + } + } + ] + + render() + + const separators = screen.getAllByRole('separator') + expect(separators).toHaveLength(1) + expect(separators[0]).toHaveAccessibleName('Lines not shown') + }) + + it('offers no empty body for a delete, which names the file and nothing else', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'apply_patch', + input: { input: '*** Begin Patch\n*** Delete File: gone.ts\n*** End Patch' }, + state: 'completed' + } + ] + + render() + + expect(screen.getByTitle('gone.ts')).toBeInTheDocument() + // The header states the change; there is no body behind a disclosure. + expect(screen.getByText('Deleted file').closest('button')).not.toHaveAttribute('aria-expanded') + }) + + it('says a diff was clipped even while the card is collapsed', () => { + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Diff', + input: { path: 'src/a.ts' }, + state: 'completed' + }, + { type: 'tool-result', output: '@@ -1,3 +1,3 @@\n ctx\n-was\n+now\n… (48210 bytes)' } + ] + + // A defined expandOverride opens the run while leaving each card closed. + render() + + expect(screen.getByText('Diff truncated')).toBeInTheDocument() + expect(screen.queryByText('was')).toBeNull() + }) + + it('copies the diff as signed rows, with the region breaks left out', () => { + const writeClipboardText = vi.fn() + Object.assign(window, { api: { ui: { writeClipboardText } } }) + const blocks: NativeChatBlock[] = [ + { + type: 'tool-call', + name: 'Edit', + input: { file_path: '/repo/a.ts' }, + state: 'completed' + }, + { + type: 'tool-result', + output: 'ok', + editPatch: { + filePath: '/repo/a.ts', + hunks: [ + { oldStart: 1, oldLines: 2, newStart: 1, newLines: 2, lines: [' ctx', '-was', '+now'] }, + { oldStart: 90, oldLines: 1, newStart: 90, newLines: 1, lines: ['+tail'] } + ] + } + } + ] + + render() + fireEvent.click(screen.getByRole('button', { name: 'Copy diff' })) + + expect(writeClipboardText).toHaveBeenCalledWith(' ctx\n-was\n+now\n+tail') + }) + it('keeps a grouped active run to one stable row showing only the latest tool', () => { const blocks: NativeChatBlock[] = [ { type: 'tool-call', name: 'shell', input: { command: 'date' }, state: 'completed' }, diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx index 53c71f1d235..3e89cd00f0c 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx @@ -1,4 +1,4 @@ -import { useEffect, useState } from 'react' +import { useEffect, useMemo, useState } from 'react' import { Check, ChevronRight, SquareTerminal, Wrench } from 'lucide-react' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' @@ -9,57 +9,41 @@ import { type NativeChatSubagentGroupBlock } from '../../../../shared/native-chat-types' import { diffFromText, diffFromToolCall, type DiffLine } from './native-chat-diff' +import { NativeChatDiffCard } from './NativeChatDiffCard' +import { pairToolBlocks } from './native-chat-tool-fold' +import { + editFilesFromToolPair, + isEditToolName +} from '../../../../shared/native-chat-edit-normalize' +import type { NativeChatEditFile } from '../../../../shared/native-chat-edit-model' import { countToolCalls, createToolInputDisplay, summarizeToolRun, truncateToolDetail } from './native-chat-tool-summary' +import { + describeActiveToolCall, + isCommandToolName, + NATIVE_CHAT_TOOL_ACTIVITY_COPY, + selectActiveToolCall +} from '../../../../shared/native-chat-tool-activity' import { NativeChatDiffView } from './NativeChatDiffView' import { NativeChatSubagentRun } from './NativeChatSubagentRun' /** Stable empty default: a fresh array literal per render breaks memoization. */ const NO_SUBAGENT_GROUPS: NativeChatSubagentGroupBlock[] = [] -const COMMAND_TOOL_NAMES = new Set([ - 'bash', - 'shell', - 'powershell', - 'terminal', - 'execute', - 'run_command', - 'run_shell_command', - 'shell_command', - 'exec_command', - 'run_terminal_cmd', - 'run_terminal_command' -]) - -function normalizedToolName(name: string): string { - return name.trim().toLowerCase() -} - function activeToolLabel(call: Extract): string { - const preview = createToolInputDisplay(call.input).label - if (COMMAND_TOOL_NAMES.has(normalizedToolName(call.name))) { - return preview - ? translate('components.native-chat.tool.runningPreview', 'Running {{preview}}', { - preview - }) - : translate('components.native-chat.tool.runningCommand', 'Running command') - } - return preview - ? translate( - 'components.native-chat.tool.runningNamedPreview', - 'Running {{toolName}} {{preview}}', - { - toolName: call.name, - preview - } - ) - : translate('components.native-chat.tool.runningNamed', 'Running {{toolName}}', { - toolName: call.name - }) + const { key, toolName, preview } = describeActiveToolCall(call) + const copy = NATIVE_CHAT_TOOL_ACTIVITY_COPY[key] + return key === 'runningPreview' + ? translate('components.native-chat.tool.runningPreview', copy, { preview }) + : key === 'runningCommand' + ? translate('components.native-chat.tool.runningCommand', copy) + : key === 'runningNamedPreview' + ? translate('components.native-chat.tool.runningNamedPreview', copy, { toolName, preview }) + : translate('components.native-chat.tool.runningNamed', copy, { toolName }) } /** A single inline tool line — `▸ ToolName preview` — that expands in place to @@ -156,6 +140,50 @@ function ToolLine({ ) } +type EditCardModel = { + editCards: Map + /** Result blocks the card already speaks for, so they render no second row. */ + consumedResults: Set +} + +const NO_EDIT_CARDS: EditCardModel = { editCards: new Map(), consumedResults: new Set() } + +/** An edit renders as one card, so its result block is folded into the call. The + * model decides which calls have landed; a call that has not keeps the generic + * tool view, its result still visible as the provider's own error. */ +function buildEditCards(blocks: NativeChatBlock[]): EditCardModel { + const editCards: EditCardModel['editCards'] = new Map() + const consumedResults: EditCardModel['consumedResults'] = new Set() + for (const [index, pair] of pairToolBlocks(blocks).entries()) { + const call = pair.call + if (!call || !isEditToolName(call.name)) { + continue + } + const files = editFilesFromToolPair({ + name: call.name, + input: call.input, + ...(call.state ? { state: call.state } : {}), + ...(pair.result + ? { + result: { + output: pair.result.output, + isError: pair.result.isError, + editPatch: pair.result.editPatch + } + } + : {}) + }) + if (!files || files.length === 0) { + continue + } + editCards.set(call, { files, key: `${call.name}:${index}` }) + if (pair.result) { + consumedResults.add(pair.result) + } + } + return { editCards, consumedResults } +} + /** A run of a message's tool calls/results, collapsed to a one-line summary that * expands to the individual inline tool lines. `expandSignal` lets the global * toolbar toggle drive every run at once while still allowing per-run override. */ @@ -191,27 +219,25 @@ export function NativeChatToolRun({ )) const callCount = countToolCalls(blocks) || blocks.length const summary = summarizeToolRun(blocks) - const calls = blocks.filter(isToolCallBlock) - const activeCalls = structuredActivityUi - ? calls.filter( - (call) => - (call.state === 'running' || (call.state == null && activeTurnIsWorking === true)) && - activeTurnIsWorking !== false - ) - : [] - const latestActiveCall = activeCalls.at(-1) + const latestActiveCall = structuredActivityUi + ? selectActiveToolCall(blocks, { activeTurnIsWorking }) + : null const isSettled = latestActiveCall == null // The turn caret opens the activity group, while each child tool remains // collapsed. The global expand toolbar still opens child details together. const expandToolLines = expandOverride === undefined ? open : false + // Diffing every edit is the run's most expensive work, so a collapsed run — + // which renders none of it — never pays for it. + const { editCards, consumedResults } = useMemo( + () => (open ? buildEditCards(blocks) : NO_EDIT_CARDS), + [open, blocks] + ) const ActiveToolIcon = - latestActiveCall && COMMAND_TOOL_NAMES.has(normalizedToolName(latestActiveCall.name)) - ? SquareTerminal - : Wrench + latestActiveCall && isCommandToolName(latestActiveCall.name) ? SquareTerminal : Wrench const fallbackLabel = callCount === 1 - ? translate('components.native-chat.tool.countOne', '1 tool call') - : translate('components.native-chat.tool.countN', '{{value0}} tool calls', { + ? translate('components.native-chat.tool.countOne', NATIVE_CHAT_TOOL_ACTIVITY_COPY.countOne) + : translate('components.native-chat.tool.countN', NATIVE_CHAT_TOOL_ACTIVITY_COPY.countN, { value0: callCount }) @@ -286,6 +312,23 @@ export function NativeChatToolRun({ {(() => { const seen = new Map() return blocks.map((block) => { + const edit = editCards.get(block) + if (edit) { + return ( +
+ {edit.files.map((file, fileIndex) => ( + + ))} +
+ ) + } + if (consumedResults.has(block)) { + return null + } const signature = block.type === 'tool-call' ? `${block.type}:${block.name}:${JSON.stringify(block.input)}` diff --git a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx index 21145e94c94..e6c38de83f5 100644 --- a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx @@ -2,21 +2,14 @@ import { useState } from 'react' import { ChevronRight } from 'lucide-react' import { translate } from '@/i18n/i18n' import { useNow } from '@/hooks/use-now' +import { + describeNativeChatTurnStatus, + formatNativeChatDuration, + NATIVE_CHAT_TURN_STATUS_COPY, + nativeChatElapsedSeconds +} from '../../../../shared/native-chat-turn-status' -/** Format turn time without exposing an ever-growing raw seconds count. */ -export function formatNativeChatDuration(seconds: number): string { - const totalSeconds = Number.isFinite(seconds) ? Math.max(0, Math.floor(seconds)) : 0 - if (totalSeconds < 60) { - return `${totalSeconds}s` - } - const minutes = Math.floor(totalSeconds / 60) - const remainingSeconds = totalSeconds % 60 - if (minutes < 60) { - return `${minutes}m ${remainingSeconds}s` - } - const hours = Math.floor(minutes / 60) - return `${hours}h ${minutes % 60}m ${remainingSeconds}s` -} +export { formatNativeChatDuration } export function NativeChatWorkingStatus({ startedAt, @@ -39,21 +32,29 @@ export function NativeChatWorkingStatus({ // Why: preserves the old effect's `startedAt ?? Date.now()` epoch for the // single frame before the turn's startedAt lands. const [mountedAt] = useState(() => Date.now()) - const elapsedSeconds = counting - ? Math.max(0, Math.floor((now - (startedAt ?? mountedAt)) / 1000)) - : 0 + const elapsedSeconds = counting ? nativeChatElapsedSeconds(startedAt, mountedAt, now) : 0 + const { key, duration } = describeNativeChatTurnStatus({ + thinking, + workedSeconds, + elapsedSeconds + }) const label = - workedSeconds != null - ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}}', { - value0: formatNativeChatDuration(workedSeconds) - }) - : thinking - ? translate('components.native-chat.status.thinking', 'Thinking') - : translate('components.native-chat.status.workingFor', 'Working for {{value0}}', { - value0: formatNativeChatDuration(elapsedSeconds) - }) - + key === 'workedFor' + ? translate( + 'components.native-chat.status.workedFor', + NATIVE_CHAT_TURN_STATUS_COPY.workedFor, + { + value0: duration + } + ) + : key === 'thinking' + ? translate('components.native-chat.status.thinking', NATIVE_CHAT_TURN_STATUS_COPY.thinking) + : translate( + 'components.native-chat.status.workingFor', + NATIVE_CHAT_TURN_STATUS_COPY.workingFor, + { value0: duration } + ) const className = `flex min-h-8 items-center gap-1 text-sm text-muted-foreground${thinking ? '' : ' border-b border-border'}` const caret = workedSeconds != null ? ( @@ -67,7 +68,10 @@ export function NativeChatWorkingStatus({