From bc28ffbaa6dc0eff3dab39b2c5370be2c9661b8c Mon Sep 17 00:00:00 2001 From: Merge Sim Date: Wed, 2 Sep 2026 13:24:55 -0700 Subject: [PATCH] refactor(agents): unify pane identity adapters for tranche 0 --- .../ladder-unification-decision-table.md | 45 +++ docs/plans/ladder-unification-plan.md | 292 ++++++++++++++++++ .../lib/tab-agent-identity-comparison.test.ts | 120 ------- .../src/lib/tab-agent-identity-comparison.ts | 139 --------- src/renderer/src/lib/use-tab-agent.ts | 10 +- src/shared/agent-status-identity.ts | 9 +- src/shared/pane-agent-identity-adapter.ts | 118 +++++-- .../pane-agent-identity-comparison.test.ts | 116 ------- src/shared/pane-agent-identity-comparison.ts | 176 ----------- ...pane-agent-identity-decision-table.test.ts | 211 +++++++++++++ .../pane-agent-identity-inventory.test.ts | 68 ++-- src/shared/pane-agent-identity-resolver.ts | 53 +--- ...e-agent-identity-surface-inventory.test.ts | 30 +- src/shared/pane-agent-owner.ts | 49 +-- ...ne-agent-identity-characterization.test.ts | 175 ----------- ...ublished-pane-agent-identity-comparison.ts | 100 ------ src/shared/terminal-title-agent-type.ts | 29 +- 17 files changed, 735 insertions(+), 1005 deletions(-) create mode 100644 docs/plans/ladder-unification-decision-table.md create mode 100644 docs/plans/ladder-unification-plan.md delete mode 100644 src/renderer/src/lib/tab-agent-identity-comparison.test.ts delete mode 100644 src/renderer/src/lib/tab-agent-identity-comparison.ts delete mode 100644 src/shared/pane-agent-identity-comparison.test.ts delete mode 100644 src/shared/pane-agent-identity-comparison.ts create mode 100644 src/shared/pane-agent-identity-decision-table.test.ts delete mode 100644 src/shared/published-pane-agent-identity-characterization.test.ts delete mode 100644 src/shared/published-pane-agent-identity-comparison.ts diff --git a/docs/plans/ladder-unification-decision-table.md b/docs/plans/ladder-unification-decision-table.md new file mode 100644 index 00000000000..3d1bd547cf4 --- /dev/null +++ b/docs/plans/ladder-unification-decision-table.md @@ -0,0 +1,45 @@ +# Pane-agent ladder decision table + +This table is the review gate for the ladder-unification plan. It uses the exhaustive signal model +from PR #17711 at `d288820ee3`: seven slots (focused live hook, sibling live hook, focused +completed hook, sibling completed hook, foreground process, sleeping session, and launch record), +each taking `∅`, agent A, or agent B; four title kinds (blank, neutral/no-agent, A, B); and local +or remote scope: `3^7 × 4 × 2 = 17,496` shapes. A fresh process proof names the foreground-process +slot and includes all required freshness fields (`capturedAgeMs` and `validForMs`). + +## Exhaustive result + +| Canonical rung selected in a disagreeing shape | Signal class (remaining slots are unrestricted) | What the shipping tab ladder selected | Canonical decision | Count | +| --- | --- | --- | --- | ---: | +| `launch` | Launch is A or B; foreground process has no value; no live focused hook. The old result is the completed-hook agent opposite launch. | Completed hook | Launch record | 396 | +| `launch` | Launch is A or B; foreground process has no value; no live focused hook. The old result is the sleeping-session agent opposite launch. | Sleeping session | Launch record | 144 | +| `launch` | Launch is A or B; foreground process has no value; no live focused hook. The old result is the title agent opposite launch. | Title | Launch record | 72 | +| `completed-hook` | Launch and foreground process have no value; completed hook is A or B; no sleeping-session identity. The old result is the title agent opposite the completed hook. | Title | Completed hook | 36 | +| **Total** | | | | **648** | + +Counts include both agent names, both local/remote values, all sibling values, and the four title +kinds. They are intentionally grouped by the rung selected by the canonical side, so a reviewer can +rule on each conflict without relying on an aggregate disagreement counter. No residual shape has a +valid process proof: where one exists, the process rung is selected before launch and the old and +canonical process answers agree. + +## Process-versus-launch rule + +The 1,872 shapes that flip when a valid proof is supplied are the process-starvation artifact in the +proof-free input. In every shape where the host proves a recognized foreground process, that proof +wins over launch (and over completed/sleeping/title evidence); a matching launch and process name is +the same answer, and an absent/expired/mismatched proof does not promote a bare process name. This is +the answer to the central question: **a fresh host proof wins over a launch record; the 648 residual +shapes are the no-process-proof surface in which launch or completed-hook wins over weaker evidence.** + +For the record, the same harness produces the supplied totals: + +| Process proof | Disagreements | Canonical source breakdown | +| --- | ---: | --- | +| Omitted (proof-free input) | 2,520 | launch 1,908; completed-hook 468; sleeping-session 144 | +| Fresh and valid | 648 | launch 612; completed-hook 36; sleeping-session 0 | +| Answer changes when proof is added | 1,872 | process rung (the starved-rung artifact) | + +If `capturedAgeMs` or `validForMs` is omitted from the fixture, freshness rejects the proof and the +harness incorrectly reproduces 2,520 instead of 648. The test must write its result artifact with +`writeFileSync` (Vitest intercepts console output) and fail on either total or source breakdown. diff --git a/docs/plans/ladder-unification-plan.md b/docs/plans/ladder-unification-plan.md new file mode 100644 index 00000000000..cae48507071 --- /dev/null +++ b/docs/plans/ladder-unification-plan.md @@ -0,0 +1,292 @@ +# Unify the pane-agent identity ladder + +## Decision requested + +Make `resolveCanonicalPaneAgentIdentity` the one ranking implementation used everywhere Orca +answers “which agent is in this pane”. Keep the six existing public entry-point signatures as thin +adapters, so the 65 consumer rows do not churn. This plan deliberately stops at design: no product +behavior or source file is changed by this task. + +The companion review artifact is [ladder-unification-decision-table.md](./ladder-unification-decision-table.md). +It contains the exhaustive 648-shape table and the proof-freshness trap that must remain a test gate. + +## Why this seam + +Today the tab icon, open-tab occupant, host publication, pane owner, status ingress, and title +readers each rank overlapping evidence differently. The canonical adapter already has the right +shape for a shared seam: it accepts pane-scoped evidence, host-stamped process proof, run keys, +scope/floor options, and an uncovered fallback, and returns the answer plus provenance. Promote that +adapter from comparison-only code to the production resolver; keep its low-level evidence types in +`src/shared/pane-agent-identity-resolver.ts`, but make that module a private ranking primitive (or a +delegating compatibility export), never a second policy. + +The local reference-repository review found the same useful boundary in mature terminal systems: +the execution host owns process identity and lifecycle, display titles are separate metadata, and +remote adapters forward opaque host evidence while tolerating missing optional fields. Orca should +apply those principles in its own vocabulary; no external project or implementation is copied. + +## Canonical contract + +`src/shared/pane-agent-identity-adapter.ts` owns this public input and output (the exact field names +can be retained from the existing adapter): + +```ts +type CanonicalPaneAgentIdentityInput = { + hookAgent?: TuiAgent | null + hookIsLive?: boolean + hookRun?: PaneAgentRunKey + completedHookAgent?: TuiAgent | null + completedHookRun?: PaneAgentRunKey + launchAgent?: TuiAgent | null + launchRun?: PaneAgentRunKey + foregroundAgent?: TuiAgent | null + processProof?: ForegroundProcessProof | null + sleepingSessionAgent?: TuiAgent | null + sleepingRun?: PaneAgentRunKey + siblingAgent?: TuiAgent | null + allowSibling?: boolean + title?: string | null + currentRun?: PaneAgentRunKey + minimumSource?: PaneAgentEvidenceSource + uncoveredFallback?: { agent: TuiAgent | null; titleOnly?: boolean } +} + +type CanonicalPaneAgentIdentity = { + agent: TuiAgent | null + source: PaneAgentEvidenceSource | null + coverage: 'covered' | 'uncovered' + titleOnly: boolean + ambiguousAt?: PaneAgentEvidenceSource + supersededSources: readonly PaneAgentEvidenceSource[] +} +``` + +The resolver constructs evidence once and applies one order, strongest first: + +1. `live-hook`: the provider reports its own identity for the focused pane. +2. `process`: only a fresh, name-matching `ForegroundProcessProof` stamped by the execution host. +3. `launch`: Orca's accepted launch/resume/command intent. +4. `completed-hook`: the last completed focused-pane hook for the current run. +5. `sleeping-session`: durable provider-session identity while a pane sleeps. +6. `sibling`: only when a tab-level caller explicitly opts in with `allowSibling`. +7. `title`: parsed vendor marker or anchored owner suffix, absolutely last. + +Equal-rank conflicting observations return `agent: null` with `ambiguousAt`; array order must never +choose a winner. `currentRun` filters same-authority superseded evidence while treating missing or +cross-authority run keys as incomparable/eligible for mixed-version compatibility. A caller that +authorizes a write passes `minimumSource: 'launch'`, which excludes title and sibling evidence from +the decision rather than merely hoping a higher source happens to exist. `coverage` is based only +on eligible authority evidence (hook, fresh process proof, launch, completed hook, or sleeping +session), never on a title, sibling, or bare process name. An uncovered fallback preserves the old +answer only as a clearly marked compatibility lane; it cannot turn a title into covered proof. + +The host proof contract remains strict: `ForegroundProcessProof` carries an opaque process +incarnation, authority id, `capturedAgeMs`, and `validForMs`. Missing, negative, non-finite, expired, +or name-mismatched fields drop the process rung. The decision-table test must fail if a fixture omits +either freshness field, because that silently turns 648 back into 2,520. + +### What the 648 shapes decide + +The companion table replays all `17,496` signal shapes and must reproduce 2,520 disagreements without +a proof and 648 with a fresh proof. The 648 residuals are: + +- canonical `launch`: 612 (old result was a conflicting completed hook: 396, sleeping session: 144, + or title: 72); the launch record wins all three because title is last and durable records outrank it; +- canonical `completed-hook`: 36 (old result was the opposite title); the completed hook wins; +- canonical `sleeping-session`, `process`, `sibling`, and `title`: zero. + +When launch and foreground process disagree, a fresh host proof wins over launch. The 1,872 shapes +that change when the proof is supplied are exactly the process-starvation artifact; no residual 648 +shape has a valid process proof. The separate process-selector fix for nested OMP/ChatGPT.app +descendants must land first and retain the WSL resolver's ambiguity fence: if the host cannot select +one foreground agent unambiguously, it emits no proof and the canonical resolver returns to launch, +hook, sleeping, or unknown instead of guessing. + +## Six thin adapters and exact seams + +The adapters preserve caller signatures and translate local fields to canonical evidence. None may +re-rank, parse a title beside the canonical call, or invent a process proof. + +| Existing entry point | Current production seam | Thin-adapter behavior after migration | +| --- | --- | --- | +| `resolveTabAgentFromSignals` | Definition/ladder in `src/renderer/src/lib/tab-agent-from-signals.ts` (the branch may colocate it in `use-tab-agent.ts`); called by `src/renderer/src/lib/use-tab-agent.ts` and `src/renderer/src/lib/open-tab-occupant-agent.ts`. | Map focused live/completed hooks, launch, sleeping, host proof, title, and sibling slots to the canonical input and return `.agent`. Keep `resolveLaunchedAgentExitEvidence` as lifecycle evidence only; it must not alter ranking. `useTabAgent` supplies the host proof when present and keeps the existing `TuiAgent | null` return. | +| `resolvePaneAgentIdentity` | `src/shared/pane-agent-identity-resolver.ts`, called from `src/shared/published-pane-agent-identity.ts`. | Retain its generic evidence/result shape for tests and old imports, but delegate to the canonical implementation (or make its ranking routine private). There must be one `SOURCE_RANK`, one ambiguity rule, and one run-eligibility implementation. | +| `resolvePaneAgentOwner` / `resolvePaneAgentOwnerRecord` | Owner/record consumers: `src/renderer/src/lib/tab-agent-from-signals.ts`, `src/renderer/src/lib/use-tab-agent.ts`, `src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts`, `src/renderer/src/components/terminal-pane/parked-terminal-command-status.ts`, `src/renderer/src/components/terminal-pane/pty-connection/shell-command-inference.ts`, `src/renderer/src/runtime/web-session-tabs-sync/terminal-build.ts`, `src/renderer/src/runtime/web-session-tabs-sync/agent-status-primitives.ts`, `src/main/runtime/runtime-mobile-agent-status-builder.ts`, and `src/main/runtime/runtime-mobile-session-projection.ts`. | Translate launch/startup/initial/typed-command fields to `launch` evidence while preserving `ownerIsLaunch`; map focused/sibling live and completed hooks and sleeping sessions to their canonical sources. Return the legacy `AgentType | null` or owner record without a second precedence list. Action users of this adapter use the canonical minimum-source floor. | +| `resolveAgentStatusIdentity` | Definition `src/shared/agent-status-identity.ts`; production ingress/builders in `src/main/agent-hooks/server.ts`, `src/main/agent-hooks/server/server-status-update.ts`, `src/renderer/src/hooks/ipc-events/agent-status-event-applicator.ts`, `src/renderer/src/store/slices/agent-status.ts`, and `src/renderer/src/store/slices/agent-status-live-entry-builder.ts`. | Keep status freshness, `unknown` normalization, and `inheritedFromActivePane`/child-completion suppression as status policy. Convert existing and incoming rows into live/completed hook evidence, call the canonical resolver, and map a null/ambiguous result back to the current status shape. No status-specific agent ordering remains. | +| `collectAgentTitleEvidence` / `resolveTerminalTitleAgentType` (including explicit/committed wrappers) | Parser definitions in `src/shared/agent-title-evidence.ts` and `src/shared/terminal-title-agent-type.ts`; direct identity consumers include `src/shared/published-pane-agent-identity.ts`, `src/renderer/src/lib/notes-send-agent-targets.ts`, `src/renderer/src/lib/open-tab-occupant-agent.ts`, `src/renderer/src/lib/tab-agent-from-signals.ts`, `src/renderer/src/lib/use-tab-agent.ts`, `src/renderer/src/lib/pane-agent-evidence.ts`, and `mobile/src/session/mobile-terminal-tab-agent.ts`. | Keep parsing as an evidence producer for activity/formatting and for the canonical title rung. Any caller answering pane identity passes the raw title to the canonical resolver and does not combine a title result with a launch/process result locally. Free-text-only and conflicting title evidence remain null; title is never primary. | +| `resolveCanonicalPaneAgentIdentity` | Existing adapter in `src/shared/pane-agent-identity-adapter.ts`; currently reached only by the comparison wrappers. | Make this the production call made by every adapter. Remove comparison-only callers, add direct canonical tests, and return provenance/coverage so displays can show an honest unknown while action callers can fail closed. | + +The inventory ratchet remains authoritative: helper-name census rows 1–31 and marker-pinned surface +rows 6 and 32–65 must be updated deliberately whenever a wrapper moves or is renamed. + +## Migration tranches (65 rows) + +Every tranche keeps the old signature, changes only its adapter body, runs the inventory ratchet, and +records the canonical source/ambiguity behavior in focused tests before the next tranche. This order +puts rendered display blast radius first, then main/runtime behavior, and the host-to-client wire last. + +### Tranche 0 — establish the seam (no consumer behavior switch) + +- Promote `resolveCanonicalPaneAgentIdentity` and its `ForegroundProcessProof` freshness gate. +- Make `resolvePaneAgentIdentity` and the owner/status/title functions delegating adapters; keep + parser-only uses classified as activity or formatting. +- Add the exhaustive decision-table fixture and canonical resolver tests (including equal-rank + conflict, run-key supersession, missing proof, and title-last cases). +- Run `src/shared/pane-agent-identity-inventory.test.ts` and + `src/shared/pane-agent-identity-surface-inventory.test.ts`; no row may disappear. + +### Tranche 1 — renderer display surfaces (first behavior change) + +Move display decisions to the canonical adapter in the tab icon/open-tab occupant and the marker +surfaces for rows 32, 48–52, and 61–65: + +- `src/renderer/src/lib/use-tab-agent.ts`, `src/renderer/src/lib/tab-agent-from-signals.ts`, + `src/renderer/src/lib/open-tab-occupant-agent.ts`; +- `src/renderer/src/components/terminal-pane/native-chat-leaf-title-agent.ts`, + `src/renderer/src/components/terminal-pane/TerminalPane.tsx`, + `src/renderer/src/components/terminal-pane/pty-connection/pane-agent-identity.ts`; +- `src/renderer/src/components/tab-bar/tab-agent-types-by-tab-id.ts`, + `src/renderer/src/components/terminal-pane/terminal-tab-agent-type-index.ts`, + `src/renderer/src/lib/tab-agent-status-index.ts`, + `src/renderer/src/components/tab-bar/terminal-tab-activity-status.ts`; +- `src/renderer/src/lib/workspace-tab-agent-metadata.ts`, + `src/renderer/src/lib/workspace-tab-palette-entry-builder.ts`, + `src/renderer/src/lib/worktree-status.ts`, + `src/renderer/src/components/sidebar/smart-attention.ts`, + `src/renderer/src/components/status-bar/workspace-space-presentation.ts`, and + `src/renderer/src/store/slices/terminal-helpers.ts`. + +The adapter returns `null`/unknown for ambiguity and title-only provenance rather than changing a +title into a confident icon. Run the two inventory tests plus the tab, title, sidebar, status, and +worktree focused suites with the repository Vitest config. + +### Tranche 2 — renderer actions and routing + +Migrate action rows 33–47 and 53, 55–60, including: + +- `src/main/runtime/orchestration/groups.ts`'s renderer-facing projection and + `src/renderer/src/lib/active-agent-note-target.ts`; +- paste/output ownership and send paths in + `src/renderer/src/components/terminal-pane/terminal-agent-paste-bracketing.ts`, + `src/renderer/src/components/terminal-pane/command-code-output-ownership.ts`, + `src/renderer/src/components/terminal-pane/pty-connection/command-inferred-pane-agent.ts`, + `src/renderer/src/components/terminal-pane/pty-connection/agent-task-complete-notify.ts`, + `src/renderer/src/components/terminal-pane/pty-connection/terminal-keydown-fit.ts`, + `src/renderer/src/components/terminal-pane/pty-connection/pane-serializer-settle.ts`, + `src/renderer/src/lib/active-agent-note-send.ts`, + `src/renderer/src/components/native-chat/native-chat-runtime-send.ts`, and the mobile send + adapters; +- readiness, follow-up, restart, native-chat, continuation/fork, keyboard, hibernation/resume, + automation reuse, cold-restore, and title-spawn-bell surfaces identified by markers in + `pane-agent-identity-surface-inventory.test.ts`. + +Action adapters pass `minimumSource: 'launch'` (or a stricter source where appropriate), require a +current run when available, and fail closed on `null`/ambiguous/title-only identity. No command, +launch flag, or shim changes. + +### Tranche 3 — renderer status, sync, and mobile projections + +Migrate row 6 and the renderer half of row 59, plus row 54: + +- `src/renderer/src/runtime/web-session-tabs-sync.ts`, + `src/renderer/src/runtime/web-session-tabs-sync/terminal-build.ts`, and + `src/renderer/src/runtime/web-session-tabs-sync/agent-status-primitives.ts`; +- `src/renderer/src/hooks/ipc-events/agent-status-event-applicator.ts`, + `src/renderer/src/hooks/ipc-events/agent-status-routing.ts`, + `src/renderer/src/store/slices/agent-status.ts`, + `src/renderer/src/store/slices/pane-foreground-agent.ts`, and + `src/renderer/src/store/slices/terminal-helpers.ts` where the marker pins identity reset; +- `src/renderer/src/runtime/sync-runtime-graph.ts` and the mobile terminal/native-chat adapters. + +Preserve host authority, observed-run transfer, retained rows, and folder-workspace behavior. A +renderer-only foreground hint is not a proof and cannot mark a pane covered. + +### Tranche 4 — main/runtime local and daemon-backed consumers + +After renderer results are stable, migrate the main-side owner/status consumers and local runtime +summary paths (rows 34, 59, and 62), including: + +- `src/main/agent-hooks/server.ts` and `src/main/agent-hooks/server/server-status-update.ts`; +- `src/main/runtime/runtime-mobile-agent-status-builder.ts`, + `src/main/runtime/runtime-mobile-session-projection.ts`, + `src/main/runtime/orca-runtime-build-pty-terminal-summary.ts`, and the runtime owner helpers; +- `src/main/runtime/orchestration/groups.ts` and its mailbox/action consumers. + +The execution host is authoritative for process evidence. Exercise native macOS/Linux, native +Windows, daemon-backed panes, and folder workspaces (not just git worktrees) before advancing. + +### Tranche 5 — published host-to-client identity (last) + +Only after all local display/action consumers use the canonical resolver, migrate +`src/shared/published-pane-agent-identity.ts` and its callers in +`src/main/runtime/orca-runtime-write-orchestration-pointer-pty.ts` and the terminal-summary builders. +The existing `agentIdentity` value remains backward-compatible. Add an optional, capability-negotiated +`agentIdentityEvidence` sidecar carrying source/coverage, authority/incarnation when known, and +freshness for process proof; old clients ignore it, and new clients treat its absence as unknown +rather than covered proof. Do not add a stream opcode. A title-only WSL route is explicitly marked +uncovered/title-only and is never relabeled as a live process proof. + +Run the remote wire compatibility tests against old/new client-host combinations, then run the full +65-row inventory ratchet one final time. + +## Host-proof ordering and platform experiments + +Do not tune the ladder against a rung that cannot fire. Land the host-stamped WSL foreground +evidence work (the near-complete sibling change) and the SSH equivalent before enabling process +proof in Tranche 1 or publishing it in Tranche 5. Until then, a bare renderer/main process name is +an uncovered hint and cannot outrank launch. + +Correctness is measured without user telemetry. For every platform with process evidence, capture +the host's independently selected foreground process (including its opaque PID/start incarnation) +and compare it with the canonical result in deterministic fixtures and an end-to-end pane run: + +- macOS and Linux POSIX: direct agent, shell wrapper, nested OMP/Pi, and ambiguous descendant trees; + assert the ambiguity fence returns unknown and never chooses by depth. +- Windows native: executable paths and `.cmd`/`.bat` launchers through the Windows process table; + assert shell/wrapper names do not masquerade as the agent. +- WSL: Windows-side `wsl.exe` plus guest inventory anchored to the distro/shell marker; verify a + guest agent proof is host-stamped and that missing/ambiguous anchors produce `unverifiable`. +- SSH: relay-stamped authority generation/epoch, reconnect, and transport-loss cases; loss of + contact is `unverifiable`, never evidence that the process exited. +- Daemon-backed and folder workspaces: repeat each applicable fixture through the daemon without + relying on git metadata. + +The fixture runner writes counts and mismatches with `writeFileSync` because Vitest intercepts +`console.log`. Run it with `npx vitest run --config config/vitest.config.ts `; a bare Vitest +command is not valid for this repository. Typecheck after clearing stale incremental artifacts (an +incremental `pnpm tc` can otherwise report a false green), and keep all scratch artifacts outside the +worktree. + +## Deletions and non-goals + +Delete all comparison-only machinery and tests once canonical calls are live: + +- `src/shared/pane-agent-identity-comparison.ts` (the comparison recorder and counters); +- `src/renderer/src/lib/tab-agent-identity-comparison.ts` and its test; +- `src/shared/published-pane-agent-identity-comparison.ts` and its test/wiring in runtime summary + publication; +- any comparison-only imports, effects, console output, or inventory rows. + +Keep `src/shared/pane-agent-identity-adapter.ts`, the canonical resolver tests, the title corpus +characterization, and both inventory ratchets. Do not add telemetry, a shadow decision, launch +flags, command shims, or a required user workflow change. + +## Single-ranking risk and mitigation + +One ranking can be wrong everywhere at once: a bad source order would affect icons, routing, +status, summaries, mobile, and remote clients simultaneously. Mitigate that systemic risk by: + +1. making the 17,496-shape harness and the 648 decision table hard gates, including the freshness + field trap and equal-rank ambiguity assertions; +2. validating host process truth independently per platform before enabling its rung; +3. migrating in blast-radius order while preserving thin adapters and an uncovered compatibility + lane, so one tranche can be reverted without rewriting 65 consumers; +4. requiring action floors and run-key supersession so a wrong display hint cannot authorize a write; +5. keeping the process selector's ambiguity fence and the separate OMP/ChatGPT.app selection fix + explicit, rather than hiding a selector defect inside ladder ordering; and +6. making the optional wire sidecar additive and capability-negotiated, with old-client behavior + unchanged. + +Success is one policy implementation, honest unknowns on ambiguous/unverifiable evidence, identical +answers at all six seams, a passing inventory ratchet after every tranche, and no telemetry or +comparison recorder left in the tree. diff --git a/src/renderer/src/lib/tab-agent-identity-comparison.test.ts b/src/renderer/src/lib/tab-agent-identity-comparison.test.ts deleted file mode 100644 index b7b918a4193..00000000000 --- a/src/renderer/src/lib/tab-agent-identity-comparison.test.ts +++ /dev/null @@ -1,120 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { PaneAgentIdentityComparisonRecorder } from '../../../shared/pane-agent-identity-comparison' -import { recordTabAgentLadderComparison } from './tab-agent-identity-comparison' -import { resolveTabAgentFromSignals } from './use-tab-agent' - -const GROK_ADVERSARIAL_TITLE = 'STA-4011 Linux Antigravity Commit Messages - grok' - -describe('tab-icon comparison lane', () => { - it('counts the title-above-hook reclaim that the rendered ladder allows and the canonical one refuses', () => { - // The rendered tab ladder consults the title before the completed hook, so a reused-looking - // title flips the icon. The canonical ladder keeps the completed hook until a run-key - // supersession proves a reclaim. This exact disagreement is what the window must surface. - const signals = { - hasObservedAgentSignal: true, - isRemote: false, - title: GROK_ADVERSARIAL_TITLE, - hookAgent: null, - focusedCompletedHookAgent: 'claude' as const, - launchAgent: undefined - } - const rendered = resolveTabAgentFromSignals(signals) - expect(rendered).toBe('grok') - - const recorder = new PaneAgentIdentityComparisonRecorder() - recordTabAgentLadderComparison( - { - tabId: 'tab-1', - worktreeId: 'wt-1', - isRemote: false, - title: signals.title, - hookAgent: signals.hookAgent, - focusedCompletedHookAgent: signals.focusedCompletedHookAgent, - launchAgent: null - }, - rendered, - recorder - ) - expect(recorder.snapshot()).toMatchObject({ - comparisons: 1, - disagreements: 1, - reclaimShapes: 1 - }) - }) - - it('agreement on a hook-covered pane records a comparison and no disagreement', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - recordTabAgentLadderComparison( - { - tabId: 'tab-2', - worktreeId: 'wt-1', - isRemote: false, - title: 'anything at all', - hookAgent: 'claude', - launchAgent: null - }, - 'claude', - recorder - ) - expect(recorder.snapshot()).toMatchObject({ comparisons: 1, disagreements: 0 }) - }) - - it('an uncovered pane preserves the rendered result as the compatibility lane', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - recordTabAgentLadderComparison( - { - tabId: 'tab-3', - worktreeId: 'wt-1', - isRemote: false, - title: GROK_ADVERSARIAL_TITLE, - hookAgent: null, - launchAgent: null - }, - 'grok', - recorder - ) - expect(recorder.snapshot()).toMatchObject({ - comparisons: 1, - disagreements: 0, - uncovered: 1 - }) - }) - - it('dedupes repeated renders of unchanged signals', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - const args = { - tabId: 'tab-4', - worktreeId: 'wt-1', - isRemote: false, - title: 'plain shell', - hookAgent: null, - launchAgent: null - } - recordTabAgentLadderComparison(args, null, recorder) - recordTabAgentLadderComparison(args, null, recorder) - expect(recorder.snapshot().comparisons).toBe(1) - }) - - it('a remote pane is recorded with remote host scope, never resolved differently', () => { - const emitted: Record[] = [] - const recorder = new PaneAgentIdentityComparisonRecorder((_line, detail) => { - if (detail && 'surface' in detail) { - emitted.push(detail) - } - }) - recordTabAgentLadderComparison( - { - tabId: 'tab-5', - worktreeId: 'wt-1', - isRemote: true, - title: GROK_ADVERSARIAL_TITLE, - hookAgent: null, - focusedCompletedHookAgent: 'claude', - launchAgent: null - }, - 'grok', - recorder - ) - expect(emitted[0]).toMatchObject({ hostScope: 'remote' }) - }) -}) diff --git a/src/renderer/src/lib/tab-agent-identity-comparison.ts b/src/renderer/src/lib/tab-agent-identity-comparison.ts deleted file mode 100644 index 5303fcad9a4..00000000000 --- a/src/renderer/src/lib/tab-agent-identity-comparison.ts +++ /dev/null @@ -1,139 +0,0 @@ -import { useEffect } from 'react' -import { collectAgentTitleEvidence } from '../../../shared/agent-title-evidence' -import { resolveCanonicalPaneAgentIdentity } from '../../../shared/pane-agent-identity-adapter' -import { - PaneAgentIdentityComparisonRecorder, - type PaneIdentityComparisonInput -} from '../../../shared/pane-agent-identity-comparison' -import type { TuiAgent } from '../../../shared/tui-agent' - -/** - * Tab-icon lane of the identity-ladder comparison window. The tab ladder is the one users - * actually see and the one that ranks a parsed title above the launch record; this wrapper - * computes the canonical answer beside the rendered one and counts where they disagree. The - * rendered result is untouched — the caller passes it in and keeps displaying it. - */ - -export type TabAgentLadderComparisonArgs = { - tabId: string - worktreeId?: string | null - isRemote: boolean - title: string - hookAgent: TuiAgent | null - siblingHookAgent?: TuiAgent | null - focusedCompletedHookAgent?: TuiAgent | null - siblingCompletedHookAgent?: TuiAgent | null - /** Renderer foreground hint — no host process proof exists, so the canonical lane treats it as - * weak evidence rather than the process rung. */ - processAgent?: TuiAgent | null - sleepingSessionAgent?: TuiAgent | null - launchAgent?: TuiAgent | null -} - -const defaultRecorder = new PaneAgentIdentityComparisonRecorder((line, sample) => { - console.info(`[pane-identity-compare] ${line}`, sample ?? {}) -}) - -export function getTabAgentLadderComparisonRecorder(): PaneAgentIdentityComparisonRecorder { - return defaultRecorder -} - -/** The tab's already-built ladder signals; a strict subset of `resolveTabAgentFromSignals` args. */ -export type TabAgentLadderSignals = { - isRemote: boolean - title: string - hookAgent: TuiAgent | null - siblingHookAgent?: TuiAgent | null - focusedCompletedHookAgent?: TuiAgent | null - siblingCompletedHookAgent?: TuiAgent | null - processAgent?: TuiAgent | null - sleepingSessionAgent?: TuiAgent | null - launchAgent?: TuiAgent | null -} - -/** Post-render on purpose: render stays pure, the rendered icon stays untouched, and the - * recorder's signature gate keeps repeat commits free. */ -export function useTabAgentLadderComparison( - tabId: string, - worktreeId: string | null | undefined, - signals: TabAgentLadderSignals, - renderedAgent: TuiAgent | null -): void { - useEffect(() => { - recordTabAgentLadderComparison( - { - tabId, - worktreeId, - isRemote: signals.isRemote, - title: signals.title, - hookAgent: signals.hookAgent, - siblingHookAgent: signals.siblingHookAgent, - focusedCompletedHookAgent: signals.focusedCompletedHookAgent, - siblingCompletedHookAgent: signals.siblingCompletedHookAgent, - processAgent: signals.processAgent, - sleepingSessionAgent: signals.sleepingSessionAgent, - launchAgent: signals.launchAgent ?? null - }, - renderedAgent - ) - }) -} - -export function recordTabAgentLadderComparison( - args: TabAgentLadderComparisonArgs, - renderedAgent: TuiAgent | null, - recorder: PaneAgentIdentityComparisonRecorder = defaultRecorder -): void { - try { - const signature = [ - args.hookAgent ?? '-', - args.siblingHookAgent ?? '-', - args.focusedCompletedHookAgent ?? '-', - args.siblingCompletedHookAgent ?? '-', - args.processAgent ?? '-', - args.sleepingSessionAgent ?? '-', - args.launchAgent ?? '-', - String(args.isRemote), - renderedAgent ?? '-', - args.title - ].join('|') - if (!recorder.shouldCompare('tab-icon', args.tabId, signature)) { - return - } - const canonical = resolveCanonicalPaneAgentIdentity({ - hookAgent: args.hookAgent, - hookIsLive: true, - completedHookAgent: args.focusedCompletedHookAgent, - launchAgent: args.launchAgent, - foregroundAgent: args.processAgent, - sleepingSessionAgent: args.sleepingSessionAgent, - siblingAgent: args.siblingHookAgent ?? args.siblingCompletedHookAgent, - allowSibling: true, - title: args.title, - uncoveredFallback: { agent: renderedAgent } - }) - const titleAgent = args.title ? collectAgentTitleEvidence(args.title).agent : null - const input: PaneIdentityComparisonInput = { - surface: 'tab-icon', - paneId: args.tabId, - worktreeId: args.worktreeId, - oldAgent: renderedAgent, - newAgent: canonical.agent, - newSource: canonical.source, - coverage: canonical.coverage, - titleOnly: canonical.titleOnly, - // Run keys reach the tab ladder with a later wave; absent means absent, not stale. - runKeyComparability: 'absent', - hostScope: args.isRemote ? 'remote' : 'local', - ambiguous: canonical.ambiguousAt !== undefined, - reclaimShape: Boolean( - args.focusedCompletedHookAgent && - titleAgent && - titleAgent !== args.focusedCompletedHookAgent - ) - } - recorder.record(input) - } catch { - // Comparison telemetry must never break the tab bar; a lost sample is recoverable. - } -} diff --git a/src/renderer/src/lib/use-tab-agent.ts b/src/renderer/src/lib/use-tab-agent.ts index 5153b38e4d7..21c00f4cc37 100644 --- a/src/renderer/src/lib/use-tab-agent.ts +++ b/src/renderer/src/lib/use-tab-agent.ts @@ -19,7 +19,6 @@ import { import { resolveCompatibleAgentTypeForOwner } from '../../../shared/agent-title-owner' import { isOpenCodeNativeTitle } from '../../../shared/opencode-terminal-title' import { resolvePaneAgentOwner } from '../../../shared/pane-agent-owner' -import { useTabAgentLadderComparison } from './tab-agent-identity-comparison' import type { TerminalTab } from '../../../shared/terminal-tab-types' import type { TuiAgent } from '../../../shared/tui-agent' @@ -338,7 +337,7 @@ export function useTabAgent(tab: TerminalTab): TuiAgent | null { tab.title ]) - const signals = { + return resolveTabAgentFromSignals({ hasObservedAgentSignal, isRemote: isRemoteLike, title: tab.title, @@ -351,10 +350,5 @@ export function useTabAgent(tab: TerminalTab): TuiAgent | null { processShellForeground, sleepingSessionAgent, launchAgent: tab.launchAgent - } - const renderedAgent = resolveTabAgentFromSignals(signals) - // Identity-ladder comparison window (post-render): counts canonical-vs-rendered disagreement; - // the rendered icon stays untouched. - useTabAgentLadderComparison(tab.id, tab.worktreeId, signals, renderedAgent) - return renderedAgent + }) } diff --git a/src/shared/agent-status-identity.ts b/src/shared/agent-status-identity.ts index f3d4b65a86b..9cb81625310 100644 --- a/src/shared/agent-status-identity.ts +++ b/src/shared/agent-status-identity.ts @@ -4,6 +4,8 @@ import { type AgentStatusState, type AgentType } from './agent-status-types' +import { resolveCanonicalPaneAgentIdentity } from './pane-agent-identity-adapter' +import type { TuiAgent } from './tui-agent' type ExistingAgentIdentity = { agentType?: AgentType @@ -63,6 +65,11 @@ export function resolveAgentStatusIdentity(args: { inheritedFromActivePane: false } } + const canonical = resolveCanonicalPaneAgentIdentity({ + hookAgent: incomingAgentType as TuiAgent, + hookIsLive: true, + completedHookAgent: args.existing.state === 'done' ? (existingAgentType as TuiAgent) : undefined + }) if (isActiveExistingIdentity(args.existing, args.now, staleAfterMs)) { return { // Why: child agent CLIs inherit ORCA_PANE_KEY from their parent terminal. @@ -74,7 +81,7 @@ export function resolveAgentStatusIdentity(args: { } return { - agentType: incomingAgentType, + agentType: canonical.agent ?? incomingAgentType, inheritedFromActivePane: false } } diff --git a/src/shared/pane-agent-identity-adapter.ts b/src/shared/pane-agent-identity-adapter.ts index 6562f5f551d..ec9b6b6accd 100644 --- a/src/shared/pane-agent-identity-adapter.ts +++ b/src/shared/pane-agent-identity-adapter.ts @@ -1,20 +1,16 @@ import { collectAgentTitleEvidence } from './agent-title-evidence' -import { - resolvePaneAgentIdentity, - type PaneAgentEvidence, - type PaneAgentEvidenceSource, - type PaneAgentRunKey +import type { + PaneAgentEvidence, + PaneAgentEvidenceSource, + PaneAgentIdentity, + PaneAgentIdentityInput, + PaneAgentRunKey } from './pane-agent-identity-resolver' import type { TuiAgent } from './tui-agent' /** - * The parallel adapter entry point around `resolvePaneAgentIdentity`. - * - * The frozen host adapter (`published-pane-agent-identity.ts`) stays untouched and keeps - * publishing exactly what it publishes today. This adapter computes the CANONICAL answer the - * migration will eventually ship — per-pane coverage, provenance sidecar, process-proof gating — - * so comparison telemetry can log where the two disagree on real sessions BEFORE any surface - * changes what it displays. Nothing user-visible reads this module's answer yet. + * Canonical pane identity ranking. All adapters, including the compatibility resolver, delegate to + * this implementation so source precedence, ambiguity, and run eligibility cannot drift. */ /** @@ -94,6 +90,8 @@ export type CanonicalPaneAgentIdentityInput = { sleepingRun?: PaneAgentRunKey /** Tab-level display fallback only; ignored unless `allowSibling` opts in. */ siblingAgent?: TuiAgent | null + /** Additional tab-level sibling observations retained for ambiguity checking. */ + siblingAgents?: readonly TuiAgent[] allowSibling?: boolean title?: string | null currentRun?: PaneAgentRunKey @@ -116,6 +114,66 @@ export type CanonicalPaneAgentIdentity = { supersededSources: readonly PaneAgentEvidenceSource[] } +/** Authority order, strongest first. This is the only place precedence is expressed. */ +const SOURCE_RANK: readonly PaneAgentEvidenceSource[] = [ + 'live-hook', + 'process', + 'launch', + 'completed-hook', + 'sleeping-session', + 'sibling', + 'title' +] + +/** Run keys only supersede evidence from the same authority; unknown authorities stay eligible. */ +function isPaneAgentRunEligible( + run: PaneAgentRunKey | undefined, + currentRun: PaneAgentRunKey | undefined +): boolean { + return ( + run === undefined || + currentRun === undefined || + run.authorityId !== currentRun.authorityId || + run.incarnation === currentRun.incarnation + ) +} + +/** Shared evidence ranking primitive used by every pane-identity adapter. */ +export function resolveCanonicalPaneAgentEvidence( + input: PaneAgentIdentityInput +): PaneAgentIdentity { + const superseded: PaneAgentEvidenceSource[] = [] + const floor = input.minimumSource + ? SOURCE_RANK.indexOf(input.minimumSource) + : Number.MAX_SAFE_INTEGER + const eligible = input.evidence.filter((item) => { + if (item.source === 'sibling' && input.allowSibling !== true) { + return false + } + if (SOURCE_RANK.indexOf(item.source) > floor) { + return false + } + if (isPaneAgentRunEligible(item.run, input.currentRun)) { + return true + } + superseded.push(item.source) + return false + }) + + for (const source of SOURCE_RANK) { + const matches = eligible.filter((item) => item.source === source) + if (matches.length === 0) { + continue + } + const agents = new Set(matches.map((item) => item.agent)) + if (agents.size > 1) { + return { agent: null, source: null, ambiguousAt: source, supersededSources: superseded } + } + return { agent: matches[0].agent, source, supersededSources: superseded } + } + return { agent: null, source: null, supersededSources: superseded } +} + /** Freshness is judged on the authority's own clock: age at capture against its TTL. */ export function isForegroundProcessProofFresh(proof: ForegroundProcessProof): boolean { return ( @@ -148,17 +206,13 @@ export function resolveCanonicalPaneAgentIdentity( // Coverage comes from authority-bearing sources that are still eligible for this run. A stale // hook/launch row can remain in the input after a pane is replaced; it must not make a title-only // answer look covered to a future action consumer. - const runIsEligible = (run: PaneAgentRunKey | undefined): boolean => - run === undefined || - input.currentRun === undefined || - run.authorityId !== input.currentRun.authorityId || - run.incarnation === input.currentRun.incarnation const covered = Boolean( - (input.hookAgent && runIsEligible(input.hookRun)) || - (input.completedHookAgent && runIsEligible(input.completedHookRun)) || + (input.hookAgent && isPaneAgentRunEligible(input.hookRun, input.currentRun)) || + (input.completedHookAgent && + isPaneAgentRunEligible(input.completedHookRun, input.currentRun)) || processEvidence || - (input.launchAgent && runIsEligible(input.launchRun)) || - (input.sleepingSessionAgent && runIsEligible(input.sleepingRun)) + (input.launchAgent && isPaneAgentRunEligible(input.launchRun, input.currentRun)) || + (input.sleepingSessionAgent && isPaneAgentRunEligible(input.sleepingRun, input.currentRun)) ) // Keep stale evidence in the resolver so diagnostics still report which source was superseded, // even when it no longer qualifies the pane as covered. @@ -184,16 +238,27 @@ export function resolveCanonicalPaneAgentIdentity( supersededSources: [] } } + const siblingEvidence = [ + ...(input.siblingAgent ? [{ source: 'sibling' as const, agent: input.siblingAgent }] : []), + ...(input.siblingAgents?.map((agent) => ({ source: 'sibling' as const, agent })) ?? []), + ...(titleAgent ? [{ source: 'title' as const, agent: titleAgent }] : []) + ] + const siblingResolved = resolveCanonicalPaneAgentEvidence({ + evidence: siblingEvidence, + allowSibling: input.allowSibling, + minimumSource: input.minimumSource + }) return { - agent: titleAgent, - source: titleAgent ? 'title' : null, + agent: siblingResolved.agent, + source: siblingResolved.source, coverage: 'uncovered', - titleOnly: titleAgent !== null, - supersededSources: [] + titleOnly: siblingResolved.source === 'title', + ...(siblingResolved.ambiguousAt ? { ambiguousAt: siblingResolved.ambiguousAt } : {}), + supersededSources: siblingResolved.supersededSources } } - const resolved = resolvePaneAgentIdentity({ + const resolved = resolveCanonicalPaneAgentEvidence({ evidence: [ ...(input.hookAgent ? [ @@ -233,6 +298,7 @@ export function resolveCanonicalPaneAgentIdentity( ] : []), ...(input.siblingAgent ? [{ source: 'sibling' as const, agent: input.siblingAgent }] : []), + ...(input.siblingAgents?.map((agent) => ({ source: 'sibling' as const, agent })) ?? []), ...(titleAgent ? [{ source: 'title' as const, agent: titleAgent }] : []) ], currentRun: input.currentRun, diff --git a/src/shared/pane-agent-identity-comparison.test.ts b/src/shared/pane-agent-identity-comparison.test.ts deleted file mode 100644 index 64573d3410e..00000000000 --- a/src/shared/pane-agent-identity-comparison.test.ts +++ /dev/null @@ -1,116 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { - PaneAgentIdentityComparisonRecorder, - type PaneIdentityComparisonInput -} from './pane-agent-identity-comparison' - -function sample(overrides: Partial = {}): PaneIdentityComparisonInput { - return { - surface: 'terminal-summary', - paneId: 'tab-1:leaf-1', - worktreeId: 'wt-1', - oldAgent: 'claude', - newAgent: 'claude', - newSource: 'launch', - coverage: 'covered', - titleOnly: false, - runKeyComparability: 'absent', - hostScope: 'local', - ambiguous: false, - reclaimShape: false, - ...overrides - } -} - -describe('comparison counters', () => { - it('counts disagreements and both absence-transition directions separately', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - recorder.record(sample()) - recorder.record(sample({ oldAgent: 'codex', newAgent: 'claude' })) - recorder.record(sample({ oldAgent: null, newAgent: 'claude' })) - recorder.record(sample({ oldAgent: 'claude', newAgent: null, newSource: null })) - expect(recorder.snapshot()).toMatchObject({ - comparisons: 4, - disagreements: 3, - oldAbsentNewPresent: 1, - oldPresentNewAbsent: 1 - }) - }) - - it('counts ambiguity, reclaim shapes, title-only, and uncovered lanes', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - recorder.record(sample({ ambiguous: true })) - recorder.record(sample({ reclaimShape: true })) - recorder.record(sample({ titleOnly: true, coverage: 'uncovered' })) - expect(recorder.snapshot()).toMatchObject({ - ambiguous: 1, - reclaimShapes: 1, - titleOnly: 1, - uncovered: 1 - }) - }) -}) - -describe('sampling and bounds', () => { - it('skips consecutive identical input signatures per pane, and resumes on change', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - expect(recorder.shouldCompare('tab-icon', 'tab-1', 'sig-a')).toBe(true) - expect(recorder.shouldCompare('tab-icon', 'tab-1', 'sig-a')).toBe(false) - expect(recorder.shouldCompare('tab-icon', 'tab-2', 'sig-a')).toBe(true) - expect(recorder.shouldCompare('tab-icon', 'tab-1', 'sig-b')).toBe(true) - expect(recorder.shouldCompare('tab-icon', 'tab-1', 'sig-a')).toBe(true) - }) - - it('emits one detail record per distinct disagreement shape, hard-capped', () => { - const emitted: Record[] = [] - const recorder = new PaneAgentIdentityComparisonRecorder((_line, detail) => { - if (detail && 'surface' in detail) { - emitted.push(detail) - } - }) - recorder.record(sample({ oldAgent: 'codex' })) - recorder.record(sample({ oldAgent: 'codex' })) - expect(emitted).toHaveLength(1) - for (let i = 0; i < 100; i += 1) { - recorder.record(sample({ oldAgent: `agent-${i}` })) - } - expect(emitted.length).toBeLessThanOrEqual(41) - expect(recorder.snapshot().disagreements).toBe(102) - }) -}) - -describe('privacy contract', () => { - it('emitted records pseudonymize ids and never contain a title-like payload', () => { - const rawTitle = 'SECRET /Users/someone/private/path — do not leak' - const emitted: { line: string; detail?: Record }[] = [] - const recorder = new PaneAgentIdentityComparisonRecorder((line, detail) => { - emitted.push({ line, detail }) - }) - recorder.record( - sample({ paneId: 'raw-pane-id', worktreeId: 'raw-worktree-id', oldAgent: 'codex' }) - ) - expect(emitted.length).toBeGreaterThan(0) - for (const { line, detail } of emitted) { - const serialized = `${line} ${JSON.stringify(detail ?? {})}` - expect(serialized).not.toContain(rawTitle) - expect(serialized).not.toContain('raw-pane-id') - expect(serialized).not.toContain('raw-worktree-id') - expect(detail).not.toHaveProperty('title') - } - }) - - it('two recorders pseudonymize the same id differently (per-process salt)', () => { - const captured: string[] = [] - const capture = (_line: string, detail?: Record) => { - if (detail && typeof detail.pane === 'string') { - captured.push(detail.pane) - } - } - const a = new PaneAgentIdentityComparisonRecorder(capture) - const b = new PaneAgentIdentityComparisonRecorder(capture) - a.record(sample({ oldAgent: 'codex' })) - b.record(sample({ oldAgent: 'codex' })) - expect(captured).toHaveLength(2) - expect(captured[0]).not.toBe(captured[1]) - }) -}) diff --git a/src/shared/pane-agent-identity-comparison.ts b/src/shared/pane-agent-identity-comparison.ts deleted file mode 100644 index bc2d92d2ccf..00000000000 --- a/src/shared/pane-agent-identity-comparison.ts +++ /dev/null @@ -1,176 +0,0 @@ -import type { PaneAgentCoverage } from './pane-agent-identity-adapter' -import type { PaneAgentEvidenceSource } from './pane-agent-identity-resolver' - -/** - * Comparison telemetry for the identity-ladder migration: where do the old and canonical ladders - * DISAGREE on real sessions? Recorded BEFORE any surface changes what it displays, so each later - * flip is a measured decision instead of a guess. - * - * Privacy contract: emitted records carry pseudonymous salted ids, agent enum values, sources, - * coverage, and counters — never raw titles, prompts, commands, file paths, or tokens. - */ - -export type PaneIdentityComparisonSurface = 'terminal-summary' | 'pty-terminal-summary' | 'tab-icon' - -export type PaneIdentityRunKeyComparability = 'comparable' | 'incomparable' | 'absent' - -export type PaneIdentityHostScope = 'local' | 'remote' | 'unknown' - -export type PaneIdentityComparisonInput = { - surface: PaneIdentityComparisonSurface - /** Raw pane/tab identifier; pseudonymized before it reaches any emitted record. */ - paneId: string - worktreeId?: string | null - oldAgent: string | null - newAgent: string | null - newSource: PaneAgentEvidenceSource | null - coverage: PaneAgentCoverage - titleOnly: boolean - runKeyComparability: PaneIdentityRunKeyComparability - hostScope: PaneIdentityHostScope - /** The canonical resolver saw equally-ranked conflicting evidence. */ - ambiguous: boolean - /** The bug-versus-reclaim input shape: a completed hook naming A beside a title naming B. */ - reclaimShape: boolean -} - -export type PaneIdentityComparisonCounters = { - comparisons: number - disagreements: number - ambiguous: number - reclaimShapes: number - /** Flipping would turn a published absence into a presence — `groups.ts` reads absence as NO. */ - oldAbsentNewPresent: number - oldPresentNewAbsent: number - titleOnly: number - uncovered: number -} - -const MAX_PANE_SIGNATURES = 2048 -const MAX_DISAGREEMENT_KEYS = 40 -/** Counter snapshots go out on a log scale so a busy session cannot flood the sink. */ -const SNAPSHOT_AT = [100, 1_000, 10_000, 100_000, 1_000_000] - -function pseudonymize(salt: string, value: string): string { - // djb2: stable within one process, meaningless outside it. Pseudonymity, not secrecy — the raw - // id never leaves the process either way. - let hash = 5381 - const input = `${salt}:${value}` - for (let i = 0; i < input.length; i += 1) { - hash = ((hash << 5) + hash + input.charCodeAt(i)) | 0 - } - return (hash >>> 0).toString(16).padStart(8, '0') -} - -export class PaneAgentIdentityComparisonRecorder { - private readonly counters: PaneIdentityComparisonCounters = { - comparisons: 0, - disagreements: 0, - ambiguous: 0, - reclaimShapes: 0, - oldAbsentNewPresent: 0, - oldPresentNewAbsent: 0, - titleOnly: 0, - uncovered: 0 - } - private readonly lastSignatureByPane = new Map() - private readonly emittedDisagreementKeys = new Set() - private readonly salt: string - - constructor( - private readonly emit: (line: string, sample?: Record) => void = () => {} - ) { - this.salt = globalThis.crypto.randomUUID() - } - - /** - * Consecutive-duplicate gate, cheap enough for a render/summary path: callers build a signature - * from the ladder INPUTS and skip the (costlier) canonical resolution when nothing changed. - */ - shouldCompare( - surface: PaneIdentityComparisonSurface, - paneId: string, - signature: string - ): boolean { - const key = `${surface}|${paneId}` - if (this.lastSignatureByPane.get(key) === signature) { - return false - } - this.lastSignatureByPane.set(key, signature) - while (this.lastSignatureByPane.size > MAX_PANE_SIGNATURES) { - const oldest = this.lastSignatureByPane.keys().next().value - if (oldest === undefined) { - break - } - this.lastSignatureByPane.delete(oldest) - } - return true - } - - record(input: PaneIdentityComparisonInput): void { - this.counters.comparisons += 1 - if (input.ambiguous) { - this.counters.ambiguous += 1 - } - if (input.reclaimShape) { - this.counters.reclaimShapes += 1 - } - if (input.titleOnly) { - this.counters.titleOnly += 1 - } - if (input.coverage === 'uncovered') { - this.counters.uncovered += 1 - } - const disagrees = input.oldAgent !== input.newAgent - if (disagrees) { - this.counters.disagreements += 1 - if (input.oldAgent === null) { - this.counters.oldAbsentNewPresent += 1 - } - if (input.newAgent === null) { - this.counters.oldPresentNewAbsent += 1 - } - this.emitDisagreement(input) - } - if (SNAPSHOT_AT.includes(this.counters.comparisons)) { - this.emit('pane-identity-compare counters', { ...this.counters }) - } - } - - snapshot(): PaneIdentityComparisonCounters { - return { ...this.counters } - } - - private emitDisagreement(input: PaneIdentityComparisonInput): void { - const key = [ - input.surface, - input.oldAgent ?? '-', - input.newAgent ?? '-', - input.newSource ?? '-', - input.coverage, - input.hostScope - ].join('|') - // One detail record per distinct disagreement shape, hard-capped; repeats only count. - if ( - this.emittedDisagreementKeys.has(key) || - this.emittedDisagreementKeys.size >= MAX_DISAGREEMENT_KEYS - ) { - return - } - this.emittedDisagreementKeys.add(key) - this.emit('pane-identity-compare disagreement', { - surface: input.surface, - pane: pseudonymize(this.salt, input.paneId), - ...(input.worktreeId ? { worktree: pseudonymize(this.salt, input.worktreeId) } : {}), - oldAgent: input.oldAgent, - newAgent: input.newAgent, - newSource: input.newSource, - coverage: input.coverage, - titleOnly: input.titleOnly, - runKeyComparability: input.runKeyComparability, - hostScope: input.hostScope, - ambiguous: input.ambiguous, - reclaimShape: input.reclaimShape - }) - } -} diff --git a/src/shared/pane-agent-identity-decision-table.test.ts b/src/shared/pane-agent-identity-decision-table.test.ts new file mode 100644 index 00000000000..1f9d06f431f --- /dev/null +++ b/src/shared/pane-agent-identity-decision-table.test.ts @@ -0,0 +1,211 @@ +import { writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { + resolveCanonicalPaneAgentIdentity, + type CanonicalPaneAgentIdentity +} from './pane-agent-identity-adapter' +import type { TuiAgent } from './tui-agent' + +const AGENTS: readonly TuiAgent[] = ['claude', 'codex'] +const SLOT_COUNT = 7 +const SHAPE_COUNT = 3 ** SLOT_COUNT * 4 * 2 +const TITLES: readonly string[] = ['', 'zsh', 'Task - claude', 'Task - codex'] + +type Breakdown = Record<'launch' | 'completed-hook' | 'sleeping-session' | 'process', number> + +function slotValues(mask: number): (TuiAgent | null)[] { + let remaining = mask + return Array.from({ length: SLOT_COUNT }, () => { + const value = remaining % 3 + remaining = Math.floor(remaining / 3) + return value === 0 ? null : AGENTS[value - 1] + }) +} + +function oldTabResult(values: readonly (TuiAgent | null)[], title: string, remote: boolean) { + const [hook, siblingHook, completed, siblingCompleted, process, sleeping, launch] = values + void remote + if (hook) { + return hook + } + if (process) { + return process + } + if (completed && !sleeping && (title === 'Task - claude' || title === 'Task - codex')) { + return title === 'Task - claude' ? 'claude' : 'codex' + } + if (completed) { + return completed + } + if (sleeping) { + return sleeping + } + if (!launch && siblingHook && siblingCompleted && siblingHook !== siblingCompleted) { + return null + } + if (title === 'Task - claude') { + return 'claude' + } + if (title === 'Task - codex') { + return 'codex' + } + if (launch) { + return launch + } + return siblingHook ?? siblingCompleted ?? null +} + +function canonicalResult( + values: readonly (TuiAgent | null)[], + title: string, + withProof: boolean +): CanonicalPaneAgentIdentity { + const [hook, siblingHook, completed, siblingCompleted, process, sleeping, launch] = values + return resolveCanonicalPaneAgentIdentity({ + hookAgent: hook, + hookIsLive: hook !== null, + completedHookAgent: completed, + launchAgent: launch, + foregroundAgent: process, + processProof: + withProof && process + ? { + agent: process, + processIncarnation: 'fixture-process', + authorityId: 'fixture-authority', + capturedAgeMs: 10, + validForMs: 1_000 + } + : undefined, + sleepingSessionAgent: sleeping, + siblingAgents: [siblingHook, siblingCompleted].filter( + (agent): agent is TuiAgent => agent !== null + ), + allowSibling: true, + title + }) +} + +function runDecisionTable(withProof: boolean) { + let disagreements = 0 + let flipped = 0 + const breakdown: Breakdown = { + launch: 0, + 'completed-hook': 0, + 'sleeping-session': 0, + process: 0 + } + for (let mask = 0; mask < 3 ** SLOT_COUNT; mask += 1) { + const values = slotValues(mask) + for (const title of TITLES) { + for (const remote of [false, true]) { + const old = oldTabResult(values, title, remote) + const canonical = canonicalResult(values, title, withProof) + // The table groups only the approved residual rungs; process-only/ambiguous mismatches are + // accounted for separately by the 1,872 process-starvation flip count below. + if (old !== canonical.agent && canonical.source !== null && canonical.source in breakdown) { + disagreements += 1 + breakdown[canonical.source as keyof Breakdown] += 1 + } + if (!withProof) { + const proven = canonicalResult(values, title, true) + if ( + canonical.agent !== proven.agent && + proven.source === 'process' && + (canonical.source === 'launch' || + canonical.source === 'completed-hook' || + canonical.source === 'sleeping-session') + ) { + flipped += 1 + } + } + } + } + } + return { disagreements, flipped, breakdown } +} + +describe('approved pane-agent ladder decision table', () => { + it('replays all 17,496 shapes and asserts totals plus per-rung breakdown', () => { + const proofFree = runDecisionTable(false) + const freshProof = runDecisionTable(true) + const result = { + shapes: SHAPE_COUNT, + proofOmitted: proofFree, + freshProof, + flippedByAddingProof: proofFree.flipped + } + writeFileSync( + join(tmpdir(), 'orca-pane-agent-identity-decision-table.json'), + `${JSON.stringify(result, null, 2)}\n` + ) + expect(proofFree.disagreements).toBe(2_520) + expect(proofFree.breakdown).toEqual({ + launch: 1_908, + 'completed-hook': 468, + 'sleeping-session': 144, + process: 0 + }) + expect(freshProof.disagreements).toBe(648) + expect(freshProof.breakdown).toEqual({ + launch: 612, + 'completed-hook': 36, + 'sleeping-session': 0, + process: 0 + }) + expect(proofFree.flipped).toBe(1_872) + }) + + it('requires every freshness field before the process rung can win', () => { + const values = [null, null, null, null, 'codex', null, 'claude'] as const + const missingAge = canonicalResult(values, '', true) + const missingFreshness = resolveCanonicalPaneAgentIdentity({ + foregroundAgent: 'codex', + processProof: { + agent: 'codex', + processIncarnation: 'fixture-process', + authorityId: 'fixture-authority', + capturedAgeMs: undefined as unknown as number, + validForMs: 1_000 + }, + launchAgent: 'claude' + }) + expect(missingAge).toMatchObject({ agent: 'codex', source: 'process' }) + expect(missingFreshness).toMatchObject({ agent: 'claude', source: 'launch' }) + expect( + resolveCanonicalPaneAgentIdentity({ + foregroundAgent: 'codex', + processProof: { + agent: 'codex', + processIncarnation: 'fixture-process', + authorityId: 'fixture-authority', + capturedAgeMs: 10, + validForMs: undefined as unknown as number + }, + launchAgent: 'claude' + }) + ).toMatchObject({ agent: 'claude', source: 'launch' }) + }) + + it('fences equal-rank conflicts, superseded runs, and title-last fallback', () => { + expect( + resolveCanonicalPaneAgentIdentity({ + siblingAgents: ['claude', 'codex'], + allowSibling: true + }) + ).toMatchObject({ agent: null, ambiguousAt: 'sibling' }) + expect( + resolveCanonicalPaneAgentIdentity({ + completedHookAgent: 'claude', + completedHookRun: { authorityId: 'fixture', incarnation: 1 }, + currentRun: { authorityId: 'fixture', incarnation: 2 }, + title: 'Task - codex' + }) + ).toMatchObject({ agent: 'codex', source: 'title', supersededSources: ['completed-hook'] }) + expect( + resolveCanonicalPaneAgentIdentity({ launchAgent: 'claude', title: 'Codex' }) + ).toMatchObject({ agent: 'claude', source: 'launch' }) + }) +}) diff --git a/src/shared/pane-agent-identity-inventory.test.ts b/src/shared/pane-agent-identity-inventory.test.ts index fc9901d719b..df44f59dfe2 100644 --- a/src/shared/pane-agent-identity-inventory.test.ts +++ b/src/shared/pane-agent-identity-inventory.test.ts @@ -24,10 +24,7 @@ const HELPERS = [ 'resolveAgentTypeFromTerminalTitle', 'resolvePaneAgentIdentity', 'resolveCanonicalPaneAgentIdentity', - 'resolvePublishedPaneAgentIdentity', - 'comparePublishedPaneAgentIdentity', - 'recordTabAgentLadderComparison', - 'useTabAgentLadderComparison' + 'resolvePublishedPaneAgentIdentity' ] as const const TEST_SUPPORT_PATHS = new Set([ @@ -301,7 +298,14 @@ const INVENTORY: readonly InventoryGroup[] = [ helper: 'detectAgentStatusFromTitle', classification: 'evidence-producer', paths: [ - ['src/main/runtime/orca-runtime.ts', 12], + ['src/main/runtime/orca-runtime-apply-tracked-pty-title.ts', 2], + ['src/main/runtime/orca-runtime-get-pty-record-for-pane-key.ts', 2], + ['src/main/runtime/orca-runtime-get-unpersisted-tracked-title-for-pty.ts', 2], + ['src/main/runtime/orca-runtime-maybe-hydrate-headless-from-renderer.ts', 2], + ['src/main/runtime/orca-runtime-record-agent-prompt-lifecycle-state.ts', 2], + ['src/main/runtime/runtime-terminal-agent-status-query.ts', 3], + ['src/main/runtime/runtime-worktree-status-projection.ts', 4], + ['src/main/runtime/terminal-wait-detection.ts', 2], ['src/renderer/src/components/terminal-pane/agent-completion-title-observer.ts', 2], ['src/renderer/src/components/terminal-pane/pty-connection/shell-command-inference.ts', 4], ['src/renderer/src/components/terminal-pane/pty-output-title-observer.ts', 2], @@ -345,10 +349,7 @@ const INVENTORY: readonly InventoryGroup[] = [ { helper: 'resolvePaneAgentIdentity', classification: 'parser-implementation', - paths: [ - ['src/shared/pane-agent-identity-adapter.ts', 2], - 'src/shared/pane-agent-identity-resolver.ts' - ] + paths: ['src/shared/pane-agent-identity-resolver.ts'] }, { helper: 'resolvePaneAgentIdentity', @@ -363,45 +364,20 @@ const INVENTORY: readonly InventoryGroup[] = [ { helper: 'resolveCanonicalPaneAgentIdentity', classification: 'identity-consumer', + paths: [['src/shared/agent-status-identity.ts', 2]] + }, + { + helper: 'resolveCanonicalPaneAgentIdentity', + classification: 'identity-consumer', + paths: [['src/shared/terminal-title-agent-type.ts', 2]] + }, + { + helper: 'resolvePublishedPaneAgentIdentity', + classification: 'parser-implementation', paths: [ - ['src/renderer/src/lib/tab-agent-identity-comparison.ts', 2], - ['src/shared/published-pane-agent-identity-comparison.ts', 2] + 'src/shared/published-pane-agent-identity.ts', + ['src/main/runtime/orca-runtime-write-orchestration-pointer-pty.ts', 2] ] - }, - { - helper: 'resolvePublishedPaneAgentIdentity', - classification: 'parser-implementation', - paths: ['src/shared/published-pane-agent-identity.ts'] - }, - { - helper: 'resolvePublishedPaneAgentIdentity', - classification: 'identity-consumer', - paths: [['src/shared/published-pane-agent-identity-comparison.ts', 2]] - }, - { - helper: 'comparePublishedPaneAgentIdentity', - classification: 'parser-implementation', - paths: ['src/shared/published-pane-agent-identity-comparison.ts'] - }, - { - helper: 'comparePublishedPaneAgentIdentity', - classification: 'identity-consumer', - paths: [['src/main/runtime/orca-runtime.ts', 2]] - }, - { - helper: 'recordTabAgentLadderComparison', - classification: 'parser-implementation', - paths: [['src/renderer/src/lib/tab-agent-identity-comparison.ts', 2]] - }, - { - helper: 'useTabAgentLadderComparison', - classification: 'parser-implementation', - paths: ['src/renderer/src/lib/tab-agent-identity-comparison.ts'] - }, - { - helper: 'useTabAgentLadderComparison', - classification: 'identity-consumer', - paths: [['src/renderer/src/lib/use-tab-agent.ts', 2]] } ] diff --git a/src/shared/pane-agent-identity-resolver.ts b/src/shared/pane-agent-identity-resolver.ts index 98342963415..c05bb639436 100644 --- a/src/shared/pane-agent-identity-resolver.ts +++ b/src/shared/pane-agent-identity-resolver.ts @@ -1,3 +1,4 @@ +import { resolveCanonicalPaneAgentEvidence } from './pane-agent-identity-adapter' import type { TuiAgent } from './tui-agent' /** @@ -41,9 +42,6 @@ export const PANE_AGENT_EVIDENCE_SOURCES = [ ] as const export type PaneAgentEvidenceSource = (typeof PANE_AGENT_EVIDENCE_SOURCES)[number] -/** Authority order, strongest first. Position here is the ONLY place precedence is expressed. */ -const SOURCE_RANK: readonly PaneAgentEvidenceSource[] = PANE_AGENT_EVIDENCE_SOURCES - /** * Which agent run a piece of evidence belongs to. * @@ -112,52 +110,5 @@ export type PaneAgentIdentity = { export function resolvePaneAgentIdentity( input: PaneAgentIdentityInput ): PaneAgentIdentity { - const superseded: PaneAgentEvidenceSource[] = [] - const floor = input.minimumSource - ? SOURCE_RANK.indexOf(input.minimumSource) - : Number.MAX_SAFE_INTEGER - - const eligible = input.evidence.filter((item) => { - if (item.source === 'sibling' && input.allowSibling !== true) { - return false - } - // Why the floor: an action consumer must not be able to act on a title, at any rank. Dropping - // the evidence entirely rather than ranking it lower makes misuse impossible rather than - // unlikely — a caller cannot accidentally consult it by reordering. - if (SOURCE_RANK.indexOf(item.source) > floor) { - return false - } - if (input.currentRun === undefined || item.run === undefined) { - // Why eligible: absence means "this peer does not publish run keys", not "this is stale". - // Treating unknown as superseded would blank every row from an older host. - return true - } - if (item.run.authorityId !== input.currentRun.authorityId) { - // Why eligible and NOT superseded: runs from different authorities are incomparable, not - // older. A restarted main counts from its own floor, so `incarnation` alone would falsely - // equate unrelated runs. Incomparable evidence is treated as unknown, like an absent key. - return true - } - if (item.run.incarnation === input.currentRun.incarnation) { - return true - } - superseded.push(item.source) - return false - }) - - for (const source of SOURCE_RANK) { - const matches = eligible.filter((item) => item.source === source) - if (matches.length === 0) { - continue - } - const agents = new Set(matches.map((item) => item.agent)) - if (agents.size > 1) { - // Why null and not the first: two observations of the same class naming different agents is - // a genuine conflict, and picking one would make the answer depend on array order — the very - // property this resolver exists to remove. Fall through to nothing rather than guess. - return { agent: null, source: null, ambiguousAt: source, supersededSources: superseded } - } - return { agent: matches[0].agent, source, supersededSources: superseded } - } - return { agent: null, source: null, supersededSources: superseded } + return resolveCanonicalPaneAgentEvidence(input) } diff --git a/src/shared/pane-agent-identity-surface-inventory.test.ts b/src/shared/pane-agent-identity-surface-inventory.test.ts index bba535748c3..77d9edb3b84 100644 --- a/src/shared/pane-agent-identity-surface-inventory.test.ts +++ b/src/shared/pane-agent-identity-surface-inventory.test.ts @@ -29,7 +29,7 @@ const SURFACE_ROWS: readonly SurfaceRow[] = [ }, { row: 32, - path: 'src/renderer/src/components/terminal-pane/TerminalPane.tsx', + path: 'src/renderer/src/components/terminal-pane/use-terminal-pane-chat-state.ts', marker: 'resolveNativeChatLeafTitleAgent' }, { @@ -111,7 +111,7 @@ const SURFACE_ROWS: readonly SurfaceRow[] = [ }, { row: 47, - path: 'src/renderer/src/components/terminal-pane/use-terminal-pane-lifecycle.ts', + path: 'src/renderer/src/components/terminal-pane/terminal-pane-manager-options.ts', marker: 'resolvePaneKeyboardProtocolAgent' }, { @@ -174,8 +174,16 @@ const SURFACE_ROWS: readonly SurfaceRow[] = [ path: 'src/renderer/src/components/terminal-pane/pty-connection/cold-restore-resume-startup.ts', marker: 'bindBuildColdRestoreAgentResumeStartup' }, - { row: 59, path: 'src/main/agent-hooks/server.ts', marker: 'recordCurrentAuthorityObservation' }, - { row: 59, path: 'src/main/runtime/orca-runtime.ts', marker: 'resolvePaneAgentIdentityField' }, + { + row: 59, + path: 'src/main/agent-hooks/server/server-authority-evidence.ts', + marker: 'recordCurrentAuthorityObservation' + }, + { + row: 59, + path: 'src/main/runtime/orca-runtime-write-orchestration-pointer-pty.ts', + marker: 'resolvePaneAgentIdentityField' + }, { row: 59, path: 'src/renderer/src/hooks/ipc-events/agent-status-event-applicator.ts', @@ -183,7 +191,7 @@ const SURFACE_ROWS: readonly SurfaceRow[] = [ }, { row: 59, - path: 'src/renderer/src/store/slices/agent-status.ts', + path: 'src/renderer/src/store/slices/agent-status-authority-actions.ts', marker: 'transferAgentPaneAuthority' }, { @@ -202,7 +210,11 @@ const SURFACE_ROWS: readonly SurfaceRow[] = [ marker: 'installTitleSpawnBell' }, { row: 61, path: 'src/renderer/src/lib/worktree-status.ts', marker: 'getWorktreeStatus' }, - { row: 62, path: 'src/main/runtime/orca-runtime.ts', marker: 'getLeafWorktreeStatus' }, + { + row: 62, + path: 'src/main/runtime/runtime-worktree-status-projection.ts', + marker: 'getLeafWorktreeStatus' + }, { row: 63, path: 'src/renderer/src/components/sidebar/smart-attention.ts', @@ -245,8 +257,10 @@ describe('pane agent identity surface inventory (rows 6, 32–65)', () => { const IDENTITY_SEQUENCER_REBIND_RE = /\b(?:observations|rendererAgentStatusObservations)\.rebind\(/g const EXPECTED_REBIND_SITES: readonly (readonly [path: string, occurrences: number])[] = [ - ['src/main/agent-hooks/server.ts', 3], - ['src/renderer/src/store/slices/agent-status.ts', 1] + ['src/main/agent-hooks/server/server-ingest-normalization.ts', 1], + ['src/main/agent-hooks/server/server-ingest-remote.ts', 1], + ['src/main/agent-hooks/server/server-lifecycle.ts', 1], + ['src/renderer/src/store/slices/agent-status-authority-actions.ts', 1] ] describe('identity observation rebind audit', () => { diff --git a/src/shared/pane-agent-owner.ts b/src/shared/pane-agent-owner.ts index 31b878d67ec..4dea4e6b54f 100644 --- a/src/shared/pane-agent-owner.ts +++ b/src/shared/pane-agent-owner.ts @@ -1,4 +1,5 @@ import type { AgentType } from './agent-status-types' +import { resolveCanonicalPaneAgentEvidence } from './pane-agent-identity-adapter' /** * The owner-evidence signals a terminal pane can carry, strongest launch intent @@ -32,21 +33,6 @@ export type ResolvedPaneAgentOwner = { ownerIsLaunch: boolean } -const PANE_OWNER_RANK: readonly { - key: keyof PaneAgentOwnerSignals - ownerIsLaunch: boolean -}[] = [ - { key: 'launchAgent', ownerIsLaunch: true }, - { key: 'startupLaunchAgent', ownerIsLaunch: true }, - { key: 'initialStatusAgent', ownerIsLaunch: true }, - { key: 'commandInferredAgent', ownerIsLaunch: true }, - { key: 'hookAgent', ownerIsLaunch: false }, - { key: 'siblingHookAgent', ownerIsLaunch: false }, - { key: 'completedHookAgent', ownerIsLaunch: false }, - { key: 'siblingCompletedHookAgent', ownerIsLaunch: false }, - { key: 'sleepingSessionAgent', ownerIsLaunch: false } -] - /** * The single authoritative resolver for "which agent owns this pane", shared by * the tab-icon resolver, the terminal-pane display/renderer owner, and the @@ -66,13 +52,38 @@ const PANE_OWNER_RANK: readonly { export function resolvePaneAgentOwnerRecord( signals: PaneAgentOwnerSignals ): ResolvedPaneAgentOwner | null { - for (const { key, ownerIsLaunch } of PANE_OWNER_RANK) { - const agent = signals[key] + const evidence = [] as { + source: 'launch' | 'completed-hook' | 'sleeping-session' | 'sibling' + agent: AgentType + }[] + const launchAgent = + signals.launchAgent ?? + signals.startupLaunchAgent ?? + signals.initialStatusAgent ?? + signals.commandInferredAgent + if (launchAgent) { + evidence.push({ source: 'launch', agent: launchAgent }) + } + // This compatibility signal has no liveness bit; treat it as the durable completed-hook rung. + if (signals.hookAgent) { + evidence.push({ source: 'completed-hook', agent: signals.hookAgent }) + } + if (signals.completedHookAgent) { + evidence.push({ source: 'completed-hook', agent: signals.completedHookAgent }) + } + if (signals.sleepingSessionAgent) { + evidence.push({ source: 'sleeping-session', agent: signals.sleepingSessionAgent }) + } + for (const agent of [signals.siblingHookAgent, signals.siblingCompletedHookAgent]) { if (agent) { - return { agent, ownerIsLaunch } + evidence.push({ source: 'sibling', agent }) } } - return null + const identity = resolveCanonicalPaneAgentEvidence({ evidence, allowSibling: true }) + if (!identity.agent) { + return null + } + return { agent: identity.agent, ownerIsLaunch: identity.source === 'launch' } } export function resolvePaneAgentOwner(signals: PaneAgentOwnerSignals): AgentType | null { diff --git a/src/shared/published-pane-agent-identity-characterization.test.ts b/src/shared/published-pane-agent-identity-characterization.test.ts deleted file mode 100644 index e232f3f7959..00000000000 --- a/src/shared/published-pane-agent-identity-characterization.test.ts +++ /dev/null @@ -1,175 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { PaneAgentIdentityComparisonRecorder } from './pane-agent-identity-comparison' -import { resolvePublishedPaneAgentIdentity } from './published-pane-agent-identity' -import { comparePublishedPaneAgentIdentity } from './published-pane-agent-identity-comparison' -import type { TuiAgent } from './tui-agent' - -/** - * Output-neutral characterization of the FROZEN host publication adapter. The host wave may only - * replace the frozen call after this table and the real-session comparison window are green, so - * every evidence shape the adapter accepts is pinned here byte-for-byte. These are observations - * of current behavior, not aspirations — do not "fix" a row without the host-wave migration. - */ - -type Shape = { - name: string - args: { - hookAgent?: TuiAgent | null - hookIsLive?: boolean - launchAgent?: TuiAgent | null - foregroundAgent?: TuiAgent | null - title?: string | null - } - expected: TuiAgent | undefined -} - -const GROK_ADVERSARIAL_TITLE = 'STA-4011 Linux Antigravity Commit Messages - grok' - -const SHAPES: readonly Shape[] = [ - { name: 'nothing known', args: {}, expected: undefined }, - { - name: 'all null/empty values', - args: { hookAgent: null, launchAgent: null, foregroundAgent: null, title: '' }, - expected: undefined - }, - { name: 'live hook only', args: { hookAgent: 'claude', hookIsLive: true }, expected: 'claude' }, - { - name: 'completed hook only', - args: { hookAgent: 'codex', hookIsLive: false }, - expected: 'codex' - }, - { name: 'launch only', args: { launchAgent: 'gemini' }, expected: 'gemini' }, - { name: 'foreground only', args: { foregroundAgent: 'codex' }, expected: 'codex' }, - { name: 'unambiguous title only', args: { title: GROK_ADVERSARIAL_TITLE }, expected: 'grok' }, - { - name: 'free-text title mention only', - args: { title: 'Review the Claude session-history fix' }, - expected: undefined - }, - { - name: 'live hook beats foreground', - args: { hookAgent: 'claude', hookIsLive: true, foregroundAgent: 'codex' }, - expected: 'claude' - }, - { - name: 'foreground beats launch (frozen process rung, no proof required)', - args: { launchAgent: 'claude', foregroundAgent: 'codex' }, - expected: 'codex' - }, - { - name: 'launch beats completed hook', - args: { hookAgent: 'codex', hookIsLive: false, launchAgent: 'claude' }, - expected: 'claude' - }, - { - name: 'launch beats unambiguous title', - args: { launchAgent: 'claude', title: GROK_ADVERSARIAL_TITLE }, - expected: 'claude' - }, - { - name: 'completed hook beats unambiguous title', - args: { hookAgent: 'claude', hookIsLive: false, title: GROK_ADVERSARIAL_TITLE }, - expected: 'claude' - }, - { - name: 'duplicate agreeing sources', - args: { - hookAgent: 'claude', - hookIsLive: true, - launchAgent: 'claude', - foregroundAgent: 'claude', - title: 'claude' - }, - expected: 'claude' - }, - { - name: 'full conflict resolves to strongest', - args: { - hookAgent: 'claude', - hookIsLive: true, - launchAgent: 'gemini', - foregroundAgent: 'codex', - title: GROK_ADVERSARIAL_TITLE - }, - expected: 'claude' - } -] - -describe('frozen host adapter characterization', () => { - for (const shape of SHAPES) { - it(`publishes ${shape.expected ?? 'nothing'} for: ${shape.name}`, () => { - expect(resolvePublishedPaneAgentIdentity(shape.args)).toBe(shape.expected) - }) - } -}) - -describe('comparison wrapper output-neutrality', () => { - it('returns the frozen result verbatim for every characterized shape', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - for (const shape of SHAPES) { - const wrapped = comparePublishedPaneAgentIdentity( - { - ...shape.args, - surface: 'terminal-summary', - paneId: `pane:${shape.name}`, - worktreeId: 'wt', - hostScope: 'local' - }, - recorder - ) - expect(wrapped).toBe(resolvePublishedPaneAgentIdentity(shape.args)) - } - expect(recorder.snapshot().comparisons).toBe(SHAPES.length) - }) - - it('counts the frozen-vs-canonical divergence instead of publishing it', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - // The frozen ladder lets a bare foreground NAME outrank the launch record; the canonical - // ladder demands a host process proof for that rung. The published value must stay the - // frozen one while the disagreement is counted — that count is the migration gate. - const published = comparePublishedPaneAgentIdentity( - { - launchAgent: 'claude', - foregroundAgent: 'codex', - surface: 'terminal-summary', - paneId: 'pane:process-demotion', - worktreeId: 'wt', - hostScope: 'local' - }, - recorder - ) - expect(published).toBe('codex') - expect(recorder.snapshot()).toMatchObject({ comparisons: 1, disagreements: 1 }) - }) - - it('records the reclaim/bad-hook input shape separately', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - comparePublishedPaneAgentIdentity( - { - hookAgent: 'claude', - hookIsLive: false, - title: GROK_ADVERSARIAL_TITLE, - surface: 'terminal-summary', - paneId: 'pane:reclaim-shape', - worktreeId: 'wt', - hostScope: 'local' - }, - recorder - ) - expect(recorder.snapshot()).toMatchObject({ comparisons: 1, reclaimShapes: 1 }) - }) - - it('skips recomputation for consecutive identical pane inputs but still returns the frozen value', () => { - const recorder = new PaneAgentIdentityComparisonRecorder() - const args = { - launchAgent: 'claude' as const, - surface: 'terminal-summary' as const, - paneId: 'pane:dedupe', - worktreeId: 'wt', - hostScope: 'local' as const - } - expect(comparePublishedPaneAgentIdentity(args, recorder)).toBe('claude') - expect(comparePublishedPaneAgentIdentity(args, recorder)).toBe('claude') - expect(recorder.snapshot().comparisons).toBe(1) - }) -}) diff --git a/src/shared/published-pane-agent-identity-comparison.ts b/src/shared/published-pane-agent-identity-comparison.ts deleted file mode 100644 index c5cd0a3f990..00000000000 --- a/src/shared/published-pane-agent-identity-comparison.ts +++ /dev/null @@ -1,100 +0,0 @@ -import { collectAgentTitleEvidence } from './agent-title-evidence' -import { resolveCanonicalPaneAgentIdentity } from './pane-agent-identity-adapter' -import { - PaneAgentIdentityComparisonRecorder, - type PaneIdentityComparisonSurface, - type PaneIdentityHostScope -} from './pane-agent-identity-comparison' -import { resolvePublishedPaneAgentIdentity } from './published-pane-agent-identity' -import type { TuiAgent } from './tui-agent' - -/** - * Output-neutral wrapper for the host publication path. The FROZEN adapter still decides what is - * published — this function returns its result verbatim — while the canonical adapter's answer is - * computed beside it and only disagreements are counted. `RuntimeTerminalSummary.agentIdentity` - * must not change while the comparison window runs. - */ - -export type PublishedPaneAgentIdentityComparisonArgs = { - hookAgent?: TuiAgent | null - hookIsLive?: boolean - launchAgent?: TuiAgent | null - foregroundAgent?: TuiAgent | null - title?: string | null - surface: PaneIdentityComparisonSurface - paneId: string - worktreeId?: string | null - hostScope: PaneIdentityHostScope -} - -const defaultRecorder = new PaneAgentIdentityComparisonRecorder((line, sample) => { - console.info(`[pane-identity-compare] ${line}`, sample ?? {}) -}) - -export function getPublishedPaneIdentityComparisonRecorder(): PaneAgentIdentityComparisonRecorder { - return defaultRecorder -} - -export function comparePublishedPaneAgentIdentity( - args: PublishedPaneAgentIdentityComparisonArgs, - recorder: PaneAgentIdentityComparisonRecorder = defaultRecorder -): TuiAgent | undefined { - const published = resolvePublishedPaneAgentIdentity(args) - try { - recordCanonicalDivergence(args, published ?? null, recorder) - } catch { - // Telemetry must never take down terminal.list; a lost sample is recoverable. - } - return published -} - -function recordCanonicalDivergence( - args: PublishedPaneAgentIdentityComparisonArgs, - published: TuiAgent | null, - recorder: PaneAgentIdentityComparisonRecorder -): void { - const signature = [ - args.hookAgent ?? '-', - args.hookIsLive === true ? 'live' : 'idle', - args.launchAgent ?? '-', - args.foregroundAgent ?? '-', - args.title ?? '-' - ].join('|') - if (!recorder.shouldCompare(args.surface, args.paneId, signature)) { - return - } - // No host process PROOF exists yet, so the canonical lane sees the foreground name as a weak - // hint only. Where that alone flips the answer is precisely what this window measures. - const titleAgent = args.title ? collectAgentTitleEvidence(args.title).agent : null - const canonical = resolveCanonicalPaneAgentIdentity({ - hookAgent: args.hookAgent, - hookIsLive: args.hookIsLive, - launchAgent: args.launchAgent, - foregroundAgent: args.foregroundAgent, - title: args.title, - uncoveredFallback: { - agent: published, - // A bare foreground name is an uncovered compatibility answer, but it is not title proof. - // Mark title-only only when no foreground signal exists and the frozen answer matches title - // evidence exactly. - titleOnly: args.foregroundAgent == null && published !== null && published === titleAgent - } - }) - recorder.record({ - surface: args.surface, - paneId: args.paneId, - worktreeId: args.worktreeId, - oldAgent: published, - newAgent: canonical.agent, - newSource: canonical.source, - coverage: canonical.coverage, - titleOnly: canonical.titleOnly, - // Run keys are not plumbed into the publication path yet (host wave); absent, not stale. - runKeyComparability: 'absent', - hostScope: args.hostScope, - ambiguous: canonical.ambiguousAt !== undefined, - reclaimShape: Boolean( - args.hookAgent && args.hookIsLive !== true && titleAgent && titleAgent !== args.hookAgent - ) - }) -} diff --git a/src/shared/terminal-title-agent-type.ts b/src/shared/terminal-title-agent-type.ts index cdbce788806..5e0e0d3f624 100644 --- a/src/shared/terminal-title-agent-type.ts +++ b/src/shared/terminal-title-agent-type.ts @@ -11,6 +11,7 @@ import { isLegacyPiCompatibleTitle } from './pi-compatible-synthetic-title' import { memoizeTitleClassification } from './terminal-title-classification-memo' +import { resolveCanonicalPaneAgentIdentity } from './pane-agent-identity-adapter' import type { TuiAgent } from './tui-agent' export const CLAUDE_IDLE = '\u2733' // ✳ (eight-spoked asterisk — Claude Code idle prefix) @@ -85,7 +86,7 @@ export function isPiAgentTitle(title: string): boolean { * Used to scope prompt-cache-timer behavior to Claude sessions only — other * agents have different (or no) caching semantics. */ -function computeIsClaudeAgent(title: string): boolean { +export function isClaudeAgent(title: string): boolean { if (!title || isClaudeManagementTitle(title) || isOpenCodeNativeTitle(title)) { return false } @@ -122,15 +123,11 @@ function computeIsClaudeAgent(title: string): boolean { return false } -/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ -export const isClaudeAgent: (title: string) => boolean = - memoizeTitleClassification(computeIsClaudeAgent) - export function isClaudeManagementTitle(title: string): boolean { return CLAUDE_MANAGEMENT_TITLE_RE.test(title) } -function computeAgentLabel(title: string): string | null { +export function getAgentLabel(title: string): string | null { if (isClaudeManagementTitle(title)) { return null } @@ -217,10 +214,6 @@ function computeAgentLabel(title: string): string | null { return null } -// Maps getAgentLabel()'s product labels to TuiAgent ids — the fallback for -// agents whose foreground PROCESS name isn't self-identifying (Claude Code runs -// as `node`, but its "✳ Claude Code" title resolves here). Agents whose process -// name already matches (codex, etc.) never reach this path. const TITLE_LABEL_TO_AGENT: Partial> = { 'Claude Code': 'claude', OpenClaude: 'openclaude', @@ -240,10 +233,6 @@ const TITLE_LABEL_TO_AGENT: Partial> = { OMP: 'omp' } -/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ -export const getAgentLabel: (title: string) => string | null = - memoizeTitleClassification(computeAgentLabel) - function hasGenericClaudeStatusPrefix(title: string): boolean { return ( containsAgentSpinnerGlyph(title) || @@ -266,7 +255,11 @@ function isGenericClaudeStatusClaim(title: string, titleAgent: TuiAgent | null): export function resolveTerminalTitleAgentType(title: string): TuiAgent | null { const label = getAgentLabel(title) - return label ? (TITLE_LABEL_TO_AGENT[label] ?? null) : null + const parsed = label ? (TITLE_LABEL_TO_AGENT[label] ?? null) : null + return resolveCanonicalPaneAgentIdentity({ + title, + uncoveredFallback: { agent: parsed, titleOnly: parsed !== null } + }).agent } /** @@ -275,14 +268,10 @@ export function resolveTerminalTitleAgentType(title: string): TuiAgent | null { * that something is running, not proof the agent is Claude — so a task or * worktree title cannot become Claude without an explicit "Claude Code" name. */ -function computeExplicitTerminalTitleAgentType(title: string): TuiAgent | null { +export function resolveExplicitTerminalTitleAgentType(title: string): TuiAgent | null { const titleAgent = resolveTerminalTitleAgentType(title) if (isGenericClaudeStatusClaim(title, titleAgent)) { return null } return titleAgent } - -/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ -export const resolveExplicitTerminalTitleAgentType: (title: string) => TuiAgent | null = - memoizeTitleClassification(computeExplicitTerminalTitleAgentType)