Files
orca/src/shared/muse-session-log.test.ts
T
NeilandAdrien De oliveira ebed0964a2 feat(agents): add first-class Muse Code harness (#22216)
* feat(agents): add first-class Muse Code harness

Add Muse as a supervised Orca agent across desktop, mobile, session history, source control, local hooks, SSH, WSL, and native Windows. Preserve user settings, support Muse 1.3 hook environment allowlists, and recognize versioned foreground processes. Include question, waiting, completion, resume, and readiness coverage.

Co-authored-by: homesh-dev <300847526+homesh-dev@users.noreply.github.com>

Co-authored-by: jeffhuen <32542276+jeffhuen@users.noreply.github.com>

Co-authored-by: John Cusack <johncusackccm@gmail.com>

Co-authored-by: Adrien De oliveira <75085839+adriendeoliveira@users.noreply.github.com>

* test(agents): cover Muse remote hook registration

* test(agents): cover Muse hook and source-control contracts

* test(agents): exclude Muse hook metadata from script mode check

* test(agents): keep Muse skill picker coverage stable

* test(ai-vault): include Muse in every-agent fixture

* test(mobile): repin Muse agent icon closure

* fix(muse): detect questions and approvals from structured Muse signals

Muse 1.3 fires no hook for request_user_input, so a pending question left
the pane "working". Its internal reminder subagents also post hooks with
their own session ids (even after Stop), which surfaced "tool failed" rows
and flipped finished panes back to working.

- Read pending questions from Muse's session log
  (user_input_prompt_requested/settled) via the existing transcript poll,
  now generalized from Codex subagents to Muse on main and relay.
- Drop child-session hooks (SubagentStart ids, or turn_id === session_id).
- Treat Notification permission_prompt as the approval wait; PermissionRequest
  also fires for auto-approved calls, so it only caches the approval card.
- Ignore Notification copy as the prompt; poll replays are not new prompts
  or turn boundaries.
- Allowlist USERPROFILE so Windows cmd AutoRun doesn't fail every hook.

* perf(muse): parse only question events from the session log

Most Muse session-log lines are large model/tool records. Filter raw lines
by the user_input_prompt_ marker before JSON.parse via an optional
readJsonlCursor line filter.

* fix(muse): unwrap batched log records and scope questions to the live turn

Review follow-ups: question events inside retained_frame batches were
skipped, and a question left open by a crash or interrupt stayed pending
for the pane's life. Share the history scanner's retained_frame unwrapper,
and only report a pending question whose run_id matches the hook turn_id.

* refactor(muse): drop type assertion in retained_frame unwrap

* fix(agent-hooks): satisfy exhaustive-switch lint in transcript poll policy

---------

Co-authored-by: Adrien De oliveira <75085839+adriendeoliveira@users.noreply.github.com>
2026-09-22 19:13:11 -07:00

141 lines
5.2 KiB
TypeScript

import { afterEach, beforeEach, describe, expect, it } from 'vitest'
import { appendFileSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { join } from 'node:path'
import {
createMuseSessionLogState,
findMuseSessionLogPath,
readMusePendingUserInput
} from './muse-session-log'
const SESSION_ID = '01a0caa3-0e77-7d41-bad7-46283a45633d'
const PROMPT_ID = '01a0caa3-a25a-7810-8229-4de04b2e7ca3'
const QUESTION = {
id: 'fav_color',
header: 'Color',
question: 'What is your favorite color?',
options: [{ label: 'Blue' }, { label: 'Green' }, { label: 'Red' }]
}
function logLine(event: Record<string, unknown>): string {
return `${JSON.stringify({
schema_version: 1,
stream: { kind: 'session', id: SESSION_ID },
record_type: 'event',
payload_type: 'runtime.session',
payload: { kind: 'run', run_id: 'fdada6f1-d403-41d8-b610-52f1d8489334', event }
})}\n`
}
const requested = (promptId = PROMPT_ID): string =>
logLine({
kind: 'user_input_prompt_requested',
prompt_id: promptId,
tool_name: 'request_user_input',
questions: [QUESTION]
})
const settled = (promptId = PROMPT_ID): string =>
logLine({ kind: 'user_input_prompt_settled', prompt_id: promptId, outcome: 'answered' })
function localShard(sessionId: string): string[] {
const hex = sessionId.replace(/-/g, '').slice(0, 12)
const date = new Date(Number.parseInt(hex, 16))
return [
String(date.getFullYear()),
String(date.getMonth() + 1).padStart(2, '0'),
String(date.getDate()).padStart(2, '0')
]
}
describe('muse session log', () => {
let sessionsDir: string
let logPath: string
beforeEach(() => {
sessionsDir = mkdtempSync(join(tmpdir(), 'muse-session-log-'))
const dir = join(sessionsDir, ...localShard(SESSION_ID), SESSION_ID)
mkdirSync(dir, { recursive: true })
logPath = join(dir, 'session.jsonl')
})
afterEach(() => {
rmSync(sessionsDir, { recursive: true, force: true })
})
it('finds the log in the date shard named by the UUIDv7 timestamp', () => {
writeFileSync(logPath, '')
expect(findMuseSessionLogPath(SESSION_ID, sessionsDir)).toBe(logPath)
})
it('returns undefined for a missing log or a non-v7 session id', () => {
expect(findMuseSessionLogPath(SESSION_ID, sessionsDir)).toBeUndefined()
expect(
findMuseSessionLogPath('1471f5e6-bd60-41a2-bfcf-c9086dbd4092', sessionsDir)
).toBeUndefined()
})
it('reads prompts batched inside a retained_frame', () => {
const frame = (line: string): string =>
`${JSON.stringify({ record_type: 'retained_frame', children: [{ record_json: line.trim() }] })}\n`
writeFileSync(logPath, frame(requested()))
const log = createMuseSessionLogState(SESSION_ID)
expect(readMusePendingUserInput(log, undefined, sessionsDir)?.promptId).toBe(PROMPT_ID)
appendFileSync(logPath, frame(settled()))
expect(readMusePendingUserInput(log, undefined, sessionsDir)).toBeUndefined()
})
it('ignores a prompt left open by an earlier run', () => {
writeFileSync(logPath, requested())
const log = createMuseSessionLogState(SESSION_ID)
expect(
readMusePendingUserInput(log, 'fdada6f1-d403-41d8-b610-52f1d8489334', sessionsDir)?.promptId
).toBe(PROMPT_ID)
expect(
readMusePendingUserInput(log, '11111111-2222-4333-8444-555555555555', sessionsDir)
).toBeUndefined()
})
it('returns the open prompt and clears it once settled', () => {
writeFileSync(logPath, requested())
const log = createMuseSessionLogState(SESSION_ID)
expect(readMusePendingUserInput(log, undefined, sessionsDir)).toEqual({
promptId: PROMPT_ID,
runId: 'fdada6f1-d403-41d8-b610-52f1d8489334',
questions: [QUESTION]
})
// Re-reading with no new bytes keeps the prompt pending.
expect(readMusePendingUserInput(log, undefined, sessionsDir)?.promptId).toBe(PROMPT_ID)
appendFileSync(logPath, settled())
expect(readMusePendingUserInput(log, undefined, sessionsDir)).toBeUndefined()
})
it('holds a partial trailing line until it is completed', () => {
const line = requested()
writeFileSync(logPath, line.slice(0, 40))
const log = createMuseSessionLogState(SESSION_ID)
expect(readMusePendingUserInput(log, undefined, sessionsDir)).toBeUndefined()
appendFileSync(logPath, line.slice(40))
expect(readMusePendingUserInput(log, undefined, sessionsDir)?.questions).toEqual([QUESTION])
})
it('reports the newest of several open prompts', () => {
const second = '01a0caa4-0000-7000-8000-000000000001'
writeFileSync(logPath, `${requested()}${requested(second)}`)
const log = createMuseSessionLogState(SESSION_ID)
expect(readMusePendingUserInput(log, undefined, sessionsDir)?.promptId).toBe(second)
appendFileSync(logPath, settled(second))
expect(readMusePendingUserInput(log, undefined, sessionsDir)?.promptId).toBe(PROMPT_ID)
})
it('locates a log created after the first read', () => {
const log = createMuseSessionLogState(SESSION_ID)
expect(readMusePendingUserInput(log, undefined, sessionsDir)).toBeUndefined()
writeFileSync(logPath, requested())
expect(readMusePendingUserInput(log, undefined, sessionsDir)?.promptId).toBe(PROMPT_ID)
})
})