mirror of
https://github.com/stablyai/orca.git
synced 2026-09-28 08:02:43 +00:00
fix(agent-status): clear answered Claude question waits at answer time (#9074)
* fix(agent-status): clear answered Claude question waits at answer time An answered AskUserQuestion left the amber "waiting" indicator on sidebar rows and tabs until the agent's next tool hook or turn end — unbounded linger while the model thinks or streams after the answer (measured 17s for a 1000-word reply, 44s for 3000 words). Root cause is an event-shape change: newer Claude reports the AskUserQuestion wait as PermissionRequest (not the PreToolUse shape #7852 special-cased), so the wait inherited real-permission stickiness and shouldKeepClaudePermissionVisible swallowed the answer-time PostToolUse(AskUserQuestion) working event — the identity match can never succeed because the question's PermissionRequest carries no inheritable tool_use_id. That silently undid #8311 for questions. Two scoped changes, both keyed on the tool name rather than the hook event name: - Sticky permission hold now exempts AskUserQuestion waits, so the real answer-time hook (when Claude sends one) clears the wait as #8311 intended. - New guarded inference for the hook Claude may never send: the submit keystroke (Enter or digit quick-select) into a pane whose fresh status is a waiting AskUserQuestion synthesizes the post-answer state, exactly mirroring the existing interrupt inference (renderer baseline capture, main-process re-validation, listener lead-state sync so child-driven refreshes cannot resurrect the dismissed question). Real permission waits (other tools) keep their sticky semantics; batched input and pastes never match the submit classifier. Verified live against a real claude CLI: waiting -> working within ~50ms of both Enter and digit answers, question card dropped, unanswered questions still hold amber, permission stickiness covered by tests. * fix(agent-status): guard question answer inference Keep multi-question, multi-select, and free-text selector interactions waiting until the full prompt is submitted. Wire native-chat answers into the same guarded inference only after every paced runtime write succeeds, with cancellation and delivery-failure coverage. * chore(skills): refresh manifest for rc.2 * fix(agent-status): verify native chat answer delivery * fix(agent-status): await verified question delivery * fix(agent-status): pin native-chat answer baseline before delivery The native-chat question-answered inference read the live pane status at settle time (after the paced send + remote acceptance, which can span seconds on SSH). If a replacement AskUserQuestion became current in that window, the settle callback minted a fresh baseline from the new question and the server cleared *its* wait — dismissing a question the user never answered. Capture the answered question's baseline before delivery and have the inference getter return it, so the server re-validates against the pinned baseline and rejects a changed status — the same capture-then-revalidate contract the terminal keystroke path already uses. Also hoist the shouldStepNativeChatAskAnswer predicate to a single evaluation. Regression test swaps the live status between sendAnswer and settle and asserts the answered question's baseline is used (fails against the prior live-read getter).
This commit is contained in:
@@ -6,6 +6,7 @@ import { mkdirSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync }
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import {
|
||||
clearClaudeAnsweredQuestionWait,
|
||||
clearPaneCacheState,
|
||||
createHookListenerState,
|
||||
getEndpointFileName,
|
||||
@@ -2781,4 +2782,60 @@ describe('shared agent-hook-listener', () => {
|
||||
expect(ok).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('clearClaudeAnsweredQuestionWait', () => {
|
||||
const claudeEvent = (
|
||||
payload: Record<string, unknown>
|
||||
): ReturnType<typeof normalizeHookPayload> =>
|
||||
normalizeHookPayload(state, 'claude', { paneKey: PANE_KEY, payload }, 'production')
|
||||
|
||||
it('restores working for an answered lead question and drops the card', () => {
|
||||
claudeEvent({ hook_event_name: 'UserPromptSubmit', prompt: 'pick a color' })
|
||||
const wait = claudeEvent({
|
||||
hook_event_name: 'PreToolUse',
|
||||
tool_name: 'AskUserQuestion',
|
||||
tool_input: { questions: [{ question: 'Red or Blue?' }] }
|
||||
})
|
||||
expect(wait?.payload.state).toBe('waiting')
|
||||
expect(wait?.payload.interactivePrompt).toBeDefined()
|
||||
|
||||
expect(clearClaudeAnsweredQuestionWait(state, PANE_KEY)).toEqual({ state: 'working' })
|
||||
|
||||
// Why: a child-driven refresh re-emits the cached lead state; the linger
|
||||
// bug would come back if it could resurrect the dismissed question.
|
||||
const childDriven = claudeEvent({
|
||||
hook_event_name: 'SubagentStart',
|
||||
agent_id: 'a1',
|
||||
agent_type: 'probe'
|
||||
})
|
||||
expect(childDriven?.payload.state).toBe('working')
|
||||
expect(childDriven?.payload.toolName).toBeUndefined()
|
||||
expect(childDriven?.payload.interactivePrompt).toBeUndefined()
|
||||
})
|
||||
|
||||
it('restores the stashed lead state for an answered child question', () => {
|
||||
claudeEvent({ hook_event_name: 'UserPromptSubmit', prompt: 'go' })
|
||||
claudeEvent({ hook_event_name: 'SubagentStart', agent_id: 'a1', agent_type: 'probe' })
|
||||
claudeEvent({ hook_event_name: 'Stop' })
|
||||
const wait = claudeEvent({
|
||||
hook_event_name: 'PreToolUse',
|
||||
tool_name: 'AskUserQuestion',
|
||||
agent_id: 'a1',
|
||||
tool_input: { questions: [{ question: 'Continue?' }] }
|
||||
})
|
||||
expect(wait?.payload.state).toBe('waiting')
|
||||
|
||||
// Why: the lead already finished; the answer resumes the child, so the
|
||||
// emitted state is gated up to working only while that child still runs.
|
||||
expect(clearClaudeAnsweredQuestionWait(state, PANE_KEY)).toEqual({ state: 'working' })
|
||||
expect(state.claudeLeadStateByPaneKey.get(PANE_KEY)).toEqual({ state: 'done' })
|
||||
|
||||
const drained = claudeEvent({ hook_event_name: 'SubagentStop', agent_id: 'a1' })
|
||||
expect(drained?.payload.state).toBe('done')
|
||||
})
|
||||
|
||||
it('falls back to working when no lead record exists', () => {
|
||||
expect(clearClaudeAnsweredQuestionWait(state, PANE_KEY)).toEqual({ state: 'working' })
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
@@ -34,6 +34,7 @@ import {
|
||||
type AgentSubagentSnapshot,
|
||||
type ParsedAgentStatusPayload
|
||||
} from './agent-status-types'
|
||||
import { isAskUserQuestionTool } from './agent-question-answered-intent'
|
||||
import {
|
||||
claudeRosterHasWorkingSubagent,
|
||||
claudeRosterToSnapshots,
|
||||
@@ -744,14 +745,6 @@ function clearActiveToolFieldsUpdate(): ToolSnapshot {
|
||||
)
|
||||
}
|
||||
|
||||
/** True for the AskUserQuestion tool across the casing variants different
|
||||
* agents emit (`AskUserQuestion` / `ask_user_question` / `askUserQuestion`).
|
||||
* Why: this is the structured "pick an option" prompt whose full input the
|
||||
* clients render as a live card. */
|
||||
function isAskUserQuestionTool(toolName: string | undefined): boolean {
|
||||
return toolName?.replaceAll(/[^a-z0-9]/gi, '').toLowerCase() === 'askuserquestion'
|
||||
}
|
||||
|
||||
/** Capture the full AskUserQuestion tool input as a JSON string when the tool
|
||||
* is an AskUserQuestion variant; otherwise undefined so resolveToolState
|
||||
* clears any prior prompt. Kept agent-generic: callers pass whatever raw
|
||||
@@ -2525,6 +2518,36 @@ function clearClaudePendingWaitForAgent(
|
||||
state.claudeLeadStateByPaneKey.set(paneKey, lead.stateBeforeWait ?? { state: 'working' })
|
||||
}
|
||||
|
||||
/** Clear an AskUserQuestion wait after the user's answer was typed into the
|
||||
* terminal. Answering emits no hook event, so the caller infers it from the
|
||||
* submit keystroke. Restores the stashed pre-wait lead state (child-induced
|
||||
* question) or falls back to 'working' (lead question), and drops the cached
|
||||
* question card so later child-driven refreshes cannot re-emit the stale
|
||||
* wait. Returns the pane state to emit, gated up to 'working' while children
|
||||
* still run. */
|
||||
export function clearClaudeAnsweredQuestionWait(
|
||||
state: HookListenerState,
|
||||
paneKey: string
|
||||
): Pick<ClaudeLeadTurnState, 'state' | 'interrupted'> {
|
||||
const lead = state.claudeLeadStateByPaneKey.get(paneKey)
|
||||
const restored =
|
||||
lead?.state === 'waiting'
|
||||
? (lead.stateBeforeWait ?? { state: 'working' as const })
|
||||
: { state: 'working' as const }
|
||||
state.claudeLeadStateByPaneKey.set(paneKey, { ...restored })
|
||||
const previousTool = state.lastToolByPaneKey.get(paneKey)
|
||||
state.lastToolByPaneKey.set(
|
||||
paneKey,
|
||||
previousTool?.lastAssistantMessage
|
||||
? { lastAssistantMessage: previousTool.lastAssistantMessage }
|
||||
: {}
|
||||
)
|
||||
const roster = state.claudeSubagentRosterByPaneKey.get(paneKey)
|
||||
return restored.state === 'done' && claudeRosterHasWorkingSubagent(roster)
|
||||
? { state: 'working' }
|
||||
: restored
|
||||
}
|
||||
|
||||
/** Emit a pane status refresh driven by child activity (lifecycle events and
|
||||
* child-origin tool events): the lead's cached state is re-emitted — gated up
|
||||
* to 'working' while a child works — without touching the lead's tool/prompt
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
import type { AgentType } from './agent-status-types'
|
||||
|
||||
/** Baseline snapshot the renderer captured when it observed the submit
|
||||
* keystroke. The main process re-validates every field against its own
|
||||
* cached status so a racing real hook always wins over the inference. */
|
||||
export type AgentQuestionAnsweredInferenceRequest = {
|
||||
paneKey: string
|
||||
baselineUpdatedAt: number
|
||||
baselineStateStartedAt: number
|
||||
baselinePrompt: string
|
||||
baselineAgentType: AgentType | undefined
|
||||
}
|
||||
|
||||
/** True for the AskUserQuestion tool across the casing variants different
|
||||
* agents emit (`AskUserQuestion` / `ask_user_question` / `askUserQuestion`).
|
||||
* Why: this is the structured "pick an option" prompt whose full input the
|
||||
* clients render as a live card. */
|
||||
export function isAskUserQuestionTool(toolName: string | undefined): boolean {
|
||||
return toolName?.replaceAll(/[^a-z0-9]/gi, '').toLowerCase() === 'askuserquestion'
|
||||
}
|
||||
|
||||
const QUESTION_ANSWER_ENTER_INPUTS: ReadonlySet<string> = new Set([
|
||||
'\r',
|
||||
'\n',
|
||||
'\r\n',
|
||||
'\x1b[13u',
|
||||
'\x1b[13;1u'
|
||||
])
|
||||
const QUESTION_ANSWER_DIGIT_INPUTS: ReadonlySet<string> = new Set('123456789')
|
||||
|
||||
export function isPotentialQuestionAnsweredSubmitInput(data: string): boolean {
|
||||
return QUESTION_ANSWER_ENTER_INPUTS.has(data) || QUESTION_ANSWER_DIGIT_INPUTS.has(data)
|
||||
}
|
||||
|
||||
function readSingleSelectOptionCount(interactivePrompt: string | undefined): number | null {
|
||||
if (!interactivePrompt) {
|
||||
return null
|
||||
}
|
||||
try {
|
||||
const parsed = JSON.parse(interactivePrompt) as { questions?: unknown }
|
||||
if (!Array.isArray(parsed.questions) || parsed.questions.length !== 1) {
|
||||
return -1
|
||||
}
|
||||
const [question] = parsed.questions as { multiSelect?: unknown; options?: unknown }[]
|
||||
if (!question || question.multiSelect === true || !Array.isArray(question.options)) {
|
||||
return -1
|
||||
}
|
||||
return question.options.length
|
||||
} catch {
|
||||
// Why: malformed JSON can be a length-capped multi-question payload. It is
|
||||
// not equivalent to an older hook omitting tool input, so fail closed.
|
||||
return -1
|
||||
}
|
||||
}
|
||||
|
||||
/** True only when one keystroke is enough to finish the whole prompt.
|
||||
* Why: digits merely advance multi-question prompts and toggle multi-selects;
|
||||
* the synthetic "Type something" row also opens an editor instead of
|
||||
* submitting. Without the prompt-shape gate those partial choices clear the
|
||||
* waiting indicator while Claude is still blocked on more input. */
|
||||
export function isQuestionAnsweredSubmitInput(
|
||||
data: string,
|
||||
interactivePrompt: string | undefined
|
||||
): boolean {
|
||||
if (!isPotentialQuestionAnsweredSubmitInput(data)) {
|
||||
return false
|
||||
}
|
||||
const optionCount = readSingleSelectOptionCount(interactivePrompt)
|
||||
if (optionCount === -1) {
|
||||
return false
|
||||
}
|
||||
if (QUESTION_ANSWER_ENTER_INPUTS.has(data)) {
|
||||
// Older hook payloads can omit tool input; Enter remains the conservative
|
||||
// fallback because it is the ordinary submit path for a question.
|
||||
return true
|
||||
}
|
||||
if (optionCount === null) {
|
||||
return false
|
||||
}
|
||||
// Claude adds a final "Type something" row after the declared options.
|
||||
// Only declared option numbers complete a single-select immediately.
|
||||
return Number(data) <= optionCount
|
||||
}
|
||||
Reference in New Issue
Block a user