mirror of
https://github.com/stablyai/orca.git
synced 2026-10-02 16:02:15 +00:00
* refactor(native-chat): give the structured chat host one required logger The structured chat runtime took an optional onError callback that the desktop never passed, so a late dispatch settlement, an unanswered-dispatch release, a journal event-sink write and a provider lifecycle delivery that failed were dropped with no trace. Other host failures went to scattered console.warn calls, which reach nothing in a packaged desktop build. The runtime and host now take one required logger (warn/error with a scope and fields). The production logger writes each entry as a failed span to <userData>/logs/main.trace.ndjson, which the diagnostic bundle collects, and to the console (stderr under a supervised headless host). The runtime and the host wrap it so a logger that throws never fails what it reports, and the install refuses without one. Sites that deliberately kept a recovery-capsule error out of the log still log no error object. * refactor(native-chat): hand the chat host's collaborators the logger, and give orcad its trace file The delivery loop, idle sweep, queued-message drain, lease renewer, event sink, conversation map and provider start/exit settlement each took an internal error callback that the host mapped onto the logger. They now take the logger itself and log under their own scope. The event sink keeps one onFailed hook, which decides whether to stop the provider, not whether to report. The dead-generation settlement returns its failure so each caller logs it under its own scope. orcad now installs the desktop's local trace sink under its own data root, so a headless host's chat failures reach <data-root>/logs/main.trace.ndjson as well as stderr. Also passes the logger in the test fixtures the first commit missed, which tc:node caught. * fix(native-chat): keep repeated chat failures from flooding the trace file, and record their causes - The production structured-chat logger writes a repeated failure (same level, scope, session, message and error text) once per 5 minutes, carrying how many repeats it swallowed; the tracked set is capped at 256. - Trace entries now carry the error's code (and SQLite errcode) and up to three causes by name and message. - A chat read whose conversation will not open is logged through the host's logger (open-for-read), and so are the runtime's chat-tab bookkeeping failures that already hold the host. - orcad writes its own orcad.trace.ndjson, closes it after every quit handler, and flushes it on process exit; a trace file that cannot be opened leaves tracing off instead of stopping the app or orcad. - Tests: the desktop wiring test proves the logger reaches the trace sink, and the privacy tests read every level the logger received. * fix(native-chat): log a created chat's tab-publication and launch-prompt failures through the host's logger * fix(native-chat): key a repeated chat failure on everything its entry writes The repeat suppression keyed on the message and the error's text, so two refusals with the same code but different causes, a plain error and a refusal of one code, or two object-valued errors shared a key and the second was swallowed for five minutes. The key is now the entry's whole written content (fields, code, errcode, refusal reason, cause chain, a stable rendering of a non-error value) plus the error's name and message; a refusal's reason is also written. * test(native-chat): pin that an error's name keeps two repeated failures apart * test(native-chat): build the refusal in the repeat-key test as the wire does * fix(native-chat): read an error's code and a refusal's reason by narrowing, not Reflect.get
265 lines
9.2 KiB
TypeScript
265 lines
9.2 KiB
TypeScript
// A Codex ask with several questions, driven from the real host journal through
|
|
// the client's session reducer to the rows the transcript list draws.
|
|
import { mkdtemp, rm } from 'node:fs/promises'
|
|
import { tmpdir } from 'node:os'
|
|
import { join } from 'node:path'
|
|
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
|
import { computeAgentSessionPayloadFingerprint } from '../../shared/agent-session-mutation-envelope'
|
|
import {
|
|
AGENT_JOURNAL_THREAD_SCOPE,
|
|
type AgentJournalRenderItem
|
|
} from '../../shared/agent-session-journal-types'
|
|
import {
|
|
EMPTY_STRUCTURED_AGENT_SESSION,
|
|
reduceStructuredAgentSession,
|
|
type StructuredAgentSessionState
|
|
} from '../../shared/structured-agent-session-reducer'
|
|
import { projectStructuredAgentSessionMessages } from '../../shared/structured-agent-session-message-projection'
|
|
import { projectNativeChatTranscriptMessages } from '../../shared/native-chat-transcript-projection'
|
|
import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store'
|
|
import { openTestAgentSessionRecordStore } from '../runtime/agent-session-record-store-test-harness'
|
|
import { CodexJournalPrompts } from './codex-structured-journal-prompts'
|
|
import { CODEX_USER_INPUT_METHOD } from './codex-structured-prompt-replies'
|
|
import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter'
|
|
import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink'
|
|
import { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host'
|
|
import {
|
|
HOST_TEST_NOW,
|
|
HOST_TEST_SESSION as SESSION,
|
|
HOST_TEST_THREAD as THREAD,
|
|
hostTestAttachParams,
|
|
hostTestOperationId,
|
|
resetHostTestOperationIds
|
|
} from '../native-chat/agent-session-wire/structured-agent-session-host-test-data'
|
|
import {
|
|
projectStructuredQuestionMessages,
|
|
structuredQuestionTranscript
|
|
} from '../../renderer/src/components/native-chat/structured-agent-question-projection'
|
|
import { openTestJournalHostDatabase } from '../native-chat/agent-session-journal/journal-host-database-test-support'
|
|
import { createStructuredAgentSessionLogger } from '../native-chat/agent-session-wire/structured-agent-session-logger'
|
|
|
|
const CALLER = { callerKey: 'client-1' }
|
|
|
|
type Asked = readonly { id: string; question: string }[]
|
|
|
|
// Codex's question ids are the model's own words, so their text order is not
|
|
// the order it asked in. These two asks spell the two orders live QA saw.
|
|
const ASKED: Asked = [
|
|
{ id: 'scope', question: 'Which files are in scope?' },
|
|
{ id: 'priority', question: 'What matters most?' },
|
|
{ id: 'deadline', question: 'When is it due?' }
|
|
]
|
|
const ASKED_OUT_OF_ORDER: Asked = [
|
|
{ id: 'format', question: 'Which format?' },
|
|
{ id: 'audience', question: 'Who reads it?' },
|
|
{ id: 'length', question: 'How long?' }
|
|
]
|
|
|
|
let root: string
|
|
let store: AgentSessionRecordStore
|
|
let host: StructuredAgentSessionHost
|
|
let sink: StructuredAgentSessionEventSink | null
|
|
let client: StructuredAgentSessionState
|
|
let clock: number
|
|
|
|
beforeEach(async () => {
|
|
root = await mkdtemp(join(tmpdir(), 'orca-codex-question-order-'))
|
|
resetHostTestOperationIds()
|
|
sink = null
|
|
clock = HOST_TEST_NOW
|
|
const adapter: StructuredAgentSessionAdapter = {
|
|
acquire: vi.fn<StructuredAgentSessionAdapter['acquire']>(async ({ fence, events }) => {
|
|
sink = events ?? null
|
|
return {
|
|
process: {
|
|
hostId: 'local',
|
|
pid: 4242,
|
|
processStartTimeMs: 1_700_000_000_000,
|
|
spawnToken: store.getRecord(SESSION)?.lease.reservedSpawnToken ?? 'spawn-a'
|
|
},
|
|
link: {
|
|
linkId: `link-${fence}`,
|
|
handle: { provider: 'codex', threadId: THREAD },
|
|
origin: 'created',
|
|
mintedAtFence: fence,
|
|
observedAt: HOST_TEST_NOW
|
|
}
|
|
}
|
|
}),
|
|
releaseAcquisition: vi.fn(async () => true),
|
|
dispatch: vi.fn(),
|
|
cancelTurn: vi.fn(async () => ({ cancelled: true })),
|
|
answerPrompt: vi.fn(async ({ commit }) => commit()),
|
|
setOption: vi.fn(async () => undefined)
|
|
}
|
|
store = await openTestAgentSessionRecordStore(root)
|
|
host = new StructuredAgentSessionHost({
|
|
logger: createStructuredAgentSessionLogger(),
|
|
store,
|
|
adapter,
|
|
journalDatabase: openTestJournalHostDatabase(root),
|
|
claimKeyId: 'key-1',
|
|
mintSpawnToken: () => 'spawn-a',
|
|
// Every write lands on its own millisecond, as it does live.
|
|
now: () => (clock += 1)
|
|
})
|
|
expect((await host.attach(CALLER, hostTestAttachParams(null))).ok).toBe(true)
|
|
const page = await host.history({ sessionId: SESSION, direction: 'tail' })
|
|
if (!page.ok) {
|
|
throw new Error('no history page')
|
|
}
|
|
client = reduceStructuredAgentSession(EMPTY_STRUCTURED_AGENT_SESSION, {
|
|
type: 'history-page',
|
|
page: page.page
|
|
})
|
|
host.subscribe({
|
|
id: 'client',
|
|
sessionId: SESSION,
|
|
cursor: page.page.liveCursor ?? page.page.window.nextCursor,
|
|
emit: (event) => {
|
|
client = reduceStructuredAgentSession(client, { type: 'event', event })
|
|
}
|
|
})
|
|
})
|
|
|
|
afterEach(async () => {
|
|
await host.flushAllStreamedEvents()
|
|
await rm(root, { recursive: true, force: true })
|
|
})
|
|
|
|
function codexPrompts(): CodexJournalPrompts {
|
|
if (!sink) {
|
|
throw new Error('session was never acquired')
|
|
}
|
|
return new CodexJournalPrompts(
|
|
{ sink, attributionFor: () => ({ turnScope: AGENT_JOURNAL_THREAD_SCOPE }) },
|
|
() => null,
|
|
() => 'turn-1'
|
|
)
|
|
}
|
|
|
|
async function ask(prompts: CodexJournalPrompts, asked: Asked = ASKED): Promise<void> {
|
|
prompts.handle({
|
|
threadId: THREAD,
|
|
method: CODEX_USER_INPUT_METHOD,
|
|
codexItemId: 'codex-item-1',
|
|
promptKey: '7',
|
|
params: {
|
|
threadId: THREAD,
|
|
turnId: 'turn-1',
|
|
questions: asked.map((question) => ({
|
|
...question,
|
|
options: [
|
|
{ label: 'Yes', description: '' },
|
|
{ label: 'No', description: '' }
|
|
]
|
|
}))
|
|
}
|
|
})
|
|
await host.flushStreamedEvents(SESSION)
|
|
}
|
|
|
|
function questionItem(question: string): AgentJournalRenderItem {
|
|
const item = client.items.find(
|
|
(candidate) => candidate.body.kind === 'question' && candidate.body.question === question
|
|
)
|
|
if (!item || item.body.kind !== 'question') {
|
|
throw new Error(`no journal item for ${question}`)
|
|
}
|
|
return item
|
|
}
|
|
|
|
async function answer(question: string): Promise<void> {
|
|
const item = questionItem(question)
|
|
if (item.body.kind !== 'question') {
|
|
return
|
|
}
|
|
const fields = {
|
|
itemId: item.itemId,
|
|
expectedRevision: item.revision,
|
|
optionId: item.body.options[0]!.id
|
|
}
|
|
const result = await host.respondToPrompt(CALLER, {
|
|
envelope: {
|
|
sessionId: SESSION,
|
|
clientOperationId: hostTestOperationId(),
|
|
expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1,
|
|
payloadFingerprint: computeAgentSessionPayloadFingerprint({
|
|
method: 'agentSession.respondTo:question',
|
|
sessionId: SESSION,
|
|
fields
|
|
})
|
|
},
|
|
kind: 'question',
|
|
...fields
|
|
})
|
|
expect(result.ok).toBe(true)
|
|
await host.flushStreamedEvents(SESSION)
|
|
}
|
|
|
|
/** The prompt rows the transcript list draws, top to bottom, as the questions each one shows. */
|
|
function drawnPromptRows(): string[][] {
|
|
const { receipts } = structuredQuestionTranscript(client.items)
|
|
// The desktop list's projection: its comparator adds only a rank for rows the host never writes.
|
|
const rows = projectNativeChatTranscriptMessages(
|
|
projectStructuredAgentSessionMessages(
|
|
client.items,
|
|
[],
|
|
client.submissions,
|
|
projectStructuredQuestionMessages
|
|
)
|
|
)
|
|
return rows.flatMap((row) => {
|
|
const prompt = receipts.get(row.id)
|
|
if (!prompt || prompt.kind !== 'question') {
|
|
return []
|
|
}
|
|
const questions = prompt.questions?.length ? prompt.questions : [prompt]
|
|
return [questions.map(({ question }) => `${question} (${prompt.resolution.state})`)]
|
|
})
|
|
}
|
|
|
|
describe('a Codex ask with several questions', () => {
|
|
it('keeps the pending rest of the ask below the question already answered', async () => {
|
|
await ask(codexPrompts())
|
|
await answer(ASKED[0]!.question)
|
|
|
|
expect(drawnPromptRows()).toEqual([
|
|
['Which files are in scope? (resolved)'],
|
|
['What matters most? (pending)', 'When is it due? (pending)']
|
|
])
|
|
})
|
|
|
|
it('draws every answered question in the order Codex asked it', async () => {
|
|
await ask(codexPrompts())
|
|
for (const { question } of ASKED) {
|
|
await answer(question)
|
|
}
|
|
|
|
expect(drawnPromptRows()).toEqual([
|
|
['Which files are in scope? (resolved)'],
|
|
['What matters most? (resolved)'],
|
|
['When is it due? (resolved)']
|
|
])
|
|
// Mobile draws the shared projection in journal order, one row per question.
|
|
expect(
|
|
projectStructuredAgentSessionMessages(client.items, [], client.submissions).map(
|
|
({ blocks }) => (blocks[0]?.type === 'text' ? blocks[0].text.split('\n')[0] : null)
|
|
)
|
|
).toEqual(ASKED.map(({ question }) => question))
|
|
})
|
|
|
|
it('draws a cancelled ask in the order Codex asked it', async () => {
|
|
const prompts = codexPrompts()
|
|
await ask(prompts, ASKED_OUT_OF_ORDER)
|
|
prompts.cancel(questionItem(ASKED_OUT_OF_ORDER[0]!.question).itemId)
|
|
await host.flushStreamedEvents(SESSION)
|
|
|
|
expect(drawnPromptRows()).toEqual([
|
|
['Which format? (cancelled)'],
|
|
['Who reads it? (cancelled)'],
|
|
['How long? (cancelled)']
|
|
])
|
|
})
|
|
})
|