fix(orchestration): validate worker models on the execution target

This commit is contained in:
Merge Sim
2026-09-10 16:48:20 -07:00
parent 36015ff005
commit 81405ed509
23 changed files with 408 additions and 138 deletions
+1 -1
View File
@@ -34,7 +34,7 @@ export const ORCHESTRATION_WORKER_COMMAND_SPECS: CommandSpec[] = [
notes: [
'Current and existing worktrees never rerun setup; a fresh agent terminal is created unless --terminal is explicit.',
'When reusing --terminal, pass --worktree for that terminal; current means the coordinator worktree.',
'--model takes an id the agent CLI accepts, such as sonnet or opus[1m]. For Claude, whose CLI publishes its whole list, an id that host does not list is rejected and the error names the accepted ones; for Codex and Cursor, whose lists depend on the account and are not exhaustive, the id is passed through and the agent CLI reports any error itself. --effort requires --model and must be a level that model lists. Neither can combine with --terminal.',
'--model takes an id the agent CLI accepts, such as sonnet, opus[1m], or a full model name. When Claude publishes its model catalog, ids that are neither listed choices, their resolved full names, nor stable aliases are rejected; for Codex and Cursor, whose lists are not exhaustive, the id is passed through and the agent CLI reports any error itself. --effort requires --model and must be supported for that model. Neither can combine with --terminal.',
'New worktrees use agent-first creation and default --setup to run. Repository start-immediately runs setup beside the agent; wait-for-setup gates agent readiness and task input.',
'Creation flags (--name, --repo, --base-branch, --display-name, --comment, --setup) are rejected for current/existing worktrees. Use exact --repo on the selected server; project/host convenience routing remains on worktree create.',
"How the worker runs follows the user's own setting for new agent tabs; there is no flag for it and no caller needs to ask. A dispatch the setting cannot apply to still starts, so the placement, agent, and launch options passed here are always the ones honoured.",
@@ -22,6 +22,10 @@ import { ClientHostedBrowserRowPublisher } from './client-hosted-browser-row-pub
import { getRuntimeBrowserPageRegistry } from './runtime-browser-page-registry'
import { getBrowserHostLeaseRegistry } from './browser-host-lease-registry-instance'
import type { RuntimeLeafRecord } from './runtime-terminal-state-records'
import { LOCAL_EXECUTION_HOST_ID, toSshExecutionHostId } from '../../shared/execution-host'
import { parseWslPath } from '../wsl'
import { requireWorktreeCreateRoute } from '../worktree-create-execution-host-route'
import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options'
export class OrcaRuntimeWithFileCommands extends OrcaRuntimeWithPreservedBranchCleanup {
protected readonly fileCommands = new RuntimeFileCommands({
@@ -71,6 +75,36 @@ export class OrcaRuntimeWithFileCommands extends OrcaRuntimeWithPreservedBranchC
protected readonly gitCommands = new RuntimeGitCommands({
resolveRuntimeGitTarget: (selector) => this.resolveRuntimeGitTarget(selector),
resolveRuntimeModelDiscoveryTarget: async (selector) => {
if (typeof selector !== 'string') {
const repo = await this.resolveRepoSelector(selector.repoSelector)
const route = requireWorktreeCreateRoute(repo)
return {
cwd: repo.path,
executionHostId: route.hostId,
...(route.kind === 'local'
? { localGitOptions: getLocalProjectWorktreeGitOptions(this.requireStore(), repo) }
: {})
}
}
const folderScope = await this.resolveFolderWorkspaceLaunchScope(selector)
if (folderScope) {
const wslDistro = parseWslPath(folderScope.path)?.distro
return {
cwd: folderScope.path,
executionHostId: folderScope.connectionId
? toSshExecutionHostId(folderScope.connectionId)
: LOCAL_EXECUTION_HOST_ID,
...(wslDistro ? { localGitOptions: { wslDistro } } : {})
}
}
const target = await this.resolveRuntimeGitTarget(selector)
return {
cwd: target.worktree.path,
executionHostId: target.executionHostId,
localGitOptions: target.localGitOptions
}
},
getRuntimeSettings: () => this.requireStore().getSettings() as GlobalSettings,
getCommitMessageAgentEnvironment: () => this.accounts.getCommitMessageAgentEnvironment(),
// Why: resolved worktrees are cached for a second, so link/unlink would lag
@@ -167,6 +167,27 @@ describe('orchestration worker workspace resolution', () => {
})
})
it('resolves model-discovery hosts from destination repos before a worktree exists', async () => {
const remoteRepo = {
id: 'repo-remote',
path: '/srv/repo',
displayName: 'Remote app',
badgeColor: 'blue',
addedAt: 1,
connectionId: 'ssh-1'
} satisfies Repo
const runtime = new OrcaRuntimeService(
makeStore({ repos: [makeStore().getRepos()[0], remoteRepo] }) as never
)
await expect(
runtime.resolveRuntimeCommitMessageDiscoveryHostKey({ repoSelector: `id:${REPO_ID}` })
).resolves.toBe('local')
await expect(
runtime.resolveRuntimeCommitMessageDiscoveryHostKey({ repoSelector: 'id:repo-remote' })
).resolves.toBe('ssh:ssh-1')
})
it('does not fall back from the floating terminal sentinel to another workspace', async () => {
const runtime = new OrcaRuntimeService(makeStore() as never)
@@ -272,6 +293,12 @@ describe('orchestration worker workspace resolution', () => {
await expect(
runtime.showManagedTerminalWorkspace('id:folder:remote-folder')
).resolves.toMatchObject({ id: 'folder:remote-folder', hostId: 'ssh:ssh-folder' })
await expect(
runtime.resolveRuntimeCommitMessageDiscoveryHostKey('folder:local-folder')
).resolves.toBe('local')
await expect(
runtime.resolveRuntimeCommitMessageDiscoveryHostKey('id:folder:remote-folder')
).resolves.toBe('ssh:ssh-folder')
} finally {
unregisterSshFilesystemProvider('ssh-folder')
}
@@ -104,6 +104,7 @@ export const GitCommit = WorktreeSelector.extend({
const CommitMessageModelCapability = z.object({
id: z.string(),
resolvedModel: z.string().optional(),
label: z.string(),
thinkingLevels: z.array(z.object({ id: z.string(), label: z.string() })).optional(),
defaultThinkingLevel: z.string().optional()
@@ -13,11 +13,15 @@ import { resolveWorkerLaunchPreferences } from './worker-launch-preferences'
const CLAUDE_CATALOG = getAgentSessionOptionCatalog('claude')!
const CODEX_CATALOG = getAgentSessionOptionCatalog('codex')!
const GROK_CATALOG = getAgentSessionOptionCatalog('grok')!
function liveModel(id: string, effortLevels: readonly string[] = []): CommitMessageModelCapability {
function liveModel(
id: string,
effortLevels: readonly string[] = [],
resolvedModel?: string
): CommitMessageModelCapability {
return {
id,
...(resolvedModel ? { resolvedModel } : {}),
label: id,
...(effortLevels.length > 0
? { thinkingLevels: effortLevels.map((level) => ({ id: level, label: level })) }
@@ -27,7 +31,7 @@ function liveModel(id: string, effortLevels: readonly string[] = []): CommitMess
function probeRuntime(
respond: (worktreeSelector: string) => unknown,
hostKeyFor: (worktreeSelector: string) => string = () => 'local'
hostKeyFor: (worktreeSelector: string) => string | Promise<string> = () => 'local'
): {
runtime: WorkerLaunchModelDiscoveryRuntime
discover: ReturnType<typeof vi.fn>
@@ -66,9 +70,12 @@ describe('worker launch model authority', () => {
clearWorkerLaunchModelAuthorityCacheForTests()
})
it('takes the live Claude CLI list as the whole membership, dropping seed ids it omits', async () => {
it('combines the live Claude CLI list with stable aliases accepted by its model flag', async () => {
const { runtime } = probeRuntime(() =>
probeSuccess([liveModel('opus[1m]', ['low', 'high', 'max']), liveModel('haiku')])
probeSuccess([
liveModel('opus[1m]', ['low', 'high', 'max'], 'claude-opus-5[1m]'),
liveModel('haiku')
])
)
const authority = await resolveWorkerLaunchModelAuthority({
@@ -78,7 +85,10 @@ describe('worker launch model authority', () => {
worktreeSelector: 'id:wt_local'
})
expect(authority).toEqual({ source: 'live', modelIds: ['opus[1m]', 'haiku'] })
expect(authority).toEqual({
source: 'live',
modelIds: ['opus[1m]', 'haiku', 'claude-opus-5[1m]', 'opus']
})
})
it('never asks an agent whose probe only extends the seed, and so refuses nothing', async () => {
@@ -114,7 +124,28 @@ describe('worker launch model authority', () => {
worktreeSelector: 'id:wt_local'
})
// The point of the PR: a resolved Claude id the CLI never offers stays refused.
// Stable aliases remain valid even though list_models only displays concrete choices.
expect(
resolveWorkerLaunchPreferences({ agent: 'claude', model: 'opus', authority: claude })
.preferences
).toEqual({ model: 'opus' })
const resolved = await resolveWorkerLaunchModelAuthority({
catalog: CLAUDE_CATALOG,
agent: 'claude',
runtime: probeRuntime(
() => probeSuccess([liveModel('fable', [], 'claude-fable-5')]),
() => 'ssh:other'
).runtime,
worktreeSelector: 'id:wt_other_host'
})
expect(
resolveWorkerLaunchPreferences({
agent: 'claude',
model: 'claude-fable-5',
authority: resolved
}).preferences
).toEqual({ model: 'claude-fable-5' })
// The point of the PR: an id neither listed nor documented as an alias stays refused.
expect(() =>
resolveWorkerLaunchPreferences({ agent: 'claude', model: 'claude-opus-5', authority: claude })
).toThrow('Agent claude does not accept model claude-opus-5')
@@ -163,7 +194,7 @@ describe('worker launch model authority', () => {
expect(discover).not.toHaveBeenCalled()
})
it('seeds without probing when the selector names no host this client can resolve', async () => {
it('seeds without probing when the selector names no workspace this client can resolve', async () => {
const { runtime, discover } = probeRuntime(
() => probeSuccess([liveModel('opus[1m]')]),
() => {
@@ -175,7 +206,7 @@ describe('worker launch model authority', () => {
catalog: CLAUDE_CATALOG,
agent: 'claude',
runtime,
worktreeSelector: 'id:wt_folder_workspace'
worktreeSelector: 'id:wt_missing'
})
expect(authority.source).toBe('seed')
@@ -259,7 +290,7 @@ describe('worker launch model authority', () => {
})
expect(discover).toHaveBeenCalledTimes(1)
expect(second.modelIds).toEqual(['opus[1m]'])
expect(second.modelIds).toContain('opus[1m]')
})
it('reuses one host answer instead of probing on every dispatch', async () => {
@@ -275,7 +306,7 @@ describe('worker launch model authority', () => {
const second = await resolveWorkerLaunchModelAuthority(args)
expect(discover).toHaveBeenCalledTimes(1)
expect(second.modelIds).toEqual(['opus[1m]'])
expect(second.modelIds).toContain('opus[1m]')
})
it('shares one in-flight probe across dispatches that race it', async () => {
@@ -302,8 +333,8 @@ describe('worker launch model authority', () => {
const [first, second] = await both
expect(discover).toHaveBeenCalledTimes(1)
expect(first.modelIds).toEqual(['opus[1m]'])
expect(second.modelIds).toEqual(['opus[1m]'])
expect(first.modelIds).toContain('opus[1m]')
expect(second.modelIds).toContain('opus[1m]')
})
it('answers with the seed past the dispatch budget, leaving the probe to fill the cache', async () => {
@@ -338,7 +369,41 @@ describe('worker launch model authority', () => {
// The abandoned probe still landed in the cache, so the next dispatch pays nothing.
expect(discover).toHaveBeenCalledTimes(1)
expect(next).toEqual({ source: 'live', modelIds: ['opus[1m]'] })
expect(next.source).toBe('live')
expect(next.modelIds).toContain('opus[1m]')
} finally {
vi.useRealTimers()
}
})
it('applies one dispatch budget to host resolution and model discovery together', async () => {
vi.useFakeTimers()
try {
const never = new Promise<unknown>(() => {})
const { runtime, discover } = probeRuntime(
() => never,
async () => {
await new Promise((resolve) => setTimeout(resolve, 9_000))
return 'local'
}
)
let settled: WorkerLaunchModelAuthority | 'waiting' = 'waiting'
const dispatch = resolveWorkerLaunchModelAuthority({
catalog: CLAUDE_CATALOG,
agent: 'claude',
runtime,
worktreeSelector: 'id:wt_local'
}).then((value) => {
settled = value
})
await vi.advanceTimersByTimeAsync(9_999)
expect(settled).toBe('waiting')
expect(discover).toHaveBeenCalledOnce()
await vi.advanceTimersByTimeAsync(1)
expect(settled).toEqual(SEED_WORKER_LAUNCH_MODEL_AUTHORITY)
await dispatch
} finally {
vi.useRealTimers()
}
@@ -368,7 +433,9 @@ describe('worker launch model authority', () => {
it('keys a remote host separately from the local one', async () => {
const { runtime, discover } = probeRuntime(
(worktreeSelector) =>
probeSuccess([liveModel(worktreeSelector === 'id:wt_remote' ? 'opus' : 'opus[1m]')]),
probeSuccess([
liveModel(worktreeSelector === 'id:wt_remote' ? 'remote-model' : 'local-model')
]),
(worktreeSelector) => (worktreeSelector === 'id:wt_remote' ? 'ssh:box' : 'local')
)
@@ -386,31 +453,13 @@ describe('worker launch model authority', () => {
})
expect(discover).toHaveBeenCalledTimes(2)
expect(local.modelIds).toEqual(['opus[1m]'])
expect(remote.modelIds).toEqual(['opus'])
expect(local.modelIds).toContain('local-model')
expect(local.modelIds).not.toContain('remote-model')
expect(remote.modelIds).toContain('remote-model')
expect(remote.modelIds).not.toContain('local-model')
})
it('keys each agent separately on the same host', async () => {
const { runtime, discover } = probeRuntime(() => probeSuccess([liveModel('opus[1m]')]))
await resolveWorkerLaunchModelAuthority({
catalog: CLAUDE_CATALOG,
agent: 'claude',
runtime,
worktreeSelector: 'id:wt_local'
})
// Grok, not Codex: only an agent whose discovery replaces the seed is probed at all.
await resolveWorkerLaunchModelAuthority({
catalog: GROK_CATALOG,
agent: 'grok',
runtime,
worktreeSelector: 'id:wt_local'
})
expect(discover).toHaveBeenCalledTimes(2)
})
it('names the agent, the rejected id and the sorted ids the host actually listed', () => {
it('names the agent, rejected id and sorted ids the host accepts', () => {
expect(
describeWorkerLaunchModelRejection({
agent: 'claude',
@@ -418,7 +467,7 @@ describe('worker launch model authority', () => {
authority: { source: 'live', modelIds: ['sonnet', 'opus'] }
})
).toBe(
'Agent claude does not accept model claude-opus-5. Accepted ids (listed by the claude CLI on the executing host): opus, sonnet.'
'Agent claude does not accept model claude-opus-5. Accepted ids for the claude CLI on the executing host: opus, sonnet.'
)
})
})
@@ -4,8 +4,8 @@
* The agent CLI on the host that will run the worker is the only authority for which ids exist
* there, so this reuses the same probe the native chat model picker uses
* (`runtime.discoverRuntimeCommitMessageModels`), which already routes local / WSL / SSH from the
* worktree selector, and the same catalog policy that decides what the picker offers — so the set
* `worker-start` accepts is the set the picker lists.
* worktree selector. Dynamic membership is combined with stable CLI aliases: a picker need not
* display an alias such as `opus`, but the launch flag still accepts it.
*
* Two things answer `seed`, which claims no membership at all and so refuses nothing.
*
@@ -32,7 +32,7 @@ export type WorkerLaunchModelSource = 'live' | 'seed'
export type WorkerLaunchModelAuthority = {
source: WorkerLaunchModelSource
/** The host's whole membership. Empty and meaningless unless `source` is `live`. */
/** The host's listed ids plus known CLI aliases. Empty unless `source` is `live`. */
modelIds: readonly string[]
}
@@ -40,10 +40,12 @@ export type WorkerLaunchModelDiscoveryRuntime = Pick<
OrcaRuntimeService,
'discoverRuntimeCommitMessageModels' | 'resolveRuntimeCommitMessageDiscoveryHostKey'
>
export type WorkerLaunchModelDiscoveryTarget = string | { repoSelector: string }
const DISCOVERY_TTL_MS = 3 * 60_000
/** A dispatch may not wait out the probe's own 60s budget; the seed answers past this. */
const DISCOVERY_BUDGET_MS = 10_000
const DISCOVERY_BUDGET_EXPIRED = Symbol('worker-launch-model-discovery-budget-expired')
export const SEED_WORKER_LAUNCH_MODEL_AUTHORITY: WorkerLaunchModelAuthority = {
source: 'seed',
@@ -72,21 +74,34 @@ function liveWorkerLaunchModelAuthority(args: {
models: readonly CommitMessageModelCapability[]
}): WorkerLaunchModelAuthority {
const discovered = args.models.map(discoveredCatalogModel)
const listedIds = resolveDiscoveredCatalogModels(args.agent, args.catalog, discovered).map(
({ id }) => id
)
const listedAliases = args.catalog.models
.filter(
(model) =>
model.isCliAlias && listedIds.some((id) => id === model.id || id.startsWith(`${model.id}[`))
)
.map(({ id }) => id)
return {
source: 'live',
modelIds: resolveDiscoveredCatalogModels(args.agent, args.catalog, discovered).map(
({ id }) => id
)
modelIds: [
...new Set([
...listedIds,
...args.models.flatMap(({ resolvedModel }) => (resolvedModel ? [resolvedModel] : [])),
...listedAliases
])
]
}
}
async function probeHostModels(
runtime: WorkerLaunchModelDiscoveryRuntime,
agent: TuiAgent,
worktreeSelector: string
target: WorkerLaunchModelDiscoveryTarget
): Promise<readonly CommitMessageModelCapability[] | null> {
try {
const result = await runtime.discoverRuntimeCommitMessageModels(worktreeSelector, agent)
const result = await runtime.discoverRuntimeCommitMessageModels(target, agent)
// `catalogOrigin: 'spec'` is the probe falling back to Orca's own list, not a CLI answer.
return result.success && result.catalogOrigin === 'probe' && result.models.length > 0
? result.models
@@ -96,20 +111,24 @@ async function probeHostModels(
}
}
function withDiscoveryBudget(
pending: Promise<readonly CommitMessageModelCapability[] | null>
): Promise<readonly CommitMessageModelCapability[] | null> {
return new Promise((resolve) => {
const timer = setTimeout(() => resolve(null), DISCOVERY_BUDGET_MS)
function withDiscoveryDeadline<T>(
pending: Promise<T>,
deadlineAt: number
): Promise<T | typeof DISCOVERY_BUDGET_EXPIRED> {
return new Promise((resolve, reject) => {
const timer = setTimeout(
() => resolve(DISCOVERY_BUDGET_EXPIRED),
Math.max(0, deadlineAt - Date.now())
)
timer.unref?.()
void pending.then(
(value) => {
clearTimeout(timer)
resolve(value)
},
() => {
(error) => {
clearTimeout(timer)
resolve(null)
reject(error)
}
)
})
@@ -126,29 +145,37 @@ function readCachedModels(scope: string): readonly CommitMessageModelCapability[
}
/**
* `worktreeSelector` names the worktree whose host will run the worker; pass null when no
* worktree exists yet on that host, which leaves the seed as the only honest answer.
* `worktreeSelector` names the existing workspace or destination repo whose host will run the
* worker. The historical name is retained because workspace callers pass string selectors.
*/
export async function resolveWorkerLaunchModelAuthority(args: {
catalog: AgentSessionOptionCatalog
agent: TuiAgent
runtime: WorkerLaunchModelDiscoveryRuntime | null
worktreeSelector: string | null
worktreeSelector: WorkerLaunchModelDiscoveryTarget | null
}): Promise<WorkerLaunchModelAuthority> {
const { catalog, agent, runtime, worktreeSelector } = args
const { catalog, agent, runtime, worktreeSelector: target } = args
// Why: for an agent whose probe only EXTENDS the seed, the host's list is known not to be
// exhaustive, so it can refuse nothing — and there is correspondingly nothing to ask it.
if (!discoveredModelsReplaceSeed(agent, catalog)) {
return SEED_WORKER_LAUNCH_MODEL_AUTHORITY
}
if (!runtime || !worktreeSelector) {
if (!runtime || !target) {
return SEED_WORKER_LAUNCH_MODEL_AUTHORITY
}
// A selector this host cannot resolve (an unknown worktree, a folder workspace) has no host to
// ask; the probe would fail the same way, so skip it rather than spend the budget.
const deadlineAt = Date.now() + DISCOVERY_BUDGET_MS
// A selector this host cannot resolve has no host to ask; the probe would fail the same way, so
// skip it rather than spend the budget.
let hostKey: string
try {
hostKey = await runtime.resolveRuntimeCommitMessageDiscoveryHostKey(worktreeSelector)
const resolvedHostKey = await withDiscoveryDeadline(
runtime.resolveRuntimeCommitMessageDiscoveryHostKey(target),
deadlineAt
)
if (resolvedHostKey === DISCOVERY_BUDGET_EXPIRED) {
return SEED_WORKER_LAUNCH_MODEL_AUTHORITY
}
hostKey = resolvedHostKey
} catch (error) {
// An unresolvable selector and a runtime that no longer carries this method both land here,
// and only the second makes `--model` validation a permanent no-op. A missing method is the
@@ -166,7 +193,7 @@ export async function resolveWorkerLaunchModelAuthority(args: {
let pending = inFlightByHost.get(scope)
if (!pending) {
// Failures are never cached, so the next dispatch retries rather than inheriting a miss.
pending = probeHostModels(runtime, agent, worktreeSelector).then((models) => {
pending = probeHostModels(runtime, agent, target).then((models) => {
inFlightByHost.delete(scope)
if (models) {
cachedByHost.set(scope, { expiresAt: Date.now() + DISCOVERY_TTL_MS, models })
@@ -176,8 +203,8 @@ export async function resolveWorkerLaunchModelAuthority(args: {
inFlightByHost.set(scope, pending)
}
// A dispatch that gives up on the budget still leaves the probe running for the next one.
const models = await withDiscoveryBudget(pending)
return models
const models = await withDiscoveryDeadline(pending, deadlineAt)
return models && models !== DISCOVERY_BUDGET_EXPIRED
? liveWorkerLaunchModelAuthority({ catalog, agent, models })
: SEED_WORKER_LAUNCH_MODEL_AUTHORITY
}
@@ -188,7 +215,7 @@ export function describeWorkerLaunchModelRejection(args: {
authority: WorkerLaunchModelAuthority
}): string {
const ids = [...args.authority.modelIds].sort()
return `Agent ${args.agent} does not accept model ${args.model}. Accepted ids (listed by the ${args.agent} CLI on the executing host): ${ids.join(', ')}.`
return `Agent ${args.agent} does not accept model ${args.model}. Accepted ids for the ${args.agent} CLI on the executing host: ${ids.join(', ')}.`
}
export function clearWorkerLaunchModelAuthorityCacheForTests(): void {
@@ -30,8 +30,11 @@ describe('orchestration worker launch preferences', () => {
})
})
it('accepts an id only the live CLI list carries, and refuses one it has dropped', () => {
const authority = { source: 'live' as const, modelIds: ['opus[1m]'] }
it('accepts a listed id and a stable alias, but refuses an unknown id', () => {
const authority = {
source: 'live' as const,
modelIds: ['opus[1m]', 'fable', 'opus', 'sonnet', 'haiku']
}
expect(
resolveWorkerLaunchPreferences({
@@ -41,11 +44,13 @@ describe('orchestration worker launch preferences', () => {
authority
}).preferences
).toEqual({ model: 'opus[1m]', effort: 'max' })
// The live list is the whole membership: a seed id the CLI no longer lists is gone.
expect(
resolveWorkerLaunchPreferences({ agent: 'claude', model: 'opus', authority }).preferences
).toEqual({ model: 'opus' })
expect(() =>
resolveWorkerLaunchPreferences({ agent: 'claude', model: 'sonnet', authority })
resolveWorkerLaunchPreferences({ agent: 'claude', model: 'claude-opus-5', authority })
).toThrow(
'Agent claude does not accept model sonnet. Accepted ids (listed by the claude CLI on the executing host): opus[1m].'
'Agent claude does not accept model claude-opus-5. Accepted ids for the claude CLI on the executing host: fable, haiku, opus, opus[1m], sonnet.'
)
})
@@ -62,27 +67,25 @@ describe('orchestration worker launch preferences', () => {
).toEqual({ model: 'opus[1m]', effort: 'max' })
})
it('keeps a seeded model’s own effort menu when the probe advertises fewer levels', () => {
// The Codex probe reports one generic level list for every model; narrowing to it would
// refuse `ultra` on a warm cache and accept it on a cold one.
const authority = { source: 'live' as const, modelIds: ['gpt-5.6-sol'] }
it('keeps a seeded alias’s effort menu when live membership accepts it', () => {
const authority = { source: 'live' as const, modelIds: ['opus'] }
expect(
resolveWorkerLaunchPreferences({
agent: 'codex',
model: 'gpt-5.6-sol',
effort: 'ultra',
agent: 'claude',
model: 'opus',
effort: 'max',
authority
}).preferences
).toEqual({ model: 'gpt-5.6-sol', effort: 'ultra' })
).toEqual({ model: 'opus', effort: 'max' })
expect(() =>
resolveWorkerLaunchPreferences({
agent: 'codex',
model: 'gpt-5.6-sol',
agent: 'claude',
model: 'opus',
effort: 'not-a-level',
authority
})
).toThrow('Agent codex model gpt-5.6-sol does not support effort not-a-level.')
).toThrow('Agent claude model opus does not support effort not-a-level.')
})
it('accepts a seed id when the host could not be listed', () => {
@@ -133,22 +136,6 @@ describe('orchestration worker launch preferences', () => {
}
})
it.each(['gpt-5.4', 'gpt-5.4-mini', 'gpt-5.3-codex-spark', 'future-codex-model'])(
'refuses an unlisted Codex id only once the host has answered: %s',
(model) => {
expect(resolveWorkerLaunchPreferences({ agent: 'codex', model }).preferences).toEqual({
model
})
expect(() =>
resolveWorkerLaunchPreferences({
agent: 'codex',
model,
authority: { source: 'live', modelIds: ['gpt-5.6-sol'] }
})
).toThrow(`Agent codex does not accept model ${model}.`)
}
)
it('rejects effort without a model', () => {
expect(() => resolveWorkerLaunchPreferences({ agent: 'codex', effort: 'high' })).toThrow(
'--effort requires --model'
@@ -52,8 +52,8 @@ export function createPendingWorkerLaunchReceipt(args: {
}
}
/** `authority` names the ids the executing host's CLI lists. Only a `live` one may refuse a
* model: a seed fallback (or no authority at all) means the host was never listed. */
/** `authority` names the ids the executing host accepts from its live list and known aliases.
* Only a `live` one may refuse; a seed fallback means the host was never listed. */
export function resolveWorkerLaunchPreferences(args: {
agent: TuiAgent
model?: string
@@ -15,17 +15,19 @@ import type { WorkerStartInput } from './worker-start-schema'
*/
function validationRuntime(): {
runtime: OrcaRuntimeService
discover: ReturnType<typeof vi.fn>
resolveHostKey: ReturnType<typeof vi.fn>
} {
const resolveHostKey = vi.fn(async () => 'local')
const discover = vi.fn(async () => ({ success: false, error: 'no CLI' }))
const runtime = {
validateOrchestrationAgentLauncher: vi.fn(),
showTerminal: vi.fn(async () => ({ worktreeId: 'wt_coordinator' })),
getOrchestrationDispatchAuthority: vi.fn(() => null),
resolveRuntimeCommitMessageDiscoveryHostKey: resolveHostKey,
discoverRuntimeCommitMessageModels: vi.fn(async () => ({ success: false, error: 'no CLI' }))
discoverRuntimeCommitMessageModels: discover
} as unknown as OrcaRuntimeService
return { runtime, resolveHostKey }
return { runtime, discover, resolveHostKey }
}
function localParams(overrides: Partial<WorkerStartInput>): WorkerStartInput {
@@ -78,6 +80,22 @@ describe('worker start placement to probe selector', () => {
}
)
it.each(['new-child', 'new-top-level'])(
'probes an explicit destination repo for placement %s',
async (worktree) => {
const { runtime, resolveHostKey } = validationRuntime()
await prepareLocalWorkerStart({
params: localParams({ worktree, name: 'child', repo: 'id:repo_remote' }),
createsWorktree: true,
runtime
})
expect(resolveHostKey).toHaveBeenCalledWith({ repoSelector: 'id:repo_remote' })
expect(runtime.showTerminal).not.toHaveBeenCalled()
}
)
it('probes the named worktree itself when the placement names one', async () => {
const { runtime, resolveHostKey } = validationRuntime()
@@ -117,6 +135,37 @@ describe('worker start placement to probe selector', () => {
expect(runtime.showTerminal).not.toHaveBeenCalled()
})
it('rejects an unsupported model option before resolving or probing a host', async () => {
const { runtime, discover, resolveHostKey } = validationRuntime()
await expect(
prepareLocalWorkerStart({
params: localParams({ agent: 'grok', model: 'grok-code-fast-1' }),
createsWorktree: false,
runtime
})
).rejects.toThrow('Agent grok does not support launch-time model selection.')
expect(runtime.showTerminal).not.toHaveBeenCalled()
expect(resolveHostKey).not.toHaveBeenCalled()
expect(discover).not.toHaveBeenCalled()
})
it('does not resolve or probe a host for a pass-through model catalog', async () => {
const { runtime, discover, resolveHostKey } = validationRuntime()
const plan = await prepareLocalWorkerStart({
params: localParams({ agent: 'codex', model: 'account-only-model' }),
createsWorktree: false,
runtime
})
expect(plan.launch.preferences).toEqual({ model: 'account-only-model' })
expect(runtime.showTerminal).not.toHaveBeenCalled()
expect(resolveHostKey).not.toHaveBeenCalled()
expect(discover).not.toHaveBeenCalled()
})
it('probes the remote worktree a federated attachment names', async () => {
const { runtime, resolveHostKey } = validationRuntime()
@@ -129,7 +178,7 @@ describe('worker start placement to probe selector', () => {
expect(resolveHostKey).toHaveBeenCalledWith('remote-worktree')
})
it('probes nothing for a federated new-top-level, which has no host until the remote makes it', async () => {
it('probes the destination repo for a federated new-top-level', async () => {
const { runtime, resolveHostKey } = validationRuntime()
await prepareFederationAttachmentWorkerStart({
@@ -138,6 +187,6 @@ describe('worker start placement to probe selector', () => {
runtime
})
expect(resolveHostKey).not.toHaveBeenCalled()
expect(resolveHostKey).toHaveBeenCalledWith({ repoSelector: 'repo_1' })
})
})
@@ -2,10 +2,16 @@ import { isTuiAgent } from '../../../../../../shared/tui-agent-config'
import type { TuiAgent } from '../../../../../../shared/tui-agent'
import type { OrcaRuntimeService } from '../../../../orca-runtime'
import { OrchestrationError } from '../../../../orchestration/orchestration-error'
import { getAgentSessionOptionCatalog } from '../../../../../../shared/agent-session-option-catalog'
import {
discoveredModelsReplaceSeed,
getAgentSessionOptionCatalog
} from '../../../../../../shared/agent-session-option-catalog'
import type { FederationAttachStartInput } from '../federation/federation-start-schema'
import { resolveDispatchCallerWorktreeId } from '../../orchestration-caller-workspace'
import { resolveWorkerLaunchModelAuthority } from './worker-launch-model-authority'
import {
resolveWorkerLaunchModelAuthority,
type WorkerLaunchModelDiscoveryTarget
} from './worker-launch-model-authority'
import {
assertWorkerLaunchPreferencesCreateTerminal,
createWorkerLaunchReceipt,
@@ -24,6 +30,21 @@ type WorkerStartAgentPlan = {
const COORDINATOR_HOSTED_PLACEMENTS = new Set(['current', 'new-child', 'new-top-level'])
function canResolveWorkerLaunchModelAuthority(
agent: string | undefined,
model: string | undefined
): boolean {
if (!model || !agent || !isTuiAgent(agent)) {
return false
}
const catalog = getAgentSessionOptionCatalog(agent)
return Boolean(
catalog?.supportsWorkerLaunchPreferences &&
catalog.modelApply.launchArgs &&
discoveredModelsReplaceSeed(agent, catalog)
)
}
/**
* The worktree whose host will run the worker, as a selector the model probe can resolve.
* A worktree that does not exist yet inherits the coordinator's host, which is where it is made.
@@ -31,8 +52,11 @@ const COORDINATOR_HOSTED_PLACEMENTS = new Set(['current', 'new-child', 'new-top-
async function resolveLocalLaunchHost(
runtime: OrcaRuntimeService,
params: WorkerStartInput
): Promise<{ selector: string | null; callerWorktreeId?: string }> {
): Promise<{ selector: WorkerLaunchModelDiscoveryTarget | null; callerWorktreeId?: string }> {
const requested = params.worktree ?? 'current'
if ((requested === 'new-child' || requested === 'new-top-level') && params.repo) {
return { selector: { repoSelector: params.repo } }
}
if (!COORDINATOR_HOSTED_PLACEMENTS.has(requested)) {
return { selector: requested }
}
@@ -109,7 +133,10 @@ export async function prepareLocalWorkerStart(args: {
'Creation and setup options apply only to new-child or new-top-level worktrees.'
)
}
const host: { selector: string | null; callerWorktreeId?: string } = params.model
const host: {
selector: WorkerLaunchModelDiscoveryTarget | null
callerWorktreeId?: string
} = canResolveWorkerLaunchModelAuthority(params.agent, params.model)
? await resolveLocalLaunchHost(runtime, params)
: { selector: null }
const plan = await resolveWorkerStartAgent({
@@ -164,8 +191,7 @@ export async function prepareFederationAttachmentWorkerStart(args: {
agent: params.agent,
model: params.model,
effort: params.effort,
// A remote new-top-level worktree has no host to probe until the remote makes it.
worktreeSelector: createsWorktree ? null : params.worktree,
worktreeSelector: createsWorktree ? { repoSelector: params.repo as string } : params.worktree,
missingAgentMessage:
'A configured --agent is required when federated worker-start creates a terminal.'
})
@@ -177,7 +203,7 @@ async function resolveWorkerStartAgent(args: {
agent?: string
model?: string
effort?: string
worktreeSelector: string | null
worktreeSelector: WorkerLaunchModelDiscoveryTarget | null
missingAgentMessage: string
}): Promise<WorkerStartAgentPlan> {
if (!args.terminal && (!args.agent || !isTuiAgent(args.agent))) {
@@ -187,22 +213,26 @@ async function resolveWorkerStartAgent(args: {
if (agent) {
args.runtime.validateOrchestrationAgentLauncher(agent)
const catalog = args.model ? getAgentSessionOptionCatalog(agent) : null
const launch = resolveWorkerLaunchPreferences({
agent,
model: args.model,
effort: args.effort
})
if (!catalog || !discoveredModelsReplaceSeed(agent, catalog)) {
return { agent, launch }
}
return {
agent,
launch: resolveWorkerLaunchPreferences({
agent,
model: args.model,
effort: args.effort,
...(catalog
? {
authority: await resolveWorkerLaunchModelAuthority({
catalog,
agent,
runtime: args.runtime,
worktreeSelector: args.worktreeSelector
})
}
: {})
authority: await resolveWorkerLaunchModelAuthority({
catalog,
agent,
runtime: args.runtime,
worktreeSelector: args.worktreeSelector
})
})
}
}
@@ -169,21 +169,21 @@ describe('orchestration new-worktree workers', () => {
mockCreatedWorktree()
const { result } = await startWorker({
model: 'gpt-5.5',
model: 'custom-codex-model',
effort: 'high'
})
expect(runtime.createManagedWorktree).toHaveBeenCalledWith(
expect.objectContaining({
startupAgent: 'codex',
startupLaunchPreferences: { model: 'gpt-5.5', effort: 'high' }
startupLaunchPreferences: { model: 'custom-codex-model', effort: 'high' }
})
)
expect(result).toMatchObject({
state: 'ready',
launch: {
requested: { agent: 'codex', model: 'gpt-5.5', effort: 'high' },
effective: { agent: 'codex', model: 'gpt-5.5', effort: 'high' }
requested: { agent: 'codex', model: 'custom-codex-model', effort: 'high' },
effective: { agent: 'codex', model: 'custom-codex-model', effort: 'high' }
}
})
})
+18 -2
View File
@@ -35,8 +35,20 @@ export type RuntimeGitTarget = {
localGitOptions?: GitRuntimeOptions
}
export type RuntimeModelDiscoverySelector = string | { repoSelector: string }
export type RuntimeModelDiscoveryTarget = {
cwd: string
executionHostId: ExecutionHostId
localGitOptions?: GitRuntimeOptions
}
export type RuntimeGitCommandHost = {
resolveRuntimeGitTarget(selector: string): Promise<RuntimeGitTarget>
/** Model discovery needs an execution host and cwd, not an existing Git worktree. */
resolveRuntimeModelDiscoveryTarget?(
selector: RuntimeModelDiscoverySelector
): Promise<RuntimeModelDiscoveryTarget>
getRuntimeSettings(): GlobalSettings
getCommitMessageAgentEnvironment?(): CommitMessageAgentEnvironmentResolvers | undefined
/** `undefined` keeps cached metadata; `null` is the authoritative unlinked answer. */
@@ -63,7 +75,9 @@ export type RuntimeGitRoute =
/** `provider: null` is "remote and currently unreachable" — never "run it here". */
| { kind: 'ssh'; connectionId: string; provider: SshGitProvider | null }
export function runtimeGitRouteForTarget(target: RuntimeGitTarget): RuntimeGitRoute {
export function runtimeGitRouteForTarget(
target: Pick<RuntimeGitTarget, 'executionHostId'>
): RuntimeGitRoute {
const route = resolveGitRouteForHost(target.executionHostId)
switch (route.kind) {
case 'local':
@@ -90,7 +104,9 @@ export function requireRuntimeGitProvider(target: RuntimeGitTarget): SshGitProvi
return route.provider
}
export function localGitOptionsForTarget(target: RuntimeGitTarget): GitRuntimeOptions {
export function localGitOptionsForTarget(
target: Pick<RuntimeGitTarget, 'executionHostId' | 'localGitOptions'>
): GitRuntimeOptions {
// WSL routing describes *this* machine; no remote host may inherit it.
return target.executionHostId === LOCAL_EXECUTION_HOST_ID ? (target.localGitOptions ?? {}) : {}
}
@@ -27,8 +27,10 @@ import { getPullRequestDraftContext } from '../text-generation/pull-request-cont
import {
localGitOptionsForTarget,
runtimeGitRouteForTarget,
type RuntimeGitCommandHost
type RuntimeGitCommandHost,
type RuntimeModelDiscoverySelector
} from './runtime-git-command-target'
import { resolveRuntimeModelDiscoveryTarget } from './runtime-model-discovery-target'
import {
getRuntimeGitGenerationSettings,
linkedIssueForTarget,
@@ -254,10 +256,12 @@ export class RuntimeGitGenerationCommands {
/**
* Which host a discovery for `worktreeSelector` would run its agent CLI on, in the same
* `local` / `wsl:<distro>` / `ssh:<id>` vocabulary the renderer caches this fact under. Every
* worktree on one host answers the same key, so a caller can cache one probe per host.
* workspace or destination repo on one host answers the same key, so callers cache one probe.
*/
async resolveRuntimeCommitMessageDiscoveryHostKey(worktreeSelector: string): Promise<string> {
const target = await this.host.resolveRuntimeGitTarget(worktreeSelector)
async resolveRuntimeCommitMessageDiscoveryHostKey(
selector: RuntimeModelDiscoverySelector
): Promise<string> {
const target = await resolveRuntimeModelDiscoveryTarget(this.host, selector)
const route = runtimeGitRouteForTarget(target)
return route.kind === 'ssh'
? getCommitMessageModelDiscoveryHostKey(route.connectionId)
@@ -267,11 +271,11 @@ export class RuntimeGitGenerationCommands {
}
async discoverRuntimeCommitMessageModels(
worktreeSelector: string,
selector: RuntimeModelDiscoverySelector,
agentId: string,
settingsOverride?: Pick<RuntimeCommitMessageSettingsOverride, 'agentCmdOverrides'>
): Promise<DiscoverCommitMessageModelsResult> {
const target = await this.host.resolveRuntimeGitTarget(worktreeSelector)
const target = await resolveRuntimeModelDiscoveryTarget(this.host, selector)
const typedAgentId = agentId as TuiAgent
const agentCommandOverride =
settingsOverride?.agentCmdOverrides?.[typedAgentId] ??
@@ -284,7 +288,7 @@ export class RuntimeGitGenerationCommands {
}
return discoverCommitMessageModelsRemote(
typedAgentId,
target.worktree.path,
target.cwd,
(plan, cwd, timeoutMs) => provider.executeCommitMessagePlan(plan, cwd, timeoutMs),
agentCommandOverride
)
@@ -300,7 +304,7 @@ export class RuntimeGitGenerationCommands {
const localOptions = localGitOptionsForTarget(target)
return localOptions.wslDistro
? discoverCommitMessageModelsLocal(typedAgentId, localEnv.env, agentCommandOverride, {
cwd: target.worktree.path,
cwd: target.cwd,
wslDistro: localOptions.wslDistro
})
: discoverCommitMessageModelsLocal(typedAgentId, localEnv.env, agentCommandOverride)
@@ -74,7 +74,7 @@ export function getRuntimeGitGenerationSettings(
}
export function localAgentRuntimeTargetForTarget(
target: RuntimeGitTarget
target: Pick<RuntimeGitTarget, 'executionHostId' | 'localGitOptions'>
): CommitMessageAgentRuntimeTarget {
const wslDistro = localGitOptionsForTarget(target).wslDistro
return wslDistro ? { runtime: 'wsl', wslDistro } : { runtime: 'host' }
@@ -0,0 +1,23 @@
import type {
RuntimeGitCommandHost,
RuntimeModelDiscoverySelector,
RuntimeModelDiscoveryTarget
} from './runtime-git-command-target'
export async function resolveRuntimeModelDiscoveryTarget(
host: RuntimeGitCommandHost,
selector: RuntimeModelDiscoverySelector
): Promise<RuntimeModelDiscoveryTarget> {
if (host.resolveRuntimeModelDiscoveryTarget) {
return host.resolveRuntimeModelDiscoveryTarget(selector)
}
if (typeof selector !== 'string') {
throw new Error('repo_model_discovery_unsupported')
}
const target = await host.resolveRuntimeGitTarget(selector)
return {
cwd: target.worktree.path,
executionHostId: target.executionHostId,
localGitOptions: target.localGitOptions
}
}
@@ -138,12 +138,14 @@ export const CLAUDE_SESSION_OPTION_CATALOG: AgentSessionOptionCatalog = {
id: 'fable',
label: 'Fable',
description: 'Most capable for the hardest, longest-running tasks',
isCliAlias: true,
options: [claudeEffort(true)]
},
{
id: 'opus',
label: 'Opus',
description: 'Best for everyday, complex tasks',
isCliAlias: true,
options: [claudeEffort(true), CLAUDE_FAST_MODE]
},
{
@@ -151,12 +153,14 @@ export const CLAUDE_SESSION_OPTION_CATALOG: AgentSessionOptionCatalog = {
label: 'Sonnet',
description: 'Efficient for routine tasks',
isDefault: true,
isCliAlias: true,
options: [claudeEffort(true)]
},
{
id: 'haiku',
label: 'Haiku',
description: 'Fastest for quick answers',
isCliAlias: true,
options: []
}
],
@@ -50,6 +50,8 @@ export type CatalogModel = {
label: string
description?: string
isDefault?: boolean
/** A stable CLI alias accepted when discovery lists it or a bracket-qualified variant. */
isCliAlias?: true
options: CatalogOption[]
}
+2 -1
View File
@@ -96,7 +96,8 @@ export function mergeDiscoveredAuthoritativeModels(
* Its scope is CLI-probe membership only — what `listModels` stdout claims. A probe that merely
* extends is, by construction, not a complete list — the Codex catalog says so of itself — so
* nothing may be refused against it. Both the picker's merge below and `worker-start`'s reject
* gate read this, so what is offered and what is accepted cannot drift apart.
* gate read this to decide whether a negative membership answer is possible. Stable CLI aliases
* can remain accepted even when an authoritative picker list does not display them.
*
* A live session's own model list is a DIFFERENT authority, outside this function's scope: Codex's
* app-server `model/list` and Claude's SDK `supportedModels()` each speak for one connected
@@ -55,6 +55,7 @@ describe('parseClaudeModelList', () => {
expect(parsed.map(({ id }) => id)).toEqual(['opus[1m]', 'sonnet', 'haiku'])
expect(parsed[0]).toEqual({
id: 'opus[1m]',
resolvedModel: 'claude-opus-5[1m]',
label: 'Opus (1M context)',
description: 'Opus 5 with 1M context · Best for everyday, complex tasks · $5/$25 per Mtok',
effortLevels: ['low', 'medium', 'high', 'xhigh', 'max'],
+8
View File
@@ -24,6 +24,8 @@ export const CLAUDE_MODEL_LIST_ARGS = [
export type ClaudeListedModel = {
/** Value the CLI accepts for `--model` and `/model` (e.g. `opus[1m]`). */
id: string
/** Full model id the alias resolves to on this host, also accepted by `--model`. */
resolvedModel?: string
/** The CLI's own picker label (e.g. `Opus (1M context)`). */
label: string
/** Names what the value resolves to on this host (e.g. `Opus 5 with 1M context …`). */
@@ -48,6 +50,7 @@ type RawControlResponse = {
type RawListedModel = {
value?: unknown
resolvedModel?: unknown
displayName?: unknown
description?: unknown
supportsEffort?: unknown
@@ -69,6 +72,10 @@ function toListedModel(value: unknown): ClaudeListedModel | null {
return null
}
const label = typeof raw.displayName === 'string' && raw.displayName.trim() ? raw.displayName : id
const resolvedModel =
typeof raw.resolvedModel === 'string' && raw.resolvedModel.trim()
? raw.resolvedModel.trim()
: undefined
const description =
typeof raw.description === 'string' && raw.description.trim() ? raw.description : undefined
const effortLevels =
@@ -77,6 +84,7 @@ function toListedModel(value: unknown): ClaudeListedModel | null {
: []
return {
id,
...(resolvedModel ? { resolvedModel } : {}),
label,
...(description ? { description } : {}),
effortLevels,
@@ -253,6 +253,7 @@ describe('model discovery parsers', () => {
},
{
value: 'opus[1m]',
resolvedModel: 'claude-opus-5[1m]',
displayName: 'Opus (1M context)',
description: 'Opus 5 with 1M context · $5/$25 per Mtok',
supportsEffort: true,
@@ -267,6 +268,7 @@ describe('model discovery parsers', () => {
expect(parseClaudeModels(stdout)).toEqual([
{
id: 'opus[1m]',
resolvedModel: 'claude-opus-5[1m]',
label: 'Opus (1M context)',
description: 'Opus 5 with 1M context · $5/$25 per Mtok',
thinkingLevels: [
+4
View File
@@ -26,6 +26,8 @@ export type ThinkingLevel = { id: string; label: string }
export type CommitMessageModel = {
/** Value passed to the agent CLI's --model flag. */
id: string
/** Full model id a discovered alias resolves to, when the CLI reports one. */
resolvedModel?: string
/** Visible label in the model dropdown. */
label: string
/** Discovery-provided detail, e.g. what a CLI alias resolves to on this host. */
@@ -70,6 +72,7 @@ export type CommitMessageAgentSpec = {
export type CommitMessageModelCapability = {
id: string
resolvedModel?: string
label: string
description?: string
thinkingLevels?: ThinkingLevel[]
@@ -174,6 +177,7 @@ function toCommitMessageAgentCapability(
// swap this source without leaking binary/argv details into UI code.
models: spec.models.map((model) => ({
id: model.id,
...(model.resolvedModel ? { resolvedModel: model.resolvedModel } : {}),
label: model.label,
...(model.description ? { description: model.description } : {}),
...(model.thinkingLevels ? { thinkingLevels: [...model.thinkingLevels] } : {}),
@@ -77,6 +77,7 @@ export function parseClaudeModels(stdout: string): CommitMessageModel[] {
)
return {
id: model.id,
...(model.resolvedModel ? { resolvedModel: model.resolvedModel } : {}),
label: model.label,
...(model.description ? { description: model.description } : {}),
...(thinkingLevels.length > 0