fix: stop the chat repeating run results the card already shows (#11293)

* fix: tell the chat not to repeat run results the card already shows

* fix: add the run result note on completion, background runs included

* fix: skip the run result note on failed runs

* fix: keep the run result note off background completions
This commit is contained in:
Guilhem
2026-09-22 15:29:20 +00:00
committed by GitHub
parent 592ef73610
commit ea0b947644
2 changed files with 59 additions and 1 deletions
@@ -1510,6 +1510,50 @@ describe('pollJobCompletion detach', () => {
})
})
describe('executeTestRun result note', () => {
// A failed run's card lands on the error, with the logs a tab away: the model must stay
// free to quote them, so only a success says the user already sees the run.
it('tells the model not to repeat a successful result, and not a failed one', async () => {
vi.useFakeTimers()
try {
const { executeTestRun } = await import('./shared')
const { JobService } = await import('$lib/gen')
const getJobUpdates = vi.mocked(JobService.getJobUpdates)
getJobUpdates.mockReset()
getJobUpdates.mockResolvedValue({ completed: true, running: false } as any)
const run = async (success: boolean) => {
vi.mocked(JobService.getJob).mockReset()
vi.mocked(JobService.getJob).mockResolvedValue({
type: 'CompletedJob',
success,
result: [1, 2],
logs: 'ran'
} as any)
const promise = executeTestRun({
jobStarter: async () => 'job1',
workspace: 'w',
toolId: 'tool1',
startMessage: 'Starting...',
contextName: 'script',
// The job hooks are what mark the global/sessions chat.
toolCallbacks: {
setToolStatus: vi.fn(),
removeToolStatus: vi.fn(),
onJobStatus: vi.fn(),
onJobStarted: vi.fn()
} as any
})
await vi.advanceTimersByTimeAsync(1000)
return promise
}
expect(await run(true)).toContain('Do not repeat')
expect(await run(false)).not.toContain('Do not repeat')
} finally {
vi.useRealTimers()
}
})
})
describe('deriveChatJobStatus', () => {
// CompletedJob is discriminated by the presence of a `success` key; the branch
// order deliberately mirrors JobStatusIcon so the badge and scalar never drift.
@@ -1871,6 +1871,15 @@ export async function buildTestRunArgs(
return parsedArgs
}
// Said with the result rather than in the system prompt: there, models still copied a
// result's table or JSON into the reply, under a card that already renders it. Only for
// a successful run in the generic card completedJobToolStatus fills: a failed run's card
// lands on the error with the logs a tab away, and a formatCompletion card shows its own.
// Never on a background completion: its card can be far up the chat by then, so the
// reply may be the only place the user reads the result.
const RESULT_SHOWN_NOTE =
"The user already sees this run's result in its card. Do not repeat the result or logs in your reply: say what it shows, quoting only the values your conclusion rests on."
// The string handed back to the model when a job is backgrounded. It carries the
// job id so the model can pull status/args/result/logs on demand (get_run / list_runs),
// and tells it the completion will be reported later (notify-only wake).
@@ -2015,7 +2024,12 @@ export async function executeTestRun(config: TestRunConfig): Promise<string> {
...(job.success ? {} : { error: getErrorMessage(job.result) })
})
const summary = formatResultSummary(job.result, job.logs, job.success)
// detachEnabled marks the global/sessions chat; the in-editor chats read the summary
// alone. The card opening on the result is up to each tool: a run card does, a
// generic one only with `autoCollapseDetails: false` at its registration.
const summary =
(detachEnabled && job.success ? `${RESULT_SHOWN_NOTE}\n` : '') +
formatResultSummary(job.result, job.logs, job.success)
// get_run only exists in the global/sessions chat (the same hosts that wire
// the job hooks) — don't advertise it to in-editor chats.
if (detachEnabled && config.contextName === 'flow' && !job.success) {