mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-08-20 08:01:35 +00:00
b0ddcf31e4
* ci: add path-gated AI agent integration tests workflow Runs integration_tests/ai_agent_tests against real LLM providers (Anthropic/OpenAI/Google) only when AI-agent backend code or the tests change, since runs make paid LLM calls. Adds a conftest fixture that skips provider-parametrized cases whose API keys are absent, so CI exercises only the providers it has secrets for. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: add path-gated ai_evals global-mode smoke workflow Runs the global AI chat eval (global-test1) across one cheap model per provider (Anthropic/OpenAI/Google/DeepSeek) only when the eval harness or copilot chat code change, since runs make paid LLM calls. Builds Windmill CE from source as the AI proxy; global tools/drafts run in the Vitest bridge. Gates on the deterministic draft pipeline (run succeeded + produced a draft + used write_script), not the variable LLM judge score. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: run AI smokes on PR ready-for-review instead of every push Switch the pull_request trigger from `synchronize` (every commit) to `ready_for_review`, with a job guard skipping draft PRs, so the paid LLM runs only fire when a PR is marked ready to merge (plus push-to-main and manual dispatch). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(ai_evals): lazily load cli mode so non-cli evals skip the cli toolchain The entrypoint eagerly imported modes/cli, which pulls the wmill CLI guidance modules and their JSR deps (@cliffy/*). Global/flow/script/app runs then crashed with "Cannot find module '@cliffy/ansi/colors'" when the cli workspace deps were not installed. Import createCliModeRunner dynamically inside runCliBenchmark instead. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * test(ai_agent): raise low max_completion_tokens to OpenAI's 16 minimum OpenAI's /v1/responses rejects max_output_tokens < 16 with a 400, failing test_low_max_tokens for openai. 16 still exercises a truncated response. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: run ai_evals workflow on Node 22 for the frontend undici 8.x dep The Vitest bridge loads frontend/node_modules/undici@8.x, which requires Node >=22.19; Node 20 failed with "webidl.util.markAsUncloneable is not a function" when loading vitest.config.ts. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(ai_evals): run frontend evals autonomously + give global-test1 more turns Frontend evals (flow/script/app/global) ran the production chat prompt, which assumes an interactive human — so cheaper models burned their turn budget asking for confirmation, waiting for approval, or presenting a plan, sometimes hitting maxTurns without producing a draft. Append a shared autonomy note in baseEvalRunner (the path all frontend modes share, mirroring cli mode): act directly on clear requests; only ask on genuinely ambiguous ones (preserving the askUserQuestion cases). Also raise global-test1's maxTurns 8 -> 10 so a model that over-explores still converges. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci(ai_evals): watch draft/prompt deps outside copilot/ The global eval runs production frontend code in-process, so the smoke's behavior depends on files outside frontend/src/lib/components/copilot/**: the draft model (userDraft.svelte.ts, userDraftDbSyncer.svelte.ts), script inference (infer.ts), and the chat system prompts ($system_prompts -> system_prompts/auto-generated). Add them to both push and PR path filters so a change there actually triggers the smoke that gates on draft production. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix: skip direct provider tests without credentials * feat: add ai evals skip judge flag * fix: simplify ai evals ci gate * fix: simplify ai evals smoke gate * fix: handle ai eval workflow triggers --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
347 lines
11 KiB
TypeScript
347 lines
11 KiB
TypeScript
import { judgeOutput, DEFAULT_JUDGE_MODEL } from "./judge";
|
|
import type {
|
|
BenchmarkAttemptResult,
|
|
BenchmarkCaseResult,
|
|
BenchmarkCheck,
|
|
EvalCase,
|
|
FrontendBenchmarkProgressEvent,
|
|
ModeRunner,
|
|
} from "./types";
|
|
import { validateToolExpectations } from "./validators";
|
|
|
|
export async function runSuite<TInitial, TExpected, TActual>(input: {
|
|
modeRunner: ModeRunner<TInitial, TExpected, TActual>;
|
|
cases: EvalCase[];
|
|
runs: number;
|
|
runModel: string | null;
|
|
judgeModel?: string | null;
|
|
executionOnly?: boolean;
|
|
concurrency?: number;
|
|
verbose?: boolean;
|
|
onProgress?: (event: FrontendBenchmarkProgressEvent) => void;
|
|
}): Promise<BenchmarkCaseResult[]> {
|
|
const judgeModel =
|
|
input.judgeModel === undefined ? DEFAULT_JUDGE_MODEL : input.judgeModel;
|
|
const concurrency = Math.max(1, input.concurrency ?? input.modeRunner.concurrency);
|
|
const results = new Array<BenchmarkCaseResult>(input.cases.length);
|
|
let cursor = 0;
|
|
|
|
if (input.modeRunner.mode !== "cli") {
|
|
input.onProgress?.({
|
|
type: "run-start",
|
|
surface: input.modeRunner.mode,
|
|
totalCases: input.cases.length,
|
|
runs: input.runs,
|
|
concurrency,
|
|
});
|
|
}
|
|
|
|
async function worker(): Promise<void> {
|
|
while (true) {
|
|
const caseIndex = cursor++;
|
|
if (caseIndex >= input.cases.length) {
|
|
return;
|
|
}
|
|
const evalCase = input.cases[caseIndex];
|
|
results[caseIndex] = {
|
|
id: evalCase.id,
|
|
prompt: evalCase.prompt,
|
|
initialPath: evalCase.initialPath,
|
|
expectedPath: evalCase.expectedPath,
|
|
attempts: await runCaseAttempts({
|
|
caseIndex,
|
|
evalCase,
|
|
runs: input.runs,
|
|
judgeModel,
|
|
judgeThreshold: input.modeRunner.judgeThreshold ?? 80,
|
|
executionOnly: input.executionOnly ?? false,
|
|
modeRunner: input.modeRunner,
|
|
totalCases: input.cases.length,
|
|
verbose: input.verbose ?? false,
|
|
onProgress: input.onProgress,
|
|
}),
|
|
};
|
|
}
|
|
}
|
|
|
|
await Promise.all(
|
|
Array.from({ length: Math.min(concurrency, input.cases.length) }, () => worker())
|
|
);
|
|
|
|
return results;
|
|
}
|
|
|
|
async function runCaseAttempts<TInitial, TExpected, TActual>(input: {
|
|
caseIndex: number;
|
|
evalCase: EvalCase;
|
|
runs: number;
|
|
judgeModel: string | null;
|
|
judgeThreshold: number;
|
|
executionOnly: boolean;
|
|
modeRunner: ModeRunner<TInitial, TExpected, TActual>;
|
|
totalCases: number;
|
|
verbose: boolean;
|
|
onProgress?: (event: FrontendBenchmarkProgressEvent) => void;
|
|
}): Promise<BenchmarkAttemptResult[]> {
|
|
const attempts: BenchmarkAttemptResult[] = [];
|
|
const surface = input.modeRunner.mode === "cli" ? null : input.modeRunner.mode;
|
|
|
|
for (let attempt = 1; attempt <= input.runs; attempt += 1) {
|
|
if (surface) {
|
|
input.onProgress?.({
|
|
type: "attempt-start",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
});
|
|
}
|
|
|
|
const startedAt = Date.now();
|
|
|
|
try {
|
|
const initial = await input.modeRunner.loadInitial(input.evalCase.initialPath);
|
|
const expected = input.executionOnly
|
|
? undefined
|
|
: await input.modeRunner.loadExpected(input.evalCase.expectedPath);
|
|
const run = await input.modeRunner.run(input.evalCase.prompt, initial, {
|
|
evalCase: input.evalCase,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
verbose: input.verbose,
|
|
onAssistantMessageStart: input.verbose && surface
|
|
? () =>
|
|
input.onProgress?.({
|
|
type: "assistant-message-start",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
})
|
|
: undefined,
|
|
onAssistantChunk: input.verbose && surface
|
|
? (chunk: string) =>
|
|
input.onProgress?.({
|
|
type: "assistant-chunk",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
chunk,
|
|
})
|
|
: undefined,
|
|
onAssistantMessageEnd: input.verbose && surface
|
|
? () =>
|
|
input.onProgress?.({
|
|
type: "assistant-message-end",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
})
|
|
: undefined,
|
|
onToolCall: input.verbose && surface
|
|
? ({ toolName, argumentsText }) =>
|
|
input.onProgress?.({
|
|
type: "tool-call",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
toolName,
|
|
argumentsText,
|
|
})
|
|
: undefined,
|
|
});
|
|
const checks: BenchmarkCheck[] = [
|
|
buildCheck("run succeeded", run.success, run.error),
|
|
];
|
|
if (!input.executionOnly) {
|
|
checks.push(
|
|
...input.modeRunner.validate({
|
|
evalCase: input.evalCase,
|
|
prompt: input.evalCase.prompt,
|
|
initial,
|
|
expected,
|
|
actual: run.actual,
|
|
run,
|
|
}),
|
|
...validateToolExpectations({
|
|
run,
|
|
toolExpect: input.evalCase.toolExpect,
|
|
})
|
|
);
|
|
}
|
|
const artifactFiles = input.modeRunner.buildArtifacts?.(run.actual) ?? [];
|
|
|
|
if (
|
|
run.success &&
|
|
!input.executionOnly &&
|
|
input.modeRunner.backendValidate
|
|
) {
|
|
try {
|
|
const backendValidation = await input.modeRunner.backendValidate({
|
|
evalCase: input.evalCase,
|
|
prompt: input.evalCase.prompt,
|
|
initial,
|
|
expected,
|
|
actual: run.actual,
|
|
run,
|
|
context: {
|
|
evalCase: input.evalCase,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
verbose: input.verbose,
|
|
onAssistantMessageStart: undefined,
|
|
onAssistantChunk: undefined,
|
|
onAssistantMessageEnd: undefined,
|
|
},
|
|
});
|
|
|
|
if (backendValidation) {
|
|
checks.push(...backendValidation.checks);
|
|
artifactFiles.push(...(backendValidation.artifactFiles ?? []));
|
|
}
|
|
} catch (error) {
|
|
checks.push(
|
|
buildCheck(
|
|
"backend validation succeeded",
|
|
false,
|
|
error instanceof Error ? error.message : String(error)
|
|
)
|
|
);
|
|
}
|
|
}
|
|
|
|
let judgeScore: number | null = null;
|
|
let judgeSummary: string | null = null;
|
|
|
|
if (
|
|
run.success &&
|
|
!input.executionOnly &&
|
|
input.judgeModel !== null &&
|
|
!input.evalCase.skipJudge
|
|
) {
|
|
const judge = await judgeOutput({
|
|
mode: input.modeRunner.mode,
|
|
prompt: input.evalCase.prompt,
|
|
checklist: input.evalCase.judgeChecklist,
|
|
initial,
|
|
expected: input.modeRunner.mode === "cli" ? undefined : expected,
|
|
actual: input.modeRunner.prepareJudgeActual
|
|
? input.modeRunner.prepareJudgeActual(run.actual)
|
|
: run.actual,
|
|
model: input.judgeModel,
|
|
});
|
|
|
|
judgeScore = judge.success ? judge.score : null;
|
|
judgeSummary = judge.summary;
|
|
checks.push(buildCheck("judge succeeded", judge.success, judge.error));
|
|
checks.push(
|
|
buildCheck(
|
|
`judge score >= ${input.judgeThreshold}`,
|
|
(judgeScore ?? 0) >= input.judgeThreshold,
|
|
judge.success ? `score=${judgeScore}` : judge.error
|
|
)
|
|
);
|
|
}
|
|
|
|
const attemptResult: BenchmarkAttemptResult = {
|
|
attempt,
|
|
passed: checks.every((check) => check.passed),
|
|
durationMs: Date.now() - startedAt,
|
|
assistantMessageCount: run.assistantMessageCount,
|
|
toolCallCount: run.toolCallCount,
|
|
toolsUsed: uniqueStrings(run.toolsUsed),
|
|
toolCallDetails: run.toolCallDetails,
|
|
skillsInvoked: uniqueStrings(run.skillsInvoked),
|
|
checks,
|
|
judgeScore,
|
|
judgeSummary,
|
|
error: run.error ?? null,
|
|
tokenUsage: run.tokenUsage ?? null,
|
|
finalContextTokens: run.finalContextTokens ?? null,
|
|
artifactsPath: null,
|
|
artifactFiles,
|
|
};
|
|
|
|
if (surface) {
|
|
input.onProgress?.({
|
|
type: "attempt-finish",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
passed: attemptResult.passed,
|
|
durationMs: attemptResult.durationMs,
|
|
judgeScore: attemptResult.judgeScore,
|
|
error: attemptResult.error,
|
|
});
|
|
}
|
|
|
|
attempts.push(attemptResult);
|
|
} catch (error) {
|
|
const message = error instanceof Error ? error.message : String(error);
|
|
const failedAttempt: BenchmarkAttemptResult = {
|
|
attempt,
|
|
passed: false,
|
|
durationMs: Date.now() - startedAt,
|
|
assistantMessageCount: 0,
|
|
toolCallCount: 0,
|
|
toolsUsed: [],
|
|
skillsInvoked: [],
|
|
checks: [buildCheck("run crashed", false, message)],
|
|
judgeScore: null,
|
|
judgeSummary: null,
|
|
error: message,
|
|
tokenUsage: null,
|
|
finalContextTokens: null,
|
|
};
|
|
if (surface) {
|
|
input.onProgress?.({
|
|
type: "attempt-finish",
|
|
surface,
|
|
caseId: input.evalCase.id,
|
|
caseNumber: input.caseIndex + 1,
|
|
totalCases: input.totalCases,
|
|
attempt,
|
|
runs: input.runs,
|
|
passed: false,
|
|
durationMs: failedAttempt.durationMs,
|
|
judgeScore: null,
|
|
error: message,
|
|
});
|
|
}
|
|
attempts.push(failedAttempt);
|
|
}
|
|
}
|
|
|
|
return attempts;
|
|
}
|
|
|
|
function buildCheck(name: string, passed: boolean, details?: string): BenchmarkCheck {
|
|
return details ? { name, passed, details } : { name, passed };
|
|
}
|
|
|
|
function uniqueStrings(values: string[]): string[] {
|
|
return [...new Set(values)];
|
|
}
|