mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-08-20 00:02:19 +00:00
b0ddcf31e4
* ci: add path-gated AI agent integration tests workflow Runs integration_tests/ai_agent_tests against real LLM providers (Anthropic/OpenAI/Google) only when AI-agent backend code or the tests change, since runs make paid LLM calls. Adds a conftest fixture that skips provider-parametrized cases whose API keys are absent, so CI exercises only the providers it has secrets for. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: add path-gated ai_evals global-mode smoke workflow Runs the global AI chat eval (global-test1) across one cheap model per provider (Anthropic/OpenAI/Google/DeepSeek) only when the eval harness or copilot chat code change, since runs make paid LLM calls. Builds Windmill CE from source as the AI proxy; global tools/drafts run in the Vitest bridge. Gates on the deterministic draft pipeline (run succeeded + produced a draft + used write_script), not the variable LLM judge score. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: run AI smokes on PR ready-for-review instead of every push Switch the pull_request trigger from `synchronize` (every commit) to `ready_for_review`, with a job guard skipping draft PRs, so the paid LLM runs only fire when a PR is marked ready to merge (plus push-to-main and manual dispatch). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(ai_evals): lazily load cli mode so non-cli evals skip the cli toolchain The entrypoint eagerly imported modes/cli, which pulls the wmill CLI guidance modules and their JSR deps (@cliffy/*). Global/flow/script/app runs then crashed with "Cannot find module '@cliffy/ansi/colors'" when the cli workspace deps were not installed. Import createCliModeRunner dynamically inside runCliBenchmark instead. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * test(ai_agent): raise low max_completion_tokens to OpenAI's 16 minimum OpenAI's /v1/responses rejects max_output_tokens < 16 with a 400, failing test_low_max_tokens for openai. 16 still exercises a truncated response. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: run ai_evals workflow on Node 22 for the frontend undici 8.x dep The Vitest bridge loads frontend/node_modules/undici@8.x, which requires Node >=22.19; Node 20 failed with "webidl.util.markAsUncloneable is not a function" when loading vitest.config.ts. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(ai_evals): run frontend evals autonomously + give global-test1 more turns Frontend evals (flow/script/app/global) ran the production chat prompt, which assumes an interactive human — so cheaper models burned their turn budget asking for confirmation, waiting for approval, or presenting a plan, sometimes hitting maxTurns without producing a draft. Append a shared autonomy note in baseEvalRunner (the path all frontend modes share, mirroring cli mode): act directly on clear requests; only ask on genuinely ambiguous ones (preserving the askUserQuestion cases). Also raise global-test1's maxTurns 8 -> 10 so a model that over-explores still converges. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci(ai_evals): watch draft/prompt deps outside copilot/ The global eval runs production frontend code in-process, so the smoke's behavior depends on files outside frontend/src/lib/components/copilot/**: the draft model (userDraft.svelte.ts, userDraftDbSyncer.svelte.ts), script inference (infer.ts), and the chat system prompts ($system_prompts -> system_prompts/auto-generated). Add them to both push and PR path filters so a change there actually triggers the smoke that gates on draft production. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix: skip direct provider tests without credentials * feat: add ai evals skip judge flag * fix: simplify ai evals ci gate * fix: simplify ai evals smoke gate * fix: handle ai eval workflow triggers --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
103 lines
2.8 KiB
TypeScript
103 lines
2.8 KiB
TypeScript
import { describe, expect, it } from "bun:test";
|
|
import { runSuite } from "./runSuite";
|
|
import type { ModeRunner } from "./types";
|
|
|
|
const modeRunner: ModeRunner<undefined, undefined, { ok: boolean }> = {
|
|
mode: "global",
|
|
concurrency: 1,
|
|
loadInitial: async () => undefined,
|
|
loadExpected: async () => undefined,
|
|
run: async () => ({
|
|
success: true,
|
|
actual: { ok: true },
|
|
assistantMessageCount: 1,
|
|
toolCallCount: 0,
|
|
toolsUsed: [],
|
|
skillsInvoked: [],
|
|
tokenUsage: null,
|
|
}),
|
|
validate: () => [],
|
|
};
|
|
|
|
describe("runSuite", () => {
|
|
it("skips judge checks when the run disables judge scoring", async () => {
|
|
const [caseResult] = await runSuite({
|
|
modeRunner,
|
|
cases: [
|
|
{
|
|
id: "case-1",
|
|
prompt: "Create a draft script",
|
|
judgeChecklist: ["the output satisfies the prompt"],
|
|
},
|
|
],
|
|
runs: 1,
|
|
runModel: "model-under-test",
|
|
judgeModel: null,
|
|
});
|
|
|
|
const [attempt] = caseResult.attempts;
|
|
expect(attempt.passed).toBe(true);
|
|
expect(attempt.judgeScore).toBeNull();
|
|
expect(attempt.judgeSummary).toBeNull();
|
|
expect(attempt.checks.map((check) => check.name)).toEqual([
|
|
"run succeeded",
|
|
]);
|
|
});
|
|
|
|
it("only requires run success when execution-only is enabled", async () => {
|
|
let loadExpectedCalls = 0;
|
|
let validateCalls = 0;
|
|
let backendValidateCalls = 0;
|
|
|
|
const executionOnlyRunner: ModeRunner<
|
|
undefined,
|
|
undefined,
|
|
{ ok: boolean }
|
|
> = {
|
|
...modeRunner,
|
|
loadExpected: async () => {
|
|
loadExpectedCalls++;
|
|
return undefined;
|
|
},
|
|
validate: () => {
|
|
validateCalls++;
|
|
return [{ name: "validator failed", passed: false }];
|
|
},
|
|
backendValidate: async () => {
|
|
backendValidateCalls++;
|
|
return {
|
|
checks: [{ name: "backend validation failed", passed: false }],
|
|
};
|
|
},
|
|
};
|
|
|
|
const [caseResult] = await runSuite({
|
|
modeRunner: executionOnlyRunner,
|
|
cases: [
|
|
{
|
|
id: "case-1",
|
|
prompt: "Create a draft script",
|
|
expectedPath: "fixtures/expected.json",
|
|
toolExpect: { requiredToolsUsed: ["write_script"] },
|
|
judgeChecklist: ["the output satisfies the prompt"],
|
|
},
|
|
],
|
|
runs: 1,
|
|
runModel: "model-under-test",
|
|
judgeModel: "judge-model",
|
|
executionOnly: true,
|
|
});
|
|
|
|
const [attempt] = caseResult.attempts;
|
|
expect(attempt.passed).toBe(true);
|
|
expect(attempt.judgeScore).toBeNull();
|
|
expect(attempt.judgeSummary).toBeNull();
|
|
expect(attempt.checks.map((check) => check.name)).toEqual([
|
|
"run succeeded",
|
|
]);
|
|
expect(loadExpectedCalls).toBe(0);
|
|
expect(validateCalls).toBe(0);
|
|
expect(backendValidateCalls).toBe(0);
|
|
});
|
|
});
|