From ade5159e90c2280d3690772fb226ca9968cd955c Mon Sep 17 00:00:00 2001 From: centdix Date: Tue, 31 Mar 2026 18:19:22 +0200 Subject: [PATCH] test: add cli artifact eval runner Co-Authored-By: Claude Opus 4.5 --- ai_evals/cases/cli/script.json | 22 +++ cli/test-skills/README.md | 32 +++- cli/test-skills/package.json | 3 +- cli/test-skills/src/artifact-eval.test.ts | 54 +++++++ cli/test-skills/src/artifact-eval.ts | 186 ++++++++++++++++++++++ 5 files changed, 290 insertions(+), 7 deletions(-) create mode 100644 ai_evals/cases/cli/script.json create mode 100644 cli/test-skills/src/artifact-eval.test.ts create mode 100644 cli/test-skills/src/artifact-eval.ts diff --git a/ai_evals/cases/cli/script.json b/ai_evals/cases/cli/script.json new file mode 100644 index 0000000000..068a60b46f --- /dev/null +++ b/ai_evals/cases/cli/script.json @@ -0,0 +1,22 @@ +[ + { + "id": "bun-hello-script", + "description": "Create a minimal Bun script in a fresh CLI workspace.", + "prompt": "This is a benchmark harness. Create exactly one Windmill Bun/TypeScript script at {{workspace_root}}/f/evals/hello.ts. The script must export async function main(name: string) and return an object { greeting: `Hello, ${name}!` }. Keep it minimal. Do not create other scripts. Do not run any CLI commands. After writing the file, tell me exactly which wmill commands I should run next.", + "maxTurns": 6, + "expectedSkill": "write-script-bun", + "expectedOutputSubstrings": [ + "wmill script generate-metadata", + "wmill sync push" + ], + "expectedFiles": [ + { + "path": "f/evals/hello.ts", + "mustContain": [ + "export async function main(name: string)", + "return { greeting: `Hello, ${name}!` };" + ] + } + ] + } +] diff --git a/cli/test-skills/README.md b/cli/test-skills/README.md index 04ee3cc1ab..6b0221e54d 100644 --- a/cli/test-skills/README.md +++ b/cli/test-skills/README.md @@ -1,10 +1,18 @@ -# Windmill Skill Invocation Tests +# Windmill CLI Prompt Tests -Test suite for verifying that Claude Code correctly invokes Windmill auto-generated skills based on user prompts. +Test suite for verifying how Claude Code behaves with Windmill auto-generated +skills. It currently contains both skill-invocation smoke tests and the first +artifact-evaluation benchmark. ## Overview -This framework tests skill invocation behavior by sending prompts through the Claude Agent SDK and verifying that the expected skills are invoked. The suite mirrors the repo's generated Windmill skills into its test workspace before each run. +This framework sends prompts through the Claude Agent SDK using the repo's +generated Windmill skills. + +It currently supports: + +- skill-invocation smoke tests +- CLI artifact evaluation in an isolated temp workspace ## Prerequisites @@ -71,6 +79,11 @@ Run only skill invocation tests: bun test:skills ``` +Run the first artifact-evaluation benchmark: +```bash +bun test:artifact +``` + ## Test Utilities The `src/test-utils.ts` module provides: @@ -81,10 +94,17 @@ The `src/test-utils.ts` module provides: - `getToolInputs(result, toolName)` - Gets all inputs for a specific tool - `getTestSkillsDir()` - Returns the test-skills directory path +The `src/artifact-eval.ts` module provides: + +- temp-workspace creation with generated skills +- prompt rendering with workspace-root placeholders +- file-based artifact scoring for benchmark cases + ## Notes - Tests have extended timeouts (120 seconds) due to API latency - Tests run against the actual Claude API, so they consume API credits -- Tests verify skill invocation, not skill execution -- The working directory for tests is `test-folder/` -- The suite refreshes `test-folder/.claude/skills/` from `system_prompts/auto-generated/skills/` before running +- Skill-invocation smoke tests still use `test-folder/` +- Artifact evals use isolated temp workspaces under `/tmp` +- The first artifact benchmark uses an explicit absolute target path so file + outputs are scoreable even when Claude executes inside a skill context diff --git a/cli/test-skills/package.json b/cli/test-skills/package.json index 1d839ab23c..d95b6c29b8 100644 --- a/cli/test-skills/package.json +++ b/cli/test-skills/package.json @@ -4,7 +4,8 @@ "type": "module", "scripts": { "test": "bun test", - "test:skills": "bun test src/skill-invocation.test.ts" + "test:skills": "bun test src/skill-invocation.test.ts", + "test:artifact": "bun test src/artifact-eval.test.ts" }, "dependencies": { "@anthropic-ai/claude-agent-sdk": "^0.2.25" diff --git a/cli/test-skills/src/artifact-eval.test.ts b/cli/test-skills/src/artifact-eval.test.ts new file mode 100644 index 0000000000..c06eb956cb --- /dev/null +++ b/cli/test-skills/src/artifact-eval.test.ts @@ -0,0 +1,54 @@ +import { beforeAll, describe, expect, test } from "bun:test"; +import { + cleanupWorkspace, + loadCliArtifactEvalCases, + runCliArtifactEvalCase, + shouldKeepWorkspace +} from "./artifact-eval"; + +const evalCases = await loadCliArtifactEvalCases(); + +describe("Windmill CLI Artifact Evals", () => { + beforeAll(() => { + if (!process.env.ANTHROPIC_API_KEY) { + throw new Error("ANTHROPIC_API_KEY environment variable is required"); + } + }); + + for (const evalCase of evalCases) { + test( + evalCase.id, + async () => { + const result = await runCliArtifactEvalCase(evalCase); + + try { + console.log( + JSON.stringify( + { + caseId: evalCase.id, + workspaceDir: result.workspaceDir, + passed: result.passed, + checks: result.checks, + skillsInvoked: result.run.skillsInvoked, + toolsUsed: result.run.toolsUsed.map((tool) => tool.tool), + files: result.expectedFiles.map((file) => ({ + path: file.path, + exists: file.exists + })) + }, + null, + 2 + ) + ); + + expect(result.passed).toBe(true); + } finally { + if (!shouldKeepWorkspace()) { + await cleanupWorkspace(result.workspaceDir); + } + } + }, + { timeout: 180000 } + ); + } +}); diff --git a/cli/test-skills/src/artifact-eval.ts b/cli/test-skills/src/artifact-eval.ts new file mode 100644 index 0000000000..b020fe06e0 --- /dev/null +++ b/cli/test-skills/src/artifact-eval.ts @@ -0,0 +1,186 @@ +import { cp, mkdtemp, mkdir, readFile, rm, writeFile } from "fs/promises"; +import { existsSync } from "fs"; +import { tmpdir } from "os"; +import { dirname, join } from "path"; +import { fileURLToPath } from "url"; +import { runPromptAndCapture, type TestResult, wasSkillInvoked, getGeneratedSkillsSource } from "./test-utils"; + +export interface ExpectedFile { + path: string; + mustContain?: string[]; + mustNotContain?: string[]; +} + +export interface CliArtifactEvalCase { + id: string; + description?: string; + prompt: string; + maxTurns?: number; + expectedSkill?: string; + expectedOutputSubstrings?: string[]; + expectedFiles: ExpectedFile[]; +} + +export interface ArtifactCheck { + name: string; + passed: boolean; + details?: string; +} + +export interface FileArtifactResult { + path: string; + exists: boolean; + content?: string; +} + +export interface CliArtifactEvalResult { + workspaceDir: string; + renderedPrompt: string; + run: TestResult; + checks: ArtifactCheck[]; + expectedFiles: FileArtifactResult[]; + passed: boolean; +} + +const CASES_FILE = fileURLToPath( + new URL("../../../ai_evals/cases/cli/script.json", import.meta.url) +); + +export async function loadCliArtifactEvalCases(): Promise { + const raw = await readFile(CASES_FILE, "utf8"); + const parsed = JSON.parse(raw) as CliArtifactEvalCase[]; + + if (!Array.isArray(parsed) || parsed.length === 0) { + throw new Error(`No CLI artifact eval cases found in ${CASES_FILE}`); + } + + return parsed; +} + +export async function runCliArtifactEvalCase( + evalCase: CliArtifactEvalCase +): Promise { + const workspaceDir = await createIsolatedWorkspace(evalCase.id); + + try { + const renderedPrompt = renderPrompt(evalCase.prompt, workspaceDir); + const run = await runPromptAndCapture( + renderedPrompt, + workspaceDir, + evalCase.maxTurns ?? 6 + ); + const fileResults = await collectExpectedFiles(workspaceDir, evalCase.expectedFiles); + const checks = buildChecks(evalCase, run, fileResults); + + return { + workspaceDir, + renderedPrompt, + run, + checks, + expectedFiles: fileResults, + passed: checks.every((check) => check.passed) + }; + } catch (error) { + if (!shouldKeepWorkspace()) { + await cleanupWorkspace(workspaceDir); + } + throw error; + } +} + +export async function cleanupWorkspace(workspaceDir: string): Promise { + await rm(workspaceDir, { recursive: true, force: true }); +} + +export function shouldKeepWorkspace(): boolean { + return process.env.WMILL_CLI_EVAL_KEEP_WORKSPACE === "1"; +} + +async function createIsolatedWorkspace(caseId: string): Promise { + const workspaceDir = await mkdtemp(join(tmpdir(), `wmill-cli-artifact-${caseId}-`)); + const skillsDir = join(workspaceDir, ".claude", "skills"); + const generatedSkillsSource = getGeneratedSkillsSource(); + + await mkdir(dirname(skillsDir), { recursive: true }); + await cp(generatedSkillsSource, skillsDir, { recursive: true }); + await writeFile(join(workspaceDir, "rt.d.ts"), "export namespace RT {}\n", "utf8"); + + return workspaceDir; +} + +function renderPrompt(prompt: string, workspaceDir: string): string { + return prompt.replaceAll("{{workspace_root}}", workspaceDir); +} + +async function collectExpectedFiles( + workspaceDir: string, + expectedFiles: ExpectedFile[] +): Promise { + const results: FileArtifactResult[] = []; + + for (const expectedFile of expectedFiles) { + const absolutePath = join(workspaceDir, expectedFile.path); + const exists = existsSync(absolutePath); + if (!exists) { + results.push({ path: expectedFile.path, exists: false }); + continue; + } + + results.push({ + path: expectedFile.path, + exists: true, + content: await readFile(absolutePath, "utf8") + }); + } + + return results; +} + +function buildChecks( + evalCase: CliArtifactEvalCase, + run: TestResult, + fileResults: FileArtifactResult[] +): ArtifactCheck[] { + const checks: ArtifactCheck[] = []; + + if (evalCase.expectedSkill) { + checks.push({ + name: `invokes ${evalCase.expectedSkill}`, + passed: wasSkillInvoked(run, evalCase.expectedSkill), + details: `skills invoked: ${run.skillsInvoked.join(", ")}` + }); + } + + for (const expectedOutput of evalCase.expectedOutputSubstrings ?? []) { + checks.push({ + name: `mentions '${expectedOutput}' in assistant output`, + passed: run.output.includes(expectedOutput) + }); + } + + for (const expectedFile of evalCase.expectedFiles) { + const fileResult = fileResults.find((entry) => entry.path === expectedFile.path); + const content = fileResult?.content ?? ""; + + checks.push({ + name: `creates ${expectedFile.path}`, + passed: Boolean(fileResult?.exists) + }); + + for (const requiredSnippet of expectedFile.mustContain ?? []) { + checks.push({ + name: `${expectedFile.path} contains '${requiredSnippet}'`, + passed: content.includes(requiredSnippet) + }); + } + + for (const forbiddenSnippet of expectedFile.mustNotContain ?? []) { + checks.push({ + name: `${expectedFile.path} avoids '${forbiddenSnippet}'`, + passed: !content.includes(forbiddenSnippet) + }); + } + } + + return checks; +}