diff --git a/ai_evals/adapters/frontend/mockBackend.ts b/ai_evals/adapters/frontend/mockBackend.ts index c8212326a9..be4868d8d4 100644 --- a/ai_evals/adapters/frontend/mockBackend.ts +++ b/ai_evals/adapters/frontend/mockBackend.ts @@ -32,19 +32,11 @@ type BenchmarkCompletedJob = CompletedJob & { type: 'CompletedJob' } const benchmarkWorkspaces = new Set() const benchmarkWorkspaceRunnables = new Map() const benchmarkJobs = new Map() -let lastBenchmarkFlowPreviewRequest: - | { - workspace: string - requestBody: Record - memoryId?: string - } - | undefined export function resetBenchmarkMockBackend(): void { benchmarkWorkspaces.clear() benchmarkWorkspaceRunnables.clear() benchmarkJobs.clear() - lastBenchmarkFlowPreviewRequest = undefined } export function registerBenchmarkWorkspace(workspace: string): void { @@ -235,46 +227,6 @@ export function runBenchmarkFlowByPath(input: { }) } -export function runBenchmarkFlowPreview(input: { - workspace: string - requestBody: { - path?: string - value?: Record - args?: Record - } - memoryId?: string -}): string { - lastBenchmarkFlowPreviewRequest = { - workspace: input.workspace, - requestBody: structuredClone(input.requestBody), - memoryId: input.memoryId - } - - return createBenchmarkCompletedJob({ - workspace: input.workspace, - jobKind: 'flowpreview', - success: true, - scriptPath: input.requestBody.path, - args: input.requestBody.args, - result: { - path: input.requestBody.path, - args: input.requestBody.args ?? {}, - mocked: true - }, - logs: 'Mock benchmark flow preview completed successfully.' - }) -} - -export function getLastBenchmarkFlowPreviewRequest(): - | { - workspace: string - requestBody: Record - memoryId?: string - } - | undefined { - return lastBenchmarkFlowPreviewRequest -} - export function previewBenchmarkSchedule(input: { requestBody?: Record }): Record { diff --git a/ai_evals/adapters/frontend/vitestAdapter.test.ts b/ai_evals/adapters/frontend/vitestAdapter.test.ts index 0e8457717c..92c43aab06 100644 --- a/ai_evals/adapters/frontend/vitestAdapter.test.ts +++ b/ai_evals/adapters/frontend/vitestAdapter.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { expect, it, vi } from 'vitest' // @ts-ignore - Node.js fs/promises import { mkdir, writeFile } from 'fs/promises' // @ts-ignore - Node.js path @@ -44,7 +44,6 @@ vi.mock('$lib/gen', async () => { createBenchmarkSchedule, previewBenchmarkSchedule, runBenchmarkFlowByPath, - runBenchmarkFlowPreview, runBenchmarkScriptPreview } = await import('./mockBackend') @@ -169,22 +168,6 @@ vi.mock('$lib/gen', async () => { args: data.requestBody }) : actual.JobService.runFlowByPath(data), - runFlowPreview: async (data: { - workspace: string - memoryId?: string - requestBody?: { - path?: string - value?: Record - args?: Record - } - }) => - hasBenchmarkWorkspace(data.workspace) - ? runBenchmarkFlowPreview({ - workspace: data.workspace, - requestBody: data.requestBody ?? {}, - memoryId: data.memoryId - }) - : actual.JobService.runFlowPreview(data), getJob: async (data: { workspace: string; id: string }) => { if (hasBenchmarkWorkspace(data.workspace)) { const job = getBenchmarkCompletedJob(data.workspace, data.id) @@ -379,173 +362,6 @@ vi.mock('$lib/gen', async () => { } }) -describe('global flow preview tools in ai_evals', () => { - beforeEach(async () => { - const { resetBenchmarkMockBackend } = await import('./mockBackend') - const { __resetUserDraftForTesting } = await import('../../../frontend/src/lib/userDraft.svelte') - resetBenchmarkMockBackend() - __resetUserDraftForTesting() - localStorage.removeItem('wm_dev_global_ai') - }) - - it('does not pass the AI session id as a flow preview memory id', async () => { - const { - getLastBenchmarkFlowPreviewRequest, - registerBenchmarkWorkspaceRunnables, - resetBenchmarkMockBackend, - unregisterBenchmarkWorkspace - } = await import('./mockBackend') - const { globalTools } = await import('../../../frontend/src/lib/components/copilot/chat/global/core') - const { processToolCall } = await import('../../../frontend/src/lib/components/copilot/chat/shared') - const workspace = 'ai-evals-global-flow-preview' - - registerBenchmarkWorkspaceRunnables(workspace, { - flows: [ - { - path: 'f/evals/global/current_invoice_flow', - summary: 'Current invoice flow', - value: { - modules: [{ id: 'calculate_total', value: { type: 'identity' } }] - } - } - ] - }) - - try { - const response = await processToolCall({ - tools: globalTools, - toolCall: { - id: 'tool-test-run-flow', - type: 'function', - function: { - name: 'test_run_flow', - arguments: JSON.stringify({ - path: 'f/evals/global/current_invoice_flow', - args: { subtotal: 100 } - }) - } - }, - helpers: { sessionId: 'htc1xouxd96dcyo6ruqo39' }, - workspace, - toolCallbacks: { - setToolStatus: vi.fn(), - shouldAutoAcceptToolConfirmations: () => true - } - }) - - expect(response.content).toContain('Result (SUCCESS)') - expect(getLastBenchmarkFlowPreviewRequest()).toMatchObject({ - workspace, - memoryId: undefined, - requestBody: { - path: 'f/evals/global/current_invoice_flow', - args: { subtotal: 100 } - } - }) - } finally { - unregisterBenchmarkWorkspace(workspace) - resetBenchmarkMockBackend() - } - }) - - it('uses the active flow test hook with args only', async () => { - const { - createBenchmarkCompletedJob, - registerBenchmarkWorkspace, - unregisterBenchmarkWorkspace - } = await import('./mockBackend') - const { globalTools } = await import('../../../frontend/src/lib/components/copilot/chat/global/core') - const { processToolCall } = await import('../../../frontend/src/lib/components/copilot/chat/shared') - const { UserDraft } = await import('../../../frontend/src/lib/userDraft.svelte') - const workspace = 'ai-evals-global-live-flow-preview' - const path = 'f/evals/global/current_invoice_flow' - const testActiveFlow = vi.fn(async (args?: Record) => - createBenchmarkCompletedJob({ - workspace, - jobKind: 'flowpreview', - args, - result: { path, args: args ?? {}, liveEditor: true }, - logs: 'Mock live editor flow preview completed successfully.' - }) - ) - - registerBenchmarkWorkspace(workspace) - UserDraft.setLiveEditorDraft({ - workspace, - itemKind: 'flow', - storagePath: '', - effectivePath: path - }) - - try { - const response = await processToolCall({ - tools: globalTools, - toolCall: { - id: 'tool-live-flow-test', - type: 'function', - function: { - name: 'test_run_flow', - arguments: JSON.stringify({ - path, - args: { subtotal: 200 } - }) - } - }, - helpers: { - sessionId: 'htc1xouxd96dcyo6ruqo39', - testActiveFlow - }, - workspace, - toolCallbacks: { - setToolStatus: vi.fn(), - shouldAutoAcceptToolConfirmations: () => true - } - }) - - expect(response.content).toContain('Result (SUCCESS)') - expect(testActiveFlow.mock.calls).toEqual([[{ subtotal: 200 }]]) - } finally { - UserDraft.clearLiveEditorDraft('flow', { workspace, storagePath: '' }) - unregisterBenchmarkWorkspace(workspace) - } - }) - - it('wires session global flow tests without a conversation id', async () => { - const { AIChatManager, AIMode } = await import( - '../../../frontend/src/lib/components/copilot/chat/AIChatManager.svelte' - ) - const testFlow = vi.fn(async () => 'job-flow-preview') - const manager = new AIChatManager() - - localStorage.setItem('wm_dev_global_ai', '1') - manager.isSessionChat = true - manager.sessionId = 'htc1xouxd96dcyo6ruqo39' - manager.setFlowHelpers({ - getFlowAndSelectedId: vi.fn(), - getRootModules: vi.fn(), - inlineScriptSession: { get: vi.fn(), set: vi.fn(), clear: vi.fn() }, - setSnapshot: vi.fn(), - revertToSnapshot: vi.fn(), - setCode: vi.fn(), - setFlowJson: vi.fn(), - getFlowInputsSchema: vi.fn(), - updateExprsToSet: vi.fn(), - acceptAllModuleActions: vi.fn(), - rejectAllModuleActions: vi.fn(), - hasPendingChanges: vi.fn(() => false), - selectStep: vi.fn(), - testFlow, - getLintErrors: vi.fn() - } as any) - - manager.changeMode(AIMode.GLOBAL) - const jobId = await manager.helpers.testActiveFlow({ subtotal: 300 }) - - expect(jobId).toBe('job-flow-preview') - expect(testFlow.mock.calls).toEqual([[{ subtotal: 300 }]]) - }) -}) - const benchmarkOutputPath = process.env.WMILL_FRONTEND_AI_EVAL_OUTPUT_PATH const benchmarkIt = benchmarkOutputPath ? it : it.skip diff --git a/ai_evals/cases/flow.yaml b/ai_evals/cases/flow.yaml index af7added38..5f4abafc48 100644 --- a/ai_evals/cases/flow.yaml +++ b/ai_evals/cases/flow.yaml @@ -8,6 +8,9 @@ args: a: 4 b: 5 + toolExpect: + requiredToolsUsed: + - test_run_flow judgeChecklist: - "the flow takes `a` and `b` as inputs" - "the main step is named `sum_numbers`" @@ -25,6 +28,9 @@ args: a: 2 b: 3 + toolExpect: + requiredToolsUsed: + - test_run_flow judgeChecklist: - "the flow takes `a` and `b` as inputs" - "the main step is named `sum_numbers`" @@ -42,6 +48,9 @@ args: a: 7 b: 8 + toolExpect: + requiredToolsUsed: + - test_run_flow judgeChecklist: - "the parent flow takes `a` and `b` as inputs" - "the main step is named `call_add_numbers`" @@ -426,6 +435,7 @@ - return_schedule_status toolExpect: requiredToolsUsed: + - test_run_flow - create_schedule toolCallArgs: - tool: create_schedule @@ -453,6 +463,7 @@ - webhook_response toolExpect: requiredToolsUsed: + - test_run_flow - create_trigger toolCallArgs: - tool: create_trigger diff --git a/ai_evals/cases/script.yaml b/ai_evals/cases/script.yaml index feae74dcda..f0910e939a 100644 --- a/ai_evals/cases/script.yaml +++ b/ai_evals/cases/script.yaml @@ -5,6 +5,9 @@ Keep it simple and do not add external dependencies. initial: ai_evals/fixtures/frontend/script/initial/test1_empty_bun.json expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json + toolExpect: + requiredToolsUsed: + - test_run_script judgeChecklist: - uses the existing `name` input - returns a plain greeting string @@ -20,6 +23,7 @@ expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json toolExpect: requiredToolsUsed: + - test_run_script - create_schedule toolCallArgs: - tool: create_schedule @@ -44,6 +48,7 @@ expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json toolExpect: requiredToolsUsed: + - test_run_script - create_trigger toolCallArgs: - tool: create_trigger diff --git a/ai_evals/core/cases.test.ts b/ai_evals/core/cases.test.ts index 526ea7c70f..05e2f1527b 100644 --- a/ai_evals/core/cases.test.ts +++ b/ai_evals/core/cases.test.ts @@ -14,6 +14,21 @@ describe("loadCases", () => { }, }, }); + expect(caseEntry?.toolExpect).toEqual({ + requiredToolsUsed: ["test_run_flow"], + }); + }); + + it("loads script and flow test tool expectations", async () => { + const scriptCases = await loadCases("script"); + const flowCases = await loadCases("flow"); + + expect(scriptCases.find((entry) => entry.id === "script-test1-greet-user")?.toolExpect).toEqual({ + requiredToolsUsed: ["test_run_script"], + }); + expect(flowCases.find((entry) => entry.id === "flow-test0-sum-two-numbers")?.toolExpect).toEqual({ + requiredToolsUsed: ["test_run_flow"], + }); }); it("loads the workspace-flow preference benchmark case", async () => { @@ -238,7 +253,7 @@ describe("loadCases", () => { ); expect(caseEntry?.toolExpect).toEqual({ - requiredToolsUsed: ["create_schedule"], + requiredToolsUsed: ["test_run_script", "create_schedule"], toolCallArgs: [ { tool: "create_schedule",