mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-08-21 00:02:23 +00:00
test: require script and flow test tools
This commit is contained in:
@@ -32,19 +32,11 @@ type BenchmarkCompletedJob = CompletedJob & { type: 'CompletedJob' }
|
||||
const benchmarkWorkspaces = new Set<string>()
|
||||
const benchmarkWorkspaceRunnables = new Map<string, BenchmarkWorkspaceRunnables>()
|
||||
const benchmarkJobs = new Map<string, { workspace: string; job: BenchmarkCompletedJob }>()
|
||||
let lastBenchmarkFlowPreviewRequest:
|
||||
| {
|
||||
workspace: string
|
||||
requestBody: Record<string, unknown>
|
||||
memoryId?: string
|
||||
}
|
||||
| undefined
|
||||
|
||||
export function resetBenchmarkMockBackend(): void {
|
||||
benchmarkWorkspaces.clear()
|
||||
benchmarkWorkspaceRunnables.clear()
|
||||
benchmarkJobs.clear()
|
||||
lastBenchmarkFlowPreviewRequest = undefined
|
||||
}
|
||||
|
||||
export function registerBenchmarkWorkspace(workspace: string): void {
|
||||
@@ -235,46 +227,6 @@ export function runBenchmarkFlowByPath(input: {
|
||||
})
|
||||
}
|
||||
|
||||
export function runBenchmarkFlowPreview(input: {
|
||||
workspace: string
|
||||
requestBody: {
|
||||
path?: string
|
||||
value?: Record<string, unknown>
|
||||
args?: Record<string, unknown>
|
||||
}
|
||||
memoryId?: string
|
||||
}): string {
|
||||
lastBenchmarkFlowPreviewRequest = {
|
||||
workspace: input.workspace,
|
||||
requestBody: structuredClone(input.requestBody),
|
||||
memoryId: input.memoryId
|
||||
}
|
||||
|
||||
return createBenchmarkCompletedJob({
|
||||
workspace: input.workspace,
|
||||
jobKind: 'flowpreview',
|
||||
success: true,
|
||||
scriptPath: input.requestBody.path,
|
||||
args: input.requestBody.args,
|
||||
result: {
|
||||
path: input.requestBody.path,
|
||||
args: input.requestBody.args ?? {},
|
||||
mocked: true
|
||||
},
|
||||
logs: 'Mock benchmark flow preview completed successfully.'
|
||||
})
|
||||
}
|
||||
|
||||
export function getLastBenchmarkFlowPreviewRequest():
|
||||
| {
|
||||
workspace: string
|
||||
requestBody: Record<string, unknown>
|
||||
memoryId?: string
|
||||
}
|
||||
| undefined {
|
||||
return lastBenchmarkFlowPreviewRequest
|
||||
}
|
||||
|
||||
export function previewBenchmarkSchedule(input: {
|
||||
requestBody?: Record<string, unknown>
|
||||
}): Record<string, unknown> {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
import { expect, it, vi } from 'vitest'
|
||||
// @ts-ignore - Node.js fs/promises
|
||||
import { mkdir, writeFile } from 'fs/promises'
|
||||
// @ts-ignore - Node.js path
|
||||
@@ -44,7 +44,6 @@ vi.mock('$lib/gen', async () => {
|
||||
createBenchmarkSchedule,
|
||||
previewBenchmarkSchedule,
|
||||
runBenchmarkFlowByPath,
|
||||
runBenchmarkFlowPreview,
|
||||
runBenchmarkScriptPreview
|
||||
} = await import('./mockBackend')
|
||||
|
||||
@@ -169,22 +168,6 @@ vi.mock('$lib/gen', async () => {
|
||||
args: data.requestBody
|
||||
})
|
||||
: actual.JobService.runFlowByPath(data),
|
||||
runFlowPreview: async (data: {
|
||||
workspace: string
|
||||
memoryId?: string
|
||||
requestBody?: {
|
||||
path?: string
|
||||
value?: Record<string, unknown>
|
||||
args?: Record<string, unknown>
|
||||
}
|
||||
}) =>
|
||||
hasBenchmarkWorkspace(data.workspace)
|
||||
? runBenchmarkFlowPreview({
|
||||
workspace: data.workspace,
|
||||
requestBody: data.requestBody ?? {},
|
||||
memoryId: data.memoryId
|
||||
})
|
||||
: actual.JobService.runFlowPreview(data),
|
||||
getJob: async (data: { workspace: string; id: string }) => {
|
||||
if (hasBenchmarkWorkspace(data.workspace)) {
|
||||
const job = getBenchmarkCompletedJob(data.workspace, data.id)
|
||||
@@ -379,173 +362,6 @@ vi.mock('$lib/gen', async () => {
|
||||
}
|
||||
})
|
||||
|
||||
describe('global flow preview tools in ai_evals', () => {
|
||||
beforeEach(async () => {
|
||||
const { resetBenchmarkMockBackend } = await import('./mockBackend')
|
||||
const { __resetUserDraftForTesting } = await import('../../../frontend/src/lib/userDraft.svelte')
|
||||
resetBenchmarkMockBackend()
|
||||
__resetUserDraftForTesting()
|
||||
localStorage.removeItem('wm_dev_global_ai')
|
||||
})
|
||||
|
||||
it('does not pass the AI session id as a flow preview memory id', async () => {
|
||||
const {
|
||||
getLastBenchmarkFlowPreviewRequest,
|
||||
registerBenchmarkWorkspaceRunnables,
|
||||
resetBenchmarkMockBackend,
|
||||
unregisterBenchmarkWorkspace
|
||||
} = await import('./mockBackend')
|
||||
const { globalTools } = await import('../../../frontend/src/lib/components/copilot/chat/global/core')
|
||||
const { processToolCall } = await import('../../../frontend/src/lib/components/copilot/chat/shared')
|
||||
const workspace = 'ai-evals-global-flow-preview'
|
||||
|
||||
registerBenchmarkWorkspaceRunnables(workspace, {
|
||||
flows: [
|
||||
{
|
||||
path: 'f/evals/global/current_invoice_flow',
|
||||
summary: 'Current invoice flow',
|
||||
value: {
|
||||
modules: [{ id: 'calculate_total', value: { type: 'identity' } }]
|
||||
}
|
||||
}
|
||||
]
|
||||
})
|
||||
|
||||
try {
|
||||
const response = await processToolCall({
|
||||
tools: globalTools,
|
||||
toolCall: {
|
||||
id: 'tool-test-run-flow',
|
||||
type: 'function',
|
||||
function: {
|
||||
name: 'test_run_flow',
|
||||
arguments: JSON.stringify({
|
||||
path: 'f/evals/global/current_invoice_flow',
|
||||
args: { subtotal: 100 }
|
||||
})
|
||||
}
|
||||
},
|
||||
helpers: { sessionId: 'htc1xouxd96dcyo6ruqo39' },
|
||||
workspace,
|
||||
toolCallbacks: {
|
||||
setToolStatus: vi.fn(),
|
||||
shouldAutoAcceptToolConfirmations: () => true
|
||||
}
|
||||
})
|
||||
|
||||
expect(response.content).toContain('Result (SUCCESS)')
|
||||
expect(getLastBenchmarkFlowPreviewRequest()).toMatchObject({
|
||||
workspace,
|
||||
memoryId: undefined,
|
||||
requestBody: {
|
||||
path: 'f/evals/global/current_invoice_flow',
|
||||
args: { subtotal: 100 }
|
||||
}
|
||||
})
|
||||
} finally {
|
||||
unregisterBenchmarkWorkspace(workspace)
|
||||
resetBenchmarkMockBackend()
|
||||
}
|
||||
})
|
||||
|
||||
it('uses the active flow test hook with args only', async () => {
|
||||
const {
|
||||
createBenchmarkCompletedJob,
|
||||
registerBenchmarkWorkspace,
|
||||
unregisterBenchmarkWorkspace
|
||||
} = await import('./mockBackend')
|
||||
const { globalTools } = await import('../../../frontend/src/lib/components/copilot/chat/global/core')
|
||||
const { processToolCall } = await import('../../../frontend/src/lib/components/copilot/chat/shared')
|
||||
const { UserDraft } = await import('../../../frontend/src/lib/userDraft.svelte')
|
||||
const workspace = 'ai-evals-global-live-flow-preview'
|
||||
const path = 'f/evals/global/current_invoice_flow'
|
||||
const testActiveFlow = vi.fn(async (args?: Record<string, unknown>) =>
|
||||
createBenchmarkCompletedJob({
|
||||
workspace,
|
||||
jobKind: 'flowpreview',
|
||||
args,
|
||||
result: { path, args: args ?? {}, liveEditor: true },
|
||||
logs: 'Mock live editor flow preview completed successfully.'
|
||||
})
|
||||
)
|
||||
|
||||
registerBenchmarkWorkspace(workspace)
|
||||
UserDraft.setLiveEditorDraft({
|
||||
workspace,
|
||||
itemKind: 'flow',
|
||||
storagePath: '',
|
||||
effectivePath: path
|
||||
})
|
||||
|
||||
try {
|
||||
const response = await processToolCall({
|
||||
tools: globalTools,
|
||||
toolCall: {
|
||||
id: 'tool-live-flow-test',
|
||||
type: 'function',
|
||||
function: {
|
||||
name: 'test_run_flow',
|
||||
arguments: JSON.stringify({
|
||||
path,
|
||||
args: { subtotal: 200 }
|
||||
})
|
||||
}
|
||||
},
|
||||
helpers: {
|
||||
sessionId: 'htc1xouxd96dcyo6ruqo39',
|
||||
testActiveFlow
|
||||
},
|
||||
workspace,
|
||||
toolCallbacks: {
|
||||
setToolStatus: vi.fn(),
|
||||
shouldAutoAcceptToolConfirmations: () => true
|
||||
}
|
||||
})
|
||||
|
||||
expect(response.content).toContain('Result (SUCCESS)')
|
||||
expect(testActiveFlow.mock.calls).toEqual([[{ subtotal: 200 }]])
|
||||
} finally {
|
||||
UserDraft.clearLiveEditorDraft('flow', { workspace, storagePath: '' })
|
||||
unregisterBenchmarkWorkspace(workspace)
|
||||
}
|
||||
})
|
||||
|
||||
it('wires session global flow tests without a conversation id', async () => {
|
||||
const { AIChatManager, AIMode } = await import(
|
||||
'../../../frontend/src/lib/components/copilot/chat/AIChatManager.svelte'
|
||||
)
|
||||
const testFlow = vi.fn(async () => 'job-flow-preview')
|
||||
const manager = new AIChatManager()
|
||||
|
||||
localStorage.setItem('wm_dev_global_ai', '1')
|
||||
manager.isSessionChat = true
|
||||
manager.sessionId = 'htc1xouxd96dcyo6ruqo39'
|
||||
manager.setFlowHelpers({
|
||||
getFlowAndSelectedId: vi.fn(),
|
||||
getRootModules: vi.fn(),
|
||||
inlineScriptSession: { get: vi.fn(), set: vi.fn(), clear: vi.fn() },
|
||||
setSnapshot: vi.fn(),
|
||||
revertToSnapshot: vi.fn(),
|
||||
setCode: vi.fn(),
|
||||
setFlowJson: vi.fn(),
|
||||
getFlowInputsSchema: vi.fn(),
|
||||
updateExprsToSet: vi.fn(),
|
||||
acceptAllModuleActions: vi.fn(),
|
||||
rejectAllModuleActions: vi.fn(),
|
||||
hasPendingChanges: vi.fn(() => false),
|
||||
selectStep: vi.fn(),
|
||||
testFlow,
|
||||
getLintErrors: vi.fn()
|
||||
} as any)
|
||||
|
||||
manager.changeMode(AIMode.GLOBAL)
|
||||
const jobId = await manager.helpers.testActiveFlow({ subtotal: 300 })
|
||||
|
||||
expect(jobId).toBe('job-flow-preview')
|
||||
expect(testFlow.mock.calls).toEqual([[{ subtotal: 300 }]])
|
||||
})
|
||||
})
|
||||
|
||||
const benchmarkOutputPath = process.env.WMILL_FRONTEND_AI_EVAL_OUTPUT_PATH
|
||||
const benchmarkIt = benchmarkOutputPath ? it : it.skip
|
||||
|
||||
|
||||
@@ -8,6 +8,9 @@
|
||||
args:
|
||||
a: 4
|
||||
b: 5
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_flow
|
||||
judgeChecklist:
|
||||
- "the flow takes `a` and `b` as inputs"
|
||||
- "the main step is named `sum_numbers`"
|
||||
@@ -25,6 +28,9 @@
|
||||
args:
|
||||
a: 2
|
||||
b: 3
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_flow
|
||||
judgeChecklist:
|
||||
- "the flow takes `a` and `b` as inputs"
|
||||
- "the main step is named `sum_numbers`"
|
||||
@@ -42,6 +48,9 @@
|
||||
args:
|
||||
a: 7
|
||||
b: 8
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_flow
|
||||
judgeChecklist:
|
||||
- "the parent flow takes `a` and `b` as inputs"
|
||||
- "the main step is named `call_add_numbers`"
|
||||
@@ -426,6 +435,7 @@
|
||||
- return_schedule_status
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_flow
|
||||
- create_schedule
|
||||
toolCallArgs:
|
||||
- tool: create_schedule
|
||||
@@ -453,6 +463,7 @@
|
||||
- webhook_response
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_flow
|
||||
- create_trigger
|
||||
toolCallArgs:
|
||||
- tool: create_trigger
|
||||
|
||||
@@ -5,6 +5,9 @@
|
||||
Keep it simple and do not add external dependencies.
|
||||
initial: ai_evals/fixtures/frontend/script/initial/test1_empty_bun.json
|
||||
expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_script
|
||||
judgeChecklist:
|
||||
- uses the existing `name` input
|
||||
- returns a plain greeting string
|
||||
@@ -20,6 +23,7 @@
|
||||
expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_script
|
||||
- create_schedule
|
||||
toolCallArgs:
|
||||
- tool: create_schedule
|
||||
@@ -44,6 +48,7 @@
|
||||
expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json
|
||||
toolExpect:
|
||||
requiredToolsUsed:
|
||||
- test_run_script
|
||||
- create_trigger
|
||||
toolCallArgs:
|
||||
- tool: create_trigger
|
||||
|
||||
@@ -14,6 +14,21 @@ describe("loadCases", () => {
|
||||
},
|
||||
},
|
||||
});
|
||||
expect(caseEntry?.toolExpect).toEqual({
|
||||
requiredToolsUsed: ["test_run_flow"],
|
||||
});
|
||||
});
|
||||
|
||||
it("loads script and flow test tool expectations", async () => {
|
||||
const scriptCases = await loadCases("script");
|
||||
const flowCases = await loadCases("flow");
|
||||
|
||||
expect(scriptCases.find((entry) => entry.id === "script-test1-greet-user")?.toolExpect).toEqual({
|
||||
requiredToolsUsed: ["test_run_script"],
|
||||
});
|
||||
expect(flowCases.find((entry) => entry.id === "flow-test0-sum-two-numbers")?.toolExpect).toEqual({
|
||||
requiredToolsUsed: ["test_run_flow"],
|
||||
});
|
||||
});
|
||||
|
||||
it("loads the workspace-flow preference benchmark case", async () => {
|
||||
@@ -238,7 +253,7 @@ describe("loadCases", () => {
|
||||
);
|
||||
|
||||
expect(caseEntry?.toolExpect).toEqual({
|
||||
requiredToolsUsed: ["create_schedule"],
|
||||
requiredToolsUsed: ["test_run_script", "create_schedule"],
|
||||
toolCallArgs: [
|
||||
{
|
||||
tool: "create_schedule",
|
||||
|
||||
Reference in New Issue
Block a user