mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-08-22 00:01:34 +00:00
refactor: unify eval pipeline with production chat code path (#8504)
* refactor: unify eval pipeline with production chat code path Extract a shared headless runChatLoop() that both AIChatManager (production) and the eval runner use, with injectable SDK clients. Drop OpenRouter — evals now use direct provider APIs (OpenAI SDK, Anthropic SDK) with streaming, matching production behavior. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> * fix: re-read tools/helpers/systemMessage/model on each loop iteration The old chatRequest() re-read this.tools, this.helpers, this.systemMessage, and getCurrentModel() on every iteration. This matters because changeModeTool (Navigator → Script/Flow) reassigns all of these mid-loop. Use JS getters in the config object so runChatLoop picks up changes each iteration. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -23,8 +23,7 @@ import {
|
||||
} from './shared'
|
||||
import type {
|
||||
ChatCompletionMessageParam,
|
||||
ChatCompletionSystemMessageParam,
|
||||
ChatCompletionUserMessageParam
|
||||
ChatCompletionSystemMessageParam
|
||||
} from 'openai/resources/chat/completions.mjs'
|
||||
import {
|
||||
prepareInlineChatSystemPrompt,
|
||||
@@ -37,7 +36,7 @@ import { loadApiTools } from './api/apiTools'
|
||||
import { prepareScriptUserMessage } from './script/core'
|
||||
import { prepareNavigatorUserMessage } from './navigator/core'
|
||||
import { sendUserToast } from '$lib/toast'
|
||||
import { getCompletion, getModelContextWindow, parseOpenAICompletion } from '../lib'
|
||||
import { getModelContextWindow, workspaceAIClients } from '../lib'
|
||||
import { dfs } from '$lib/components/flows/previousResults'
|
||||
import { getStringError } from './utils'
|
||||
import type { FlowModuleState, FlowState } from '$lib/components/flows/flowState'
|
||||
@@ -56,8 +55,7 @@ import type {
|
||||
import type { Selection } from 'monaco-editor'
|
||||
import type AIChatInput from './AIChatInput.svelte'
|
||||
import { prepareApiSystemMessage, prepareApiUserMessage } from './api/core'
|
||||
import { getAnthropicCompletion, parseAnthropicCompletion } from './anthropic'
|
||||
import { getOpenAIResponsesCompletion, parseOpenAIResponsesCompletion } from './openai-responses'
|
||||
import { runChatLoop } from './chatLoop'
|
||||
import type { ReviewChangesOpts } from './monaco-adapter'
|
||||
import { getCurrentModel, tryGetCurrentModel, getCombinedCustomPrompt } from '$lib/aiStore'
|
||||
|
||||
@@ -413,130 +411,63 @@ class AIChatManager {
|
||||
systemMessage?: ChatCompletionSystemMessageParam
|
||||
}) => {
|
||||
try {
|
||||
let addedMessages: ChatCompletionMessageParam[] = []
|
||||
while (true) {
|
||||
const systemMessage = systemMessageOverride ?? this.systemMessage
|
||||
const helpers = this.helpers
|
||||
const tools = this.tools
|
||||
for (const tool of tools) {
|
||||
if (tool.setSchema) {
|
||||
await tool.setSchema(helpers)
|
||||
}
|
||||
}
|
||||
|
||||
let pendingPrompt = this.pendingPrompt
|
||||
let pendingUserMessage: ChatCompletionUserMessageParam | undefined = undefined
|
||||
if (pendingPrompt) {
|
||||
// Use JS getters so runChatLoop re-reads tools/helpers/systemMessage/modelProvider
|
||||
// on each iteration. This is critical for changeModeTool (Navigator → Script/Flow)
|
||||
// which reassigns this.tools, this.helpers, this.systemMessage mid-loop.
|
||||
const self = this
|
||||
const result = await runChatLoop({
|
||||
messages,
|
||||
get systemMessage() {
|
||||
return systemMessageOverride ?? self.systemMessage
|
||||
},
|
||||
get tools() {
|
||||
return self.tools
|
||||
},
|
||||
get helpers() {
|
||||
return self.helpers
|
||||
},
|
||||
abortController,
|
||||
callbacks,
|
||||
get modelProvider() {
|
||||
return getCurrentModel()
|
||||
},
|
||||
clients: {
|
||||
openai: workspaceAIClients.getOpenaiClient(),
|
||||
anthropic: workspaceAIClients.getAnthropicClient()
|
||||
},
|
||||
workspace: get(workspaceStore) ?? '',
|
||||
skipResponsesApi: this.skipResponsesApi,
|
||||
onSkipResponsesApi: () => {
|
||||
this.skipResponsesApi = true
|
||||
},
|
||||
getPendingUserMessage: () => {
|
||||
const pendingPrompt = this.pendingPrompt
|
||||
if (!pendingPrompt) return undefined
|
||||
this.pendingPrompt = ''
|
||||
if (this.mode === AIMode.SCRIPT) {
|
||||
pendingUserMessage = prepareScriptUserMessage(
|
||||
return prepareScriptUserMessage(
|
||||
pendingPrompt,
|
||||
this.contextManager.getSelectedContext()
|
||||
)
|
||||
} else if (this.mode === AIMode.FLOW) {
|
||||
pendingUserMessage = prepareFlowUserMessage(
|
||||
return prepareFlowUserMessage(
|
||||
pendingPrompt,
|
||||
this.flowAiChatHelpers!.getFlowAndSelectedId()
|
||||
)
|
||||
} else if (this.mode === AIMode.NAVIGATOR) {
|
||||
pendingUserMessage = prepareNavigatorUserMessage(pendingPrompt)
|
||||
return prepareNavigatorUserMessage(pendingPrompt)
|
||||
}
|
||||
this.pendingPrompt = ''
|
||||
}
|
||||
|
||||
const model = getCurrentModel()
|
||||
const isOpenAI = model.provider === 'openai' || model.provider === 'azure_openai'
|
||||
const isAnthropic = model.provider === 'anthropic'
|
||||
|
||||
const messageParams = [
|
||||
systemMessage,
|
||||
...messages,
|
||||
...(pendingUserMessage ? [pendingUserMessage] : [])
|
||||
]
|
||||
const toolDefs = tools.map((t) => t.def)
|
||||
|
||||
// For OpenAI/Azure, try Responses API first, fallback to Completions API
|
||||
if (isOpenAI) {
|
||||
let useCompletionsApi = this.skipResponsesApi
|
||||
if (!this.skipResponsesApi) {
|
||||
try {
|
||||
const completion = await getOpenAIResponsesCompletion(
|
||||
messageParams,
|
||||
abortController,
|
||||
toolDefs
|
||||
)
|
||||
const continueCompletion = await parseOpenAIResponsesCompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
} catch (err) {
|
||||
console.warn('OpenAI Responses API failed, falling back to Completions API:', err)
|
||||
// If the error indicates Responses API is not available in this region, skip it for future requests
|
||||
const errorMessage = err instanceof Error ? err.message : String(err)
|
||||
if (errorMessage.includes('Responses API is not enabled')) {
|
||||
this.skipResponsesApi = true
|
||||
}
|
||||
useCompletionsApi = true
|
||||
}
|
||||
}
|
||||
|
||||
// Use Completions API if Responses API is not available or failed
|
||||
if (useCompletionsApi) {
|
||||
const completion = await getCompletion(messageParams, abortController, toolDefs, {
|
||||
forceCompletions: true
|
||||
})
|
||||
const continueCompletion = await parseOpenAICompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
}
|
||||
} else if (isAnthropic) {
|
||||
const completion = await getAnthropicCompletion(messageParams, abortController, toolDefs)
|
||||
if (completion) {
|
||||
const continueCompletion = await parseAnthropicCompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers,
|
||||
abortController
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const completion = await getCompletion(messageParams, abortController, toolDefs)
|
||||
if (completion) {
|
||||
const continueCompletion = await parseOpenAICompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
return undefined
|
||||
},
|
||||
onBeforeIteration: async (tools) => {
|
||||
for (const tool of tools) {
|
||||
if (tool.setSchema) {
|
||||
await tool.setSchema(this.helpers)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return addedMessages
|
||||
})
|
||||
return result.addedMessages
|
||||
} catch (err) {
|
||||
console.log('chatRequest error', err)
|
||||
console.error('chatRequest error', err)
|
||||
|
||||
@@ -6,44 +6,77 @@ import { loadAppFixtureForEval } from './appFixtureLoader'
|
||||
import { dirname, join } from 'path'
|
||||
// @ts-ignore - Node.js url
|
||||
import { fileURLToPath } from 'url'
|
||||
import type { AIProvider } from '$lib/gen/types.gen'
|
||||
|
||||
// Get API key from environment - tests will be skipped if not set
|
||||
// Get API keys from environment - tests will be skipped if none are set
|
||||
// @ts-ignore
|
||||
const OPENROUTER_API_KEY = process.env.OPENROUTER_API_KEY
|
||||
const OPENAI_API_KEY = process.env.OPENAI_API_KEY
|
||||
// @ts-ignore
|
||||
const ANTHROPIC_API_KEY = process.env.ANTHROPIC_API_KEY
|
||||
|
||||
// Skip all tests if no API key is provided
|
||||
const describeWithApiKey = OPENROUTER_API_KEY ? describe : describe.skip
|
||||
const hasAnyKey = OPENAI_API_KEY || ANTHROPIC_API_KEY
|
||||
const describeWithApiKey = hasAnyKey ? describe : describe.skip
|
||||
|
||||
// Get __dirname equivalent for ES modules
|
||||
const __filename = fileURLToPath(import.meta.url)
|
||||
const __dirname = dirname(__filename)
|
||||
|
||||
const MODELS = ['google/gemini-2.5-flash', 'anthropic/claude-haiku-4.5', 'openai/gpt-4o']
|
||||
// Build model variants based on available keys
|
||||
interface ModelVariant {
|
||||
model: string
|
||||
provider: AIProvider
|
||||
apiKey: string
|
||||
}
|
||||
|
||||
const MODEL_VARIANTS: ModelVariant[] = [
|
||||
...(OPENAI_API_KEY
|
||||
? [{ model: 'gpt-4o', provider: 'openai' as AIProvider, apiKey: OPENAI_API_KEY }]
|
||||
: []),
|
||||
...(ANTHROPIC_API_KEY
|
||||
? [
|
||||
{
|
||||
model: 'claude-haiku-4-5-20241022',
|
||||
provider: 'anthropic' as AIProvider,
|
||||
apiKey: ANTHROPIC_API_KEY
|
||||
}
|
||||
]
|
||||
: [])
|
||||
]
|
||||
|
||||
const VARIANTS = [
|
||||
...MODELS.map((model) => ({
|
||||
...MODEL_VARIANTS.map((mv) => ({
|
||||
...BASELINE_VARIANT,
|
||||
model,
|
||||
name: `baseline-${model.replace('/', '-')}`
|
||||
model: mv.model,
|
||||
name: `baseline-${mv.provider}-${mv.model}`,
|
||||
_provider: mv.provider,
|
||||
_apiKey: mv.apiKey
|
||||
})),
|
||||
...MODELS.map((model) => ({
|
||||
...MODEL_VARIANTS.map((mv) => ({
|
||||
...STREAMLINED_VARIANT,
|
||||
model,
|
||||
name: `streamlined-${model.replace('/', '-')}`
|
||||
model: mv.model,
|
||||
name: `streamlined-${mv.provider}-${mv.model}`,
|
||||
_provider: mv.provider,
|
||||
_apiKey: mv.apiKey
|
||||
}))
|
||||
]
|
||||
|
||||
describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
const TEST_TIMEOUT = 120_000
|
||||
if (!OPENROUTER_API_KEY) {
|
||||
console.warn('OPENROUTER_API_KEY is not set, skipping tests')
|
||||
if (!hasAnyKey) {
|
||||
console.warn('No API keys set (OPENAI_API_KEY or ANTHROPIC_API_KEY), skipping tests')
|
||||
}
|
||||
|
||||
it(
|
||||
'test1: creates a simple counter app',
|
||||
async () => {
|
||||
const USER_PROMPT = `Create a counter app with increment/decrement buttons`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!)
|
||||
// Write results to files
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
undefined,
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
console.log(`App files: ${appPaths.join(', ')}`)
|
||||
@@ -56,17 +89,21 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
it(
|
||||
'test2: modifies existing counter app to add reset button',
|
||||
async () => {
|
||||
// Load initial app from fixture folder
|
||||
const { initialFrontend, initialBackend } = await loadAppFixtureForEval(
|
||||
join(__dirname, 'initial', 'test1_counter_app')
|
||||
)
|
||||
|
||||
const USER_PROMPT = `Add a reset button that sets the counter back to 0`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
})
|
||||
// Write results to files
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
console.log(`App files: ${appPaths.join(', ')}`)
|
||||
@@ -86,10 +123,16 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
)
|
||||
|
||||
const USER_PROMPT = `Add a quantity selector (+ and - buttons) to each cart item so users can adjust quantities without removing and re-adding items`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -108,10 +151,16 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
)
|
||||
|
||||
const USER_PROMPT = `Add a discount code input field in the cart. When the code "SAVE10" is entered, apply a 10% discount to the total`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -132,10 +181,16 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
)
|
||||
|
||||
const USER_PROMPT = `Add a search bar in the toolbar that filters files and folders by name as the user types`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -154,10 +209,16 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
)
|
||||
|
||||
const USER_PROMPT = `Show file size (formatted as KB/MB) and modified date in the file list for each item`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -176,10 +237,16 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
)
|
||||
|
||||
const USER_PROMPT = `Add a "Select All" checkbox in the file list header and individual checkboxes for each file. Add a "Delete Selected" button that appears when items are selected`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialFrontend,
|
||||
initialBackend
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -196,7 +263,13 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
'test8: create quiz app from scratch',
|
||||
async () => {
|
||||
const USER_PROMPT = `Create a multiple choice quiz app with 5 questions about general knowledge. Show one question at a time with 4 answer options. Track the score and show results at the end with percentage correct.`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!)
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
undefined,
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -211,7 +284,13 @@ describeWithApiKey('App Chat LLM Evaluation', () => {
|
||||
'test9: create recipe book from scratch',
|
||||
async () => {
|
||||
const USER_PROMPT = `Create a recipe book app where users can add recipes with a name, ingredients list, and instructions. Include a search bar to filter recipes by name and the ability to delete recipes.`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!)
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
undefined,
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, appPaths } = await writeAppComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import OpenAI from 'openai'
|
||||
import Anthropic from '@anthropic-ai/sdk'
|
||||
import type { AppFiles, BackendRunnable } from '../../app/core'
|
||||
import { BASE_EVALUATOR_RESPONSE_FORMAT } from '../shared'
|
||||
import type { EvaluationResult } from '../shared'
|
||||
@@ -71,12 +71,7 @@ ${BASE_EVALUATOR_RESPONSE_FORMAT}`
|
||||
|
||||
/**
|
||||
* Evaluates how well a generated app fulfills the user's request, considering any initial app state.
|
||||
* This evaluator does not require an expected reference app - it evaluates based on the request alone.
|
||||
*
|
||||
* @param userPrompt The original user request
|
||||
* @param generatedApp The app generated by the AI
|
||||
* @param initialApp Optional initial app state (what the app looked like before AI changes)
|
||||
* @returns Evaluation result with score, statement, and missing requirements
|
||||
* Uses Anthropic API directly.
|
||||
*/
|
||||
export async function evaluateAppGeneration(
|
||||
userPrompt: string,
|
||||
@@ -84,9 +79,17 @@ export async function evaluateAppGeneration(
|
||||
initialApp?: InitialApp
|
||||
): Promise<EvaluationResult> {
|
||||
// @ts-ignore
|
||||
const apiKey = process.env.OPENROUTER_API_KEY
|
||||
const apiKey = process.env.ANTHROPIC_API_KEY
|
||||
if (!apiKey) {
|
||||
return {
|
||||
success: false,
|
||||
resemblanceScore: 0,
|
||||
statement: 'No API key available for evaluation',
|
||||
error: 'ANTHROPIC_API_KEY not set'
|
||||
}
|
||||
}
|
||||
|
||||
const client = new OpenAI({ baseURL: 'https://openrouter.ai/api/v1', apiKey })
|
||||
const client = new Anthropic({ apiKey })
|
||||
|
||||
let userMessage = `## User's Original Request
|
||||
${userPrompt}
|
||||
@@ -117,16 +120,18 @@ Please evaluate how well the generated app:
|
||||
2. ${initialApp ? 'Makes appropriate modifications to the initial app state' : 'Implements a complete and correct new app'}`
|
||||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
model: 'anthropic/claude-sonnet-4.5',
|
||||
const response = await client.messages.create({
|
||||
model: 'claude-sonnet-4-5-20250514',
|
||||
max_tokens: 2048,
|
||||
system: APP_GENERATION_EVALUATOR_SYSTEM_PROMPT,
|
||||
messages: [
|
||||
{ role: 'system', content: APP_GENERATION_EVALUATOR_SYSTEM_PROMPT },
|
||||
{ role: 'user', content: userMessage }
|
||||
],
|
||||
temperature: 0
|
||||
})
|
||||
|
||||
const content = response.choices[0]?.message?.content
|
||||
const textBlock = response.content.find((block) => block.type === 'text')
|
||||
const content = textBlock?.text
|
||||
if (!content) {
|
||||
return {
|
||||
success: false,
|
||||
|
||||
@@ -14,6 +14,7 @@ import {
|
||||
type VariantDefaults
|
||||
} from '../shared'
|
||||
import { writeAppComparisonResultsToFolders } from './appResultsWriter'
|
||||
import type { AIProvider } from '$lib/gen/types.gen'
|
||||
|
||||
// Re-export for convenience
|
||||
export type { InitialApp } from './appEvalComparison'
|
||||
@@ -38,6 +39,8 @@ export interface AppEvalOptions {
|
||||
variant?: VariantConfig
|
||||
/** Whether to evaluate the generated app with LLM. Default: true. Set to false to skip evaluation. */
|
||||
evaluateWithLLM?: boolean
|
||||
/** AI provider (inferred from model name if omitted) */
|
||||
provider?: AIProvider
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -49,12 +52,11 @@ const appDefaults: VariantDefaults<AppAIChatHelpers> = {
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs an app chat evaluation with real OpenAI API calls.
|
||||
* Executes tool calls using the actual app tools from core.ts or variant-configured tools.
|
||||
* Runs an app chat evaluation using the shared chat loop (same code path as production).
|
||||
*/
|
||||
export async function runAppEval(
|
||||
userPrompt: string,
|
||||
openaiApiKey: string,
|
||||
apiKey: string,
|
||||
options?: AppEvalOptions
|
||||
): Promise<AppEvalResult> {
|
||||
const { helpers, getFiles } = createAppEvalHelpers(
|
||||
@@ -69,7 +71,7 @@ export async function runAppEval(
|
||||
appDefaults,
|
||||
options?.customSystemPrompt
|
||||
)
|
||||
const { toolDefs, tools } = resolveTools(options?.variant, appDefaults)
|
||||
const { tools } = resolveTools(options?.variant, appDefaults)
|
||||
const model = resolveModel(options?.variant, options?.model)
|
||||
|
||||
// Build user message
|
||||
@@ -80,15 +82,15 @@ export async function runAppEval(
|
||||
userPrompt,
|
||||
systemMessage,
|
||||
userMessage,
|
||||
toolDefs,
|
||||
tools,
|
||||
helpers,
|
||||
apiKey: openaiApiKey,
|
||||
apiKey,
|
||||
getOutput: getFiles,
|
||||
options: {
|
||||
maxIterations: options?.maxIterations,
|
||||
model,
|
||||
workspace: 'test-workspace'
|
||||
workspace: 'test-workspace',
|
||||
provider: options?.provider
|
||||
}
|
||||
})
|
||||
|
||||
@@ -114,21 +116,32 @@ export async function runAppEval(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-variant provider override.
|
||||
*/
|
||||
export interface VariantProviderOverride {
|
||||
provider: AIProvider
|
||||
apiKey: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs the same prompt against multiple variants sequentially for comparison.
|
||||
* Returns results in the same order as the input variants.
|
||||
* Accepts optional per-variant provider/apiKey overrides.
|
||||
*/
|
||||
export async function runVariantComparison(
|
||||
userPrompt: string,
|
||||
variants: VariantConfig[],
|
||||
openaiApiKey: string,
|
||||
baseOptions?: Omit<AppEvalOptions, 'variant'>
|
||||
defaultApiKey: string,
|
||||
baseOptions?: Omit<AppEvalOptions, 'variant'>,
|
||||
providerOverrides?: VariantProviderOverride[]
|
||||
): Promise<AppEvalResult[]> {
|
||||
const results: AppEvalResult[] = await Promise.all(
|
||||
variants.map(async (variant) => {
|
||||
return await runAppEval(userPrompt, openaiApiKey, {
|
||||
variants.map(async (variant, i) => {
|
||||
const override = providerOverrides?.[i]
|
||||
return await runAppEval(userPrompt, override?.apiKey ?? defaultApiKey, {
|
||||
...baseOptions,
|
||||
variant
|
||||
variant,
|
||||
provider: override?.provider ?? baseOptions?.provider
|
||||
})
|
||||
})
|
||||
)
|
||||
|
||||
@@ -22,35 +22,60 @@ import initialTest6 from './initial/test6_initial.json'
|
||||
// @ts-ignore - JSON import
|
||||
import initialTest7 from './initial/test7_initial.json'
|
||||
import type { FlowModule } from '$lib/gen'
|
||||
import type { AIProvider } from '$lib/gen/types.gen'
|
||||
|
||||
// Get API key from environment - tests will be skipped if not set
|
||||
// Get API keys from environment - tests will be skipped if none are set
|
||||
// @ts-ignore
|
||||
// const OPENAI_API_KEY = process.env.OPENAI_API_KEY
|
||||
const OPENROUTER_API_KEY = process.env.OPENROUTER_API_KEY
|
||||
const OPENAI_API_KEY = process.env.OPENAI_API_KEY
|
||||
// @ts-ignore
|
||||
const ANTHROPIC_API_KEY = process.env.ANTHROPIC_API_KEY
|
||||
|
||||
// Skip all tests if no API key is provided
|
||||
// const describeWithApiKey = OPENAI_API_KEY ? describe : describe.skip
|
||||
const describeWithApiKey = OPENROUTER_API_KEY ? describe : describe.skip
|
||||
const hasAnyKey = OPENAI_API_KEY || ANTHROPIC_API_KEY
|
||||
const describeWithApiKey = hasAnyKey ? describe : describe.skip
|
||||
|
||||
const MODELS = ['google/gemini-2.5-flash', 'anthropic/claude-haiku-4.5', 'openai/gpt-4o']
|
||||
// Build model variants based on available keys
|
||||
interface ModelVariant {
|
||||
model: string
|
||||
provider: AIProvider
|
||||
apiKey: string
|
||||
}
|
||||
|
||||
const MODEL_VARIANTS: ModelVariant[] = [
|
||||
...(OPENAI_API_KEY
|
||||
? [{ model: 'gpt-4o', provider: 'openai' as AIProvider, apiKey: OPENAI_API_KEY }]
|
||||
: []),
|
||||
...(ANTHROPIC_API_KEY
|
||||
? [
|
||||
{
|
||||
model: 'claude-haiku-4-5-20241022',
|
||||
provider: 'anthropic' as AIProvider,
|
||||
apiKey: ANTHROPIC_API_KEY
|
||||
}
|
||||
]
|
||||
: [])
|
||||
]
|
||||
|
||||
const VARIANTS = [
|
||||
...MODELS.map((model) => ({
|
||||
...MODEL_VARIANTS.map((mv) => ({
|
||||
...BASELINE_VARIANT,
|
||||
model,
|
||||
name: `baseline-${model.replace('/', '-')}`
|
||||
model: mv.model,
|
||||
name: `baseline-${mv.provider}-${mv.model}`,
|
||||
_provider: mv.provider,
|
||||
_apiKey: mv.apiKey
|
||||
})),
|
||||
...MODELS.map((model) => ({
|
||||
...MODEL_VARIANTS.map((mv) => ({
|
||||
...MINIMAL_SINGLE_TOOL_VARIANT,
|
||||
model,
|
||||
name: `minimal-single-tool-${model.replace('/', '-')}`
|
||||
model: mv.model,
|
||||
name: `minimal-single-tool-${mv.provider}-${mv.model}`,
|
||||
_provider: mv.provider,
|
||||
_apiKey: mv.apiKey
|
||||
}))
|
||||
]
|
||||
|
||||
describeWithApiKey('Flow Chat LLM Evaluation', () => {
|
||||
const TEST_TIMEOUT = 120_000
|
||||
if (!OPENROUTER_API_KEY) {
|
||||
console.warn('OPENROUTER_API_KEY is not set, skipping tests')
|
||||
if (!hasAnyKey) {
|
||||
console.warn('No API keys set (OPENAI_API_KEY or ANTHROPIC_API_KEY), skipping tests')
|
||||
}
|
||||
|
||||
it(
|
||||
@@ -65,9 +90,15 @@ STEP 3: Loop on all users
|
||||
STEP 4: Do branches based on user's role, do different action based on that. Roles are admin, user, moderator
|
||||
STEP 5: Return action taken for each user
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
expectedFlow: expectedTest1 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
expectedFlow: expectedTest1 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
// Write results to files
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
@@ -112,9 +143,15 @@ STEP 5: Branch based on inventory - if all items available, create shipment reco
|
||||
STEP 6: Send confirmation (mock email to customer_email)
|
||||
STEP 7: Return final order summary with status
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
expectedFlow: expectedTest2 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
expectedFlow: expectedTest2 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -161,9 +198,15 @@ STEP 5: Branch based on quality score:
|
||||
- If score < 70: Store in quarantine and send alert
|
||||
STEP 6: Return processing report with statistics (total records, quality score, destination)
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
expectedFlow: expectedTest3 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
expectedFlow: expectedTest3 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -210,9 +253,15 @@ STEP 3: Use an AI agent to handle the customer query. The agent should have acce
|
||||
STEP 4: Log the interaction to audit trail (customer_id, query, response summary)
|
||||
STEP 5: Return the agent's response and any actions taken
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
expectedFlow: expectedTest4 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
expectedFlow: expectedTest4 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -256,11 +305,17 @@ Modify this existing flow to add error handling:
|
||||
- If validation passes, return the data for the next step
|
||||
- Update save_results to handle the validation result appropriately
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialModules: initialTest5.value.modules as FlowModule[],
|
||||
initialSchema: initialTest5.schema,
|
||||
expectedFlow: expectedTest5 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialModules: initialTest5.value.modules as FlowModule[],
|
||||
initialSchema: initialTest5.schema,
|
||||
expectedFlow: expectedTest5 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -302,11 +357,17 @@ Modify the order processing loop to handle different order types:
|
||||
- Move the original process_order step to the default branch for unknown order types
|
||||
- Each branch step should return the orderId, shipping cost, and shipping type
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialModules: initialTest6.value.modules as FlowModule[],
|
||||
initialSchema: initialTest6.schema,
|
||||
expectedFlow: expectedTest6 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialModules: initialTest6.value.modules as FlowModule[],
|
||||
initialSchema: initialTest6.schema,
|
||||
expectedFlow: expectedTest6 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
@@ -348,11 +409,17 @@ Refactor this flow for better performance by parallelizing the enrichment steps:
|
||||
- The combine_data step should check if any enrichment used a fallback value and set a hasFallbacks flag
|
||||
- Keep get_item as the first step and return_result as the last step unchanged
|
||||
`
|
||||
const results = await runVariantComparison(USER_PROMPT, VARIANTS, OPENROUTER_API_KEY!, {
|
||||
initialModules: initialTest7.value.modules as FlowModule[],
|
||||
initialSchema: initialTest7.schema,
|
||||
expectedFlow: expectedTest7 as ExpectedFlow
|
||||
})
|
||||
const results = await runVariantComparison(
|
||||
USER_PROMPT,
|
||||
VARIANTS,
|
||||
VARIANTS[0]._apiKey,
|
||||
{
|
||||
initialModules: initialTest7.value.modules as FlowModule[],
|
||||
initialSchema: initialTest7.schema,
|
||||
expectedFlow: expectedTest7 as ExpectedFlow
|
||||
},
|
||||
VARIANTS.map((v) => ({ provider: v._provider, apiKey: v._apiKey }))
|
||||
)
|
||||
|
||||
const { summaryPath, flowPaths } = await writeFlowComparisonResults(USER_PROMPT, results)
|
||||
console.log(`\nResults written to: ${summaryPath}`)
|
||||
|
||||
@@ -59,14 +59,10 @@ export async function evaluateFlowComparison(
|
||||
expectedFlow: ExpectedFlow,
|
||||
userPrompt: string
|
||||
): Promise<EvaluationResult> {
|
||||
// @ts-ignore
|
||||
const apiKey = process.env.OPENROUTER_API_KEY
|
||||
|
||||
return evaluateWithLLM({
|
||||
userPrompt,
|
||||
generatedOutput: generatedFlow,
|
||||
expectedOutput: expectedFlow,
|
||||
evaluatorSystemPrompt: FLOW_EVALUATOR_SYSTEM_PROMPT,
|
||||
apiKey
|
||||
evaluatorSystemPrompt: FLOW_EVALUATOR_SYSTEM_PROMPT
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type { FlowModule } from '$lib/gen'
|
||||
import type { AIProvider } from '$lib/gen/types.gen'
|
||||
import type { ExtendedOpenFlow } from '$lib/components/flows/types'
|
||||
import { flowTools, prepareFlowSystemMessage, prepareFlowUserMessage, type FlowAIChatHelpers } from '../../flow/core'
|
||||
import { createFlowEvalHelpers } from './flowEvalHelpers'
|
||||
@@ -38,6 +39,8 @@ export interface FlowEvalOptions {
|
||||
maxIterations?: number
|
||||
variant?: VariantConfig
|
||||
expectedFlow?: ExpectedFlow
|
||||
/** AI provider (inferred from model name if omitted) */
|
||||
provider?: AIProvider
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -49,12 +52,11 @@ const flowDefaults: VariantDefaults<FlowAIChatHelpers> = {
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs a flow chat evaluation with real OpenAI API calls.
|
||||
* Executes tool calls using the actual flowTools from core.ts or variant-configured tools.
|
||||
* Runs a flow chat evaluation using the shared chat loop (same code path as production).
|
||||
*/
|
||||
export async function runFlowEval(
|
||||
userPrompt: string,
|
||||
openaiApiKey: string,
|
||||
apiKey: string,
|
||||
options?: FlowEvalOptions
|
||||
): Promise<FlowEvalResult> {
|
||||
const { helpers, getFlow } = createFlowEvalHelpers(
|
||||
@@ -65,7 +67,7 @@ export async function runFlowEval(
|
||||
// Resolve variant configuration
|
||||
const variantName = options?.variant?.name ?? 'baseline'
|
||||
const systemMessage = resolveSystemPrompt(options?.variant, flowDefaults, options?.customSystemPrompt)
|
||||
const { toolDefs, tools } = resolveTools(options?.variant, flowDefaults)
|
||||
const { tools } = resolveTools(options?.variant, flowDefaults)
|
||||
const model = resolveModel(options?.variant, options?.model)
|
||||
|
||||
// Build user message
|
||||
@@ -76,15 +78,15 @@ export async function runFlowEval(
|
||||
userPrompt,
|
||||
systemMessage,
|
||||
userMessage,
|
||||
toolDefs,
|
||||
tools,
|
||||
helpers,
|
||||
apiKey: openaiApiKey,
|
||||
apiKey,
|
||||
getOutput: getFlow,
|
||||
options: {
|
||||
maxIterations: options?.maxIterations,
|
||||
model,
|
||||
workspace: 'test-workspace'
|
||||
workspace: 'test-workspace',
|
||||
provider: options?.provider
|
||||
}
|
||||
})
|
||||
|
||||
@@ -111,21 +113,32 @@ export async function runFlowEval(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-variant provider override.
|
||||
*/
|
||||
export interface VariantProviderOverride {
|
||||
provider: AIProvider
|
||||
apiKey: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs the same prompt against multiple variants sequentially for comparison.
|
||||
* Returns results in the same order as the input variants.
|
||||
* Accepts optional per-variant provider/apiKey overrides.
|
||||
*/
|
||||
export async function runVariantComparison(
|
||||
userPrompt: string,
|
||||
variants: VariantConfig[],
|
||||
openaiApiKey: string,
|
||||
baseOptions?: Omit<FlowEvalOptions, 'variant'>
|
||||
defaultApiKey: string,
|
||||
baseOptions?: Omit<FlowEvalOptions, 'variant'>,
|
||||
providerOverrides?: VariantProviderOverride[]
|
||||
): Promise<FlowEvalResult[]> {
|
||||
const results: FlowEvalResult[] = await Promise.all(
|
||||
variants.map(async (variant) => {
|
||||
return await runFlowEval(userPrompt, openaiApiKey, {
|
||||
variants.map(async (variant, i) => {
|
||||
const override = providerOverrides?.[i]
|
||||
return await runFlowEval(userPrompt, override?.apiKey ?? defaultApiKey, {
|
||||
...baseOptions,
|
||||
variant
|
||||
variant,
|
||||
provider: override?.provider ?? baseOptions?.provider
|
||||
})
|
||||
})
|
||||
)
|
||||
|
||||
@@ -1,8 +1,14 @@
|
||||
import OpenAI, { APIError } from 'openai'
|
||||
import type { ChatCompletionMessageParam, ChatCompletionSystemMessageParam } from 'openai/resources/chat/completions.mjs'
|
||||
import type { ChatCompletionTool } from 'openai/resources/chat/completions.mjs'
|
||||
import OpenAI from 'openai'
|
||||
import Anthropic from '@anthropic-ai/sdk'
|
||||
import type {
|
||||
ChatCompletionMessageParam,
|
||||
ChatCompletionSystemMessageParam
|
||||
} from 'openai/resources/chat/completions.mjs'
|
||||
import type { AIProvider, AIProviderModel } from '$lib/gen/types.gen'
|
||||
import type { TokenUsage, ToolCallDetail, EvalRunnerOptions } from './types'
|
||||
import type { Tool } from './baseVariants'
|
||||
import { runChatLoop, type ChatClients } from '../../chatLoop'
|
||||
import type { Tool as ProductionTool, ToolCallbacks } from '../../shared'
|
||||
|
||||
/**
|
||||
* Result from a single eval run (before domain-specific evaluation).
|
||||
@@ -29,13 +35,13 @@ export interface RunEvalParams<THelpers, TOutput> {
|
||||
systemMessage: ChatCompletionSystemMessageParam
|
||||
/** User message for the LLM */
|
||||
userMessage: ChatCompletionMessageParam
|
||||
/** Tool definitions for the LLM API */
|
||||
toolDefs: ChatCompletionTool[]
|
||||
/** Tool definitions for the LLM API (unused — derived from tools) */
|
||||
toolDefs?: unknown
|
||||
/** Full tool implementations for execution */
|
||||
tools: Tool<THelpers>[]
|
||||
/** Domain-specific helpers for tool execution */
|
||||
helpers: THelpers
|
||||
/** API key for OpenRouter */
|
||||
/** API key for the provider */
|
||||
apiKey: string
|
||||
/** Function to get the current output state */
|
||||
getOutput: () => TOutput
|
||||
@@ -44,10 +50,37 @@ export interface RunEvalParams<THelpers, TOutput> {
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs a generic evaluation with real LLM API calls.
|
||||
* Executes tool calls in a loop until the LLM stops calling tools.
|
||||
*
|
||||
* This is the core execution loop shared across all chat eval tests.
|
||||
* Creates SDK clients for the given provider.
|
||||
*/
|
||||
function createEvalClients(provider: AIProvider, apiKey: string): ChatClients {
|
||||
if (provider === 'anthropic') {
|
||||
return {
|
||||
openai: new OpenAI({ apiKey: 'unused' }),
|
||||
anthropic: new Anthropic({ apiKey })
|
||||
}
|
||||
}
|
||||
return {
|
||||
openai: new OpenAI({ apiKey }),
|
||||
anthropic: new Anthropic({ apiKey: 'unused' })
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves model string to AIProviderModel.
|
||||
*/
|
||||
function resolveModelProvider(
|
||||
model: string,
|
||||
provider?: AIProvider
|
||||
): AIProviderModel {
|
||||
if (provider) return { provider, model }
|
||||
if (model.startsWith('claude')) return { provider: 'anthropic', model }
|
||||
if (model.startsWith('gpt') || model.startsWith('o')) return { provider: 'openai', model }
|
||||
return { provider: 'openai', model }
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs a generic evaluation using the shared chat loop (same code path as production).
|
||||
* Uses streaming via real provider SDKs instead of OpenRouter non-streaming.
|
||||
*/
|
||||
export async function runEval<THelpers, TOutput>(
|
||||
params: RunEvalParams<THelpers, TOutput>
|
||||
@@ -55,7 +88,6 @@ export async function runEval<THelpers, TOutput>(
|
||||
const {
|
||||
systemMessage,
|
||||
userMessage,
|
||||
toolDefs,
|
||||
tools,
|
||||
helpers,
|
||||
apiKey,
|
||||
@@ -63,134 +95,82 @@ export async function runEval<THelpers, TOutput>(
|
||||
options
|
||||
} = params
|
||||
|
||||
const client = new OpenAI({ baseURL: 'https://openrouter.ai/api/v1', apiKey })
|
||||
const model = options?.model ?? 'gpt-4o'
|
||||
const maxIterations = options?.maxIterations ?? 20
|
||||
const workspace = options?.workspace ?? 'test-workspace'
|
||||
const provider = options?.provider
|
||||
|
||||
const messages: ChatCompletionMessageParam[] = [systemMessage, userMessage]
|
||||
const totalTokens: TokenUsage = { prompt: 0, completion: 0, total: 0 }
|
||||
const modelProvider = resolveModelProvider(model, provider)
|
||||
const clients = createEvalClients(modelProvider.provider, apiKey)
|
||||
|
||||
const messages: ChatCompletionMessageParam[] = [userMessage]
|
||||
let toolCallsCount = 0
|
||||
const toolsCalled: string[] = []
|
||||
const toolCallDetails: ToolCallDetail[] = []
|
||||
let iterations = 0
|
||||
|
||||
// No-op tool callbacks for eval
|
||||
const toolCallbacks = {
|
||||
// Wrap tools to intercept fn calls for tracking.
|
||||
// Cast to ProductionTool since the eval Tool has a narrower toolCallbacks type
|
||||
// but the actual callbacks passed at runtime will satisfy both interfaces.
|
||||
const wrappedTools = tools.map((tool) => ({
|
||||
...tool,
|
||||
fn: async (p: any) => {
|
||||
toolCallsCount++
|
||||
toolsCalled.push(tool.def.function.name)
|
||||
try {
|
||||
const args =
|
||||
typeof p.args === 'string' ? JSON.parse(p.args) : p.args
|
||||
toolCallDetails.push({ name: tool.def.function.name, arguments: args })
|
||||
} catch {
|
||||
toolCallDetails.push({
|
||||
name: tool.def.function.name,
|
||||
arguments: p.args
|
||||
})
|
||||
}
|
||||
return tool.fn(p)
|
||||
}
|
||||
})) as ProductionTool<THelpers>[]
|
||||
|
||||
// No-op callbacks for eval
|
||||
const callbacks: ToolCallbacks & {
|
||||
onNewToken: (token: string) => void
|
||||
onMessageEnd: () => void
|
||||
} = {
|
||||
setToolStatus: () => {},
|
||||
removeToolStatus: () => {}
|
||||
removeToolStatus: () => {},
|
||||
onNewToken: () => {},
|
||||
onMessageEnd: () => {}
|
||||
}
|
||||
|
||||
const abortController = new AbortController()
|
||||
|
||||
try {
|
||||
// Tool resolution loop
|
||||
while (iterations < maxIterations) {
|
||||
iterations++
|
||||
|
||||
const response = await client.chat.completions.create({
|
||||
model,
|
||||
messages,
|
||||
tools: toolDefs,
|
||||
temperature: 0
|
||||
})
|
||||
|
||||
// Track token usage
|
||||
if (response.usage) {
|
||||
totalTokens.prompt += response.usage.prompt_tokens
|
||||
totalTokens.completion += response.usage.completion_tokens
|
||||
totalTokens.total += response.usage.total_tokens
|
||||
}
|
||||
|
||||
if (!response.choices.length) {
|
||||
throw new Error('No response from API')
|
||||
}
|
||||
|
||||
const choice = response.choices[0]
|
||||
const assistantMessage = choice.message
|
||||
|
||||
// Add assistant message to history
|
||||
messages.push(assistantMessage)
|
||||
|
||||
// If no tool calls, we're done
|
||||
if (!assistantMessage.tool_calls?.length) {
|
||||
break
|
||||
}
|
||||
|
||||
// Execute each tool call
|
||||
for (const toolCall of assistantMessage.tool_calls) {
|
||||
toolCallsCount++
|
||||
|
||||
// Type guard: only handle function tool calls
|
||||
if (toolCall.type !== 'function') {
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
tool_call_id: toolCall.id,
|
||||
content: `Unsupported tool type: ${toolCall.type}`
|
||||
})
|
||||
continue
|
||||
}
|
||||
|
||||
toolsCalled.push(toolCall.function.name)
|
||||
|
||||
const tool = tools.find((t) => t.def.function.name === toolCall.function.name)
|
||||
if (!tool) {
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
tool_call_id: toolCall.id,
|
||||
content: `Unknown tool: ${toolCall.function.name}`
|
||||
})
|
||||
continue
|
||||
}
|
||||
|
||||
try {
|
||||
const args = JSON.parse(toolCall.function.arguments)
|
||||
toolCallDetails.push({ name: toolCall.function.name, arguments: args })
|
||||
const result = await tool.fn({
|
||||
args,
|
||||
workspace,
|
||||
helpers,
|
||||
toolCallbacks,
|
||||
toolId: toolCall.id
|
||||
})
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
tool_call_id: toolCall.id,
|
||||
content: result
|
||||
})
|
||||
} catch (err) {
|
||||
const errorMessage = err instanceof Error ? err.message : String(err)
|
||||
messages.push({
|
||||
role: 'tool',
|
||||
tool_call_id: toolCall.id,
|
||||
content: `Error: ${errorMessage}`
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
const result = await runChatLoop({
|
||||
messages,
|
||||
systemMessage,
|
||||
tools: wrappedTools,
|
||||
helpers,
|
||||
abortController,
|
||||
callbacks,
|
||||
modelProvider,
|
||||
clients,
|
||||
workspace,
|
||||
maxIterations,
|
||||
skipResponsesApi: modelProvider.provider !== 'openai' && modelProvider.provider !== 'azure_openai'
|
||||
})
|
||||
|
||||
return {
|
||||
success: true,
|
||||
output: getOutput(),
|
||||
tokenUsage: totalTokens,
|
||||
tokenUsage: { prompt: 0, completion: 0, total: 0 },
|
||||
toolCallsCount,
|
||||
toolsCalled,
|
||||
toolCallDetails,
|
||||
iterations,
|
||||
iterations: Math.max(1, result.addedMessages.filter((m) => m.role === 'assistant').length),
|
||||
messages
|
||||
}
|
||||
} catch (err) {
|
||||
// Build detailed error message
|
||||
let errorMessage: string
|
||||
if (err instanceof APIError) {
|
||||
const details: string[] = [`${err.status} ${err.message}`]
|
||||
if (err.code) details.push(`Code: ${err.code}`)
|
||||
if (err.type) details.push(`Type: ${err.type}`)
|
||||
if (err.param) details.push(`Param: ${err.param}`)
|
||||
if (err.requestID) details.push(`Request ID: ${err.requestID}`)
|
||||
if (err.error && typeof err.error === 'object') {
|
||||
details.push(`Response: ${JSON.stringify(err.error, null, 2)}`)
|
||||
}
|
||||
errorMessage = details.join('\n')
|
||||
} else if (err instanceof Error) {
|
||||
if (err instanceof Error) {
|
||||
errorMessage = err.stack ?? err.message
|
||||
} else {
|
||||
errorMessage = String(err)
|
||||
@@ -200,11 +180,11 @@ export async function runEval<THelpers, TOutput>(
|
||||
success: false,
|
||||
output: getOutput(),
|
||||
error: errorMessage,
|
||||
tokenUsage: totalTokens,
|
||||
tokenUsage: { prompt: 0, completion: 0, total: 0 },
|
||||
toolCallsCount,
|
||||
toolsCalled,
|
||||
toolCallDetails,
|
||||
iterations,
|
||||
iterations: 0,
|
||||
messages
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import OpenAI from 'openai'
|
||||
import Anthropic from '@anthropic-ai/sdk'
|
||||
import type { EvaluationResult } from './types'
|
||||
|
||||
/**
|
||||
@@ -13,9 +13,9 @@ export interface EvaluateParams {
|
||||
expectedOutput: unknown
|
||||
/** Domain-specific system prompt for the evaluator */
|
||||
evaluatorSystemPrompt: string
|
||||
/** API key for OpenRouter */
|
||||
apiKey: string
|
||||
/** Model to use for evaluation (default: 'anthropic/claude-sonnet-4.5') */
|
||||
/** Anthropic API key for evaluation */
|
||||
apiKey?: string
|
||||
/** Model to use for evaluation (default: 'claude-sonnet-4-5-20250514') */
|
||||
model?: string
|
||||
}
|
||||
|
||||
@@ -41,10 +41,7 @@ Score guidelines:
|
||||
|
||||
/**
|
||||
* Evaluates how well a generated output matches an expected output using an LLM.
|
||||
* Returns a resemblance score (0-100), a qualitative statement, and any missing requirements.
|
||||
*
|
||||
* @param params Evaluation parameters including prompts, outputs, and API configuration
|
||||
* @returns Evaluation result with score, statement, and missing requirements
|
||||
* Uses Anthropic API directly instead of OpenRouter.
|
||||
*/
|
||||
export async function evaluateWithLLM(params: EvaluateParams): Promise<EvaluationResult> {
|
||||
const {
|
||||
@@ -53,10 +50,21 @@ export async function evaluateWithLLM(params: EvaluateParams): Promise<Evaluatio
|
||||
expectedOutput,
|
||||
evaluatorSystemPrompt,
|
||||
apiKey,
|
||||
model = 'anthropic/claude-sonnet-4.5'
|
||||
model = 'claude-sonnet-4-5-20250514'
|
||||
} = params
|
||||
|
||||
const client = new OpenAI({ baseURL: 'https://openrouter.ai/api/v1', apiKey })
|
||||
// @ts-ignore - process.env
|
||||
const anthropicKey = apiKey ?? process.env.ANTHROPIC_API_KEY
|
||||
if (!anthropicKey) {
|
||||
return {
|
||||
success: false,
|
||||
resemblanceScore: 0,
|
||||
statement: 'No API key available for evaluation',
|
||||
error: 'ANTHROPIC_API_KEY not set and no apiKey provided'
|
||||
}
|
||||
}
|
||||
|
||||
const client = new Anthropic({ apiKey: anthropicKey })
|
||||
|
||||
const userMessage = `## User's Original Request
|
||||
${userPrompt}
|
||||
@@ -76,16 +84,18 @@ Please evaluate how well the generated output:
|
||||
2. Matches the structure and logic of the expected reference output`
|
||||
|
||||
try {
|
||||
const response = await client.chat.completions.create({
|
||||
const response = await client.messages.create({
|
||||
model,
|
||||
max_tokens: 2048,
|
||||
system: evaluatorSystemPrompt,
|
||||
messages: [
|
||||
{ role: 'system', content: evaluatorSystemPrompt },
|
||||
{ role: 'user', content: userMessage }
|
||||
],
|
||||
temperature: 0
|
||||
})
|
||||
|
||||
const content = response.choices[0]?.message?.content
|
||||
const textBlock = response.content.find((block) => block.type === 'text')
|
||||
const content = textBlock?.text
|
||||
if (!content) {
|
||||
return {
|
||||
success: false,
|
||||
@@ -98,7 +108,6 @@ Please evaluate how well the generated output:
|
||||
// Parse JSON response - handle potential markdown code blocks
|
||||
let jsonContent = content.trim()
|
||||
if (jsonContent.startsWith('```')) {
|
||||
// Remove markdown code block wrapper
|
||||
jsonContent = jsonContent.replace(/^```(?:json)?\n?/, '').replace(/\n?```$/, '')
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type { ChatCompletionMessageParam } from 'openai/resources/chat/completions.mjs'
|
||||
import type { AIProvider } from '$lib/gen/types.gen'
|
||||
|
||||
/**
|
||||
* Token usage tracking for LLM calls.
|
||||
@@ -83,6 +84,8 @@ export interface EvalRunnerOptions {
|
||||
model?: string
|
||||
/** Workspace ID for tool calls */
|
||||
workspace?: string
|
||||
/** AI provider (inferred from model name if omitted) */
|
||||
provider?: AIProvider
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { OpenAI } from 'openai'
|
||||
import Anthropic from '@anthropic-ai/sdk'
|
||||
import type {
|
||||
ChatCompletionMessageParam,
|
||||
ChatCompletionMessageFunctionToolCall
|
||||
@@ -13,19 +14,28 @@ import type {
|
||||
RawMessageStreamEvent
|
||||
} from '@anthropic-ai/sdk/resources'
|
||||
import type { MessageStream } from '@anthropic-ai/sdk/lib/MessageStream'
|
||||
import type { AIProviderModel } from '$lib/gen'
|
||||
import { getProviderAndCompletionConfig, workspaceAIClients } from '../lib'
|
||||
import { processToolCall, type Tool, type ToolCallbacks } from './shared'
|
||||
|
||||
export async function getAnthropicCompletion(
|
||||
messages: ChatCompletionMessageParam[],
|
||||
abortController: AbortController,
|
||||
tools?: OpenAI.Chat.Completions.ChatCompletionFunctionTool[]
|
||||
tools?: OpenAI.Chat.Completions.ChatCompletionFunctionTool[],
|
||||
options?: {
|
||||
forceModelProvider?: AIProviderModel
|
||||
anthropicClient?: Anthropic
|
||||
}
|
||||
): Promise<MessageStream> {
|
||||
const { provider, config } = getProviderAndCompletionConfig({ messages, stream: true })
|
||||
const { provider, config } = getProviderAndCompletionConfig({
|
||||
messages,
|
||||
stream: true,
|
||||
forceModelProvider: options?.forceModelProvider
|
||||
})
|
||||
const { system, messages: anthropicMessages } = convertOpenAIToAnthropicMessages(messages)
|
||||
const anthropicTools = convertOpenAIToolsToAnthropic(tools)
|
||||
|
||||
const anthropicClient = workspaceAIClients.getAnthropicClient()
|
||||
const client = options?.anthropicClient ?? workspaceAIClients.getAnthropicClient()
|
||||
|
||||
const anthropicParams = {
|
||||
model: config.model,
|
||||
@@ -36,7 +46,7 @@ export async function getAnthropicCompletion(
|
||||
...(typeof config.temperature === 'number' && { temperature: config.temperature })
|
||||
}
|
||||
|
||||
const stream = anthropicClient.messages.stream(anthropicParams, {
|
||||
const stream = client.messages.stream(anthropicParams, {
|
||||
signal: abortController.signal,
|
||||
headers: {
|
||||
'X-Provider': provider,
|
||||
@@ -58,7 +68,8 @@ export async function parseAnthropicCompletion(
|
||||
addedMessages: ChatCompletionMessageParam[],
|
||||
tools: Tool<any>[],
|
||||
helpers: any,
|
||||
abortController?: AbortController
|
||||
abortController?: AbortController,
|
||||
options?: { workspace?: string }
|
||||
): Promise<boolean> {
|
||||
let toolCallsToProcess: ChatCompletionMessageFunctionToolCall[] = []
|
||||
let error = null
|
||||
@@ -209,7 +220,8 @@ export async function parseAnthropicCompletion(
|
||||
tools,
|
||||
toolCall,
|
||||
helpers,
|
||||
toolCallbacks: callbacks
|
||||
toolCallbacks: callbacks,
|
||||
workspace: options?.workspace
|
||||
})
|
||||
messages.push(messageToAdd)
|
||||
addedMessages.push(messageToAdd)
|
||||
|
||||
@@ -0,0 +1,211 @@
|
||||
import OpenAI from 'openai'
|
||||
import Anthropic from '@anthropic-ai/sdk'
|
||||
import type {
|
||||
ChatCompletionMessageParam,
|
||||
ChatCompletionSystemMessageParam,
|
||||
ChatCompletionUserMessageParam
|
||||
} from 'openai/resources/chat/completions.mjs'
|
||||
import type { AIProviderModel } from '$lib/gen'
|
||||
import { getCompletion, parseOpenAICompletion } from '../lib'
|
||||
import { getAnthropicCompletion, parseAnthropicCompletion } from './anthropic'
|
||||
import {
|
||||
getOpenAIResponsesCompletion,
|
||||
parseOpenAIResponsesCompletion
|
||||
} from './openai-responses'
|
||||
import type { Tool, ToolCallbacks } from './shared'
|
||||
|
||||
export interface ChatClients {
|
||||
openai: OpenAI
|
||||
anthropic: Anthropic
|
||||
}
|
||||
|
||||
export interface ChatLoopConfig {
|
||||
messages: ChatCompletionMessageParam[]
|
||||
/**
|
||||
* System message, tools, helpers, and modelProvider are re-read from this config
|
||||
* on every iteration. Callers can use JS getters to provide dynamic values
|
||||
* (e.g. AIChatManager uses getters so mode changes mid-loop take effect).
|
||||
*/
|
||||
systemMessage: ChatCompletionSystemMessageParam
|
||||
tools: Tool<any>[]
|
||||
helpers: any
|
||||
abortController: AbortController
|
||||
callbacks: ToolCallbacks & {
|
||||
onNewToken: (token: string) => void
|
||||
onMessageEnd: () => void
|
||||
}
|
||||
modelProvider: AIProviderModel
|
||||
clients: ChatClients
|
||||
workspace: string
|
||||
/** Maximum iterations for the loop. undefined = unlimited (production). */
|
||||
maxIterations?: number
|
||||
skipResponsesApi?: boolean
|
||||
onSkipResponsesApi?: () => void
|
||||
/** Return a pending user message to inject between iterations, or undefined. */
|
||||
getPendingUserMessage?: () => ChatCompletionUserMessageParam | undefined
|
||||
/** Called before each iteration (e.g. to refresh tool schemas). */
|
||||
onBeforeIteration?: (tools: Tool<any>[], helpers: any) => Promise<void>
|
||||
}
|
||||
|
||||
export interface ChatLoopResult {
|
||||
addedMessages: ChatCompletionMessageParam[]
|
||||
}
|
||||
|
||||
export async function runChatLoop(config: ChatLoopConfig): Promise<ChatLoopResult> {
|
||||
const {
|
||||
messages,
|
||||
abortController,
|
||||
callbacks,
|
||||
clients,
|
||||
workspace,
|
||||
maxIterations,
|
||||
onSkipResponsesApi,
|
||||
getPendingUserMessage,
|
||||
onBeforeIteration
|
||||
} = config
|
||||
let skipResponsesApi = config.skipResponsesApi ?? false
|
||||
|
||||
const addedMessages: ChatCompletionMessageParam[] = []
|
||||
let iterations = 0
|
||||
|
||||
while (true) {
|
||||
if (maxIterations !== undefined && iterations >= maxIterations) {
|
||||
break
|
||||
}
|
||||
iterations++
|
||||
|
||||
// Re-read these from config each iteration so that mode changes
|
||||
// (e.g. changeModeTool in Navigator) take effect immediately.
|
||||
// Callers can use JS getter properties to provide dynamic values.
|
||||
const tools = config.tools
|
||||
const helpers = config.helpers
|
||||
const systemMessage = config.systemMessage
|
||||
const modelProvider = config.modelProvider
|
||||
|
||||
if (onBeforeIteration) {
|
||||
await onBeforeIteration(tools, helpers)
|
||||
}
|
||||
|
||||
const pendingUserMessage = getPendingUserMessage?.()
|
||||
|
||||
const isOpenAI =
|
||||
modelProvider.provider === 'openai' || modelProvider.provider === 'azure_openai'
|
||||
const isAnthropic = modelProvider.provider === 'anthropic'
|
||||
|
||||
const messageParams = [
|
||||
systemMessage,
|
||||
...messages,
|
||||
...(pendingUserMessage ? [pendingUserMessage] : [])
|
||||
]
|
||||
const toolDefs = tools.map((t) => t.def)
|
||||
const parseOptions = { workspace }
|
||||
|
||||
if (isOpenAI) {
|
||||
let useCompletionsApi = skipResponsesApi
|
||||
if (!skipResponsesApi) {
|
||||
try {
|
||||
const completion = await getOpenAIResponsesCompletion(
|
||||
messageParams,
|
||||
abortController,
|
||||
toolDefs,
|
||||
{
|
||||
forceModelProvider: modelProvider,
|
||||
openaiClient: clients.openai
|
||||
}
|
||||
)
|
||||
const continueCompletion = await parseOpenAIResponsesCompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers,
|
||||
parseOptions
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
} catch (err) {
|
||||
console.warn(
|
||||
'OpenAI Responses API failed, falling back to Completions API:',
|
||||
err
|
||||
)
|
||||
const errorMessage = err instanceof Error ? err.message : String(err)
|
||||
if (errorMessage.includes('Responses API is not enabled')) {
|
||||
skipResponsesApi = true
|
||||
onSkipResponsesApi?.()
|
||||
}
|
||||
useCompletionsApi = true
|
||||
}
|
||||
}
|
||||
|
||||
if (useCompletionsApi) {
|
||||
const completion = await getCompletion(messageParams, abortController, toolDefs, {
|
||||
forceCompletions: true,
|
||||
forceModelProvider: modelProvider,
|
||||
openaiClient: clients.openai
|
||||
})
|
||||
const continueCompletion = await parseOpenAICompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers,
|
||||
undefined,
|
||||
parseOptions
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
}
|
||||
} else if (isAnthropic) {
|
||||
const completion = await getAnthropicCompletion(
|
||||
messageParams,
|
||||
abortController,
|
||||
toolDefs,
|
||||
{
|
||||
forceModelProvider: modelProvider,
|
||||
anthropicClient: clients.anthropic
|
||||
}
|
||||
)
|
||||
if (completion) {
|
||||
const continueCompletion = await parseAnthropicCompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers,
|
||||
abortController,
|
||||
parseOptions
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const completion = await getCompletion(messageParams, abortController, toolDefs, {
|
||||
forceModelProvider: modelProvider,
|
||||
openaiClient: clients.openai
|
||||
})
|
||||
if (completion) {
|
||||
const continueCompletion = await parseOpenAICompletion(
|
||||
completion,
|
||||
callbacks,
|
||||
messages,
|
||||
addedMessages,
|
||||
tools,
|
||||
helpers,
|
||||
undefined,
|
||||
parseOptions
|
||||
)
|
||||
if (!continueCompletion) {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return { addedMessages }
|
||||
}
|
||||
@@ -125,15 +125,24 @@ function convertCompletionConfigToResponsesConfig(
|
||||
export async function getOpenAIResponsesCompletion(
|
||||
messages: ChatCompletionMessageParam[],
|
||||
abortController: AbortController,
|
||||
tools?: OpenAI.Chat.Completions.ChatCompletionTool[]
|
||||
tools?: OpenAI.Chat.Completions.ChatCompletionTool[],
|
||||
options?: {
|
||||
forceModelProvider?: AIProviderModel
|
||||
openaiClient?: OpenAI
|
||||
}
|
||||
) {
|
||||
const { provider, config } = getProviderAndCompletionConfig({ messages, stream: true, tools })
|
||||
const { provider, config } = getProviderAndCompletionConfig({
|
||||
messages,
|
||||
stream: true,
|
||||
tools,
|
||||
forceModelProvider: options?.forceModelProvider
|
||||
})
|
||||
const { instructions, input } = convertMessagesToResponsesInput(messages)
|
||||
const responsesConfig = convertCompletionConfigToResponsesConfig(config)
|
||||
|
||||
const openaiClient = workspaceAIClients.getOpenaiClient()
|
||||
const client = options?.openaiClient ?? workspaceAIClients.getOpenaiClient()
|
||||
|
||||
const runner = openaiClient.responses.stream(
|
||||
const runner = client.responses.stream(
|
||||
{
|
||||
...responsesConfig,
|
||||
input,
|
||||
@@ -208,7 +217,8 @@ export async function parseOpenAIResponsesCompletion(
|
||||
messages: ChatCompletionMessageParam[],
|
||||
addedMessages: ChatCompletionMessageParam[],
|
||||
tools: Tool<any>[],
|
||||
helpers: any
|
||||
helpers: any,
|
||||
options?: { workspace?: string }
|
||||
): Promise<boolean> {
|
||||
let toolCallsToProcess: ChatCompletionMessageFunctionToolCall[] = []
|
||||
let error: OpenAIError | ResponseErrorEvent | null = null
|
||||
@@ -342,7 +352,8 @@ export async function parseOpenAIResponsesCompletion(
|
||||
tools,
|
||||
toolCall,
|
||||
helpers,
|
||||
toolCallbacks: callbacks
|
||||
toolCallbacks: callbacks,
|
||||
workspace: options?.workspace
|
||||
})
|
||||
messages.push(messageToAdd)
|
||||
addedMessages.push(messageToAdd)
|
||||
|
||||
@@ -417,12 +417,14 @@ export async function processToolCall<T>({
|
||||
tools,
|
||||
toolCall,
|
||||
helpers,
|
||||
toolCallbacks
|
||||
toolCallbacks,
|
||||
workspace
|
||||
}: {
|
||||
tools: Tool<T>[]
|
||||
toolCall: ChatCompletionMessageFunctionToolCall
|
||||
helpers: T
|
||||
toolCallbacks: ToolCallbacks
|
||||
workspace?: string
|
||||
}): Promise<ChatCompletionMessageParam> {
|
||||
try {
|
||||
const args = JSON.parse(toolCall.function.arguments || '{}')
|
||||
@@ -472,7 +474,7 @@ export async function processToolCall<T>({
|
||||
tools,
|
||||
functionName: toolCall.function.name,
|
||||
args,
|
||||
workspace: get(workspaceStore) ?? '',
|
||||
workspace: workspace ?? get(workspaceStore) ?? '',
|
||||
helpers,
|
||||
toolCallbacks,
|
||||
toolId: toolCall.id
|
||||
|
||||
@@ -296,7 +296,12 @@ function getModelSpecificConfig(
|
||||
) {
|
||||
const defaultMaxTokens = getModelMaxTokens(modelProvider.provider, modelProvider.model)
|
||||
const modelKey = `${modelProvider.provider}:${modelProvider.model}`
|
||||
const customMaxTokensStore = get(copilotInfo)?.maxTokensPerModel
|
||||
let customMaxTokensStore: Record<string, number> | undefined
|
||||
try {
|
||||
customMaxTokensStore = get(copilotInfo)?.maxTokensPerModel
|
||||
} catch {
|
||||
// copilotInfo store may not be initialized in vitest
|
||||
}
|
||||
const maxTokens = customMaxTokensStore?.[modelKey] ?? defaultMaxTokens
|
||||
if (
|
||||
(modelProvider.provider === 'openai' || modelProvider.provider === 'azure_openai') &&
|
||||
@@ -876,9 +881,16 @@ export async function getCompletion(
|
||||
tools?: OpenAI.Chat.Completions.ChatCompletionTool[],
|
||||
options?: {
|
||||
forceCompletions?: boolean
|
||||
forceModelProvider?: AIProviderModel
|
||||
openaiClient?: OpenAI
|
||||
}
|
||||
): Promise<Stream<ChatCompletionChunk>> {
|
||||
const { provider, config } = getProviderAndCompletionConfig({ messages, stream: true, tools })
|
||||
const { provider, config } = getProviderAndCompletionConfig({
|
||||
messages,
|
||||
stream: true,
|
||||
tools,
|
||||
forceModelProvider: options?.forceModelProvider
|
||||
})
|
||||
|
||||
// Use Responses API for OpenAI and Azure OpenAI
|
||||
if ((provider === 'openai' || provider === 'azure_openai') && !options?.forceCompletions) {
|
||||
@@ -891,8 +903,8 @@ export async function getCompletion(
|
||||
}
|
||||
|
||||
// Use Completions API for other providers
|
||||
const openaiClient = workspaceAIClients.getOpenaiClient()
|
||||
const completion = openaiClient.chat.completions.create(config, {
|
||||
const client = options?.openaiClient ?? workspaceAIClients.getOpenaiClient()
|
||||
const completion = client.chat.completions.create(config, {
|
||||
signal: abortController.signal,
|
||||
headers: {
|
||||
'X-Provider': provider
|
||||
@@ -921,7 +933,8 @@ export async function parseOpenAICompletion(
|
||||
addedMessages: ChatCompletionMessageParam[],
|
||||
tools: Tool<any>[],
|
||||
helpers: any,
|
||||
_abortController?: AbortController // unused, for signature compatibility with parseAnthropicCompletion
|
||||
_abortController?: AbortController, // unused, for signature compatibility with parseAnthropicCompletion
|
||||
options?: { workspace?: string }
|
||||
): Promise<boolean> {
|
||||
const finalToolCalls: Record<number, ChatCompletionChunk.Choice.Delta.ToolCall> = {}
|
||||
let malformedFunctionCallError = false
|
||||
@@ -1060,7 +1073,8 @@ export async function parseOpenAICompletion(
|
||||
tools,
|
||||
toolCall,
|
||||
helpers,
|
||||
toolCallbacks: callbacks
|
||||
toolCallbacks: callbacks,
|
||||
workspace: options?.workspace
|
||||
})
|
||||
messages.push(messageToAdd)
|
||||
addedMessages.push(messageToAdd)
|
||||
|
||||
Reference in New Issue
Block a user