mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-10-03 16:02:12 +00:00
weekly ai evals on current models, and claude 5.5/gpt-6 support (#11409)
* feat: run ai evals weekly on current models and post results to a dashboard Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * feat: add current flagship models, a reasoning flag and claude 5.5 defaults Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: never send a reasoning disable claude 5.5 or gpt-6-astra reject, and treat gpt-6 as a reasoning model Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: address review on gpt-6 support, chat completions tools and model metadata Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: leave tiered gpt-6 unpriced and drop the off sentinel on gpt-5 and o-series Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * docs: point the ai_evals readme at the model registry instead of copying it Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * refactor: encode the reasoning rules as per-family maps with a shared parity fixture Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: keep the chat completions tools rule open-ended past gpt-5.6 and scope the parity fixture Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
94ac34a489
commit
e117871bec
@@ -0,0 +1,171 @@
|
||||
name: AI Evals (scheduled)
|
||||
|
||||
# Full ai_evals suites on current models, posted to the AI evals dashboard in
|
||||
# the windmill-prod workspace (f/ai/ai_evals_dashboard) so quality is tracked
|
||||
# over time. Weekly, since one pass of every suite at --runs 3 costs about 50M
|
||||
# tokens per model: every suite runs on MODELS, and global (the mode users get)
|
||||
# also runs on GLOBAL_EXTRA_MODELS, one flagship per other provider. Run it by
|
||||
# hand to measure a branch against main.
|
||||
on:
|
||||
schedule:
|
||||
- cron: "0 3 * * 1"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
modes:
|
||||
description: "Space-separated modes"
|
||||
default: "global flow app script cli"
|
||||
models:
|
||||
description: "Space-separated model aliases (bun run cli -- models)"
|
||||
default: "sonnet-5.5"
|
||||
global_extra_models:
|
||||
description: "Extra model aliases for global mode only"
|
||||
default: "gpt-6-astra gemini-3.8-flash"
|
||||
runs:
|
||||
description: "Runs per case"
|
||||
default: "3"
|
||||
reasoning:
|
||||
description: "Reasoning effort for frontend modes (empty: the product default)"
|
||||
default: ""
|
||||
|
||||
concurrency:
|
||||
group: ai-evals-scheduled-${{ github.ref }}
|
||||
|
||||
env:
|
||||
MODELS: ${{ inputs.models || 'sonnet-5.5' }}
|
||||
GLOBAL_EXTRA_MODELS: ${{ inputs.global_extra_models || 'gpt-6-astra gemini-3.8-flash' }}
|
||||
RUNS: ${{ inputs.runs || '3' }}
|
||||
REASONING: ${{ inputs.reasoning }}
|
||||
INGEST_URL: https://app.windmill.dev/api/r/f/ai/ingest_ai_eval_run
|
||||
|
||||
jobs:
|
||||
setup:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
modes: ${{ steps.modes.outputs.modes }}
|
||||
steps:
|
||||
- id: modes
|
||||
env:
|
||||
MODES: ${{ inputs.modes || 'global flow app script cli' }}
|
||||
run: |
|
||||
echo "modes=$(jq -cn --arg m "$MODES" '$m | split(" ")
|
||||
| map(select(IN("global", "flow", "app", "script", "cli")))')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
evals:
|
||||
needs: setup
|
||||
runs-on: ubicloud-standard-16
|
||||
timeout-minutes: 330
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
mode: ${{ fromJSON(needs.setup.outputs.modes) }}
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:16
|
||||
ports:
|
||||
- 5432:5432
|
||||
env:
|
||||
POSTGRES_DB: windmill
|
||||
POSTGRES_PASSWORD: changeme
|
||||
options: >-
|
||||
--health-cmd pg_isready --health-interval 10s --health-timeout 5s
|
||||
--health-retries 5
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||
with:
|
||||
cache-workspaces: backend
|
||||
toolchain: 1.97.0
|
||||
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: 1.4.0
|
||||
|
||||
- uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: "24"
|
||||
|
||||
- name: Build Windmill
|
||||
working-directory: ./backend
|
||||
env:
|
||||
SQLX_OFFLINE: true
|
||||
CARGO_BUILD_JOBS: 12
|
||||
RUSTFLAGS: ""
|
||||
run: cargo build --features quickjs
|
||||
|
||||
- name: Start Windmill
|
||||
working-directory: ./backend
|
||||
env:
|
||||
DATABASE_URL: postgres://postgres:changeme@localhost:5432/windmill
|
||||
RUST_LOG: info
|
||||
run: |
|
||||
mkdir -p ../ai_evals/logs
|
||||
./target/debug/windmill > ../ai_evals/logs/windmill.log 2>&1 &
|
||||
for i in $(seq 1 60); do
|
||||
curl -sf http://localhost:8000/api/version > /dev/null 2>&1 && break
|
||||
sleep 2
|
||||
done
|
||||
curl -sf http://localhost:8000/api/version > /dev/null || { tail -50 ../ai_evals/logs/windmill.log; exit 1; }
|
||||
|
||||
- name: Install frontend deps + generate client
|
||||
working-directory: ./frontend
|
||||
run: |
|
||||
npm ci
|
||||
npm run generate-backend-client
|
||||
|
||||
- name: Install CLI deps + generate CLI client
|
||||
working-directory: ./cli
|
||||
run: bun install && ./gen_wm_client.sh && ./windmill-utils-internal/gen_wm_client.sh
|
||||
|
||||
- name: Run ${{ matrix.mode }} evals
|
||||
working-directory: ./ai_evals
|
||||
env:
|
||||
MODE: ${{ matrix.mode }}
|
||||
WMILL_AI_EVAL_BACKEND_URL: http://localhost:8000
|
||||
WMILL_AI_EVAL_BACKEND_WORKSPACE: integration-tests
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
|
||||
DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }}
|
||||
run: |
|
||||
bun install
|
||||
mkdir -p results
|
||||
models="$MODELS"
|
||||
[ "$MODE" = global ] && models="$models $GLOBAL_EXTRA_MODELS"
|
||||
reasoning=()
|
||||
[ -n "$REASONING" ] && [ "$MODE" != cli ] && reasoning=(--reasoning "$REASONING")
|
||||
for m in $models; do
|
||||
bun run cli -- run "$MODE" --model "$m" --runs "$RUNS" "${reasoning[@]}" \
|
||||
--output "$PWD/results/$MODE-$m.json" || echo "::warning::$MODE on $m errored"
|
||||
done
|
||||
|
||||
- name: Post results to the dashboard
|
||||
if: always()
|
||||
working-directory: ./ai_evals
|
||||
env:
|
||||
MODE: ${{ matrix.mode }}
|
||||
INGEST_TOKEN: ${{ secrets.AI_EVALS_INGEST_TOKEN }}
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
run: |
|
||||
[ -n "$INGEST_TOKEN" ] || { echo "::warning::AI_EVALS_INGEST_TOKEN is not set"; exit 0; }
|
||||
shopt -s nullglob
|
||||
for f in results/"$MODE"-*.json; do
|
||||
# Keep what the dashboard reads; the full traces stay in the run's artifacts.
|
||||
jq -c --arg ref "$GITHUB_REF" --arg trigger "$GITHUB_EVENT_NAME" --arg url "$RUN_URL" '
|
||||
{result_json: (del(.cases[].prompt, .cases[].initialPath, .cases[].expectedPath)
|
||||
| .cases[].attempts[] |= {attempt, passed, durationMs, toolCallCount, judgeScore,
|
||||
judgeSummary, error, tokenUsage, checks: [.checks[]? | {name, passed}]}),
|
||||
git_ref: $ref, trigger: $trigger, run_url: $url}' "$f" \
|
||||
| curl -sf --retry 3 -X POST "$INGEST_URL" \
|
||||
-H "Authorization: Bearer $INGEST_TOKEN" -H 'content-type: application/json' \
|
||||
--data-binary @- && echo " <- $f" || echo "::warning::failed to post $f"
|
||||
done
|
||||
|
||||
- name: Archive logs and results
|
||||
uses: actions/upload-artifact@v4
|
||||
if: always()
|
||||
with:
|
||||
name: ai-evals-${{ matrix.mode }}
|
||||
path: |
|
||||
ai_evals/logs
|
||||
ai_evals/results
|
||||
+3
-14
@@ -77,31 +77,20 @@ Public CLI surface:
|
||||
- `--verbose`: stream assistant output for frontend runs
|
||||
- `--skip-judge`: skip LLM judge scoring for the run
|
||||
- `--execution-only`: only require the model/proxy/frontend loop to complete; skip validators, tool expectations, backend artifact validation, and judge scoring
|
||||
- `--reasoning <effort>`: reasoning effort for frontend modes (`off`, `low`, `medium`, `high`, `max`, …); without it the product's default applies (`high` on models that can reason). The effort is appended to the recorded model label (`anthropic:claude-sonnet-5-5@max`)
|
||||
- `--record`: append a compact tracked summary line to `ai_evals/history/<mode>.jsonl` for full-suite runs only
|
||||
- `--backend-validation <mode>`: optional backend smoke validation (`off` or `preview`) for `script` and `flow` evals
|
||||
|
||||
## Models
|
||||
|
||||
Use `bun run cli -- models` to see the current aliases.
|
||||
|
||||
Today:
|
||||
|
||||
- `haiku`
|
||||
- `sonnet`
|
||||
- `opus`
|
||||
- `4o`
|
||||
- `gpt-5.5`
|
||||
- `gemini-3-flash-preview`
|
||||
- `gemini-3.1-pro-preview`
|
||||
- `deepseek-v4-flash`
|
||||
- `deepseek-v4-pro`
|
||||
Use `bun run cli -- models` to see the current aliases; `core/models.ts` is the list.
|
||||
|
||||
Notes:
|
||||
|
||||
- the command also prints accepted alias spellings such as `gpt-4o`, `gpt-55`, `claude-opus-4.6`, and `claude-haiku-4.5`
|
||||
- frontend modes (`flow`, `script`, `app`, `global`) can use Anthropic, OpenAI, Gemini, and DeepSeek-backed aliases
|
||||
- `cli` mode always uses the Anthropic agent SDK, so only Anthropic aliases are valid there
|
||||
- the judge model is separate and currently defaults to `claude-sonnet-4-6`; use `--skip-judge` for deterministic-only runs
|
||||
- the judge model is separate and currently defaults to `claude-sonnet-5-5`; use `--skip-judge` for deterministic-only runs
|
||||
|
||||
## Case Format
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ import { runEval } from "../shared";
|
||||
import type { ModeRunContext } from "../../../../core/types";
|
||||
import type { TokenUsage, ToolCallDetail } from "../shared/types";
|
||||
import type { WindmillBackendSettings } from "../../../../core/windmillBackendSettings";
|
||||
import { evalReasoningEffort } from "../shared/providerConfig";
|
||||
|
||||
export interface ScriptEvalResult {
|
||||
success: boolean;
|
||||
@@ -41,14 +42,15 @@ export interface ScriptEvalOptions {
|
||||
function resolveModelProvider(
|
||||
model: string,
|
||||
provider?: AIProvider,
|
||||
): AIProviderModel {
|
||||
): AIProviderModel & { reasoning?: string } {
|
||||
const reasoning = evalReasoningEffort();
|
||||
if (provider) {
|
||||
return { provider, model };
|
||||
return { provider, model, reasoning };
|
||||
}
|
||||
if (model.startsWith("claude")) {
|
||||
return { provider: "anthropic", model };
|
||||
return { provider: "anthropic", model, reasoning };
|
||||
}
|
||||
return { provider: "openai", model };
|
||||
return { provider: "openai", model, reasoning };
|
||||
}
|
||||
|
||||
export async function runScriptEval(
|
||||
|
||||
@@ -12,6 +12,12 @@ export interface EvalClients {
|
||||
export interface ResolvedEvalModelProvider {
|
||||
provider: FrontendEvalProvider;
|
||||
model: string;
|
||||
reasoning?: string;
|
||||
}
|
||||
|
||||
/** `run --reasoning` hands the effort to the frontend runtime through the environment. */
|
||||
export function evalReasoningEffort(): string | undefined {
|
||||
return process.env.WMILL_AI_EVAL_REASONING || undefined;
|
||||
}
|
||||
|
||||
export interface WindmillAiProxyClientConfig {
|
||||
@@ -73,6 +79,13 @@ export function createEvalClients(input: {
|
||||
export function resolveEvalModelProvider(
|
||||
model: string,
|
||||
provider?: FrontendEvalProvider,
|
||||
): ResolvedEvalModelProvider {
|
||||
return { ...resolveProvider(model, provider), reasoning: evalReasoningEffort() };
|
||||
}
|
||||
|
||||
function resolveProvider(
|
||||
model: string,
|
||||
provider?: FrontendEvalProvider,
|
||||
): ResolvedEvalModelProvider {
|
||||
if (provider) {
|
||||
return { provider, model };
|
||||
|
||||
+29
-91
@@ -5,8 +5,8 @@
|
||||
"": {
|
||||
"name": "windmill-ai-evals",
|
||||
"dependencies": {
|
||||
"@anthropic-ai/claude-agent-sdk": "^0.2.25",
|
||||
"@anthropic-ai/sdk": "^0.39.0",
|
||||
"@anthropic-ai/claude-agent-sdk": "^0.3.284",
|
||||
"@anthropic-ai/sdk": "^0.129.0",
|
||||
"commander": "^14.0.3",
|
||||
"openai": "^6.9.1",
|
||||
"yaml": "^2.8.3",
|
||||
@@ -18,66 +18,44 @@
|
||||
},
|
||||
},
|
||||
"packages": {
|
||||
"@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.2.87", "", { "dependencies": { "@anthropic-ai/sdk": "^0.74.0", "@modelcontextprotocol/sdk": "^1.27.1" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "^0.34.2", "@img/sharp-darwin-x64": "^0.34.2", "@img/sharp-linux-arm": "^0.34.2", "@img/sharp-linux-arm64": "^0.34.2", "@img/sharp-linux-x64": "^0.34.2", "@img/sharp-linuxmusl-arm64": "^0.34.2", "@img/sharp-linuxmusl-x64": "^0.34.2", "@img/sharp-win32-arm64": "^0.34.2", "@img/sharp-win32-x64": "^0.34.2" }, "peerDependencies": { "zod": "^4.0.0" } }, "sha512-WWmgBPxPhBOvNT0ujI8vPTI2lK+w5YEkEZ/y1mH0EDkK/0kBnxVJNhCtG5vnueiAViwLoUOFn66pbkDiivijdA=="],
|
||||
"@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.3.284", "", { "optionalDependencies": { "@anthropic-ai/claude-agent-sdk-darwin-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-darwin-x64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-x64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-x64-musl": "0.3.284", "@anthropic-ai/claude-agent-sdk-win32-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-win32-x64": "0.3.284" }, "peerDependencies": { "@anthropic-ai/sdk": ">=0.93.0", "@modelcontextprotocol/sdk": "^1.29.0", "zod": "^4.0.0" } }, "sha512-NSoJwEq6nFSf8dtaacYx37QdGgqApI3eHFUlxclMLYi8irb6ZJwEaUnPnLUCyAyg9W/tMkgZC0GXWLRCc01I0w=="],
|
||||
|
||||
"@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.39.0", "", { "dependencies": { "@types/node": "^18.11.18", "@types/node-fetch": "^2.6.4", "abort-controller": "^3.0.0", "agentkeepalive": "^4.2.1", "form-data-encoder": "1.7.2", "formdata-node": "^4.3.2", "node-fetch": "^2.6.7" } }, "sha512-eMyDIPRZbt1CCLErRCi3exlAvNkBtRe+kW5vvJyef93PmNr/clstYgHhtvmkxN82nlKgzyGPCyGxrm0JQ1ZIdg=="],
|
||||
"@anthropic-ai/claude-agent-sdk-darwin-arm64": ["@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.284", "", { "os": "darwin", "cpu": "arm64" }, "sha512-gKY9MUjY83398uCiPLHsd87kyzu7agIM7ApqWpJkpSINep6hxx4rNoR8bUbNWg49Aoe/PW/DJkQByAQuNzR9rg=="],
|
||||
|
||||
"@babel/runtime": ["@babel/runtime@7.29.2", "", {}, "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g=="],
|
||||
"@anthropic-ai/claude-agent-sdk-darwin-x64": ["@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.284", "", { "os": "darwin", "cpu": "x64" }, "sha512-P+q6Z7sKeYz99uE7RJc4au1IK+4SiWe69pyHY3QXMnjZq8hSFU/5Mo8iSurLk9jRlihsMltV1rLA5naouYyCAQ=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk-linux-arm64": ["@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.284", "", { "os": "linux", "cpu": "arm64" }, "sha512-LDpuYDaz+pCdG29iy1pw4P1D2YVtpZOb4rWOPxxSpg2fFyUeQ5WrfnuszOxhDWpNh1k1ekDiKm3A9OzGcrkdtA=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk-linux-arm64-musl": ["@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.284", "", { "os": "linux", "cpu": "arm64" }, "sha512-U3/uAv1TS8sMWsgWQSxYbdND5AaRXXpP5RxHU1oYC5ZizlZE8/xAVum9dj1QX3rVZJg8EzsqZveq/AkEv5ABfQ=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk-linux-x64": ["@anthropic-ai/claude-agent-sdk-linux-x64@0.3.284", "", { "os": "linux", "cpu": "x64" }, "sha512-yGytBCCwJvWeg1FzFpgnLM9BOM2vKVExtv6pMJHMtF82J5yhKHcxodRaYlQVd/HzJJ2nruKb9EnrhRBZ//4yRw=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk-linux-x64-musl": ["@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.284", "", { "os": "linux", "cpu": "x64" }, "sha512-4x0Q8CFkCNbeENE7pRhuUmkwh/sQfTq1yWi4ZE56stve9pgoKzu6ZDa1SEtvK1YIooGE/qM5rVS8xFLE1gbQ0w=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk-win32-arm64": ["@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.284", "", { "os": "win32", "cpu": "arm64" }, "sha512-zOXkdPHhElyxyFJ6eY5R+YGUswqBDZ+ZHNaRmQGJ28RkQYgwG5imNy/nfFR+/CPGlMnH9ZzbGc2rKYApm1K4Tg=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk-win32-x64": ["@anthropic-ai/claude-agent-sdk-win32-x64@0.3.284", "", { "os": "win32", "cpu": "x64" }, "sha512-VGaFRDCOPloj5IjvJTyy5JsSl9sE2HR3W+A6f/cjgh23/7Y5vK7/ka+1JQD+AkoGO9tbHqP2O6god3IBrLvOaw=="],
|
||||
|
||||
"@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.129.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1", "standardwebhooks": "^1.0.0" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-MH7LB20kNpGLUpPTe2OG5XvL/KhX8HUO9XyBaFjgFk6TLS4Xjt0VPIuMKywe7POKKPslX+QxSlVok9e+8kG5Nw=="],
|
||||
|
||||
"@babel/runtime": ["@babel/runtime@7.29.7", "", {}, "sha512-Nq8OhGWiZIZGV6hLHoyAKLLcJihP/xFeBMGJoUrxTX2psI8dCifzLhZISFb+VWS3wFMRDmCGw5R+dOySCqPLhw=="],
|
||||
|
||||
"@hono/node-server": ["@hono/node-server@1.19.12", "", { "peerDependencies": { "hono": "^4" } }, "sha512-txsUW4SQ1iilgE0l9/e9VQWmELXifEFvmdA1j6WFh/aFPj99hIntrSsq/if0UWyGVkmrRPKA1wCeP+UCr1B9Uw=="],
|
||||
|
||||
"@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="],
|
||||
|
||||
"@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.2.4" }, "os": "darwin", "cpu": "x64" }, "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw=="],
|
||||
|
||||
"@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.2.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g=="],
|
||||
|
||||
"@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.2.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg=="],
|
||||
|
||||
"@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.2.4", "", { "os": "linux", "cpu": "arm" }, "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A=="],
|
||||
|
||||
"@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw=="],
|
||||
|
||||
"@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw=="],
|
||||
|
||||
"@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw=="],
|
||||
|
||||
"@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg=="],
|
||||
|
||||
"@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.2.4" }, "os": "linux", "cpu": "arm" }, "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw=="],
|
||||
|
||||
"@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg=="],
|
||||
|
||||
"@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ=="],
|
||||
|
||||
"@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg=="],
|
||||
|
||||
"@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q=="],
|
||||
|
||||
"@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.34.5", "", { "os": "win32", "cpu": "arm64" }, "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g=="],
|
||||
|
||||
"@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.34.5", "", { "os": "win32", "cpu": "x64" }, "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw=="],
|
||||
|
||||
"@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.29.0", "", { "dependencies": { "@hono/node-server": "^1.19.9", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.2.1", "express-rate-limit": "^8.2.1", "hono": "^4.11.4", "jose": "^6.1.3", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.1" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-zo37mZA9hJWpULgkRpowewez1y6ML5GsXJPY8FI0tBBCd77HEvza4jDqRKOXgHNn867PVGCyTdzqpz0izu5ZjQ=="],
|
||||
|
||||
"@stablelib/base64": ["@stablelib/base64@1.0.1", "", {}, "sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ=="],
|
||||
|
||||
"@types/bun": ["@types/bun@1.3.11", "", { "dependencies": { "bun-types": "1.3.11" } }, "sha512-5vPne5QvtpjGpsGYXiFyycfpDF2ECyPcTSsFBMa0fraoxiQyMJ3SmuQIGhzPg2WJuWxVBoxWJ2kClYTcw/4fAg=="],
|
||||
|
||||
"@types/node": ["@types/node@18.19.130", "", { "dependencies": { "undici-types": "~5.26.4" } }, "sha512-GRaXQx6jGfL8sKfaIDD6OupbIHBr9jv7Jnaml9tB7l4v068PAOXqfcujMMo5PhbIs6ggR1XODELqahT2R8v0fg=="],
|
||||
|
||||
"@types/node-fetch": ["@types/node-fetch@2.6.13", "", { "dependencies": { "@types/node": "*", "form-data": "^4.0.4" } }, "sha512-QGpRVpzSaUs30JBSGPjOg4Uveu384erbHBoT1zeONvyCfwQxIkUshLAOqN/k9EjGviPRmWTTe6aH2qySWKTVSw=="],
|
||||
|
||||
"abort-controller": ["abort-controller@3.0.0", "", { "dependencies": { "event-target-shim": "^5.0.0" } }, "sha512-h8lQ8tacZYnR3vNQTgibj+tODHI5/+l06Au2Pcriv/Gmet0eaj4TwWH41sO9wnHDiQsEj19q0drzdWdeAHtweg=="],
|
||||
"@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="],
|
||||
|
||||
"accepts": ["accepts@2.0.0", "", { "dependencies": { "mime-types": "^3.0.0", "negotiator": "^1.0.0" } }, "sha512-5cvg6CtKwfgdmVqY1WIiXKc3Q1bkRqGLi+2W/6ao+6Y7gu/RCwRuAhGEzh5B4KlszSuTLgZYuqFqo5bImjNKng=="],
|
||||
|
||||
"agentkeepalive": ["agentkeepalive@4.6.0", "", { "dependencies": { "humanize-ms": "^1.2.1" } }, "sha512-kja8j7PjmncONqaTsB8fQ+wE2mSU2DJ9D4XKoJ5PFWIdRMa6SLSN1ff4mOr4jCbfRSsxR4keIiySJU0N9T5hIQ=="],
|
||||
|
||||
"ajv": ["ajv@8.18.0", "", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-PlXPeEWMXMZ7sPYOHqmDyCJzcfNrUr3fGNKtezX14ykXOEIvyK81d+qydx89KY5O71FKMPaQ2vBfBFI5NHR63A=="],
|
||||
|
||||
"ajv-formats": ["ajv-formats@3.0.1", "", { "dependencies": { "ajv": "^8.0.0" } }, "sha512-8iUql50EUR+uUcdRQ3HDqa6EVyo3docL8g5WJ3FNcWmu62IbkGUue/pEyLBW8VGKKucTPgqeks4fIU1DA4yowQ=="],
|
||||
|
||||
"asynckit": ["asynckit@0.4.0", "", {}, "sha512-Oei9OH4tRh0YqU3GxhX79dM/mwVgvbZJaSNaRk+bshkj0S5cfHcgYakreBjrHwatXKbz+IoIdYLxrKim2MjW0Q=="],
|
||||
|
||||
"body-parser": ["body-parser@2.2.2", "", { "dependencies": { "bytes": "^3.1.2", "content-type": "^1.0.5", "debug": "^4.4.3", "http-errors": "^2.0.0", "iconv-lite": "^0.7.0", "on-finished": "^2.4.1", "qs": "^6.14.1", "raw-body": "^3.0.1", "type-is": "^2.0.1" } }, "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA=="],
|
||||
|
||||
"bun-types": ["bun-types@1.3.11", "", { "dependencies": { "@types/node": "*" } }, "sha512-1KGPpoxQWl9f6wcZh57LvrPIInQMn2TQ7jsgxqpRzg+l0QPOFvJVH7HmvHo/AiPgwXy+/Thf6Ov3EdVn1vOabg=="],
|
||||
@@ -88,8 +66,6 @@
|
||||
|
||||
"call-bound": ["call-bound@1.0.4", "", { "dependencies": { "call-bind-apply-helpers": "^1.0.2", "get-intrinsic": "^1.3.0" } }, "sha512-+ys997U96po4Kx/ABpBCqhA9EuxJaQWDQg7295H4hBphv3IZg0boBKuwYpt4YXp6MZ5AmZQnU/tyMTlRpaSejg=="],
|
||||
|
||||
"combined-stream": ["combined-stream@1.0.8", "", { "dependencies": { "delayed-stream": "~1.0.0" } }, "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg=="],
|
||||
|
||||
"commander": ["commander@14.0.3", "", {}, "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw=="],
|
||||
|
||||
"content-disposition": ["content-disposition@1.0.1", "", {}, "sha512-oIXISMynqSqm241k6kcQ5UwttDILMK4BiurCfGEREw6+X9jkkpEe5T9FZaApyLGGOnFuyMWZpdolTXMtvEJ08Q=="],
|
||||
@@ -106,8 +82,6 @@
|
||||
|
||||
"debug": ["debug@4.4.3", "", { "dependencies": { "ms": "^2.1.3" } }, "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA=="],
|
||||
|
||||
"delayed-stream": ["delayed-stream@1.0.0", "", {}, "sha512-ZySD7Nf91aLB0RxL4KGrKHBXl7Eds1DAmEdcoVawXnLD7SDhpNgtuII2aAkg7a7QS41jxPSZ17p4VdGnMHk3MQ=="],
|
||||
|
||||
"depd": ["depd@2.0.0", "", {}, "sha512-g7nH6P6dyDioJogAAGprGpCtVImJhpPk/roCzdb3fIh61/s/nPsfR6onyMwkCAR/OlC3yBC0lESvUoQEAssIrw=="],
|
||||
|
||||
"dunder-proto": ["dunder-proto@1.0.1", "", { "dependencies": { "call-bind-apply-helpers": "^1.0.1", "es-errors": "^1.3.0", "gopd": "^1.2.0" } }, "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A=="],
|
||||
@@ -122,14 +96,10 @@
|
||||
|
||||
"es-object-atoms": ["es-object-atoms@1.1.1", "", { "dependencies": { "es-errors": "^1.3.0" } }, "sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA=="],
|
||||
|
||||
"es-set-tostringtag": ["es-set-tostringtag@2.1.0", "", { "dependencies": { "es-errors": "^1.3.0", "get-intrinsic": "^1.2.6", "has-tostringtag": "^1.0.2", "hasown": "^2.0.2" } }, "sha512-j6vWzfrGVfyXxge+O0x5sh6cvxAog0a/4Rdd2K36zCMV5eJ+/+tOAngRO8cODMNWbVRdVlmGZQL2YS3yR8bIUA=="],
|
||||
|
||||
"escape-html": ["escape-html@1.0.3", "", {}, "sha512-NiSupZ4OeuGwr68lGIeym/ksIZMJodUGOSCZ/FSnTxcrekbvqrgdUxlJOMpijaKZVjAJrWrGs/6Jy8OMuyj9ow=="],
|
||||
|
||||
"etag": ["etag@1.8.1", "", {}, "sha512-aIL5Fx7mawVa300al2BnEE4iNvo1qETxLrPI/o05L7z6go7fCw1J6EQmbK4FmJ2AS7kgVF/KEZWufBfdClMcPg=="],
|
||||
|
||||
"event-target-shim": ["event-target-shim@5.0.1", "", {}, "sha512-i/2XbnSz/uxRCU6+NdVJgKWDTM427+MqYbkQzD321DuCQJUqOuJKIA0IM2+W2xtYHdKOmZ4dR6fExsd4SXL+WQ=="],
|
||||
|
||||
"eventsource": ["eventsource@3.0.7", "", { "dependencies": { "eventsource-parser": "^3.0.1" } }, "sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA=="],
|
||||
|
||||
"eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="],
|
||||
@@ -140,16 +110,12 @@
|
||||
|
||||
"fast-deep-equal": ["fast-deep-equal@3.1.3", "", {}, "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q=="],
|
||||
|
||||
"fast-sha256": ["fast-sha256@1.3.0", "", {}, "sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ=="],
|
||||
|
||||
"fast-uri": ["fast-uri@3.1.0", "", {}, "sha512-iPeeDKJSWf4IEOasVVrknXpaBV0IApz/gp7S2bb7Z4Lljbl2MGJRqInZiUrQwV16cpzw/D3S5j5Julj/gT52AA=="],
|
||||
|
||||
"finalhandler": ["finalhandler@2.1.1", "", { "dependencies": { "debug": "^4.4.0", "encodeurl": "^2.0.0", "escape-html": "^1.0.3", "on-finished": "^2.4.1", "parseurl": "^1.3.3", "statuses": "^2.0.1" } }, "sha512-S8KoZgRZN+a5rNwqTxlZZePjT/4cnm0ROV70LedRHZ0p8u9fRID0hJUZQpkKLzro8LfmC8sx23bY6tVNxv8pQA=="],
|
||||
|
||||
"form-data": ["form-data@4.0.5", "", { "dependencies": { "asynckit": "^0.4.0", "combined-stream": "^1.0.8", "es-set-tostringtag": "^2.1.0", "hasown": "^2.0.2", "mime-types": "^2.1.12" } }, "sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w=="],
|
||||
|
||||
"form-data-encoder": ["form-data-encoder@1.7.2", "", {}, "sha512-qfqtYan3rxrnCk1VYaA4H+Ms9xdpPqvLZa6xmMgFvhO32x7/3J/ExcTd6qpxM0vH2GdMI+poehyBZvqfMTto8A=="],
|
||||
|
||||
"formdata-node": ["formdata-node@4.4.1", "", { "dependencies": { "node-domexception": "1.0.0", "web-streams-polyfill": "4.0.0-beta.3" } }, "sha512-0iirZp3uVDjVGt9p49aTaqjk84TrglENEDuqfdlZQ1roC9CWlPk6Avf8EEnZNcAqPonwkG35x4n3ww/1THYAeQ=="],
|
||||
|
||||
"forwarded": ["forwarded@0.2.0", "", {}, "sha512-buRG0fpBtRHSTCOASe6hD258tEubFoRLb4ZNA6NxMVHNw2gOcwHo9wyablzMzOA5z9xA9L1KNjk/Nt6MT9aYow=="],
|
||||
|
||||
"fresh": ["fresh@2.0.0", "", {}, "sha512-Rx/WycZ60HOaqLKAi6cHRKKI7zxWbJ31MhntmtwMoaTeF7XFH9hhBp8vITaMidfljRQ6eYWCKkaTK+ykVJHP2A=="],
|
||||
@@ -164,16 +130,12 @@
|
||||
|
||||
"has-symbols": ["has-symbols@1.1.0", "", {}, "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ=="],
|
||||
|
||||
"has-tostringtag": ["has-tostringtag@1.0.2", "", { "dependencies": { "has-symbols": "^1.0.3" } }, "sha512-NqADB8VjPFLM2V0VvHUewwwsw0ZWBaIdgo+ieHtK3hasLz4qeCRjYcqfB6AQrBggRKppKF8L52/VqdVsO47Dlw=="],
|
||||
|
||||
"hasown": ["hasown@2.0.2", "", { "dependencies": { "function-bind": "^1.1.2" } }, "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ=="],
|
||||
|
||||
"hono": ["hono@4.12.9", "", {}, "sha512-wy3T8Zm2bsEvxKZM5w21VdHDDcwVS1yUFFY6i8UobSsKfFceT7TOwhbhfKsDyx7tYQlmRM5FLpIuYvNFyjctiA=="],
|
||||
|
||||
"http-errors": ["http-errors@2.0.1", "", { "dependencies": { "depd": "~2.0.0", "inherits": "~2.0.4", "setprototypeof": "~1.2.0", "statuses": "~2.0.2", "toidentifier": "~1.0.1" } }, "sha512-4FbRdAX+bSdmo4AUFuS0WNiPz8NgFt+r8ThgNWmlrjQjt1Q7ZR9+zTlce2859x4KSXrwIsaeTqDoKQmtP8pLmQ=="],
|
||||
|
||||
"humanize-ms": ["humanize-ms@1.2.1", "", { "dependencies": { "ms": "^2.0.0" } }, "sha512-Fl70vYtsAFb/C06PTS9dZBo7ihau+Tu/DNCk/OyHhea07S+aeMWpFFkUaXRa8fI+ScZbEI8dfSxwY7gxZ9SAVQ=="],
|
||||
|
||||
"iconv-lite": ["iconv-lite@0.7.2", "", { "dependencies": { "safer-buffer": ">= 2.1.2 < 3.0.0" } }, "sha512-im9DjEDQ55s9fL4EYzOAv0yMqmMBSZp6G0VvFyTMPKWxiSBHUj9NW/qqLmXUwXrrM7AvqSlTCfvqRb0cM8yYqw=="],
|
||||
|
||||
"inherits": ["inherits@2.0.4", "", {}, "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ=="],
|
||||
@@ -208,10 +170,6 @@
|
||||
|
||||
"negotiator": ["negotiator@1.0.0", "", {}, "sha512-8Ofs/AUQh8MaEcrlq5xOX0CQ9ypTF5dl78mjlMNfOK08fzpgTHQRQPBxcPlEtIw0yRpws+Zo/3r+5WRby7u3Gg=="],
|
||||
|
||||
"node-domexception": ["node-domexception@1.0.0", "", {}, "sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ=="],
|
||||
|
||||
"node-fetch": ["node-fetch@2.7.0", "", { "dependencies": { "whatwg-url": "^5.0.0" }, "peerDependencies": { "encoding": "^0.1.0" }, "optionalPeers": ["encoding"] }, "sha512-c4FRfUm/dbcWZ7U+1Wq0AwCyFL+3nt2bEw05wfxSz+DWpWsitgmSgYmy2dQdWyKC1694ELPqMs/YzUSNozLt8A=="],
|
||||
|
||||
"object-assign": ["object-assign@4.1.1", "", {}, "sha512-rJgTQnkUnH1sFw8yT6VSU3zD3sWmu6sZhIseY8VX+GRu3P6F7Fu+JNDoXfklElbLJSnc3FUQHVe4cU5hj+BcUg=="],
|
||||
|
||||
"object-inspect": ["object-inspect@1.13.4", "", {}, "sha512-W67iLl4J2EXEGTbfeHCffrjDfitvLANg0UlX3wFUUSTx92KXRFegMHUVgSqE+wvhAbi4WqjGg9czysTV2Epbew=="],
|
||||
@@ -262,30 +220,24 @@
|
||||
|
||||
"side-channel-weakmap": ["side-channel-weakmap@1.0.2", "", { "dependencies": { "call-bound": "^1.0.2", "es-errors": "^1.3.0", "get-intrinsic": "^1.2.5", "object-inspect": "^1.13.3", "side-channel-map": "^1.0.1" } }, "sha512-WPS/HvHQTYnHisLo9McqBHOJk2FkHO/tlpvldyrnem4aeQp4hai3gythswg6p01oSoTl58rcpiFAjF2br2Ak2A=="],
|
||||
|
||||
"standardwebhooks": ["standardwebhooks@1.1.1", "", { "dependencies": { "@stablelib/base64": "^1.0.0", "fast-sha256": "^1.3.0" } }, "sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ=="],
|
||||
|
||||
"statuses": ["statuses@2.0.2", "", {}, "sha512-DvEy55V3DB7uknRo+4iOGT5fP1slR8wQohVdknigZPMpMstaKJQWhwiYBACJE3Ul2pTnATihhBYnRhZQHGBiRw=="],
|
||||
|
||||
"toidentifier": ["toidentifier@1.0.1", "", {}, "sha512-o5sSPKEkg/DIQNmH43V0/uerLrpzVedkUh8tGNvaeXpfpuwjKenlSox/2O/BTlZUtEe+JG7s5YhEz608PlAHRA=="],
|
||||
|
||||
"tr46": ["tr46@0.0.3", "", {}, "sha512-N3WMsuqV66lT30CrXNbEjx4GEwlow3v6rr4mCcv6prnfwhS01rkgyFdjPNBYd9br7LpXV1+Emh01fHnq2Gdgrw=="],
|
||||
|
||||
"ts-algebra": ["ts-algebra@2.0.0", "", {}, "sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw=="],
|
||||
|
||||
"type-is": ["type-is@2.0.1", "", { "dependencies": { "content-type": "^1.0.5", "media-typer": "^1.1.0", "mime-types": "^3.0.0" } }, "sha512-OZs6gsjF4vMp32qrCbiVSkrFmXtG/AZhY3t0iAMrMBiAZyV9oALtXO8hsrHbMXF9x6L3grlFuwW2oAz7cav+Gw=="],
|
||||
|
||||
"typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="],
|
||||
|
||||
"undici-types": ["undici-types@5.26.5", "", {}, "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA=="],
|
||||
"undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
|
||||
"unpipe": ["unpipe@1.0.0", "", {}, "sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ=="],
|
||||
|
||||
"vary": ["vary@1.1.2", "", {}, "sha512-BNGbWLfd0eUPabhkXUVm0j8uuvREyTh5ovRa/dyow/BqAbZJyC+5fU+IzQOzmAKzYqYRAISoRhdQr3eIZ/PXqg=="],
|
||||
|
||||
"web-streams-polyfill": ["web-streams-polyfill@4.0.0-beta.3", "", {}, "sha512-QW95TCTaHmsYfHDybGMwO5IJIM93I/6vTRk+daHTWFPhwh+C8Cg7j7XyKrwrj8Ib6vYXe0ocYNrmzY4xAAN6ug=="],
|
||||
|
||||
"webidl-conversions": ["webidl-conversions@3.0.1", "", {}, "sha512-2JAn3z8AR6rjK8Sm8orRC0h/bcl/DqL7tRPdGZ4I1CjdF+EaMLmYxBHyXuKL849eucPFhvBoxMsflfOb8kxaeQ=="],
|
||||
|
||||
"whatwg-url": ["whatwg-url@5.0.0", "", { "dependencies": { "tr46": "~0.0.3", "webidl-conversions": "^3.0.0" } }, "sha512-saE57nupxk6v3HY35+jzBwYa0rKSy0XR8JSxZPwgLr7ys0IBzhGviA1/TUGJLmSVqs8pb9AnvICXEuOHLprYTw=="],
|
||||
|
||||
"which": ["which@2.0.2", "", { "dependencies": { "isexe": "^2.0.0" }, "bin": { "node-which": "./bin/node-which" } }, "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA=="],
|
||||
|
||||
"wrappy": ["wrappy@1.0.2", "", {}, "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ=="],
|
||||
@@ -295,19 +247,5 @@
|
||||
"zod": ["zod@4.3.6", "", {}, "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg=="],
|
||||
|
||||
"zod-to-json-schema": ["zod-to-json-schema@3.25.2", "", { "peerDependencies": { "zod": "^3.25.28 || ^4" } }, "sha512-O/PgfnpT1xKSDeQYSCfRI5Gy3hPf91mKVDuYLUHZJMiDFptvP41MSnWofm8dnCm0256ZNfZIM7DSzuSMAFnjHA=="],
|
||||
|
||||
"@anthropic-ai/claude-agent-sdk/@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.74.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-srbJV7JKsc5cQ6eVuFzjZO7UR3xEPJqPamHFIe29bs38Ij2IripoAhC0S5NslNbaFUYqBKypmmpzMTpqfHEUDw=="],
|
||||
|
||||
"@types/node-fetch/@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="],
|
||||
|
||||
"bun-types/@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="],
|
||||
|
||||
"form-data/mime-types": ["mime-types@2.1.35", "", { "dependencies": { "mime-db": "1.52.0" } }, "sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw=="],
|
||||
|
||||
"@types/node-fetch/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
|
||||
"bun-types/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
|
||||
|
||||
"form-data/mime-types/mime-db": ["mime-db@1.52.0", "", {}, "sha512-sPU4uV7dYlvtWJxwwxHD0PuihVNiE7TyAbQ5SWxDCB9mUYvOgroQOwYQQOKPJ8CIbE+1ETVlOoK1UC2nU3gYvg=="],
|
||||
}
|
||||
}
|
||||
|
||||
@@ -108,6 +108,10 @@ async function main() {
|
||||
"--record",
|
||||
"append a compact summary line to ai_evals/history/<mode>.jsonl",
|
||||
)
|
||||
.option(
|
||||
"--reasoning <effort>",
|
||||
"reasoning effort for frontend modes (e.g. off, low, medium, high, max); default: the product's",
|
||||
)
|
||||
.option(
|
||||
"--backend-validation <mode>",
|
||||
`backend smoke validation (${BACKEND_VALIDATION_MODES.join(", ")})`,
|
||||
@@ -126,6 +130,7 @@ async function main() {
|
||||
executionOnly?: boolean;
|
||||
record?: boolean;
|
||||
backendValidation?: string;
|
||||
reasoning?: string;
|
||||
},
|
||||
) => {
|
||||
await handleRun({
|
||||
@@ -140,6 +145,7 @@ async function main() {
|
||||
executionOnly: options.executionOnly ?? false,
|
||||
record: options.record ?? false,
|
||||
backendValidation: options.backendValidation,
|
||||
reasoning: options.reasoning,
|
||||
});
|
||||
},
|
||||
);
|
||||
@@ -190,6 +196,7 @@ async function handleRun(input: {
|
||||
executionOnly: boolean;
|
||||
record: boolean;
|
||||
backendValidation?: string;
|
||||
reasoning?: string;
|
||||
}) {
|
||||
if (input.record && input.caseIds.length > 0) {
|
||||
throw new Error(
|
||||
@@ -217,6 +224,13 @@ async function handleRun(input: {
|
||||
"--backend-validation currently supports only flow and script modes",
|
||||
);
|
||||
}
|
||||
if (input.reasoning) {
|
||||
if (input.mode === "cli") {
|
||||
throw new Error("--reasoning only applies to frontend modes");
|
||||
}
|
||||
// The frontend runtime runs in a child process, which inherits it.
|
||||
process.env.WMILL_AI_EVAL_REASONING = input.reasoning;
|
||||
}
|
||||
if (input.mode !== "cli") {
|
||||
await assertWindmillBackendReachable(resolveWindmillBackendSettings());
|
||||
}
|
||||
@@ -257,6 +271,10 @@ async function handleRun(input: {
|
||||
backendValidation,
|
||||
});
|
||||
|
||||
if (input.reasoning && result.runModel) {
|
||||
result.runModel = `${result.runModel}@${input.reasoning}`;
|
||||
}
|
||||
|
||||
const resolvedOutputPath =
|
||||
models.length === 1
|
||||
? resolveRunOutputPath(input.mode, input.outputPath)
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import Anthropic from "@anthropic-ai/sdk";
|
||||
import type { EvalMode, JudgeResult } from "./types";
|
||||
|
||||
export const DEFAULT_JUDGE_MODEL = "claude-sonnet-4-6";
|
||||
export const DEFAULT_JUDGE_MODEL = "claude-sonnet-5-5";
|
||||
|
||||
const JUDGE_TOOL_NAME = "submit_judgement";
|
||||
|
||||
@@ -29,6 +29,7 @@ export async function judgeOutput(input: {
|
||||
|
||||
const system = [
|
||||
"You evaluate benchmark outputs for Windmill AI generation.",
|
||||
`Always answer by calling the ${JUDGE_TOOL_NAME} tool.`,
|
||||
"Deterministic checks already run separately. Focus on whether the final output satisfies the user request.",
|
||||
"If expected state is provided, treat it as a valid example and reward semantically equivalent outputs.",
|
||||
"If a checklist is provided, treat it as the explicit acceptance criteria for this case.",
|
||||
@@ -69,8 +70,8 @@ export async function judgeOutput(input: {
|
||||
try {
|
||||
const response = await client.messages.create({
|
||||
model,
|
||||
max_tokens: 1024,
|
||||
temperature: 0,
|
||||
// The judge thinks by default, and thinking shares this budget with the verdict.
|
||||
max_tokens: 16000,
|
||||
system,
|
||||
messages: [{ role: "user", content: user }],
|
||||
tools: [
|
||||
@@ -93,11 +94,8 @@ export async function judgeOutput(input: {
|
||||
},
|
||||
},
|
||||
],
|
||||
tool_choice: {
|
||||
type: "tool",
|
||||
name: JUDGE_TOOL_NAME,
|
||||
disable_parallel_tool_use: true,
|
||||
},
|
||||
// Current models refuse a forced tool_choice ("tool"/"any").
|
||||
tool_choice: { type: "auto", disable_parallel_tool_use: true },
|
||||
});
|
||||
|
||||
const toolUseBlock = response.content.find(
|
||||
|
||||
@@ -78,6 +78,32 @@ export const EVAL_MODELS: EvalModelSpec[] = [
|
||||
model: "opus",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "sonnet-5.5",
|
||||
label: "Claude Sonnet 5.5",
|
||||
aliases: ["sonnet-5.5", "claude-sonnet-5.5", "claude-sonnet-5-5"],
|
||||
frontend: {
|
||||
provider: "anthropic",
|
||||
model: "claude-sonnet-5-5",
|
||||
},
|
||||
cli: {
|
||||
provider: "anthropic",
|
||||
model: "claude-sonnet-5-5",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "opus-5.5",
|
||||
label: "Claude Opus 5.5",
|
||||
aliases: ["opus-5.5", "claude-opus-5.5", "claude-opus-5-5"],
|
||||
frontend: {
|
||||
provider: "anthropic",
|
||||
model: "claude-opus-5-5",
|
||||
},
|
||||
cli: {
|
||||
provider: "anthropic",
|
||||
model: "claude-opus-5-5",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "4o",
|
||||
label: "GPT-4o",
|
||||
@@ -96,6 +122,51 @@ export const EVAL_MODELS: EvalModelSpec[] = [
|
||||
model: "gpt-5.5",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "gpt-5.6-sol",
|
||||
label: "GPT-5.6 Sol",
|
||||
aliases: ["gpt-5.6-sol"],
|
||||
frontend: {
|
||||
provider: "openai",
|
||||
model: "gpt-5.6-sol",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "gpt-6-astra",
|
||||
label: "GPT-6 Astra",
|
||||
aliases: ["gpt-6-astra", "gpt-6"],
|
||||
frontend: {
|
||||
provider: "openai",
|
||||
model: "gpt-6-astra",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "gpt-6-sol",
|
||||
label: "GPT-6 Sol",
|
||||
aliases: ["gpt-6-sol"],
|
||||
frontend: {
|
||||
provider: "openai",
|
||||
model: "gpt-6-sol",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "gpt-6-luna",
|
||||
label: "GPT-6 Luna",
|
||||
aliases: ["gpt-6-luna"],
|
||||
frontend: {
|
||||
provider: "openai",
|
||||
model: "gpt-6-luna",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "gemini-3.8-flash",
|
||||
label: "Gemini 3.8 Flash",
|
||||
aliases: ["gemini-3.8-flash"],
|
||||
frontend: {
|
||||
provider: "googleai",
|
||||
model: "gemini-3.8-flash",
|
||||
},
|
||||
},
|
||||
{
|
||||
id: "gemini-3-flash-preview",
|
||||
label: "Gemini 3 Flash Preview",
|
||||
|
||||
@@ -8,8 +8,8 @@
|
||||
"test:frontend-graph": "cd ../frontend && node_modules/.bin/vitest run --project server --config ../ai_evals/adapters/frontend/vitest.unit.config.ts"
|
||||
},
|
||||
"dependencies": {
|
||||
"@anthropic-ai/claude-agent-sdk": "^0.2.25",
|
||||
"@anthropic-ai/sdk": "^0.39.0",
|
||||
"@anthropic-ai/claude-agent-sdk": "^0.3.284",
|
||||
"@anthropic-ai/sdk": "^0.129.0",
|
||||
"commander": "^14.0.3",
|
||||
"openai": "^6.9.1",
|
||||
"yaml": "^2.8.3"
|
||||
|
||||
@@ -14,6 +14,130 @@ use std::time::{Duration, Instant};
|
||||
/// its `thinking` param, Gemini to a zero budget or the model's floor).
|
||||
pub(crate) const REASONING_OFF_SENTINEL: &str = "none";
|
||||
|
||||
/// The effort to send for a model, dropping the off sentinel on a model that rejects every
|
||||
/// disable: the model then reasons at its default instead of failing the request. The UI
|
||||
/// never offers off on these, so this guards an agent step saved before it stopped, or an
|
||||
/// effort passed in as a flow input.
|
||||
pub fn effective_reasoning_effort<'a>(model: &str, effort: Option<&'a str>) -> Option<&'a str> {
|
||||
match effort {
|
||||
Some(effort)
|
||||
if effort == REASONING_OFF_SENTINEL
|
||||
&& reasoning_rule(model).is_some_and(|rule| !rule.can_disable) =>
|
||||
{
|
||||
None
|
||||
}
|
||||
effort => effort,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a request with function tools must send the off sentinel on Chat Completions:
|
||||
/// the model refuses tools there while it reasons, and does accept being turned off.
|
||||
pub(crate) fn completions_tools_need_reasoning_off(model: &str) -> bool {
|
||||
reasoning_rule(model).is_some_and(|rule| rule.completions_tools_need_off && rule.can_disable)
|
||||
}
|
||||
|
||||
/// What the backend needs to know about a model's reasoning: the rows of `REASONING_RULES`
|
||||
/// in the frontend's `reasoningRegistry.ts`, cut down to the two facts the wire needs.
|
||||
/// `reasoningParity.json` next to that file is checked by both sides' tests.
|
||||
struct ReasoningRule {
|
||||
matches: fn(&str) -> bool,
|
||||
/// False when the provider rejects every disable, so the off sentinel must not be sent.
|
||||
can_disable: bool,
|
||||
/// Chat Completions refuses function tools while the model reasons, even with the
|
||||
/// effort omitted (live-verified); the Responses API has no such limit.
|
||||
completions_tools_need_off: bool,
|
||||
}
|
||||
|
||||
/// Matched in order against the lowercased model id, first match wins. A model no row
|
||||
/// matches keeps whatever effort it was given.
|
||||
const REASONING_RULES: &[ReasoningRule] = &[
|
||||
// Live-verified: Fable, Mythos and the 5.x point releases reject `thinking: disabled`.
|
||||
ReasoningRule {
|
||||
matches: |m| m.contains("claude-fable") || m.contains("claude-mythos"),
|
||||
can_disable: false,
|
||||
completions_tools_need_off: false,
|
||||
},
|
||||
ReasoningRule {
|
||||
matches: is_claude_5_point_release,
|
||||
can_disable: false,
|
||||
completions_tools_need_off: false,
|
||||
},
|
||||
// Live-verified: astra takes low..max only, where sol and luna also take `none`.
|
||||
ReasoningRule {
|
||||
matches: |m| base_id(m).starts_with("gpt-6-astra"),
|
||||
can_disable: false,
|
||||
completions_tools_need_off: true,
|
||||
},
|
||||
ReasoningRule {
|
||||
matches: |m| gpt_version(m).is_some_and(|(major, _)| major >= 6),
|
||||
can_disable: true,
|
||||
completions_tools_need_off: true,
|
||||
},
|
||||
ReasoningRule {
|
||||
matches: |m| matches!(gpt_version(m), Some((5, Some(minor))) if minor >= 5),
|
||||
can_disable: true,
|
||||
completions_tools_need_off: true,
|
||||
},
|
||||
ReasoningRule {
|
||||
matches: |m| matches!(gpt_version(m), Some((5, Some(_)))),
|
||||
can_disable: true,
|
||||
completions_tools_need_off: false,
|
||||
},
|
||||
// gpt-5 and the o-series reject `none`.
|
||||
ReasoningRule {
|
||||
matches: |m| matches!(gpt_version(m), Some((5, None))),
|
||||
can_disable: false,
|
||||
completions_tools_need_off: false,
|
||||
},
|
||||
ReasoningRule {
|
||||
matches: |m| {
|
||||
let base = base_id(m);
|
||||
base.starts_with('o') && base[1..].starts_with(|c: char| c.is_ascii_digit())
|
||||
},
|
||||
can_disable: false,
|
||||
completions_tools_need_off: false,
|
||||
},
|
||||
];
|
||||
|
||||
fn reasoning_rule(model: &str) -> Option<&'static ReasoningRule> {
|
||||
let model = model.to_lowercase();
|
||||
REASONING_RULES.iter().find(|rule| (rule.matches)(&model))
|
||||
}
|
||||
|
||||
/// The id after a gateway's `vendor/` and before a `:variant`.
|
||||
fn base_id(model: &str) -> &str {
|
||||
let last = model.rsplit('/').next().unwrap_or(model);
|
||||
last.split(':').next().unwrap_or(last)
|
||||
}
|
||||
|
||||
/// `(major, minor)` of a `gpt-` id. The major is one digit then a separator or the end,
|
||||
/// since Azure names gpt-3.5 `gpt-35-turbo`.
|
||||
fn gpt_version(model: &str) -> Option<(u32, Option<u32>)> {
|
||||
let rest = base_id(model).strip_prefix("gpt-")?;
|
||||
let mut chars = rest.chars();
|
||||
let major = chars.next()?.to_digit(10)?;
|
||||
match chars.next() {
|
||||
None | Some('-') => Some((major, None)),
|
||||
Some('.') => {
|
||||
let minor: String = chars.take_while(char::is_ascii_digit).collect();
|
||||
Some((major, minor.parse().ok()))
|
||||
}
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Sonnet or Opus 5.x with x >= 1. The version match stops at one digit so a dated id
|
||||
/// (`claude-sonnet-5-20260101`) stays Sonnet 5.
|
||||
fn is_claude_5_point_release(model: &str) -> bool {
|
||||
let model = model.replace('.', "-");
|
||||
["claude-opus-5-", "claude-sonnet-5-"].iter().any(|prefix| {
|
||||
model.split(prefix).skip(1).any(|rest| {
|
||||
let mut chars = rest.chars();
|
||||
matches!(chars.next(), Some('1'..='9')) && !matches!(chars.next(), Some('0'..='9'))
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether a Claude model removed the sampling params (`temperature`, `top_p`,
|
||||
/// `top_k`). On these, any value is a hard 400 — `temperature is deprecated for
|
||||
/// this model` — whatever the thinking mode, so the param has to be dropped on
|
||||
@@ -119,6 +243,46 @@ pub fn remember_chat_completions_only(base_url: &str, model: &str) {
|
||||
CHAT_COMPLETIONS_ONLY.insert((base_url.to_string(), model.to_string()), Instant::now());
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod reasoning_rule_tests {
|
||||
use super::*;
|
||||
|
||||
/// The frontend registry's test reads the same file. These rules see the model id
|
||||
/// alone, so it holds only rows whose answer doesn't depend on the provider.
|
||||
#[test]
|
||||
fn agrees_with_the_frontend_registry() {
|
||||
let rows: Vec<serde_json::Value> = serde_json::from_str(include_str!(
|
||||
"../../../../frontend/src/lib/components/copilot/reasoningParity.json"
|
||||
))
|
||||
.unwrap();
|
||||
for row in rows {
|
||||
let model = row["model"].as_str().unwrap();
|
||||
let can_disable = row["canDisable"].as_bool().unwrap();
|
||||
let tools_need_off = row["completionsToolsNeedOff"].as_bool().unwrap();
|
||||
let sent = effective_reasoning_effort(model, Some(REASONING_OFF_SENTINEL));
|
||||
assert_eq!(sent.is_some(), can_disable, "{model}");
|
||||
assert_eq!(
|
||||
effective_reasoning_effort(model, Some("low")),
|
||||
Some("low"),
|
||||
"{model}"
|
||||
);
|
||||
assert_eq!(
|
||||
completions_tools_need_reasoning_off(model),
|
||||
tools_need_off && can_disable,
|
||||
"{model}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reads_ids_the_frontend_resolves_first() {
|
||||
// Azure's gpt-3.5 is not major 35, and a gateway prefix is not part of the id.
|
||||
assert_eq!(gpt_version("gpt-35-turbo"), None);
|
||||
assert!(completions_tools_need_reasoning_off("openai/gpt-6-sol"));
|
||||
assert_eq!(effective_reasoning_effort("openai/o3", Some("none")), None);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod chat_completions_only_tests {
|
||||
use super::*;
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
use super::{completions_tools_need_reasoning_off, REASONING_OFF_SENTINEL};
|
||||
use crate::{
|
||||
ai_providers::AIProvider,
|
||||
image_handler::prepare_messages_for_api,
|
||||
@@ -143,6 +144,15 @@ impl OtherQueryBuilder {
|
||||
|
||||
let (reasoning_effort, thinking, temperature) =
|
||||
provider_reasoning_fields(&self.provider_kind, args.reasoning_effort, args.temperature);
|
||||
// With tools, gpt-5.5+ only run here with reasoning off: turn it off where the model
|
||||
// can, rather than failing every turn.
|
||||
let reasoning_effort = if args.tools.is_some_and(|tools| !tools.is_empty())
|
||||
&& completions_tools_need_reasoning_off(args.model)
|
||||
{
|
||||
Some(REASONING_OFF_SENTINEL)
|
||||
} else {
|
||||
reasoning_effort
|
||||
};
|
||||
|
||||
// Build request with stream_options for usage tracking
|
||||
let request_with_usage = OpenAICompletionRequest {
|
||||
|
||||
@@ -294,7 +294,10 @@ impl ProviderWithResource {
|
||||
/// The reasoning effort to thread to the provider, treating an empty string
|
||||
/// (e.g. a cleared flow input) as unset.
|
||||
pub fn get_reasoning_effort(&self) -> Option<&str> {
|
||||
self.reasoning_effort.as_deref().filter(|s| !s.is_empty())
|
||||
crate::providers::effective_reasoning_effort(
|
||||
&self.model,
|
||||
self.reasoning_effort.as_deref().filter(|s| !s.is_empty()),
|
||||
)
|
||||
}
|
||||
|
||||
pub async fn get_base_url(&self, db: &DB) -> Result<String, Error> {
|
||||
|
||||
@@ -22,6 +22,8 @@ import {
|
||||
} from './modelConfig'
|
||||
import {
|
||||
applyReasoningToConfig,
|
||||
completionsRejectsToolsWithReasoning,
|
||||
explicitOffToken,
|
||||
requestsReasoning,
|
||||
stripLegacyThinkingSuffix,
|
||||
type ReasoningEffort
|
||||
@@ -69,6 +71,9 @@ interface AIProviderDetails {
|
||||
// the frontier model. The gpt-5 family is deprecated (retires 2026-12-11) but
|
||||
// still served, so it stays in the list below the 5.6 models.
|
||||
const OPENAI_MODELS = [
|
||||
'gpt-6-sol',
|
||||
'gpt-6-astra',
|
||||
'gpt-6-luna',
|
||||
'gpt-5.6-terra',
|
||||
'gpt-5.6-sol',
|
||||
'gpt-5.6-luna',
|
||||
@@ -87,7 +92,14 @@ export const AI_PROVIDERS: Record<AIProvider, AIProviderDetails> = {
|
||||
},
|
||||
anthropic: {
|
||||
label: 'Anthropic',
|
||||
defaultModels: ['claude-sonnet-5', 'claude-opus-5', 'claude-opus-4-8', 'claude-haiku-4-5']
|
||||
defaultModels: [
|
||||
'claude-sonnet-5-5',
|
||||
'claude-opus-5-5',
|
||||
'claude-sonnet-5',
|
||||
'claude-opus-5',
|
||||
'claude-opus-4-8',
|
||||
'claude-haiku-4-5'
|
||||
]
|
||||
},
|
||||
googleai: {
|
||||
label: 'Google AI',
|
||||
@@ -1160,6 +1172,12 @@ export async function getCompletion(
|
||||
|
||||
// Use Completions API for other providers
|
||||
const client = options?.openaiClient ?? workspaceAIClients.getOpenaiClient()
|
||||
// gpt-5.5+ refuse function tools here unless reasoning is off, so the Responses API
|
||||
// fallback turns it off where the model can rather than failing the turn.
|
||||
const reasoningEffort =
|
||||
tools?.length && completionsRejectsToolsWithReasoning(provider, modelProvider.model)
|
||||
? (explicitOffToken(provider, modelProvider.model) ?? options?.reasoningEffort)
|
||||
: options?.reasoningEffort
|
||||
const completionConfig = applyReasoningToConfig(
|
||||
config.stream && STREAM_USAGE_PROVIDERS.has(provider)
|
||||
? {
|
||||
@@ -1175,7 +1193,7 @@ export async function getCompletion(
|
||||
}
|
||||
: config,
|
||||
provider === 'deepseek' ? 'deepseek' : provider === 'mistral' ? 'mistral' : 'completions',
|
||||
options?.reasoningEffort
|
||||
reasoningEffort
|
||||
)
|
||||
const completion = client.chat.completions.create(completionConfig, {
|
||||
signal: abortController.signal,
|
||||
|
||||
@@ -21,6 +21,13 @@ describe('workspace context window overrides', () => {
|
||||
expect(getConfiguredModelContextWindow('openai', 'qwen-local', overrides)).toBeUndefined()
|
||||
expect(getEffectiveModelContextWindow('openai', 'qwen-local', overrides)).toBe(128_000)
|
||||
})
|
||||
|
||||
it('knows the gpt-6 and Claude 5.5 windows, so they do not compact at the assumed one', () => {
|
||||
expect(getConfiguredModelContextWindow('openai', 'gpt-6-sol', undefined)).toBe(1_050_000)
|
||||
expect(getConfiguredModelContextWindow('anthropic', 'claude-sonnet-5-5', undefined)).toBe(
|
||||
1_000_000
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
describe('usesAnthropicMessagesApi', () => {
|
||||
|
||||
@@ -58,7 +58,7 @@ export function usesOpenRouterPromptCaching(provider: AIProvider, model: string)
|
||||
// so it does not catch unrelated ids like Mistral's "open-mistral-*" or "optimus-*".
|
||||
export function requiresMaxCompletionTokens(model: string) {
|
||||
const baseModel = parseModelId(model).base
|
||||
return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel)
|
||||
return Number(/^gpt-(\d)(?:[.-]|$)/.exec(baseModel)?.[1] ?? 0) >= 5 || /^o\d/.test(baseModel)
|
||||
}
|
||||
|
||||
// Context windows of the models we know, most specific entry first — the first
|
||||
@@ -87,6 +87,7 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
|
||||
['claude', 200_000],
|
||||
// OpenAI — gpt-5 covers the base family (-mini / -nano) and the 5.1/5.2
|
||||
// revisions, all 400K; 5.4/5.5 moved to 1M and 5.6 to 1.05M
|
||||
['gpt-6', 1_050_000],
|
||||
['gpt-5.6', 1_050_000],
|
||||
['gpt-5.5', 1_000_000],
|
||||
['gpt-5.4', 1_000_000],
|
||||
@@ -140,6 +141,7 @@ const MODEL_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [
|
||||
['claude-fable', 64_000],
|
||||
['claude-mythos', 64_000],
|
||||
// OpenAI
|
||||
['gpt-6', 128_000],
|
||||
['gpt-5', 128_000],
|
||||
['gpt-4.1', 32_768],
|
||||
['gpt-4o', 16_384],
|
||||
|
||||
@@ -35,6 +35,16 @@ describe('resolveModelPrice', () => {
|
||||
expect(resolveModelPrice('googleai', 'gemini-3.7-flash', undefined)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('prices Claude 5.5 flat and leaves the tiered gpt-6 unpriced', () => {
|
||||
expect(resolveModelPrice('anthropic', 'claude-opus-5-5', undefined)?.price).toMatchObject({
|
||||
input: 4,
|
||||
output: 20,
|
||||
cacheRead: 0.2
|
||||
})
|
||||
expect(resolveModelPrice('anthropic', 'claude-sonnet-5-5', undefined)?.price.input).toBe(2)
|
||||
expect(resolveModelPrice('openai', 'gpt-6-sol', undefined)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('reports an unknown model as unpriced rather than guessing', () => {
|
||||
expect(resolveModelPrice('customai', 'some-in-house-model', undefined)).toBeUndefined()
|
||||
})
|
||||
|
||||
@@ -69,6 +69,8 @@ const MODEL_PRICES: [name: string, price: PriceEntry | null][] = [
|
||||
// fallback sits below the explicit entries rather than covering them.
|
||||
['claude-fable-5', { input: 10, output: 50 }],
|
||||
['claude-mythos-5', { input: 10, output: 50 }],
|
||||
['claude-opus-5-5', { input: 4, output: 20, cacheRead: 0.2 }],
|
||||
['claude-sonnet-5-5', { input: 2, output: 10, cacheRead: 0.2 }],
|
||||
['claude-opus-5', { input: 5, output: 25 }],
|
||||
['claude-opus-4-8', { input: 5, output: 25 }],
|
||||
['claude-opus-4-7', { input: 5, output: 25 }],
|
||||
@@ -98,6 +100,9 @@ const MODEL_PRICES: [name: string, price: PriceEntry | null][] = [
|
||||
// Revisions past gpt-5 are priced separately by OpenAI and are not tracked here.
|
||||
// The matcher's revision guard already keeps them off the family rate; these
|
||||
// entries stay so a revision the guard admits still resolves to no rate.
|
||||
// gpt-6 bills the whole request at a higher rate above 272K input tokens, which a
|
||||
// per-model rate cannot express.
|
||||
['gpt-6', null],
|
||||
['gpt-5.6', null],
|
||||
['gpt-5.5', null],
|
||||
['gpt-5.4', null],
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
[
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-sonnet-5-5",
|
||||
"canDisable": false,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-5-5",
|
||||
"canDisable": false,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-fable-5",
|
||||
"canDisable": false,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "aws_bedrock",
|
||||
"model": "global.anthropic.claude-opus-5-5-v1:0",
|
||||
"canDisable": false,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-sonnet-5",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-5-20260101",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "anthropic",
|
||||
"model": "claude-opus-4-8",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{
|
||||
"provider": "openai",
|
||||
"model": "gpt-6-astra",
|
||||
"canDisable": false,
|
||||
"completionsToolsNeedOff": true
|
||||
},
|
||||
{
|
||||
"provider": "openai",
|
||||
"model": "gpt-6-sol",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": true
|
||||
},
|
||||
{
|
||||
"provider": "azure_openai",
|
||||
"model": "gpt-6-luna",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": true
|
||||
},
|
||||
{
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.6-sol",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": true
|
||||
},
|
||||
{ "provider": "openai", "model": "gpt-5.5", "canDisable": true, "completionsToolsNeedOff": true },
|
||||
{ "provider": "openai", "model": "gpt-5.7", "canDisable": true, "completionsToolsNeedOff": true },
|
||||
{
|
||||
"provider": "openai",
|
||||
"model": "gpt-5.1",
|
||||
"canDisable": true,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{ "provider": "openai", "model": "gpt-5", "canDisable": false, "completionsToolsNeedOff": false },
|
||||
{
|
||||
"provider": "openai",
|
||||
"model": "gpt-5-mini",
|
||||
"canDisable": false,
|
||||
"completionsToolsNeedOff": false
|
||||
},
|
||||
{ "provider": "openai", "model": "o3", "canDisable": false, "completionsToolsNeedOff": false }
|
||||
]
|
||||
@@ -1,6 +1,8 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import {
|
||||
applyReasoningToConfig,
|
||||
completionsRejectsToolsWithReasoning,
|
||||
explicitOffToken,
|
||||
getReasoningCapability,
|
||||
REASONING_OFF,
|
||||
resolveEffectiveReasoning,
|
||||
@@ -8,6 +10,8 @@ import {
|
||||
stripLegacyThinkingSuffix,
|
||||
supportsReasoning
|
||||
} from './reasoningRegistry'
|
||||
import type { AIProvider } from '$lib/gen'
|
||||
import parity from './reasoningParity.json'
|
||||
|
||||
describe('stripLegacyThinkingSuffix', () => {
|
||||
it('removes the deprecated /thinking suffix', () => {
|
||||
@@ -261,6 +265,42 @@ describe('supportsReasoning (static registry)', () => {
|
||||
expect(getReasoningCapability('openrouter', 'x-ai/grok-4').canDisable).toBe(false)
|
||||
expect(getReasoningCapability('openrouter', 'deepseek/deepseek-r1').canDisable).toBe(false)
|
||||
})
|
||||
it('never sends a disable the 5.5 point releases and gpt-6-astra reject', () => {
|
||||
// Live-verified: Claude 5.5 rejects `thinking: disabled`, gpt-6-astra rejects `none`.
|
||||
for (const [provider, model] of [
|
||||
['anthropic', 'claude-sonnet-5-5'],
|
||||
['anthropic', 'claude-opus-5-5'],
|
||||
['aws_bedrock', 'global.anthropic.claude-opus-5-5-v1:0'],
|
||||
['openai', 'gpt-6-astra']
|
||||
] as const) {
|
||||
expect(getReasoningCapability(provider, model).canDisable, model).toBe(false)
|
||||
expect(explicitOffToken(provider, model), model).toBeUndefined()
|
||||
}
|
||||
expect(getReasoningCapability('openrouter', 'anthropic/claude-sonnet-5.5').canDisable).toBe(
|
||||
false
|
||||
)
|
||||
expect(explicitOffToken('openrouter', 'anthropic/claude-sonnet-5.5')).toBeUndefined()
|
||||
// A dated Claude 5 id is not a point release.
|
||||
expect(explicitOffToken('anthropic', 'claude-sonnet-5-20260101')).toBe('none')
|
||||
expect(explicitOffToken('openai', 'gpt-6-sol')).toBe('none')
|
||||
expect(getReasoningCapability('openai', 'gpt-6-luna')).toMatchObject({
|
||||
supported: true,
|
||||
canDisable: true,
|
||||
levels: ['low', 'medium', 'high', 'xhigh', 'max']
|
||||
})
|
||||
})
|
||||
it("reads Azure's gpt-35-turbo as gpt-3.5, not a gpt-5+ reasoning model", () => {
|
||||
expect(supportsReasoning('azure_openai', 'gpt-35-turbo')).toBe(false)
|
||||
expect(supportsReasoning('openai', 'gpt-35-turbo-16k')).toBe(false)
|
||||
})
|
||||
it('finds the models that refuse function tools with reasoning on Chat Completions', () => {
|
||||
for (const model of ['gpt-5.5', 'gpt-5.6-sol', 'gpt-6-astra']) {
|
||||
expect(completionsRejectsToolsWithReasoning('openai', model), model).toBe(true)
|
||||
}
|
||||
for (const model of ['gpt-5', 'gpt-5.1', 'gpt-35-turbo', 'o3']) {
|
||||
expect(completionsRejectsToolsWithReasoning('azure_openai', model), model).toBe(false)
|
||||
}
|
||||
})
|
||||
it('forwards an explicit off as effort none through OpenRouter', () => {
|
||||
expect(
|
||||
resolveRequestReasoning({
|
||||
@@ -353,6 +393,23 @@ describe('Azure AI Foundry reasoning follows the model family', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('backend parity', () => {
|
||||
// windmill-ai's `providers/mod.rs` test reads the same file. The backend rules see the
|
||||
// model id alone, so the file holds only rows whose answer doesn't depend on the
|
||||
// provider: a Bedrock- or Gemini-Pro-specific row belongs in the tests above.
|
||||
it.each(parity)(
|
||||
'$provider $model',
|
||||
({ provider, model, canDisable, completionsToolsNeedOff }) => {
|
||||
const capability = getReasoningCapability(provider as AIProvider, model)
|
||||
expect(capability.supported).toBe(true)
|
||||
expect(capability.canDisable).toBe(canDisable)
|
||||
expect(completionsRejectsToolsWithReasoning(provider as AIProvider, model)).toBe(
|
||||
completionsToolsNeedOff
|
||||
)
|
||||
}
|
||||
)
|
||||
})
|
||||
|
||||
describe('resolveEffectiveReasoning', () => {
|
||||
it('defaults capable models to high when unset', () => {
|
||||
expect(resolveEffectiveReasoning({ provider: 'anthropic', model: 'claude-sonnet-4-6' })).toBe(
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { AIProvider, AIProviderModel } from '$lib/gen'
|
||||
import { parseModelId, usesAnthropicMessagesApi } from './modelConfig'
|
||||
import { usesAnthropicMessagesApi } from './modelConfig'
|
||||
|
||||
/**
|
||||
* Reasoning effort is provider/model-specific. We never normalize a single
|
||||
@@ -33,11 +33,6 @@ export function stripLegacyThinkingSuffix(model: string): string {
|
||||
: model
|
||||
}
|
||||
|
||||
/** Bare model id without any provider/gateway prefix (e.g. OpenRouter's `openai/o3`). */
|
||||
function baseModelId(model: string): string {
|
||||
return parseModelId(model).base
|
||||
}
|
||||
|
||||
/**
|
||||
* Azure AI Foundry hosts multiple model families under one provider, so reasoning
|
||||
* support follows the underlying model rather than the provider: Claude deployments
|
||||
@@ -53,172 +48,218 @@ function reasoningProviderFamily(provider: AIProvider, model: string): AIProvide
|
||||
}
|
||||
|
||||
/**
|
||||
* Suggested effort levels per provider, sourced from each provider SDK's own
|
||||
* vocabulary.
|
||||
* Sentinel sent for the deepseek off case. It never reaches the wire as an
|
||||
* effort: the 'deepseek' branch of `applyReasoningToConfig` translates it to
|
||||
* the provider's `thinking: {type: "disabled"}` param (`reasoning_effort:
|
||||
* "none"` is rejected by their API).
|
||||
*/
|
||||
const PROVIDER_REASONING_LEVELS: Partial<Record<AIProvider, ReasoningEffort[]>> = {
|
||||
// DeepSeek accepts the full five-token vocabulary but only two levels are
|
||||
// real: low/medium are server-mapped to high and xhigh to max — offering
|
||||
// them would be a no-op knob.
|
||||
deepseek: ['high', 'max'],
|
||||
// Mistral's only effort token besides the 'none' disable is 'high'
|
||||
// (anything else is rejected), so the knob is effectively on/off.
|
||||
mistral: ['high']
|
||||
}
|
||||
export const DEEPSEEK_OFF_SENTINEL: ReasoningEffort = 'none'
|
||||
|
||||
/**
|
||||
* OpenRouter validates effort against its own vocabulary
|
||||
* (minimal..xhigh + none) and translates per underlying provider, so the
|
||||
* real ladder depends on the model family: Anthropic gets all five as
|
||||
* distinct budget ratios (minimal 10% .. xhigh 95% of max_tokens); Gemini
|
||||
* maps to thinkingLevel with xhigh clamped to high (a no-op vs high);
|
||||
* OpenAI gets the token passed through verbatim, so the per-model OpenAI
|
||||
* scoping applies; DeepSeek server-maps low/medium to high and xhigh to max.
|
||||
* Sentinel for the Anthropic off case. Like the DeepSeek one it never reaches
|
||||
* the wire as an effort: the 'anthropic' branch of `applyReasoningToConfig`
|
||||
* translates it to `thinking: {type: "disabled"}`, the only off that Opus and
|
||||
* Sonnet 5 respect.
|
||||
*/
|
||||
function openrouterReasoningLevels(model: string): ReasoningEffort[] {
|
||||
const m = model.toLowerCase()
|
||||
const base = baseModelId(model)
|
||||
if (/claude-(opus|sonnet)-(4|5)/.test(m)) {
|
||||
return ['minimal', 'low', 'medium', 'high', 'xhigh']
|
||||
}
|
||||
if (m.includes('gemini-')) {
|
||||
return geminiReasoningLevels(m)
|
||||
}
|
||||
if (base.startsWith('gpt-5') || /^o\d/.test(base)) {
|
||||
return openaiReasoningLevels(base)
|
||||
}
|
||||
if (m.includes('deepseek-v4')) {
|
||||
return ['high', 'xhigh']
|
||||
}
|
||||
return ['low', 'medium', 'high']
|
||||
}
|
||||
export const ANTHROPIC_OFF_SENTINEL: ReasoningEffort = 'none'
|
||||
|
||||
/**
|
||||
* OpenAI's effort vocabulary is model-dependent: `minimal` exists on gpt-5 but
|
||||
* not on gpt-5.1+, `xhigh` arrived on gpt-5.5 and `max` on gpt-5.6; o-series
|
||||
* take low/medium/high. An unsupported level is rejected, so scope the list to
|
||||
* the model. (`none` is the disable token, handled by `explicitOffToken`.)
|
||||
* What the registry knows about one group of models. The rows below are matched in
|
||||
* order against the lowercased model id, first match wins, so a narrower row goes above
|
||||
* the row it overrides. A model no row matches does not reason, as far as we know.
|
||||
*/
|
||||
function openaiReasoningLevels(model: string): ReasoningEffort[] {
|
||||
const base = baseModelId(model)
|
||||
if (/^gpt-5\.6/.test(base)) {
|
||||
return ['low', 'medium', 'high', 'xhigh', 'max']
|
||||
}
|
||||
if (/^gpt-5\.5/.test(base)) {
|
||||
return ['low', 'medium', 'high', 'xhigh']
|
||||
}
|
||||
if (/^gpt-5\./.test(base)) {
|
||||
return ['low', 'medium', 'high']
|
||||
}
|
||||
if (/^gpt-5/.test(base)) {
|
||||
return ['minimal', 'low', 'medium', 'high']
|
||||
}
|
||||
return ['low', 'medium', 'high']
|
||||
type ReasoningRule = {
|
||||
match: RegExp
|
||||
/** The levels the UI offers. An unsupported level is a 400, so this is per model. */
|
||||
levels: readonly ReasoningEffort[]
|
||||
/**
|
||||
* Whether "off" really stops the model reasoning. When false the UI offers no off:
|
||||
* the provider would reject it, or coerce it to the lowest level.
|
||||
*/
|
||||
canDisable: boolean
|
||||
/**
|
||||
* The effort to send for off on a model that reasons when the field is omitted.
|
||||
* Unset where omission is already off. Always `'none'`: `requestsReasoning` reads
|
||||
* that as off, and each wire format translates it (see `applyReasoningToConfig`).
|
||||
*/
|
||||
offToken?: ReasoningEffort
|
||||
/**
|
||||
* gpt-5.5 and later refuse function tools on Chat Completions while they reason,
|
||||
* even with the effort omitted (live-verified); the Responses API has no such limit.
|
||||
*/
|
||||
completionsToolsNeedOff?: boolean
|
||||
}
|
||||
|
||||
const LOW_TO_HIGH = ['low', 'medium', 'high']
|
||||
const LOW_TO_XHIGH = ['low', 'medium', 'high', 'xhigh']
|
||||
const LOW_TO_MAX = ['low', 'medium', 'high', 'xhigh', 'max']
|
||||
const MINIMAL_TO_HIGH = ['minimal', 'low', 'medium', 'high']
|
||||
|
||||
/**
|
||||
* Gemini's level ladder is model-dependent: Gemini 3+ Flash / Flash-Lite accept
|
||||
* `minimal`, while 3.x Pro does not (and cannot disable thinking). Gemini 2.5
|
||||
* uses numeric budgets — the proxy maps the three tiers to budget values, so
|
||||
* `minimal` is not offered there.
|
||||
* Claude models whose thinking cannot be turned off: an explicit disable 400s. Fable,
|
||||
* Mythos, and the 5.x point releases (Sonnet 5.5, Opus 5.5), whose lowest setting is
|
||||
* adaptive thinking at `low` (live-verified). The version match stops at one digit so a
|
||||
* dated id (`claude-sonnet-5-20260101`) stays Sonnet 5.
|
||||
*/
|
||||
function geminiReasoningLevels(model: string): ReasoningEffort[] {
|
||||
const m = model.toLowerCase()
|
||||
const isGemini3Plus = !m.includes('gemini-2.5')
|
||||
if (isGemini3Plus && (m.includes('flash') || m.includes('lite'))) {
|
||||
return ['minimal', 'low', 'medium', 'high']
|
||||
}
|
||||
return ['low', 'medium', 'high']
|
||||
const CLAUDE_ALWAYS_THINKING = /fable|mythos|claude-(opus|sonnet)-5[-.][1-9](?!\d)/
|
||||
|
||||
// Anthropic ids are matched anywhere in the id: Bedrock prefixes them
|
||||
// (`us.anthropic.claude-opus-4-6-v1`). Opus 4.5 and older reject adaptive thinking.
|
||||
const ANTHROPIC_RULES: ReasoningRule[] = [
|
||||
{ match: CLAUDE_ALWAYS_THINKING, levels: LOW_TO_MAX, canDisable: false },
|
||||
// The 5 family thinks when the field is absent, so off is an explicit disable.
|
||||
{
|
||||
match: /claude-(opus|sonnet)-5/,
|
||||
levels: LOW_TO_MAX,
|
||||
canDisable: true,
|
||||
offToken: ANTHROPIC_OFF_SENTINEL
|
||||
},
|
||||
// 4.6-4.8 only think when asked, so omission is already off.
|
||||
{ match: /claude-opus-4-[78]/, levels: LOW_TO_MAX, canDisable: true },
|
||||
{ match: /claude-(opus|sonnet)-4-6/, levels: ['low', 'medium', 'high', 'max'], canDisable: true }
|
||||
]
|
||||
|
||||
const BEDROCK_RULES: ReasoningRule[] = [
|
||||
ANTHROPIC_RULES[0],
|
||||
// AWS documents Sonnet 5 on Bedrock as always thinking, where the native API
|
||||
// accepts a disable for it.
|
||||
{ match: /claude-sonnet-5/, levels: LOW_TO_MAX, canDisable: false },
|
||||
...ANTHROPIC_RULES.slice(1)
|
||||
]
|
||||
|
||||
// Anchored at the start or after a gateway's `vendor/`, and the major is one digit:
|
||||
// Azure names gpt-3.5 `gpt-35-turbo`. `minimal` exists on gpt-5 only, `xhigh` from
|
||||
// gpt-5.5, `max` from gpt-5.6.
|
||||
const OPENAI_RULES: ReasoningRule[] = [
|
||||
// Live-verified: astra takes low..max only, where sol and luna also take `none`.
|
||||
{
|
||||
match: /(?:^|\/)gpt-6-astra/,
|
||||
levels: LOW_TO_MAX,
|
||||
canDisable: false,
|
||||
completionsToolsNeedOff: true
|
||||
},
|
||||
{
|
||||
match: /(?:^|\/)gpt-[6-9](?:[.:-]|$)/,
|
||||
levels: LOW_TO_MAX,
|
||||
canDisable: true,
|
||||
offToken: 'none',
|
||||
completionsToolsNeedOff: true
|
||||
},
|
||||
{
|
||||
match: /(?:^|\/)gpt-5\.6/,
|
||||
levels: LOW_TO_MAX,
|
||||
canDisable: true,
|
||||
offToken: 'none',
|
||||
completionsToolsNeedOff: true
|
||||
},
|
||||
{
|
||||
match: /(?:^|\/)gpt-5\.5/,
|
||||
levels: LOW_TO_XHIGH,
|
||||
canDisable: true,
|
||||
offToken: 'none',
|
||||
completionsToolsNeedOff: true
|
||||
},
|
||||
// A later gpt-5 minor keeps the tools limit, which holds for every version from 5.5,
|
||||
// but only the levels every gpt-5.x takes until it has a row of its own.
|
||||
{
|
||||
match: /(?:^|\/)gpt-5\.(?:[5-9]|\d{2,})/,
|
||||
levels: LOW_TO_HIGH,
|
||||
canDisable: true,
|
||||
offToken: 'none',
|
||||
completionsToolsNeedOff: true
|
||||
},
|
||||
// gpt-5.1+ are off only through `none`: omitted, they reason at medium.
|
||||
{ match: /(?:^|\/)gpt-5\./, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' },
|
||||
// gpt-5 and the o-series reject `none` and reason when it is omitted.
|
||||
{ match: /(?:^|\/)gpt-5(?:[:-]|$)/, levels: MINIMAL_TO_HIGH, canDisable: false },
|
||||
{ match: /(?:^|\/)o\d/, levels: LOW_TO_HIGH, canDisable: false }
|
||||
]
|
||||
|
||||
// Gemini 2.5/3 think by default; the backend proxy maps `none` to off on Flash, or to
|
||||
// the floor on Pro, which enforces one (level `low` on 3.x, 128 tokens on 2.5).
|
||||
// Gemini 3+ Flash / Flash-Lite accept `minimal`; 2.5 takes numeric budgets the proxy
|
||||
// maps from three tiers.
|
||||
const GEMINI_RULES: ReasoningRule[] = [
|
||||
{ match: /gemini-2\.5.*pro/, levels: LOW_TO_HIGH, canDisable: false, offToken: 'none' },
|
||||
{ match: /gemini-2\.5/, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' },
|
||||
{ match: /gemini-3.*pro/, levels: LOW_TO_HIGH, canDisable: false, offToken: 'none' },
|
||||
{ match: /gemini-3.*(flash|lite)/, levels: MINIMAL_TO_HIGH, canDisable: true, offToken: 'none' },
|
||||
{ match: /gemini-3/, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' }
|
||||
]
|
||||
|
||||
const REASONING_RULES: Partial<Record<AIProvider, ReasoningRule[]>> = {
|
||||
anthropic: ANTHROPIC_RULES,
|
||||
aws_bedrock: BEDROCK_RULES,
|
||||
openai: OPENAI_RULES,
|
||||
azure_openai: OPENAI_RULES,
|
||||
googleai: GEMINI_RULES,
|
||||
deepseek: [
|
||||
// Every current API model takes reasoning_effort, but only two levels are real:
|
||||
// low/medium are server-mapped to high, xhigh to max. The retired
|
||||
// `deepseek-chat` alias means "non-thinking mode", so a saved selection on it
|
||||
// must not silently become a thinking request. Off is a separate `thinking` param.
|
||||
{
|
||||
match: /(?:^|\/)deepseek(?!-chat(?::|$))/,
|
||||
levels: ['high', 'max'],
|
||||
canDisable: true,
|
||||
offToken: DEEPSEEK_OFF_SENTINEL
|
||||
}
|
||||
],
|
||||
mistral: [
|
||||
// Only the ids verified to accept reasoning_effort (large, magistral, ministral
|
||||
// and pinned versions reject it), whose only effort besides off is `high`.
|
||||
{
|
||||
match: /(?:^|\/)mistral-(?:(?:small|medium)-latest(?::|$)|medium-3[-.]5)/,
|
||||
levels: ['high'],
|
||||
canDisable: true
|
||||
}
|
||||
],
|
||||
// OpenRouter validates effort against its own vocabulary (minimal..xhigh + none) and
|
||||
// translates it per underlying provider, which scopes the ladder: Anthropic gets all
|
||||
// five as budget ratios, OpenAI gets the token verbatim, DeepSeek server-maps. `none`
|
||||
// is its documented off, more reliable than omission; it can only disable a model
|
||||
// whose upstream can.
|
||||
openrouter: [
|
||||
{
|
||||
match: /claude-(opus|sonnet)-5[-.][1-9](?!\d)/,
|
||||
levels: ['minimal', ...LOW_TO_XHIGH],
|
||||
canDisable: false
|
||||
},
|
||||
{
|
||||
match: /claude-(opus|sonnet)-(4|5)/,
|
||||
levels: ['minimal', ...LOW_TO_XHIGH],
|
||||
canDisable: true,
|
||||
offToken: 'none'
|
||||
},
|
||||
...GEMINI_RULES,
|
||||
...OPENAI_RULES.map(({ completionsToolsNeedOff: _, ...rule }) => ({
|
||||
...rule,
|
||||
offToken: 'none'
|
||||
})),
|
||||
{ match: /deepseek-v4/, levels: LOW_TO_XHIGH.slice(2), canDisable: true, offToken: 'none' },
|
||||
// deepseek-r1, grok-4 and :thinking variants reason unconditionally.
|
||||
{
|
||||
match: /deepseek-r|grok-4|:thinking/,
|
||||
levels: LOW_TO_HIGH,
|
||||
canDisable: false,
|
||||
offToken: 'none'
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
/**
|
||||
* Gemini Pro models cannot turn thinking off — the API enforces a floor
|
||||
* (level `low` on 3.x Pro, a 128-token budget on 2.5 Pro), so an off option
|
||||
* would silently mean "lowest". Flash / Flash-Lite can truly disable
|
||||
* (budget 0 / level `minimal`).
|
||||
*/
|
||||
function geminiCanDisable(model: string): boolean {
|
||||
return !model.toLowerCase().includes('pro')
|
||||
/** The row that describes a model, and whether its provider family has rows at all. */
|
||||
function findReasoningRule(
|
||||
provider: AIProvider,
|
||||
model: string
|
||||
): { rule: ReasoningRule | undefined; known: boolean } {
|
||||
const rules = REASONING_RULES[reasoningProviderFamily(provider, model)]
|
||||
const id = stripLegacyThinkingSuffix(model).toLowerCase()
|
||||
return { rule: rules?.find((rule) => rule.match.test(id)), known: rules !== undefined }
|
||||
}
|
||||
|
||||
/**
|
||||
* Anthropic's effort ladder is model-dependent: `xhigh` exists on Opus 4.7/4.8,
|
||||
* the 5 family and Fable/Mythos; `max` also on Opus 4.6 and Sonnet 4.6.
|
||||
* Offering an unsupported level would 400, so scope the list to the model.
|
||||
*/
|
||||
function anthropicReasoningLevels(model: string): ReasoningEffort[] {
|
||||
const m = model.toLowerCase()
|
||||
if (
|
||||
/claude-(opus|sonnet)-5/.test(m) ||
|
||||
/claude-opus-4-(7|8)/.test(m) ||
|
||||
m.includes('fable') ||
|
||||
m.includes('mythos')
|
||||
) {
|
||||
return ['low', 'medium', 'high', 'xhigh', 'max']
|
||||
}
|
||||
return ['low', 'medium', 'high', 'max']
|
||||
}
|
||||
|
||||
/** Mistral writes both `mistral-medium-3.5` and `mistral-medium-3-5`. */
|
||||
function normalizeMistralId(model: string): string {
|
||||
return baseModelId(model).replace(/\./g, '-')
|
||||
}
|
||||
|
||||
/**
|
||||
* Conservative static predicate for whether a model accepts an effort knob.
|
||||
* Kept tight to avoid 400s on models that reject reasoning params.
|
||||
*/
|
||||
function supportsReasoningStatic(provider: AIProvider, model: string): boolean {
|
||||
const m = model.toLowerCase()
|
||||
const base = baseModelId(model)
|
||||
switch (reasoningProviderFamily(provider, model)) {
|
||||
case 'anthropic':
|
||||
// Bedrock serves the same Claude models under prefixed ids
|
||||
// (e.g. `us.anthropic.claude-opus-4-6-v1`), so match on the full string.
|
||||
case 'aws_bedrock':
|
||||
// 4.6+ only: Opus 4.5 rejects adaptive thinking (and, on Bedrock,
|
||||
// the whole output_config surface) — live-verified hard 400.
|
||||
return (
|
||||
/claude-opus-(4-(6|7|8)|5)/.test(m) ||
|
||||
/claude-sonnet-(4-6|5)/.test(m) ||
|
||||
m.includes('fable') ||
|
||||
m.includes('mythos')
|
||||
)
|
||||
case 'openai':
|
||||
case 'azure_openai':
|
||||
return base.startsWith('gpt-5') || /^o\d/.test(base)
|
||||
case 'openrouter':
|
||||
// Best-effort markers for models whose `supported_parameters` include
|
||||
// `reasoning` in OpenRouter's catalog; OpenRouter translates the effort
|
||||
// per underlying provider.
|
||||
return (
|
||||
base.startsWith('gpt-5') ||
|
||||
/^o\d/.test(base) ||
|
||||
/claude-(opus|sonnet)-(4|5)/.test(m) ||
|
||||
/gemini-(2\.5|3)/.test(m) ||
|
||||
m.includes('deepseek-r') ||
|
||||
m.includes('deepseek-v4') ||
|
||||
m.includes('grok-4') ||
|
||||
m.includes(':thinking')
|
||||
)
|
||||
case 'googleai':
|
||||
return /gemini-(2\.5|3)/.test(m)
|
||||
case 'deepseek':
|
||||
// All current API models take reasoning_effort (live-verified). The
|
||||
// retired `deepseek-chat` alias stays excluded: its documented meaning
|
||||
// is "non-thinking mode", so a saved selection on it must not silently
|
||||
// become a thinking request.
|
||||
return base.startsWith('deepseek') && base !== 'deepseek-chat'
|
||||
case 'mistral':
|
||||
// Only the ids verified to accept reasoning_effort; other models
|
||||
// (large, magistral, ministral, pinned versions) reject the param.
|
||||
return (
|
||||
/^mistral-(small|medium)-latest$/.test(base) ||
|
||||
normalizeMistralId(model).startsWith('mistral-medium-3-5')
|
||||
)
|
||||
default:
|
||||
return false
|
||||
}
|
||||
/** Whether Chat Completions needs reasoning off for this model to take function tools. */
|
||||
export function completionsRejectsToolsWithReasoning(provider: AIProvider, model: string): boolean {
|
||||
return findReasoningRule(provider, model).rule?.completionsToolsNeedOff ?? false
|
||||
}
|
||||
|
||||
export type ReasoningCapability = {
|
||||
@@ -241,89 +282,12 @@ export type ReasoningCapability = {
|
||||
known: boolean
|
||||
}
|
||||
|
||||
/** Provider families the registry has real rules for; everything else is a shrug. */
|
||||
const KNOWN_REASONING_FAMILIES: ReadonlySet<string> = new Set([
|
||||
'anthropic',
|
||||
'aws_bedrock',
|
||||
'openai',
|
||||
'azure_openai',
|
||||
'openrouter',
|
||||
'googleai',
|
||||
'deepseek',
|
||||
'mistral'
|
||||
])
|
||||
|
||||
/** Resolve the reasoning capability of a model from the static registry. */
|
||||
export function getReasoningCapability(provider: AIProvider, model: string): ReasoningCapability {
|
||||
const bareModel = stripLegacyThinkingSuffix(model)
|
||||
const known = KNOWN_REASONING_FAMILIES.has(reasoningProviderFamily(provider, bareModel))
|
||||
const supported = supportsReasoningStatic(provider, bareModel)
|
||||
if (!supported) {
|
||||
return { supported: false, levels: [], canDisable: false, known }
|
||||
}
|
||||
const family = reasoningProviderFamily(provider, bareModel)
|
||||
const levels =
|
||||
family === 'anthropic' || family === 'aws_bedrock'
|
||||
? anthropicReasoningLevels(bareModel)
|
||||
: family === 'googleai'
|
||||
? geminiReasoningLevels(bareModel)
|
||||
: family === 'openai' || family === 'azure_openai'
|
||||
? openaiReasoningLevels(bareModel)
|
||||
: family === 'openrouter'
|
||||
? openrouterReasoningLevels(bareModel)
|
||||
: (PROVIDER_REASONING_LEVELS[family] ?? ['low', 'medium', 'high'])
|
||||
return { supported, levels, canDisable: canDisableReasoning(provider, bareModel), known }
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether selecting "off" truly disables reasoning for the model. Off is
|
||||
* sent either as an explicit provider disable (see `explicitOffToken`) or by
|
||||
* omitting the effort — which only works where the model doesn't reason by
|
||||
* default.
|
||||
*/
|
||||
function canDisableReasoning(provider: AIProvider, model: string): boolean {
|
||||
const m = model.toLowerCase()
|
||||
const base = baseModelId(model)
|
||||
switch (reasoningProviderFamily(provider, model)) {
|
||||
case 'anthropic':
|
||||
// Every Claude but Fable and Mythos can stop thinking: 4.6-4.8 by
|
||||
// omission, and the 5 family through the explicit disable that
|
||||
// `explicitOffToken` sends.
|
||||
return !ANTHROPIC_ALWAYS_THINKING.test(m)
|
||||
case 'aws_bedrock':
|
||||
// Same models, different answer: AWS documents Sonnet 5 on Bedrock as
|
||||
// always thinking, where the native API accepts a disable for it.
|
||||
return !(ANTHROPIC_ALWAYS_THINKING.test(m) || m.includes('claude-sonnet-5'))
|
||||
case 'googleai':
|
||||
return geminiCanDisable(model)
|
||||
case 'openai':
|
||||
case 'azure_openai':
|
||||
// gpt-5.1+ accept effort 'none'; gpt-5 and o-series reject it and
|
||||
// reason at `medium` by default, so omission isn't off either.
|
||||
return /^gpt-5\./.test(base)
|
||||
case 'openrouter':
|
||||
// 'none' is in OpenRouter's vocabulary, but the gateway can't
|
||||
// disable a model whose upstream can't — scope off per underlying
|
||||
// family, like the levels.
|
||||
// The 5 family thinks by default, but its upstream takes an explicit
|
||||
// disable, so the gateway's 'none' has something to translate to.
|
||||
if (/claude-(opus|sonnet)-(4|5)/.test(m)) {
|
||||
return true
|
||||
}
|
||||
if (m.includes('gemini-')) {
|
||||
return geminiCanDisable(m)
|
||||
}
|
||||
if (base.startsWith('gpt-5') || /^o\d/.test(base)) {
|
||||
return /^gpt-5\./.test(base)
|
||||
}
|
||||
if (m.includes('deepseek-v4')) {
|
||||
return true
|
||||
}
|
||||
// grok-4, deepseek-r1 and :thinking variants reason unconditionally.
|
||||
return false
|
||||
default:
|
||||
return true
|
||||
}
|
||||
const { rule, known } = findReasoningRule(provider, model)
|
||||
return rule
|
||||
? { supported: true, levels: [...rule.levels], canDisable: rule.canDisable, known }
|
||||
: { supported: false, levels: [], canDisable: false, known }
|
||||
}
|
||||
|
||||
export function supportsReasoning(provider: AIProvider, model: string): boolean {
|
||||
@@ -352,64 +316,14 @@ export function resolveEffectiveReasoning(
|
||||
: undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Sentinel sent for the deepseek off case. It never reaches the wire as an
|
||||
* effort: the 'deepseek' branch of `applyReasoningToConfig` translates it to
|
||||
* the provider's `thinking: {type: "disabled"}` param (`reasoning_effort:
|
||||
* "none"` is rejected by their API).
|
||||
*/
|
||||
export const DEEPSEEK_OFF_SENTINEL: ReasoningEffort = 'none'
|
||||
|
||||
/**
|
||||
* Sentinel for the Anthropic off case. Like the DeepSeek one it never reaches
|
||||
* the wire as an effort: the 'anthropic' branch of `applyReasoningToConfig`
|
||||
* translates it to `thinking: {type: "disabled"}`, which is the only off the
|
||||
* always-on 5 family respects.
|
||||
*/
|
||||
export const ANTHROPIC_OFF_SENTINEL: ReasoningEffort = 'none'
|
||||
|
||||
/** Claude models whose thinking cannot be turned off — an explicit disable 400s. */
|
||||
const ANTHROPIC_ALWAYS_THINKING = /fable|mythos/
|
||||
|
||||
/**
|
||||
* Disable token to forward when the user explicitly turns reasoning off on a
|
||||
* model that reasons *by default* — omitting the field would silently keep
|
||||
* the default-on behavior. Undefined means omission is the correct off. Every
|
||||
* token is `'none'`: `requestsReasoning` reads that value as off.
|
||||
* the default-on behavior. Undefined means omission is the correct off, or that
|
||||
* the model cannot be turned off at all.
|
||||
*/
|
||||
export function explicitOffToken(provider: AIProvider, model: string): ReasoningEffort | undefined {
|
||||
switch (reasoningProviderFamily(provider, model)) {
|
||||
case 'anthropic':
|
||||
// Claude 4.6-4.8 only think when asked, so omission is already a
|
||||
// real off there and stays the wire form. Only the 5 family, which
|
||||
// thinks when the field is absent, needs the explicit disable —
|
||||
// Fable and Mythos reject it outright and get no off token at all.
|
||||
return /claude-(opus|sonnet)-5/.test(model.toLowerCase()) ? ANTHROPIC_OFF_SENTINEL : undefined
|
||||
case 'aws_bedrock':
|
||||
// Bedrock's Sonnet 5 cannot be disabled at all, so only Opus 5 gets
|
||||
// the sentinel; the rest keep omission.
|
||||
return model.toLowerCase().includes('claude-opus-5') ? ANTHROPIC_OFF_SENTINEL : undefined
|
||||
case 'googleai':
|
||||
// Gemini 2.5/3 think by default (dynamic budget / level). The backend
|
||||
// proxy maps 'none' to off on Flash, or the floor on Pro (only
|
||||
// reachable via a stale persisted preference — see canDisableReasoning).
|
||||
return 'none'
|
||||
case 'deepseek':
|
||||
return DEEPSEEK_OFF_SENTINEL
|
||||
case 'openai':
|
||||
case 'azure_openai':
|
||||
// gpt-5.1+ reasoning is off only via the explicit 'none' effort
|
||||
// (gpt-5.5 defaults to medium when the field is omitted).
|
||||
return /^gpt-5\./.test(baseModelId(model)) ? 'none' : undefined
|
||||
case 'openrouter':
|
||||
// OpenRouter validates effort against xhigh..minimal|none and
|
||||
// documents 'none' as disabling reasoning, translated per the
|
||||
// underlying provider — more reliable than omission, which keeps
|
||||
// reasoning-by-default models thinking.
|
||||
return 'none'
|
||||
default:
|
||||
return undefined
|
||||
}
|
||||
return findReasoningRule(provider, model).rule?.offToken
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
Reference in New Issue
Block a user