weekly ai evals on current models, and claude 5.5/gpt-6 support (#11409)

* feat: run ai evals weekly on current models and post results to a dashboard

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* feat: add current flagship models, a reasoning flag and claude 5.5 defaults

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: never send a reasoning disable claude 5.5 or gpt-6-astra reject, and treat gpt-6 as a reasoning model

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: address review on gpt-6 support, chat completions tools and model metadata

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: leave tiered gpt-6 unpriced and drop the off sentinel on gpt-5 and o-series

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* docs: point the ai_evals readme at the model registry instead of copying it

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* refactor: encode the reasoning rules as per-family maps with a shared parity fixture

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: keep the chat completions tools rule open-ended past gpt-5.6 and scope the parity fixture

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
hugocasa
2026-09-30 13:10:24 +02:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 94ac34a489
commit e117871bec
20 changed files with 888 additions and 414 deletions
+171
View File
@@ -0,0 +1,171 @@
name: AI Evals (scheduled)
# Full ai_evals suites on current models, posted to the AI evals dashboard in
# the windmill-prod workspace (f/ai/ai_evals_dashboard) so quality is tracked
# over time. Weekly, since one pass of every suite at --runs 3 costs about 50M
# tokens per model: every suite runs on MODELS, and global (the mode users get)
# also runs on GLOBAL_EXTRA_MODELS, one flagship per other provider. Run it by
# hand to measure a branch against main.
on:
schedule:
- cron: "0 3 * * 1"
workflow_dispatch:
inputs:
modes:
description: "Space-separated modes"
default: "global flow app script cli"
models:
description: "Space-separated model aliases (bun run cli -- models)"
default: "sonnet-5.5"
global_extra_models:
description: "Extra model aliases for global mode only"
default: "gpt-6-astra gemini-3.8-flash"
runs:
description: "Runs per case"
default: "3"
reasoning:
description: "Reasoning effort for frontend modes (empty: the product default)"
default: ""
concurrency:
group: ai-evals-scheduled-${{ github.ref }}
env:
MODELS: ${{ inputs.models || 'sonnet-5.5' }}
GLOBAL_EXTRA_MODELS: ${{ inputs.global_extra_models || 'gpt-6-astra gemini-3.8-flash' }}
RUNS: ${{ inputs.runs || '3' }}
REASONING: ${{ inputs.reasoning }}
INGEST_URL: https://app.windmill.dev/api/r/f/ai/ingest_ai_eval_run
jobs:
setup:
runs-on: ubuntu-latest
outputs:
modes: ${{ steps.modes.outputs.modes }}
steps:
- id: modes
env:
MODES: ${{ inputs.modes || 'global flow app script cli' }}
run: |
echo "modes=$(jq -cn --arg m "$MODES" '$m | split(" ")
| map(select(IN("global", "flow", "app", "script", "cli")))')" >> "$GITHUB_OUTPUT"
evals:
needs: setup
runs-on: ubicloud-standard-16
timeout-minutes: 330
strategy:
fail-fast: false
matrix:
mode: ${{ fromJSON(needs.setup.outputs.modes) }}
services:
postgres:
image: postgres:16
ports:
- 5432:5432
env:
POSTGRES_DB: windmill
POSTGRES_PASSWORD: changeme
options: >-
--health-cmd pg_isready --health-interval 10s --health-timeout 5s
--health-retries 5
steps:
- uses: actions/checkout@v4
- uses: actions-rust-lang/setup-rust-toolchain@v1
with:
cache-workspaces: backend
toolchain: 1.97.0
- uses: oven-sh/setup-bun@v2
with:
bun-version: 1.4.0
- uses: actions/setup-node@v7
with:
node-version: "24"
- name: Build Windmill
working-directory: ./backend
env:
SQLX_OFFLINE: true
CARGO_BUILD_JOBS: 12
RUSTFLAGS: ""
run: cargo build --features quickjs
- name: Start Windmill
working-directory: ./backend
env:
DATABASE_URL: postgres://postgres:changeme@localhost:5432/windmill
RUST_LOG: info
run: |
mkdir -p ../ai_evals/logs
./target/debug/windmill > ../ai_evals/logs/windmill.log 2>&1 &
for i in $(seq 1 60); do
curl -sf http://localhost:8000/api/version > /dev/null 2>&1 && break
sleep 2
done
curl -sf http://localhost:8000/api/version > /dev/null || { tail -50 ../ai_evals/logs/windmill.log; exit 1; }
- name: Install frontend deps + generate client
working-directory: ./frontend
run: |
npm ci
npm run generate-backend-client
- name: Install CLI deps + generate CLI client
working-directory: ./cli
run: bun install && ./gen_wm_client.sh && ./windmill-utils-internal/gen_wm_client.sh
- name: Run ${{ matrix.mode }} evals
working-directory: ./ai_evals
env:
MODE: ${{ matrix.mode }}
WMILL_AI_EVAL_BACKEND_URL: http://localhost:8000
WMILL_AI_EVAL_BACKEND_WORKSPACE: integration-tests
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }}
run: |
bun install
mkdir -p results
models="$MODELS"
[ "$MODE" = global ] && models="$models $GLOBAL_EXTRA_MODELS"
reasoning=()
[ -n "$REASONING" ] && [ "$MODE" != cli ] && reasoning=(--reasoning "$REASONING")
for m in $models; do
bun run cli -- run "$MODE" --model "$m" --runs "$RUNS" "${reasoning[@]}" \
--output "$PWD/results/$MODE-$m.json" || echo "::warning::$MODE on $m errored"
done
- name: Post results to the dashboard
if: always()
working-directory: ./ai_evals
env:
MODE: ${{ matrix.mode }}
INGEST_TOKEN: ${{ secrets.AI_EVALS_INGEST_TOKEN }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
[ -n "$INGEST_TOKEN" ] || { echo "::warning::AI_EVALS_INGEST_TOKEN is not set"; exit 0; }
shopt -s nullglob
for f in results/"$MODE"-*.json; do
# Keep what the dashboard reads; the full traces stay in the run's artifacts.
jq -c --arg ref "$GITHUB_REF" --arg trigger "$GITHUB_EVENT_NAME" --arg url "$RUN_URL" '
{result_json: (del(.cases[].prompt, .cases[].initialPath, .cases[].expectedPath)
| .cases[].attempts[] |= {attempt, passed, durationMs, toolCallCount, judgeScore,
judgeSummary, error, tokenUsage, checks: [.checks[]? | {name, passed}]}),
git_ref: $ref, trigger: $trigger, run_url: $url}' "$f" \
| curl -sf --retry 3 -X POST "$INGEST_URL" \
-H "Authorization: Bearer $INGEST_TOKEN" -H 'content-type: application/json' \
--data-binary @- && echo " <- $f" || echo "::warning::failed to post $f"
done
- name: Archive logs and results
uses: actions/upload-artifact@v4
if: always()
with:
name: ai-evals-${{ matrix.mode }}
path: |
ai_evals/logs
ai_evals/results
+3 -14
View File
@@ -77,31 +77,20 @@ Public CLI surface:
- `--verbose`: stream assistant output for frontend runs
- `--skip-judge`: skip LLM judge scoring for the run
- `--execution-only`: only require the model/proxy/frontend loop to complete; skip validators, tool expectations, backend artifact validation, and judge scoring
- `--reasoning <effort>`: reasoning effort for frontend modes (`off`, `low`, `medium`, `high`, `max`, …); without it the product's default applies (`high` on models that can reason). The effort is appended to the recorded model label (`anthropic:claude-sonnet-5-5@max`)
- `--record`: append a compact tracked summary line to `ai_evals/history/<mode>.jsonl` for full-suite runs only
- `--backend-validation <mode>`: optional backend smoke validation (`off` or `preview`) for `script` and `flow` evals
## Models
Use `bun run cli -- models` to see the current aliases.
Today:
- `haiku`
- `sonnet`
- `opus`
- `4o`
- `gpt-5.5`
- `gemini-3-flash-preview`
- `gemini-3.1-pro-preview`
- `deepseek-v4-flash`
- `deepseek-v4-pro`
Use `bun run cli -- models` to see the current aliases; `core/models.ts` is the list.
Notes:
- the command also prints accepted alias spellings such as `gpt-4o`, `gpt-55`, `claude-opus-4.6`, and `claude-haiku-4.5`
- frontend modes (`flow`, `script`, `app`, `global`) can use Anthropic, OpenAI, Gemini, and DeepSeek-backed aliases
- `cli` mode always uses the Anthropic agent SDK, so only Anthropic aliases are valid there
- the judge model is separate and currently defaults to `claude-sonnet-4-6`; use `--skip-judge` for deterministic-only runs
- the judge model is separate and currently defaults to `claude-sonnet-5-5`; use `--skip-judge` for deterministic-only runs
## Case Format
@@ -15,6 +15,7 @@ import { runEval } from "../shared";
import type { ModeRunContext } from "../../../../core/types";
import type { TokenUsage, ToolCallDetail } from "../shared/types";
import type { WindmillBackendSettings } from "../../../../core/windmillBackendSettings";
import { evalReasoningEffort } from "../shared/providerConfig";
export interface ScriptEvalResult {
success: boolean;
@@ -41,14 +42,15 @@ export interface ScriptEvalOptions {
function resolveModelProvider(
model: string,
provider?: AIProvider,
): AIProviderModel {
): AIProviderModel & { reasoning?: string } {
const reasoning = evalReasoningEffort();
if (provider) {
return { provider, model };
return { provider, model, reasoning };
}
if (model.startsWith("claude")) {
return { provider: "anthropic", model };
return { provider: "anthropic", model, reasoning };
}
return { provider: "openai", model };
return { provider: "openai", model, reasoning };
}
export async function runScriptEval(
@@ -12,6 +12,12 @@ export interface EvalClients {
export interface ResolvedEvalModelProvider {
provider: FrontendEvalProvider;
model: string;
reasoning?: string;
}
/** `run --reasoning` hands the effort to the frontend runtime through the environment. */
export function evalReasoningEffort(): string | undefined {
return process.env.WMILL_AI_EVAL_REASONING || undefined;
}
export interface WindmillAiProxyClientConfig {
@@ -73,6 +79,13 @@ export function createEvalClients(input: {
export function resolveEvalModelProvider(
model: string,
provider?: FrontendEvalProvider,
): ResolvedEvalModelProvider {
return { ...resolveProvider(model, provider), reasoning: evalReasoningEffort() };
}
function resolveProvider(
model: string,
provider?: FrontendEvalProvider,
): ResolvedEvalModelProvider {
if (provider) {
return { provider, model };
+29 -91
View File
@@ -5,8 +5,8 @@
"": {
"name": "windmill-ai-evals",
"dependencies": {
"@anthropic-ai/claude-agent-sdk": "^0.2.25",
"@anthropic-ai/sdk": "^0.39.0",
"@anthropic-ai/claude-agent-sdk": "^0.3.284",
"@anthropic-ai/sdk": "^0.129.0",
"commander": "^14.0.3",
"openai": "^6.9.1",
"yaml": "^2.8.3",
@@ -18,66 +18,44 @@
},
},
"packages": {
"@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.2.87", "", { "dependencies": { "@anthropic-ai/sdk": "^0.74.0", "@modelcontextprotocol/sdk": "^1.27.1" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "^0.34.2", "@img/sharp-darwin-x64": "^0.34.2", "@img/sharp-linux-arm": "^0.34.2", "@img/sharp-linux-arm64": "^0.34.2", "@img/sharp-linux-x64": "^0.34.2", "@img/sharp-linuxmusl-arm64": "^0.34.2", "@img/sharp-linuxmusl-x64": "^0.34.2", "@img/sharp-win32-arm64": "^0.34.2", "@img/sharp-win32-x64": "^0.34.2" }, "peerDependencies": { "zod": "^4.0.0" } }, "sha512-WWmgBPxPhBOvNT0ujI8vPTI2lK+w5YEkEZ/y1mH0EDkK/0kBnxVJNhCtG5vnueiAViwLoUOFn66pbkDiivijdA=="],
"@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.3.284", "", { "optionalDependencies": { "@anthropic-ai/claude-agent-sdk-darwin-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-darwin-x64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-x64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-x64-musl": "0.3.284", "@anthropic-ai/claude-agent-sdk-win32-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-win32-x64": "0.3.284" }, "peerDependencies": { "@anthropic-ai/sdk": ">=0.93.0", "@modelcontextprotocol/sdk": "^1.29.0", "zod": "^4.0.0" } }, "sha512-NSoJwEq6nFSf8dtaacYx37QdGgqApI3eHFUlxclMLYi8irb6ZJwEaUnPnLUCyAyg9W/tMkgZC0GXWLRCc01I0w=="],
"@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.39.0", "", { "dependencies": { "@types/node": "^18.11.18", "@types/node-fetch": "^2.6.4", "abort-controller": "^3.0.0", "agentkeepalive": "^4.2.1", "form-data-encoder": "1.7.2", "formdata-node": "^4.3.2", "node-fetch": "^2.6.7" } }, "sha512-eMyDIPRZbt1CCLErRCi3exlAvNkBtRe+kW5vvJyef93PmNr/clstYgHhtvmkxN82nlKgzyGPCyGxrm0JQ1ZIdg=="],
"@anthropic-ai/claude-agent-sdk-darwin-arm64": ["@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.284", "", { "os": "darwin", "cpu": "arm64" }, "sha512-gKY9MUjY83398uCiPLHsd87kyzu7agIM7ApqWpJkpSINep6hxx4rNoR8bUbNWg49Aoe/PW/DJkQByAQuNzR9rg=="],
"@babel/runtime": ["@babel/runtime@7.29.2", "", {}, "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g=="],
"@anthropic-ai/claude-agent-sdk-darwin-x64": ["@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.284", "", { "os": "darwin", "cpu": "x64" }, "sha512-P+q6Z7sKeYz99uE7RJc4au1IK+4SiWe69pyHY3QXMnjZq8hSFU/5Mo8iSurLk9jRlihsMltV1rLA5naouYyCAQ=="],
"@anthropic-ai/claude-agent-sdk-linux-arm64": ["@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.284", "", { "os": "linux", "cpu": "arm64" }, "sha512-LDpuYDaz+pCdG29iy1pw4P1D2YVtpZOb4rWOPxxSpg2fFyUeQ5WrfnuszOxhDWpNh1k1ekDiKm3A9OzGcrkdtA=="],
"@anthropic-ai/claude-agent-sdk-linux-arm64-musl": ["@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.284", "", { "os": "linux", "cpu": "arm64" }, "sha512-U3/uAv1TS8sMWsgWQSxYbdND5AaRXXpP5RxHU1oYC5ZizlZE8/xAVum9dj1QX3rVZJg8EzsqZveq/AkEv5ABfQ=="],
"@anthropic-ai/claude-agent-sdk-linux-x64": ["@anthropic-ai/claude-agent-sdk-linux-x64@0.3.284", "", { "os": "linux", "cpu": "x64" }, "sha512-yGytBCCwJvWeg1FzFpgnLM9BOM2vKVExtv6pMJHMtF82J5yhKHcxodRaYlQVd/HzJJ2nruKb9EnrhRBZ//4yRw=="],
"@anthropic-ai/claude-agent-sdk-linux-x64-musl": ["@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.284", "", { "os": "linux", "cpu": "x64" }, "sha512-4x0Q8CFkCNbeENE7pRhuUmkwh/sQfTq1yWi4ZE56stve9pgoKzu6ZDa1SEtvK1YIooGE/qM5rVS8xFLE1gbQ0w=="],
"@anthropic-ai/claude-agent-sdk-win32-arm64": ["@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.284", "", { "os": "win32", "cpu": "arm64" }, "sha512-zOXkdPHhElyxyFJ6eY5R+YGUswqBDZ+ZHNaRmQGJ28RkQYgwG5imNy/nfFR+/CPGlMnH9ZzbGc2rKYApm1K4Tg=="],
"@anthropic-ai/claude-agent-sdk-win32-x64": ["@anthropic-ai/claude-agent-sdk-win32-x64@0.3.284", "", { "os": "win32", "cpu": "x64" }, "sha512-VGaFRDCOPloj5IjvJTyy5JsSl9sE2HR3W+A6f/cjgh23/7Y5vK7/ka+1JQD+AkoGO9tbHqP2O6god3IBrLvOaw=="],
"@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.129.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1", "standardwebhooks": "^1.0.0" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-MH7LB20kNpGLUpPTe2OG5XvL/KhX8HUO9XyBaFjgFk6TLS4Xjt0VPIuMKywe7POKKPslX+QxSlVok9e+8kG5Nw=="],
"@babel/runtime": ["@babel/runtime@7.29.7", "", {}, "sha512-Nq8OhGWiZIZGV6hLHoyAKLLcJihP/xFeBMGJoUrxTX2psI8dCifzLhZISFb+VWS3wFMRDmCGw5R+dOySCqPLhw=="],
"@hono/node-server": ["@hono/node-server@1.19.12", "", { "peerDependencies": { "hono": "^4" } }, "sha512-txsUW4SQ1iilgE0l9/e9VQWmELXifEFvmdA1j6WFh/aFPj99hIntrSsq/if0UWyGVkmrRPKA1wCeP+UCr1B9Uw=="],
"@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="],
"@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.2.4" }, "os": "darwin", "cpu": "x64" }, "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw=="],
"@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.2.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g=="],
"@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.2.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg=="],
"@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.2.4", "", { "os": "linux", "cpu": "arm" }, "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A=="],
"@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw=="],
"@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw=="],
"@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw=="],
"@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg=="],
"@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.2.4" }, "os": "linux", "cpu": "arm" }, "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw=="],
"@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg=="],
"@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ=="],
"@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg=="],
"@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q=="],
"@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.34.5", "", { "os": "win32", "cpu": "arm64" }, "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g=="],
"@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.34.5", "", { "os": "win32", "cpu": "x64" }, "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw=="],
"@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.29.0", "", { "dependencies": { "@hono/node-server": "^1.19.9", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.2.1", "express-rate-limit": "^8.2.1", "hono": "^4.11.4", "jose": "^6.1.3", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.1" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-zo37mZA9hJWpULgkRpowewez1y6ML5GsXJPY8FI0tBBCd77HEvza4jDqRKOXgHNn867PVGCyTdzqpz0izu5ZjQ=="],
"@stablelib/base64": ["@stablelib/base64@1.0.1", "", {}, "sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ=="],
"@types/bun": ["@types/bun@1.3.11", "", { "dependencies": { "bun-types": "1.3.11" } }, "sha512-5vPne5QvtpjGpsGYXiFyycfpDF2ECyPcTSsFBMa0fraoxiQyMJ3SmuQIGhzPg2WJuWxVBoxWJ2kClYTcw/4fAg=="],
"@types/node": ["@types/node@18.19.130", "", { "dependencies": { "undici-types": "~5.26.4" } }, "sha512-GRaXQx6jGfL8sKfaIDD6OupbIHBr9jv7Jnaml9tB7l4v068PAOXqfcujMMo5PhbIs6ggR1XODELqahT2R8v0fg=="],
"@types/node-fetch": ["@types/node-fetch@2.6.13", "", { "dependencies": { "@types/node": "*", "form-data": "^4.0.4" } }, "sha512-QGpRVpzSaUs30JBSGPjOg4Uveu384erbHBoT1zeONvyCfwQxIkUshLAOqN/k9EjGviPRmWTTe6aH2qySWKTVSw=="],
"abort-controller": ["abort-controller@3.0.0", "", { "dependencies": { "event-target-shim": "^5.0.0" } }, "sha512-h8lQ8tacZYnR3vNQTgibj+tODHI5/+l06Au2Pcriv/Gmet0eaj4TwWH41sO9wnHDiQsEj19q0drzdWdeAHtweg=="],
"@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="],
"accepts": ["accepts@2.0.0", "", { "dependencies": { "mime-types": "^3.0.0", "negotiator": "^1.0.0" } }, "sha512-5cvg6CtKwfgdmVqY1WIiXKc3Q1bkRqGLi+2W/6ao+6Y7gu/RCwRuAhGEzh5B4KlszSuTLgZYuqFqo5bImjNKng=="],
"agentkeepalive": ["agentkeepalive@4.6.0", "", { "dependencies": { "humanize-ms": "^1.2.1" } }, "sha512-kja8j7PjmncONqaTsB8fQ+wE2mSU2DJ9D4XKoJ5PFWIdRMa6SLSN1ff4mOr4jCbfRSsxR4keIiySJU0N9T5hIQ=="],
"ajv": ["ajv@8.18.0", "", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-PlXPeEWMXMZ7sPYOHqmDyCJzcfNrUr3fGNKtezX14ykXOEIvyK81d+qydx89KY5O71FKMPaQ2vBfBFI5NHR63A=="],
"ajv-formats": ["ajv-formats@3.0.1", "", { "dependencies": { "ajv": "^8.0.0" } }, "sha512-8iUql50EUR+uUcdRQ3HDqa6EVyo3docL8g5WJ3FNcWmu62IbkGUue/pEyLBW8VGKKucTPgqeks4fIU1DA4yowQ=="],
"asynckit": ["asynckit@0.4.0", "", {}, "sha512-Oei9OH4tRh0YqU3GxhX79dM/mwVgvbZJaSNaRk+bshkj0S5cfHcgYakreBjrHwatXKbz+IoIdYLxrKim2MjW0Q=="],
"body-parser": ["body-parser@2.2.2", "", { "dependencies": { "bytes": "^3.1.2", "content-type": "^1.0.5", "debug": "^4.4.3", "http-errors": "^2.0.0", "iconv-lite": "^0.7.0", "on-finished": "^2.4.1", "qs": "^6.14.1", "raw-body": "^3.0.1", "type-is": "^2.0.1" } }, "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA=="],
"bun-types": ["bun-types@1.3.11", "", { "dependencies": { "@types/node": "*" } }, "sha512-1KGPpoxQWl9f6wcZh57LvrPIInQMn2TQ7jsgxqpRzg+l0QPOFvJVH7HmvHo/AiPgwXy+/Thf6Ov3EdVn1vOabg=="],
@@ -88,8 +66,6 @@
"call-bound": ["call-bound@1.0.4", "", { "dependencies": { "call-bind-apply-helpers": "^1.0.2", "get-intrinsic": "^1.3.0" } }, "sha512-+ys997U96po4Kx/ABpBCqhA9EuxJaQWDQg7295H4hBphv3IZg0boBKuwYpt4YXp6MZ5AmZQnU/tyMTlRpaSejg=="],
"combined-stream": ["combined-stream@1.0.8", "", { "dependencies": { "delayed-stream": "~1.0.0" } }, "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg=="],
"commander": ["commander@14.0.3", "", {}, "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw=="],
"content-disposition": ["content-disposition@1.0.1", "", {}, "sha512-oIXISMynqSqm241k6kcQ5UwttDILMK4BiurCfGEREw6+X9jkkpEe5T9FZaApyLGGOnFuyMWZpdolTXMtvEJ08Q=="],
@@ -106,8 +82,6 @@
"debug": ["debug@4.4.3", "", { "dependencies": { "ms": "^2.1.3" } }, "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA=="],
"delayed-stream": ["delayed-stream@1.0.0", "", {}, "sha512-ZySD7Nf91aLB0RxL4KGrKHBXl7Eds1DAmEdcoVawXnLD7SDhpNgtuII2aAkg7a7QS41jxPSZ17p4VdGnMHk3MQ=="],
"depd": ["depd@2.0.0", "", {}, "sha512-g7nH6P6dyDioJogAAGprGpCtVImJhpPk/roCzdb3fIh61/s/nPsfR6onyMwkCAR/OlC3yBC0lESvUoQEAssIrw=="],
"dunder-proto": ["dunder-proto@1.0.1", "", { "dependencies": { "call-bind-apply-helpers": "^1.0.1", "es-errors": "^1.3.0", "gopd": "^1.2.0" } }, "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A=="],
@@ -122,14 +96,10 @@
"es-object-atoms": ["es-object-atoms@1.1.1", "", { "dependencies": { "es-errors": "^1.3.0" } }, "sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA=="],
"es-set-tostringtag": ["es-set-tostringtag@2.1.0", "", { "dependencies": { "es-errors": "^1.3.0", "get-intrinsic": "^1.2.6", "has-tostringtag": "^1.0.2", "hasown": "^2.0.2" } }, "sha512-j6vWzfrGVfyXxge+O0x5sh6cvxAog0a/4Rdd2K36zCMV5eJ+/+tOAngRO8cODMNWbVRdVlmGZQL2YS3yR8bIUA=="],
"escape-html": ["escape-html@1.0.3", "", {}, "sha512-NiSupZ4OeuGwr68lGIeym/ksIZMJodUGOSCZ/FSnTxcrekbvqrgdUxlJOMpijaKZVjAJrWrGs/6Jy8OMuyj9ow=="],
"etag": ["etag@1.8.1", "", {}, "sha512-aIL5Fx7mawVa300al2BnEE4iNvo1qETxLrPI/o05L7z6go7fCw1J6EQmbK4FmJ2AS7kgVF/KEZWufBfdClMcPg=="],
"event-target-shim": ["event-target-shim@5.0.1", "", {}, "sha512-i/2XbnSz/uxRCU6+NdVJgKWDTM427+MqYbkQzD321DuCQJUqOuJKIA0IM2+W2xtYHdKOmZ4dR6fExsd4SXL+WQ=="],
"eventsource": ["eventsource@3.0.7", "", { "dependencies": { "eventsource-parser": "^3.0.1" } }, "sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA=="],
"eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="],
@@ -140,16 +110,12 @@
"fast-deep-equal": ["fast-deep-equal@3.1.3", "", {}, "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q=="],
"fast-sha256": ["fast-sha256@1.3.0", "", {}, "sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ=="],
"fast-uri": ["fast-uri@3.1.0", "", {}, "sha512-iPeeDKJSWf4IEOasVVrknXpaBV0IApz/gp7S2bb7Z4Lljbl2MGJRqInZiUrQwV16cpzw/D3S5j5Julj/gT52AA=="],
"finalhandler": ["finalhandler@2.1.1", "", { "dependencies": { "debug": "^4.4.0", "encodeurl": "^2.0.0", "escape-html": "^1.0.3", "on-finished": "^2.4.1", "parseurl": "^1.3.3", "statuses": "^2.0.1" } }, "sha512-S8KoZgRZN+a5rNwqTxlZZePjT/4cnm0ROV70LedRHZ0p8u9fRID0hJUZQpkKLzro8LfmC8sx23bY6tVNxv8pQA=="],
"form-data": ["form-data@4.0.5", "", { "dependencies": { "asynckit": "^0.4.0", "combined-stream": "^1.0.8", "es-set-tostringtag": "^2.1.0", "hasown": "^2.0.2", "mime-types": "^2.1.12" } }, "sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w=="],
"form-data-encoder": ["form-data-encoder@1.7.2", "", {}, "sha512-qfqtYan3rxrnCk1VYaA4H+Ms9xdpPqvLZa6xmMgFvhO32x7/3J/ExcTd6qpxM0vH2GdMI+poehyBZvqfMTto8A=="],
"formdata-node": ["formdata-node@4.4.1", "", { "dependencies": { "node-domexception": "1.0.0", "web-streams-polyfill": "4.0.0-beta.3" } }, "sha512-0iirZp3uVDjVGt9p49aTaqjk84TrglENEDuqfdlZQ1roC9CWlPk6Avf8EEnZNcAqPonwkG35x4n3ww/1THYAeQ=="],
"forwarded": ["forwarded@0.2.0", "", {}, "sha512-buRG0fpBtRHSTCOASe6hD258tEubFoRLb4ZNA6NxMVHNw2gOcwHo9wyablzMzOA5z9xA9L1KNjk/Nt6MT9aYow=="],
"fresh": ["fresh@2.0.0", "", {}, "sha512-Rx/WycZ60HOaqLKAi6cHRKKI7zxWbJ31MhntmtwMoaTeF7XFH9hhBp8vITaMidfljRQ6eYWCKkaTK+ykVJHP2A=="],
@@ -164,16 +130,12 @@
"has-symbols": ["has-symbols@1.1.0", "", {}, "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ=="],
"has-tostringtag": ["has-tostringtag@1.0.2", "", { "dependencies": { "has-symbols": "^1.0.3" } }, "sha512-NqADB8VjPFLM2V0VvHUewwwsw0ZWBaIdgo+ieHtK3hasLz4qeCRjYcqfB6AQrBggRKppKF8L52/VqdVsO47Dlw=="],
"hasown": ["hasown@2.0.2", "", { "dependencies": { "function-bind": "^1.1.2" } }, "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ=="],
"hono": ["hono@4.12.9", "", {}, "sha512-wy3T8Zm2bsEvxKZM5w21VdHDDcwVS1yUFFY6i8UobSsKfFceT7TOwhbhfKsDyx7tYQlmRM5FLpIuYvNFyjctiA=="],
"http-errors": ["http-errors@2.0.1", "", { "dependencies": { "depd": "~2.0.0", "inherits": "~2.0.4", "setprototypeof": "~1.2.0", "statuses": "~2.0.2", "toidentifier": "~1.0.1" } }, "sha512-4FbRdAX+bSdmo4AUFuS0WNiPz8NgFt+r8ThgNWmlrjQjt1Q7ZR9+zTlce2859x4KSXrwIsaeTqDoKQmtP8pLmQ=="],
"humanize-ms": ["humanize-ms@1.2.1", "", { "dependencies": { "ms": "^2.0.0" } }, "sha512-Fl70vYtsAFb/C06PTS9dZBo7ihau+Tu/DNCk/OyHhea07S+aeMWpFFkUaXRa8fI+ScZbEI8dfSxwY7gxZ9SAVQ=="],
"iconv-lite": ["iconv-lite@0.7.2", "", { "dependencies": { "safer-buffer": ">= 2.1.2 < 3.0.0" } }, "sha512-im9DjEDQ55s9fL4EYzOAv0yMqmMBSZp6G0VvFyTMPKWxiSBHUj9NW/qqLmXUwXrrM7AvqSlTCfvqRb0cM8yYqw=="],
"inherits": ["inherits@2.0.4", "", {}, "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ=="],
@@ -208,10 +170,6 @@
"negotiator": ["negotiator@1.0.0", "", {}, "sha512-8Ofs/AUQh8MaEcrlq5xOX0CQ9ypTF5dl78mjlMNfOK08fzpgTHQRQPBxcPlEtIw0yRpws+Zo/3r+5WRby7u3Gg=="],
"node-domexception": ["node-domexception@1.0.0", "", {}, "sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ=="],
"node-fetch": ["node-fetch@2.7.0", "", { "dependencies": { "whatwg-url": "^5.0.0" }, "peerDependencies": { "encoding": "^0.1.0" }, "optionalPeers": ["encoding"] }, "sha512-c4FRfUm/dbcWZ7U+1Wq0AwCyFL+3nt2bEw05wfxSz+DWpWsitgmSgYmy2dQdWyKC1694ELPqMs/YzUSNozLt8A=="],
"object-assign": ["object-assign@4.1.1", "", {}, "sha512-rJgTQnkUnH1sFw8yT6VSU3zD3sWmu6sZhIseY8VX+GRu3P6F7Fu+JNDoXfklElbLJSnc3FUQHVe4cU5hj+BcUg=="],
"object-inspect": ["object-inspect@1.13.4", "", {}, "sha512-W67iLl4J2EXEGTbfeHCffrjDfitvLANg0UlX3wFUUSTx92KXRFegMHUVgSqE+wvhAbi4WqjGg9czysTV2Epbew=="],
@@ -262,30 +220,24 @@
"side-channel-weakmap": ["side-channel-weakmap@1.0.2", "", { "dependencies": { "call-bound": "^1.0.2", "es-errors": "^1.3.0", "get-intrinsic": "^1.2.5", "object-inspect": "^1.13.3", "side-channel-map": "^1.0.1" } }, "sha512-WPS/HvHQTYnHisLo9McqBHOJk2FkHO/tlpvldyrnem4aeQp4hai3gythswg6p01oSoTl58rcpiFAjF2br2Ak2A=="],
"standardwebhooks": ["standardwebhooks@1.1.1", "", { "dependencies": { "@stablelib/base64": "^1.0.0", "fast-sha256": "^1.3.0" } }, "sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ=="],
"statuses": ["statuses@2.0.2", "", {}, "sha512-DvEy55V3DB7uknRo+4iOGT5fP1slR8wQohVdknigZPMpMstaKJQWhwiYBACJE3Ul2pTnATihhBYnRhZQHGBiRw=="],
"toidentifier": ["toidentifier@1.0.1", "", {}, "sha512-o5sSPKEkg/DIQNmH43V0/uerLrpzVedkUh8tGNvaeXpfpuwjKenlSox/2O/BTlZUtEe+JG7s5YhEz608PlAHRA=="],
"tr46": ["tr46@0.0.3", "", {}, "sha512-N3WMsuqV66lT30CrXNbEjx4GEwlow3v6rr4mCcv6prnfwhS01rkgyFdjPNBYd9br7LpXV1+Emh01fHnq2Gdgrw=="],
"ts-algebra": ["ts-algebra@2.0.0", "", {}, "sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw=="],
"type-is": ["type-is@2.0.1", "", { "dependencies": { "content-type": "^1.0.5", "media-typer": "^1.1.0", "mime-types": "^3.0.0" } }, "sha512-OZs6gsjF4vMp32qrCbiVSkrFmXtG/AZhY3t0iAMrMBiAZyV9oALtXO8hsrHbMXF9x6L3grlFuwW2oAz7cav+Gw=="],
"typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="],
"undici-types": ["undici-types@5.26.5", "", {}, "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA=="],
"undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
"unpipe": ["unpipe@1.0.0", "", {}, "sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ=="],
"vary": ["vary@1.1.2", "", {}, "sha512-BNGbWLfd0eUPabhkXUVm0j8uuvREyTh5ovRa/dyow/BqAbZJyC+5fU+IzQOzmAKzYqYRAISoRhdQr3eIZ/PXqg=="],
"web-streams-polyfill": ["web-streams-polyfill@4.0.0-beta.3", "", {}, "sha512-QW95TCTaHmsYfHDybGMwO5IJIM93I/6vTRk+daHTWFPhwh+C8Cg7j7XyKrwrj8Ib6vYXe0ocYNrmzY4xAAN6ug=="],
"webidl-conversions": ["webidl-conversions@3.0.1", "", {}, "sha512-2JAn3z8AR6rjK8Sm8orRC0h/bcl/DqL7tRPdGZ4I1CjdF+EaMLmYxBHyXuKL849eucPFhvBoxMsflfOb8kxaeQ=="],
"whatwg-url": ["whatwg-url@5.0.0", "", { "dependencies": { "tr46": "~0.0.3", "webidl-conversions": "^3.0.0" } }, "sha512-saE57nupxk6v3HY35+jzBwYa0rKSy0XR8JSxZPwgLr7ys0IBzhGviA1/TUGJLmSVqs8pb9AnvICXEuOHLprYTw=="],
"which": ["which@2.0.2", "", { "dependencies": { "isexe": "^2.0.0" }, "bin": { "node-which": "./bin/node-which" } }, "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA=="],
"wrappy": ["wrappy@1.0.2", "", {}, "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ=="],
@@ -295,19 +247,5 @@
"zod": ["zod@4.3.6", "", {}, "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg=="],
"zod-to-json-schema": ["zod-to-json-schema@3.25.2", "", { "peerDependencies": { "zod": "^3.25.28 || ^4" } }, "sha512-O/PgfnpT1xKSDeQYSCfRI5Gy3hPf91mKVDuYLUHZJMiDFptvP41MSnWofm8dnCm0256ZNfZIM7DSzuSMAFnjHA=="],
"@anthropic-ai/claude-agent-sdk/@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.74.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-srbJV7JKsc5cQ6eVuFzjZO7UR3xEPJqPamHFIe29bs38Ij2IripoAhC0S5NslNbaFUYqBKypmmpzMTpqfHEUDw=="],
"@types/node-fetch/@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="],
"bun-types/@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="],
"form-data/mime-types": ["mime-types@2.1.35", "", { "dependencies": { "mime-db": "1.52.0" } }, "sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw=="],
"@types/node-fetch/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
"bun-types/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="],
"form-data/mime-types/mime-db": ["mime-db@1.52.0", "", {}, "sha512-sPU4uV7dYlvtWJxwwxHD0PuihVNiE7TyAbQ5SWxDCB9mUYvOgroQOwYQQOKPJ8CIbE+1ETVlOoK1UC2nU3gYvg=="],
}
}
+18
View File
@@ -108,6 +108,10 @@ async function main() {
"--record",
"append a compact summary line to ai_evals/history/<mode>.jsonl",
)
.option(
"--reasoning <effort>",
"reasoning effort for frontend modes (e.g. off, low, medium, high, max); default: the product's",
)
.option(
"--backend-validation <mode>",
`backend smoke validation (${BACKEND_VALIDATION_MODES.join(", ")})`,
@@ -126,6 +130,7 @@ async function main() {
executionOnly?: boolean;
record?: boolean;
backendValidation?: string;
reasoning?: string;
},
) => {
await handleRun({
@@ -140,6 +145,7 @@ async function main() {
executionOnly: options.executionOnly ?? false,
record: options.record ?? false,
backendValidation: options.backendValidation,
reasoning: options.reasoning,
});
},
);
@@ -190,6 +196,7 @@ async function handleRun(input: {
executionOnly: boolean;
record: boolean;
backendValidation?: string;
reasoning?: string;
}) {
if (input.record && input.caseIds.length > 0) {
throw new Error(
@@ -217,6 +224,13 @@ async function handleRun(input: {
"--backend-validation currently supports only flow and script modes",
);
}
if (input.reasoning) {
if (input.mode === "cli") {
throw new Error("--reasoning only applies to frontend modes");
}
// The frontend runtime runs in a child process, which inherits it.
process.env.WMILL_AI_EVAL_REASONING = input.reasoning;
}
if (input.mode !== "cli") {
await assertWindmillBackendReachable(resolveWindmillBackendSettings());
}
@@ -257,6 +271,10 @@ async function handleRun(input: {
backendValidation,
});
if (input.reasoning && result.runModel) {
result.runModel = `${result.runModel}@${input.reasoning}`;
}
const resolvedOutputPath =
models.length === 1
? resolveRunOutputPath(input.mode, input.outputPath)
+6 -8
View File
@@ -1,7 +1,7 @@
import Anthropic from "@anthropic-ai/sdk";
import type { EvalMode, JudgeResult } from "./types";
export const DEFAULT_JUDGE_MODEL = "claude-sonnet-4-6";
export const DEFAULT_JUDGE_MODEL = "claude-sonnet-5-5";
const JUDGE_TOOL_NAME = "submit_judgement";
@@ -29,6 +29,7 @@ export async function judgeOutput(input: {
const system = [
"You evaluate benchmark outputs for Windmill AI generation.",
`Always answer by calling the ${JUDGE_TOOL_NAME} tool.`,
"Deterministic checks already run separately. Focus on whether the final output satisfies the user request.",
"If expected state is provided, treat it as a valid example and reward semantically equivalent outputs.",
"If a checklist is provided, treat it as the explicit acceptance criteria for this case.",
@@ -69,8 +70,8 @@ export async function judgeOutput(input: {
try {
const response = await client.messages.create({
model,
max_tokens: 1024,
temperature: 0,
// The judge thinks by default, and thinking shares this budget with the verdict.
max_tokens: 16000,
system,
messages: [{ role: "user", content: user }],
tools: [
@@ -93,11 +94,8 @@ export async function judgeOutput(input: {
},
},
],
tool_choice: {
type: "tool",
name: JUDGE_TOOL_NAME,
disable_parallel_tool_use: true,
},
// Current models refuse a forced tool_choice ("tool"/"any").
tool_choice: { type: "auto", disable_parallel_tool_use: true },
});
const toolUseBlock = response.content.find(
+71
View File
@@ -78,6 +78,32 @@ export const EVAL_MODELS: EvalModelSpec[] = [
model: "opus",
},
},
{
id: "sonnet-5.5",
label: "Claude Sonnet 5.5",
aliases: ["sonnet-5.5", "claude-sonnet-5.5", "claude-sonnet-5-5"],
frontend: {
provider: "anthropic",
model: "claude-sonnet-5-5",
},
cli: {
provider: "anthropic",
model: "claude-sonnet-5-5",
},
},
{
id: "opus-5.5",
label: "Claude Opus 5.5",
aliases: ["opus-5.5", "claude-opus-5.5", "claude-opus-5-5"],
frontend: {
provider: "anthropic",
model: "claude-opus-5-5",
},
cli: {
provider: "anthropic",
model: "claude-opus-5-5",
},
},
{
id: "4o",
label: "GPT-4o",
@@ -96,6 +122,51 @@ export const EVAL_MODELS: EvalModelSpec[] = [
model: "gpt-5.5",
},
},
{
id: "gpt-5.6-sol",
label: "GPT-5.6 Sol",
aliases: ["gpt-5.6-sol"],
frontend: {
provider: "openai",
model: "gpt-5.6-sol",
},
},
{
id: "gpt-6-astra",
label: "GPT-6 Astra",
aliases: ["gpt-6-astra", "gpt-6"],
frontend: {
provider: "openai",
model: "gpt-6-astra",
},
},
{
id: "gpt-6-sol",
label: "GPT-6 Sol",
aliases: ["gpt-6-sol"],
frontend: {
provider: "openai",
model: "gpt-6-sol",
},
},
{
id: "gpt-6-luna",
label: "GPT-6 Luna",
aliases: ["gpt-6-luna"],
frontend: {
provider: "openai",
model: "gpt-6-luna",
},
},
{
id: "gemini-3.8-flash",
label: "Gemini 3.8 Flash",
aliases: ["gemini-3.8-flash"],
frontend: {
provider: "googleai",
model: "gemini-3.8-flash",
},
},
{
id: "gemini-3-flash-preview",
label: "Gemini 3 Flash Preview",
+2 -2
View File
@@ -8,8 +8,8 @@
"test:frontend-graph": "cd ../frontend && node_modules/.bin/vitest run --project server --config ../ai_evals/adapters/frontend/vitest.unit.config.ts"
},
"dependencies": {
"@anthropic-ai/claude-agent-sdk": "^0.2.25",
"@anthropic-ai/sdk": "^0.39.0",
"@anthropic-ai/claude-agent-sdk": "^0.3.284",
"@anthropic-ai/sdk": "^0.129.0",
"commander": "^14.0.3",
"openai": "^6.9.1",
"yaml": "^2.8.3"
+164
View File
@@ -14,6 +14,130 @@ use std::time::{Duration, Instant};
/// its `thinking` param, Gemini to a zero budget or the model's floor).
pub(crate) const REASONING_OFF_SENTINEL: &str = "none";
/// The effort to send for a model, dropping the off sentinel on a model that rejects every
/// disable: the model then reasons at its default instead of failing the request. The UI
/// never offers off on these, so this guards an agent step saved before it stopped, or an
/// effort passed in as a flow input.
pub fn effective_reasoning_effort<'a>(model: &str, effort: Option<&'a str>) -> Option<&'a str> {
match effort {
Some(effort)
if effort == REASONING_OFF_SENTINEL
&& reasoning_rule(model).is_some_and(|rule| !rule.can_disable) =>
{
None
}
effort => effort,
}
}
/// Whether a request with function tools must send the off sentinel on Chat Completions:
/// the model refuses tools there while it reasons, and does accept being turned off.
pub(crate) fn completions_tools_need_reasoning_off(model: &str) -> bool {
reasoning_rule(model).is_some_and(|rule| rule.completions_tools_need_off && rule.can_disable)
}
/// What the backend needs to know about a model's reasoning: the rows of `REASONING_RULES`
/// in the frontend's `reasoningRegistry.ts`, cut down to the two facts the wire needs.
/// `reasoningParity.json` next to that file is checked by both sides' tests.
struct ReasoningRule {
matches: fn(&str) -> bool,
/// False when the provider rejects every disable, so the off sentinel must not be sent.
can_disable: bool,
/// Chat Completions refuses function tools while the model reasons, even with the
/// effort omitted (live-verified); the Responses API has no such limit.
completions_tools_need_off: bool,
}
/// Matched in order against the lowercased model id, first match wins. A model no row
/// matches keeps whatever effort it was given.
const REASONING_RULES: &[ReasoningRule] = &[
// Live-verified: Fable, Mythos and the 5.x point releases reject `thinking: disabled`.
ReasoningRule {
matches: |m| m.contains("claude-fable") || m.contains("claude-mythos"),
can_disable: false,
completions_tools_need_off: false,
},
ReasoningRule {
matches: is_claude_5_point_release,
can_disable: false,
completions_tools_need_off: false,
},
// Live-verified: astra takes low..max only, where sol and luna also take `none`.
ReasoningRule {
matches: |m| base_id(m).starts_with("gpt-6-astra"),
can_disable: false,
completions_tools_need_off: true,
},
ReasoningRule {
matches: |m| gpt_version(m).is_some_and(|(major, _)| major >= 6),
can_disable: true,
completions_tools_need_off: true,
},
ReasoningRule {
matches: |m| matches!(gpt_version(m), Some((5, Some(minor))) if minor >= 5),
can_disable: true,
completions_tools_need_off: true,
},
ReasoningRule {
matches: |m| matches!(gpt_version(m), Some((5, Some(_)))),
can_disable: true,
completions_tools_need_off: false,
},
// gpt-5 and the o-series reject `none`.
ReasoningRule {
matches: |m| matches!(gpt_version(m), Some((5, None))),
can_disable: false,
completions_tools_need_off: false,
},
ReasoningRule {
matches: |m| {
let base = base_id(m);
base.starts_with('o') && base[1..].starts_with(|c: char| c.is_ascii_digit())
},
can_disable: false,
completions_tools_need_off: false,
},
];
fn reasoning_rule(model: &str) -> Option<&'static ReasoningRule> {
let model = model.to_lowercase();
REASONING_RULES.iter().find(|rule| (rule.matches)(&model))
}
/// The id after a gateway's `vendor/` and before a `:variant`.
fn base_id(model: &str) -> &str {
let last = model.rsplit('/').next().unwrap_or(model);
last.split(':').next().unwrap_or(last)
}
/// `(major, minor)` of a `gpt-` id. The major is one digit then a separator or the end,
/// since Azure names gpt-3.5 `gpt-35-turbo`.
fn gpt_version(model: &str) -> Option<(u32, Option<u32>)> {
let rest = base_id(model).strip_prefix("gpt-")?;
let mut chars = rest.chars();
let major = chars.next()?.to_digit(10)?;
match chars.next() {
None | Some('-') => Some((major, None)),
Some('.') => {
let minor: String = chars.take_while(char::is_ascii_digit).collect();
Some((major, minor.parse().ok()))
}
_ => None,
}
}
/// Sonnet or Opus 5.x with x >= 1. The version match stops at one digit so a dated id
/// (`claude-sonnet-5-20260101`) stays Sonnet 5.
fn is_claude_5_point_release(model: &str) -> bool {
let model = model.replace('.', "-");
["claude-opus-5-", "claude-sonnet-5-"].iter().any(|prefix| {
model.split(prefix).skip(1).any(|rest| {
let mut chars = rest.chars();
matches!(chars.next(), Some('1'..='9')) && !matches!(chars.next(), Some('0'..='9'))
})
})
}
/// Whether a Claude model removed the sampling params (`temperature`, `top_p`,
/// `top_k`). On these, any value is a hard 400 — `temperature is deprecated for
/// this model` — whatever the thinking mode, so the param has to be dropped on
@@ -119,6 +243,46 @@ pub fn remember_chat_completions_only(base_url: &str, model: &str) {
CHAT_COMPLETIONS_ONLY.insert((base_url.to_string(), model.to_string()), Instant::now());
}
#[cfg(test)]
mod reasoning_rule_tests {
use super::*;
/// The frontend registry's test reads the same file. These rules see the model id
/// alone, so it holds only rows whose answer doesn't depend on the provider.
#[test]
fn agrees_with_the_frontend_registry() {
let rows: Vec<serde_json::Value> = serde_json::from_str(include_str!(
"../../../../frontend/src/lib/components/copilot/reasoningParity.json"
))
.unwrap();
for row in rows {
let model = row["model"].as_str().unwrap();
let can_disable = row["canDisable"].as_bool().unwrap();
let tools_need_off = row["completionsToolsNeedOff"].as_bool().unwrap();
let sent = effective_reasoning_effort(model, Some(REASONING_OFF_SENTINEL));
assert_eq!(sent.is_some(), can_disable, "{model}");
assert_eq!(
effective_reasoning_effort(model, Some("low")),
Some("low"),
"{model}"
);
assert_eq!(
completions_tools_need_reasoning_off(model),
tools_need_off && can_disable,
"{model}"
);
}
}
#[test]
fn reads_ids_the_frontend_resolves_first() {
// Azure's gpt-3.5 is not major 35, and a gateway prefix is not part of the id.
assert_eq!(gpt_version("gpt-35-turbo"), None);
assert!(completions_tools_need_reasoning_off("openai/gpt-6-sol"));
assert_eq!(effective_reasoning_effort("openai/o3", Some("none")), None);
}
}
#[cfg(test)]
mod chat_completions_only_tests {
use super::*;
@@ -1,3 +1,4 @@
use super::{completions_tools_need_reasoning_off, REASONING_OFF_SENTINEL};
use crate::{
ai_providers::AIProvider,
image_handler::prepare_messages_for_api,
@@ -143,6 +144,15 @@ impl OtherQueryBuilder {
let (reasoning_effort, thinking, temperature) =
provider_reasoning_fields(&self.provider_kind, args.reasoning_effort, args.temperature);
// With tools, gpt-5.5+ only run here with reasoning off: turn it off where the model
// can, rather than failing every turn.
let reasoning_effort = if args.tools.is_some_and(|tools| !tools.is_empty())
&& completions_tools_need_reasoning_off(args.model)
{
Some(REASONING_OFF_SENTINEL)
} else {
reasoning_effort
};
// Build request with stream_options for usage tracking
let request_with_usage = OpenAICompletionRequest {
+4 -1
View File
@@ -294,7 +294,10 @@ impl ProviderWithResource {
/// The reasoning effort to thread to the provider, treating an empty string
/// (e.g. a cleared flow input) as unset.
pub fn get_reasoning_effort(&self) -> Option<&str> {
self.reasoning_effort.as_deref().filter(|s| !s.is_empty())
crate::providers::effective_reasoning_effort(
&self.model,
self.reasoning_effort.as_deref().filter(|s| !s.is_empty()),
)
}
pub async fn get_base_url(&self, db: &DB) -> Result<String, Error> {
+20 -2
View File
@@ -22,6 +22,8 @@ import {
} from './modelConfig'
import {
applyReasoningToConfig,
completionsRejectsToolsWithReasoning,
explicitOffToken,
requestsReasoning,
stripLegacyThinkingSuffix,
type ReasoningEffort
@@ -69,6 +71,9 @@ interface AIProviderDetails {
// the frontier model. The gpt-5 family is deprecated (retires 2026-12-11) but
// still served, so it stays in the list below the 5.6 models.
const OPENAI_MODELS = [
'gpt-6-sol',
'gpt-6-astra',
'gpt-6-luna',
'gpt-5.6-terra',
'gpt-5.6-sol',
'gpt-5.6-luna',
@@ -87,7 +92,14 @@ export const AI_PROVIDERS: Record<AIProvider, AIProviderDetails> = {
},
anthropic: {
label: 'Anthropic',
defaultModels: ['claude-sonnet-5', 'claude-opus-5', 'claude-opus-4-8', 'claude-haiku-4-5']
defaultModels: [
'claude-sonnet-5-5',
'claude-opus-5-5',
'claude-sonnet-5',
'claude-opus-5',
'claude-opus-4-8',
'claude-haiku-4-5'
]
},
googleai: {
label: 'Google AI',
@@ -1160,6 +1172,12 @@ export async function getCompletion(
// Use Completions API for other providers
const client = options?.openaiClient ?? workspaceAIClients.getOpenaiClient()
// gpt-5.5+ refuse function tools here unless reasoning is off, so the Responses API
// fallback turns it off where the model can rather than failing the turn.
const reasoningEffort =
tools?.length && completionsRejectsToolsWithReasoning(provider, modelProvider.model)
? (explicitOffToken(provider, modelProvider.model) ?? options?.reasoningEffort)
: options?.reasoningEffort
const completionConfig = applyReasoningToConfig(
config.stream && STREAM_USAGE_PROVIDERS.has(provider)
? {
@@ -1175,7 +1193,7 @@ export async function getCompletion(
}
: config,
provider === 'deepseek' ? 'deepseek' : provider === 'mistral' ? 'mistral' : 'completions',
options?.reasoningEffort
reasoningEffort
)
const completion = client.chat.completions.create(completionConfig, {
signal: abortController.signal,
@@ -21,6 +21,13 @@ describe('workspace context window overrides', () => {
expect(getConfiguredModelContextWindow('openai', 'qwen-local', overrides)).toBeUndefined()
expect(getEffectiveModelContextWindow('openai', 'qwen-local', overrides)).toBe(128_000)
})
it('knows the gpt-6 and Claude 5.5 windows, so they do not compact at the assumed one', () => {
expect(getConfiguredModelContextWindow('openai', 'gpt-6-sol', undefined)).toBe(1_050_000)
expect(getConfiguredModelContextWindow('anthropic', 'claude-sonnet-5-5', undefined)).toBe(
1_000_000
)
})
})
describe('usesAnthropicMessagesApi', () => {
@@ -58,7 +58,7 @@ export function usesOpenRouterPromptCaching(provider: AIProvider, model: string)
// so it does not catch unrelated ids like Mistral's "open-mistral-*" or "optimus-*".
export function requiresMaxCompletionTokens(model: string) {
const baseModel = parseModelId(model).base
return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel)
return Number(/^gpt-(\d)(?:[.-]|$)/.exec(baseModel)?.[1] ?? 0) >= 5 || /^o\d/.test(baseModel)
}
// Context windows of the models we know, most specific entry first — the first
@@ -87,6 +87,7 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [
['claude', 200_000],
// OpenAI — gpt-5 covers the base family (-mini / -nano) and the 5.1/5.2
// revisions, all 400K; 5.4/5.5 moved to 1M and 5.6 to 1.05M
['gpt-6', 1_050_000],
['gpt-5.6', 1_050_000],
['gpt-5.5', 1_000_000],
['gpt-5.4', 1_000_000],
@@ -140,6 +141,7 @@ const MODEL_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [
['claude-fable', 64_000],
['claude-mythos', 64_000],
// OpenAI
['gpt-6', 128_000],
['gpt-5', 128_000],
['gpt-4.1', 32_768],
['gpt-4o', 16_384],
@@ -35,6 +35,16 @@ describe('resolveModelPrice', () => {
expect(resolveModelPrice('googleai', 'gemini-3.7-flash', undefined)).toBeUndefined()
})
it('prices Claude 5.5 flat and leaves the tiered gpt-6 unpriced', () => {
expect(resolveModelPrice('anthropic', 'claude-opus-5-5', undefined)?.price).toMatchObject({
input: 4,
output: 20,
cacheRead: 0.2
})
expect(resolveModelPrice('anthropic', 'claude-sonnet-5-5', undefined)?.price.input).toBe(2)
expect(resolveModelPrice('openai', 'gpt-6-sol', undefined)).toBeUndefined()
})
it('reports an unknown model as unpriced rather than guessing', () => {
expect(resolveModelPrice('customai', 'some-in-house-model', undefined)).toBeUndefined()
})
@@ -69,6 +69,8 @@ const MODEL_PRICES: [name: string, price: PriceEntry | null][] = [
// fallback sits below the explicit entries rather than covering them.
['claude-fable-5', { input: 10, output: 50 }],
['claude-mythos-5', { input: 10, output: 50 }],
['claude-opus-5-5', { input: 4, output: 20, cacheRead: 0.2 }],
['claude-sonnet-5-5', { input: 2, output: 10, cacheRead: 0.2 }],
['claude-opus-5', { input: 5, output: 25 }],
['claude-opus-4-8', { input: 5, output: 25 }],
['claude-opus-4-7', { input: 5, output: 25 }],
@@ -98,6 +100,9 @@ const MODEL_PRICES: [name: string, price: PriceEntry | null][] = [
// Revisions past gpt-5 are priced separately by OpenAI and are not tracked here.
// The matcher's revision guard already keeps them off the family rate; these
// entries stay so a revision the guard admits still resolves to no rate.
// gpt-6 bills the whole request at a higher rate above 272K input tokens, which a
// per-model rate cannot express.
['gpt-6', null],
['gpt-5.6', null],
['gpt-5.5', null],
['gpt-5.4', null],
@@ -0,0 +1,84 @@
[
{
"provider": "anthropic",
"model": "claude-sonnet-5-5",
"canDisable": false,
"completionsToolsNeedOff": false
},
{
"provider": "anthropic",
"model": "claude-opus-5-5",
"canDisable": false,
"completionsToolsNeedOff": false
},
{
"provider": "anthropic",
"model": "claude-fable-5",
"canDisable": false,
"completionsToolsNeedOff": false
},
{
"provider": "aws_bedrock",
"model": "global.anthropic.claude-opus-5-5-v1:0",
"canDisable": false,
"completionsToolsNeedOff": false
},
{
"provider": "anthropic",
"model": "claude-sonnet-5",
"canDisable": true,
"completionsToolsNeedOff": false
},
{
"provider": "anthropic",
"model": "claude-opus-5-20260101",
"canDisable": true,
"completionsToolsNeedOff": false
},
{
"provider": "anthropic",
"model": "claude-opus-4-8",
"canDisable": true,
"completionsToolsNeedOff": false
},
{
"provider": "openai",
"model": "gpt-6-astra",
"canDisable": false,
"completionsToolsNeedOff": true
},
{
"provider": "openai",
"model": "gpt-6-sol",
"canDisable": true,
"completionsToolsNeedOff": true
},
{
"provider": "azure_openai",
"model": "gpt-6-luna",
"canDisable": true,
"completionsToolsNeedOff": true
},
{
"provider": "openai",
"model": "gpt-5.6-sol",
"canDisable": true,
"completionsToolsNeedOff": true
},
{ "provider": "openai", "model": "gpt-5.5", "canDisable": true, "completionsToolsNeedOff": true },
{ "provider": "openai", "model": "gpt-5.7", "canDisable": true, "completionsToolsNeedOff": true },
{
"provider": "openai",
"model": "gpt-5.1",
"canDisable": true,
"completionsToolsNeedOff": false
},
{ "provider": "openai", "model": "gpt-5", "canDisable": false, "completionsToolsNeedOff": false },
{
"provider": "openai",
"model": "gpt-5-mini",
"canDisable": false,
"completionsToolsNeedOff": false
},
{ "provider": "openai", "model": "o3", "canDisable": false, "completionsToolsNeedOff": false }
]
@@ -1,6 +1,8 @@
import { describe, expect, it } from 'vitest'
import {
applyReasoningToConfig,
completionsRejectsToolsWithReasoning,
explicitOffToken,
getReasoningCapability,
REASONING_OFF,
resolveEffectiveReasoning,
@@ -8,6 +10,8 @@ import {
stripLegacyThinkingSuffix,
supportsReasoning
} from './reasoningRegistry'
import type { AIProvider } from '$lib/gen'
import parity from './reasoningParity.json'
describe('stripLegacyThinkingSuffix', () => {
it('removes the deprecated /thinking suffix', () => {
@@ -261,6 +265,42 @@ describe('supportsReasoning (static registry)', () => {
expect(getReasoningCapability('openrouter', 'x-ai/grok-4').canDisable).toBe(false)
expect(getReasoningCapability('openrouter', 'deepseek/deepseek-r1').canDisable).toBe(false)
})
it('never sends a disable the 5.5 point releases and gpt-6-astra reject', () => {
// Live-verified: Claude 5.5 rejects `thinking: disabled`, gpt-6-astra rejects `none`.
for (const [provider, model] of [
['anthropic', 'claude-sonnet-5-5'],
['anthropic', 'claude-opus-5-5'],
['aws_bedrock', 'global.anthropic.claude-opus-5-5-v1:0'],
['openai', 'gpt-6-astra']
] as const) {
expect(getReasoningCapability(provider, model).canDisable, model).toBe(false)
expect(explicitOffToken(provider, model), model).toBeUndefined()
}
expect(getReasoningCapability('openrouter', 'anthropic/claude-sonnet-5.5').canDisable).toBe(
false
)
expect(explicitOffToken('openrouter', 'anthropic/claude-sonnet-5.5')).toBeUndefined()
// A dated Claude 5 id is not a point release.
expect(explicitOffToken('anthropic', 'claude-sonnet-5-20260101')).toBe('none')
expect(explicitOffToken('openai', 'gpt-6-sol')).toBe('none')
expect(getReasoningCapability('openai', 'gpt-6-luna')).toMatchObject({
supported: true,
canDisable: true,
levels: ['low', 'medium', 'high', 'xhigh', 'max']
})
})
it("reads Azure's gpt-35-turbo as gpt-3.5, not a gpt-5+ reasoning model", () => {
expect(supportsReasoning('azure_openai', 'gpt-35-turbo')).toBe(false)
expect(supportsReasoning('openai', 'gpt-35-turbo-16k')).toBe(false)
})
it('finds the models that refuse function tools with reasoning on Chat Completions', () => {
for (const model of ['gpt-5.5', 'gpt-5.6-sol', 'gpt-6-astra']) {
expect(completionsRejectsToolsWithReasoning('openai', model), model).toBe(true)
}
for (const model of ['gpt-5', 'gpt-5.1', 'gpt-35-turbo', 'o3']) {
expect(completionsRejectsToolsWithReasoning('azure_openai', model), model).toBe(false)
}
})
it('forwards an explicit off as effort none through OpenRouter', () => {
expect(
resolveRequestReasoning({
@@ -353,6 +393,23 @@ describe('Azure AI Foundry reasoning follows the model family', () => {
})
})
describe('backend parity', () => {
// windmill-ai's `providers/mod.rs` test reads the same file. The backend rules see the
// model id alone, so the file holds only rows whose answer doesn't depend on the
// provider: a Bedrock- or Gemini-Pro-specific row belongs in the tests above.
it.each(parity)(
'$provider $model',
({ provider, model, canDisable, completionsToolsNeedOff }) => {
const capability = getReasoningCapability(provider as AIProvider, model)
expect(capability.supported).toBe(true)
expect(capability.canDisable).toBe(canDisable)
expect(completionsRejectsToolsWithReasoning(provider as AIProvider, model)).toBe(
completionsToolsNeedOff
)
}
)
})
describe('resolveEffectiveReasoning', () => {
it('defaults capable models to high when unset', () => {
expect(resolveEffectiveReasoning({ provider: 'anthropic', model: 'claude-sonnet-4-6' })).toBe(
@@ -1,5 +1,5 @@
import type { AIProvider, AIProviderModel } from '$lib/gen'
import { parseModelId, usesAnthropicMessagesApi } from './modelConfig'
import { usesAnthropicMessagesApi } from './modelConfig'
/**
* Reasoning effort is provider/model-specific. We never normalize a single
@@ -33,11 +33,6 @@ export function stripLegacyThinkingSuffix(model: string): string {
: model
}
/** Bare model id without any provider/gateway prefix (e.g. OpenRouter's `openai/o3`). */
function baseModelId(model: string): string {
return parseModelId(model).base
}
/**
* Azure AI Foundry hosts multiple model families under one provider, so reasoning
* support follows the underlying model rather than the provider: Claude deployments
@@ -53,172 +48,218 @@ function reasoningProviderFamily(provider: AIProvider, model: string): AIProvide
}
/**
* Suggested effort levels per provider, sourced from each provider SDK's own
* vocabulary.
* Sentinel sent for the deepseek off case. It never reaches the wire as an
* effort: the 'deepseek' branch of `applyReasoningToConfig` translates it to
* the provider's `thinking: {type: "disabled"}` param (`reasoning_effort:
* "none"` is rejected by their API).
*/
const PROVIDER_REASONING_LEVELS: Partial<Record<AIProvider, ReasoningEffort[]>> = {
// DeepSeek accepts the full five-token vocabulary but only two levels are
// real: low/medium are server-mapped to high and xhigh to max — offering
// them would be a no-op knob.
deepseek: ['high', 'max'],
// Mistral's only effort token besides the 'none' disable is 'high'
// (anything else is rejected), so the knob is effectively on/off.
mistral: ['high']
}
export const DEEPSEEK_OFF_SENTINEL: ReasoningEffort = 'none'
/**
* OpenRouter validates effort against its own vocabulary
* (minimal..xhigh + none) and translates per underlying provider, so the
* real ladder depends on the model family: Anthropic gets all five as
* distinct budget ratios (minimal 10% .. xhigh 95% of max_tokens); Gemini
* maps to thinkingLevel with xhigh clamped to high (a no-op vs high);
* OpenAI gets the token passed through verbatim, so the per-model OpenAI
* scoping applies; DeepSeek server-maps low/medium to high and xhigh to max.
* Sentinel for the Anthropic off case. Like the DeepSeek one it never reaches
* the wire as an effort: the 'anthropic' branch of `applyReasoningToConfig`
* translates it to `thinking: {type: "disabled"}`, the only off that Opus and
* Sonnet 5 respect.
*/
function openrouterReasoningLevels(model: string): ReasoningEffort[] {
const m = model.toLowerCase()
const base = baseModelId(model)
if (/claude-(opus|sonnet)-(4|5)/.test(m)) {
return ['minimal', 'low', 'medium', 'high', 'xhigh']
}
if (m.includes('gemini-')) {
return geminiReasoningLevels(m)
}
if (base.startsWith('gpt-5') || /^o\d/.test(base)) {
return openaiReasoningLevels(base)
}
if (m.includes('deepseek-v4')) {
return ['high', 'xhigh']
}
return ['low', 'medium', 'high']
}
export const ANTHROPIC_OFF_SENTINEL: ReasoningEffort = 'none'
/**
* OpenAI's effort vocabulary is model-dependent: `minimal` exists on gpt-5 but
* not on gpt-5.1+, `xhigh` arrived on gpt-5.5 and `max` on gpt-5.6; o-series
* take low/medium/high. An unsupported level is rejected, so scope the list to
* the model. (`none` is the disable token, handled by `explicitOffToken`.)
* What the registry knows about one group of models. The rows below are matched in
* order against the lowercased model id, first match wins, so a narrower row goes above
* the row it overrides. A model no row matches does not reason, as far as we know.
*/
function openaiReasoningLevels(model: string): ReasoningEffort[] {
const base = baseModelId(model)
if (/^gpt-5\.6/.test(base)) {
return ['low', 'medium', 'high', 'xhigh', 'max']
}
if (/^gpt-5\.5/.test(base)) {
return ['low', 'medium', 'high', 'xhigh']
}
if (/^gpt-5\./.test(base)) {
return ['low', 'medium', 'high']
}
if (/^gpt-5/.test(base)) {
return ['minimal', 'low', 'medium', 'high']
}
return ['low', 'medium', 'high']
type ReasoningRule = {
match: RegExp
/** The levels the UI offers. An unsupported level is a 400, so this is per model. */
levels: readonly ReasoningEffort[]
/**
* Whether "off" really stops the model reasoning. When false the UI offers no off:
* the provider would reject it, or coerce it to the lowest level.
*/
canDisable: boolean
/**
* The effort to send for off on a model that reasons when the field is omitted.
* Unset where omission is already off. Always `'none'`: `requestsReasoning` reads
* that as off, and each wire format translates it (see `applyReasoningToConfig`).
*/
offToken?: ReasoningEffort
/**
* gpt-5.5 and later refuse function tools on Chat Completions while they reason,
* even with the effort omitted (live-verified); the Responses API has no such limit.
*/
completionsToolsNeedOff?: boolean
}
const LOW_TO_HIGH = ['low', 'medium', 'high']
const LOW_TO_XHIGH = ['low', 'medium', 'high', 'xhigh']
const LOW_TO_MAX = ['low', 'medium', 'high', 'xhigh', 'max']
const MINIMAL_TO_HIGH = ['minimal', 'low', 'medium', 'high']
/**
* Gemini's level ladder is model-dependent: Gemini 3+ Flash / Flash-Lite accept
* `minimal`, while 3.x Pro does not (and cannot disable thinking). Gemini 2.5
* uses numeric budgets — the proxy maps the three tiers to budget values, so
* `minimal` is not offered there.
* Claude models whose thinking cannot be turned off: an explicit disable 400s. Fable,
* Mythos, and the 5.x point releases (Sonnet 5.5, Opus 5.5), whose lowest setting is
* adaptive thinking at `low` (live-verified). The version match stops at one digit so a
* dated id (`claude-sonnet-5-20260101`) stays Sonnet 5.
*/
function geminiReasoningLevels(model: string): ReasoningEffort[] {
const m = model.toLowerCase()
const isGemini3Plus = !m.includes('gemini-2.5')
if (isGemini3Plus && (m.includes('flash') || m.includes('lite'))) {
return ['minimal', 'low', 'medium', 'high']
}
return ['low', 'medium', 'high']
const CLAUDE_ALWAYS_THINKING = /fable|mythos|claude-(opus|sonnet)-5[-.][1-9](?!\d)/
// Anthropic ids are matched anywhere in the id: Bedrock prefixes them
// (`us.anthropic.claude-opus-4-6-v1`). Opus 4.5 and older reject adaptive thinking.
const ANTHROPIC_RULES: ReasoningRule[] = [
{ match: CLAUDE_ALWAYS_THINKING, levels: LOW_TO_MAX, canDisable: false },
// The 5 family thinks when the field is absent, so off is an explicit disable.
{
match: /claude-(opus|sonnet)-5/,
levels: LOW_TO_MAX,
canDisable: true,
offToken: ANTHROPIC_OFF_SENTINEL
},
// 4.6-4.8 only think when asked, so omission is already off.
{ match: /claude-opus-4-[78]/, levels: LOW_TO_MAX, canDisable: true },
{ match: /claude-(opus|sonnet)-4-6/, levels: ['low', 'medium', 'high', 'max'], canDisable: true }
]
const BEDROCK_RULES: ReasoningRule[] = [
ANTHROPIC_RULES[0],
// AWS documents Sonnet 5 on Bedrock as always thinking, where the native API
// accepts a disable for it.
{ match: /claude-sonnet-5/, levels: LOW_TO_MAX, canDisable: false },
...ANTHROPIC_RULES.slice(1)
]
// Anchored at the start or after a gateway's `vendor/`, and the major is one digit:
// Azure names gpt-3.5 `gpt-35-turbo`. `minimal` exists on gpt-5 only, `xhigh` from
// gpt-5.5, `max` from gpt-5.6.
const OPENAI_RULES: ReasoningRule[] = [
// Live-verified: astra takes low..max only, where sol and luna also take `none`.
{
match: /(?:^|\/)gpt-6-astra/,
levels: LOW_TO_MAX,
canDisable: false,
completionsToolsNeedOff: true
},
{
match: /(?:^|\/)gpt-[6-9](?:[.:-]|$)/,
levels: LOW_TO_MAX,
canDisable: true,
offToken: 'none',
completionsToolsNeedOff: true
},
{
match: /(?:^|\/)gpt-5\.6/,
levels: LOW_TO_MAX,
canDisable: true,
offToken: 'none',
completionsToolsNeedOff: true
},
{
match: /(?:^|\/)gpt-5\.5/,
levels: LOW_TO_XHIGH,
canDisable: true,
offToken: 'none',
completionsToolsNeedOff: true
},
// A later gpt-5 minor keeps the tools limit, which holds for every version from 5.5,
// but only the levels every gpt-5.x takes until it has a row of its own.
{
match: /(?:^|\/)gpt-5\.(?:[5-9]|\d{2,})/,
levels: LOW_TO_HIGH,
canDisable: true,
offToken: 'none',
completionsToolsNeedOff: true
},
// gpt-5.1+ are off only through `none`: omitted, they reason at medium.
{ match: /(?:^|\/)gpt-5\./, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' },
// gpt-5 and the o-series reject `none` and reason when it is omitted.
{ match: /(?:^|\/)gpt-5(?:[:-]|$)/, levels: MINIMAL_TO_HIGH, canDisable: false },
{ match: /(?:^|\/)o\d/, levels: LOW_TO_HIGH, canDisable: false }
]
// Gemini 2.5/3 think by default; the backend proxy maps `none` to off on Flash, or to
// the floor on Pro, which enforces one (level `low` on 3.x, 128 tokens on 2.5).
// Gemini 3+ Flash / Flash-Lite accept `minimal`; 2.5 takes numeric budgets the proxy
// maps from three tiers.
const GEMINI_RULES: ReasoningRule[] = [
{ match: /gemini-2\.5.*pro/, levels: LOW_TO_HIGH, canDisable: false, offToken: 'none' },
{ match: /gemini-2\.5/, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' },
{ match: /gemini-3.*pro/, levels: LOW_TO_HIGH, canDisable: false, offToken: 'none' },
{ match: /gemini-3.*(flash|lite)/, levels: MINIMAL_TO_HIGH, canDisable: true, offToken: 'none' },
{ match: /gemini-3/, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' }
]
const REASONING_RULES: Partial<Record<AIProvider, ReasoningRule[]>> = {
anthropic: ANTHROPIC_RULES,
aws_bedrock: BEDROCK_RULES,
openai: OPENAI_RULES,
azure_openai: OPENAI_RULES,
googleai: GEMINI_RULES,
deepseek: [
// Every current API model takes reasoning_effort, but only two levels are real:
// low/medium are server-mapped to high, xhigh to max. The retired
// `deepseek-chat` alias means "non-thinking mode", so a saved selection on it
// must not silently become a thinking request. Off is a separate `thinking` param.
{
match: /(?:^|\/)deepseek(?!-chat(?::|$))/,
levels: ['high', 'max'],
canDisable: true,
offToken: DEEPSEEK_OFF_SENTINEL
}
],
mistral: [
// Only the ids verified to accept reasoning_effort (large, magistral, ministral
// and pinned versions reject it), whose only effort besides off is `high`.
{
match: /(?:^|\/)mistral-(?:(?:small|medium)-latest(?::|$)|medium-3[-.]5)/,
levels: ['high'],
canDisable: true
}
],
// OpenRouter validates effort against its own vocabulary (minimal..xhigh + none) and
// translates it per underlying provider, which scopes the ladder: Anthropic gets all
// five as budget ratios, OpenAI gets the token verbatim, DeepSeek server-maps. `none`
// is its documented off, more reliable than omission; it can only disable a model
// whose upstream can.
openrouter: [
{
match: /claude-(opus|sonnet)-5[-.][1-9](?!\d)/,
levels: ['minimal', ...LOW_TO_XHIGH],
canDisable: false
},
{
match: /claude-(opus|sonnet)-(4|5)/,
levels: ['minimal', ...LOW_TO_XHIGH],
canDisable: true,
offToken: 'none'
},
...GEMINI_RULES,
...OPENAI_RULES.map(({ completionsToolsNeedOff: _, ...rule }) => ({
...rule,
offToken: 'none'
})),
{ match: /deepseek-v4/, levels: LOW_TO_XHIGH.slice(2), canDisable: true, offToken: 'none' },
// deepseek-r1, grok-4 and :thinking variants reason unconditionally.
{
match: /deepseek-r|grok-4|:thinking/,
levels: LOW_TO_HIGH,
canDisable: false,
offToken: 'none'
}
]
}
/**
* Gemini Pro models cannot turn thinking off — the API enforces a floor
* (level `low` on 3.x Pro, a 128-token budget on 2.5 Pro), so an off option
* would silently mean "lowest". Flash / Flash-Lite can truly disable
* (budget 0 / level `minimal`).
*/
function geminiCanDisable(model: string): boolean {
return !model.toLowerCase().includes('pro')
/** The row that describes a model, and whether its provider family has rows at all. */
function findReasoningRule(
provider: AIProvider,
model: string
): { rule: ReasoningRule | undefined; known: boolean } {
const rules = REASONING_RULES[reasoningProviderFamily(provider, model)]
const id = stripLegacyThinkingSuffix(model).toLowerCase()
return { rule: rules?.find((rule) => rule.match.test(id)), known: rules !== undefined }
}
/**
* Anthropic's effort ladder is model-dependent: `xhigh` exists on Opus 4.7/4.8,
* the 5 family and Fable/Mythos; `max` also on Opus 4.6 and Sonnet 4.6.
* Offering an unsupported level would 400, so scope the list to the model.
*/
function anthropicReasoningLevels(model: string): ReasoningEffort[] {
const m = model.toLowerCase()
if (
/claude-(opus|sonnet)-5/.test(m) ||
/claude-opus-4-(7|8)/.test(m) ||
m.includes('fable') ||
m.includes('mythos')
) {
return ['low', 'medium', 'high', 'xhigh', 'max']
}
return ['low', 'medium', 'high', 'max']
}
/** Mistral writes both `mistral-medium-3.5` and `mistral-medium-3-5`. */
function normalizeMistralId(model: string): string {
return baseModelId(model).replace(/\./g, '-')
}
/**
* Conservative static predicate for whether a model accepts an effort knob.
* Kept tight to avoid 400s on models that reject reasoning params.
*/
function supportsReasoningStatic(provider: AIProvider, model: string): boolean {
const m = model.toLowerCase()
const base = baseModelId(model)
switch (reasoningProviderFamily(provider, model)) {
case 'anthropic':
// Bedrock serves the same Claude models under prefixed ids
// (e.g. `us.anthropic.claude-opus-4-6-v1`), so match on the full string.
case 'aws_bedrock':
// 4.6+ only: Opus 4.5 rejects adaptive thinking (and, on Bedrock,
// the whole output_config surface) — live-verified hard 400.
return (
/claude-opus-(4-(6|7|8)|5)/.test(m) ||
/claude-sonnet-(4-6|5)/.test(m) ||
m.includes('fable') ||
m.includes('mythos')
)
case 'openai':
case 'azure_openai':
return base.startsWith('gpt-5') || /^o\d/.test(base)
case 'openrouter':
// Best-effort markers for models whose `supported_parameters` include
// `reasoning` in OpenRouter's catalog; OpenRouter translates the effort
// per underlying provider.
return (
base.startsWith('gpt-5') ||
/^o\d/.test(base) ||
/claude-(opus|sonnet)-(4|5)/.test(m) ||
/gemini-(2\.5|3)/.test(m) ||
m.includes('deepseek-r') ||
m.includes('deepseek-v4') ||
m.includes('grok-4') ||
m.includes(':thinking')
)
case 'googleai':
return /gemini-(2\.5|3)/.test(m)
case 'deepseek':
// All current API models take reasoning_effort (live-verified). The
// retired `deepseek-chat` alias stays excluded: its documented meaning
// is "non-thinking mode", so a saved selection on it must not silently
// become a thinking request.
return base.startsWith('deepseek') && base !== 'deepseek-chat'
case 'mistral':
// Only the ids verified to accept reasoning_effort; other models
// (large, magistral, ministral, pinned versions) reject the param.
return (
/^mistral-(small|medium)-latest$/.test(base) ||
normalizeMistralId(model).startsWith('mistral-medium-3-5')
)
default:
return false
}
/** Whether Chat Completions needs reasoning off for this model to take function tools. */
export function completionsRejectsToolsWithReasoning(provider: AIProvider, model: string): boolean {
return findReasoningRule(provider, model).rule?.completionsToolsNeedOff ?? false
}
export type ReasoningCapability = {
@@ -241,89 +282,12 @@ export type ReasoningCapability = {
known: boolean
}
/** Provider families the registry has real rules for; everything else is a shrug. */
const KNOWN_REASONING_FAMILIES: ReadonlySet<string> = new Set([
'anthropic',
'aws_bedrock',
'openai',
'azure_openai',
'openrouter',
'googleai',
'deepseek',
'mistral'
])
/** Resolve the reasoning capability of a model from the static registry. */
export function getReasoningCapability(provider: AIProvider, model: string): ReasoningCapability {
const bareModel = stripLegacyThinkingSuffix(model)
const known = KNOWN_REASONING_FAMILIES.has(reasoningProviderFamily(provider, bareModel))
const supported = supportsReasoningStatic(provider, bareModel)
if (!supported) {
return { supported: false, levels: [], canDisable: false, known }
}
const family = reasoningProviderFamily(provider, bareModel)
const levels =
family === 'anthropic' || family === 'aws_bedrock'
? anthropicReasoningLevels(bareModel)
: family === 'googleai'
? geminiReasoningLevels(bareModel)
: family === 'openai' || family === 'azure_openai'
? openaiReasoningLevels(bareModel)
: family === 'openrouter'
? openrouterReasoningLevels(bareModel)
: (PROVIDER_REASONING_LEVELS[family] ?? ['low', 'medium', 'high'])
return { supported, levels, canDisable: canDisableReasoning(provider, bareModel), known }
}
/**
* Whether selecting "off" truly disables reasoning for the model. Off is
* sent either as an explicit provider disable (see `explicitOffToken`) or by
* omitting the effort — which only works where the model doesn't reason by
* default.
*/
function canDisableReasoning(provider: AIProvider, model: string): boolean {
const m = model.toLowerCase()
const base = baseModelId(model)
switch (reasoningProviderFamily(provider, model)) {
case 'anthropic':
// Every Claude but Fable and Mythos can stop thinking: 4.6-4.8 by
// omission, and the 5 family through the explicit disable that
// `explicitOffToken` sends.
return !ANTHROPIC_ALWAYS_THINKING.test(m)
case 'aws_bedrock':
// Same models, different answer: AWS documents Sonnet 5 on Bedrock as
// always thinking, where the native API accepts a disable for it.
return !(ANTHROPIC_ALWAYS_THINKING.test(m) || m.includes('claude-sonnet-5'))
case 'googleai':
return geminiCanDisable(model)
case 'openai':
case 'azure_openai':
// gpt-5.1+ accept effort 'none'; gpt-5 and o-series reject it and
// reason at `medium` by default, so omission isn't off either.
return /^gpt-5\./.test(base)
case 'openrouter':
// 'none' is in OpenRouter's vocabulary, but the gateway can't
// disable a model whose upstream can't — scope off per underlying
// family, like the levels.
// The 5 family thinks by default, but its upstream takes an explicit
// disable, so the gateway's 'none' has something to translate to.
if (/claude-(opus|sonnet)-(4|5)/.test(m)) {
return true
}
if (m.includes('gemini-')) {
return geminiCanDisable(m)
}
if (base.startsWith('gpt-5') || /^o\d/.test(base)) {
return /^gpt-5\./.test(base)
}
if (m.includes('deepseek-v4')) {
return true
}
// grok-4, deepseek-r1 and :thinking variants reason unconditionally.
return false
default:
return true
}
const { rule, known } = findReasoningRule(provider, model)
return rule
? { supported: true, levels: [...rule.levels], canDisable: rule.canDisable, known }
: { supported: false, levels: [], canDisable: false, known }
}
export function supportsReasoning(provider: AIProvider, model: string): boolean {
@@ -352,64 +316,14 @@ export function resolveEffectiveReasoning(
: undefined
}
/**
* Sentinel sent for the deepseek off case. It never reaches the wire as an
* effort: the 'deepseek' branch of `applyReasoningToConfig` translates it to
* the provider's `thinking: {type: "disabled"}` param (`reasoning_effort:
* "none"` is rejected by their API).
*/
export const DEEPSEEK_OFF_SENTINEL: ReasoningEffort = 'none'
/**
* Sentinel for the Anthropic off case. Like the DeepSeek one it never reaches
* the wire as an effort: the 'anthropic' branch of `applyReasoningToConfig`
* translates it to `thinking: {type: "disabled"}`, which is the only off the
* always-on 5 family respects.
*/
export const ANTHROPIC_OFF_SENTINEL: ReasoningEffort = 'none'
/** Claude models whose thinking cannot be turned off — an explicit disable 400s. */
const ANTHROPIC_ALWAYS_THINKING = /fable|mythos/
/**
* Disable token to forward when the user explicitly turns reasoning off on a
* model that reasons *by default* — omitting the field would silently keep
* the default-on behavior. Undefined means omission is the correct off. Every
* token is `'none'`: `requestsReasoning` reads that value as off.
* the default-on behavior. Undefined means omission is the correct off, or that
* the model cannot be turned off at all.
*/
export function explicitOffToken(provider: AIProvider, model: string): ReasoningEffort | undefined {
switch (reasoningProviderFamily(provider, model)) {
case 'anthropic':
// Claude 4.6-4.8 only think when asked, so omission is already a
// real off there and stays the wire form. Only the 5 family, which
// thinks when the field is absent, needs the explicit disable —
// Fable and Mythos reject it outright and get no off token at all.
return /claude-(opus|sonnet)-5/.test(model.toLowerCase()) ? ANTHROPIC_OFF_SENTINEL : undefined
case 'aws_bedrock':
// Bedrock's Sonnet 5 cannot be disabled at all, so only Opus 5 gets
// the sentinel; the rest keep omission.
return model.toLowerCase().includes('claude-opus-5') ? ANTHROPIC_OFF_SENTINEL : undefined
case 'googleai':
// Gemini 2.5/3 think by default (dynamic budget / level). The backend
// proxy maps 'none' to off on Flash, or the floor on Pro (only
// reachable via a stale persisted preference — see canDisableReasoning).
return 'none'
case 'deepseek':
return DEEPSEEK_OFF_SENTINEL
case 'openai':
case 'azure_openai':
// gpt-5.1+ reasoning is off only via the explicit 'none' effort
// (gpt-5.5 defaults to medium when the field is omitted).
return /^gpt-5\./.test(baseModelId(model)) ? 'none' : undefined
case 'openrouter':
// OpenRouter validates effort against xhigh..minimal|none and
// documents 'none' as disabling reasoning, translated per the
// underlying provider — more reliable than omission, which keeps
// reasoning-by-default models thinking.
return 'none'
default:
return undefined
}
return findReasoningRule(provider, model).rule?.offToken
}
/**