From e117871becc04a84baabe2aa0db6ae6e58cdd058 Mon Sep 17 00:00:00 2001 From: hugocasa Date: Wed, 30 Sep 2026 13:10:24 +0200 Subject: [PATCH] weekly ai evals on current models, and claude 5.5/gpt-6 support (#11409) * feat: run ai evals weekly on current models and post results to a dashboard Co-Authored-By: Claude Opus 5.5 * feat: add current flagship models, a reasoning flag and claude 5.5 defaults Co-Authored-By: Claude Opus 5.5 * fix: never send a reasoning disable claude 5.5 or gpt-6-astra reject, and treat gpt-6 as a reasoning model Co-Authored-By: Claude Opus 5.5 * fix: address review on gpt-6 support, chat completions tools and model metadata Co-Authored-By: Claude Opus 5.5 * fix: leave tiered gpt-6 unpriced and drop the off sentinel on gpt-5 and o-series Co-Authored-By: Claude Opus 5.5 * docs: point the ai_evals readme at the model registry instead of copying it Co-Authored-By: Claude Opus 5.5 * refactor: encode the reasoning rules as per-family maps with a shared parity fixture Co-Authored-By: Claude Opus 5.5 * fix: keep the chat completions tools rule open-ended past gpt-5.6 and scope the parity fixture Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Opus 5.5 --- .github/workflows/ai-evals-scheduled.yml | 171 ++++++ ai_evals/README.md | 17 +- .../frontend/core/script/scriptEvalRunner.ts | 10 +- .../frontend/core/shared/providerConfig.ts | 13 + ai_evals/bun.lock | 120 +---- ai_evals/cli/index.ts | 18 + ai_evals/core/judge.ts | 14 +- ai_evals/core/models.ts | 71 +++ ai_evals/package.json | 4 +- backend/windmill-ai/src/providers/mod.rs | 164 ++++++ backend/windmill-ai/src/providers/other.rs | 10 + backend/windmill-ai/src/types.rs | 5 +- frontend/src/lib/components/copilot/lib.ts | 22 +- .../components/copilot/modelConfig.test.ts | 7 + .../src/lib/components/copilot/modelConfig.ts | 4 +- .../components/copilot/modelPricing.test.ts | 10 + .../lib/components/copilot/modelPricing.ts | 5 + .../components/copilot/reasoningParity.json | 84 +++ .../copilot/reasoningRegistry.test.ts | 57 ++ .../components/copilot/reasoningRegistry.ts | 496 ++++++++---------- 20 files changed, 888 insertions(+), 414 deletions(-) create mode 100644 .github/workflows/ai-evals-scheduled.yml create mode 100644 frontend/src/lib/components/copilot/reasoningParity.json diff --git a/.github/workflows/ai-evals-scheduled.yml b/.github/workflows/ai-evals-scheduled.yml new file mode 100644 index 0000000000..964e5d0cab --- /dev/null +++ b/.github/workflows/ai-evals-scheduled.yml @@ -0,0 +1,171 @@ +name: AI Evals (scheduled) + +# Full ai_evals suites on current models, posted to the AI evals dashboard in +# the windmill-prod workspace (f/ai/ai_evals_dashboard) so quality is tracked +# over time. Weekly, since one pass of every suite at --runs 3 costs about 50M +# tokens per model: every suite runs on MODELS, and global (the mode users get) +# also runs on GLOBAL_EXTRA_MODELS, one flagship per other provider. Run it by +# hand to measure a branch against main. +on: + schedule: + - cron: "0 3 * * 1" + workflow_dispatch: + inputs: + modes: + description: "Space-separated modes" + default: "global flow app script cli" + models: + description: "Space-separated model aliases (bun run cli -- models)" + default: "sonnet-5.5" + global_extra_models: + description: "Extra model aliases for global mode only" + default: "gpt-6-astra gemini-3.8-flash" + runs: + description: "Runs per case" + default: "3" + reasoning: + description: "Reasoning effort for frontend modes (empty: the product default)" + default: "" + +concurrency: + group: ai-evals-scheduled-${{ github.ref }} + +env: + MODELS: ${{ inputs.models || 'sonnet-5.5' }} + GLOBAL_EXTRA_MODELS: ${{ inputs.global_extra_models || 'gpt-6-astra gemini-3.8-flash' }} + RUNS: ${{ inputs.runs || '3' }} + REASONING: ${{ inputs.reasoning }} + INGEST_URL: https://app.windmill.dev/api/r/f/ai/ingest_ai_eval_run + +jobs: + setup: + runs-on: ubuntu-latest + outputs: + modes: ${{ steps.modes.outputs.modes }} + steps: + - id: modes + env: + MODES: ${{ inputs.modes || 'global flow app script cli' }} + run: | + echo "modes=$(jq -cn --arg m "$MODES" '$m | split(" ") + | map(select(IN("global", "flow", "app", "script", "cli")))')" >> "$GITHUB_OUTPUT" + + evals: + needs: setup + runs-on: ubicloud-standard-16 + timeout-minutes: 330 + strategy: + fail-fast: false + matrix: + mode: ${{ fromJSON(needs.setup.outputs.modes) }} + services: + postgres: + image: postgres:16 + ports: + - 5432:5432 + env: + POSTGRES_DB: windmill + POSTGRES_PASSWORD: changeme + options: >- + --health-cmd pg_isready --health-interval 10s --health-timeout 5s + --health-retries 5 + steps: + - uses: actions/checkout@v4 + + - uses: actions-rust-lang/setup-rust-toolchain@v1 + with: + cache-workspaces: backend + toolchain: 1.97.0 + + - uses: oven-sh/setup-bun@v2 + with: + bun-version: 1.4.0 + + - uses: actions/setup-node@v7 + with: + node-version: "24" + + - name: Build Windmill + working-directory: ./backend + env: + SQLX_OFFLINE: true + CARGO_BUILD_JOBS: 12 + RUSTFLAGS: "" + run: cargo build --features quickjs + + - name: Start Windmill + working-directory: ./backend + env: + DATABASE_URL: postgres://postgres:changeme@localhost:5432/windmill + RUST_LOG: info + run: | + mkdir -p ../ai_evals/logs + ./target/debug/windmill > ../ai_evals/logs/windmill.log 2>&1 & + for i in $(seq 1 60); do + curl -sf http://localhost:8000/api/version > /dev/null 2>&1 && break + sleep 2 + done + curl -sf http://localhost:8000/api/version > /dev/null || { tail -50 ../ai_evals/logs/windmill.log; exit 1; } + + - name: Install frontend deps + generate client + working-directory: ./frontend + run: | + npm ci + npm run generate-backend-client + + - name: Install CLI deps + generate CLI client + working-directory: ./cli + run: bun install && ./gen_wm_client.sh && ./windmill-utils-internal/gen_wm_client.sh + + - name: Run ${{ matrix.mode }} evals + working-directory: ./ai_evals + env: + MODE: ${{ matrix.mode }} + WMILL_AI_EVAL_BACKEND_URL: http://localhost:8000 + WMILL_AI_EVAL_BACKEND_WORKSPACE: integration-tests + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + GEMINI_API_KEY: ${{ secrets.GOOGLE_API_KEY }} + DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }} + run: | + bun install + mkdir -p results + models="$MODELS" + [ "$MODE" = global ] && models="$models $GLOBAL_EXTRA_MODELS" + reasoning=() + [ -n "$REASONING" ] && [ "$MODE" != cli ] && reasoning=(--reasoning "$REASONING") + for m in $models; do + bun run cli -- run "$MODE" --model "$m" --runs "$RUNS" "${reasoning[@]}" \ + --output "$PWD/results/$MODE-$m.json" || echo "::warning::$MODE on $m errored" + done + + - name: Post results to the dashboard + if: always() + working-directory: ./ai_evals + env: + MODE: ${{ matrix.mode }} + INGEST_TOKEN: ${{ secrets.AI_EVALS_INGEST_TOKEN }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + run: | + [ -n "$INGEST_TOKEN" ] || { echo "::warning::AI_EVALS_INGEST_TOKEN is not set"; exit 0; } + shopt -s nullglob + for f in results/"$MODE"-*.json; do + # Keep what the dashboard reads; the full traces stay in the run's artifacts. + jq -c --arg ref "$GITHUB_REF" --arg trigger "$GITHUB_EVENT_NAME" --arg url "$RUN_URL" ' + {result_json: (del(.cases[].prompt, .cases[].initialPath, .cases[].expectedPath) + | .cases[].attempts[] |= {attempt, passed, durationMs, toolCallCount, judgeScore, + judgeSummary, error, tokenUsage, checks: [.checks[]? | {name, passed}]}), + git_ref: $ref, trigger: $trigger, run_url: $url}' "$f" \ + | curl -sf --retry 3 -X POST "$INGEST_URL" \ + -H "Authorization: Bearer $INGEST_TOKEN" -H 'content-type: application/json' \ + --data-binary @- && echo " <- $f" || echo "::warning::failed to post $f" + done + + - name: Archive logs and results + uses: actions/upload-artifact@v4 + if: always() + with: + name: ai-evals-${{ matrix.mode }} + path: | + ai_evals/logs + ai_evals/results diff --git a/ai_evals/README.md b/ai_evals/README.md index 603f28e5f9..bff4294e52 100644 --- a/ai_evals/README.md +++ b/ai_evals/README.md @@ -77,31 +77,20 @@ Public CLI surface: - `--verbose`: stream assistant output for frontend runs - `--skip-judge`: skip LLM judge scoring for the run - `--execution-only`: only require the model/proxy/frontend loop to complete; skip validators, tool expectations, backend artifact validation, and judge scoring +- `--reasoning `: reasoning effort for frontend modes (`off`, `low`, `medium`, `high`, `max`, …); without it the product's default applies (`high` on models that can reason). The effort is appended to the recorded model label (`anthropic:claude-sonnet-5-5@max`) - `--record`: append a compact tracked summary line to `ai_evals/history/.jsonl` for full-suite runs only - `--backend-validation `: optional backend smoke validation (`off` or `preview`) for `script` and `flow` evals ## Models -Use `bun run cli -- models` to see the current aliases. - -Today: - -- `haiku` -- `sonnet` -- `opus` -- `4o` -- `gpt-5.5` -- `gemini-3-flash-preview` -- `gemini-3.1-pro-preview` -- `deepseek-v4-flash` -- `deepseek-v4-pro` +Use `bun run cli -- models` to see the current aliases; `core/models.ts` is the list. Notes: - the command also prints accepted alias spellings such as `gpt-4o`, `gpt-55`, `claude-opus-4.6`, and `claude-haiku-4.5` - frontend modes (`flow`, `script`, `app`, `global`) can use Anthropic, OpenAI, Gemini, and DeepSeek-backed aliases - `cli` mode always uses the Anthropic agent SDK, so only Anthropic aliases are valid there -- the judge model is separate and currently defaults to `claude-sonnet-4-6`; use `--skip-judge` for deterministic-only runs +- the judge model is separate and currently defaults to `claude-sonnet-5-5`; use `--skip-judge` for deterministic-only runs ## Case Format diff --git a/ai_evals/adapters/frontend/core/script/scriptEvalRunner.ts b/ai_evals/adapters/frontend/core/script/scriptEvalRunner.ts index b9c80f3554..0ff352f3d4 100644 --- a/ai_evals/adapters/frontend/core/script/scriptEvalRunner.ts +++ b/ai_evals/adapters/frontend/core/script/scriptEvalRunner.ts @@ -15,6 +15,7 @@ import { runEval } from "../shared"; import type { ModeRunContext } from "../../../../core/types"; import type { TokenUsage, ToolCallDetail } from "../shared/types"; import type { WindmillBackendSettings } from "../../../../core/windmillBackendSettings"; +import { evalReasoningEffort } from "../shared/providerConfig"; export interface ScriptEvalResult { success: boolean; @@ -41,14 +42,15 @@ export interface ScriptEvalOptions { function resolveModelProvider( model: string, provider?: AIProvider, -): AIProviderModel { +): AIProviderModel & { reasoning?: string } { + const reasoning = evalReasoningEffort(); if (provider) { - return { provider, model }; + return { provider, model, reasoning }; } if (model.startsWith("claude")) { - return { provider: "anthropic", model }; + return { provider: "anthropic", model, reasoning }; } - return { provider: "openai", model }; + return { provider: "openai", model, reasoning }; } export async function runScriptEval( diff --git a/ai_evals/adapters/frontend/core/shared/providerConfig.ts b/ai_evals/adapters/frontend/core/shared/providerConfig.ts index 9d9e315217..49d5cb6ef8 100644 --- a/ai_evals/adapters/frontend/core/shared/providerConfig.ts +++ b/ai_evals/adapters/frontend/core/shared/providerConfig.ts @@ -12,6 +12,12 @@ export interface EvalClients { export interface ResolvedEvalModelProvider { provider: FrontendEvalProvider; model: string; + reasoning?: string; +} + +/** `run --reasoning` hands the effort to the frontend runtime through the environment. */ +export function evalReasoningEffort(): string | undefined { + return process.env.WMILL_AI_EVAL_REASONING || undefined; } export interface WindmillAiProxyClientConfig { @@ -73,6 +79,13 @@ export function createEvalClients(input: { export function resolveEvalModelProvider( model: string, provider?: FrontendEvalProvider, +): ResolvedEvalModelProvider { + return { ...resolveProvider(model, provider), reasoning: evalReasoningEffort() }; +} + +function resolveProvider( + model: string, + provider?: FrontendEvalProvider, ): ResolvedEvalModelProvider { if (provider) { return { provider, model }; diff --git a/ai_evals/bun.lock b/ai_evals/bun.lock index eaed1db99a..798022d49a 100644 --- a/ai_evals/bun.lock +++ b/ai_evals/bun.lock @@ -5,8 +5,8 @@ "": { "name": "windmill-ai-evals", "dependencies": { - "@anthropic-ai/claude-agent-sdk": "^0.2.25", - "@anthropic-ai/sdk": "^0.39.0", + "@anthropic-ai/claude-agent-sdk": "^0.3.284", + "@anthropic-ai/sdk": "^0.129.0", "commander": "^14.0.3", "openai": "^6.9.1", "yaml": "^2.8.3", @@ -18,66 +18,44 @@ }, }, "packages": { - "@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.2.87", "", { "dependencies": { "@anthropic-ai/sdk": "^0.74.0", "@modelcontextprotocol/sdk": "^1.27.1" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "^0.34.2", "@img/sharp-darwin-x64": "^0.34.2", "@img/sharp-linux-arm": "^0.34.2", "@img/sharp-linux-arm64": "^0.34.2", "@img/sharp-linux-x64": "^0.34.2", "@img/sharp-linuxmusl-arm64": "^0.34.2", "@img/sharp-linuxmusl-x64": "^0.34.2", "@img/sharp-win32-arm64": "^0.34.2", "@img/sharp-win32-x64": "^0.34.2" }, "peerDependencies": { "zod": "^4.0.0" } }, "sha512-WWmgBPxPhBOvNT0ujI8vPTI2lK+w5YEkEZ/y1mH0EDkK/0kBnxVJNhCtG5vnueiAViwLoUOFn66pbkDiivijdA=="], + "@anthropic-ai/claude-agent-sdk": ["@anthropic-ai/claude-agent-sdk@0.3.284", "", { "optionalDependencies": { "@anthropic-ai/claude-agent-sdk-darwin-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-darwin-x64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-x64": "0.3.284", "@anthropic-ai/claude-agent-sdk-linux-x64-musl": "0.3.284", "@anthropic-ai/claude-agent-sdk-win32-arm64": "0.3.284", "@anthropic-ai/claude-agent-sdk-win32-x64": "0.3.284" }, "peerDependencies": { "@anthropic-ai/sdk": ">=0.93.0", "@modelcontextprotocol/sdk": "^1.29.0", "zod": "^4.0.0" } }, "sha512-NSoJwEq6nFSf8dtaacYx37QdGgqApI3eHFUlxclMLYi8irb6ZJwEaUnPnLUCyAyg9W/tMkgZC0GXWLRCc01I0w=="], - "@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.39.0", "", { "dependencies": { "@types/node": "^18.11.18", "@types/node-fetch": "^2.6.4", "abort-controller": "^3.0.0", "agentkeepalive": "^4.2.1", "form-data-encoder": "1.7.2", "formdata-node": "^4.3.2", "node-fetch": "^2.6.7" } }, "sha512-eMyDIPRZbt1CCLErRCi3exlAvNkBtRe+kW5vvJyef93PmNr/clstYgHhtvmkxN82nlKgzyGPCyGxrm0JQ1ZIdg=="], + "@anthropic-ai/claude-agent-sdk-darwin-arm64": ["@anthropic-ai/claude-agent-sdk-darwin-arm64@0.3.284", "", { "os": "darwin", "cpu": "arm64" }, "sha512-gKY9MUjY83398uCiPLHsd87kyzu7agIM7ApqWpJkpSINep6hxx4rNoR8bUbNWg49Aoe/PW/DJkQByAQuNzR9rg=="], - "@babel/runtime": ["@babel/runtime@7.29.2", "", {}, "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g=="], + "@anthropic-ai/claude-agent-sdk-darwin-x64": ["@anthropic-ai/claude-agent-sdk-darwin-x64@0.3.284", "", { "os": "darwin", "cpu": "x64" }, "sha512-P+q6Z7sKeYz99uE7RJc4au1IK+4SiWe69pyHY3QXMnjZq8hSFU/5Mo8iSurLk9jRlihsMltV1rLA5naouYyCAQ=="], + + "@anthropic-ai/claude-agent-sdk-linux-arm64": ["@anthropic-ai/claude-agent-sdk-linux-arm64@0.3.284", "", { "os": "linux", "cpu": "arm64" }, "sha512-LDpuYDaz+pCdG29iy1pw4P1D2YVtpZOb4rWOPxxSpg2fFyUeQ5WrfnuszOxhDWpNh1k1ekDiKm3A9OzGcrkdtA=="], + + "@anthropic-ai/claude-agent-sdk-linux-arm64-musl": ["@anthropic-ai/claude-agent-sdk-linux-arm64-musl@0.3.284", "", { "os": "linux", "cpu": "arm64" }, "sha512-U3/uAv1TS8sMWsgWQSxYbdND5AaRXXpP5RxHU1oYC5ZizlZE8/xAVum9dj1QX3rVZJg8EzsqZveq/AkEv5ABfQ=="], + + "@anthropic-ai/claude-agent-sdk-linux-x64": ["@anthropic-ai/claude-agent-sdk-linux-x64@0.3.284", "", { "os": "linux", "cpu": "x64" }, "sha512-yGytBCCwJvWeg1FzFpgnLM9BOM2vKVExtv6pMJHMtF82J5yhKHcxodRaYlQVd/HzJJ2nruKb9EnrhRBZ//4yRw=="], + + "@anthropic-ai/claude-agent-sdk-linux-x64-musl": ["@anthropic-ai/claude-agent-sdk-linux-x64-musl@0.3.284", "", { "os": "linux", "cpu": "x64" }, "sha512-4x0Q8CFkCNbeENE7pRhuUmkwh/sQfTq1yWi4ZE56stve9pgoKzu6ZDa1SEtvK1YIooGE/qM5rVS8xFLE1gbQ0w=="], + + "@anthropic-ai/claude-agent-sdk-win32-arm64": ["@anthropic-ai/claude-agent-sdk-win32-arm64@0.3.284", "", { "os": "win32", "cpu": "arm64" }, "sha512-zOXkdPHhElyxyFJ6eY5R+YGUswqBDZ+ZHNaRmQGJ28RkQYgwG5imNy/nfFR+/CPGlMnH9ZzbGc2rKYApm1K4Tg=="], + + "@anthropic-ai/claude-agent-sdk-win32-x64": ["@anthropic-ai/claude-agent-sdk-win32-x64@0.3.284", "", { "os": "win32", "cpu": "x64" }, "sha512-VGaFRDCOPloj5IjvJTyy5JsSl9sE2HR3W+A6f/cjgh23/7Y5vK7/ka+1JQD+AkoGO9tbHqP2O6god3IBrLvOaw=="], + + "@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.129.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1", "standardwebhooks": "^1.0.0" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-MH7LB20kNpGLUpPTe2OG5XvL/KhX8HUO9XyBaFjgFk6TLS4Xjt0VPIuMKywe7POKKPslX+QxSlVok9e+8kG5Nw=="], + + "@babel/runtime": ["@babel/runtime@7.29.7", "", {}, "sha512-Nq8OhGWiZIZGV6hLHoyAKLLcJihP/xFeBMGJoUrxTX2psI8dCifzLhZISFb+VWS3wFMRDmCGw5R+dOySCqPLhw=="], "@hono/node-server": ["@hono/node-server@1.19.12", "", { "peerDependencies": { "hono": "^4" } }, "sha512-txsUW4SQ1iilgE0l9/e9VQWmELXifEFvmdA1j6WFh/aFPj99hIntrSsq/if0UWyGVkmrRPKA1wCeP+UCr1B9Uw=="], - "@img/sharp-darwin-arm64": ["@img/sharp-darwin-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-arm64": "1.2.4" }, "os": "darwin", "cpu": "arm64" }, "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w=="], - - "@img/sharp-darwin-x64": ["@img/sharp-darwin-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-darwin-x64": "1.2.4" }, "os": "darwin", "cpu": "x64" }, "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw=="], - - "@img/sharp-libvips-darwin-arm64": ["@img/sharp-libvips-darwin-arm64@1.2.4", "", { "os": "darwin", "cpu": "arm64" }, "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g=="], - - "@img/sharp-libvips-darwin-x64": ["@img/sharp-libvips-darwin-x64@1.2.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg=="], - - "@img/sharp-libvips-linux-arm": ["@img/sharp-libvips-linux-arm@1.2.4", "", { "os": "linux", "cpu": "arm" }, "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A=="], - - "@img/sharp-libvips-linux-arm64": ["@img/sharp-libvips-linux-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw=="], - - "@img/sharp-libvips-linux-x64": ["@img/sharp-libvips-linux-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw=="], - - "@img/sharp-libvips-linuxmusl-arm64": ["@img/sharp-libvips-linuxmusl-arm64@1.2.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw=="], - - "@img/sharp-libvips-linuxmusl-x64": ["@img/sharp-libvips-linuxmusl-x64@1.2.4", "", { "os": "linux", "cpu": "x64" }, "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg=="], - - "@img/sharp-linux-arm": ["@img/sharp-linux-arm@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm": "1.2.4" }, "os": "linux", "cpu": "arm" }, "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw=="], - - "@img/sharp-linux-arm64": ["@img/sharp-linux-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg=="], - - "@img/sharp-linux-x64": ["@img/sharp-linux-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linux-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ=="], - - "@img/sharp-linuxmusl-arm64": ["@img/sharp-linuxmusl-arm64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" }, "os": "linux", "cpu": "arm64" }, "sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg=="], - - "@img/sharp-linuxmusl-x64": ["@img/sharp-linuxmusl-x64@0.34.5", "", { "optionalDependencies": { "@img/sharp-libvips-linuxmusl-x64": "1.2.4" }, "os": "linux", "cpu": "x64" }, "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q=="], - - "@img/sharp-win32-arm64": ["@img/sharp-win32-arm64@0.34.5", "", { "os": "win32", "cpu": "arm64" }, "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g=="], - - "@img/sharp-win32-x64": ["@img/sharp-win32-x64@0.34.5", "", { "os": "win32", "cpu": "x64" }, "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw=="], - "@modelcontextprotocol/sdk": ["@modelcontextprotocol/sdk@1.29.0", "", { "dependencies": { "@hono/node-server": "^1.19.9", "ajv": "^8.17.1", "ajv-formats": "^3.0.1", "content-type": "^1.0.5", "cors": "^2.8.5", "cross-spawn": "^7.0.5", "eventsource": "^3.0.2", "eventsource-parser": "^3.0.0", "express": "^5.2.1", "express-rate-limit": "^8.2.1", "hono": "^4.11.4", "jose": "^6.1.3", "json-schema-typed": "^8.0.2", "pkce-challenge": "^5.0.0", "raw-body": "^3.0.0", "zod": "^3.25 || ^4.0", "zod-to-json-schema": "^3.25.1" }, "peerDependencies": { "@cfworker/json-schema": "^4.1.1" }, "optionalPeers": ["@cfworker/json-schema"] }, "sha512-zo37mZA9hJWpULgkRpowewez1y6ML5GsXJPY8FI0tBBCd77HEvza4jDqRKOXgHNn867PVGCyTdzqpz0izu5ZjQ=="], + "@stablelib/base64": ["@stablelib/base64@1.0.1", "", {}, "sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ=="], + "@types/bun": ["@types/bun@1.3.11", "", { "dependencies": { "bun-types": "1.3.11" } }, "sha512-5vPne5QvtpjGpsGYXiFyycfpDF2ECyPcTSsFBMa0fraoxiQyMJ3SmuQIGhzPg2WJuWxVBoxWJ2kClYTcw/4fAg=="], - "@types/node": ["@types/node@18.19.130", "", { "dependencies": { "undici-types": "~5.26.4" } }, "sha512-GRaXQx6jGfL8sKfaIDD6OupbIHBr9jv7Jnaml9tB7l4v068PAOXqfcujMMo5PhbIs6ggR1XODELqahT2R8v0fg=="], - - "@types/node-fetch": ["@types/node-fetch@2.6.13", "", { "dependencies": { "@types/node": "*", "form-data": "^4.0.4" } }, "sha512-QGpRVpzSaUs30JBSGPjOg4Uveu384erbHBoT1zeONvyCfwQxIkUshLAOqN/k9EjGviPRmWTTe6aH2qySWKTVSw=="], - - "abort-controller": ["abort-controller@3.0.0", "", { "dependencies": { "event-target-shim": "^5.0.0" } }, "sha512-h8lQ8tacZYnR3vNQTgibj+tODHI5/+l06Au2Pcriv/Gmet0eaj4TwWH41sO9wnHDiQsEj19q0drzdWdeAHtweg=="], + "@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="], "accepts": ["accepts@2.0.0", "", { "dependencies": { "mime-types": "^3.0.0", "negotiator": "^1.0.0" } }, "sha512-5cvg6CtKwfgdmVqY1WIiXKc3Q1bkRqGLi+2W/6ao+6Y7gu/RCwRuAhGEzh5B4KlszSuTLgZYuqFqo5bImjNKng=="], - "agentkeepalive": ["agentkeepalive@4.6.0", "", { "dependencies": { "humanize-ms": "^1.2.1" } }, "sha512-kja8j7PjmncONqaTsB8fQ+wE2mSU2DJ9D4XKoJ5PFWIdRMa6SLSN1ff4mOr4jCbfRSsxR4keIiySJU0N9T5hIQ=="], - "ajv": ["ajv@8.18.0", "", { "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", "json-schema-traverse": "^1.0.0", "require-from-string": "^2.0.2" } }, "sha512-PlXPeEWMXMZ7sPYOHqmDyCJzcfNrUr3fGNKtezX14ykXOEIvyK81d+qydx89KY5O71FKMPaQ2vBfBFI5NHR63A=="], "ajv-formats": ["ajv-formats@3.0.1", "", { "dependencies": { "ajv": "^8.0.0" } }, "sha512-8iUql50EUR+uUcdRQ3HDqa6EVyo3docL8g5WJ3FNcWmu62IbkGUue/pEyLBW8VGKKucTPgqeks4fIU1DA4yowQ=="], - "asynckit": ["asynckit@0.4.0", "", {}, "sha512-Oei9OH4tRh0YqU3GxhX79dM/mwVgvbZJaSNaRk+bshkj0S5cfHcgYakreBjrHwatXKbz+IoIdYLxrKim2MjW0Q=="], - "body-parser": ["body-parser@2.2.2", "", { "dependencies": { "bytes": "^3.1.2", "content-type": "^1.0.5", "debug": "^4.4.3", "http-errors": "^2.0.0", "iconv-lite": "^0.7.0", "on-finished": "^2.4.1", "qs": "^6.14.1", "raw-body": "^3.0.1", "type-is": "^2.0.1" } }, "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA=="], "bun-types": ["bun-types@1.3.11", "", { "dependencies": { "@types/node": "*" } }, "sha512-1KGPpoxQWl9f6wcZh57LvrPIInQMn2TQ7jsgxqpRzg+l0QPOFvJVH7HmvHo/AiPgwXy+/Thf6Ov3EdVn1vOabg=="], @@ -88,8 +66,6 @@ "call-bound": ["call-bound@1.0.4", "", { "dependencies": { "call-bind-apply-helpers": "^1.0.2", "get-intrinsic": "^1.3.0" } }, "sha512-+ys997U96po4Kx/ABpBCqhA9EuxJaQWDQg7295H4hBphv3IZg0boBKuwYpt4YXp6MZ5AmZQnU/tyMTlRpaSejg=="], - "combined-stream": ["combined-stream@1.0.8", "", { "dependencies": { "delayed-stream": "~1.0.0" } }, "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg=="], - "commander": ["commander@14.0.3", "", {}, "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw=="], "content-disposition": ["content-disposition@1.0.1", "", {}, "sha512-oIXISMynqSqm241k6kcQ5UwttDILMK4BiurCfGEREw6+X9jkkpEe5T9FZaApyLGGOnFuyMWZpdolTXMtvEJ08Q=="], @@ -106,8 +82,6 @@ "debug": ["debug@4.4.3", "", { "dependencies": { "ms": "^2.1.3" } }, "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA=="], - "delayed-stream": ["delayed-stream@1.0.0", "", {}, "sha512-ZySD7Nf91aLB0RxL4KGrKHBXl7Eds1DAmEdcoVawXnLD7SDhpNgtuII2aAkg7a7QS41jxPSZ17p4VdGnMHk3MQ=="], - "depd": ["depd@2.0.0", "", {}, "sha512-g7nH6P6dyDioJogAAGprGpCtVImJhpPk/roCzdb3fIh61/s/nPsfR6onyMwkCAR/OlC3yBC0lESvUoQEAssIrw=="], "dunder-proto": ["dunder-proto@1.0.1", "", { "dependencies": { "call-bind-apply-helpers": "^1.0.1", "es-errors": "^1.3.0", "gopd": "^1.2.0" } }, "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A=="], @@ -122,14 +96,10 @@ "es-object-atoms": ["es-object-atoms@1.1.1", "", { "dependencies": { "es-errors": "^1.3.0" } }, "sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA=="], - "es-set-tostringtag": ["es-set-tostringtag@2.1.0", "", { "dependencies": { "es-errors": "^1.3.0", "get-intrinsic": "^1.2.6", "has-tostringtag": "^1.0.2", "hasown": "^2.0.2" } }, "sha512-j6vWzfrGVfyXxge+O0x5sh6cvxAog0a/4Rdd2K36zCMV5eJ+/+tOAngRO8cODMNWbVRdVlmGZQL2YS3yR8bIUA=="], - "escape-html": ["escape-html@1.0.3", "", {}, "sha512-NiSupZ4OeuGwr68lGIeym/ksIZMJodUGOSCZ/FSnTxcrekbvqrgdUxlJOMpijaKZVjAJrWrGs/6Jy8OMuyj9ow=="], "etag": ["etag@1.8.1", "", {}, "sha512-aIL5Fx7mawVa300al2BnEE4iNvo1qETxLrPI/o05L7z6go7fCw1J6EQmbK4FmJ2AS7kgVF/KEZWufBfdClMcPg=="], - "event-target-shim": ["event-target-shim@5.0.1", "", {}, "sha512-i/2XbnSz/uxRCU6+NdVJgKWDTM427+MqYbkQzD321DuCQJUqOuJKIA0IM2+W2xtYHdKOmZ4dR6fExsd4SXL+WQ=="], - "eventsource": ["eventsource@3.0.7", "", { "dependencies": { "eventsource-parser": "^3.0.1" } }, "sha512-CRT1WTyuQoD771GW56XEZFQ/ZoSfWid1alKGDYMmkt2yl8UXrVR4pspqWNEcqKvVIzg6PAltWjxcSSPrboA4iA=="], "eventsource-parser": ["eventsource-parser@3.0.6", "", {}, "sha512-Vo1ab+QXPzZ4tCa8SwIHJFaSzy4R6SHf7BY79rFBDf0idraZWAkYrDjDj8uWaSm3S2TK+hJ7/t1CEmZ7jXw+pg=="], @@ -140,16 +110,12 @@ "fast-deep-equal": ["fast-deep-equal@3.1.3", "", {}, "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q=="], + "fast-sha256": ["fast-sha256@1.3.0", "", {}, "sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ=="], + "fast-uri": ["fast-uri@3.1.0", "", {}, "sha512-iPeeDKJSWf4IEOasVVrknXpaBV0IApz/gp7S2bb7Z4Lljbl2MGJRqInZiUrQwV16cpzw/D3S5j5Julj/gT52AA=="], "finalhandler": ["finalhandler@2.1.1", "", { "dependencies": { "debug": "^4.4.0", "encodeurl": "^2.0.0", "escape-html": "^1.0.3", "on-finished": "^2.4.1", "parseurl": "^1.3.3", "statuses": "^2.0.1" } }, "sha512-S8KoZgRZN+a5rNwqTxlZZePjT/4cnm0ROV70LedRHZ0p8u9fRID0hJUZQpkKLzro8LfmC8sx23bY6tVNxv8pQA=="], - "form-data": ["form-data@4.0.5", "", { "dependencies": { "asynckit": "^0.4.0", "combined-stream": "^1.0.8", "es-set-tostringtag": "^2.1.0", "hasown": "^2.0.2", "mime-types": "^2.1.12" } }, "sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w=="], - - "form-data-encoder": ["form-data-encoder@1.7.2", "", {}, "sha512-qfqtYan3rxrnCk1VYaA4H+Ms9xdpPqvLZa6xmMgFvhO32x7/3J/ExcTd6qpxM0vH2GdMI+poehyBZvqfMTto8A=="], - - "formdata-node": ["formdata-node@4.4.1", "", { "dependencies": { "node-domexception": "1.0.0", "web-streams-polyfill": "4.0.0-beta.3" } }, "sha512-0iirZp3uVDjVGt9p49aTaqjk84TrglENEDuqfdlZQ1roC9CWlPk6Avf8EEnZNcAqPonwkG35x4n3ww/1THYAeQ=="], - "forwarded": ["forwarded@0.2.0", "", {}, "sha512-buRG0fpBtRHSTCOASe6hD258tEubFoRLb4ZNA6NxMVHNw2gOcwHo9wyablzMzOA5z9xA9L1KNjk/Nt6MT9aYow=="], "fresh": ["fresh@2.0.0", "", {}, "sha512-Rx/WycZ60HOaqLKAi6cHRKKI7zxWbJ31MhntmtwMoaTeF7XFH9hhBp8vITaMidfljRQ6eYWCKkaTK+ykVJHP2A=="], @@ -164,16 +130,12 @@ "has-symbols": ["has-symbols@1.1.0", "", {}, "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ=="], - "has-tostringtag": ["has-tostringtag@1.0.2", "", { "dependencies": { "has-symbols": "^1.0.3" } }, "sha512-NqADB8VjPFLM2V0VvHUewwwsw0ZWBaIdgo+ieHtK3hasLz4qeCRjYcqfB6AQrBggRKppKF8L52/VqdVsO47Dlw=="], - "hasown": ["hasown@2.0.2", "", { "dependencies": { "function-bind": "^1.1.2" } }, "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ=="], "hono": ["hono@4.12.9", "", {}, "sha512-wy3T8Zm2bsEvxKZM5w21VdHDDcwVS1yUFFY6i8UobSsKfFceT7TOwhbhfKsDyx7tYQlmRM5FLpIuYvNFyjctiA=="], "http-errors": ["http-errors@2.0.1", "", { "dependencies": { "depd": "~2.0.0", "inherits": "~2.0.4", "setprototypeof": "~1.2.0", "statuses": "~2.0.2", "toidentifier": "~1.0.1" } }, "sha512-4FbRdAX+bSdmo4AUFuS0WNiPz8NgFt+r8ThgNWmlrjQjt1Q7ZR9+zTlce2859x4KSXrwIsaeTqDoKQmtP8pLmQ=="], - "humanize-ms": ["humanize-ms@1.2.1", "", { "dependencies": { "ms": "^2.0.0" } }, "sha512-Fl70vYtsAFb/C06PTS9dZBo7ihau+Tu/DNCk/OyHhea07S+aeMWpFFkUaXRa8fI+ScZbEI8dfSxwY7gxZ9SAVQ=="], - "iconv-lite": ["iconv-lite@0.7.2", "", { "dependencies": { "safer-buffer": ">= 2.1.2 < 3.0.0" } }, "sha512-im9DjEDQ55s9fL4EYzOAv0yMqmMBSZp6G0VvFyTMPKWxiSBHUj9NW/qqLmXUwXrrM7AvqSlTCfvqRb0cM8yYqw=="], "inherits": ["inherits@2.0.4", "", {}, "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ=="], @@ -208,10 +170,6 @@ "negotiator": ["negotiator@1.0.0", "", {}, "sha512-8Ofs/AUQh8MaEcrlq5xOX0CQ9ypTF5dl78mjlMNfOK08fzpgTHQRQPBxcPlEtIw0yRpws+Zo/3r+5WRby7u3Gg=="], - "node-domexception": ["node-domexception@1.0.0", "", {}, "sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ=="], - - "node-fetch": ["node-fetch@2.7.0", "", { "dependencies": { "whatwg-url": "^5.0.0" }, "peerDependencies": { "encoding": "^0.1.0" }, "optionalPeers": ["encoding"] }, "sha512-c4FRfUm/dbcWZ7U+1Wq0AwCyFL+3nt2bEw05wfxSz+DWpWsitgmSgYmy2dQdWyKC1694ELPqMs/YzUSNozLt8A=="], - "object-assign": ["object-assign@4.1.1", "", {}, "sha512-rJgTQnkUnH1sFw8yT6VSU3zD3sWmu6sZhIseY8VX+GRu3P6F7Fu+JNDoXfklElbLJSnc3FUQHVe4cU5hj+BcUg=="], "object-inspect": ["object-inspect@1.13.4", "", {}, "sha512-W67iLl4J2EXEGTbfeHCffrjDfitvLANg0UlX3wFUUSTx92KXRFegMHUVgSqE+wvhAbi4WqjGg9czysTV2Epbew=="], @@ -262,30 +220,24 @@ "side-channel-weakmap": ["side-channel-weakmap@1.0.2", "", { "dependencies": { "call-bound": "^1.0.2", "es-errors": "^1.3.0", "get-intrinsic": "^1.2.5", "object-inspect": "^1.13.3", "side-channel-map": "^1.0.1" } }, "sha512-WPS/HvHQTYnHisLo9McqBHOJk2FkHO/tlpvldyrnem4aeQp4hai3gythswg6p01oSoTl58rcpiFAjF2br2Ak2A=="], + "standardwebhooks": ["standardwebhooks@1.1.1", "", { "dependencies": { "@stablelib/base64": "^1.0.0", "fast-sha256": "^1.3.0" } }, "sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ=="], + "statuses": ["statuses@2.0.2", "", {}, "sha512-DvEy55V3DB7uknRo+4iOGT5fP1slR8wQohVdknigZPMpMstaKJQWhwiYBACJE3Ul2pTnATihhBYnRhZQHGBiRw=="], "toidentifier": ["toidentifier@1.0.1", "", {}, "sha512-o5sSPKEkg/DIQNmH43V0/uerLrpzVedkUh8tGNvaeXpfpuwjKenlSox/2O/BTlZUtEe+JG7s5YhEz608PlAHRA=="], - "tr46": ["tr46@0.0.3", "", {}, "sha512-N3WMsuqV66lT30CrXNbEjx4GEwlow3v6rr4mCcv6prnfwhS01rkgyFdjPNBYd9br7LpXV1+Emh01fHnq2Gdgrw=="], - "ts-algebra": ["ts-algebra@2.0.0", "", {}, "sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw=="], "type-is": ["type-is@2.0.1", "", { "dependencies": { "content-type": "^1.0.5", "media-typer": "^1.1.0", "mime-types": "^3.0.0" } }, "sha512-OZs6gsjF4vMp32qrCbiVSkrFmXtG/AZhY3t0iAMrMBiAZyV9oALtXO8hsrHbMXF9x6L3grlFuwW2oAz7cav+Gw=="], "typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="], - "undici-types": ["undici-types@5.26.5", "", {}, "sha512-JlCMO+ehdEIKqlFxk6IfVoAUVmgz7cU7zD/h9XZ0qzeosSHmUJVOzSQvvYSYWXkFXC+IfLKSIffhv0sVZup6pA=="], + "undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="], "unpipe": ["unpipe@1.0.0", "", {}, "sha512-pjy2bYhSsufwWlKwPc+l3cN7+wuJlK6uz0YdJEOlQDbl6jo/YlPi4mb8agUkVC8BF7V8NuzeyPNqRksA3hztKQ=="], "vary": ["vary@1.1.2", "", {}, "sha512-BNGbWLfd0eUPabhkXUVm0j8uuvREyTh5ovRa/dyow/BqAbZJyC+5fU+IzQOzmAKzYqYRAISoRhdQr3eIZ/PXqg=="], - "web-streams-polyfill": ["web-streams-polyfill@4.0.0-beta.3", "", {}, "sha512-QW95TCTaHmsYfHDybGMwO5IJIM93I/6vTRk+daHTWFPhwh+C8Cg7j7XyKrwrj8Ib6vYXe0ocYNrmzY4xAAN6ug=="], - - "webidl-conversions": ["webidl-conversions@3.0.1", "", {}, "sha512-2JAn3z8AR6rjK8Sm8orRC0h/bcl/DqL7tRPdGZ4I1CjdF+EaMLmYxBHyXuKL849eucPFhvBoxMsflfOb8kxaeQ=="], - - "whatwg-url": ["whatwg-url@5.0.0", "", { "dependencies": { "tr46": "~0.0.3", "webidl-conversions": "^3.0.0" } }, "sha512-saE57nupxk6v3HY35+jzBwYa0rKSy0XR8JSxZPwgLr7ys0IBzhGviA1/TUGJLmSVqs8pb9AnvICXEuOHLprYTw=="], - "which": ["which@2.0.2", "", { "dependencies": { "isexe": "^2.0.0" }, "bin": { "node-which": "./bin/node-which" } }, "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA=="], "wrappy": ["wrappy@1.0.2", "", {}, "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ=="], @@ -295,19 +247,5 @@ "zod": ["zod@4.3.6", "", {}, "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg=="], "zod-to-json-schema": ["zod-to-json-schema@3.25.2", "", { "peerDependencies": { "zod": "^3.25.28 || ^4" } }, "sha512-O/PgfnpT1xKSDeQYSCfRI5Gy3hPf91mKVDuYLUHZJMiDFptvP41MSnWofm8dnCm0256ZNfZIM7DSzuSMAFnjHA=="], - - "@anthropic-ai/claude-agent-sdk/@anthropic-ai/sdk": ["@anthropic-ai/sdk@0.74.0", "", { "dependencies": { "json-schema-to-ts": "^3.1.1" }, "peerDependencies": { "zod": "^3.25.0 || ^4.0.0" }, "optionalPeers": ["zod"], "bin": { "anthropic-ai-sdk": "bin/cli" } }, "sha512-srbJV7JKsc5cQ6eVuFzjZO7UR3xEPJqPamHFIe29bs38Ij2IripoAhC0S5NslNbaFUYqBKypmmpzMTpqfHEUDw=="], - - "@types/node-fetch/@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="], - - "bun-types/@types/node": ["@types/node@25.5.0", "", { "dependencies": { "undici-types": "~7.18.0" } }, "sha512-jp2P3tQMSxWugkCUKLRPVUpGaL5MVFwF8RDuSRztfwgN1wmqJeMSbKlnEtQqU8UrhTmzEmZdu2I6v2dpp7XIxw=="], - - "form-data/mime-types": ["mime-types@2.1.35", "", { "dependencies": { "mime-db": "1.52.0" } }, "sha512-ZDY+bPm5zTTF+YpCrAU9nK0UgICYPT0QtT1NZWFv4s++TNkcgVaT0g6+4R2uI4MjQjzysHB1zxuWL50hzaeXiw=="], - - "@types/node-fetch/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="], - - "bun-types/@types/node/undici-types": ["undici-types@7.18.2", "", {}, "sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w=="], - - "form-data/mime-types/mime-db": ["mime-db@1.52.0", "", {}, "sha512-sPU4uV7dYlvtWJxwwxHD0PuihVNiE7TyAbQ5SWxDCB9mUYvOgroQOwYQQOKPJ8CIbE+1ETVlOoK1UC2nU3gYvg=="], } } diff --git a/ai_evals/cli/index.ts b/ai_evals/cli/index.ts index 504b8a8d13..8a5f63c8b2 100644 --- a/ai_evals/cli/index.ts +++ b/ai_evals/cli/index.ts @@ -108,6 +108,10 @@ async function main() { "--record", "append a compact summary line to ai_evals/history/.jsonl", ) + .option( + "--reasoning ", + "reasoning effort for frontend modes (e.g. off, low, medium, high, max); default: the product's", + ) .option( "--backend-validation ", `backend smoke validation (${BACKEND_VALIDATION_MODES.join(", ")})`, @@ -126,6 +130,7 @@ async function main() { executionOnly?: boolean; record?: boolean; backendValidation?: string; + reasoning?: string; }, ) => { await handleRun({ @@ -140,6 +145,7 @@ async function main() { executionOnly: options.executionOnly ?? false, record: options.record ?? false, backendValidation: options.backendValidation, + reasoning: options.reasoning, }); }, ); @@ -190,6 +196,7 @@ async function handleRun(input: { executionOnly: boolean; record: boolean; backendValidation?: string; + reasoning?: string; }) { if (input.record && input.caseIds.length > 0) { throw new Error( @@ -217,6 +224,13 @@ async function handleRun(input: { "--backend-validation currently supports only flow and script modes", ); } + if (input.reasoning) { + if (input.mode === "cli") { + throw new Error("--reasoning only applies to frontend modes"); + } + // The frontend runtime runs in a child process, which inherits it. + process.env.WMILL_AI_EVAL_REASONING = input.reasoning; + } if (input.mode !== "cli") { await assertWindmillBackendReachable(resolveWindmillBackendSettings()); } @@ -257,6 +271,10 @@ async function handleRun(input: { backendValidation, }); + if (input.reasoning && result.runModel) { + result.runModel = `${result.runModel}@${input.reasoning}`; + } + const resolvedOutputPath = models.length === 1 ? resolveRunOutputPath(input.mode, input.outputPath) diff --git a/ai_evals/core/judge.ts b/ai_evals/core/judge.ts index 5e67510027..f1df60f286 100644 --- a/ai_evals/core/judge.ts +++ b/ai_evals/core/judge.ts @@ -1,7 +1,7 @@ import Anthropic from "@anthropic-ai/sdk"; import type { EvalMode, JudgeResult } from "./types"; -export const DEFAULT_JUDGE_MODEL = "claude-sonnet-4-6"; +export const DEFAULT_JUDGE_MODEL = "claude-sonnet-5-5"; const JUDGE_TOOL_NAME = "submit_judgement"; @@ -29,6 +29,7 @@ export async function judgeOutput(input: { const system = [ "You evaluate benchmark outputs for Windmill AI generation.", + `Always answer by calling the ${JUDGE_TOOL_NAME} tool.`, "Deterministic checks already run separately. Focus on whether the final output satisfies the user request.", "If expected state is provided, treat it as a valid example and reward semantically equivalent outputs.", "If a checklist is provided, treat it as the explicit acceptance criteria for this case.", @@ -69,8 +70,8 @@ export async function judgeOutput(input: { try { const response = await client.messages.create({ model, - max_tokens: 1024, - temperature: 0, + // The judge thinks by default, and thinking shares this budget with the verdict. + max_tokens: 16000, system, messages: [{ role: "user", content: user }], tools: [ @@ -93,11 +94,8 @@ export async function judgeOutput(input: { }, }, ], - tool_choice: { - type: "tool", - name: JUDGE_TOOL_NAME, - disable_parallel_tool_use: true, - }, + // Current models refuse a forced tool_choice ("tool"/"any"). + tool_choice: { type: "auto", disable_parallel_tool_use: true }, }); const toolUseBlock = response.content.find( diff --git a/ai_evals/core/models.ts b/ai_evals/core/models.ts index 295cd36135..1580cf5b0b 100644 --- a/ai_evals/core/models.ts +++ b/ai_evals/core/models.ts @@ -78,6 +78,32 @@ export const EVAL_MODELS: EvalModelSpec[] = [ model: "opus", }, }, + { + id: "sonnet-5.5", + label: "Claude Sonnet 5.5", + aliases: ["sonnet-5.5", "claude-sonnet-5.5", "claude-sonnet-5-5"], + frontend: { + provider: "anthropic", + model: "claude-sonnet-5-5", + }, + cli: { + provider: "anthropic", + model: "claude-sonnet-5-5", + }, + }, + { + id: "opus-5.5", + label: "Claude Opus 5.5", + aliases: ["opus-5.5", "claude-opus-5.5", "claude-opus-5-5"], + frontend: { + provider: "anthropic", + model: "claude-opus-5-5", + }, + cli: { + provider: "anthropic", + model: "claude-opus-5-5", + }, + }, { id: "4o", label: "GPT-4o", @@ -96,6 +122,51 @@ export const EVAL_MODELS: EvalModelSpec[] = [ model: "gpt-5.5", }, }, + { + id: "gpt-5.6-sol", + label: "GPT-5.6 Sol", + aliases: ["gpt-5.6-sol"], + frontend: { + provider: "openai", + model: "gpt-5.6-sol", + }, + }, + { + id: "gpt-6-astra", + label: "GPT-6 Astra", + aliases: ["gpt-6-astra", "gpt-6"], + frontend: { + provider: "openai", + model: "gpt-6-astra", + }, + }, + { + id: "gpt-6-sol", + label: "GPT-6 Sol", + aliases: ["gpt-6-sol"], + frontend: { + provider: "openai", + model: "gpt-6-sol", + }, + }, + { + id: "gpt-6-luna", + label: "GPT-6 Luna", + aliases: ["gpt-6-luna"], + frontend: { + provider: "openai", + model: "gpt-6-luna", + }, + }, + { + id: "gemini-3.8-flash", + label: "Gemini 3.8 Flash", + aliases: ["gemini-3.8-flash"], + frontend: { + provider: "googleai", + model: "gemini-3.8-flash", + }, + }, { id: "gemini-3-flash-preview", label: "Gemini 3 Flash Preview", diff --git a/ai_evals/package.json b/ai_evals/package.json index 6720c910cb..424e49ffae 100644 --- a/ai_evals/package.json +++ b/ai_evals/package.json @@ -8,8 +8,8 @@ "test:frontend-graph": "cd ../frontend && node_modules/.bin/vitest run --project server --config ../ai_evals/adapters/frontend/vitest.unit.config.ts" }, "dependencies": { - "@anthropic-ai/claude-agent-sdk": "^0.2.25", - "@anthropic-ai/sdk": "^0.39.0", + "@anthropic-ai/claude-agent-sdk": "^0.3.284", + "@anthropic-ai/sdk": "^0.129.0", "commander": "^14.0.3", "openai": "^6.9.1", "yaml": "^2.8.3" diff --git a/backend/windmill-ai/src/providers/mod.rs b/backend/windmill-ai/src/providers/mod.rs index 908f007978..b5b6281264 100644 --- a/backend/windmill-ai/src/providers/mod.rs +++ b/backend/windmill-ai/src/providers/mod.rs @@ -14,6 +14,130 @@ use std::time::{Duration, Instant}; /// its `thinking` param, Gemini to a zero budget or the model's floor). pub(crate) const REASONING_OFF_SENTINEL: &str = "none"; +/// The effort to send for a model, dropping the off sentinel on a model that rejects every +/// disable: the model then reasons at its default instead of failing the request. The UI +/// never offers off on these, so this guards an agent step saved before it stopped, or an +/// effort passed in as a flow input. +pub fn effective_reasoning_effort<'a>(model: &str, effort: Option<&'a str>) -> Option<&'a str> { + match effort { + Some(effort) + if effort == REASONING_OFF_SENTINEL + && reasoning_rule(model).is_some_and(|rule| !rule.can_disable) => + { + None + } + effort => effort, + } +} + +/// Whether a request with function tools must send the off sentinel on Chat Completions: +/// the model refuses tools there while it reasons, and does accept being turned off. +pub(crate) fn completions_tools_need_reasoning_off(model: &str) -> bool { + reasoning_rule(model).is_some_and(|rule| rule.completions_tools_need_off && rule.can_disable) +} + +/// What the backend needs to know about a model's reasoning: the rows of `REASONING_RULES` +/// in the frontend's `reasoningRegistry.ts`, cut down to the two facts the wire needs. +/// `reasoningParity.json` next to that file is checked by both sides' tests. +struct ReasoningRule { + matches: fn(&str) -> bool, + /// False when the provider rejects every disable, so the off sentinel must not be sent. + can_disable: bool, + /// Chat Completions refuses function tools while the model reasons, even with the + /// effort omitted (live-verified); the Responses API has no such limit. + completions_tools_need_off: bool, +} + +/// Matched in order against the lowercased model id, first match wins. A model no row +/// matches keeps whatever effort it was given. +const REASONING_RULES: &[ReasoningRule] = &[ + // Live-verified: Fable, Mythos and the 5.x point releases reject `thinking: disabled`. + ReasoningRule { + matches: |m| m.contains("claude-fable") || m.contains("claude-mythos"), + can_disable: false, + completions_tools_need_off: false, + }, + ReasoningRule { + matches: is_claude_5_point_release, + can_disable: false, + completions_tools_need_off: false, + }, + // Live-verified: astra takes low..max only, where sol and luna also take `none`. + ReasoningRule { + matches: |m| base_id(m).starts_with("gpt-6-astra"), + can_disable: false, + completions_tools_need_off: true, + }, + ReasoningRule { + matches: |m| gpt_version(m).is_some_and(|(major, _)| major >= 6), + can_disable: true, + completions_tools_need_off: true, + }, + ReasoningRule { + matches: |m| matches!(gpt_version(m), Some((5, Some(minor))) if minor >= 5), + can_disable: true, + completions_tools_need_off: true, + }, + ReasoningRule { + matches: |m| matches!(gpt_version(m), Some((5, Some(_)))), + can_disable: true, + completions_tools_need_off: false, + }, + // gpt-5 and the o-series reject `none`. + ReasoningRule { + matches: |m| matches!(gpt_version(m), Some((5, None))), + can_disable: false, + completions_tools_need_off: false, + }, + ReasoningRule { + matches: |m| { + let base = base_id(m); + base.starts_with('o') && base[1..].starts_with(|c: char| c.is_ascii_digit()) + }, + can_disable: false, + completions_tools_need_off: false, + }, +]; + +fn reasoning_rule(model: &str) -> Option<&'static ReasoningRule> { + let model = model.to_lowercase(); + REASONING_RULES.iter().find(|rule| (rule.matches)(&model)) +} + +/// The id after a gateway's `vendor/` and before a `:variant`. +fn base_id(model: &str) -> &str { + let last = model.rsplit('/').next().unwrap_or(model); + last.split(':').next().unwrap_or(last) +} + +/// `(major, minor)` of a `gpt-` id. The major is one digit then a separator or the end, +/// since Azure names gpt-3.5 `gpt-35-turbo`. +fn gpt_version(model: &str) -> Option<(u32, Option)> { + let rest = base_id(model).strip_prefix("gpt-")?; + let mut chars = rest.chars(); + let major = chars.next()?.to_digit(10)?; + match chars.next() { + None | Some('-') => Some((major, None)), + Some('.') => { + let minor: String = chars.take_while(char::is_ascii_digit).collect(); + Some((major, minor.parse().ok())) + } + _ => None, + } +} + +/// Sonnet or Opus 5.x with x >= 1. The version match stops at one digit so a dated id +/// (`claude-sonnet-5-20260101`) stays Sonnet 5. +fn is_claude_5_point_release(model: &str) -> bool { + let model = model.replace('.', "-"); + ["claude-opus-5-", "claude-sonnet-5-"].iter().any(|prefix| { + model.split(prefix).skip(1).any(|rest| { + let mut chars = rest.chars(); + matches!(chars.next(), Some('1'..='9')) && !matches!(chars.next(), Some('0'..='9')) + }) + }) +} + /// Whether a Claude model removed the sampling params (`temperature`, `top_p`, /// `top_k`). On these, any value is a hard 400 — `temperature is deprecated for /// this model` — whatever the thinking mode, so the param has to be dropped on @@ -119,6 +243,46 @@ pub fn remember_chat_completions_only(base_url: &str, model: &str) { CHAT_COMPLETIONS_ONLY.insert((base_url.to_string(), model.to_string()), Instant::now()); } +#[cfg(test)] +mod reasoning_rule_tests { + use super::*; + + /// The frontend registry's test reads the same file. These rules see the model id + /// alone, so it holds only rows whose answer doesn't depend on the provider. + #[test] + fn agrees_with_the_frontend_registry() { + let rows: Vec = serde_json::from_str(include_str!( + "../../../../frontend/src/lib/components/copilot/reasoningParity.json" + )) + .unwrap(); + for row in rows { + let model = row["model"].as_str().unwrap(); + let can_disable = row["canDisable"].as_bool().unwrap(); + let tools_need_off = row["completionsToolsNeedOff"].as_bool().unwrap(); + let sent = effective_reasoning_effort(model, Some(REASONING_OFF_SENTINEL)); + assert_eq!(sent.is_some(), can_disable, "{model}"); + assert_eq!( + effective_reasoning_effort(model, Some("low")), + Some("low"), + "{model}" + ); + assert_eq!( + completions_tools_need_reasoning_off(model), + tools_need_off && can_disable, + "{model}" + ); + } + } + + #[test] + fn reads_ids_the_frontend_resolves_first() { + // Azure's gpt-3.5 is not major 35, and a gateway prefix is not part of the id. + assert_eq!(gpt_version("gpt-35-turbo"), None); + assert!(completions_tools_need_reasoning_off("openai/gpt-6-sol")); + assert_eq!(effective_reasoning_effort("openai/o3", Some("none")), None); + } +} + #[cfg(test)] mod chat_completions_only_tests { use super::*; diff --git a/backend/windmill-ai/src/providers/other.rs b/backend/windmill-ai/src/providers/other.rs index b6040d8d79..1ea54a7e79 100644 --- a/backend/windmill-ai/src/providers/other.rs +++ b/backend/windmill-ai/src/providers/other.rs @@ -1,3 +1,4 @@ +use super::{completions_tools_need_reasoning_off, REASONING_OFF_SENTINEL}; use crate::{ ai_providers::AIProvider, image_handler::prepare_messages_for_api, @@ -143,6 +144,15 @@ impl OtherQueryBuilder { let (reasoning_effort, thinking, temperature) = provider_reasoning_fields(&self.provider_kind, args.reasoning_effort, args.temperature); + // With tools, gpt-5.5+ only run here with reasoning off: turn it off where the model + // can, rather than failing every turn. + let reasoning_effort = if args.tools.is_some_and(|tools| !tools.is_empty()) + && completions_tools_need_reasoning_off(args.model) + { + Some(REASONING_OFF_SENTINEL) + } else { + reasoning_effort + }; // Build request with stream_options for usage tracking let request_with_usage = OpenAICompletionRequest { diff --git a/backend/windmill-ai/src/types.rs b/backend/windmill-ai/src/types.rs index 03476240a3..7e2de09e01 100644 --- a/backend/windmill-ai/src/types.rs +++ b/backend/windmill-ai/src/types.rs @@ -294,7 +294,10 @@ impl ProviderWithResource { /// The reasoning effort to thread to the provider, treating an empty string /// (e.g. a cleared flow input) as unset. pub fn get_reasoning_effort(&self) -> Option<&str> { - self.reasoning_effort.as_deref().filter(|s| !s.is_empty()) + crate::providers::effective_reasoning_effort( + &self.model, + self.reasoning_effort.as_deref().filter(|s| !s.is_empty()), + ) } pub async fn get_base_url(&self, db: &DB) -> Result { diff --git a/frontend/src/lib/components/copilot/lib.ts b/frontend/src/lib/components/copilot/lib.ts index 655a872f96..ba718bf760 100644 --- a/frontend/src/lib/components/copilot/lib.ts +++ b/frontend/src/lib/components/copilot/lib.ts @@ -22,6 +22,8 @@ import { } from './modelConfig' import { applyReasoningToConfig, + completionsRejectsToolsWithReasoning, + explicitOffToken, requestsReasoning, stripLegacyThinkingSuffix, type ReasoningEffort @@ -69,6 +71,9 @@ interface AIProviderDetails { // the frontier model. The gpt-5 family is deprecated (retires 2026-12-11) but // still served, so it stays in the list below the 5.6 models. const OPENAI_MODELS = [ + 'gpt-6-sol', + 'gpt-6-astra', + 'gpt-6-luna', 'gpt-5.6-terra', 'gpt-5.6-sol', 'gpt-5.6-luna', @@ -87,7 +92,14 @@ export const AI_PROVIDERS: Record = { }, anthropic: { label: 'Anthropic', - defaultModels: ['claude-sonnet-5', 'claude-opus-5', 'claude-opus-4-8', 'claude-haiku-4-5'] + defaultModels: [ + 'claude-sonnet-5-5', + 'claude-opus-5-5', + 'claude-sonnet-5', + 'claude-opus-5', + 'claude-opus-4-8', + 'claude-haiku-4-5' + ] }, googleai: { label: 'Google AI', @@ -1160,6 +1172,12 @@ export async function getCompletion( // Use Completions API for other providers const client = options?.openaiClient ?? workspaceAIClients.getOpenaiClient() + // gpt-5.5+ refuse function tools here unless reasoning is off, so the Responses API + // fallback turns it off where the model can rather than failing the turn. + const reasoningEffort = + tools?.length && completionsRejectsToolsWithReasoning(provider, modelProvider.model) + ? (explicitOffToken(provider, modelProvider.model) ?? options?.reasoningEffort) + : options?.reasoningEffort const completionConfig = applyReasoningToConfig( config.stream && STREAM_USAGE_PROVIDERS.has(provider) ? { @@ -1175,7 +1193,7 @@ export async function getCompletion( } : config, provider === 'deepseek' ? 'deepseek' : provider === 'mistral' ? 'mistral' : 'completions', - options?.reasoningEffort + reasoningEffort ) const completion = client.chat.completions.create(completionConfig, { signal: abortController.signal, diff --git a/frontend/src/lib/components/copilot/modelConfig.test.ts b/frontend/src/lib/components/copilot/modelConfig.test.ts index 4f54eb223a..8bcdf70f54 100644 --- a/frontend/src/lib/components/copilot/modelConfig.test.ts +++ b/frontend/src/lib/components/copilot/modelConfig.test.ts @@ -21,6 +21,13 @@ describe('workspace context window overrides', () => { expect(getConfiguredModelContextWindow('openai', 'qwen-local', overrides)).toBeUndefined() expect(getEffectiveModelContextWindow('openai', 'qwen-local', overrides)).toBe(128_000) }) + + it('knows the gpt-6 and Claude 5.5 windows, so they do not compact at the assumed one', () => { + expect(getConfiguredModelContextWindow('openai', 'gpt-6-sol', undefined)).toBe(1_050_000) + expect(getConfiguredModelContextWindow('anthropic', 'claude-sonnet-5-5', undefined)).toBe( + 1_000_000 + ) + }) }) describe('usesAnthropicMessagesApi', () => { diff --git a/frontend/src/lib/components/copilot/modelConfig.ts b/frontend/src/lib/components/copilot/modelConfig.ts index 83e7771873..f9653bff25 100644 --- a/frontend/src/lib/components/copilot/modelConfig.ts +++ b/frontend/src/lib/components/copilot/modelConfig.ts @@ -58,7 +58,7 @@ export function usesOpenRouterPromptCaching(provider: AIProvider, model: string) // so it does not catch unrelated ids like Mistral's "open-mistral-*" or "optimus-*". export function requiresMaxCompletionTokens(model: string) { const baseModel = parseModelId(model).base - return baseModel.startsWith('gpt-5') || /^o\d/.test(baseModel) + return Number(/^gpt-(\d)(?:[.-]|$)/.exec(baseModel)?.[1] ?? 0) >= 5 || /^o\d/.test(baseModel) } // Context windows of the models we know, most specific entry first — the first @@ -87,6 +87,7 @@ const MODEL_CONTEXT_WINDOWS: [name: string, contextWindow: number][] = [ ['claude', 200_000], // OpenAI — gpt-5 covers the base family (-mini / -nano) and the 5.1/5.2 // revisions, all 400K; 5.4/5.5 moved to 1M and 5.6 to 1.05M + ['gpt-6', 1_050_000], ['gpt-5.6', 1_050_000], ['gpt-5.5', 1_000_000], ['gpt-5.4', 1_000_000], @@ -140,6 +141,7 @@ const MODEL_MAX_OUTPUT_TOKENS: [name: string, maxOutputTokens: number][] = [ ['claude-fable', 64_000], ['claude-mythos', 64_000], // OpenAI + ['gpt-6', 128_000], ['gpt-5', 128_000], ['gpt-4.1', 32_768], ['gpt-4o', 16_384], diff --git a/frontend/src/lib/components/copilot/modelPricing.test.ts b/frontend/src/lib/components/copilot/modelPricing.test.ts index e295682187..4eded135ee 100644 --- a/frontend/src/lib/components/copilot/modelPricing.test.ts +++ b/frontend/src/lib/components/copilot/modelPricing.test.ts @@ -35,6 +35,16 @@ describe('resolveModelPrice', () => { expect(resolveModelPrice('googleai', 'gemini-3.7-flash', undefined)).toBeUndefined() }) + it('prices Claude 5.5 flat and leaves the tiered gpt-6 unpriced', () => { + expect(resolveModelPrice('anthropic', 'claude-opus-5-5', undefined)?.price).toMatchObject({ + input: 4, + output: 20, + cacheRead: 0.2 + }) + expect(resolveModelPrice('anthropic', 'claude-sonnet-5-5', undefined)?.price.input).toBe(2) + expect(resolveModelPrice('openai', 'gpt-6-sol', undefined)).toBeUndefined() + }) + it('reports an unknown model as unpriced rather than guessing', () => { expect(resolveModelPrice('customai', 'some-in-house-model', undefined)).toBeUndefined() }) diff --git a/frontend/src/lib/components/copilot/modelPricing.ts b/frontend/src/lib/components/copilot/modelPricing.ts index 1e5b63247b..b191a0bcaa 100644 --- a/frontend/src/lib/components/copilot/modelPricing.ts +++ b/frontend/src/lib/components/copilot/modelPricing.ts @@ -69,6 +69,8 @@ const MODEL_PRICES: [name: string, price: PriceEntry | null][] = [ // fallback sits below the explicit entries rather than covering them. ['claude-fable-5', { input: 10, output: 50 }], ['claude-mythos-5', { input: 10, output: 50 }], + ['claude-opus-5-5', { input: 4, output: 20, cacheRead: 0.2 }], + ['claude-sonnet-5-5', { input: 2, output: 10, cacheRead: 0.2 }], ['claude-opus-5', { input: 5, output: 25 }], ['claude-opus-4-8', { input: 5, output: 25 }], ['claude-opus-4-7', { input: 5, output: 25 }], @@ -98,6 +100,9 @@ const MODEL_PRICES: [name: string, price: PriceEntry | null][] = [ // Revisions past gpt-5 are priced separately by OpenAI and are not tracked here. // The matcher's revision guard already keeps them off the family rate; these // entries stay so a revision the guard admits still resolves to no rate. + // gpt-6 bills the whole request at a higher rate above 272K input tokens, which a + // per-model rate cannot express. + ['gpt-6', null], ['gpt-5.6', null], ['gpt-5.5', null], ['gpt-5.4', null], diff --git a/frontend/src/lib/components/copilot/reasoningParity.json b/frontend/src/lib/components/copilot/reasoningParity.json new file mode 100644 index 0000000000..93b8ac7eb2 --- /dev/null +++ b/frontend/src/lib/components/copilot/reasoningParity.json @@ -0,0 +1,84 @@ +[ + { + "provider": "anthropic", + "model": "claude-sonnet-5-5", + "canDisable": false, + "completionsToolsNeedOff": false + }, + { + "provider": "anthropic", + "model": "claude-opus-5-5", + "canDisable": false, + "completionsToolsNeedOff": false + }, + { + "provider": "anthropic", + "model": "claude-fable-5", + "canDisable": false, + "completionsToolsNeedOff": false + }, + { + "provider": "aws_bedrock", + "model": "global.anthropic.claude-opus-5-5-v1:0", + "canDisable": false, + "completionsToolsNeedOff": false + }, + { + "provider": "anthropic", + "model": "claude-sonnet-5", + "canDisable": true, + "completionsToolsNeedOff": false + }, + { + "provider": "anthropic", + "model": "claude-opus-5-20260101", + "canDisable": true, + "completionsToolsNeedOff": false + }, + { + "provider": "anthropic", + "model": "claude-opus-4-8", + "canDisable": true, + "completionsToolsNeedOff": false + }, + { + "provider": "openai", + "model": "gpt-6-astra", + "canDisable": false, + "completionsToolsNeedOff": true + }, + { + "provider": "openai", + "model": "gpt-6-sol", + "canDisable": true, + "completionsToolsNeedOff": true + }, + { + "provider": "azure_openai", + "model": "gpt-6-luna", + "canDisable": true, + "completionsToolsNeedOff": true + }, + { + "provider": "openai", + "model": "gpt-5.6-sol", + "canDisable": true, + "completionsToolsNeedOff": true + }, + { "provider": "openai", "model": "gpt-5.5", "canDisable": true, "completionsToolsNeedOff": true }, + { "provider": "openai", "model": "gpt-5.7", "canDisable": true, "completionsToolsNeedOff": true }, + { + "provider": "openai", + "model": "gpt-5.1", + "canDisable": true, + "completionsToolsNeedOff": false + }, + { "provider": "openai", "model": "gpt-5", "canDisable": false, "completionsToolsNeedOff": false }, + { + "provider": "openai", + "model": "gpt-5-mini", + "canDisable": false, + "completionsToolsNeedOff": false + }, + { "provider": "openai", "model": "o3", "canDisable": false, "completionsToolsNeedOff": false } +] diff --git a/frontend/src/lib/components/copilot/reasoningRegistry.test.ts b/frontend/src/lib/components/copilot/reasoningRegistry.test.ts index faf6f60264..30543e6c2f 100644 --- a/frontend/src/lib/components/copilot/reasoningRegistry.test.ts +++ b/frontend/src/lib/components/copilot/reasoningRegistry.test.ts @@ -1,6 +1,8 @@ import { describe, expect, it } from 'vitest' import { applyReasoningToConfig, + completionsRejectsToolsWithReasoning, + explicitOffToken, getReasoningCapability, REASONING_OFF, resolveEffectiveReasoning, @@ -8,6 +10,8 @@ import { stripLegacyThinkingSuffix, supportsReasoning } from './reasoningRegistry' +import type { AIProvider } from '$lib/gen' +import parity from './reasoningParity.json' describe('stripLegacyThinkingSuffix', () => { it('removes the deprecated /thinking suffix', () => { @@ -261,6 +265,42 @@ describe('supportsReasoning (static registry)', () => { expect(getReasoningCapability('openrouter', 'x-ai/grok-4').canDisable).toBe(false) expect(getReasoningCapability('openrouter', 'deepseek/deepseek-r1').canDisable).toBe(false) }) + it('never sends a disable the 5.5 point releases and gpt-6-astra reject', () => { + // Live-verified: Claude 5.5 rejects `thinking: disabled`, gpt-6-astra rejects `none`. + for (const [provider, model] of [ + ['anthropic', 'claude-sonnet-5-5'], + ['anthropic', 'claude-opus-5-5'], + ['aws_bedrock', 'global.anthropic.claude-opus-5-5-v1:0'], + ['openai', 'gpt-6-astra'] + ] as const) { + expect(getReasoningCapability(provider, model).canDisable, model).toBe(false) + expect(explicitOffToken(provider, model), model).toBeUndefined() + } + expect(getReasoningCapability('openrouter', 'anthropic/claude-sonnet-5.5').canDisable).toBe( + false + ) + expect(explicitOffToken('openrouter', 'anthropic/claude-sonnet-5.5')).toBeUndefined() + // A dated Claude 5 id is not a point release. + expect(explicitOffToken('anthropic', 'claude-sonnet-5-20260101')).toBe('none') + expect(explicitOffToken('openai', 'gpt-6-sol')).toBe('none') + expect(getReasoningCapability('openai', 'gpt-6-luna')).toMatchObject({ + supported: true, + canDisable: true, + levels: ['low', 'medium', 'high', 'xhigh', 'max'] + }) + }) + it("reads Azure's gpt-35-turbo as gpt-3.5, not a gpt-5+ reasoning model", () => { + expect(supportsReasoning('azure_openai', 'gpt-35-turbo')).toBe(false) + expect(supportsReasoning('openai', 'gpt-35-turbo-16k')).toBe(false) + }) + it('finds the models that refuse function tools with reasoning on Chat Completions', () => { + for (const model of ['gpt-5.5', 'gpt-5.6-sol', 'gpt-6-astra']) { + expect(completionsRejectsToolsWithReasoning('openai', model), model).toBe(true) + } + for (const model of ['gpt-5', 'gpt-5.1', 'gpt-35-turbo', 'o3']) { + expect(completionsRejectsToolsWithReasoning('azure_openai', model), model).toBe(false) + } + }) it('forwards an explicit off as effort none through OpenRouter', () => { expect( resolveRequestReasoning({ @@ -353,6 +393,23 @@ describe('Azure AI Foundry reasoning follows the model family', () => { }) }) +describe('backend parity', () => { + // windmill-ai's `providers/mod.rs` test reads the same file. The backend rules see the + // model id alone, so the file holds only rows whose answer doesn't depend on the + // provider: a Bedrock- or Gemini-Pro-specific row belongs in the tests above. + it.each(parity)( + '$provider $model', + ({ provider, model, canDisable, completionsToolsNeedOff }) => { + const capability = getReasoningCapability(provider as AIProvider, model) + expect(capability.supported).toBe(true) + expect(capability.canDisable).toBe(canDisable) + expect(completionsRejectsToolsWithReasoning(provider as AIProvider, model)).toBe( + completionsToolsNeedOff + ) + } + ) +}) + describe('resolveEffectiveReasoning', () => { it('defaults capable models to high when unset', () => { expect(resolveEffectiveReasoning({ provider: 'anthropic', model: 'claude-sonnet-4-6' })).toBe( diff --git a/frontend/src/lib/components/copilot/reasoningRegistry.ts b/frontend/src/lib/components/copilot/reasoningRegistry.ts index 9915e9c42f..68934c22f3 100644 --- a/frontend/src/lib/components/copilot/reasoningRegistry.ts +++ b/frontend/src/lib/components/copilot/reasoningRegistry.ts @@ -1,5 +1,5 @@ import type { AIProvider, AIProviderModel } from '$lib/gen' -import { parseModelId, usesAnthropicMessagesApi } from './modelConfig' +import { usesAnthropicMessagesApi } from './modelConfig' /** * Reasoning effort is provider/model-specific. We never normalize a single @@ -33,11 +33,6 @@ export function stripLegacyThinkingSuffix(model: string): string { : model } -/** Bare model id without any provider/gateway prefix (e.g. OpenRouter's `openai/o3`). */ -function baseModelId(model: string): string { - return parseModelId(model).base -} - /** * Azure AI Foundry hosts multiple model families under one provider, so reasoning * support follows the underlying model rather than the provider: Claude deployments @@ -53,172 +48,218 @@ function reasoningProviderFamily(provider: AIProvider, model: string): AIProvide } /** - * Suggested effort levels per provider, sourced from each provider SDK's own - * vocabulary. + * Sentinel sent for the deepseek off case. It never reaches the wire as an + * effort: the 'deepseek' branch of `applyReasoningToConfig` translates it to + * the provider's `thinking: {type: "disabled"}` param (`reasoning_effort: + * "none"` is rejected by their API). */ -const PROVIDER_REASONING_LEVELS: Partial> = { - // DeepSeek accepts the full five-token vocabulary but only two levels are - // real: low/medium are server-mapped to high and xhigh to max — offering - // them would be a no-op knob. - deepseek: ['high', 'max'], - // Mistral's only effort token besides the 'none' disable is 'high' - // (anything else is rejected), so the knob is effectively on/off. - mistral: ['high'] -} +export const DEEPSEEK_OFF_SENTINEL: ReasoningEffort = 'none' /** - * OpenRouter validates effort against its own vocabulary - * (minimal..xhigh + none) and translates per underlying provider, so the - * real ladder depends on the model family: Anthropic gets all five as - * distinct budget ratios (minimal 10% .. xhigh 95% of max_tokens); Gemini - * maps to thinkingLevel with xhigh clamped to high (a no-op vs high); - * OpenAI gets the token passed through verbatim, so the per-model OpenAI - * scoping applies; DeepSeek server-maps low/medium to high and xhigh to max. + * Sentinel for the Anthropic off case. Like the DeepSeek one it never reaches + * the wire as an effort: the 'anthropic' branch of `applyReasoningToConfig` + * translates it to `thinking: {type: "disabled"}`, the only off that Opus and + * Sonnet 5 respect. */ -function openrouterReasoningLevels(model: string): ReasoningEffort[] { - const m = model.toLowerCase() - const base = baseModelId(model) - if (/claude-(opus|sonnet)-(4|5)/.test(m)) { - return ['minimal', 'low', 'medium', 'high', 'xhigh'] - } - if (m.includes('gemini-')) { - return geminiReasoningLevels(m) - } - if (base.startsWith('gpt-5') || /^o\d/.test(base)) { - return openaiReasoningLevels(base) - } - if (m.includes('deepseek-v4')) { - return ['high', 'xhigh'] - } - return ['low', 'medium', 'high'] -} +export const ANTHROPIC_OFF_SENTINEL: ReasoningEffort = 'none' /** - * OpenAI's effort vocabulary is model-dependent: `minimal` exists on gpt-5 but - * not on gpt-5.1+, `xhigh` arrived on gpt-5.5 and `max` on gpt-5.6; o-series - * take low/medium/high. An unsupported level is rejected, so scope the list to - * the model. (`none` is the disable token, handled by `explicitOffToken`.) + * What the registry knows about one group of models. The rows below are matched in + * order against the lowercased model id, first match wins, so a narrower row goes above + * the row it overrides. A model no row matches does not reason, as far as we know. */ -function openaiReasoningLevels(model: string): ReasoningEffort[] { - const base = baseModelId(model) - if (/^gpt-5\.6/.test(base)) { - return ['low', 'medium', 'high', 'xhigh', 'max'] - } - if (/^gpt-5\.5/.test(base)) { - return ['low', 'medium', 'high', 'xhigh'] - } - if (/^gpt-5\./.test(base)) { - return ['low', 'medium', 'high'] - } - if (/^gpt-5/.test(base)) { - return ['minimal', 'low', 'medium', 'high'] - } - return ['low', 'medium', 'high'] +type ReasoningRule = { + match: RegExp + /** The levels the UI offers. An unsupported level is a 400, so this is per model. */ + levels: readonly ReasoningEffort[] + /** + * Whether "off" really stops the model reasoning. When false the UI offers no off: + * the provider would reject it, or coerce it to the lowest level. + */ + canDisable: boolean + /** + * The effort to send for off on a model that reasons when the field is omitted. + * Unset where omission is already off. Always `'none'`: `requestsReasoning` reads + * that as off, and each wire format translates it (see `applyReasoningToConfig`). + */ + offToken?: ReasoningEffort + /** + * gpt-5.5 and later refuse function tools on Chat Completions while they reason, + * even with the effort omitted (live-verified); the Responses API has no such limit. + */ + completionsToolsNeedOff?: boolean } +const LOW_TO_HIGH = ['low', 'medium', 'high'] +const LOW_TO_XHIGH = ['low', 'medium', 'high', 'xhigh'] +const LOW_TO_MAX = ['low', 'medium', 'high', 'xhigh', 'max'] +const MINIMAL_TO_HIGH = ['minimal', 'low', 'medium', 'high'] + /** - * Gemini's level ladder is model-dependent: Gemini 3+ Flash / Flash-Lite accept - * `minimal`, while 3.x Pro does not (and cannot disable thinking). Gemini 2.5 - * uses numeric budgets — the proxy maps the three tiers to budget values, so - * `minimal` is not offered there. + * Claude models whose thinking cannot be turned off: an explicit disable 400s. Fable, + * Mythos, and the 5.x point releases (Sonnet 5.5, Opus 5.5), whose lowest setting is + * adaptive thinking at `low` (live-verified). The version match stops at one digit so a + * dated id (`claude-sonnet-5-20260101`) stays Sonnet 5. */ -function geminiReasoningLevels(model: string): ReasoningEffort[] { - const m = model.toLowerCase() - const isGemini3Plus = !m.includes('gemini-2.5') - if (isGemini3Plus && (m.includes('flash') || m.includes('lite'))) { - return ['minimal', 'low', 'medium', 'high'] - } - return ['low', 'medium', 'high'] +const CLAUDE_ALWAYS_THINKING = /fable|mythos|claude-(opus|sonnet)-5[-.][1-9](?!\d)/ + +// Anthropic ids are matched anywhere in the id: Bedrock prefixes them +// (`us.anthropic.claude-opus-4-6-v1`). Opus 4.5 and older reject adaptive thinking. +const ANTHROPIC_RULES: ReasoningRule[] = [ + { match: CLAUDE_ALWAYS_THINKING, levels: LOW_TO_MAX, canDisable: false }, + // The 5 family thinks when the field is absent, so off is an explicit disable. + { + match: /claude-(opus|sonnet)-5/, + levels: LOW_TO_MAX, + canDisable: true, + offToken: ANTHROPIC_OFF_SENTINEL + }, + // 4.6-4.8 only think when asked, so omission is already off. + { match: /claude-opus-4-[78]/, levels: LOW_TO_MAX, canDisable: true }, + { match: /claude-(opus|sonnet)-4-6/, levels: ['low', 'medium', 'high', 'max'], canDisable: true } +] + +const BEDROCK_RULES: ReasoningRule[] = [ + ANTHROPIC_RULES[0], + // AWS documents Sonnet 5 on Bedrock as always thinking, where the native API + // accepts a disable for it. + { match: /claude-sonnet-5/, levels: LOW_TO_MAX, canDisable: false }, + ...ANTHROPIC_RULES.slice(1) +] + +// Anchored at the start or after a gateway's `vendor/`, and the major is one digit: +// Azure names gpt-3.5 `gpt-35-turbo`. `minimal` exists on gpt-5 only, `xhigh` from +// gpt-5.5, `max` from gpt-5.6. +const OPENAI_RULES: ReasoningRule[] = [ + // Live-verified: astra takes low..max only, where sol and luna also take `none`. + { + match: /(?:^|\/)gpt-6-astra/, + levels: LOW_TO_MAX, + canDisable: false, + completionsToolsNeedOff: true + }, + { + match: /(?:^|\/)gpt-[6-9](?:[.:-]|$)/, + levels: LOW_TO_MAX, + canDisable: true, + offToken: 'none', + completionsToolsNeedOff: true + }, + { + match: /(?:^|\/)gpt-5\.6/, + levels: LOW_TO_MAX, + canDisable: true, + offToken: 'none', + completionsToolsNeedOff: true + }, + { + match: /(?:^|\/)gpt-5\.5/, + levels: LOW_TO_XHIGH, + canDisable: true, + offToken: 'none', + completionsToolsNeedOff: true + }, + // A later gpt-5 minor keeps the tools limit, which holds for every version from 5.5, + // but only the levels every gpt-5.x takes until it has a row of its own. + { + match: /(?:^|\/)gpt-5\.(?:[5-9]|\d{2,})/, + levels: LOW_TO_HIGH, + canDisable: true, + offToken: 'none', + completionsToolsNeedOff: true + }, + // gpt-5.1+ are off only through `none`: omitted, they reason at medium. + { match: /(?:^|\/)gpt-5\./, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' }, + // gpt-5 and the o-series reject `none` and reason when it is omitted. + { match: /(?:^|\/)gpt-5(?:[:-]|$)/, levels: MINIMAL_TO_HIGH, canDisable: false }, + { match: /(?:^|\/)o\d/, levels: LOW_TO_HIGH, canDisable: false } +] + +// Gemini 2.5/3 think by default; the backend proxy maps `none` to off on Flash, or to +// the floor on Pro, which enforces one (level `low` on 3.x, 128 tokens on 2.5). +// Gemini 3+ Flash / Flash-Lite accept `minimal`; 2.5 takes numeric budgets the proxy +// maps from three tiers. +const GEMINI_RULES: ReasoningRule[] = [ + { match: /gemini-2\.5.*pro/, levels: LOW_TO_HIGH, canDisable: false, offToken: 'none' }, + { match: /gemini-2\.5/, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' }, + { match: /gemini-3.*pro/, levels: LOW_TO_HIGH, canDisable: false, offToken: 'none' }, + { match: /gemini-3.*(flash|lite)/, levels: MINIMAL_TO_HIGH, canDisable: true, offToken: 'none' }, + { match: /gemini-3/, levels: LOW_TO_HIGH, canDisable: true, offToken: 'none' } +] + +const REASONING_RULES: Partial> = { + anthropic: ANTHROPIC_RULES, + aws_bedrock: BEDROCK_RULES, + openai: OPENAI_RULES, + azure_openai: OPENAI_RULES, + googleai: GEMINI_RULES, + deepseek: [ + // Every current API model takes reasoning_effort, but only two levels are real: + // low/medium are server-mapped to high, xhigh to max. The retired + // `deepseek-chat` alias means "non-thinking mode", so a saved selection on it + // must not silently become a thinking request. Off is a separate `thinking` param. + { + match: /(?:^|\/)deepseek(?!-chat(?::|$))/, + levels: ['high', 'max'], + canDisable: true, + offToken: DEEPSEEK_OFF_SENTINEL + } + ], + mistral: [ + // Only the ids verified to accept reasoning_effort (large, magistral, ministral + // and pinned versions reject it), whose only effort besides off is `high`. + { + match: /(?:^|\/)mistral-(?:(?:small|medium)-latest(?::|$)|medium-3[-.]5)/, + levels: ['high'], + canDisable: true + } + ], + // OpenRouter validates effort against its own vocabulary (minimal..xhigh + none) and + // translates it per underlying provider, which scopes the ladder: Anthropic gets all + // five as budget ratios, OpenAI gets the token verbatim, DeepSeek server-maps. `none` + // is its documented off, more reliable than omission; it can only disable a model + // whose upstream can. + openrouter: [ + { + match: /claude-(opus|sonnet)-5[-.][1-9](?!\d)/, + levels: ['minimal', ...LOW_TO_XHIGH], + canDisable: false + }, + { + match: /claude-(opus|sonnet)-(4|5)/, + levels: ['minimal', ...LOW_TO_XHIGH], + canDisable: true, + offToken: 'none' + }, + ...GEMINI_RULES, + ...OPENAI_RULES.map(({ completionsToolsNeedOff: _, ...rule }) => ({ + ...rule, + offToken: 'none' + })), + { match: /deepseek-v4/, levels: LOW_TO_XHIGH.slice(2), canDisable: true, offToken: 'none' }, + // deepseek-r1, grok-4 and :thinking variants reason unconditionally. + { + match: /deepseek-r|grok-4|:thinking/, + levels: LOW_TO_HIGH, + canDisable: false, + offToken: 'none' + } + ] } -/** - * Gemini Pro models cannot turn thinking off — the API enforces a floor - * (level `low` on 3.x Pro, a 128-token budget on 2.5 Pro), so an off option - * would silently mean "lowest". Flash / Flash-Lite can truly disable - * (budget 0 / level `minimal`). - */ -function geminiCanDisable(model: string): boolean { - return !model.toLowerCase().includes('pro') +/** The row that describes a model, and whether its provider family has rows at all. */ +function findReasoningRule( + provider: AIProvider, + model: string +): { rule: ReasoningRule | undefined; known: boolean } { + const rules = REASONING_RULES[reasoningProviderFamily(provider, model)] + const id = stripLegacyThinkingSuffix(model).toLowerCase() + return { rule: rules?.find((rule) => rule.match.test(id)), known: rules !== undefined } } -/** - * Anthropic's effort ladder is model-dependent: `xhigh` exists on Opus 4.7/4.8, - * the 5 family and Fable/Mythos; `max` also on Opus 4.6 and Sonnet 4.6. - * Offering an unsupported level would 400, so scope the list to the model. - */ -function anthropicReasoningLevels(model: string): ReasoningEffort[] { - const m = model.toLowerCase() - if ( - /claude-(opus|sonnet)-5/.test(m) || - /claude-opus-4-(7|8)/.test(m) || - m.includes('fable') || - m.includes('mythos') - ) { - return ['low', 'medium', 'high', 'xhigh', 'max'] - } - return ['low', 'medium', 'high', 'max'] -} - -/** Mistral writes both `mistral-medium-3.5` and `mistral-medium-3-5`. */ -function normalizeMistralId(model: string): string { - return baseModelId(model).replace(/\./g, '-') -} - -/** - * Conservative static predicate for whether a model accepts an effort knob. - * Kept tight to avoid 400s on models that reject reasoning params. - */ -function supportsReasoningStatic(provider: AIProvider, model: string): boolean { - const m = model.toLowerCase() - const base = baseModelId(model) - switch (reasoningProviderFamily(provider, model)) { - case 'anthropic': - // Bedrock serves the same Claude models under prefixed ids - // (e.g. `us.anthropic.claude-opus-4-6-v1`), so match on the full string. - case 'aws_bedrock': - // 4.6+ only: Opus 4.5 rejects adaptive thinking (and, on Bedrock, - // the whole output_config surface) — live-verified hard 400. - return ( - /claude-opus-(4-(6|7|8)|5)/.test(m) || - /claude-sonnet-(4-6|5)/.test(m) || - m.includes('fable') || - m.includes('mythos') - ) - case 'openai': - case 'azure_openai': - return base.startsWith('gpt-5') || /^o\d/.test(base) - case 'openrouter': - // Best-effort markers for models whose `supported_parameters` include - // `reasoning` in OpenRouter's catalog; OpenRouter translates the effort - // per underlying provider. - return ( - base.startsWith('gpt-5') || - /^o\d/.test(base) || - /claude-(opus|sonnet)-(4|5)/.test(m) || - /gemini-(2\.5|3)/.test(m) || - m.includes('deepseek-r') || - m.includes('deepseek-v4') || - m.includes('grok-4') || - m.includes(':thinking') - ) - case 'googleai': - return /gemini-(2\.5|3)/.test(m) - case 'deepseek': - // All current API models take reasoning_effort (live-verified). The - // retired `deepseek-chat` alias stays excluded: its documented meaning - // is "non-thinking mode", so a saved selection on it must not silently - // become a thinking request. - return base.startsWith('deepseek') && base !== 'deepseek-chat' - case 'mistral': - // Only the ids verified to accept reasoning_effort; other models - // (large, magistral, ministral, pinned versions) reject the param. - return ( - /^mistral-(small|medium)-latest$/.test(base) || - normalizeMistralId(model).startsWith('mistral-medium-3-5') - ) - default: - return false - } +/** Whether Chat Completions needs reasoning off for this model to take function tools. */ +export function completionsRejectsToolsWithReasoning(provider: AIProvider, model: string): boolean { + return findReasoningRule(provider, model).rule?.completionsToolsNeedOff ?? false } export type ReasoningCapability = { @@ -241,89 +282,12 @@ export type ReasoningCapability = { known: boolean } -/** Provider families the registry has real rules for; everything else is a shrug. */ -const KNOWN_REASONING_FAMILIES: ReadonlySet = new Set([ - 'anthropic', - 'aws_bedrock', - 'openai', - 'azure_openai', - 'openrouter', - 'googleai', - 'deepseek', - 'mistral' -]) - /** Resolve the reasoning capability of a model from the static registry. */ export function getReasoningCapability(provider: AIProvider, model: string): ReasoningCapability { - const bareModel = stripLegacyThinkingSuffix(model) - const known = KNOWN_REASONING_FAMILIES.has(reasoningProviderFamily(provider, bareModel)) - const supported = supportsReasoningStatic(provider, bareModel) - if (!supported) { - return { supported: false, levels: [], canDisable: false, known } - } - const family = reasoningProviderFamily(provider, bareModel) - const levels = - family === 'anthropic' || family === 'aws_bedrock' - ? anthropicReasoningLevels(bareModel) - : family === 'googleai' - ? geminiReasoningLevels(bareModel) - : family === 'openai' || family === 'azure_openai' - ? openaiReasoningLevels(bareModel) - : family === 'openrouter' - ? openrouterReasoningLevels(bareModel) - : (PROVIDER_REASONING_LEVELS[family] ?? ['low', 'medium', 'high']) - return { supported, levels, canDisable: canDisableReasoning(provider, bareModel), known } -} - -/** - * Whether selecting "off" truly disables reasoning for the model. Off is - * sent either as an explicit provider disable (see `explicitOffToken`) or by - * omitting the effort — which only works where the model doesn't reason by - * default. - */ -function canDisableReasoning(provider: AIProvider, model: string): boolean { - const m = model.toLowerCase() - const base = baseModelId(model) - switch (reasoningProviderFamily(provider, model)) { - case 'anthropic': - // Every Claude but Fable and Mythos can stop thinking: 4.6-4.8 by - // omission, and the 5 family through the explicit disable that - // `explicitOffToken` sends. - return !ANTHROPIC_ALWAYS_THINKING.test(m) - case 'aws_bedrock': - // Same models, different answer: AWS documents Sonnet 5 on Bedrock as - // always thinking, where the native API accepts a disable for it. - return !(ANTHROPIC_ALWAYS_THINKING.test(m) || m.includes('claude-sonnet-5')) - case 'googleai': - return geminiCanDisable(model) - case 'openai': - case 'azure_openai': - // gpt-5.1+ accept effort 'none'; gpt-5 and o-series reject it and - // reason at `medium` by default, so omission isn't off either. - return /^gpt-5\./.test(base) - case 'openrouter': - // 'none' is in OpenRouter's vocabulary, but the gateway can't - // disable a model whose upstream can't — scope off per underlying - // family, like the levels. - // The 5 family thinks by default, but its upstream takes an explicit - // disable, so the gateway's 'none' has something to translate to. - if (/claude-(opus|sonnet)-(4|5)/.test(m)) { - return true - } - if (m.includes('gemini-')) { - return geminiCanDisable(m) - } - if (base.startsWith('gpt-5') || /^o\d/.test(base)) { - return /^gpt-5\./.test(base) - } - if (m.includes('deepseek-v4')) { - return true - } - // grok-4, deepseek-r1 and :thinking variants reason unconditionally. - return false - default: - return true - } + const { rule, known } = findReasoningRule(provider, model) + return rule + ? { supported: true, levels: [...rule.levels], canDisable: rule.canDisable, known } + : { supported: false, levels: [], canDisable: false, known } } export function supportsReasoning(provider: AIProvider, model: string): boolean { @@ -352,64 +316,14 @@ export function resolveEffectiveReasoning( : undefined } -/** - * Sentinel sent for the deepseek off case. It never reaches the wire as an - * effort: the 'deepseek' branch of `applyReasoningToConfig` translates it to - * the provider's `thinking: {type: "disabled"}` param (`reasoning_effort: - * "none"` is rejected by their API). - */ -export const DEEPSEEK_OFF_SENTINEL: ReasoningEffort = 'none' - -/** - * Sentinel for the Anthropic off case. Like the DeepSeek one it never reaches - * the wire as an effort: the 'anthropic' branch of `applyReasoningToConfig` - * translates it to `thinking: {type: "disabled"}`, which is the only off the - * always-on 5 family respects. - */ -export const ANTHROPIC_OFF_SENTINEL: ReasoningEffort = 'none' - -/** Claude models whose thinking cannot be turned off — an explicit disable 400s. */ -const ANTHROPIC_ALWAYS_THINKING = /fable|mythos/ - /** * Disable token to forward when the user explicitly turns reasoning off on a * model that reasons *by default* — omitting the field would silently keep - * the default-on behavior. Undefined means omission is the correct off. Every - * token is `'none'`: `requestsReasoning` reads that value as off. + * the default-on behavior. Undefined means omission is the correct off, or that + * the model cannot be turned off at all. */ export function explicitOffToken(provider: AIProvider, model: string): ReasoningEffort | undefined { - switch (reasoningProviderFamily(provider, model)) { - case 'anthropic': - // Claude 4.6-4.8 only think when asked, so omission is already a - // real off there and stays the wire form. Only the 5 family, which - // thinks when the field is absent, needs the explicit disable — - // Fable and Mythos reject it outright and get no off token at all. - return /claude-(opus|sonnet)-5/.test(model.toLowerCase()) ? ANTHROPIC_OFF_SENTINEL : undefined - case 'aws_bedrock': - // Bedrock's Sonnet 5 cannot be disabled at all, so only Opus 5 gets - // the sentinel; the rest keep omission. - return model.toLowerCase().includes('claude-opus-5') ? ANTHROPIC_OFF_SENTINEL : undefined - case 'googleai': - // Gemini 2.5/3 think by default (dynamic budget / level). The backend - // proxy maps 'none' to off on Flash, or the floor on Pro (only - // reachable via a stale persisted preference — see canDisableReasoning). - return 'none' - case 'deepseek': - return DEEPSEEK_OFF_SENTINEL - case 'openai': - case 'azure_openai': - // gpt-5.1+ reasoning is off only via the explicit 'none' effort - // (gpt-5.5 defaults to medium when the field is omitted). - return /^gpt-5\./.test(baseModelId(model)) ? 'none' : undefined - case 'openrouter': - // OpenRouter validates effort against xhigh..minimal|none and - // documents 'none' as disabling reasoning, translated per the - // underlying provider — more reliable than omission, which keeps - // reasoning-by-default models thinking. - return 'none' - default: - return undefined - } + return findReasoningRule(provider, model).rule?.offToken } /**