weekly ai evals on current models, and claude 5.5/gpt-6 support (#11409)

* feat: run ai evals weekly on current models and post results to a dashboard

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* feat: add current flagship models, a reasoning flag and claude 5.5 defaults

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: never send a reasoning disable claude 5.5 or gpt-6-astra reject, and treat gpt-6 as a reasoning model

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: address review on gpt-6 support, chat completions tools and model metadata

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: leave tiered gpt-6 unpriced and drop the off sentinel on gpt-5 and o-series

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* docs: point the ai_evals readme at the model registry instead of copying it

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* refactor: encode the reasoning rules as per-family maps with a shared parity fixture

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix: keep the chat completions tools rule open-ended past gpt-5.6 and scope the parity fixture

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
hugocasa
2026-09-30 13:10:24 +02:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 94ac34a489
commit e117871bec
20 changed files with 888 additions and 414 deletions
+171
View File
@@ -0,0 +1,171 @@
name: AI Evals (scheduled)
# Full ai_evals suites on current models, posted to the AI evals dashboard in
# the windmill-prod workspace (f/ai/ai_evals_dashboard) so quality is tracked
# over time. Weekly, since one pass of every suite at --runs 3 costs about 50M
# tokens per model: every suite runs on MODELS, and global (the mode users get)
# also runs on GLOBAL_EXTRA_MODELS, one flagship per other provider. Run it by
# hand to measure a branch against main.
on:
schedule:
- cron: "0 3 * * 1"
workflow_dispatch:
inputs:
modes:
description: "Space-separated modes"
default: "global flow app script cli"
models:
description: "Space-separated model aliases (bun run cli -- models)"
default: "sonnet-5.5"
global_extra_models:
description: "Extra model aliases for global mode only"
default: "gpt-6-astra gemini-3.8-flash"
runs:
description: "Runs per case"
default: "3"
reasoning:
description: "Reasoning effort for frontend modes (empty: the product default)"
default: ""
concurrency:
group: ai-evals-scheduled-${{ github.ref }}
env:
MODELS: ${{ inputs.models || 'sonnet-5.5' }}
GLOBAL_EXTRA_MODELS: ${{ inputs.global_extra_models || 'gpt-6-astra gemini-3.8-flash' }}
RUNS: ${{ inputs.runs || '3' }}
REASONING: ${{ inputs.reasoning }}
INGEST_URL: https://app.windmill.dev/api/r/f/ai/ingest_ai_eval_run
jobs:
setup:
runs-on: ubuntu-latest
outputs:
modes: ${{ steps.modes.outputs.modes }}
steps:
- id: modes
env:
MODES: ${{ inputs.modes || 'global flow app script cli' }}
run: |
echo "modes=$(jq -cn --arg m "$MODES" '$m | split(" ")
| map(select(IN("global", "flow", "app", "script", "cli")))')" >> "$GITHUB_OUTPUT"
evals:
needs: setup
runs-on: ubicloud-standard-16
timeout-minutes: 330
strategy:
fail-fast: false
matrix:
mode: ${{ fromJSON(needs.setup.outputs.modes) }}
services:
postgres:
image: postgres:16
ports:
- 5432:5432
env:
POSTGRES_DB: windmill
POSTGRES_PASSWORD: changeme
options: >-
--health-cmd pg_isready --health-interval 10s --health-timeout 5s
--health-retries 5
steps:
- uses: actions/checkout@v4
- uses: actions-rust-lang/setup-rust-toolchain@v1
with:
cache-workspaces: backend
toolchain: 1.97.0
- uses: oven-sh/setup-bun@v2
with:
bun-version: 1.4.0
- uses: actions/setup-node@v7
with:
node-version: "24"
- name: Build Windmill
working-directory: ./backend
env:
SQLX_OFFLINE: true
CARGO_BUILD_JOBS: 12
RUSTFLAGS: ""
run: cargo build --features quickjs
- name: Start Windmill
working-directory: ./backend
env:
DATABASE_URL: postgres://postgres:changeme@localhost:5432/windmill
RUST_LOG: info
run: |
mkdir -p ../ai_evals/logs
./target/debug/windmill > ../ai_evals/logs/windmill.log 2>&1 &
for i in $(seq 1 60); do
curl -sf http://localhost:8000/api/version > /dev/null 2>&1 && break
sleep 2
done
curl -sf http://localhost:8000/api/version > /dev/null || { tail -50 ../ai_evals/logs/windmill.log; exit 1; }
- name: Install frontend deps + generate client
working-directory: ./frontend
run: |
npm ci
npm run generate-backend-client
- name: Install CLI deps + generate CLI client
working-directory: ./cli
run: bun install && ./gen_wm_client.sh && ./windmill-utils-internal/gen_wm_client.sh
- name: Run ${{ matrix.mode }} evals
working-directory: ./ai_evals
env:
MODE: ${{ matrix.mode }}
WMILL_AI_EVAL_BACKEND_URL: http://localhost:8000
WMILL_AI_EVAL_BACKEND_WORKSPACE: integration-tests
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
GEMINI_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }}
run: |
bun install
mkdir -p results
models="$MODELS"
[ "$MODE" = global ] && models="$models $GLOBAL_EXTRA_MODELS"
reasoning=()
[ -n "$REASONING" ] && [ "$MODE" != cli ] && reasoning=(--reasoning "$REASONING")
for m in $models; do
bun run cli -- run "$MODE" --model "$m" --runs "$RUNS" "${reasoning[@]}" \
--output "$PWD/results/$MODE-$m.json" || echo "::warning::$MODE on $m errored"
done
- name: Post results to the dashboard
if: always()
working-directory: ./ai_evals
env:
MODE: ${{ matrix.mode }}
INGEST_TOKEN: ${{ secrets.AI_EVALS_INGEST_TOKEN }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
[ -n "$INGEST_TOKEN" ] || { echo "::warning::AI_EVALS_INGEST_TOKEN is not set"; exit 0; }
shopt -s nullglob
for f in results/"$MODE"-*.json; do
# Keep what the dashboard reads; the full traces stay in the run's artifacts.
jq -c --arg ref "$GITHUB_REF" --arg trigger "$GITHUB_EVENT_NAME" --arg url "$RUN_URL" '
{result_json: (del(.cases[].prompt, .cases[].initialPath, .cases[].expectedPath)
| .cases[].attempts[] |= {attempt, passed, durationMs, toolCallCount, judgeScore,
judgeSummary, error, tokenUsage, checks: [.checks[]? | {name, passed}]}),
git_ref: $ref, trigger: $trigger, run_url: $url}' "$f" \
| curl -sf --retry 3 -X POST "$INGEST_URL" \
-H "Authorization: Bearer $INGEST_TOKEN" -H 'content-type: application/json' \
--data-binary @- && echo " <- $f" || echo "::warning::failed to post $f"
done
- name: Archive logs and results
uses: actions/upload-artifact@v4
if: always()
with:
name: ai-evals-${{ matrix.mode }}
path: |
ai_evals/logs
ai_evals/results