mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-10-03 16:02:12 +00:00
weekly ai evals on current models, and claude 5.5/gpt-6 support (#11409)
* feat: run ai evals weekly on current models and post results to a dashboard Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * feat: add current flagship models, a reasoning flag and claude 5.5 defaults Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: never send a reasoning disable claude 5.5 or gpt-6-astra reject, and treat gpt-6 as a reasoning model Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: address review on gpt-6 support, chat completions tools and model metadata Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: leave tiered gpt-6 unpriced and drop the off sentinel on gpt-5 and o-series Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * docs: point the ai_evals readme at the model registry instead of copying it Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * refactor: encode the reasoning rules as per-family maps with a shared parity fixture Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> * fix: keep the chat completions tools rule open-ended past gpt-5.6 and scope the parity fixture Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
94ac34a489
commit
e117871bec
@@ -0,0 +1,171 @@
|
||||
name: AI Evals (scheduled)
|
||||
|
||||
# Full ai_evals suites on current models, posted to the AI evals dashboard in
|
||||
# the windmill-prod workspace (f/ai/ai_evals_dashboard) so quality is tracked
|
||||
# over time. Weekly, since one pass of every suite at --runs 3 costs about 50M
|
||||
# tokens per model: every suite runs on MODELS, and global (the mode users get)
|
||||
# also runs on GLOBAL_EXTRA_MODELS, one flagship per other provider. Run it by
|
||||
# hand to measure a branch against main.
|
||||
on:
|
||||
schedule:
|
||||
- cron: "0 3 * * 1"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
modes:
|
||||
description: "Space-separated modes"
|
||||
default: "global flow app script cli"
|
||||
models:
|
||||
description: "Space-separated model aliases (bun run cli -- models)"
|
||||
default: "sonnet-5.5"
|
||||
global_extra_models:
|
||||
description: "Extra model aliases for global mode only"
|
||||
default: "gpt-6-astra gemini-3.8-flash"
|
||||
runs:
|
||||
description: "Runs per case"
|
||||
default: "3"
|
||||
reasoning:
|
||||
description: "Reasoning effort for frontend modes (empty: the product default)"
|
||||
default: ""
|
||||
|
||||
concurrency:
|
||||
group: ai-evals-scheduled-${{ github.ref }}
|
||||
|
||||
env:
|
||||
MODELS: ${{ inputs.models || 'sonnet-5.5' }}
|
||||
GLOBAL_EXTRA_MODELS: ${{ inputs.global_extra_models || 'gpt-6-astra gemini-3.8-flash' }}
|
||||
RUNS: ${{ inputs.runs || '3' }}
|
||||
REASONING: ${{ inputs.reasoning }}
|
||||
INGEST_URL: https://app.windmill.dev/api/r/f/ai/ingest_ai_eval_run
|
||||
|
||||
jobs:
|
||||
setup:
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
modes: ${{ steps.modes.outputs.modes }}
|
||||
steps:
|
||||
- id: modes
|
||||
env:
|
||||
MODES: ${{ inputs.modes || 'global flow app script cli' }}
|
||||
run: |
|
||||
echo "modes=$(jq -cn --arg m "$MODES" '$m | split(" ")
|
||||
| map(select(IN("global", "flow", "app", "script", "cli")))')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
evals:
|
||||
needs: setup
|
||||
runs-on: ubicloud-standard-16
|
||||
timeout-minutes: 330
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
mode: ${{ fromJSON(needs.setup.outputs.modes) }}
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:16
|
||||
ports:
|
||||
- 5432:5432
|
||||
env:
|
||||
POSTGRES_DB: windmill
|
||||
POSTGRES_PASSWORD: changeme
|
||||
options: >-
|
||||
--health-cmd pg_isready --health-interval 10s --health-timeout 5s
|
||||
--health-retries 5
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions-rust-lang/setup-rust-toolchain@v1
|
||||
with:
|
||||
cache-workspaces: backend
|
||||
toolchain: 1.97.0
|
||||
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: 1.4.0
|
||||
|
||||
- uses: actions/setup-node@v7
|
||||
with:
|
||||
node-version: "24"
|
||||
|
||||
- name: Build Windmill
|
||||
working-directory: ./backend
|
||||
env:
|
||||
SQLX_OFFLINE: true
|
||||
CARGO_BUILD_JOBS: 12
|
||||
RUSTFLAGS: ""
|
||||
run: cargo build --features quickjs
|
||||
|
||||
- name: Start Windmill
|
||||
working-directory: ./backend
|
||||
env:
|
||||
DATABASE_URL: postgres://postgres:changeme@localhost:5432/windmill
|
||||
RUST_LOG: info
|
||||
run: |
|
||||
mkdir -p ../ai_evals/logs
|
||||
./target/debug/windmill > ../ai_evals/logs/windmill.log 2>&1 &
|
||||
for i in $(seq 1 60); do
|
||||
curl -sf http://localhost:8000/api/version > /dev/null 2>&1 && break
|
||||
sleep 2
|
||||
done
|
||||
curl -sf http://localhost:8000/api/version > /dev/null || { tail -50 ../ai_evals/logs/windmill.log; exit 1; }
|
||||
|
||||
- name: Install frontend deps + generate client
|
||||
working-directory: ./frontend
|
||||
run: |
|
||||
npm ci
|
||||
npm run generate-backend-client
|
||||
|
||||
- name: Install CLI deps + generate CLI client
|
||||
working-directory: ./cli
|
||||
run: bun install && ./gen_wm_client.sh && ./windmill-utils-internal/gen_wm_client.sh
|
||||
|
||||
- name: Run ${{ matrix.mode }} evals
|
||||
working-directory: ./ai_evals
|
||||
env:
|
||||
MODE: ${{ matrix.mode }}
|
||||
WMILL_AI_EVAL_BACKEND_URL: http://localhost:8000
|
||||
WMILL_AI_EVAL_BACKEND_WORKSPACE: integration-tests
|
||||
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
|
||||
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
||||
GEMINI_API_KEY: ${{ secrets.GOOGLE_API_KEY }}
|
||||
DEEPSEEK_API_KEY: ${{ secrets.DEEPSEEK_API_KEY }}
|
||||
run: |
|
||||
bun install
|
||||
mkdir -p results
|
||||
models="$MODELS"
|
||||
[ "$MODE" = global ] && models="$models $GLOBAL_EXTRA_MODELS"
|
||||
reasoning=()
|
||||
[ -n "$REASONING" ] && [ "$MODE" != cli ] && reasoning=(--reasoning "$REASONING")
|
||||
for m in $models; do
|
||||
bun run cli -- run "$MODE" --model "$m" --runs "$RUNS" "${reasoning[@]}" \
|
||||
--output "$PWD/results/$MODE-$m.json" || echo "::warning::$MODE on $m errored"
|
||||
done
|
||||
|
||||
- name: Post results to the dashboard
|
||||
if: always()
|
||||
working-directory: ./ai_evals
|
||||
env:
|
||||
MODE: ${{ matrix.mode }}
|
||||
INGEST_TOKEN: ${{ secrets.AI_EVALS_INGEST_TOKEN }}
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
run: |
|
||||
[ -n "$INGEST_TOKEN" ] || { echo "::warning::AI_EVALS_INGEST_TOKEN is not set"; exit 0; }
|
||||
shopt -s nullglob
|
||||
for f in results/"$MODE"-*.json; do
|
||||
# Keep what the dashboard reads; the full traces stay in the run's artifacts.
|
||||
jq -c --arg ref "$GITHUB_REF" --arg trigger "$GITHUB_EVENT_NAME" --arg url "$RUN_URL" '
|
||||
{result_json: (del(.cases[].prompt, .cases[].initialPath, .cases[].expectedPath)
|
||||
| .cases[].attempts[] |= {attempt, passed, durationMs, toolCallCount, judgeScore,
|
||||
judgeSummary, error, tokenUsage, checks: [.checks[]? | {name, passed}]}),
|
||||
git_ref: $ref, trigger: $trigger, run_url: $url}' "$f" \
|
||||
| curl -sf --retry 3 -X POST "$INGEST_URL" \
|
||||
-H "Authorization: Bearer $INGEST_TOKEN" -H 'content-type: application/json' \
|
||||
--data-binary @- && echo " <- $f" || echo "::warning::failed to post $f"
|
||||
done
|
||||
|
||||
- name: Archive logs and results
|
||||
uses: actions/upload-artifact@v4
|
||||
if: always()
|
||||
with:
|
||||
name: ai-evals-${{ matrix.mode }}
|
||||
path: |
|
||||
ai_evals/logs
|
||||
ai_evals/results
|
||||
Reference in New Issue
Block a user