ci: move CI AI models to opus 5.5, gpt-6.1 sol and deepseek v4.1 (#11461)

* ci: move CI AI models to opus 5.5, gpt-6 and deepseek v4.1 flash

* test: point the deepseek eval alias at v4.1 flash

* ci: run the codex review on gpt-6.1-sol with codex cli 0.159.3
This commit is contained in:
Ruben Fiszel
2026-10-01 11:11:57 +02:00
committed by GitHub
parent 94924d4644
commit f1d970bd1a
10 changed files with 28 additions and 18 deletions
+3 -3
View File
@@ -1,6 +1,6 @@
---
name: local-review-codex
description: Run the CI Codex PR review locally against this branch's unpushed work (committed + uncommitted) before pushing. Same policy and reasoning effort as the codex-pr-review GitHub action, on a newer model.
description: Run the CI Codex PR review locally against this branch's unpushed work (committed + uncommitted) before pushing. Same policy, model and reasoning effort as the codex-pr-review GitHub action.
---
# Local Codex Review (pre-push)
@@ -11,18 +11,18 @@ before the PR exists. Use this before `git push` on a non-trivial change.
**Correspondence with CI** — identical:
- Policy: `REVIEW.md` (severity triage, public-surface checklist, AGENTS.md compliance, test coverage).
- Model: `gpt-6.1-sol`.
- Reasoning effort: `model_reasoning_effort="xhigh"`.
- Output: markdown starting with `## Codex Review`, findings tagged P0 / P1 / P2 with file:line.
**Differences from CI** — local-only:
- Model is `gpt-6-astra`; CI stays on `gpt-5.6-sol`. Not an oversight to reconcile: `gpt-6-astra` is confirmed on the ChatGPT auth `codex login` uses locally, while CI authenticates with `OPENAI_API_KEY` (`codex-pr-review.yml` prefers it over `CODEX_AUTH_JSON`) and that tier is unverified for the model. Move CI once API access is confirmed, or once CI switches to `CODEX_AUTH_JSON`.
- Scope is the current branch vs `main` at the merge-base, **including uncommitted changes** (CI reviews a pushed PR diff).
- Sandbox is `read-only` (CI uses `danger-full-access` on an ephemeral runner). Codex reads the diff and files but cannot modify your working tree.
- Fresh context is inherent: `codex exec` is a separate cold process, so it does not anchor on the current chat session — the same reason `local-review` insists on a subagent.
## Prerequisites
- `codex` CLI **>= 0.153.4** installed and authed via `codex login` (an `OPENAI_API_KEY` in the environment takes priority and may not reach `gpt-6-astra` — see the model note above). Older CLIs reject the model with "requires a newer version of Codex"; `run.sh` checks the version up front. Upgrade with `npm install --global @openai/codex@0.153.4` (may need `sudo` for a global install). This matches the pin in `.github/workflows/codex-pr-review.yml` — the CLI version is the same on both sides, only the model differs.
- `codex` CLI **>= 0.159.3** installed and authed via `codex login` or an `OPENAI_API_KEY` in the environment (which takes priority). Older CLIs reject the model with "requires a newer version of Codex"; `run.sh` checks the version up front. Upgrade with `npm install --global @openai/codex@0.159.3` (may need `sudo` for a global install). This matches the pin in `.github/workflows/codex-pr-review.yml` — the CLI version is the same on both sides.
- `git fetch` the base ref if it's stale, so the merge-base is accurate.
## Run
+4 -7
View File
@@ -1,17 +1,14 @@
#!/usr/bin/env bash
# Local Codex review — mirrors the .github/workflows/codex-pr-review.yml CI job,
# but scoped to this branch's unpushed work (committed + uncommitted) so you can
# review before pushing. Same policy (REVIEW.md) and reasoning effort (xhigh) as CI.
#
# The model deliberately differs from CI: gpt-6-astra is confirmed available on the
# ChatGPT auth `codex login` uses here, but CI authenticates with OPENAI_API_KEY and
# that tier is unverified for it, so codex-pr-review.yml stays on gpt-5.6-sol.
# review before pushing. Same policy (REVIEW.md), model and reasoning effort (xhigh)
# as CI.
#
# Usage: run.sh [BASE_REF] (BASE_REF defaults to "main")
set -euo pipefail
MODEL="gpt-6-astra"
CODEX_MIN="0.153.4"
MODEL="gpt-6.1-sol"
CODEX_MIN="0.159.3"
BASE_REF="${1:-main}"
REPO_ROOT="$(git rev-parse --show-toplevel)"
+1 -1
View File
@@ -147,7 +147,7 @@ jobs:
mkdir -p results
# One cheap model per provider (anthropic/openai/googleai/deepseek).
fail=0
for m in haiku 4o gemini-3-flash-preview deepseek-v4-flash; do
for m in haiku gpt-6-luna gemini-3.8-flash deepseek-flash; do
echo "::group::global-test1-script-create ($m)"
if ! bun run cli -- run global global-test1-script-create \
--model "$m" --execution-only --output "$PWD/results/ci-$m.json"; then
+1 -1
View File
@@ -49,7 +49,7 @@ jobs:
allowed_bots: 'windmill-internal-app[bot]'
trigger_phrase: '/plan'
claude_args: |
--model claude-opus-5
--model claude-opus-5-5
--system-prompt "# Claude Planning Mode
You are operating in PLANNING MODE ONLY. Your role is to create detailed, structured plans without making any code changes.
+1 -1
View File
@@ -93,4 +93,4 @@ jobs:
}
claude_args: |
--allowedTools "Bash,WebFetch,WebSearch"
--model claude-opus-5
--model claude-opus-5-5
+2 -2
View File
@@ -222,7 +222,7 @@ jobs:
- name: Install Codex CLI
if: steps.codex_config.outputs.enabled == 'true' && steps.pr.outputs.skip != 'true'
run: npm install --global @openai/codex@0.153.4
run: npm install --global @openai/codex@0.159.3
- name: Configure Codex auth
if: steps.codex_config.outputs.enabled == 'true' && steps.pr.outputs.skip != 'true'
@@ -359,7 +359,7 @@ jobs:
# e.g. a GitHub Action's index.js, which then runs with our credentials.
codex exec \
-C "$GITHUB_WORKSPACE" \
-m gpt-5.6-sol \
-m gpt-6.1-sol \
-c 'model_reasoning_effort="xhigh"' \
-s "$SANDBOX_MODE" \
-o "$RUNNER_TEMP/codex-final-message.md" \
+1 -1
View File
@@ -208,4 +208,4 @@ jobs:
${{ env.REVIEW_PROMPT }}
claude_args: |
--allowedTools "mcp__github_inline_comment__create_inline_comment,Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*)"
--model claude-opus-5
--model claude-opus-5-5
+1 -1
View File
@@ -49,7 +49,7 @@ Open-source platform for internal tools, workflows, API integrations, background
- **Backend patterns**: use the `rust-backend` skill when writing Rust code
- **Frontend patterns**: use the `svelte-frontend` skill when writing Svelte code. Do NOT edit svelte files unless you have read that skill.
- **Frontend UUIDs**: do not call `crypto.randomUUID()` in frontend code. Import `randomUUID` from `$lib/utils/uuid` instead.
- **Code review**: review the current PR or branch against the shared review policy in `REVIEW.md` (severity triage, public-surface checklist, AGENTS.md compliance, test-coverage assessment). The skill at `.agents/skills/local-review/SKILL.md` orchestrates it. All three CLIs auto-discover the same SKILL — Claude reads `.claude/skills/` (symlinked to the canonical `.agents/skills/` file), Codex and Pi read `.agents/skills/` directly. Invoke with `/local-review` in Claude Code, `$local-review` (or `/skills` selector) in Codex, or `pi --skill local-review` / `/skill:local-review` in Pi. For a Codex-driven pass that mirrors the `codex-pr-review` GitHub action against your unpushed work (committed + uncommitted) before you push, use `/local-review-codex` (`.agents/skills/local-review-codex/`) — same `REVIEW.md` policy and `xhigh` reasoning, on `gpt-6-astra` rather than the action's `gpt-5.6-sol`; requires the `codex` CLI >= 0.153.4.
- **Code review**: review the current PR or branch against the shared review policy in `REVIEW.md` (severity triage, public-surface checklist, AGENTS.md compliance, test-coverage assessment). The skill at `.agents/skills/local-review/SKILL.md` orchestrates it. All three CLIs auto-discover the same SKILL — Claude reads `.claude/skills/` (symlinked to the canonical `.agents/skills/` file), Codex and Pi read `.agents/skills/` directly. Invoke with `/local-review` in Claude Code, `$local-review` (or `/skills` selector) in Codex, or `pi --skill local-review` / `/skill:local-review` in Pi. For a Codex-driven pass that mirrors the `codex-pr-review` GitHub action against your unpushed work (committed + uncommitted) before you push, use `/local-review-codex` (`.agents/skills/local-review-codex/`) — same `REVIEW.md` policy, `gpt-6.1-sol` model and `xhigh` reasoning as the action; requires the `codex` CLI >= 0.159.3.
- **Domain guides**: `.claude/skills/native-trigger/`
- **Brand/UI guidelines**: `frontend/brand-guidelines.md`
- **Domain vocabulary**: `CONTEXT.md` — the words this codebase uses for its own concepts (step, step setting, trigger step, …). Name things the way it does.
+4
View File
@@ -34,6 +34,10 @@ describe("resolveEvalModel", () => {
it("supports DeepSeek aliases for frontend evals", () => {
expect(resolveEvalModel("flow", "deepseek").frontend).toEqual({
provider: "deepseek",
model: "deepseek-flash",
});
expect(resolveEvalModel("flow", "deepseek-v4-flash").frontend).toEqual({
provider: "deepseek",
model: "deepseek-v4-flash",
});
+10 -1
View File
@@ -192,12 +192,21 @@ export const EVAL_MODELS: EvalModelSpec[] = [
{
id: "deepseek-v4-flash",
label: "DeepSeek V4 Flash",
aliases: ["deepseek", "deepseek-v4", "deepseek-v4-flash"],
aliases: ["deepseek-v4", "deepseek-v4-flash"],
frontend: {
provider: "deepseek",
model: "deepseek-v4-flash",
},
},
{
id: "deepseek-flash",
label: "DeepSeek V4.1 Flash",
aliases: ["deepseek", "deepseek-flash", "deepseek-v4.1-flash"],
frontend: {
provider: "deepseek",
model: "deepseek-flash",
},
},
{
id: "deepseek-v4-pro",
label: "DeepSeek V4 Pro",