diff --git a/.agents/skills/local-review-codex/SKILL.md b/.agents/skills/local-review-codex/SKILL.md index ec1277fe3a..3483453f66 100644 --- a/.agents/skills/local-review-codex/SKILL.md +++ b/.agents/skills/local-review-codex/SKILL.md @@ -1,6 +1,6 @@ --- name: local-review-codex -description: Run the CI Codex PR review locally against this branch's unpushed work (committed + uncommitted) before pushing. Same policy and reasoning effort as the codex-pr-review GitHub action, on a newer model. +description: Run the CI Codex PR review locally against this branch's unpushed work (committed + uncommitted) before pushing. Same policy, model and reasoning effort as the codex-pr-review GitHub action. --- # Local Codex Review (pre-push) @@ -11,18 +11,18 @@ before the PR exists. Use this before `git push` on a non-trivial change. **Correspondence with CI** — identical: - Policy: `REVIEW.md` (severity triage, public-surface checklist, AGENTS.md compliance, test coverage). +- Model: `gpt-6.1-sol`. - Reasoning effort: `model_reasoning_effort="xhigh"`. - Output: markdown starting with `## Codex Review`, findings tagged P0 / P1 / P2 with file:line. **Differences from CI** — local-only: -- Model is `gpt-6-astra`; CI stays on `gpt-5.6-sol`. Not an oversight to reconcile: `gpt-6-astra` is confirmed on the ChatGPT auth `codex login` uses locally, while CI authenticates with `OPENAI_API_KEY` (`codex-pr-review.yml` prefers it over `CODEX_AUTH_JSON`) and that tier is unverified for the model. Move CI once API access is confirmed, or once CI switches to `CODEX_AUTH_JSON`. - Scope is the current branch vs `main` at the merge-base, **including uncommitted changes** (CI reviews a pushed PR diff). - Sandbox is `read-only` (CI uses `danger-full-access` on an ephemeral runner). Codex reads the diff and files but cannot modify your working tree. - Fresh context is inherent: `codex exec` is a separate cold process, so it does not anchor on the current chat session — the same reason `local-review` insists on a subagent. ## Prerequisites -- `codex` CLI **>= 0.153.4** installed and authed via `codex login` (an `OPENAI_API_KEY` in the environment takes priority and may not reach `gpt-6-astra` — see the model note above). Older CLIs reject the model with "requires a newer version of Codex"; `run.sh` checks the version up front. Upgrade with `npm install --global @openai/codex@0.153.4` (may need `sudo` for a global install). This matches the pin in `.github/workflows/codex-pr-review.yml` — the CLI version is the same on both sides, only the model differs. +- `codex` CLI **>= 0.159.3** installed and authed via `codex login` or an `OPENAI_API_KEY` in the environment (which takes priority). Older CLIs reject the model with "requires a newer version of Codex"; `run.sh` checks the version up front. Upgrade with `npm install --global @openai/codex@0.159.3` (may need `sudo` for a global install). This matches the pin in `.github/workflows/codex-pr-review.yml` — the CLI version is the same on both sides. - `git fetch` the base ref if it's stale, so the merge-base is accurate. ## Run diff --git a/.agents/skills/local-review-codex/run.sh b/.agents/skills/local-review-codex/run.sh index 948d3820eb..e1fef4ce8e 100755 --- a/.agents/skills/local-review-codex/run.sh +++ b/.agents/skills/local-review-codex/run.sh @@ -1,17 +1,14 @@ #!/usr/bin/env bash # Local Codex review — mirrors the .github/workflows/codex-pr-review.yml CI job, # but scoped to this branch's unpushed work (committed + uncommitted) so you can -# review before pushing. Same policy (REVIEW.md) and reasoning effort (xhigh) as CI. -# -# The model deliberately differs from CI: gpt-6-astra is confirmed available on the -# ChatGPT auth `codex login` uses here, but CI authenticates with OPENAI_API_KEY and -# that tier is unverified for it, so codex-pr-review.yml stays on gpt-5.6-sol. +# review before pushing. Same policy (REVIEW.md), model and reasoning effort (xhigh) +# as CI. # # Usage: run.sh [BASE_REF] (BASE_REF defaults to "main") set -euo pipefail -MODEL="gpt-6-astra" -CODEX_MIN="0.153.4" +MODEL="gpt-6.1-sol" +CODEX_MIN="0.159.3" BASE_REF="${1:-main}" REPO_ROOT="$(git rev-parse --show-toplevel)" diff --git a/.github/workflows/ai-evals-test.yml b/.github/workflows/ai-evals-test.yml index 573bc77df9..84b488ae46 100644 --- a/.github/workflows/ai-evals-test.yml +++ b/.github/workflows/ai-evals-test.yml @@ -147,7 +147,7 @@ jobs: mkdir -p results # One cheap model per provider (anthropic/openai/googleai/deepseek). fail=0 - for m in haiku 4o gemini-3-flash-preview deepseek-v4-flash; do + for m in haiku gpt-6-luna gemini-3.8-flash deepseek-flash; do echo "::group::global-test1-script-create ($m)" if ! bun run cli -- run global global-test1-script-create \ --model "$m" --execution-only --output "$PWD/results/ci-$m.json"; then diff --git a/.github/workflows/claude-plan.yml b/.github/workflows/claude-plan.yml index bc81801417..b3798c76f5 100644 --- a/.github/workflows/claude-plan.yml +++ b/.github/workflows/claude-plan.yml @@ -49,7 +49,7 @@ jobs: allowed_bots: 'windmill-internal-app[bot]' trigger_phrase: '/plan' claude_args: | - --model claude-opus-5 + --model claude-opus-5-5 --system-prompt "# Claude Planning Mode You are operating in PLANNING MODE ONLY. Your role is to create detailed, structured plans without making any code changes. diff --git a/.github/workflows/claude.yml b/.github/workflows/claude.yml index 5e9aab00d3..5250c5a04b 100644 --- a/.github/workflows/claude.yml +++ b/.github/workflows/claude.yml @@ -93,4 +93,4 @@ jobs: } claude_args: | --allowedTools "Bash,WebFetch,WebSearch" - --model claude-opus-5 + --model claude-opus-5-5 diff --git a/.github/workflows/codex-pr-review.yml b/.github/workflows/codex-pr-review.yml index 55036776c6..2d70ad7f7a 100644 --- a/.github/workflows/codex-pr-review.yml +++ b/.github/workflows/codex-pr-review.yml @@ -222,7 +222,7 @@ jobs: - name: Install Codex CLI if: steps.codex_config.outputs.enabled == 'true' && steps.pr.outputs.skip != 'true' - run: npm install --global @openai/codex@0.153.4 + run: npm install --global @openai/codex@0.159.3 - name: Configure Codex auth if: steps.codex_config.outputs.enabled == 'true' && steps.pr.outputs.skip != 'true' @@ -359,7 +359,7 @@ jobs: # e.g. a GitHub Action's index.js, which then runs with our credentials. codex exec \ -C "$GITHUB_WORKSPACE" \ - -m gpt-5.6-sol \ + -m gpt-6.1-sol \ -c 'model_reasoning_effort="xhigh"' \ -s "$SANDBOX_MODE" \ -o "$RUNNER_TEMP/codex-final-message.md" \ diff --git a/.github/workflows/pr-ready-review.yml b/.github/workflows/pr-ready-review.yml index 5e5c6a8c61..90714cba53 100644 --- a/.github/workflows/pr-ready-review.yml +++ b/.github/workflows/pr-ready-review.yml @@ -208,4 +208,4 @@ jobs: ${{ env.REVIEW_PROMPT }} claude_args: | --allowedTools "mcp__github_inline_comment__create_inline_comment,Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*)" - --model claude-opus-5 + --model claude-opus-5-5 diff --git a/AGENTS.md b/AGENTS.md index 54b5953748..8451a0a600 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -49,7 +49,7 @@ Open-source platform for internal tools, workflows, API integrations, background - **Backend patterns**: use the `rust-backend` skill when writing Rust code - **Frontend patterns**: use the `svelte-frontend` skill when writing Svelte code. Do NOT edit svelte files unless you have read that skill. - **Frontend UUIDs**: do not call `crypto.randomUUID()` in frontend code. Import `randomUUID` from `$lib/utils/uuid` instead. -- **Code review**: review the current PR or branch against the shared review policy in `REVIEW.md` (severity triage, public-surface checklist, AGENTS.md compliance, test-coverage assessment). The skill at `.agents/skills/local-review/SKILL.md` orchestrates it. All three CLIs auto-discover the same SKILL — Claude reads `.claude/skills/` (symlinked to the canonical `.agents/skills/` file), Codex and Pi read `.agents/skills/` directly. Invoke with `/local-review` in Claude Code, `$local-review` (or `/skills` selector) in Codex, or `pi --skill local-review` / `/skill:local-review` in Pi. For a Codex-driven pass that mirrors the `codex-pr-review` GitHub action against your unpushed work (committed + uncommitted) before you push, use `/local-review-codex` (`.agents/skills/local-review-codex/`) — same `REVIEW.md` policy and `xhigh` reasoning, on `gpt-6-astra` rather than the action's `gpt-5.6-sol`; requires the `codex` CLI >= 0.153.4. +- **Code review**: review the current PR or branch against the shared review policy in `REVIEW.md` (severity triage, public-surface checklist, AGENTS.md compliance, test-coverage assessment). The skill at `.agents/skills/local-review/SKILL.md` orchestrates it. All three CLIs auto-discover the same SKILL — Claude reads `.claude/skills/` (symlinked to the canonical `.agents/skills/` file), Codex and Pi read `.agents/skills/` directly. Invoke with `/local-review` in Claude Code, `$local-review` (or `/skills` selector) in Codex, or `pi --skill local-review` / `/skill:local-review` in Pi. For a Codex-driven pass that mirrors the `codex-pr-review` GitHub action against your unpushed work (committed + uncommitted) before you push, use `/local-review-codex` (`.agents/skills/local-review-codex/`) — same `REVIEW.md` policy, `gpt-6.1-sol` model and `xhigh` reasoning as the action; requires the `codex` CLI >= 0.159.3. - **Domain guides**: `.claude/skills/native-trigger/` - **Brand/UI guidelines**: `frontend/brand-guidelines.md` - **Domain vocabulary**: `CONTEXT.md` — the words this codebase uses for its own concepts (step, step setting, trigger step, …). Name things the way it does. diff --git a/ai_evals/core/models.test.ts b/ai_evals/core/models.test.ts index ba53c24592..d51d61d224 100644 --- a/ai_evals/core/models.test.ts +++ b/ai_evals/core/models.test.ts @@ -34,6 +34,10 @@ describe("resolveEvalModel", () => { it("supports DeepSeek aliases for frontend evals", () => { expect(resolveEvalModel("flow", "deepseek").frontend).toEqual({ + provider: "deepseek", + model: "deepseek-flash", + }); + expect(resolveEvalModel("flow", "deepseek-v4-flash").frontend).toEqual({ provider: "deepseek", model: "deepseek-v4-flash", }); diff --git a/ai_evals/core/models.ts b/ai_evals/core/models.ts index 1580cf5b0b..06689c36e1 100644 --- a/ai_evals/core/models.ts +++ b/ai_evals/core/models.ts @@ -192,12 +192,21 @@ export const EVAL_MODELS: EvalModelSpec[] = [ { id: "deepseek-v4-flash", label: "DeepSeek V4 Flash", - aliases: ["deepseek", "deepseek-v4", "deepseek-v4-flash"], + aliases: ["deepseek-v4", "deepseek-v4-flash"], frontend: { provider: "deepseek", model: "deepseek-v4-flash", }, }, + { + id: "deepseek-flash", + label: "DeepSeek V4.1 Flash", + aliases: ["deepseek", "deepseek-flash", "deepseek-v4.1-flash"], + frontend: { + provider: "deepseek", + model: "deepseek-flash", + }, + }, { id: "deepseek-v4-pro", label: "DeepSeek V4 Pro",