Merge branch 'main' into feat/asset-graph-view

# Conflicts:
#	backend/ee-repo-ref.txt
#	frontend/src/lib/components/ScriptEditor.svelte
This commit is contained in:
Ruben Fiszel
2026-06-02 09:40:18 +00:00
256 changed files with 15607 additions and 4075 deletions
+3 -1
View File
@@ -8,6 +8,7 @@ on:
- "backend/windmill-git-sync/**"
- "backend/windmill-api-integration-tests/tests/git_sync*"
- "backend/ee-repo-ref.txt"
- "backend/windmill-common/src/workspaces.rs"
- "integration_tests/test/git_sync_test.py"
- ".github/workflows/git-sync-test.yml"
pull_request:
@@ -16,6 +17,7 @@ on:
- "backend/windmill-git-sync/**"
- "backend/windmill-api-integration-tests/tests/git_sync*"
- "backend/ee-repo-ref.txt"
- "backend/windmill-common/src/workspaces.rs"
- "integration_tests/test/git_sync_test.py"
- ".github/workflows/git-sync-test.yml"
@@ -49,7 +51,7 @@ jobs:
echo "$CHANGED_FILES"
# Direct git sync file changes — always relevant
if echo "$CHANGED_FILES" | grep -qE '^(backend/windmill-git-sync/|backend/windmill-api-integration-tests/tests/git_sync|integration_tests/test/git_sync|\.github/workflows/git-sync-test\.yml)'; then
if echo "$CHANGED_FILES" | grep -qE '^(backend/windmill-git-sync/|backend/windmill-api-integration-tests/tests/git_sync|backend/windmill-common/src/workspaces\.rs|integration_tests/test/git_sync|\.github/workflows/git-sync-test\.yml)'; then
echo "should_run=true" >> "$GITHUB_OUTPUT"
echo "Relevant: direct git sync file changes"
exit 0
+76
View File
@@ -1,5 +1,81 @@
# Changelog
## [1.714.0](https://github.com/windmill-labs/windmill/compare/v1.713.1...v1.714.0) (2026-06-02)
### Features
* add global ai chat test tools ([#9391](https://github.com/windmill-labs/windmill/issues/9391)) ([5c20d6b](https://github.com/windmill-labs/windmill/commit/5c20d6b4f79f2ccc1987ce7fdaf74e6b8f697846))
* add workspace datatable tools to global AI chat mode ([#9395](https://github.com/windmill-labs/windmill/issues/9395)) ([943ef6e](https://github.com/windmill-labs/windmill/commit/943ef6eb2089f4b744cfa7945ce47f7f3b361ec7))
* **flow-ai:** constrain flow-group colors to the NoteColor palette ([#9343](https://github.com/windmill-labs/windmill/issues/9343)) ([e4213c1](https://github.com/windmill-labs/windmill/commit/e4213c1ab8c448f492f372580f5c9df37e33fffc))
* **frontend:** surface local drafts in drawer editors with an unsaved-changes banner ([#9335](https://github.com/windmill-labs/windmill/issues/9335)) ([075faab](https://github.com/windmill-labs/windmill/commit/075faabf3bba16a10a02ae3973008e5a13473085))
* handle CTRL_BREAK_EVENT for graceful shutdown on Windows ([#9400](https://github.com/windmill-labs/windmill/issues/9400)) ([2e14456](https://github.com/windmill-labs/windmill/commit/2e1445616a412c5112ad2247b4087c7ddc218845))
* refine ask-user-question chat display and keyboard nav ([#9392](https://github.com/windmill-labs/windmill/issues/9392)) ([1275487](https://github.com/windmill-labs/windmill/commit/1275487f028d4c74a9eeb18981ed05c225505be0))
* sessions page with isolated AI chat + flow editor ([#9034](https://github.com/windmill-labs/windmill/issues/9034)) ([eadeac2](https://github.com/windmill-labs/windmill/commit/eadeac248bd022c2796cfe638eb617c6143b8fc4))
### Bug Fixes
* **cli:** make encryption key push non-interactive-safe + add --skip-reencrypt-on-key-change ([#9402](https://github.com/windmill-labs/windmill/issues/9402)) ([e356bb1](https://github.com/windmill-labs/windmill/commit/e356bb1f5df92eca3fbb0ca2114b9f4c32d4c496))
* **cli:** stop git-sync promotion deploys from dropping triggers/schedules ([#9403](https://github.com/windmill-labs/windmill/issues/9403)) ([24e3ef2](https://github.com/windmill-labs/windmill/commit/24e3ef27be8498fb820c228a52febf6a0a91b487))
* **frontend:** align Monaco editor font size with text-xs ([#9161](https://github.com/windmill-labs/windmill/issues/9161)) ([de76668](https://github.com/windmill-labs/windmill/commit/de76668c10c04abe8771a8ca7bba7b2259819a1c))
* resolve username rename failing on apps with runnable deps ([#9401](https://github.com/windmill-labs/windmill/issues/9401)) ([e8ad53d](https://github.com/windmill-labs/windmill/commit/e8ad53dae92597f5a1a8b76f38a7d8c24f578a47))
### Performance Improvements
* **python:** add --compile-bytecode to uv pip install ([#9393](https://github.com/windmill-labs/windmill/issues/9393)) ([c19441b](https://github.com/windmill-labs/windmill/commit/c19441bc8cb2da064e4ad44d77dc04ab8bbb22ec))
## [1.713.1](https://github.com/windmill-labs/windmill/compare/v1.713.0...v1.713.1) (2026-06-01)
### Bug Fixes
* **api:** handle multi-version scripts when removing granular ACL ([#9388](https://github.com/windmill-labs/windmill/issues/9388)) ([9d9c503](https://github.com/windmill-labs/windmill/commit/9d9c5038ce8b0016320a670c434ef9063cb40441))
## [1.713.0](https://github.com/windmill-labs/windmill/compare/v1.712.0...v1.713.0) (2026-05-31)
### Features
* **flows:** preserve step/subflow worker tags under a custom-tagged flow ([#9375](https://github.com/windmill-labs/windmill/issues/9375)) ([f0301b1](https://github.com/windmill-labs/windmill/commit/f0301b1605cee5fba4024803555333e6fa5c40ee))
* **oauth:** support per-provider sandbox URLs ([#9358](https://github.com/windmill-labs/windmill/issues/9358)) ([2bf11dc](https://github.com/windmill-labs/windmill/commit/2bf11dcb15540c538ea2ac3cf70dcbe589060b4e))
### Bug Fixes
* **ai:** validate token_url for SSRF in OAuth credentials flow ([#9385](https://github.com/windmill-labs/windmill/issues/9385)) ([4b06881](https://github.com/windmill-labs/windmill/commit/4b06881918b76c5a411cc70b318e46efcc1393a7))
* **api:** authorize and harden log-file reading endpoints ([#9368](https://github.com/windmill-labs/windmill/issues/9368)) ([bb90f4c](https://github.com/windmill-labs/windmill/commit/bb90f4ce83a0e60af219b11c12ab4fe1d13f47a4))
* **apps:** make public apps opt into cross-origin isolation via wm_coep (GIT-884) ([#9374](https://github.com/windmill-labs/windmill/issues/9374)) ([2c0c2c4](https://github.com/windmill-labs/windmill/commit/2c0c2c467f163cd24c14c7be2db07af9cf2ce020))
* **auth:** enforce monotonic privilege on user token lifecycle endpoints ([#9371](https://github.com/windmill-labs/windmill/issues/9371)) ([2ddf93d](https://github.com/windmill-labs/windmill/commit/2ddf93de96622b2a1b2b6f59398a7a1f59360efd))
* batch encryption-key rotation into one git-sync job ([#9355](https://github.com/windmill-labs/windmill/issues/9355)) ([04a0897](https://github.com/windmill-labs/windmill/commit/04a08976aec4ba9b0516350316df303e9f96bfd3))
* **cli:** preserve user drafts on sync push and permissioned-as ([#9381](https://github.com/windmill-labs/windmill/issues/9381)) ([b0c3b01](https://github.com/windmill-labs/windmill/commit/b0c3b01d31b0ab3a6566e1f5fec60e3e230cfadb))
* **frontend:** sanitize user markdown to prevent stored XSS ([#9386](https://github.com/windmill-labs/windmill/issues/9386)) ([def01b8](https://github.com/windmill-labs/windmill/commit/def01b8ff6f331cc36ce02b947adc31c766042c4))
* **security:** re-pin cached hub scripts to CVE-patched versions (+ HUB_BASE_URL override for cache mode) ([#9387](https://github.com/windmill-labs/windmill/issues/9387)) ([edf340c](https://github.com/windmill-labs/windmill/commit/edf340c4d4f18b16b142cb7deb67afa586f10946))
## [1.712.0](https://github.com/windmill-labs/windmill/compare/v1.711.0...v1.712.0) (2026-05-28)
### Features
* add deepseek fim support ([#9365](https://github.com/windmill-labs/windmill/issues/9365)) ([2553fbf](https://github.com/windmill-labs/windmill/commit/2553fbfe31417bd985e7994eac695bf918f97ce2))
* deploy raw apps from global chat ([#9349](https://github.com/windmill-labs/windmill/issues/9349)) ([dec58e6](https://github.com/windmill-labs/windmill/commit/dec58e6c4f55062b42a752c43c89ef05903e713a))
* inject active editor into global chat ([#9361](https://github.com/windmill-labs/windmill/issues/9361)) ([9e7eaf3](https://github.com/windmill-labs/windmill/commit/9e7eaf36847ad3a004ec84e8b7d4784771b7b451))
* **queue:** duration-weighted fairness admission ([#9334](https://github.com/windmill-labs/windmill/issues/9334)) ([045d120](https://github.com/windmill-labs/windmill/commit/045d12043e7c99830ef90bc0da798c94e2094711))
* warn when custom instance db is shared across workspaces ([#9359](https://github.com/windmill-labs/windmill/issues/9359)) ([a9e5140](https://github.com/windmill-labs/windmill/commit/a9e514099585e5ee72df21bd551a223cceb20fb0))
### Bug Fixes
* **cli:** redact encryption_key diff in stdout by default ([#9347](https://github.com/windmill-labs/windmill/issues/9347)) ([88056f8](https://github.com/windmill-labs/windmill/commit/88056f8d4c91c1d14d85a08851ecf0bd97e2260d))
* **cli:** stop re-prompting on wmill refresh prompts ([#9357](https://github.com/windmill-labs/windmill/issues/9357)) ([c2b5ba8](https://github.com/windmill-labs/windmill/commit/c2b5ba8871abbbcff6de69c90e2f09fee70586c1))
* **frontend:** close other sidebar menus when hovering Help ([#9354](https://github.com/windmill-labs/windmill/issues/9354)) ([da882c5](https://github.com/windmill-labs/windmill/commit/da882c54b21e3eaf2c1d1abccd0996b243d96dce))
* **frontend:** prevent duplicate asset node ids crashing flow graph ([#9367](https://github.com/windmill-labs/windmill/issues/9367)) ([9a659b6](https://github.com/windmill-labs/windmill/commit/9a659b636d713ee8fdfbdad41c58bb3d7c79e0d9))
* **frontend:** prevent MultiSelect crash on undefined value ([#9364](https://github.com/windmill-labs/windmill/issues/9364)) ([aea0061](https://github.com/windmill-labs/windmill/commit/aea00611c41379be2afdad0eedd608c9537d03f7))
* **git-sync:** publish fork branch on only_create_branch from the CLI ([#9366](https://github.com/windmill-labs/windmill/issues/9366)) ([2fdc51e](https://github.com/windmill-labs/windmill/commit/2fdc51e62985fc755884436130bdd58e294247c8))
* infer script arg schema when deploying via AI chat ([#9356](https://github.com/windmill-labs/windmill/issues/9356)) ([4efc372](https://github.com/windmill-labs/windmill/commit/4efc37212a98571214aba135b0fbb10dc263fd4f))
* **monitor:** cleanup stale server_heartbeat background_task_state rows ([#9338](https://github.com/windmill-labs/windmill/issues/9338)) ([59ab038](https://github.com/windmill-labs/windmill/commit/59ab038d7718d8a4c25efa5928f42e1393ebbf40))
## [1.711.0](https://github.com/windmill-labs/windmill/compare/v1.710.1...v1.711.0) (2026-05-26)
+1
View File
@@ -66,6 +66,7 @@ RUN npm ci
COPY frontend .
RUN mkdir /backend
COPY /backend/windmill-api/openapi.yaml /backend/windmill-api/openapi.yaml
COPY /backend/oauth_connect.json /backend/oauth_connect.json
COPY /openflow.openapi.yaml /openflow.openapi.yaml
COPY /backend/windmill-api/build_openapi.sh /backend/windmill-api/build_openapi.sh
COPY /system_prompts/auto-generated /system_prompts/auto-generated
+22 -8
View File
@@ -56,7 +56,7 @@ bun run cli -- run flow flow-test4-order-processing-loop --model opus
bun run cli -- run flow flow-test0-sum-two-numbers --models haiku,opus,4o
bun run cli -- run flow flow-test0-sum-two-numbers --runs 3 --verbose
bun run cli -- run flow --record
GEMINI_API_KEY=... bun run cli -- run app app-test1-counter-create --model gemini-pro
GEMINI_API_KEY=... bun run cli -- run app app-test1-counter-create --model gemini-3-flash-preview
WMILL_AI_EVAL_BACKEND_URL=http://127.0.0.1:8000 bun run cli -- run flow --backend-validation preview
bun run cli -- run global global-test1-script-create
bun run cli -- run cli bun-hello-script
@@ -88,15 +88,16 @@ Today:
- `sonnet`
- `opus`
- `4o`
- `gemini-flash`
- `gemini-pro`
- `gpt-5.5`
- `gemini-3-flash-preview`
- `gemini-3.1-pro-preview`
- `deepseek-v4-flash`
- `deepseek-v4-pro`
Notes:
- the command also prints accepted alias spellings such as `gpt-4o`, `claude-opus-4.6`, and `claude-haiku-4.5`
- frontend modes (`flow`, `script`, `app`, `global`) can use Anthropic, OpenAI, and Gemini-backed aliases
- the command also prints accepted alias spellings such as `gpt-4o`, `gpt-55`, `claude-opus-4.6`, and `claude-haiku-4.5`
- frontend modes (`flow`, `script`, `app`, `global`) can use Anthropic, OpenAI, Gemini, and DeepSeek-backed aliases
- `cli` mode always uses the Anthropic agent SDK, so only Anthropic aliases are valid there
- the judge model is separate and currently defaults to `claude-sonnet-4-6`
@@ -142,6 +143,15 @@ For `global` mode, `validate` can express draft-level requirements such as:
- required or forbidden draft counts
- forbidden draft paths
Global initial fixtures can also seed `liveEditorDrafts` with `type`,
`storagePath`, `effectivePath`, and `value` fields. These drafts emulate the
currently open script, flow, or raw app editor so cases can test prompts that
refer to "this" or the "current" item.
Set `WMILL_AI_EVAL_DISABLE_ACTIVE_EDITOR_CONTEXT=1` to run those cases with
the old behavior where the live editor is only discoverable through
`list_workspace_items`.
App fixtures can also include an optional `datatables.json` file at the fixture root.
For `flow` mode, an `initial` fixture can also include a benchmark workspace catalog of
@@ -189,11 +199,15 @@ If `--record` is used, the CLI also appends one compact JSON line to:
Each recorded line contains:
- run metadata (`createdAt`, `gitSha`, `mode`, `runModel`, `judgeModel`)
- suite totals (`caseCount`, `attemptCount`, `passedAttempts`, `passRate`, `averageDurationMs`, `averageJudgeScore`)
- average token usage (`averageTokenUsagePerAttempt`)
- per-case metrics under `cases[]` (`averageDurationMs`, `averageJudgeScore`, `averageTokenUsagePerAttempt`, pass rate)
- suite totals (`caseCount`, `attemptCount`, `passedAttempts`, `passRate`, `averageDurationMs`, `averagePassedDurationMs`, `averageJudgeScore`)
- average token usage (`averageTokenUsagePerAttempt`, `averageTokenUsagePerPassedAttempt`)
- per-case metrics under `cases[]` (`averageDurationMs`, `averagePassedDurationMs`, `averageJudgeScore`, `averageTokenUsagePerAttempt`, `averageTokenUsagePerPassedAttempt`, pass rate)
- `failedCaseIds`
The CLI headline duration and token averages use passed attempts only.
All-attempt averages are still recorded to make failures auditable without
letting failed attempts skew success cost comparisons.
Example:
- summary: `ai_evals/results/2026-04-09T09-40-33.051Z__flow.json`
@@ -7,8 +7,12 @@ import {
prepareGlobalSystemMessage,
prepareGlobalUserMessage,
} from "../../../../../frontend/src/lib/components/copilot/chat/global/core";
import { globalDraftStore } from "../../../../../frontend/src/lib/components/copilot/chat/global/draftStore.svelte";
import {
clearGlobalDrafts,
listGlobalDrafts,
} from "../../../../../frontend/src/lib/components/copilot/chat/global/userDraftAdapter";
import type { Tool as ProductionTool } from "../../../../../frontend/src/lib/components/copilot/chat/shared";
import { UserDraft } from "../../../../../frontend/src/lib/userDraft.svelte";
import type { ModeRunContext } from "../../../../core/types";
import type { GlobalDraftState } from "../../../../core/validators";
import type { WindmillBackendSettings } from "../../../../core/windmillBackendSettings";
@@ -24,6 +28,21 @@ const MUTATING_GLOBAL_TOOLS = new Set([
"deploy_workspace_item",
"delete_workspace_item",
]);
const DISABLE_ACTIVE_EDITOR_CONTEXT_ENV =
"WMILL_AI_EVAL_DISABLE_ACTIVE_EDITOR_CONTEXT";
const LIVE_EDITOR_ITEM_KINDS = {
script: "script",
flow: "flow",
app: "raw_app",
} as const;
export interface GlobalLiveEditorDraftFixture {
type: keyof typeof LIVE_EDITOR_ITEM_KINDS;
storagePath?: string;
effectivePath?: string;
value?: unknown;
}
export interface GlobalEvalResult {
success: boolean;
@@ -38,6 +57,7 @@ export interface GlobalEvalResult {
export interface GlobalEvalOptions {
workspaceFixtures?: BenchmarkWorkspaceRunnables;
liveEditorDrafts?: GlobalLiveEditorDraftFixture[];
model?: string;
maxIterations?: number;
provider?: AIProvider;
@@ -55,19 +75,26 @@ export async function runGlobalEval(
options.workspaceRoot ??
(await mkdtemp(join(tmpdir(), "wmill-frontend-global-benchmark-")));
globalDraftStore.clearDrafts(workspaceRoot);
clearGlobalDrafts(workspaceRoot);
registerBenchmarkWorkspaceRunnables(workspaceRoot, options.workspaceFixtures ?? {});
seedLiveEditorDrafts(workspaceRoot, options.liveEditorDrafts ?? []);
try {
const model = options.model ?? "claude-haiku-4-5-20251001";
const injectActiveEditorContext =
process.env[DISABLE_ACTIVE_EDITOR_CONTEXT_ENV] !== "1";
const rawResult = await runEval({
userPrompt,
systemMessage: prepareGlobalSystemMessage(),
userMessage: prepareGlobalUserMessage(userPrompt),
userMessage: prepareGlobalUserMessage(
userPrompt,
[],
injectActiveEditorContext ? { workspace: workspaceRoot } : {},
),
tools: getGlobalEvalTools(),
helpers: {},
apiKey,
getOutput: () => ({ drafts: globalDraftStore.listDrafts(workspaceRoot) }),
getOutput: () => ({ drafts: listGlobalDrafts(workspaceRoot) }),
onAssistantMessageStart: options.runContext?.onAssistantMessageStart,
onAssistantToken: options.runContext?.onAssistantChunk,
onAssistantMessageEnd: options.runContext?.onAssistantMessageEnd,
@@ -94,7 +121,8 @@ export async function runGlobalEval(
tokenUsage: rawResult.tokenUsage,
};
} finally {
globalDraftStore.clearDrafts(workspaceRoot);
clearGlobalDrafts(workspaceRoot);
clearLiveEditorDrafts(workspaceRoot, options.liveEditorDrafts ?? []);
unregisterBenchmarkWorkspaceRunnables(workspaceRoot);
if (!options.workspaceRoot) {
await rm(workspaceRoot, { recursive: true, force: true });
@@ -102,6 +130,36 @@ export async function runGlobalEval(
}
}
function seedLiveEditorDrafts(
workspace: string,
fixtures: GlobalLiveEditorDraftFixture[],
): void {
for (const fixture of fixtures) {
const itemKind = LIVE_EDITOR_ITEM_KINDS[fixture.type];
const storagePath = fixture.storagePath ?? fixture.effectivePath ?? "";
if (fixture.value !== undefined) {
UserDraft.save(itemKind, storagePath, fixture.value, { workspace });
}
UserDraft.setLiveEditorDraft({
workspace,
itemKind,
storagePath,
effectivePath: fixture.effectivePath ?? fixture.storagePath,
});
}
}
function clearLiveEditorDrafts(
workspace: string,
fixtures: GlobalLiveEditorDraftFixture[],
): void {
for (const fixture of fixtures) {
const itemKind = LIVE_EDITOR_ITEM_KINDS[fixture.type];
const storagePath = fixture.storagePath ?? fixture.effectivePath ?? "";
UserDraft.clearLiveEditorDraft(itemKind, { workspace, storagePath });
}
}
function getGlobalEvalTools(): ProductionTool<{}>[] {
return (globalTools as ProductionTool<{}>[]).map((tool) => {
if (!MUTATING_GLOBAL_TOOLS.has(tool.def.function.name)) {
@@ -21,9 +21,9 @@ describe("proxy helpers", () => {
describe("resolveEvalModelProvider", () => {
it("infers googleai from Gemini model ids", () => {
expect(resolveEvalModelProvider("gemini-2.5-flash")).toEqual({
expect(resolveEvalModelProvider("gemini-3-flash-preview")).toEqual({
provider: "googleai",
model: "gemini-2.5-flash",
model: "gemini-3-flash-preview",
});
});
@@ -35,9 +35,11 @@ describe("resolveEvalModelProvider", () => {
});
it("preserves an explicit provider", () => {
expect(resolveEvalModelProvider("gemini-2.5-pro", "googleai")).toEqual({
expect(
resolveEvalModelProvider("gemini-3.1-pro-preview", "googleai"),
).toEqual({
provider: "googleai",
model: "gemini-2.5-pro",
model: "gemini-3.1-pro-preview",
});
});
});
@@ -79,6 +79,16 @@ vi.mock('$lib/gen', async () => {
}
return actual.ScriptService.getScriptByPath(data)
},
getScriptByPathWithDraft: async (data: { workspace: string; path: string }) => {
if (hasBenchmarkWorkspace(data.workspace)) {
const script = getBenchmarkScriptByPath(data.workspace, data.path)
if (!script) {
throw new Error(`Script "${data.path}" not found in benchmark workspace`)
}
return script
}
return actual.ScriptService.getScriptByPathWithDraft(data)
},
getScriptByHash: async (data: { workspace: string; hash: string }) => {
if (hasBenchmarkWorkspace(data.workspace)) {
const script = getBenchmarkScriptByHash(data.workspace, data.hash)
@@ -108,6 +118,26 @@ vi.mock('$lib/gen', async () => {
return flow
}
return actual.FlowService.getFlowByPath(data)
},
getFlowByPathWithDraft: async (data: { workspace: string; path: string }) => {
if (hasBenchmarkWorkspace(data.workspace)) {
const flow = getBenchmarkFlowByPath(data.workspace, data.path)
if (!flow) {
throw new Error(`Flow "${data.path}" not found in benchmark workspace`)
}
return flow
}
return actual.FlowService.getFlowByPathWithDraft(data)
},
getFlowLatestVersion: async (data: { workspace: string; path: string }) => {
if (hasBenchmarkWorkspace(data.workspace)) {
const flow = getBenchmarkFlowByPath(data.workspace, data.path)
if (!flow) {
throw new Error(`Flow "${data.path}" not found in benchmark workspace`)
}
return { id: 1 }
}
return actual.FlowService.getFlowLatestVersion(data)
}
}),
JobService: wrapService(actual.JobService, {
+11
View File
@@ -8,6 +8,9 @@
args:
a: 4
b: 5
toolExpect:
requiredToolsUsed:
- test_run_flow
judgeChecklist:
- "the flow takes `a` and `b` as inputs"
- "the main step is named `sum_numbers`"
@@ -25,6 +28,9 @@
args:
a: 2
b: 3
toolExpect:
requiredToolsUsed:
- test_run_flow
judgeChecklist:
- "the flow takes `a` and `b` as inputs"
- "the main step is named `sum_numbers`"
@@ -42,6 +48,9 @@
args:
a: 7
b: 8
toolExpect:
requiredToolsUsed:
- test_run_flow
judgeChecklist:
- "the parent flow takes `a` and `b` as inputs"
- "the main step is named `call_add_numbers`"
@@ -426,6 +435,7 @@
- return_schedule_status
toolExpect:
requiredToolsUsed:
- test_run_flow
- create_schedule
toolCallArgs:
- tool: create_schedule
@@ -453,6 +463,7 @@
- webhook_response
toolExpect:
requiredToolsUsed:
- test_run_flow
- create_trigger
toolCallArgs:
- tool: create_trigger
+531
View File
@@ -87,3 +87,534 @@
- the flow accepts numeric inputs a and b
- the flow returns the sum of a and b
- the result stays as an AI draft and is not deployed or saved to the workspace
- id: global-test4-multi-artifact-notification-job
prompt: |-
Set up a draft stale-trial notification job.
Create a Bun script at `f/evals/global/check_stale_trials` that accepts `max_age_days`, uses mocked inline trial account data, and returns the stale trial account IDs.
Also create a weekday 09:00 UTC schedule at `f/evals/global/check_stale_trials_weekday` for that script with `max_age_days` set to 14.
Add an HTTP POST trigger at `f/evals/global/check_stale_trials_manual` with route path `evals/check-stale-trials` that runs the same script manually.
Leave everything as AI drafts only; do not deploy or save anything to the workspace.
runtime:
maxTurns: 12
validate:
draftCountExactly: 3
requiredDrafts:
- type: script
path: f/evals/global/check_stale_trials
language: bun
valueIncludes:
- max_age_days
- trial
- type: schedule
path: f/evals/global/check_stale_trials_weekday
valueIncludes:
- f/evals/global/check_stale_trials
- UTC
- "14"
- type: trigger
triggerKind: http
path: f/evals/global/check_stale_trials_manual
valueIncludes:
- evals/check-stale-trials
- f/evals/global/check_stale_trials
toolExpect:
requiredToolsUsed:
- write_script
- write_schedule
- write_trigger
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- creates a Bun script draft for stale trial accounts
- creates a weekday 09:00 UTC schedule draft for the script with max_age_days set to 14
- creates an HTTP POST trigger draft with route path evals/check-stale-trials for the same script
- leaves all artifacts as drafts only and does not deploy
- id: global-test5-existing-flow-inline-code-edit
prompt: |-
Update the existing flow at `f/evals/global/process_invoice`.
Only change the `calculate_total` inline code so it applies 8% tax and returns an object containing `subtotal`, `tax`, and `total`.
Leave the updated flow as an AI draft only; do not deploy or save it.
initial: ai_evals/fixtures/frontend/global/initial/process_invoice_flow.json
runtime:
maxTurns: 10
validate:
draftCountExactly: 1
requiredDrafts:
- type: flow
path: f/evals/global/process_invoice
valueIncludes:
- calculate_total
- tax
- total
toolExpect:
requiredToolsUsed:
- read_workspace_item
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- reads the existing process_invoice flow before editing it
- updates the calculate_total inline code to apply 8% tax
- returns subtotal, tax, and total from the updated flow logic
- leaves the result as an AI draft only
- id: global-test6-secret-variable-draft
prompt: |-
Create a secret variable draft at `f/evals/global/slack_bot_token`.
Use the placeholder value `xoxb-redacted-test-token` and description `Slack bot token for eval notifications`.
Do not create any resource or deploy anything.
runtime:
maxTurns: 6
validate:
draftCountExactly: 1
requiredDrafts:
- type: variable
path: f/evals/global/slack_bot_token
valueIncludes:
- Slack bot token
- "true"
forbiddenDrafts:
- type: resource
path: f/evals/global/slack_bot_token
toolExpect:
requiredToolsUsed:
- write_variable
forbiddenToolsUsed:
- write_resource
- deploy_workspace_item
- delete_workspace_item
toolCallArgs:
- tool: write_variable
field: value
stringStartsWithAnyOf:
- xoxb-redacted-test-token
skipJudge: true
judgeChecklist:
- creates exactly one secret variable draft at f/evals/global/slack_bot_token
- uses the requested placeholder value and description
- does not create a resource or deploy anything
- id: global-test7-ambiguous-app-asks-question
prompt: |-
Create a new raw app for triaging support tickets.
runtime:
maxTurns: 4
validate:
draftCountExactly: 0
toolExpect:
requiredToolsUsed:
- askUserQuestion
forbiddenToolsUsed:
- init_app
- write_app_file
- write_app_runnable
- deploy_workspace_item
- delete_workspace_item
skipJudge: true
- id: global-test8-human-script-infer-path-language
prompt: |-
I need a small helper that formats a customer-facing welcome line.
It should take a person's name and return "Welcome aboard, <name>!".
Please just stage it as a draft for now.
runtime:
maxTurns: 8
validate:
draftCountExactly: 1
requiredDrafts:
- type: script
valueIncludes:
- Welcome aboard
- name
toolExpect:
requiredToolsUsed:
- write_script
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- creates a single script draft for a welcome-line helper
- accepts a person's name as input
- returns a message containing Welcome aboard, the provided name, and an exclamation mark
- chooses a reasonable workspace path and script language without needing the user to specify them
- leaves the result as an AI draft only
- id: global-test9-human-weekday-trial-job
prompt: |-
Can you set up a draft daily job that checks a few hard-coded trial accounts and returns the ones whose trial has ended?
It should run every weekday morning around 9 in UTC with a 30 day cutoff.
Keep it as draft work only.
runtime:
maxTurns: 10
validate:
draftCountExactly: 2
requiredDrafts:
- type: script
pathIncludes:
- trial
valueIncludes:
- trial
- "30"
- type: schedule
pathIncludes:
- trial
valueIncludes:
- UTC
toolExpect:
requiredToolsUsed:
- write_script
- write_schedule
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- creates a script draft that checks hard-coded trial accounts
- returns the accounts whose trial has ended based on a 30 day cutoff
- creates a schedule draft for weekday mornings around 09:00 UTC
- links the schedule to the generated script
- leaves both artifacts as drafts only
- id: global-test10-human-secret-variable
prompt: |-
I need a placeholder Slack bot token stored securely for future notification work.
Use xoxb-redacted-test-token and note that it is for eval notifications.
Only prepare a draft.
runtime:
maxTurns: 6
validate:
draftCountExactly: 1
requiredDrafts:
- type: variable
pathIncludes:
- slack
valueIncludes:
- eval notifications
- "true"
toolExpect:
requiredToolsUsed:
- write_variable
forbiddenToolsUsed:
- write_resource
- deploy_workspace_item
- delete_workspace_item
toolCallArgs:
- tool: write_variable
field: value
stringStartsWithAnyOf:
- xoxb-redacted-test-token
skipJudge: true
judgeChecklist:
- creates a single secret variable draft for the Slack bot token placeholder
- uses the requested placeholder value
- includes a note or description that it is for eval notifications
- does not create a resource or deploy anything
- id: global-test11-human-existing-flow-informal-edit
prompt: |-
There is an invoice processing flow in this workspace.
Can you adjust its total calculation so it adds 8% tax and returns subtotal, tax, and total?
Keep the change as a draft.
initial: ai_evals/fixtures/frontend/global/initial/process_invoice_flow.json
runtime:
maxTurns: 10
validate:
draftCountExactly: 1
requiredDrafts:
- type: flow
pathIncludes:
- invoice
valueIncludes:
- calculate_total
- tax
- total
toolExpect:
requiredToolsUsed:
- read_workspace_item
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- finds and edits the existing invoice processing flow without the user providing its exact path
- updates the total calculation to apply 8% tax
- returns subtotal, tax, and total from the updated flow logic
- leaves the result as an AI draft only
- id: global-test12-current-live-script-edit
prompt: |-
The script I have open formats greetings.
Can you update this script so it uppercases the name before greeting them and ends with an exclamation mark?
Keep it as draft work.
initial: ai_evals/fixtures/frontend/global/initial/current_greeting_live_script.json
runtime:
maxTurns: 8
validate:
draftCountExactly: 1
requiredDrafts:
- type: script
path: f/evals/global/current_greeting
language: bun
valueIncludes:
- toUpperCase
- "!"
forbiddenDrafts:
- type: script
path: f/evals/global/format_greeting
- type: script
path: f/evals/global/format_greeting_archive
toolExpect:
requiredToolsUsed:
- read_workspace_item
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- resolves "this script" to the active live editor script instead of another similarly named workspace script
- updates the greeting logic to uppercase the provided name
- returns a greeting ending with an exclamation mark
- leaves the result as a draft only
- id: global-test13-current-live-flow-edit
prompt: |-
I have the invoice flow open.
In the current flow, update the total calculation to add 8% tax and return subtotal, tax, and total.
Keep the change as a draft.
initial: ai_evals/fixtures/frontend/global/initial/current_invoice_live_flow.json
runtime:
maxTurns: 10
validate:
draftCountExactly: 1
requiredDrafts:
- type: flow
path: f/evals/global/current_invoice_flow
valueIncludes:
- calculate_total
- tax
- total
forbiddenDrafts:
- type: flow
path: f/evals/global/process_invoice
- type: flow
path: f/evals/global/process_refund
toolExpect:
requiredToolsUsed:
- read_workspace_item
forbiddenToolsUsed:
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- resolves "current flow" to the active live editor flow
- does not edit the similarly named deployed invoice or refund flows
- updates the calculate_total logic to apply 8% tax
- returns subtotal, tax, and total from the updated flow logic
- leaves the result as a draft only
- id: global-test14-current-without-live-editor-asks-question
prompt: |-
Please update this script so it returns `ok`.
Keep it as a draft.
runtime:
maxTurns: 4
validate:
draftCountExactly: 0
toolExpect:
forbiddenToolsUsed:
- write_script
- edit_script
- write_flow
- deploy_workspace_item
- delete_workspace_item
skipJudge: true
judgeChecklist:
- asks which script to update when the user refers to "this script" without selected or active editor context
- does not guess a path or create a new script draft
- id: global-test15-human-postgres-resource
prompt: |-
I'm wiring the eval reporting database into this workspace.
Can you stage a Postgres connection for it in the shared evals/global folder?
Use host `reports-db.internal`, port 5432, database `evals_reporting`, user `report_reader`, and password `pg-redacted-reporting-password`.
Keep the credentials safe.
This is just draft work for now.
runtime:
maxTurns: 10
validate:
draftCountExactly: 2
requiredDrafts:
- type: variable
pathStartsWith: f/evals/global/
pathIncludes:
- evals
- global
- report
- password
valueIncludes:
- "true"
- report
- type: resource
pathStartsWith: f/evals/global/
pathIncludes:
- evals
- global
- report
valueIncludes:
- postgres
- reports-db.internal
- "5432"
- evals_reporting
- report_reader
- "$var:"
valueExcludes:
- pg-redacted-reporting-password
toolExpect:
requiredToolsUsed:
- write_variable
- search_resource_types
- write_resource
forbiddenToolsUsed:
- write_schedule
- write_trigger
- deploy_workspace_item
- delete_workspace_item
toolCallArgs:
- tool: write_variable
field: value
stringStartsWithAnyOf:
- pg-redacted-reporting-password
skipJudge: true
judgeChecklist:
- creates a Postgres resource draft for the eval reporting database
- creates a secret variable draft for the database password
- puts the drafts in sensible eval/global reporting-related paths
- uses the requested host, port, database, and user
- references the secret variable from the resource instead of embedding the password
- leaves the work as a draft only
- id: global-test16-human-visible-variable
prompt: |-
We keep reusing a 30 day trial cutoff in eval notification jobs.
Can you stage that as a normal workspace variable in the shared evals/global folder, with a short description so people know what it controls?
It is not a secret.
runtime:
maxTurns: 6
validate:
draftCountExactly: 1
requiredDrafts:
- type: variable
pathStartsWith: f/evals/global/
pathIncludes:
- evals
- global
- trial
valueIncludes:
- "30"
- "false"
- trial
toolExpect:
requiredToolsUsed:
- write_variable
forbiddenToolsUsed:
- write_resource
- write_schedule
- write_trigger
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- creates exactly one non-secret variable draft for the trial cutoff
- stores the value 30
- chooses a sensible eval/global path related to trials or notifications
- includes a useful description of what the value controls
- does not create resources, schedules, triggers, or deployed workspace changes
- id: global-test17-human-schedule-existing-helper
prompt: |-
The workspace already has a report digest helper.
Can you stage a weekday 8:30 AM UTC run for it with `dry_run` turned on?
I only want the schedule draft for review.
initial: ai_evals/fixtures/frontend/global/initial/report_digest_script.json
runtime:
maxTurns: 8
validate:
draftCountExactly: 1
requiredDrafts:
- type: schedule
pathIncludes:
- digest
valueIncludes:
- f/evals/global/send_report_digest
- UTC
- dry_run
- "true"
toolExpect:
requiredToolsUsed:
- list_workspace_items
- write_schedule
forbiddenToolsUsed:
- write_script
- write_flow
- write_resource
- write_variable
- write_trigger
- deploy_workspace_item
- delete_workspace_item
judgeChecklist:
- finds the existing report digest helper rather than creating a new script or flow
- creates one schedule draft for that helper
- schedules it for weekdays around 08:30 UTC
- passes dry_run as true
- leaves only the schedule draft for review
- id: global-test18-human-slack-resource-with-secret
prompt: |-
I'm preparing Slack notifications for eval failures.
Can you stage a Slack connection in the shared evals/global folder?
The bot token is `xoxb-redacted-test-token`; keep it safe.
Don't deploy anything yet.
runtime:
maxTurns: 8
validate:
draftCountExactly: 2
requiredDrafts:
- type: variable
pathStartsWith: f/evals/global/
pathIncludes:
- evals
- global
- slack
- token
valueIncludes:
- "true"
- type: resource
pathStartsWith: f/evals/global/
pathIncludes:
- evals
- global
- slack
valueIncludes:
- slack
- "$var:"
valueExcludes:
- xoxb-redacted-test-token
toolExpect:
requiredToolsUsed:
- write_variable
- search_resource_types
- write_resource
forbiddenToolsUsed:
- write_schedule
- write_trigger
- deploy_workspace_item
- delete_workspace_item
toolCallArgs:
- tool: write_variable
field: value
stringStartsWithAnyOf:
- xoxb-redacted-test-token
skipJudge: true
judgeChecklist:
- creates a secret variable draft for the Slack bot token placeholder
- creates a Slack resource draft that references the secret variable instead of embedding the token
- keeps both drafts under a sensible eval/global Slack-related path
- does not create schedules, triggers, or deployed workspace changes
+5
View File
@@ -5,6 +5,9 @@
Keep it simple and do not add external dependencies.
initial: ai_evals/fixtures/frontend/script/initial/test1_empty_bun.json
expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json
toolExpect:
requiredToolsUsed:
- test_run_script
judgeChecklist:
- uses the existing `name` input
- returns a plain greeting string
@@ -20,6 +23,7 @@
expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json
toolExpect:
requiredToolsUsed:
- test_run_script
- create_schedule
toolCallArgs:
- tool: create_schedule
@@ -44,6 +48,7 @@
expected: ai_evals/fixtures/frontend/script/expected/test1_greet_user.json
toolExpect:
requiredToolsUsed:
- test_run_script
- create_trigger
toolCallArgs:
- tool: create_trigger
+7 -3
View File
@@ -211,7 +211,7 @@ async function handleRun(input: {
const summaries: Array<{
label: string;
passRate: number;
averageDurationMs: number;
averagePassedDurationMs: number | null;
}> = [];
for (const [index, model] of models.entries()) {
@@ -259,7 +259,7 @@ async function handleRun(input: {
summaries.push({
label: `${model.id} (${runModel})`,
passRate: result.passRate,
averageDurationMs: result.averageDurationMs,
averagePassedDurationMs: result.averagePassedDurationMs ?? null,
});
}
@@ -267,7 +267,7 @@ async function handleRun(input: {
process.stdout.write("\nModel summary\n");
for (const summary of summaries) {
process.stdout.write(
`- ${summary.label}: ${formatPercent(summary.passRate)} | ${Math.round(summary.averageDurationMs)}ms\n`,
`- ${summary.label}: ${formatPercent(summary.passRate)} | passed avg ${formatNullableDuration(summary.averagePassedDurationMs)}\n`,
);
}
}
@@ -351,6 +351,10 @@ function formatPercent(value: number): string {
return `${(value * 100).toFixed(1)}%`;
}
function formatNullableDuration(value: number | null): string {
return value === null ? "n/a" : `${Math.round(value)}ms`;
}
void main().catch((error) => {
const message = error instanceof Error ? error.message : String(error);
process.stderr.write(`${message}\n`);
+44 -1
View File
@@ -14,6 +14,21 @@ describe("loadCases", () => {
},
},
});
expect(caseEntry?.toolExpect).toEqual({
requiredToolsUsed: ["test_run_flow"],
});
});
it("loads script and flow test tool expectations", async () => {
const scriptCases = await loadCases("script");
const flowCases = await loadCases("flow");
expect(scriptCases.find((entry) => entry.id === "script-test1-greet-user")?.toolExpect).toEqual({
requiredToolsUsed: ["test_run_script"],
});
expect(flowCases.find((entry) => entry.id === "flow-test0-sum-two-numbers")?.toolExpect).toEqual({
requiredToolsUsed: ["test_run_flow"],
});
});
it("loads the workspace-flow preference benchmark case", async () => {
@@ -203,6 +218,34 @@ describe("loadCases", () => {
});
});
it("loads global active-editor eval cases", async () => {
const globalCases = await loadCases("global");
const scriptCase = globalCases.find(
(entry) => entry.id === "global-test12-current-live-script-edit"
);
const flowCase = globalCases.find(
(entry) => entry.id === "global-test13-current-live-flow-edit"
);
expect(scriptCase?.initialPath).toContain(
"ai_evals/fixtures/frontend/global/initial/current_greeting_live_script.json"
);
expect(scriptCase?.toolExpect).toMatchObject({
requiredToolsUsed: ["read_workspace_item"],
});
expect(flowCase?.initialPath).toContain(
"ai_evals/fixtures/frontend/global/initial/current_invoice_live_flow.json"
);
expect(flowCase?.validate).toMatchObject({
requiredDrafts: [
{
type: "flow",
path: "f/evals/global/current_invoice_flow",
},
],
});
});
it("loads tool expectations for workspace mutation cases", async () => {
const scriptCases = await loadCases("script");
const caseEntry = scriptCases.find(
@@ -210,7 +253,7 @@ describe("loadCases", () => {
);
expect(caseEntry?.toolExpect).toEqual({
requiredToolsUsed: ["create_schedule"],
requiredToolsUsed: ["test_run_script", "create_schedule"],
toolCallArgs: [
{
tool: "create_schedule",
+17 -10
View File
@@ -2,15 +2,22 @@ import { describe, expect, it } from "bun:test";
import { resolveEvalModel } from "./models";
describe("resolveEvalModel", () => {
it("supports GPT-5.5 aliases for frontend evals", () => {
expect(resolveEvalModel("flow", "gpt-5.5").frontend).toEqual({
provider: "openai",
model: "gpt-5.5",
});
expect(resolveEvalModel("app", "gpt-55").frontend).toEqual({
provider: "openai",
model: "gpt-5.5",
});
expect(resolveEvalModel("script", "5.5").frontend).toEqual({
provider: "openai",
model: "gpt-5.5",
});
});
it("supports Gemini aliases for frontend evals", () => {
expect(resolveEvalModel("flow", "gemini").frontend).toEqual({
provider: "googleai",
model: "gemini-2.5-flash",
});
expect(resolveEvalModel("app", "gemini-pro").frontend).toEqual({
provider: "googleai",
model: "gemini-2.5-pro",
});
expect(
resolveEvalModel("script", "gemini-3-flash-preview").frontend,
).toEqual({
@@ -37,8 +44,8 @@ describe("resolveEvalModel", () => {
});
it("rejects Gemini aliases for cli evals", () => {
expect(() => resolveEvalModel("cli", "gemini")).toThrow(
"Model gemini-flash is not supported for cli mode",
expect(() => resolveEvalModel("cli", "gemini-3-flash-preview")).toThrow(
"Model gemini-3-flash-preview is not supported for cli mode",
);
});
});
+5 -14
View File
@@ -88,21 +88,12 @@ export const EVAL_MODELS: EvalModelSpec[] = [
},
},
{
id: "gemini-flash",
label: "Gemini 2.5 Flash",
aliases: ["gemini", "gemini-flash", "gemini-2.5-flash"],
id: "gpt-5.5",
label: "GPT-5.5",
aliases: ["gpt-5.5", "gpt-55", "5.5"],
frontend: {
provider: "googleai",
model: "gemini-2.5-flash",
},
},
{
id: "gemini-pro",
label: "Gemini 2.5 Pro",
aliases: ["gemini-pro", "gemini-2.5-pro"],
frontend: {
provider: "googleai",
model: "gemini-2.5-pro",
provider: "openai",
model: "gpt-5.5",
},
},
{
+242
View File
@@ -0,0 +1,242 @@
import { mkdtemp, readFile, rm } from "node:fs/promises";
import { join } from "node:path";
import { tmpdir } from "node:os";
import { describe, expect, it } from "bun:test";
import {
appendHistoryRecord,
buildRunResult,
formatRunSummary,
} from "./results";
import type { BenchmarkCaseResult } from "./types";
function caseResult(
attempts: BenchmarkCaseResult["attempts"],
): BenchmarkCaseResult {
return {
id: "case-1",
prompt: "Do the thing",
attempts,
};
}
describe("benchmark results", () => {
it("keeps success cost metrics separate from failed attempts", () => {
const result = buildRunResult({
mode: "global",
runs: 1,
runModel: "model-under-test",
judgeModel: "judge-model",
caseResults: [
caseResult([
{
attempt: 1,
passed: true,
durationMs: 1000,
assistantMessageCount: 1,
toolCallCount: 1,
toolsUsed: ["edit_script"],
skillsInvoked: [],
checks: [{ name: "edited", passed: true }],
judgeScore: 100,
judgeSummary: "ok",
error: null,
tokenUsage: { prompt: 100, completion: 20, total: 120 },
},
{
attempt: 2,
passed: false,
durationMs: 100,
assistantMessageCount: 1,
toolCallCount: 0,
toolsUsed: [],
skillsInvoked: [],
checks: [{ name: "edited", passed: false }],
judgeScore: 10,
judgeSummary: "missed",
error: "failed",
tokenUsage: { prompt: 10, completion: 5, total: 15 },
},
]),
],
});
expect(result.attemptCount).toBe(2);
expect(result.passedAttempts).toBe(1);
expect(result.passRate).toBe(0.5);
expect(result.averageDurationMs).toBe(550);
expect(result.averagePassedDurationMs).toBe(1000);
expect(result.totalTokenUsage).toEqual({
prompt: 110,
completion: 25,
total: 135,
});
expect(result.totalPassedTokenUsage).toEqual({
prompt: 100,
completion: 20,
total: 120,
});
expect(result.averageTokenUsagePerAttempt).toEqual({
prompt: 55,
completion: 12.5,
total: 67.5,
});
expect(result.averageTokenUsagePerPassedAttempt).toEqual({
prompt: 100,
completion: 20,
total: 120,
});
const summary = formatRunSummary(result);
expect(summary).toContain("Average duration (passed): 1000ms");
expect(summary).toContain("Average tokens (passed): 120 total");
expect(summary).toContain("Average duration (all attempts): 550ms");
});
it("reports passed averages as unavailable when no attempt passes", () => {
const result = buildRunResult({
mode: "global",
runs: 1,
runModel: "model-under-test",
judgeModel: "judge-model",
caseResults: [
caseResult([
{
attempt: 1,
passed: false,
durationMs: 100,
assistantMessageCount: 1,
toolCallCount: 0,
toolsUsed: [],
skillsInvoked: [],
checks: [{ name: "edited", passed: false }],
judgeScore: 10,
judgeSummary: "missed",
error: "failed",
tokenUsage: { prompt: 10, completion: 5, total: 15 },
},
]),
],
});
expect(result.averagePassedDurationMs).toBeNull();
expect(result.totalPassedTokenUsage).toBeNull();
expect(result.averageTokenUsagePerPassedAttempt).toBeNull();
expect(formatRunSummary(result)).toContain(
"Average duration (passed): n/a",
);
});
it("normalizes passed token averages by passed attempts", () => {
const result = buildRunResult({
mode: "global",
runs: 1,
runModel: "model-under-test",
judgeModel: "judge-model",
caseResults: [
caseResult([
{
attempt: 1,
passed: true,
durationMs: 1000,
assistantMessageCount: 1,
toolCallCount: 1,
toolsUsed: ["edit_script"],
skillsInvoked: [],
checks: [{ name: "edited", passed: true }],
judgeScore: 100,
judgeSummary: "ok",
error: null,
tokenUsage: { prompt: 100, completion: 20, total: 120 },
},
{
attempt: 2,
passed: true,
durationMs: 1200,
assistantMessageCount: 1,
toolCallCount: 1,
toolsUsed: ["edit_script"],
skillsInvoked: [],
checks: [{ name: "edited", passed: true }],
judgeScore: 100,
judgeSummary: "ok",
error: null,
tokenUsage: null,
},
]),
],
});
expect(result.passedAttempts).toBe(2);
expect(result.totalPassedTokenUsage).toEqual({
prompt: 100,
completion: 20,
total: 120,
});
expect(result.averageTokenUsagePerPassedAttempt).toEqual({
prompt: 50,
completion: 10,
total: 60,
});
});
it("records passed-attempt metrics in history", async () => {
const tempDir = await mkdtemp(join(tmpdir(), "windmill-ai-evals-"));
try {
const historyPath = join(tempDir, "history.jsonl");
const result = buildRunResult({
mode: "global",
runs: 1,
runModel: "model-under-test",
judgeModel: "judge-model",
caseResults: [
caseResult([
{
attempt: 1,
passed: true,
durationMs: 1000,
assistantMessageCount: 1,
toolCallCount: 1,
toolsUsed: ["edit_script"],
skillsInvoked: [],
checks: [{ name: "edited", passed: true }],
judgeScore: 100,
judgeSummary: "ok",
error: null,
tokenUsage: { prompt: 100, completion: 20, total: 120 },
},
{
attempt: 2,
passed: false,
durationMs: 100,
assistantMessageCount: 1,
toolCallCount: 0,
toolsUsed: [],
skillsInvoked: [],
checks: [{ name: "edited", passed: false }],
judgeScore: 10,
judgeSummary: "missed",
error: "failed",
tokenUsage: { prompt: 10, completion: 5, total: 15 },
},
]),
],
});
await appendHistoryRecord(result, historyPath);
const record = JSON.parse(await readFile(historyPath, "utf8"));
expect(record.averageDurationMs).toBe(550);
expect(record.averagePassedDurationMs).toBe(1000);
expect(record.averageTokenUsagePerAttempt.total).toBe(67.5);
expect(record.averageTokenUsagePerPassedAttempt.total).toBe(120);
expect(record.cases[0].averageDurationMs).toBe(550);
expect(record.cases[0].averagePassedDurationMs).toBe(1000);
expect(record.cases[0].averageTokenUsagePerAttempt.total).toBe(67.5);
expect(record.cases[0].averageTokenUsagePerPassedAttempt.total).toBe(
120,
);
} finally {
await rm(tempDir, { recursive: true, force: true });
}
});
});
+114 -67
View File
@@ -4,12 +4,20 @@ import { execFileSync } from "node:child_process";
import { getAiEvalsRoot, getRepoRoot } from "./cases";
import type {
BenchmarkArtifactFile,
BenchmarkAttemptResult,
BenchmarkCaseResult,
BenchmarkRunResult,
BenchmarkTokenUsage,
EvalMode,
} from "./types";
type AttemptAggregate = {
attemptCount: number;
durationTotal: number;
tokenUsageAttemptCount: number;
tokenUsageTotal: BenchmarkTokenUsage | null;
};
export async function writeRunResult(
result: BenchmarkRunResult,
outputPath?: string,
@@ -77,36 +85,12 @@ export function buildRunResult(input: {
judgeModel: string | null;
caseResults: BenchmarkCaseResult[];
}): BenchmarkRunResult {
const attemptCount = input.caseResults.reduce(
(sum, entry) => sum + entry.attempts.length,
0,
);
const passedAttempts = input.caseResults.reduce(
(sum, entry) =>
sum + entry.attempts.filter((attempt) => attempt.passed).length,
0,
);
const durationTotal = input.caseResults.reduce(
(sum, entry) =>
sum +
entry.attempts.reduce((inner, attempt) => inner + attempt.durationMs, 0),
0,
);
const tokenUsageTotal = input.caseResults.reduce<BenchmarkTokenUsage | null>(
(sum, entry) => {
for (const attempt of entry.attempts) {
if (!attempt.tokenUsage) {
continue;
}
sum ??= { prompt: 0, completion: 0, total: 0 };
sum.prompt += attempt.tokenUsage.prompt;
sum.completion += attempt.tokenUsage.completion;
sum.total += attempt.tokenUsage.total;
}
return sum;
},
null,
);
const attempts = input.caseResults.flatMap((entry) => entry.attempts);
const passedAttemptResults = attempts.filter((attempt) => attempt.passed);
const attemptAggregate = aggregateAttempts(attempts);
const passedAttemptAggregate = aggregateAttempts(passedAttemptResults);
const attemptCount = attemptAggregate.attemptCount;
const passedAttempts = passedAttemptAggregate.attemptCount;
return {
version: 1,
@@ -120,16 +104,19 @@ export function buildRunResult(input: {
attemptCount,
passedAttempts,
passRate: attemptCount === 0 ? 0 : passedAttempts / attemptCount,
averageDurationMs: attemptCount === 0 ? 0 : durationTotal / attemptCount,
totalTokenUsage: tokenUsageTotal,
averageDurationMs:
attemptCount === 0 ? 0 : attemptAggregate.durationTotal / attemptCount,
averagePassedDurationMs: averageDuration(passedAttemptAggregate),
totalTokenUsage: attemptAggregate.tokenUsageTotal,
totalPassedTokenUsage: passedAttemptAggregate.tokenUsageTotal,
averageTokenUsagePerAttempt:
attemptCount === 0 || !tokenUsageTotal
attemptCount === 0
? null
: {
prompt: tokenUsageTotal.prompt / attemptCount,
completion: tokenUsageTotal.completion / attemptCount,
total: tokenUsageTotal.total / attemptCount,
},
: averageTokenUsage(attemptAggregate, attemptCount),
averageTokenUsagePerPassedAttempt: averageTokenUsage(
passedAttemptAggregate,
passedAttempts,
),
cases: input.caseResults,
};
}
@@ -138,9 +125,25 @@ export function formatRunSummary(result: BenchmarkRunResult): string {
const lines = [
`${result.mode} benchmark complete`,
`Pass rate: ${formatPercent(result.passRate)} (${result.passedAttempts}/${result.attemptCount})`,
`Average duration: ${Math.round(result.averageDurationMs)}ms`,
`Average duration (passed): ${formatNullableDuration(result.averagePassedDurationMs ?? null)}`,
];
if (result.averageTokenUsagePerPassedAttempt) {
lines.push(
`Average tokens (passed): ${formatTokenUsage(result.averageTokenUsagePerPassedAttempt)}`,
);
}
if (result.passedAttempts < result.attemptCount) {
lines.push(
`Average duration (all attempts): ${Math.round(result.averageDurationMs)}ms`,
);
if (result.averageTokenUsagePerAttempt) {
lines.push(
`Average tokens (all attempts): ${formatTokenUsage(result.averageTokenUsagePerAttempt)}`,
);
}
}
const failures = collectFailures(result);
if (failures.length > 0) {
lines.push("Failures:");
@@ -172,6 +175,60 @@ function collectFailures(result: BenchmarkRunResult): string[] {
return failures;
}
function aggregateAttempts(attempts: BenchmarkAttemptResult[]): AttemptAggregate {
const aggregate: AttemptAggregate = {
attemptCount: attempts.length,
durationTotal: 0,
tokenUsageAttemptCount: 0,
tokenUsageTotal: null,
};
for (const attempt of attempts) {
aggregate.durationTotal += attempt.durationMs;
if (!attempt.tokenUsage) {
continue;
}
aggregate.tokenUsageAttemptCount += 1;
aggregate.tokenUsageTotal ??= { prompt: 0, completion: 0, total: 0 };
aggregate.tokenUsageTotal.prompt += attempt.tokenUsage.prompt;
aggregate.tokenUsageTotal.completion += attempt.tokenUsage.completion;
aggregate.tokenUsageTotal.total += attempt.tokenUsage.total;
}
return aggregate;
}
function averageDuration(aggregate: AttemptAggregate): number | null {
return aggregate.attemptCount === 0
? null
: aggregate.durationTotal / aggregate.attemptCount;
}
function averageTokenUsage(
aggregate: AttemptAggregate,
denominator: number,
): BenchmarkTokenUsage | null {
if (denominator === 0 || !aggregate.tokenUsageTotal) {
return null;
}
return {
prompt: aggregate.tokenUsageTotal.prompt / denominator,
completion: aggregate.tokenUsageTotal.completion / denominator,
total: aggregate.tokenUsageTotal.total / denominator,
};
}
function formatNullableDuration(value: number | null): string {
return value === null ? "n/a" : `${Math.round(value)}ms`;
}
function formatTokenUsage(value: BenchmarkTokenUsage): string {
const total = Math.round(value.total);
const prompt = Math.round(value.prompt);
const completion = Math.round(value.completion);
return `${total} total (${prompt} prompt, ${completion} completion)`;
}
function defaultFileName(mode: EvalMode): string {
return `${new Date().toISOString().replaceAll(":", "-")}__${mode}.json`;
}
@@ -252,12 +309,15 @@ function toHistoryRecord(result: BenchmarkRunResult) {
passedAttempts: result.passedAttempts,
passRate: result.passRate,
averageDurationMs: result.averageDurationMs,
averagePassedDurationMs: result.averagePassedDurationMs ?? null,
averageJudgeScore:
judgeScores.length === 0
? null
: judgeScores.reduce((sum, score) => sum + score, 0) /
judgeScores.length,
averageTokenUsagePerAttempt: result.averageTokenUsagePerAttempt ?? null,
averageTokenUsagePerPassedAttempt:
result.averageTokenUsagePerPassedAttempt ?? null,
failedCaseIds: Array.from(
new Set(
result.cases
@@ -268,31 +328,15 @@ function toHistoryRecord(result: BenchmarkRunResult) {
),
),
cases: result.cases.map((caseResult) => {
const attemptCount = caseResult.attempts.length;
const passedAttempts = caseResult.attempts.filter(
(attempt) => attempt.passed,
).length;
const totalDurationMs = caseResult.attempts.reduce(
(sum, attempt) => sum + attempt.durationMs,
0,
const attemptAggregate = aggregateAttempts(caseResult.attempts);
const passedAttemptAggregate = aggregateAttempts(
caseResult.attempts.filter((attempt) => attempt.passed),
);
const attemptCount = attemptAggregate.attemptCount;
const passedAttempts = passedAttemptAggregate.attemptCount;
const judgeScores = caseResult.attempts.flatMap((attempt) =>
typeof attempt.judgeScore === "number" ? [attempt.judgeScore] : [],
);
const totalTokenUsage =
caseResult.attempts.reduce<BenchmarkTokenUsage | null>(
(sum, attempt) => {
if (!attempt.tokenUsage) {
return sum;
}
sum ??= { prompt: 0, completion: 0, total: 0 };
sum.prompt += attempt.tokenUsage.prompt;
sum.completion += attempt.tokenUsage.completion;
sum.total += attempt.tokenUsage.total;
return sum;
},
null,
);
return {
id: caseResult.id,
@@ -300,20 +344,23 @@ function toHistoryRecord(result: BenchmarkRunResult) {
passedAttempts,
passRate: attemptCount === 0 ? 0 : passedAttempts / attemptCount,
averageDurationMs:
attemptCount === 0 ? 0 : totalDurationMs / attemptCount,
attemptCount === 0
? 0
: attemptAggregate.durationTotal / attemptCount,
averagePassedDurationMs: averageDuration(passedAttemptAggregate),
averageJudgeScore:
judgeScores.length === 0
? null
: judgeScores.reduce((sum, score) => sum + score, 0) /
judgeScores.length,
averageTokenUsagePerAttempt:
attemptCount === 0 || !totalTokenUsage
attemptCount === 0
? null
: {
prompt: totalTokenUsage.prompt / attemptCount,
completion: totalTokenUsage.completion / attemptCount,
total: totalTokenUsage.total / attemptCount,
},
: averageTokenUsage(attemptAggregate, attemptCount),
averageTokenUsagePerPassedAttempt: averageTokenUsage(
passedAttemptAggregate,
passedAttempts,
),
};
}),
};
+6 -1
View File
@@ -110,7 +110,9 @@ export interface AppValidationSpec {
export interface GlobalDraftRequirement {
type: string;
path: string;
path?: string;
pathIncludes?: string[];
pathStartsWith?: string;
triggerKind?: string;
language?: string;
summaryIncludes?: string[];
@@ -324,8 +326,11 @@ export interface BenchmarkRunResult {
passedAttempts: number;
passRate: number;
averageDurationMs: number;
averagePassedDurationMs?: number | null;
totalTokenUsage?: BenchmarkTokenUsage | null;
totalPassedTokenUsage?: BenchmarkTokenUsage | null;
averageTokenUsagePerAttempt?: BenchmarkTokenUsage | null;
averageTokenUsagePerPassedAttempt?: BenchmarkTokenUsage | null;
artifactsPath?: string | null;
cases: BenchmarkCaseResult[];
}
+63
View File
@@ -195,6 +195,69 @@ describe("validateGlobalState", () => {
});
});
it("accepts a required script draft without an exact path", () => {
const checks = validateGlobalState({
actual: {
drafts: [
{
type: "script",
path: "f/team_tools/friendly_greeting",
language: "bun",
summary: "Friendly greeting helper",
value:
"export async function main(name: string) {\n return `Hello, ${name}!`\n}\n",
isDraft: true,
},
],
},
validate: {
draftCountExactly: 1,
requiredDrafts: [
{
type: "script",
pathIncludes: ["greeting"],
language: "bun",
summaryIncludes: ["Friendly"],
valueIncludes: ["Hello"],
},
],
},
});
expect(checks.every((check) => check.passed)).toBe(true);
});
it("reports flexible global draft path filters when no draft matches", () => {
const checks = validateGlobalState({
actual: {
drafts: [
{
type: "script",
path: "f/team_tools/friendly_greeting",
language: "bun",
value:
"export async function main(name: string) {\n return `Hello, ${name}!`\n}\n",
isDraft: true,
},
],
},
validate: {
requiredDrafts: [
{
type: "script",
pathIncludes: ["invoice"],
},
],
},
});
expect(checks).toContainEqual({
name: "global includes script draft (path includes invoice)",
passed: false,
details: "drafts: script:f/team_tools/friendly_greeting",
});
});
it("does not require a TypeScript entrypoint for non-TypeScript script drafts", () => {
const checks = validateGlobalState({
actual: {
+100 -15
View File
@@ -315,10 +315,11 @@ export function validateGlobalState(input: {
}
for (const required of validate.requiredDrafts ?? []) {
const draft = findGlobalDraft(drafts, required.type, required.path, required.triggerKind);
const requirementLabel = formatGlobalDraftRequirement(required);
const draft = findGlobalDraft(drafts, required);
checks.push(
check(
`global includes ${required.type} draft ${required.path}`,
`global includes ${requirementLabel}`,
Boolean(draft),
summarizeGlobalDrafts(drafts)
)
@@ -330,7 +331,7 @@ export function validateGlobalState(input: {
if (required.language !== undefined) {
checks.push(
check(
`${required.type} draft ${required.path} uses ${required.language}`,
`${requirementLabel} uses ${required.language}`,
draft.language === required.language,
`language=${draft.language ?? "(none)"}`
)
@@ -340,7 +341,7 @@ export function validateGlobalState(input: {
for (const snippet of required.summaryIncludes ?? []) {
checks.push(
check(
`${required.type} draft ${required.path} summary includes '${snippet}'`,
`${requirementLabel} summary includes '${snippet}'`,
normalizeText(draft.summary ?? "").includes(normalizeText(snippet)),
`summary=${draft.summary ?? ""}`
)
@@ -351,7 +352,7 @@ export function validateGlobalState(input: {
for (const snippet of required.valueIncludes ?? []) {
checks.push(
check(
`${required.type} draft ${required.path} value includes '${snippet}'`,
`${requirementLabel} value includes '${snippet}'`,
normalizeText(valueText).includes(normalizeText(snippet)),
truncateForDetails(valueText)
)
@@ -361,7 +362,7 @@ export function validateGlobalState(input: {
for (const snippet of required.valueExcludes ?? []) {
checks.push(
check(
`${required.type} draft ${required.path} value excludes '${snippet}'`,
`${requirementLabel} value excludes '${snippet}'`,
!normalizeText(valueText).includes(normalizeText(snippet)),
truncateForDetails(valueText)
)
@@ -373,7 +374,7 @@ export function validateGlobalState(input: {
checks.push(
check(
`global does not include ${forbidden.type} draft ${forbidden.path}`,
!findGlobalDraft(drafts, forbidden.type, forbidden.path, forbidden.triggerKind),
!findGlobalDraft(drafts, forbidden),
summarizeGlobalDrafts(drafts)
)
);
@@ -615,16 +616,100 @@ function summarizeProblems(problems: string[], limit = 5): string | undefined {
function findGlobalDraft(
drafts: GlobalDraft[],
type: string,
path: string,
triggerKind?: string
requirement: {
type: string;
path?: string;
pathIncludes?: string[];
pathStartsWith?: string;
triggerKind?: string;
summaryIncludes?: string[];
valueIncludes?: string[];
valueExcludes?: string[];
}
): GlobalDraft | undefined {
return drafts.find(
(draft) =>
draft.type === type &&
draft.path === path &&
(triggerKind === undefined || draft.triggerKind === triggerKind)
const candidates = drafts.filter((draft) =>
globalDraftMatchesLocator(draft, requirement)
);
return (
candidates.find((draft) => globalDraftMatchesContent(draft, requirement)) ??
candidates[0]
);
}
function globalDraftMatchesLocator(
draft: GlobalDraft,
requirement: {
type: string;
path?: string;
pathIncludes?: string[];
pathStartsWith?: string;
triggerKind?: string;
}
): boolean {
return (
draft.type === requirement.type &&
(requirement.path === undefined || draft.path === requirement.path) &&
(requirement.pathStartsWith === undefined ||
draft.path.startsWith(requirement.pathStartsWith)) &&
(requirement.pathIncludes ?? []).every((snippet) =>
normalizeText(draft.path).includes(normalizeText(snippet))
) &&
(requirement.triggerKind === undefined ||
draft.triggerKind === requirement.triggerKind)
);
}
function globalDraftMatchesContent(
draft: GlobalDraft,
requirement: {
summaryIncludes?: string[];
valueIncludes?: string[];
valueExcludes?: string[];
}
): boolean {
const summary = normalizeText(draft.summary ?? "");
const value = normalizeText(stringifyGlobalDraftValue(draft.value));
return (
(requirement.summaryIncludes ?? []).every((snippet) =>
summary.includes(normalizeText(snippet))
) &&
(requirement.valueIncludes ?? []).every((snippet) =>
value.includes(normalizeText(snippet))
) &&
(requirement.valueExcludes ?? []).every(
(snippet) => !value.includes(normalizeText(snippet))
)
);
}
function formatGlobalDraftRequirement(
requirement: {
type: string;
path?: string;
pathIncludes?: string[];
pathStartsWith?: string;
triggerKind?: string;
}
): string {
const typeLabel =
requirement.triggerKind === undefined
? requirement.type
: `${requirement.triggerKind} ${requirement.type}`;
if (requirement.path !== undefined) {
return `${typeLabel} draft ${requirement.path}`;
}
const filters = [
...(requirement.pathStartsWith === undefined
? []
: [`path starts with ${requirement.pathStartsWith}`]),
...(requirement.pathIncludes ?? []).map(
(snippet) => `path includes ${snippet}`
),
];
return filters.length === 0
? `${typeLabel} draft`
: `${typeLabel} draft (${filters.join(", ")})`;
}
function summarizeGlobalDrafts(drafts: GlobalDraft[]): string {
@@ -0,0 +1,66 @@
{
"workspace": {
"scripts": [
{
"path": "f/evals/global/format_greeting",
"summary": "Format a deployed greeting",
"description": "Returns a plain greeting for a provided name.",
"language": "bun",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"name": {
"type": "string"
}
},
"required": ["name"]
},
"content": "export async function main(name: string) {\n return `Hello, ${name}`\n}\n"
},
{
"path": "f/evals/global/format_greeting_archive",
"summary": "Archived greeting formatter",
"description": "Older greeting formatter kept for reference.",
"language": "bun",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"name": {
"type": "string"
}
},
"required": ["name"]
},
"content": "export async function main(name: string) {\n return `Hi, ${name}`\n}\n"
}
]
},
"liveEditorDrafts": [
{
"type": "script",
"storagePath": "f/evals/global/current_greeting",
"effectivePath": "f/evals/global/current_greeting",
"value": {
"path": "f/evals/global/current_greeting",
"summary": "Open greeting formatter",
"description": "Formats a greeting in the live editor.",
"language": "bun",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"name": {
"type": "string"
}
},
"required": ["name"]
},
"content": "export async function main(name: string) {\n return `Hello, ${name}`\n}\n",
"is_template": false,
"kind": "script"
}
}
]
}
@@ -0,0 +1,118 @@
{
"workspace": {
"flows": [
{
"path": "f/evals/global/process_invoice",
"summary": "Deployed invoice processor",
"description": "Calculates invoice totals from a subtotal.",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"subtotal": {
"type": "number"
}
},
"required": ["subtotal"]
},
"value": {
"modules": [
{
"id": "calculate_total",
"summary": "Calculate total from subtotal",
"value": {
"type": "rawscript",
"language": "bun",
"content": "export async function main(subtotal: number) {\n return { subtotal, total: subtotal }\n}\n",
"input_transforms": {
"subtotal": {
"type": "javascript",
"expr": "flow_input.subtotal"
}
}
}
}
]
}
},
{
"path": "f/evals/global/process_refund",
"summary": "Refund processor",
"description": "Calculates refund totals.",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"subtotal": {
"type": "number"
}
},
"required": ["subtotal"]
},
"value": {
"modules": [
{
"id": "calculate_total",
"summary": "Calculate refund total",
"value": {
"type": "rawscript",
"language": "bun",
"content": "export async function main(subtotal: number) {\n return { subtotal, total: subtotal }\n}\n",
"input_transforms": {
"subtotal": {
"type": "javascript",
"expr": "flow_input.subtotal"
}
}
}
}
]
}
}
]
},
"liveEditorDrafts": [
{
"type": "flow",
"storagePath": "f/evals/global/current_invoice_flow",
"effectivePath": "f/evals/global/current_invoice_flow",
"value": {
"path": "f/evals/global/current_invoice_flow",
"summary": "Open invoice processor",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"subtotal": {
"type": "number"
}
},
"required": ["subtotal"]
},
"value": {
"modules": [
{
"id": "calculate_total",
"summary": "Calculate total from subtotal",
"value": {
"type": "rawscript",
"language": "bun",
"content": "export async function main(subtotal: number) {\n return { subtotal, total: subtotal }\n}\n",
"input_transforms": {
"subtotal": {
"type": "javascript",
"expr": "flow_input.subtotal"
}
}
}
}
]
},
"edited_by": "",
"edited_at": "",
"archived": false,
"extra_perms": {}
}
}
]
}
@@ -0,0 +1,40 @@
{
"workspace": {
"flows": [
{
"path": "f/evals/global/process_invoice",
"summary": "Process an invoice subtotal",
"description": "Calculates invoice totals from a subtotal.",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"subtotal": {
"type": "number"
}
},
"required": ["subtotal"]
},
"value": {
"modules": [
{
"id": "calculate_total",
"summary": "Calculate total from subtotal",
"value": {
"type": "rawscript",
"language": "bun",
"content": "export async function main(subtotal: number) {\n return { subtotal, total: subtotal }\n}\n",
"input_transforms": {
"subtotal": {
"type": "javascript",
"expr": "flow_input.subtotal"
}
}
}
}
]
}
}
]
}
}
@@ -0,0 +1,23 @@
{
"workspace": {
"scripts": [
{
"path": "f/evals/global/send_report_digest",
"summary": "Build and send the eval report digest",
"description": "Returns a dry-run summary for eval report digest notifications.",
"language": "bun",
"schema": {
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"properties": {
"dry_run": {
"type": "boolean"
}
},
"required": ["dry_run"]
},
"content": "export async function main(dry_run: boolean) {\n return { dry_run, sent: !dry_run, message: dry_run ? 'Preview digest' : 'Digest sent' }\n}\n"
}
]
}
}
+7 -1
View File
@@ -1,5 +1,8 @@
import { readFile } from "node:fs/promises";
import { runGlobalEval } from "../adapters/frontend/core/global/globalEvalRunner";
import {
runGlobalEval,
type GlobalLiveEditorDraftFixture,
} from "../adapters/frontend/core/global/globalEvalRunner";
import type { BenchmarkWorkspaceRunnables } from "../adapters/frontend/mockBackend";
import type { FrontendEvalModelConfig } from "../core/models";
import type { BenchmarkArtifactFile, GlobalValidationSpec, ModeRunner } from "../core/types";
@@ -9,6 +12,7 @@ import { getFrontendApiKey } from "./frontendCommon";
export interface GlobalInitialFixture {
workspace?: BenchmarkWorkspaceRunnables;
liveEditorDrafts?: GlobalLiveEditorDraftFixture[];
}
export function createGlobalModeRunner(
@@ -31,6 +35,7 @@ export function createGlobalModeRunner(
getFrontendApiKey(modelConfig.provider),
{
workspaceFixtures: initial?.workspace,
liveEditorDrafts: initial?.liveEditorDrafts,
maxIterations: context.evalCase?.runtime?.maxTurns,
provider: modelConfig.provider,
model: modelConfig.model,
@@ -73,6 +78,7 @@ async function loadGlobalInitialFixture(path: string): Promise<GlobalInitialFixt
const parsed = JSON.parse(await readFile(path, "utf8")) as GlobalInitialFixture;
return {
workspace: parsed.workspace ?? {},
liveEditorDrafts: parsed.liveEditorDrafts ?? [],
};
}
@@ -46,11 +46,11 @@
]
},
"nullable": [
true,
true,
true,
true,
true,
false,
false,
false,
false,
false,
true,
true
]
@@ -0,0 +1,22 @@
{
"db_name": "PostgreSQL",
"query": "INSERT INTO token\n (token_hash, token_prefix, token, email, label, expiration, super_admin, scopes, read_only)\n VALUES ($1, $2, $3, $4, $5, now() + ($6 || ' seconds')::interval, $7, $8, $9)",
"describe": {
"columns": [],
"parameters": {
"Left": [
"Varchar",
"Varchar",
"Varchar",
"Varchar",
"Varchar",
"Text",
"Bool",
"TextArray",
"Bool"
]
},
"nullable": []
},
"hash": "7f832370916794ab0e5645053688c24678f1519d49ee7263a86dba71d45b8e8c"
}
@@ -0,0 +1,26 @@
{
"db_name": "PostgreSQL",
"query": "\n SELECT ws.workspace_id AS \"workspace_id!\", entry->'catalog'->>'resource_path' AS dbname\n FROM workspace_settings ws\n CROSS JOIN LATERAL jsonb_each(\n CASE WHEN jsonb_typeof(ws.ducklake->'ducklakes') = 'object'\n THEN ws.ducklake->'ducklakes'\n ELSE '{}'::jsonb END\n ) AS dl(k, entry)\n WHERE entry->'catalog'->>'resource_type' = 'instance'\n AND entry->'catalog'->>'resource_path' IS NOT NULL\n UNION ALL\n SELECT ws.workspace_id AS \"workspace_id!\", entry->'database'->>'resource_path' AS dbname\n FROM workspace_settings ws\n CROSS JOIN LATERAL jsonb_each(\n CASE WHEN jsonb_typeof(ws.datatable->'datatables') = 'object'\n THEN ws.datatable->'datatables'\n ELSE '{}'::jsonb END\n ) AS dt(k, entry)\n WHERE entry->'database'->>'resource_type' = 'instance'\n AND entry->'database'->>'resource_path' IS NOT NULL\n ",
"describe": {
"columns": [
{
"ordinal": 0,
"name": "workspace_id!",
"type_info": "Varchar"
},
{
"ordinal": 1,
"name": "dbname",
"type_info": "Text"
}
],
"parameters": {
"Left": []
},
"nullable": [
null,
null
]
},
"hash": "815d96aea4681490582b08630a30a168cc1191acaab96bed6a016c437059c2cd"
}
@@ -0,0 +1,16 @@
{
"db_name": "PostgreSQL",
"query": "\n INSERT INTO variable (workspace_id, path, value, is_secret, description, extra_perms, account)\n VALUES ($1, $2, $3, true, '', '{}'::jsonb, NULL)\n ON CONFLICT (workspace_id, path) DO UPDATE SET value = EXCLUDED.value\n ",
"describe": {
"columns": [],
"parameters": {
"Left": [
"Varchar",
"Varchar",
"Varchar"
]
},
"nullable": []
},
"hash": "b876f26ce90e30c3510eacddb03d9dd26fac05d18183f2d623ac91ea3876dd5c"
}
@@ -0,0 +1,28 @@
{
"db_name": "PostgreSQL",
"query": "\n SELECT j.id, j.args\n FROM v2_job j\n JOIN v2_job_queue q ON j.id = q.id\n WHERE j.runnable_path = $1\n AND j.kind = 'deploymentcallback'\n AND j.workspace_id = 'test-workspace'\n ORDER BY j.created_at DESC\n ",
"describe": {
"columns": [
{
"ordinal": 0,
"name": "id",
"type_info": "Uuid"
},
{
"ordinal": 1,
"name": "args",
"type_info": "Jsonb"
}
],
"parameters": {
"Left": [
"Text"
]
},
"nullable": [
false,
true
]
},
"hash": "bf601a919de299e44e6e418b2e711e24909fc206b7fd48541483e6316fe002d5"
}
@@ -1,16 +0,0 @@
{
"db_name": "PostgreSQL",
"query": "UPDATE workspace_runnable_dependencies SET app_path = REGEXP_REPLACE(app_path,'u/' || $2 || '/(.*)','u/' || $1 || '/\\1') WHERE app_path LIKE ('u/' || $2 || '/%') AND workspace_id = $3",
"describe": {
"columns": [],
"parameters": {
"Left": [
"Text",
"Text",
"Text"
]
},
"nullable": []
},
"hash": "f699cc3644aeb35a0588bbb3a6bf2dc0746d9f3a1bea104123188fe2921bc886"
}
+206 -244
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -1,6 +1,6 @@
[package]
name = "windmill"
version = "1.711.0"
version = "1.714.0"
authors.workspace = true
edition.workspace = true
@@ -87,7 +87,7 @@ members = [
exclude = ["./windmill-duckdb-ffi-internal", "./parsers/windmill-parser-wasm"]
[workspace.package]
version = "1.711.0"
version = "1.714.0"
authors = ["Ruben Fiszel <ruben@windmill.dev>"]
edition = "2021"
+1 -1
View File
@@ -1 +1 @@
327d23f7438968a21bac9fd42e7f6f027c61477c
3742e0659c5e97aab03b9efeea14cd94a3ac658a
+18 -1
View File
@@ -176,6 +176,23 @@
"token_url": "https://account.docusign.com/oauth/token",
"scopes": [
"signature"
]
],
"sandbox": {
"auth_url": "https://account-d.docusign.com/oauth/auth",
"token_url": "https://account-d.docusign.com/oauth/token"
}
},
"salesforce": {
"auth_url": "https://login.salesforce.com/services/oauth2/authorize",
"token_url": "https://login.salesforce.com/services/oauth2/token",
"scopes": [
"api",
"refresh_token",
"offline_access"
],
"sandbox": {
"auth_url": "https://test.salesforce.com/services/oauth2/authorize",
"token_url": "https://test.salesforce.com/services/oauth2/token"
}
}
}
+24 -24
View File
@@ -6183,7 +6183,7 @@ checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f"
[[package]]
name = "windmill-common"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"aho-corasick",
"anyhow",
@@ -6263,7 +6263,7 @@ dependencies = [
[[package]]
name = "windmill-macros"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"proc-macro2",
"quote",
@@ -6275,7 +6275,7 @@ dependencies = [
[[package]]
name = "windmill-parser"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"convert_case",
"serde",
@@ -6284,7 +6284,7 @@ dependencies = [
[[package]]
name = "windmill-parser-bash"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"lazy_static",
@@ -6296,7 +6296,7 @@ dependencies = [
[[package]]
name = "windmill-parser-csharp"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"serde_json",
@@ -6308,7 +6308,7 @@ dependencies = [
[[package]]
name = "windmill-parser-go"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"gosyn",
@@ -6320,7 +6320,7 @@ dependencies = [
[[package]]
name = "windmill-parser-graphql"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"lazy_static",
@@ -6332,7 +6332,7 @@ dependencies = [
[[package]]
name = "windmill-parser-java"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"serde_json",
@@ -6344,7 +6344,7 @@ dependencies = [
[[package]]
name = "windmill-parser-nu"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"nu-parser",
@@ -6355,7 +6355,7 @@ dependencies = [
[[package]]
name = "windmill-parser-php"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"itertools 0.14.0",
@@ -6366,7 +6366,7 @@ dependencies = [
[[package]]
name = "windmill-parser-py"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"itertools 0.14.0",
@@ -6378,7 +6378,7 @@ dependencies = [
[[package]]
name = "windmill-parser-py-asset"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"rustpython-ast",
@@ -6389,7 +6389,7 @@ dependencies = [
[[package]]
name = "windmill-parser-py-imports"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"async-recursion",
@@ -6411,7 +6411,7 @@ dependencies = [
[[package]]
name = "windmill-parser-r"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"serde_json",
@@ -6423,7 +6423,7 @@ dependencies = [
[[package]]
name = "windmill-parser-ruby"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"lazy_static",
@@ -6437,7 +6437,7 @@ dependencies = [
[[package]]
name = "windmill-parser-rust"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"convert_case",
@@ -6454,7 +6454,7 @@ dependencies = [
[[package]]
name = "windmill-parser-sql"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"lazy_static",
@@ -6467,7 +6467,7 @@ dependencies = [
[[package]]
name = "windmill-parser-sql-asset"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"serde",
@@ -6479,7 +6479,7 @@ dependencies = [
[[package]]
name = "windmill-parser-ts"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"lazy_static",
@@ -6497,7 +6497,7 @@ dependencies = [
[[package]]
name = "windmill-parser-ts-asset"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"serde-wasm-bindgen",
@@ -6513,7 +6513,7 @@ dependencies = [
[[package]]
name = "windmill-parser-wac"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"rustpython-ast",
@@ -6529,7 +6529,7 @@ dependencies = [
[[package]]
name = "windmill-parser-wasm"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"getrandom 0.2.17",
@@ -6561,7 +6561,7 @@ dependencies = [
[[package]]
name = "windmill-parser-yaml"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"serde",
@@ -6572,7 +6572,7 @@ dependencies = [
[[package]]
name = "windmill-types"
version = "1.711.0"
version = "1.714.0"
dependencies = [
"anyhow",
"bitflags",
@@ -12,7 +12,7 @@ resolver = "2"
members = ["."]
[workspace.package]
version = "1.711.0"
version = "1.714.0"
edition = "2021"
authors = ["Ruben Fiszel <ruben@windmill.dev>"]
+10
View File
@@ -258,6 +258,15 @@ pub fn main() -> anyhow::Result<()> {
}
async fn cache_hub_scripts(file_path: Option<String>) -> anyhow::Result<()> {
// The `cache` CLI mode never connects to the DB, so HUB_BASE_URL keeps its
// compiled default. Allow overriding it via env so the prebuild cache step can
// be pointed at a private/staging hub (e.g. a local proxy for testing).
if let Ok(hub_base_url) = std::env::var("HUB_BASE_URL") {
if !hub_base_url.is_empty() {
tracing::info!("Overriding hub base url from env: {hub_base_url}");
windmill_common::HUB_BASE_URL.store(std::sync::Arc::new(hub_base_url));
}
}
let file_path = file_path.unwrap_or("./hubPaths.json".to_string());
let mut file = File::open(&file_path)
.await
@@ -567,6 +576,7 @@ fn print_help() {
println!(" RUN_UPDATE_CA_CERTIFICATE_AT_START = false Run system CA update at startup");
println!(" RUN_UPDATE_CA_CERTIFICATE_PATH = /usr/sbin/update-ca-certificates Path to CA update tool");
println!(" SYNC_CACHED_RT = false Sync cached resource types to admins workspace on server start");
println!(" HUB_BASE_URL = https://hub.windmill.dev Hub to fetch scripts from in `cache` mode (server/worker use the DB setting instead)");
println!();
println!("Notes:");
println!("- Advanced and less commonly used settings are managed via the database and are omitted here.");
+2
View File
@@ -451,6 +451,7 @@ def main():
preserve_on_behalf_of: None,
ws_error_handler_muted: None,
labels: None,
skip_draft_deletion: None,
})
.send()
.await
@@ -513,6 +514,7 @@ def main():
custom_path: None,
preserve_on_behalf_of: None,
labels: None,
skip_draft_deletion: None,
})
.send()
.await
+36
View File
@@ -754,6 +754,26 @@ pub fn bedrock_stream_event_to_tool_start(
}
}
pub fn bedrock_stream_event_to_tool_start_with_block_index(
event: &ConverseStreamOutput,
) -> Option<(usize, StreamingToolCall)> {
match event {
ConverseStreamOutput::ContentBlockStart(start) => {
let block_index = usize::try_from(start.content_block_index()).ok()?;
let tool_use = start.start().and_then(|s| s.as_tool_use().ok())?;
Some((
block_index,
StreamingToolCall {
id: tool_use.tool_use_id().to_string(),
name: tool_use.name().to_string(),
arguments: String::new(),
},
))
}
_ => None,
}
}
/// Extract tool use input delta from stream
pub fn bedrock_stream_event_to_tool_delta(event: &ConverseStreamOutput) -> Option<String> {
match event {
@@ -765,6 +785,22 @@ pub fn bedrock_stream_event_to_tool_delta(event: &ConverseStreamOutput) -> Optio
}
}
pub fn bedrock_stream_event_to_tool_delta_with_block_index(
event: &ConverseStreamOutput,
) -> Option<(usize, String)> {
match event {
ConverseStreamOutput::ContentBlockDelta(delta) => {
let block_index = usize::try_from(delta.content_block_index()).ok()?;
let input = delta
.delta()
.and_then(|d| d.as_tool_use().ok())
.map(|tool_use| tool_use.input().to_string())?;
Some((block_index, input))
}
_ => None,
}
}
/// Check if stream event indicates content block stop
pub fn bedrock_stream_event_is_block_stop(event: &ConverseStreamOutput) -> bool {
matches!(event, ConverseStreamOutput::ContentBlockStop(_))
+3 -2
View File
@@ -20,13 +20,14 @@ where
lazy_static::lazy_static! {
static ref OPENAI_AZURE_BASE_PATH: Option<String> = std::env::var("OPENAI_AZURE_BASE_PATH").ok();
static ref ALLOW_PRIVATE_AI_BASE_URLS: bool = std::env::var("ALLOW_PRIVATE_AI_BASE_URLS")
pub static ref ALLOW_PRIVATE_AI_BASE_URLS: bool = std::env::var("ALLOW_PRIVATE_AI_BASE_URLS")
.ok()
.map(|v| v == "true" || v == "1")
.unwrap_or(false);
}
pub const OPENAI_BASE_URL: &str = "https://api.openai.com/v1";
pub const DEEPSEEK_BASE_URL: &str = "https://api.deepseek.com/v1";
pub const GOOGLE_AI_BASE_URL: &str = "https://generativelanguage.googleapis.com/v1beta";
/// Empty string signals BedrockClient::from_env() to use the region from AWS environment/config
@@ -106,7 +107,7 @@ impl AIProvider {
Ok(azure_base_path.unwrap_or("https://api.openai.com/v1".to_string()))
}
AIProvider::DeepSeek => Ok("https://api.deepseek.com/v1".to_string()),
AIProvider::DeepSeek => Ok(DEEPSEEK_BASE_URL.to_string()),
AIProvider::GoogleAI => Ok(GOOGLE_AI_BASE_URL.to_string()),
AIProvider::Groq => Ok("https://api.groq.com/openai/v1".to_string()),
AIProvider::OpenRouter => Ok("https://openrouter.ai/api/v1".to_string()),
+25
View File
@@ -0,0 +1,25 @@
use std::collections::HashMap;
use crate::ai_providers::{AIPlatform, AIProvider};
/// Resolved provider credentials shared by API proxy and worker execution.
///
/// Raw API resources and worker agent payloads convert into this shape at their
/// execution boundaries. Request-specific state such as the selected model stays
/// outside this type.
#[derive(Clone, Debug)]
pub struct ProviderCredentials {
pub provider: AIProvider,
pub base_url: String,
pub api_key: Option<String>,
pub access_token: Option<String>,
pub organization_id: Option<String>,
pub user: Option<String>,
pub region: Option<String>,
pub aws_access_key_id: Option<String>,
pub aws_secret_access_key: Option<String>,
pub aws_session_token: Option<String>,
pub platform: AIPlatform,
pub enable_1m_context: bool,
pub custom_headers: HashMap<String, String>,
}
+1
View File
@@ -4,6 +4,7 @@ pub mod ai_cache;
pub mod ai_google;
pub mod ai_providers;
pub mod ai_types;
pub mod credentials;
pub mod image_handler;
pub mod providers;
pub mod proxy;
@@ -729,8 +729,7 @@ impl QueryBuilder for AnthropicQueryBuilder {
mod tests {
use super::*;
use crate::{
proxy::{ProviderCredentials, ProxyBuildArgs},
query_builder::QueryBuilder,
credentials::ProviderCredentials, proxy::ProxyBuildArgs, query_builder::QueryBuilder,
};
use http::{HeaderMap, HeaderValue, Method};
use std::collections::HashMap;
+228 -128
View File
@@ -10,9 +10,10 @@ use crate::{
ai_bedrock::{
bedrock_model_supports_prompt_caching, bedrock_stream_event_is_block_stop,
bedrock_stream_event_to_text, bedrock_stream_event_to_tool_delta,
bedrock_stream_event_to_tool_start, build_tool_config, create_inference_config,
format_bedrock_error, openai_messages_to_bedrock, streaming_tool_calls_to_openai,
BearerTokenProvider, BedrockClient, StreamingToolCall,
bedrock_stream_event_to_tool_delta_with_block_index, bedrock_stream_event_to_tool_start,
bedrock_stream_event_to_tool_start_with_block_index, build_tool_config,
create_inference_config, format_bedrock_error, openai_messages_to_bedrock,
streaming_tool_calls_to_openai, BearerTokenProvider, BedrockClient, StreamingToolCall,
},
ai_providers::USE_ENV_REGION,
ai_types::{OpenAIFunction, OpenAIToolCall, ToolDefFunction},
@@ -403,137 +404,15 @@ pub fn sdk_stream_to_sse(
.unwrap()
.as_secs();
struct StreamState {
id: String,
model: String,
created: u64,
tool_calls: HashMap<usize, (String, String, String)>,
current_tool_index: usize,
}
let state = std::sync::Arc::new(tokio::sync::Mutex::new(StreamState {
id,
model,
created,
tool_calls: HashMap::new(),
current_tool_index: 0,
}));
async_stream::stream! {
let mut stream = stream;
let state = state.clone();
let mut state = BedrockSseStreamState::new(id, model, created);
loop {
match stream.recv().await {
Ok(Some(event)) => {
let mut state = state.lock().await;
if let Some(tool_call) = bedrock_stream_event_to_tool_start(&event) {
let index = state.current_tool_index;
state.tool_calls.insert(
index,
(tool_call.id.clone(), tool_call.name.clone(), String::new()),
);
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {
"tool_calls": [{
"index": index,
"id": tool_call.id,
"type": "function",
"function": {
"name": tool_call.name,
"arguments": ""
}
}]
},
"finish_reason": serde_json::Value::Null
}]
});
yield Ok(Bytes::from(format!("data: {}\n\n", chunk)));
}
if let Some(text) = bedrock_stream_event_to_text(&event) {
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {
"content": text
},
"finish_reason": serde_json::Value::Null
}]
});
yield Ok(Bytes::from(format!("data: {}\n\n", chunk)));
}
if let Some(input_delta) = bedrock_stream_event_to_tool_delta(&event) {
let index = state.current_tool_index;
if let Some((_id, _name, ref mut args)) = state.tool_calls.get_mut(&index) {
args.push_str(&input_delta);
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {
"tool_calls": [{
"index": index,
"function": {
"arguments": input_delta
}
}]
},
"finish_reason": serde_json::Value::Null
}]
});
yield Ok(Bytes::from(format!("data: {}\n\n", chunk)));
}
}
if bedrock_stream_event_is_block_stop(&event) {
state.current_tool_index += 1;
}
if let aws_sdk_bedrockruntime::types::ConverseStreamOutput::MessageStop(stop) = &event {
let stop_reason = stop.stop_reason().as_str();
let finish_reason = match stop_reason {
"end_turn" => "stop",
"max_tokens" => "length",
"tool_use" => "tool_calls",
"stop_sequence" => "stop",
"guardrail_intervened" | "content_filtered" => "content_filter",
_ => "stop",
};
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {},
"finish_reason": finish_reason
}]
});
yield Ok(Bytes::from(format!("data: {}\n\n", chunk)));
for chunk in bedrock_sse_chunks_for_event(&event, &mut state) {
yield Ok(chunk);
}
}
Ok(None) => break,
@@ -551,6 +430,149 @@ pub fn sdk_stream_to_sse(
}
}
#[derive(Debug)]
struct BedrockSseStreamState {
id: String,
model: String,
created: u64,
tool_calls: HashMap<usize, (String, String, String)>,
tool_block_indexes: HashMap<usize, usize>,
next_tool_index: usize,
}
impl BedrockSseStreamState {
fn new(id: String, model: String, created: u64) -> Self {
Self {
id,
model,
created,
tool_calls: HashMap::new(),
tool_block_indexes: HashMap::new(),
next_tool_index: 0,
}
}
}
fn bedrock_sse_chunks_for_event(
event: &aws_sdk_bedrockruntime::types::ConverseStreamOutput,
state: &mut BedrockSseStreamState,
) -> Vec<Bytes> {
let mut chunks = Vec::new();
if let Some((block_index, tool_call)) =
bedrock_stream_event_to_tool_start_with_block_index(event)
{
let index = state.next_tool_index;
state.next_tool_index += 1;
state.tool_block_indexes.insert(block_index, index);
state.tool_calls.insert(
index,
(tool_call.id.clone(), tool_call.name.clone(), String::new()),
);
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {
"tool_calls": [{
"index": index,
"id": tool_call.id,
"type": "function",
"function": {
"name": tool_call.name,
"arguments": ""
}
}]
},
"finish_reason": serde_json::Value::Null
}]
});
chunks.push(Bytes::from(format!("data: {}\n\n", chunk)));
}
if let Some(text) = bedrock_stream_event_to_text(event) {
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {
"content": text
},
"finish_reason": serde_json::Value::Null
}]
});
chunks.push(Bytes::from(format!("data: {}\n\n", chunk)));
}
if let Some((block_index, input_delta)) =
bedrock_stream_event_to_tool_delta_with_block_index(event)
{
if let Some(index) = state.tool_block_indexes.get(&block_index).copied() {
if let Some((_id, _name, ref mut args)) = state.tool_calls.get_mut(&index) {
args.push_str(&input_delta);
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {
"tool_calls": [{
"index": index,
"function": {
"arguments": input_delta
}
}]
},
"finish_reason": serde_json::Value::Null
}]
});
chunks.push(Bytes::from(format!("data: {}\n\n", chunk)));
}
}
}
if let aws_sdk_bedrockruntime::types::ConverseStreamOutput::MessageStop(stop) = event {
let stop_reason = stop.stop_reason().as_str();
let finish_reason = match stop_reason {
"end_turn" => "stop",
"max_tokens" => "length",
"tool_use" => "tool_calls",
"stop_sequence" => "stop",
"guardrail_intervened" | "content_filtered" => "content_filter",
_ => "stop",
};
let chunk = serde_json::json!({
"id": state.id,
"object": "chat.completion.chunk",
"created": state.created,
"model": state.model,
"choices": [{
"index": 0,
"delta": {},
"finish_reason": finish_reason
}]
});
chunks.push(Bytes::from(format!("data: {}\n\n", chunk)));
}
chunks
}
async fn handle_bedrock_sdk_non_streaming(
model: &str,
body: &[u8],
@@ -970,6 +992,19 @@ impl BedrockQueryBuilder {
#[cfg(test)]
mod tests {
use super::*;
use aws_sdk_bedrockruntime::types::{
ContentBlockDelta, ContentBlockDeltaEvent, ContentBlockStart, ContentBlockStartEvent,
ContentBlockStopEvent, ConverseStreamOutput, ToolUseBlockDelta, ToolUseBlockStart,
};
fn sse_json(chunk: &Bytes) -> serde_json::Value {
let chunk = std::str::from_utf8(chunk).expect("SSE chunk should be UTF-8");
let payload = chunk
.strip_prefix("data: ")
.and_then(|chunk| chunk.strip_suffix("\n\n"))
.expect("chunk should be SSE data");
serde_json::from_str(payload).expect("chunk should contain JSON")
}
#[test]
fn determine_auth_config_prioritizes_bearer_token() {
@@ -1022,4 +1057,69 @@ mod tests {
let config = determine_auth_config(None, Some("AKIA123"), None, Some("session-token"));
assert!(matches!(config, BedrockAuthConfig::Environment));
}
#[test]
fn bedrock_sse_tool_indexes_ignore_text_block_stops() {
let mut state =
BedrockSseStreamState::new("chatcmpl-test".to_string(), "model".to_string(), 1);
let text_delta = ConverseStreamOutput::ContentBlockDelta(
ContentBlockDeltaEvent::builder()
.content_block_index(0)
.delta(ContentBlockDelta::Text("hello".to_string()))
.build()
.unwrap(),
);
assert_eq!(
bedrock_sse_chunks_for_event(&text_delta, &mut state).len(),
1
);
let text_stop = ConverseStreamOutput::ContentBlockStop(
ContentBlockStopEvent::builder()
.content_block_index(0)
.build()
.unwrap(),
);
assert!(bedrock_sse_chunks_for_event(&text_stop, &mut state).is_empty());
let tool_start = ConverseStreamOutput::ContentBlockStart(
ContentBlockStartEvent::builder()
.content_block_index(1)
.start(ContentBlockStart::ToolUse(
ToolUseBlockStart::builder()
.tool_use_id("call_1")
.name("lookup")
.build()
.unwrap(),
))
.build()
.unwrap(),
);
let start_chunks = bedrock_sse_chunks_for_event(&tool_start, &mut state);
let start_json = sse_json(&start_chunks[0]);
assert_eq!(
start_json["choices"][0]["delta"]["tool_calls"][0]["index"],
0
);
let tool_delta = ConverseStreamOutput::ContentBlockDelta(
ContentBlockDeltaEvent::builder()
.content_block_index(1)
.delta(ContentBlockDelta::ToolUse(
ToolUseBlockDelta::builder()
.input("{\"city\":\"Paris\"}")
.build()
.unwrap(),
))
.build()
.unwrap(),
);
let delta_chunks = bedrock_sse_chunks_for_event(&tool_delta, &mut state);
let delta_json = sse_json(&delta_chunks[0]);
assert_eq!(
delta_json["choices"][0]["delta"]["tool_calls"][0]["index"],
0
);
}
}
@@ -691,7 +691,7 @@ impl QueryBuilder for GoogleAIQueryBuilder {
#[cfg(test)]
mod tests {
use super::*;
use crate::{ai_providers::AIProvider, proxy::ProviderCredentials};
use crate::{ai_providers::AIProvider, credentials::ProviderCredentials};
use std::collections::HashMap;
fn credentials(base_url: &str, platform: AIPlatform) -> ProviderCredentials {
+3 -1
View File
@@ -6,7 +6,9 @@ pub mod openai;
pub mod openrouter;
pub mod other;
use crate::{ai_providers::AIProvider, proxy::ProviderCredentials, query_builder::QueryBuilder};
use crate::{
ai_providers::AIProvider, credentials::ProviderCredentials, query_builder::QueryBuilder,
};
use self::{
anthropic::AnthropicQueryBuilder, google_ai::GoogleAIQueryBuilder, openai::OpenAIQueryBuilder,
+6 -22
View File
@@ -4,30 +4,11 @@ use http::{HeaderMap, Method};
use serde_json::value::RawValue;
use windmill_common::error::{Error, Result};
use crate::ai_providers::{AIPlatform, AIProvider};
use crate::ai_providers::AIProvider;
use crate::credentials::ProviderCredentials;
use crate::utils::AI_HTTP_HEADERS;
/// Resolved provider credentials shared by API proxy and worker execution.
///
/// Raw API resources and worker agent payloads convert into this shape at their
/// execution boundaries. Request-specific state such as the selected model stays
/// outside this type.
#[derive(Clone, Debug)]
pub struct ProviderCredentials {
pub provider: AIProvider,
pub base_url: String,
pub api_key: Option<String>,
pub access_token: Option<String>,
pub organization_id: Option<String>,
pub user: Option<String>,
pub region: Option<String>,
pub aws_access_key_id: Option<String>,
pub aws_secret_access_key: Option<String>,
pub aws_session_token: Option<String>,
pub platform: AIPlatform,
pub enable_1m_context: bool,
pub custom_headers: HashMap<String, String>,
}
pub mod fim;
/// Inputs needed to transform an OpenAI-compatible proxy request for a provider.
pub struct ProxyBuildArgs<'a> {
@@ -167,6 +148,9 @@ pub(crate) fn add_user_to_body(body: &[u8], user: &str) -> Result<Vec<u8>> {
#[cfg(test)]
mod tests {
use super::*;
use std::collections::HashMap;
use crate::ai_providers::AIPlatform;
fn credentials(provider: AIProvider, base_url: &str) -> ProviderCredentials {
ProviderCredentials {
+212
View File
@@ -0,0 +1,212 @@
use bytes::Bytes;
use serde::Deserialize;
use serde_json::json;
use windmill_common::error::{Error, Result};
use crate::ai_providers::{AIProvider, DEEPSEEK_BASE_URL};
#[derive(Debug, Eq, PartialEq)]
pub struct FimProxyTransform {
pub body: Bytes,
pub path: String,
pub base_url: Option<String>,
}
#[derive(Deserialize)]
struct FimRequest {
model: String,
prompt: String,
suffix: Option<String>,
temperature: Option<f32>,
max_tokens: Option<u32>,
stop: Option<Vec<String>>,
}
pub fn supports_native_fim(provider: &AIProvider) -> bool {
matches!(provider, AIProvider::Mistral | AIProvider::DeepSeek)
}
fn deepseek_fim_base_url(base_url: &str) -> String {
let trimmed = base_url.trim_end_matches('/');
let deepseek_root_base_url = DEEPSEEK_BASE_URL
.strip_suffix("/v1")
.unwrap_or(DEEPSEEK_BASE_URL);
if trimmed == DEEPSEEK_BASE_URL || trimmed == deepseek_root_base_url {
return format!("{deepseek_root_base_url}/beta");
}
if let Some(prefix) = trimmed.strip_suffix("/v1") {
return format!("{prefix}/beta");
}
trimmed.to_string()
}
pub fn maybe_transform_fim_request(
provider: &AIProvider,
path: &str,
base_url: &str,
body: &[u8],
) -> Result<Option<FimProxyTransform>> {
if !path.contains("fim/completions") {
return Ok(None);
}
if matches!(provider, AIProvider::DeepSeek) {
return Ok(Some(FimProxyTransform {
body: Bytes::copy_from_slice(body),
path: "completions".to_string(),
base_url: Some(deepseek_fim_base_url(base_url)),
}));
}
if !supports_native_fim(provider) {
return transform_fim_to_chat_completions(body).map(Some);
}
Ok(None)
}
fn transform_fim_to_chat_completions(body: &[u8]) -> Result<FimProxyTransform> {
let fim_req: FimRequest = serde_json::from_slice(body)
.map_err(|e| Error::BadRequest(format!("Failed to parse FIM request: {}", e)))?;
let suffix = fim_req.suffix.unwrap_or_default();
let system_prompt = "You are a code completion assistant. Complete the code at the <CURSOR/> position between the given prefix and suffix. Output ONLY the code that goes at the cursor - no explanations, no markdown, no repeating the prefix or suffix.";
let user_content = format!(
"<PREFIX>\n{}\n<CURSOR/>\n<SUFFIX>\n{}",
fim_req.prompt, suffix
);
let chat_req = json!({
"model": fim_req.model,
"messages": [
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_content}
],
"temperature": fim_req.temperature.unwrap_or(0.0),
"max_tokens": fim_req.max_tokens.unwrap_or(256),
"stop": fim_req.stop
});
let body = serde_json::to_vec(&chat_req)
.map_err(|e| Error::internal_err(format!("Failed to serialize chat request: {}", e)))?;
Ok(FimProxyTransform {
body: Bytes::from(body),
path: "chat/completions".to_string(),
base_url: None,
})
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn mistral_keeps_native_fim_request() {
let transformed = maybe_transform_fim_request(
&AIProvider::Mistral,
"fim/completions",
"https://api.mistral.ai/v1",
br#"{}"#,
)
.unwrap();
assert!(transformed.is_none());
assert!(supports_native_fim(&AIProvider::Mistral));
assert!(supports_native_fim(&AIProvider::DeepSeek));
assert!(!supports_native_fim(&AIProvider::OpenAI));
}
#[test]
fn deepseek_fim_base_url_uses_beta_endpoint() {
assert_eq!(
deepseek_fim_base_url("https://api.deepseek.com/v1"),
"https://api.deepseek.com/beta"
);
assert_eq!(
deepseek_fim_base_url("https://api.deepseek.com/v1/"),
"https://api.deepseek.com/beta"
);
assert_eq!(
deepseek_fim_base_url("https://api.deepseek.com"),
"https://api.deepseek.com/beta"
);
assert_eq!(
deepseek_fim_base_url("https://proxy.example/deepseek/v1"),
"https://proxy.example/deepseek/beta"
);
assert_eq!(
deepseek_fim_base_url("https://proxy.example/deepseek/beta"),
"https://proxy.example/deepseek/beta"
);
}
#[test]
fn deepseek_fim_request_uses_beta_completions_endpoint() {
let body = br#"{"model":"deepseek-v4-pro","prompt":"return ","suffix":";"}"#;
let transformed = maybe_transform_fim_request(
&AIProvider::DeepSeek,
"fim/completions",
DEEPSEEK_BASE_URL,
body,
)
.unwrap()
.expect("DeepSeek FIM should be routed to the beta completions endpoint");
assert_eq!(transformed.path, "completions");
assert_eq!(
transformed.base_url.as_deref(),
Some("https://api.deepseek.com/beta")
);
assert_eq!(transformed.body, Bytes::copy_from_slice(body));
}
#[test]
fn openai_fim_request_is_transformed_to_chat_completion() {
let transformed = maybe_transform_fim_request(
&AIProvider::OpenAI,
"fim/completions",
"https://api.openai.com/v1",
br#"{
"model": "gpt-4.1",
"prompt": "fn main() {",
"suffix": "}",
"stop": ["\n\n"]
}"#,
)
.unwrap()
.expect("OpenAI FIM should be transformed");
assert_eq!(transformed.path, "chat/completions");
assert_eq!(transformed.base_url, None);
let body: serde_json::Value = serde_json::from_slice(&transformed.body).unwrap();
assert_eq!(body["model"], "gpt-4.1");
assert_eq!(body["temperature"], 0.0);
assert_eq!(body["max_tokens"], 256);
assert_eq!(body["stop"], serde_json::json!(["\n\n"]));
assert_eq!(body["messages"][1]["role"], "user");
assert_eq!(
body["messages"][1]["content"],
"<PREFIX>\nfn main() {\n<CURSOR/>\n<SUFFIX>\n}"
);
}
#[test]
fn invalid_fim_body_is_bad_request() {
let err = maybe_transform_fim_request(
&AIProvider::OpenAI,
"fim/completions",
"https://api.openai.com/v1",
br#"{"model": 1}"#,
)
.unwrap_err();
assert!(matches!(err, Error::BadRequest(_)));
}
}
+1 -1
View File
@@ -18,7 +18,7 @@ pub struct McpToolSource {
use crate::{
ai_google::sanitize_schema_for_google,
ai_providers::{empty_string_as_none, AIProvider},
proxy::ProviderCredentials,
credentials::ProviderCredentials,
};
use windmill_common::{db::DB, error::Error, flow_status::AgentAction, flows::FlowModule};
use windmill_parser::Typ;
+477
View File
@@ -235,6 +235,202 @@ where
Ok(())
}
/// Returns the caller's "real" scope restrictions: every scope other than
/// `if_jobs:filter_tags:` tag filters. `None` means the token is unscoped and
/// has the full privileges of its user; `Some` means it is restricted to the
/// returned scopes. An empty or filter-tags-only scope list is treated as
/// unscoped, mirroring `check_scopes`/`check_route_access`.
fn scope_restrictions(scopes: Option<&[String]>) -> Option<Vec<&String>> {
let restrictions: Vec<&String> = scopes?
.iter()
.filter(|s| !s.starts_with("if_jobs:filter_tags:"))
.collect();
(!restrictions.is_empty()).then_some(restrictions)
}
/// Enforce monotonic privilege when a token lifecycle endpoint mints or rescopes
/// a credential on behalf of `authed`: the resulting credential must never be
/// more privileged than the caller's own token.
///
/// - An unscoped caller may grant any scopes (this is the existing UI/CLI flow).
/// - A scope-restricted caller may only grant scopes that are a subset of its
/// own, and may never produce an unscoped credential.
///
/// Without this, a `users:write` token could create or rescope a token to be
/// unscoped, and a `users:read` token could refresh into an unscoped session —
/// escaping its own restrictions.
pub fn ensure_scopes_within_caller(
authed: &ApiAuthed,
requested_scopes: Option<&[String]>,
) -> error::Result<()> {
if let Some(caller_restrictions) = scope_restrictions(authed.scopes.as_deref()) {
let Some(requested_restrictions) = scope_restrictions(requested_scopes) else {
return Err(Error::PermissionDenied(
"A scope-restricted token cannot create or update a token with broader (unscoped) \
privileges"
.to_string(),
));
};
// MCP scopes (`mcp:all`, `mcp:favorites`, `mcp:scripts:*`, etc.) use a
// custom format that ScopeDefinition::from_scope_string parses
// permissively but the MCP runtime interprets via its own parser
// (parse_mcp_scopes). The two views disagree — e.g. the generic parser
// accepts `mcp:scripts` as an unrestricted-resource scope, while the
// MCP runtime ignores it as unrecognized but interprets `mcp:scripts:*`
// as granting all scripts. So generic containment would silently allow
// `mcp:scripts` → `mcp:scripts:*` (a widening). Legitimate MCP token
// issuance goes through the OAuth gateway (mcp/oauth_server.rs), not
// these user-token endpoints, so require byte-identical match for MCP
// scopes here rather than trying to mirror MCP semantics in two places.
// Unparseable non-MCP caller scopes are intentionally dropped
// (fail-closed): a caller scope that fails to parse can only narrow
// the set of requested scopes that get covered, never widen it.
// Unparseable requested scopes surface as `BadRequest`, which is what
// we want — the client is sending garbage.
let parsed_caller: Vec<ScopeDefinition> = caller_restrictions
.iter()
.filter(|s| !s.starts_with("mcp:"))
.filter_map(|s| ScopeDefinition::from_scope_string(s).ok())
.collect();
let caller_mcp: std::collections::HashSet<&str> = caller_restrictions
.iter()
.filter(|s| s.starts_with("mcp:"))
.map(|s| s.as_str())
.collect();
for requested in requested_restrictions {
if requested.starts_with("mcp:") {
if !caller_mcp.contains(requested.as_str()) {
return Err(Error::PermissionDenied(format!(
"A scope-restricted token cannot grant MCP scope '{requested}' unless the \
caller holds the same scope verbatim"
)));
}
continue;
}
let requested_scope = ScopeDefinition::from_scope_string(requested)?;
let covered = parsed_caller
.iter()
.any(|caller_scope| scope_contains(caller_scope, &requested_scope));
if !covered {
return Err(Error::PermissionDenied(format!(
"A scope-restricted token cannot grant scope '{requested}' which exceeds its \
own scopes"
)));
}
}
}
// `if_jobs:filter_tags:` fences which job tags a token can run on (enforced
// at job operations as `v2_job.tag = ANY(...)`), and is checked independently
// of domain/action/resource subset. A caller restricted by filter_tags must
// not be able to mint or rescope a credential that drops or widens the fence
// — even if the caller has no other scope restrictions (filter_tags-only
// tokens otherwise look "unscoped" to `scope_restrictions`).
if let Some(caller_tags) = first_filter_tags(authed.scopes.as_deref()) {
let Some(requested_tags) = first_filter_tags(requested_scopes) else {
return Err(Error::PermissionDenied(
"A token restricted by if_jobs:filter_tags cannot mint or rescope a token that \
drops the tag restriction"
.to_string(),
));
};
let caller_set: std::collections::HashSet<&str> = caller_tags.iter().copied().collect();
for tag in &requested_tags {
if !caller_set.contains(tag) {
return Err(Error::PermissionDenied(format!(
"A token restricted by if_jobs:filter_tags cannot grant tag '{tag}' which is \
not within its own filter_tags"
)));
}
}
}
Ok(())
}
/// Tags from the first `if_jobs:filter_tags:<a,b,...>` scope, matching the
/// semantics of [`get_scope_tags`] (which is what the job runtime consults).
/// Returns `None` if no such scope is present.
fn first_filter_tags(scopes: Option<&[String]>) -> Option<Vec<&str>> {
scopes?.iter().find_map(|s| {
s.strip_prefix("if_jobs:filter_tags:")
.map(|tags| tags.split(',').collect())
})
}
/// Whether `caller` grants at least everything `requested` grants (directional
/// containment).
///
/// This is intentionally NOT `ScopeDefinition::includes`: that method answers
/// "does this scope grant access to a required action" using OR semantics over
/// resources (any overlap counts, and a `*` on either side matches), which is
/// correct for access checks but unsafe for subset checks — it would let a
/// token scoped to `scripts:read:f/team/a` mint `scripts:read:*` or
/// `scripts:read:f/team/a,f/other/b`. Subset containment instead requires that
/// EVERY requested resource is covered by SOME caller resource.
fn scope_contains(caller: &ScopeDefinition, requested: &ScopeDefinition) -> bool {
if caller.domain != requested.domain {
return false;
}
// write subsumes read; otherwise the action must match exactly.
match (caller.action.as_str(), requested.action.as_str()) {
(c, r) if c == r || (c == "write" && r == "read") => {}
_ => return false,
}
if caller.domain == "jobs" && caller.action == "run" {
match (&caller.kind, &requested.kind) {
(Some(caller_kind), Some(requested_kind)) if caller_kind != requested_kind => {
return false
}
// Caller pinned to a kind, but the request covers any kind.
(Some(_), None) => return false,
_ => {}
}
}
match (&caller.resource, &requested.resource) {
// Caller is unrestricted on resources: covers everything.
(None, _) => true,
// Caller is resource-restricted but the request is not: broader.
(Some(_), None) => false,
(Some(caller_resources), Some(requested_resources)) => {
resource_set_contains(caller_resources, requested_resources)
}
}
}
/// Every resource in `requested` must be covered by some resource in `caller`.
fn resource_set_contains(caller: &[String], requested: &[String]) -> bool {
if caller.iter().any(|r| r == "*") {
return true;
}
requested
.iter()
.all(|req| req != "*" && caller.iter().any(|c| resource_covers(c, req)))
}
/// Directional: does the single caller resource pattern cover `requested`?
/// `caller` may be an exact path or a `<prefix>/*` subtree wildcard; `requested`
/// may itself be a subtree wildcard, in which case the whole requested subtree
/// must fall within the caller's subtree.
fn resource_covers(caller: &str, requested: &str) -> bool {
if caller == requested {
return true;
}
let Some(prefix) = caller.strip_suffix("/*") else {
// An exact caller resource only covers itself (handled above).
return false;
};
let requested_base = requested.strip_suffix("/*").unwrap_or(requested);
requested_base == prefix
|| (requested_base.starts_with(prefix)
&& requested_base.as_bytes().get(prefix.len()) == Some(&b'/'))
}
/// Returns a predicate that checks whether `path` is within the token's
/// scope for `{domain}:{action}:{path}`. For tokens without scope
/// restrictions (no scopes at all, or only `if_jobs:filter_tags:*` scopes),
@@ -574,6 +770,13 @@ impl NewToken {
}
}
/// Low-level token mint shared by trusted callers (the user-facing
/// `tokens/create` handler and internal mints such as native-trigger webhook
/// tokens). It does NOT enforce that `token_config.scopes` is within the
/// caller's own scopes — callers exposed to untrusted input must call
/// [`ensure_scopes_within_caller`] first (internal narrowing mints intentionally
/// skip it, since their scopes derive from the action being authorized, not the
/// caller's token).
pub async fn create_token_internal(
tx: &mut sqlx::PgConnection,
db: &DB,
@@ -908,4 +1111,278 @@ mod tests {
assert!(allowed("u/alice/foo"));
assert!(!allowed("u/alice/bar"));
}
fn opt_scopes(scopes: Option<Vec<&str>>) -> Option<Vec<String>> {
scopes.map(|v| v.into_iter().map(String::from).collect())
}
// Regression tests for WIN-1999: scoped user tokens must not be able to
// mint or rescope credentials with broader privileges than themselves.
#[test]
fn unscoped_caller_can_grant_anything() {
let authed = authed_with_scopes(None);
assert!(ensure_scopes_within_caller(&authed, None).is_ok());
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["jobs:run:scripts"])).as_deref()
)
.is_ok());
}
#[test]
fn filter_tags_only_caller_is_unrestricted_on_domain_action_dimension() {
// The domain/action/resource subset check treats filter-tags-only as
// unrestricted, mirroring check_scopes/check_route_access. The tag
// dimension is checked separately (see filter_tags_dimension_is_monotonic).
let authed = authed_with_scopes(Some(vec!["if_jobs:filter_tags:default"]));
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["users:write", "if_jobs:filter_tags:default"])).as_deref()
)
.is_ok());
}
#[test]
fn filter_tags_dimension_is_monotonic() {
// Caller restricted to tag fence "a" cannot drop the fence …
let single = authed_with_scopes(Some(vec!["if_jobs:filter_tags:a"]));
assert!(ensure_scopes_within_caller(&single, None).is_err());
assert!(
ensure_scopes_within_caller(&single, opt_scopes(Some(vec!["users:read"])).as_deref())
.is_err(),
"minting a token without filter_tags must be rejected"
);
// … cannot widen to a tag it lacks …
assert!(ensure_scopes_within_caller(
&single,
opt_scopes(Some(vec!["if_jobs:filter_tags:a,b"])).as_deref()
)
.is_err());
// … and cannot mint a token fenced on a disjoint tag.
assert!(ensure_scopes_within_caller(
&single,
opt_scopes(Some(vec!["if_jobs:filter_tags:b"])).as_deref()
)
.is_err());
// Narrowing or matching the tag fence is allowed.
let multi = authed_with_scopes(Some(vec!["if_jobs:filter_tags:a,b"]));
assert!(ensure_scopes_within_caller(
&multi,
opt_scopes(Some(vec!["if_jobs:filter_tags:a"])).as_deref()
)
.is_ok());
assert!(ensure_scopes_within_caller(
&multi,
opt_scopes(Some(vec!["if_jobs:filter_tags:a,b"])).as_deref()
)
.is_ok());
// A caller with a real scope plus a tag fence cannot drop just the fence.
let mixed = authed_with_scopes(Some(vec!["jobs:run:scripts", "if_jobs:filter_tags:a"]));
assert!(ensure_scopes_within_caller(
&mixed,
opt_scopes(Some(vec!["jobs:run:scripts"])).as_deref()
)
.is_err());
assert!(ensure_scopes_within_caller(
&mixed,
opt_scopes(Some(vec!["jobs:run:scripts", "if_jobs:filter_tags:a"])).as_deref()
)
.is_ok());
// An unrestricted caller may grant filter_tags freely.
let unscoped = authed_with_scopes(None);
assert!(ensure_scopes_within_caller(
&unscoped,
opt_scopes(Some(vec!["if_jobs:filter_tags:x"])).as_deref()
)
.is_ok());
}
#[test]
fn scoped_caller_cannot_mint_unscoped_token() {
// Primitive 2 in the report: a users:write token minting an unscoped token.
let authed = authed_with_scopes(Some(vec!["users:write"]));
assert!(ensure_scopes_within_caller(&authed, None).is_err());
// Empty scope list is effectively unscoped and must also be rejected.
assert!(ensure_scopes_within_caller(&authed, Some(&[])).is_err());
// A scope list of only tag filters is effectively unscoped too.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["if_jobs:filter_tags:default"])).as_deref()
)
.is_err());
}
#[test]
fn scoped_caller_cannot_remove_its_own_scopes() {
// Primitive 3 in the report: a users:write token setting its scopes to null.
let authed = authed_with_scopes(Some(vec!["users:write"]));
assert!(ensure_scopes_within_caller(&authed, None).is_err());
}
#[test]
fn scoped_caller_cannot_grant_scope_it_lacks() {
let authed = authed_with_scopes(Some(vec!["users:write"]));
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["jobs:run:scripts"])).as_deref()
)
.is_err());
}
#[test]
fn scoped_caller_can_grant_subset_of_own_scopes() {
let authed = authed_with_scopes(Some(vec!["users:write", "jobs:run:scripts"]));
// Equal scope is allowed.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["jobs:run:scripts"])).as_deref()
)
.is_ok());
// write implies read, so a narrower read scope is allowed.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["users:read"])).as_deref()
)
.is_ok());
// Tag filters narrow further and are always permitted.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["users:read", "if_jobs:filter_tags:default"])).as_deref()
)
.is_ok());
}
#[test]
fn scoped_caller_cannot_broaden_resource_scope() {
let authed = authed_with_scopes(Some(vec!["scripts:read:f/team/*"]));
// Narrower resource within the subtree is allowed.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["scripts:read:f/team/sub"])).as_deref()
)
.is_ok());
// A nested subtree within the caller's subtree is allowed.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["scripts:read:f/team/sub/*"])).as_deref()
)
.is_ok());
// The subtree root itself is allowed.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["scripts:read:f/team"])).as_deref()
)
.is_ok());
// A path outside the subtree is rejected.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["scripts:read:f/other/x"])).as_deref()
)
.is_err());
// read caller cannot grant write.
assert!(ensure_scopes_within_caller(
&authed,
opt_scopes(Some(vec!["scripts:write:f/team/db"])).as_deref()
)
.is_err());
}
#[test]
fn mcp_scopes_require_byte_identical_match() {
// Regression for the access-grant-OR vs runtime-MCP-parser confusion:
// ScopeDefinition treats `mcp:scripts` as an unrestricted-resource scope
// and `mcp:scripts:*` as a strictly narrower one, so generic containment
// would silently allow widening. The MCP runtime however ignores
// `mcp:scripts` (unrecognized) while `mcp:scripts:*` grants all scripts.
// Legitimate MCP token issuance is the OAuth gateway, not these
// user-token endpoints, so MCP scopes must match the caller verbatim.
// The bypass the reviewer flagged: malformed `mcp:scripts` would widen
// into the real `mcp:scripts:*` under generic containment.
let bypass = authed_with_scopes(Some(vec!["users:write", "mcp:scripts"]));
assert!(ensure_scopes_within_caller(
&bypass,
opt_scopes(Some(vec!["users:write", "mcp:scripts:*"])).as_deref()
)
.is_err());
// A caller without any MCP scope cannot grant one (widening on the MCP
// dimension), even if the rest of the requested scopes are within reach.
let no_mcp = authed_with_scopes(Some(vec!["users:write"]));
assert!(ensure_scopes_within_caller(
&no_mcp,
opt_scopes(Some(vec!["users:write", "mcp:scripts:*"])).as_deref()
)
.is_err());
// Byte-identical MCP scope passes; an additional non-matching MCP scope
// alongside it does not.
let mcp_caller = authed_with_scopes(Some(vec!["mcp:scripts:*"]));
assert!(ensure_scopes_within_caller(
&mcp_caller,
opt_scopes(Some(vec!["mcp:scripts:*"])).as_deref()
)
.is_ok());
assert!(ensure_scopes_within_caller(
&mcp_caller,
opt_scopes(Some(vec!["mcp:scripts:*", "mcp:flows:*"])).as_deref()
)
.is_err());
// Even a narrowing within MCP semantics (`mcp:all` → `mcp:scripts:*`)
// is rejected by the byte-identical rule. This is intentional — these
// endpoints are not the legitimate path for narrowing MCP tokens.
let mcp_all = authed_with_scopes(Some(vec!["mcp:all"]));
assert!(ensure_scopes_within_caller(
&mcp_all,
opt_scopes(Some(vec!["mcp:scripts:*"])).as_deref()
)
.is_err());
}
#[test]
fn scoped_caller_cannot_escalate_to_wildcard_or_superset() {
// Regression for the access-grant-OR vs subset-containment confusion:
// ScopeDefinition::includes would (incorrectly) allow all of these.
let star = authed_with_scopes(Some(vec!["scripts:read:f/team/a"]));
// Minting `*` from a single-path scope must be rejected.
assert!(ensure_scopes_within_caller(
&star,
opt_scopes(Some(vec!["scripts:read:*"])).as_deref()
)
.is_err());
// Minting a broader subtree must be rejected.
assert!(ensure_scopes_within_caller(
&star,
opt_scopes(Some(vec!["scripts:read:f/team/*"])).as_deref()
)
.is_err());
// A comma-separated list that adds an uncovered resource must be rejected,
// even though one element overlaps the caller's scope.
let list = authed_with_scopes(Some(vec!["scripts:read:f/team/a"]));
assert!(ensure_scopes_within_caller(
&list,
opt_scopes(Some(vec!["scripts:read:f/team/a,f/other/b"])).as_deref()
)
.is_err());
// A subset of a multi-resource caller scope is allowed.
let multi = authed_with_scopes(Some(vec!["scripts:read:f/team/a,f/team/b"]));
assert!(ensure_scopes_within_caller(
&multi,
opt_scopes(Some(vec!["scripts:read:f/team/a"])).as_deref()
)
.is_ok());
// A wildcard caller covers any subset, but not `*`-less escalation rules apply
// only when the caller itself lacks `*`.
let wildcard = authed_with_scopes(Some(vec!["scripts:read:*"]));
assert!(ensure_scopes_within_caller(
&wildcard,
opt_scopes(Some(vec!["scripts:read:f/team/a"])).as_deref()
)
.is_ok());
}
}
+23 -14
View File
@@ -558,13 +558,17 @@ async fn create_flow(
w_id
).execute(&mut *tx).await?;
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'flow'",
nf.path,
&w_id
)
.execute(&mut *tx)
.await?;
// CLI / git-sync deploys ask us to preserve any existing user draft at this
// path instead of wiping it as part of the deploy.
if !nf.skip_draft_deletion.unwrap_or(false) {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'flow'",
nf.path,
&w_id
)
.execute(&mut *tx)
.await?;
}
audit_log(
&mut *tx,
@@ -1157,13 +1161,17 @@ async fn update_flow(
})?;
}
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'flow'",
flow_path,
&w_id
)
.execute(&mut *tx)
.await?;
// CLI / git-sync deploys ask us to preserve any existing user draft at this
// path instead of wiping it as part of the deploy.
if !nf.skip_draft_deletion.unwrap_or(false) {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'flow'",
flow_path,
&w_id
)
.execute(&mut *tx)
.await?;
}
audit_log(
&mut *tx,
@@ -2031,6 +2039,7 @@ mod tests {
})),
preprocessor_module: None,
same_worker: false,
preserve_step_tags: false,
skip_expr: None,
cache_ttl: None,
cache_ignore_s3_path: None,
@@ -383,6 +383,8 @@ async fn remove_granular_acl(
// workspace export.
let table = if kind == "raw_app" { "app" } else { kind };
// SAFETY: `kind` has been validated against the `KINDS` allowlist before reaching this function.
// LIMIT 1: `script` shares (workspace_id, path) across versions, so `old` can
// return >1 row, which would break the scalar subquery in RETURNING.
let obj_o = sqlx::query_scalar::<_, bool>(&format!(
"WITH old AS (
SELECT extra_perms->$1 as old_write FROM {table}
@@ -390,7 +392,7 @@ async fn remove_granular_acl(
)
UPDATE {table} SET extra_perms = extra_perms - $1
WHERE {identifier} = $2 AND workspace_id = $3 AND extra_perms ? $1
RETURNING (SELECT old_write FROM old)::bool"
RETURNING (SELECT old_write FROM old LIMIT 1)::bool"
))
.bind(&owner)
.bind(path)
@@ -0,0 +1,358 @@
/*!
* Integration test for workspace encryption key rotation triggering git sync.
*
* Regression test for windmill-labs/windmill#9344 re-encrypting all secret
* variables on workspace key change must dispatch a git-sync job that carries
* every re-encrypted variable plus the encryption_key entry, so repos with
* Secrets sync enabled receive the new ciphertexts in one commit.
*
* Run with enterprise features:
* ```bash
* cargo test --test workspace_encryption_key_git_sync --features enterprise,private
* ```
*/
use serde_json::json;
use sqlx::{Pool, Postgres};
use std::time::Duration;
#[allow(unused_imports)]
use windmill_test_utils::*;
fn client() -> reqwest::Client {
reqwest::Client::new()
}
fn authed(builder: reqwest::RequestBuilder) -> reqwest::RequestBuilder {
builder.header("Authorization", "Bearer SECRET_TOKEN")
}
#[allow(dead_code)]
async fn create_git_repo_resource(db: &Pool<Postgres>) -> anyhow::Result<()> {
sqlx::query(
r#"
INSERT INTO resource (workspace_id, path, value, resource_type, extra_perms, created_by)
VALUES ('test-workspace', 'u/test-user/test_git_repo', $1::jsonb, 'git_repository', '{}'::jsonb, 'test-user')
ON CONFLICT (workspace_id, path) DO NOTHING
"#,
)
.bind(json!({
"url": "https://github.com/test/test.git",
"branch": "main",
"token": "test-token"
}))
.execute(db)
.await?;
Ok(())
}
#[allow(dead_code)]
async fn create_folder(db: &Pool<Postgres>, name: &str) -> anyhow::Result<()> {
sqlx::query(
r#"
INSERT INTO folder (workspace_id, name, display_name, owners, extra_perms, created_by)
VALUES ('test-workspace', $1, $1, ARRAY['u/test-user'], '{}'::jsonb, 'test-user')
ON CONFLICT (workspace_id, name) DO NOTHING
"#,
)
.bind(name)
.execute(db)
.await?;
Ok(())
}
#[allow(dead_code)]
async fn create_sync_script(db: &Pool<Postgres>, path: &str) -> anyhow::Result<i64> {
let hash: i64 = rand::random::<i64>().unsigned_abs() as i64;
sqlx::query(
r#"
INSERT INTO script (workspace_id, hash, path, summary, description, content,
created_by, language, kind, lock)
VALUES ('test-workspace', $1, $2, 'sync script', '',
'export function main(items: any[]) { return { synced: items.length }; }',
'test-user', 'bun', 'script', '')
"#,
)
.bind(hash)
.bind(path)
.execute(db)
.await?;
Ok(hash)
}
#[allow(dead_code)]
async fn setup_git_sync_config(db: &Pool<Postgres>, sync_script_path: &str) -> anyhow::Result<()> {
// Include Variable + Secret + Key so the encryption rotation has a reason
// to push every re-encrypted variable. Anchor include_path to root so all
// u/... and f/... paths pass the regex filter.
let git_sync_config = json!({
"include_type": ["variable", "secret", "key"],
"include_path": ["**"],
"repositories": [{
"script_path": sync_script_path,
"git_repo_resource_path": "$res:u/test-user/test_git_repo",
"use_individual_branch": false,
"group_by_folder": false
}]
});
sqlx::query!(
"UPDATE workspace_settings SET git_sync = $1 WHERE workspace_id = $2",
git_sync_config,
"test-workspace"
)
.execute(db)
.await?;
Ok(())
}
/// Insert N secret variables, encrypting their values with the workspace's
/// current key so the re-encryption path can decrypt them.
#[allow(dead_code)]
async fn insert_secret_variables(db: &Pool<Postgres>, paths: &[&str]) -> anyhow::Result<()> {
use windmill_common::variables::{build_crypt, encrypt};
let mc = build_crypt(db, "test-workspace").await?;
for path in paths {
let plaintext = format!("secret-value-for-{path}");
let encrypted = encrypt(&mc, &plaintext);
sqlx::query!(
r#"
INSERT INTO variable (workspace_id, path, value, is_secret, description, extra_perms, account)
VALUES ($1, $2, $3, true, '', '{}'::jsonb, NULL)
ON CONFLICT (workspace_id, path) DO UPDATE SET value = EXCLUDED.value
"#,
"test-workspace",
path,
encrypted,
)
.execute(db)
.await?;
}
Ok(())
}
#[derive(Debug)]
#[allow(dead_code)]
struct DeploymentCallbackJob {
id: uuid::Uuid,
args: Option<serde_json::Value>,
}
/// Poll until at least `min_count` deployment-callback jobs exist for the
/// script path, or the timeout elapses. Returns whatever was found.
#[allow(dead_code)]
async fn wait_for_deployment_callbacks(
db: &Pool<Postgres>,
script_path: &str,
min_count: usize,
timeout: Duration,
) -> anyhow::Result<Vec<DeploymentCallbackJob>> {
let deadline = tokio::time::Instant::now() + timeout;
loop {
let rows = sqlx::query_as!(
DeploymentCallbackJob,
r#"
SELECT j.id, j.args
FROM v2_job j
JOIN v2_job_queue q ON j.id = q.id
WHERE j.runnable_path = $1
AND j.kind = 'deploymentcallback'
AND j.workspace_id = 'test-workspace'
ORDER BY j.created_at DESC
"#,
script_path,
)
.fetch_all(db)
.await?;
if rows.len() >= min_count || tokio::time::Instant::now() >= deadline {
return Ok(rows);
}
tokio::time::sleep(Duration::from_millis(100)).await;
}
}
#[cfg(all(feature = "enterprise", feature = "private"))]
#[sqlx::test(migrations = "../migrations", fixtures("base"))]
async fn test_encryption_key_rotation_dispatches_batched_git_sync(
db: Pool<Postgres>,
) -> anyhow::Result<()> {
initialize_tracing().await;
// Setup git sync repo + sync script (folder/path encodes the hub min-version)
create_folder(&db, "28103").await?;
create_git_repo_resource(&db).await?;
let sync_script_path = "f/28103/test_sync_script_encryption";
create_sync_script(&db, sync_script_path).await?;
setup_git_sync_config(&db, sync_script_path).await?;
let secret_paths = [
"u/test-user/secret_a",
"u/test-user/secret_b",
"u/test-user/secret_c",
];
insert_secret_variables(&db, &secret_paths).await?;
let server = ApiServer::start(db.clone()).await?;
let port = server.addr.port();
let base = format!("http://localhost:{port}/api/w/test-workspace/workspaces");
// 64-char alphanumeric per the route's WORKSPACE_KEY_REGEXP
let new_key = "a".repeat(64);
let resp = authed(client().post(format!("{base}/encryption_key")))
.json(&json!({"new_key": new_key, "skip_reencrypt": false}))
.send()
.await?;
assert_eq!(
resp.status(),
200,
"set_encryption_key failed: {}",
resp.text().await?
);
// The git-sync dispatch runs in a tokio::spawn'd task. Poll up to a few
// seconds for the deployment callback to land in the queue.
let jobs =
wait_for_deployment_callbacks(&db, sync_script_path, 1, Duration::from_secs(5)).await?;
assert_eq!(
jobs.len(),
1,
"expected exactly one batched deployment callback job, got {}",
jobs.len()
);
let job = &jobs[0];
let args = job.args.as_ref().expect("job should have args");
let items = args
.get("items")
.and_then(|v| v.as_array())
.expect("args.items should be a JSON array");
// Expect exactly one job carrying the Key entry + every re-encrypted variable
assert_eq!(
items.len(),
secret_paths.len() + 1,
"expected {} items (key + {} variables) in a single sync job, got {} — items: {:#?}",
secret_paths.len() + 1,
secret_paths.len(),
items.len(),
items
);
let mut variable_paths: Vec<String> = Vec::new();
let mut saw_key = false;
for item in items {
let path_type = item.get("path_type").and_then(|v| v.as_str()).unwrap_or("");
let path = item.get("path").and_then(|v| v.as_str()).unwrap_or("");
match path_type {
"variable" => variable_paths.push(path.to_string()),
"key" => saw_key = true,
other => panic!("unexpected path_type in batch: {other}"),
}
}
assert!(
saw_key,
"expected a path_type=key entry in items: {:#?}",
items
);
variable_paths.sort();
let mut expected: Vec<String> = secret_paths.iter().map(|s| s.to_string()).collect();
expected.sort();
assert_eq!(
variable_paths, expected,
"items array should contain every re-encrypted secret variable"
);
// Secrets sync is enabled (ObjectType::Secret in include_type), so the sync
// script should be invoked with skip_secret=false.
let skip_secret = args
.get("skip_secret")
.and_then(|v| v.as_bool())
.expect("args.skip_secret should be set when batch carries variables");
assert!(
!skip_secret,
"skip_secret should be false when Secret is included in the repo's types"
);
Ok(())
}
/// Regression test for the non-debouncing fallback: a workspace whose sync
/// script predates hub version 28103 must still receive git-sync jobs for the
/// encryption_key entry and every re-encrypted secret. Before the fallback was
/// added, the batch path `continue`d past such repos and queued nothing,
/// silently leaving the repo stale after a key rotation.
#[cfg(all(feature = "enterprise", feature = "private"))]
#[sqlx::test(migrations = "../migrations", fixtures("base"))]
async fn test_encryption_key_rotation_falls_back_without_debouncing(
db: Pool<Postgres>,
) -> anyhow::Result<()> {
initialize_tracing().await;
// Folder/path encodes a hub version BELOW 28103, so
// is_script_meets_min_version(28103) is false → debouncing unsupported.
create_folder(&db, "28000").await?;
create_git_repo_resource(&db).await?;
let sync_script_path = "f/28000/test_sync_script_legacy";
create_sync_script(&db, sync_script_path).await?;
setup_git_sync_config(&db, sync_script_path).await?;
let secret_paths = ["u/test-user/secret_a", "u/test-user/secret_b"];
insert_secret_variables(&db, &secret_paths).await?;
let server = ApiServer::start(db.clone()).await?;
let port = server.addr.port();
let base = format!("http://localhost:{port}/api/w/test-workspace/workspaces");
let new_key = "b".repeat(64);
let resp = authed(client().post(format!("{base}/encryption_key")))
.json(&json!({"new_key": new_key, "skip_reencrypt": false}))
.send()
.await?;
assert_eq!(
resp.status(),
200,
"set_encryption_key failed: {}",
resp.text().await?
);
// Legacy fallback pushes one job per item (flat args, no `items` array):
// the Key entry + one per re-encrypted variable.
let expected = secret_paths.len() + 1;
let jobs =
wait_for_deployment_callbacks(&db, sync_script_path, expected, Duration::from_secs(5))
.await?;
assert_eq!(
jobs.len(),
expected,
"expected {expected} legacy deployment-callback jobs (key + {} variables), got {} — a repo on an old sync script must not be silently skipped",
secret_paths.len(),
jobs.len()
);
let mut variable_paths: Vec<String> = Vec::new();
let mut saw_key = false;
for job in &jobs {
let args = job.args.as_ref().expect("job should have args");
// Legacy format: flat fields, never an `items` array.
assert!(
args.get("items").is_none(),
"fallback jobs must use the flat legacy format, not an items array: {args:#?}"
);
let path_type = args.get("path_type").and_then(|v| v.as_str()).unwrap_or("");
let path = args.get("path").and_then(|v| v.as_str()).unwrap_or("");
match path_type {
"variable" => variable_paths.push(path.to_string()),
"key" => saw_key = true,
other => panic!("unexpected path_type in fallback job: {other}"),
}
}
assert!(saw_key, "expected a path_type=key fallback job");
variable_paths.sort();
let mut expected_paths: Vec<String> = secret_paths.iter().map(|s| s.to_string()).collect();
expected_paths.sort();
assert_eq!(
variable_paths, expected_paths,
"fallback must queue a job for every re-encrypted secret variable"
);
Ok(())
}
+16 -8
View File
@@ -739,6 +739,9 @@ async fn is_noop_deploy_against_parent(
// caller-intent flag (auto-resolve parent), not script state
auto_parent: _,
labels,
// caller-intent flag (preserve user drafts on CLI/git-sync deploys);
// transient, never persisted, does not change what the script *is*
skip_draft_deletion: _,
} = ns;
if path != &parent.path {
@@ -927,6 +930,9 @@ async fn create_script_internal<'c>(
}
}
let script_path = ns.path.clone();
// Caller-intent: CLI / git-sync deploys ask us to preserve any existing
// user draft at this path instead of wiping it as part of the deploy.
let skip_draft_deletion = ns.skip_draft_deletion.unwrap_or(false);
let hash = ScriptHash(hash_script(&ns));
let authed = maybe_refresh_folders(&ns.path, &w_id, authed, &db).await;
@@ -1395,13 +1401,15 @@ async fn create_script_internal<'c>(
let p_path_opt = parent_hashes_and_perms.as_ref().map(|x| x.p_path.clone());
if let Some(ref p_path) = p_path_opt {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'script'",
p_path,
&w_id
)
.execute(&mut *tx)
.await?;
if !skip_draft_deletion {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'script'",
p_path,
&w_id
)
.execute(&mut *tx)
.await?;
}
sqlx::query!(
"UPDATE capture_config SET path = $1 WHERE path = $2 AND workspace_id = $3 AND is_flow IS FALSE",
@@ -1480,7 +1488,7 @@ async fn create_script_internal<'c>(
tx = push_scheduled_job(&db, tx, &schedule, None, None).await?;
}
}
} else {
} else if !skip_draft_deletion {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'script'",
ns.path,
+61 -9
View File
@@ -6,7 +6,10 @@
* LICENSE-AGPL for a copy of the license.
*/
use std::{collections::HashMap, time::Duration};
use std::{
collections::{BTreeSet, HashMap},
time::Duration,
};
#[cfg(feature = "parquet")]
mod audit_logs_s3;
@@ -47,6 +50,7 @@ use windmill_common::secret_backend::{
AwsSecretsManagerSettings, AzureKeyVaultSettings, SecretMigrationReport, VaultSettings,
};
use windmill_common::{
auth::is_super_admin_email,
ee_oss::{get_license_plan, LicensePlan},
email_oss::send_email_plain_text,
error::{self, JsonResult, Result},
@@ -1135,6 +1139,8 @@ struct CustomInstanceDb {
success: bool,
error: Option<String>,
tag: Option<String>,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
used_by_workspaces: Vec<String>,
}
#[derive(Deserialize, Debug, Serialize, Default)]
@@ -1154,7 +1160,7 @@ struct CustomInstanceDbLogs {
}
async fn list_custom_instance_pg_databases(
_authed: ApiAuthed,
authed: ApiAuthed,
Extension(db): Extension<DB>,
) -> JsonResult<HashMap<String, CustomInstanceDb>> {
let result = sqlx::query_scalar!(
@@ -1163,12 +1169,57 @@ async fn list_custom_instance_pg_databases(
.fetch_one(&db)
.await?
.ok_or_else(|| error::Error::ExecutionErr("Couldn't find custom_instance_pg_databases".to_string()))?;
let result = serde_json::from_value(result).map_err(|e| {
error::Error::ExecutionErr(format!(
"couldn't parse custom_instance_pg_databases.databases : {}",
e.to_string()
))
})?;
let mut result: HashMap<String, CustomInstanceDb> =
serde_json::from_value(result).map_err(|e| {
error::Error::ExecutionErr(format!(
"couldn't parse custom_instance_pg_databases.databases : {}",
e.to_string()
))
})?;
if is_super_admin_email(&db, &authed.email).await? {
// Enrich each database with the list of workspaces referencing it through
// either a ducklake catalog or a datatable database whose resource_type is
// 'instance'. Not stored in DB to avoid drift.
let usages = sqlx::query!(
r#"
SELECT ws.workspace_id AS "workspace_id!", entry->'catalog'->>'resource_path' AS dbname
FROM workspace_settings ws
CROSS JOIN LATERAL jsonb_each(
CASE WHEN jsonb_typeof(ws.ducklake->'ducklakes') = 'object'
THEN ws.ducklake->'ducklakes'
ELSE '{}'::jsonb END
) AS dl(k, entry)
WHERE entry->'catalog'->>'resource_type' = 'instance'
AND entry->'catalog'->>'resource_path' IS NOT NULL
UNION ALL
SELECT ws.workspace_id AS "workspace_id!", entry->'database'->>'resource_path' AS dbname
FROM workspace_settings ws
CROSS JOIN LATERAL jsonb_each(
CASE WHEN jsonb_typeof(ws.datatable->'datatables') = 'object'
THEN ws.datatable->'datatables'
ELSE '{}'::jsonb END
) AS dt(k, entry)
WHERE entry->'database'->>'resource_type' = 'instance'
AND entry->'database'->>'resource_path' IS NOT NULL
"#,
)
.fetch_all(&db)
.await?;
let mut by_db: HashMap<String, BTreeSet<String>> = HashMap::new();
for row in usages {
if let Some(dbname) = row.dbname {
by_db.entry(dbname).or_default().insert(row.workspace_id);
}
}
for (dbname, entry) in result.iter_mut() {
if let Some(workspaces) = by_db.remove(dbname) {
entry.used_by_workspaces = workspaces.into_iter().collect();
}
}
}
return Ok(Json(result));
}
@@ -1196,7 +1247,8 @@ async fn setup_custom_instance_pg_database(
let result = setup_custom_instance_pg_database_inner(authed, &db, &dbname, &mut logs).await;
let success = result.is_ok();
let error = result.err().map(|e| e.to_string());
let status = CustomInstanceDb { logs, success, error, tag: body.tag };
let status =
CustomInstanceDb { logs, success, error, tag: body.tag, used_by_workspaces: vec![] };
let status_json = serde_json::to_value(&status).map_err(to_anyhow)?;
// Save that the database was setup successfully
sqlx::query!(
+22 -5
View File
@@ -1964,7 +1964,8 @@ async fn login(
windmill_common::login_rate_limit::record_login_failure(&email);
Err(Error::BadRequest("Invalid login".to_string()))
} else {
let token = create_session_token(&email, super_admin, &mut tx, cookies).await?;
let token =
create_session_token(&email, super_admin, None, false, &mut tx, cookies).await?;
let audit_author = AuditAuthor {
email: email.clone(),
@@ -2036,7 +2037,15 @@ async fn refresh_token(
.await?
.unwrap_or(false);
let new_token = create_session_token(&authed.email, super_admin, &mut tx, cookies).await?;
let new_token = create_session_token(
&authed.email,
super_admin,
authed.scopes.as_deref(),
authed.read_only,
&mut tx,
cookies,
)
.await?;
audit_log(
&mut *tx,
@@ -2066,6 +2075,8 @@ lazy_static::lazy_static! {
pub async fn create_session_token<'c>(
email: &str,
super_admin: bool,
scopes: Option<&[String]>,
read_only: bool,
tx: &mut sqlx::Transaction<'c, sqlx::Postgres>,
cookies: Cookies,
) -> Result<String> {
@@ -2108,15 +2119,17 @@ pub async fn create_session_token<'c>(
sqlx::query!(
"INSERT INTO token
(token_hash, token_prefix, token, email, label, expiration, super_admin)
VALUES ($1, $2, $3, $4, $5, now() + ($6 || ' seconds')::interval, $7)",
(token_hash, token_prefix, token, email, label, expiration, super_admin, scopes, read_only)
VALUES ($1, $2, $3, $4, $5, now() + ($6 || ' seconds')::interval, $7, $8, $9)",
t_hash,
t_prefix,
plaintext as Option<&str>,
email,
"session",
&MAX_SESSION_VALIDITY_SECONDS.to_string(),
super_admin
super_admin,
scopes,
read_only,
)
.execute(&mut **tx)
.await?;
@@ -2146,6 +2159,8 @@ async fn create_token(
) -> Result<(StatusCode, String)> {
check_token_create_rate_limit(&authed.username)?;
windmill_api_auth::ensure_scopes_within_caller(&authed, token_config.scopes.as_deref())?;
let mut tx = db.begin().await?;
let token = create_token_internal(&mut *tx, &db, &authed, token_config).await?;
@@ -2353,6 +2368,8 @@ async fn update_token_scopes(
Path(token_prefix): Path<String>,
Json(req): Json<UpdateTokenScopesRequest>,
) -> Result<String> {
windmill_api_auth::ensure_scopes_within_caller(&authed, req.scopes.as_deref())?;
let mut tx = db.begin().await?;
let updated: Option<String> = sqlx::query_scalar!(
@@ -55,7 +55,10 @@ use windmill_common::{
use windmill_dep_map::scoped_dependency_map::{
DependencyDependent, DependencyMap, ScopedDependencyMap,
};
use windmill_git_sync::{handle_deployment_metadata, handle_fork_branch_creation, DeployedObject};
use windmill_git_sync::{
handle_deployment_metadata, handle_deployment_metadata_batch, handle_fork_branch_creation,
DeployedObject,
};
use windmill_types::s3::LargeFileStorage;
use hyper::StatusCode;
@@ -3403,6 +3406,7 @@ async fn set_encryption_key(
.execute(&mut *tx)
.await?;
let mut reencrypted_secret_paths: Vec<String> = Vec::new();
if !request.skip_reencrypt.unwrap_or(false) {
// Build the new cipher directly from the key string, since the transaction
// hasn't committed yet and build_crypt() would read the old key from the pool.
@@ -3448,6 +3452,7 @@ async fn set_encryption_key(
)
.execute(&mut *tx)
.await?;
reencrypted_secret_paths.push(variable.path);
}
}
@@ -3456,16 +3461,23 @@ async fn set_encryption_key(
// Invalidate the cache only after the transaction has committed
WORKSPACE_CRYPT_CACHE.remove(w_id.as_str());
// Trigger git sync for encryption key changes
handle_deployment_metadata(
// Build the batch: one event for the encryption key itself plus one per
// re-encrypted secret variable. The batch entrypoint dispatches a single
// git-sync job per repo carrying all items, so repos with Secrets sync
// enabled receive the new ciphertexts in one commit.
let mut batch: Vec<DeployedObject> = Vec::with_capacity(reencrypted_secret_paths.len() + 1);
batch.push(DeployedObject::Key { key_type: "encryption_key".to_string() });
for path in reencrypted_secret_paths {
batch.push(DeployedObject::Variable { path: path.clone(), parent_path: Some(path) });
}
handle_deployment_metadata_batch(
&authed.email,
&authed.username,
&db,
&w_id,
windmill_git_sync::DeployedObject::Key { key_type: "encryption_key".to_string() },
batch,
Some("Encryption key updated".to_string()),
false,
None,
)
.await?;
+27 -1
View File
@@ -1,7 +1,7 @@
openapi: "3.0.3"
info:
version: 1.711.0
version: 1.714.0
title: Windmill API
contact:
@@ -9751,6 +9751,9 @@ paths:
type: boolean
deployment_message:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this flow does not delete an existing user draft at the same path."
responses:
"201":
description: flow created
@@ -9792,6 +9795,9 @@ paths:
properties:
deployment_message:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this flow does not delete an existing user draft at the same path."
responses:
"200":
@@ -10290,6 +10296,9 @@ paths:
type: array
items:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this app does not delete an existing user draft at the same path."
required:
- path
- value
@@ -10342,6 +10351,9 @@ paths:
type: array
items:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this app does not delete an existing user draft at the same path."
required:
- path
- value
@@ -10660,6 +10672,9 @@ paths:
type: array
items:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this app does not delete an existing user draft at the same path."
responses:
"200":
description: app updated
@@ -10706,6 +10721,9 @@ paths:
type: array
items:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this app does not delete an existing user draft at the same path."
js:
type: string
css:
@@ -21952,6 +21970,9 @@ components:
type: array
items:
type: string
skip_draft_deletion:
type: boolean
description: "When true (set by the CLI / git sync), deploying this script does not delete an existing user draft at the same path."
required:
- path
@@ -25475,6 +25496,11 @@ components:
example: "Connection timeout"
tag:
$ref: "#/components/schemas/CustomInstanceDbTag"
used_by_workspaces:
type: array
items:
type: string
description: Workspaces that reference this database via a ducklake catalog or datatable database with resource_type 'instance'. Computed at request time, not persisted.
NewSqsTrigger:
type: object
+125 -108
View File
@@ -11,13 +11,14 @@ use http::{HeaderMap, Method};
use quick_cache::sync::Cache;
use reqwest::{Client, RequestBuilder};
use serde::{Deserialize, Serialize};
use serde_json::{json, value::RawValue};
use serde_json::value::RawValue;
use std::collections::HashMap;
use std::time::Duration;
use windmill_ai::ai_cache::current_instance_ai_config_revision;
use windmill_ai::ai_providers::{
empty_string_as_none, AIPlatform, AIProvider, ProviderConfig, ProviderModel,
};
use windmill_ai::credentials::ProviderCredentials;
#[cfg(feature = "bedrock")]
use windmill_ai::providers::bedrock::{
handle_bedrock_proxy, BedrockProxyResponse, BedrockProxyResponseBody,
@@ -30,10 +31,9 @@ use windmill_ai::providers::{
},
};
use windmill_ai::proxy::{
proxy_execution_mode, supports_query_builder_proxy, ProviderCredentials, ProxyBuildArgs,
ProxyExecutionMode, ProxyRequest,
fim::maybe_transform_fim_request, proxy_execution_mode, ProxyBuildArgs, ProxyExecutionMode,
ProxyRequest,
};
use windmill_ai::utils::AI_HTTP_HEADERS;
use windmill_audit::{audit_oss::audit_log, ActionKind};
use windmill_common::db::UserDB;
use windmill_common::error::{to_anyhow, Error, Result};
@@ -108,6 +108,13 @@ lazy_static::lazy_static! {
.timeout(std::time::Duration::from_secs(*AI_TIMEOUT_SECS))
.pool_max_idle_per_host(HTTP_POOL_MAX_IDLE_PER_HOST)
.pool_idle_timeout(Some(std::time::Duration::from_secs(HTTP_POOL_IDLE_TIMEOUT_SECS)))
// The SSRF check in `get_base_url` only validates the configured `base_url`.
// reqwest follows up to 10 redirects by default and does not revalidate the
// hops, so a public base_url could 3xx the server into a private/internal
// address. Disable redirect following so the validated host is the only one
// we ever connect to. AI APIs respond directly and do not rely on redirects,
// so this holds even for ALLOW_PRIVATE_AI_BASE_URLS deployments.
.redirect(reqwest::redirect::Policy::none())
.user_agent("windmill/beta"))
.build()
.expect("Failed to build AI HTTP client - check system TLS configuration");
@@ -298,6 +305,22 @@ async fn get_token_using_oauth(
resource.client_id = resolve_var(resource.client_id, db, w_id, user_db, authed).await?;
resource.client_secret = resolve_var(resource.client_secret, db, w_id, user_db, authed).await?;
resource.token_url = resolve_var(resource.token_url, db, w_id, user_db, authed).await?;
// Validate the resolved token_url against SSRF rules before issuing the request,
// mirroring the protection applied to base_url in `get_base_url` (same
// ALLOW_PRIVATE_AI_BASE_URLS opt-in). Without this a workspace member could
// point token_url at an internal/metadata address.
if !*windmill_ai::ai_providers::ALLOW_PRIVATE_AI_BASE_URLS {
use windmill_common::ssrf::SsrfValidationError;
windmill_common::ssrf::validate_url_for_ssrf(&resource.token_url)
.await
.map_err(|e| match e {
e @ SsrfValidationError::Private { .. } => Error::BadRequest(format!(
"{e}. If you need to use private/internal AI endpoints, \
set the ALLOW_PRIVATE_AI_BASE_URLS=true environment variable"
)),
e => Error::from(e),
})?;
}
let mut params = HashMap::new();
params.insert("grant_type", "client_credentials");
params.insert("scope", "https://cognitiveservices.azure.com/.default");
@@ -369,53 +392,6 @@ impl AIConfig {
}
}
// FIM (Fill-in-the-Middle) simulation for providers that don't support native FIM
#[derive(Deserialize, Debug)]
struct FimRequest {
model: String,
prompt: String, // code before cursor
suffix: Option<String>, // code after cursor
temperature: Option<f32>,
max_tokens: Option<u32>,
stop: Option<Vec<String>>,
}
/// Checks if the AI provider supports native FIM (Fill-in-the-Middle) endpoint
fn supports_native_fim(provider: &AIProvider) -> bool {
matches!(provider, AIProvider::Mistral)
}
/// Transforms a FIM request to chat/completions format for providers that don't support native FIM.
fn transform_fim_to_chat_completions(body: &Bytes) -> Result<(Bytes, String)> {
let fim_req: FimRequest = serde_json::from_slice(body)
.map_err(|e| Error::internal_err(format!("Failed to parse FIM request: {}", e)))?;
let suffix = fim_req.suffix.unwrap_or_default();
let system_prompt = "You are a code completion assistant. Complete the code at the <CURSOR/> position between the given prefix and suffix. Output ONLY the code that goes at the cursor - no explanations, no markdown, no repeating the prefix or suffix.";
let user_content = format!(
"<PREFIX>\n{}\n<CURSOR/>\n<SUFFIX>\n{}",
fim_req.prompt, suffix
);
let chat_req = json!({
"model": fim_req.model,
"messages": [
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_content}
],
"temperature": fim_req.temperature.unwrap_or(0.0),
"max_tokens": fim_req.max_tokens.unwrap_or(256),
"stop": fim_req.stop
});
let chat_body = serde_json::to_vec(&chat_req)
.map_err(|e| Error::internal_err(format!("Failed to serialize chat request: {}", e)))?;
Ok((Bytes::from(chat_body), "chat/completions".to_string()))
}
pub fn global_service() -> Router {
Router::new().route("/proxy/{*ai}", post(global_proxy).get(global_proxy))
}
@@ -455,6 +431,24 @@ fn proxy_request_to_request_builder(proxy_request: ProxyRequest) -> RequestBuild
request.body(proxy_request.body)
}
async fn audit_global_ai_request(db: &DB, authed: &ApiAuthed) -> Result<()> {
let mut tx = db.begin().await?;
audit_log(
&mut *tx,
authed,
"ai.global_request",
ActionKind::Execute,
"global",
Some(&authed.email),
None,
)
.await?;
tx.commit().await?;
Ok(())
}
fn google_ai_proxy_response_to_body(
response: GoogleAIProxyResponse,
) -> (http::StatusCode, HeaderMap, axum::body::Body) {
@@ -530,63 +524,78 @@ async fn global_proxy(
return Err(Error::BadRequest("API key is required".to_string()));
};
let base_url = provider.get_base_url(None, &db).await?;
let proxy_mode = proxy_execution_mode(&provider);
let request = if supports_query_builder_proxy(&provider) {
let credentials = ProviderCredentials {
provider: provider.clone(),
base_url,
api_key: Some(api_key.clone()),
access_token: None,
organization_id: None,
user: None,
region: None,
aws_access_key_id: None,
aws_secret_access_key: None,
aws_session_token: None,
platform: AIPlatform::Standard,
enable_1m_context: false,
custom_headers: HashMap::new(),
};
let query_builder = create_query_builder(&credentials);
let proxy_request = query_builder.build_proxy_request(&ProxyBuildArgs {
if matches!(proxy_mode, ProxyExecutionMode::NativeAwsBedrock) {
return Err(Error::BadRequest(
"AWS Bedrock global proxy is not supported; use a workspace AI resource with a region"
.to_string(),
));
}
let base_url = provider.get_base_url(None, &db).await?;
let credentials = ProviderCredentials {
provider: provider.clone(),
base_url,
api_key: Some(api_key.clone()),
access_token: None,
organization_id: None,
user: None,
region: None,
aws_access_key_id: None,
aws_secret_access_key: None,
aws_session_token: None,
platform: AIPlatform::Standard,
enable_1m_context: false,
custom_headers: HashMap::new(),
};
if matches!(proxy_mode, ProxyExecutionMode::NativeGoogleAi) {
let proxy_args = ProxyBuildArgs {
method: &method,
path: &ai_path,
headers: &headers,
body: &body,
credentials: &credentials,
})?;
proxy_request_to_request_builder(proxy_request)
} else {
let url = format!("{}/{}", base_url, ai_path);
let mut request = HTTP_CLIENT
.request(method, url)
.header("content-type", "application/json")
.header("Authorization", format!("Bearer {}", &api_key));
};
// Apply custom headers from AI_HTTP_HEADERS environment variable
for (header_name, header_value) in AI_HTTP_HEADERS.iter() {
request = request.header(header_name.as_str(), header_value.as_str());
audit_global_ai_request(&db, &authed).await?;
let response = match ai_path.as_str() {
"chat/completions" => handle_google_ai_chat_proxy(&HTTP_CLIENT, &proxy_args).await,
"models" => handle_google_ai_models_proxy(&HTTP_CLIENT, &proxy_args).await,
_ => Err(Error::BadRequest(format!(
"Unsupported Google AI path: {}",
ai_path
))),
}?;
return Ok(google_ai_proxy_response_to_body(response));
}
let request = match proxy_mode {
ProxyExecutionMode::HttpForward => {
let query_builder = create_query_builder(&credentials);
let proxy_request = query_builder.build_proxy_request(&ProxyBuildArgs {
method: &method,
path: &ai_path,
headers: &headers,
body: &body,
credentials: &credentials,
})?;
proxy_request_to_request_builder(proxy_request)
}
ProxyExecutionMode::NativeGoogleAi | ProxyExecutionMode::NativeAwsBedrock => {
return Err(Error::BadRequest(format!(
"Unsupported global proxy mode for provider {:?}",
provider
)))
}
request.body(body)
};
let response = request.send().await.map_err(to_anyhow)?;
let mut tx = db.begin().await?;
audit_log(
&mut *tx,
&authed,
"ai.global_request",
ActionKind::Execute,
"global",
Some(&authed.email),
None,
)
.await?;
tx.commit().await?;
audit_global_ai_request(&db, &authed).await?;
if response.error_for_status_ref().is_err() {
let err_msg = response.text().await.unwrap_or("".to_string());
@@ -641,7 +650,7 @@ async fn proxy(
check_scopes(&authed, || format!("resources:read:{}", resource_path))?;
}
let credentials = match workspace_cache {
let mut credentials = match workspace_cache {
Some(request_cache) if !request_cache.is_expired() && forced_resource_path.is_none() => {
request_cache.credentials
}
@@ -772,17 +781,25 @@ async fn proxy(
}
};
// Check if this is a FIM request to a provider that doesn't support native FIM endpoint
// For such providers, transform to use FIM sentinel tokens with the chat/completions endpoint
let is_fim_request = ai_path.contains("fim/completions");
if is_fim_request && !supports_native_fim(&provider) {
tracing::debug!(
"Transforming FIM request to chat/completions with FIM tokens for provider {:?}",
provider
);
let (chat_body, chat_path) = transform_fim_to_chat_completions(&body)?;
body = chat_body;
ai_path = chat_path;
if let Some(fim_transform) =
maybe_transform_fim_request(&provider, &ai_path, &credentials.base_url, &body)?
{
if fim_transform.base_url.is_some() {
tracing::debug!(
"Routing native FIM request through provider-specific endpoint for {:?}",
provider
);
} else {
tracing::debug!(
"Transforming FIM request to chat/completions with FIM tokens for provider {:?}",
provider
);
}
if let Some(base_url) = fim_transform.base_url {
credentials.base_url = base_url;
}
body = fim_transform.body;
ai_path = fim_transform.path;
}
let proxy_mode = proxy_execution_mode(&provider);
+32 -14
View File
@@ -306,6 +306,11 @@ pub struct CreateApp {
pub preserve_on_behalf_of: Option<bool>,
#[serde(default)]
pub labels: Option<Vec<String>>,
/// Caller-intent flag (set by the CLI / git sync): when true, deploying
/// this app must NOT delete an existing user draft at the same path.
/// Transient — never persisted.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub skip_draft_deletion: Option<bool>,
}
#[derive(Serialize, Deserialize)]
@@ -319,6 +324,11 @@ pub struct EditApp {
pub preserve_on_behalf_of: Option<bool>,
#[serde(default)]
pub labels: Option<Vec<String>>,
/// Caller-intent flag (set by the CLI / git sync): when true, deploying
/// this app must NOT delete an existing user draft at the same path.
/// Transient — never persisted.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub skip_draft_deletion: Option<bool>,
}
#[derive(Serialize, FromRow)]
@@ -1338,13 +1348,17 @@ async fn create_app_internal<'a>(
));
}
}
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'app'",
&app.path,
&w_id
)
.execute(&mut *tx)
.await?;
// CLI / git-sync deploys ask us to preserve any existing user draft at this
// path instead of wiping it as part of the deploy.
if !app.skip_draft_deletion.unwrap_or(false) {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'app'",
&app.path,
&w_id
)
.execute(&mut *tx)
.await?;
}
let id = sqlx::query_scalar!(
"INSERT INTO app
(workspace_id, path, summary, policy, versions, draft_only, custom_path, labels)
@@ -1943,13 +1957,17 @@ async fn update_app_internal<'a>(
)));
}
};
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'app'",
path,
&w_id
)
.execute(&mut *tx)
.await?;
// CLI / git-sync deploys ask us to preserve any existing user draft at this
// path instead of wiping it as part of the deploy.
if !ns.skip_draft_deletion.unwrap_or(false) {
sqlx::query!(
"DELETE FROM draft WHERE path = $1 AND workspace_id = $2 AND typ = 'app'",
path,
&w_id
)
.execute(&mut *tx)
.await?;
}
audit_log(
&mut *tx,
&authed,
+50 -15
View File
@@ -7107,7 +7107,11 @@ pub async fn run_job_by_hash_inner(
Ok((uuid, delete_after_use, delete_after_secs))
}
async fn get_log_file(Path((_w_id, file_p)): Path<(String, String)>) -> error::Result<Response> {
async fn get_log_file(
OptAuthed(opt_authed): OptAuthed,
Extension(db): Extension<DB>,
Path((w_id, file_p)): Path<(String, String)>,
) -> error::Result<Response> {
if file_p.contains("..") {
return Err(error::Error::BadRequest("Invalid path".to_string()));
}
@@ -7119,27 +7123,58 @@ async fn get_log_file(Path((_w_id, file_p)): Path<(String, String)>) -> error::R
"Invalid path: must have exactly 2 components".to_string(),
));
}
if Uuid::parse_str(parts[0]).is_err() {
return Err(error::Error::BadRequest(
"Invalid path: first component must be a valid UUID".to_string(),
));
}
let job_id = Uuid::parse_str(parts[0]).map_err(|_| {
error::Error::BadRequest("Invalid path: first component must be a valid UUID".to_string())
})?;
if !parts[1].ends_with(".txt") {
return Err(error::Error::BadRequest(
"Invalid path: file must end with .txt".to_string(),
));
}
// Authorization: the log file directory is the job id, so gate access the same
// way as get_job_logs — the caller must be able to read the job. Non-logged-in
// callers may only read logs of jobs created by the anonymous user.
let tags = opt_authed
.as_ref()
.map(|authed| get_scope_tags(authed).map(|v| v.iter().map(|s| s.to_string()).collect_vec()))
.flatten();
let created_by = sqlx::query_scalar!(
"SELECT created_by FROM v2_job WHERE id = $1 AND workspace_id = $2 AND ($3::text[] IS NULL OR tag = ANY($3))",
job_id,
w_id,
tags.as_ref().map(|v| v.as_slice())
)
.fetch_optional(&db)
.await?
.ok_or_else(|| error::Error::NotFound(format!("Job {job_id} not found")))?;
if opt_authed.is_none() && created_by != "anonymous" {
return Err(error::Error::BadRequest(
"As a non logged in user, you can only see jobs ran by anonymous users".to_string(),
));
}
let local_file = format!("{}/logs/{file_p}", *WINDMILL_DIR);
if tokio::fs::metadata(&local_file).await.is_ok() {
let mut file = tokio::fs::File::open(local_file).await.map_err(to_anyhow)?;
let mut buffer = Vec::new();
file.read_to_end(&mut buffer).await.map_err(to_anyhow)?;
let res = Response::builder()
.header(http::header::CONTENT_TYPE, "text/plain")
.body(Body::from(bytes::Bytes::from(buffer)))
.unwrap();
return Ok(res);
// SECURITY (defense in depth): refuse to read through a symlink so a planted
// symlink in the logs directory cannot be used to exfiltrate arbitrary files.
// `symlink_metadata` returns the link's own metadata without following it.
match tokio::fs::symlink_metadata(&local_file).await {
Ok(meta) if meta.file_type().is_symlink() => {
return Err(error::Error::BadRequest("Invalid path".to_string()));
}
Ok(_) => {
let mut file = tokio::fs::File::open(&local_file)
.await
.map_err(to_anyhow)?;
let mut buffer = Vec::new();
file.read_to_end(&mut buffer).await.map_err(to_anyhow)?;
let res = Response::builder()
.header(http::header::CONTENT_TYPE, "text/plain")
.body(Body::from(bytes::Bytes::from(buffer)))
.unwrap();
return Ok(res);
}
Err(_) => {}
}
#[cfg(all(feature = "enterprise", feature = "parquet"))]
+12 -1
View File
@@ -133,7 +133,18 @@ async fn get_log_file(
}
}
}
let file = tokio::fs::read(format!("{}{}", *TMP_WINDMILL_LOGS_SERVICE, path)).await;
let full_path = format!("{}{}", *TMP_WINDMILL_LOGS_SERVICE, path);
// SECURITY (defense in depth): refuse to read through a symlink so a planted
// symlink in the logs directory cannot be used to exfiltrate arbitrary files.
// `symlink_metadata` returns the link's own metadata without following it.
match tokio::fs::symlink_metadata(&full_path).await {
Ok(meta) if meta.file_type().is_symlink() => {
return Err(Error::BadRequest("Invalid path".to_string()));
}
Ok(_) => {}
Err(_) => return Err(Error::NotFound(format!("File {path} not found"))),
}
let file = tokio::fs::read(&full_path).await;
if let Ok(bytes) = file {
Ok(content_plain(Body::from(bytes::Bytes::from(bytes))))
} else {
+85 -9
View File
@@ -39,23 +39,43 @@ pub struct StaticFile(Uri);
impl IntoResponse for StaticFile {
fn into_response(self) -> Response<Body> {
let original_path = self.0.path();
let query = self.0.query();
let path = original_path.trim_start_matches('/');
serve_path(path, original_path)
serve_path(path, original_path, query)
}
}
#[cfg(feature = "static_frontend")]
const TWO_HUNDRED: &str = "200.html";
/// Check if the original path requires cross-origin isolation headers
/// Check if the original path requires cross-origin isolation headers.
///
/// These headers are needed for SharedArrayBuffer and TypeScript workers
/// Only enabled for /apps_raw paths (raw app editor)
/// (raw app editor at `/apps_raw/`, in-browser bundler at `/ui_builder/`).
///
/// Public apps (`/public/` and custom paths `/a/`) opt in via the `wm_coep`
/// query param: a public (raw) app must set COEP to be embeddable as an iframe
/// inside a cross-origin-isolated page (which requires the embedded document to
/// also set COEP). It is opt-in rather than always-on because cross-origin
/// isolation also blocks subresources without CORP (e.g. external image URLs
/// or embeds used by classic apps), so we only enable it when the embedder
/// explicitly requests it.
#[cfg(feature = "static_frontend")]
fn needs_cross_origin_isolation(original_path: &str) -> bool {
original_path.starts_with("/apps_raw/") || original_path.starts_with("/ui_builder/")
fn needs_cross_origin_isolation(original_path: &str, query: Option<&str>) -> bool {
original_path.starts_with("/apps_raw/")
|| original_path.starts_with("/ui_builder/")
|| ((original_path.starts_with("/public/") || original_path.starts_with("/a/"))
&& query_has_flag(query, "wm_coep"))
}
fn serve_path(path: &str, original_path: &str) -> Response<Body> {
/// Returns true if `query` contains the given flag key (with or without a
/// value), e.g. `?wm_coep`, `?wm_coep=on`, `?foo=1&wm_coep=1`.
#[cfg(feature = "static_frontend")]
fn query_has_flag(query: Option<&str>, flag: &str) -> bool {
query.is_some_and(|q| q.split('&').any(|kv| kv.split('=').next() == Some(flag)))
}
fn serve_path(path: &str, original_path: &str, query: Option<&str>) -> Response<Body> {
if path.starts_with("api/") {
return Response::builder().status(404).body(Body::empty()).unwrap();
}
@@ -71,7 +91,7 @@ fn serve_path(path: &str, original_path: &str) -> Response<Body> {
// Add cross-origin isolation headers only for paths that need them
// (apps_raw editor needs SharedArrayBuffer for TypeScript workers)
if needs_cross_origin_isolation(original_path) {
if needs_cross_origin_isolation(original_path, query) {
res = res
.header("Cross-Origin-Opener-Policy", "same-origin")
.header("Cross-Origin-Embedder-Policy", "require-corp")
@@ -102,12 +122,68 @@ fn serve_path(path: &str, original_path: &str) -> Response<Body> {
None if path.starts_with("_app/") => {
Response::builder().status(404).body(Body::empty()).unwrap()
}
None => serve_path(TWO_HUNDRED, original_path),
None => serve_path(TWO_HUNDRED, original_path, query),
}
#[cfg(not(feature = "static_frontend"))]
{
let _ = original_path; // suppress unused warning
let _ = (original_path, query); // suppress unused warning
Response::builder().status(404).body(Body::empty()).unwrap()
}
}
#[cfg(all(test, feature = "static_frontend"))]
mod tests {
use super::*;
#[test]
fn test_query_has_flag() {
assert!(query_has_flag(Some("wm_coep"), "wm_coep"));
assert!(query_has_flag(Some("wm_coep=on"), "wm_coep"));
assert!(query_has_flag(Some("foo=1&wm_coep=1"), "wm_coep"));
assert!(query_has_flag(Some("wm_coep&foo=1"), "wm_coep"));
assert!(!query_has_flag(Some("wm_coepx=1"), "wm_coep"));
assert!(!query_has_flag(Some("foo=wm_coep"), "wm_coep"));
assert!(!query_has_flag(Some(""), "wm_coep"));
assert!(!query_has_flag(None, "wm_coep"));
}
#[test]
fn test_needs_cross_origin_isolation() {
// editor + bundler are always isolated, regardless of query
assert!(needs_cross_origin_isolation("/apps_raw/edit/foo", None));
assert!(needs_cross_origin_isolation("/ui_builder/index.html", None));
// public apps (and custom paths) are isolated only when they opt in via wm_coep
assert!(needs_cross_origin_isolation(
"/public/ws/secret",
Some("wm_coep")
));
assert!(needs_cross_origin_isolation(
"/public/ws/secret",
Some("wm_coep=on")
));
assert!(needs_cross_origin_isolation(
"/a/ws/my/path",
Some("wm_coep=on")
));
assert!(!needs_cross_origin_isolation("/public/ws/secret", None));
assert!(!needs_cross_origin_isolation("/a/ws/my/path", None));
assert!(!needs_cross_origin_isolation(
"/public/ws/secret",
Some("foo=1")
));
// unrelated paths never get the headers
assert!(!needs_cross_origin_isolation(
"/apps/get/foo",
Some("wm_coep")
));
// `/api/` must not be caught by the `/a/` prefix
assert!(!needs_cross_origin_isolation(
"/api/version",
Some("wm_coep")
));
assert!(!needs_cross_origin_isolation("/", None));
}
}
+7 -7
View File
@@ -484,13 +484,13 @@ async fn update_username_in_workpsace<'c>(
).execute(&mut **tx)
.await?;
sqlx::query!(
r#"UPDATE workspace_runnable_dependencies SET app_path = REGEXP_REPLACE(app_path,'u/' || $2 || '/(.*)','u/' || $1 || '/\1') WHERE app_path LIKE ('u/' || $2 || '/%') AND workspace_id = $3"#,
new_username,
old_username,
w_id
).execute(&mut **tx)
.await?;
// NB: workspace_runnable_dependencies.app_path is intentionally NOT rewritten here.
// Its FK to app(path, workspace_id) is ON UPDATE CASCADE, so the `UPDATE app SET path`
// below propagates the new path automatically. Rewriting it manually here (before the
// app row is renamed) points the row at a not-yet-existing app path and violates
// fk_workspace_runnable_dependencies_app_path. (flow_path above DOES need the manual
// rewrite because flows are migrated via INSERT-new + DELETE-old, not UPDATE flow.path,
// so the cascade never fires for them.)
sqlx::query!(
r#"UPDATE workspace_runnable_dependencies SET runnable_path = REGEXP_REPLACE(runnable_path,'u/' || $2 || '/(.*)','u/' || $1 || '/\1') WHERE runnable_path LIKE ('u/' || $2 || '/%') AND workspace_id = $3"#,
@@ -586,6 +586,21 @@ pub struct OAuthConfig {
pub req_body_auth: Option<bool>,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
pub grant_types: Vec<String>,
/// Optional URL overrides for the provider's sandbox environment.
#[serde(skip_serializing_if = "Option::is_none")]
pub sandbox: Option<OAuthSandboxOverride>,
}
/// URL overrides for an OAuth provider's sandbox environment.
#[derive(Deserialize, Serialize, Clone, Debug, Default)]
#[cfg_attr(feature = "instance_config_schema", derive(schemars::JsonSchema))]
pub struct OAuthSandboxOverride {
#[serde(skip_serializing_if = "Option::is_none")]
pub auth_url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub token_url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub userinfo_url: Option<String>,
}
// ---------------------------------------------------------------------------
+27 -1
View File
@@ -283,6 +283,24 @@ pub async fn shutdown_signal(
Ok(())
}
// Defined for the whole non-unix scope (not just windows) so it can be a
// plain `tokio::select!` branch: that macro does not accept `#[cfg(...)]`
// attributes on individual branches. On non-windows non-unix targets the
// future never resolves, so the branch is effectively inert there.
#[cfg(not(any(target_os = "linux", target_os = "macos")))]
async fn ctrl_break() -> std::io::Result<()> {
#[cfg(windows)]
{
tokio::signal::windows::ctrl_break()?.recv().await;
Ok(())
}
#[cfg(not(windows))]
{
std::future::pending::<()>().await;
Ok(())
}
}
#[cfg(any(target_os = "linux", target_os = "macos"))]
tokio::select! {
_ = terminate() => {
@@ -298,7 +316,12 @@ pub async fn shutdown_signal(
#[cfg(not(any(target_os = "linux", target_os = "macos")))]
tokio::select! {
_ = tokio::signal::ctrl_c() => {},
_ = tokio::signal::ctrl_c() => {
tracing::info!("shutdown monitor received ctrl-c");
},
_ = ctrl_break() => {
tracing::info!("shutdown monitor received ctrl-break");
},
_ = rx.recv() => {
tracing::info!("shutdown monitor received killpill");
},
@@ -320,6 +343,9 @@ pub async fn shutdown_signal(
_ = tokio::signal::ctrl_c() => {
tracing::error!("2nd shutdown monitor received ctrl-c")
},
_ = ctrl_break() => {
tracing::error!("2nd shutdown monitor received ctrl-break")
},
}
tracing::info!("Second terminate signal received, forcefully exiting");
+1
View File
@@ -393,6 +393,7 @@ pub async fn clone_script<'c>(
modules: s.modules,
auto_parent: None,
labels: s.labels,
skip_draft_deletion: None,
};
let new_hash = hash_script(&ns);
+11
View File
@@ -252,6 +252,17 @@ lazy_static::lazy_static! {
pub static ref WORKSPACE_FAIRNESS_OVERLOADED: arc_swap::ArcSwap<Vec<String>> = arc_swap::ArcSwap::from_pointee(vec![]);
pub static ref WORKSPACE_FAIRNESS_LAST_REFRESH_MICROS: AtomicI64 = AtomicI64::new(0);
/// Stochastic admission probability for capped workspaces, expressed in
/// parts per 10_000 (so `420` = 4.2%). The refresh computes this from the
/// observed worker-second distribution and the configured cap so that
/// admission converges to the target *worker-second* share — independent
/// of how the capped vs uncapped workspaces compare on per-job durations.
/// See `workspace_fairness_ee::refresh_overloaded` for the derivation.
/// `10_000` (= admit all) is the default until the first refresh
/// classifies an overloaded set — before that, no workspace is capped so
/// `should_admit_capped` is moot and "admit all" is the correct no-op.
pub static ref WORKSPACE_FAIRNESS_ADMISSION_PPM: AtomicU32 = AtomicU32::new(10_000);
pub static ref SMTP_CONFIG: arc_swap::ArcSwap<Option<Smtp>> = arc_swap::ArcSwap::from_pointee(None);
pub static ref INDEXER_CONFIG: arc_swap::ArcSwap<TantivyIndexerSettings> = arc_swap::ArcSwap::from_pointee(TantivyIndexerSettings::default());
+1 -1
View File
@@ -157,7 +157,7 @@ pub enum ObjectType {
WorkspaceDependencies,
}
pub const LATEST_GIT_SYNC_SCRIPT_PATH: &str = "hub/28236/sync-script-to-git-repo-windmill";
pub const LATEST_GIT_SYNC_SCRIPT_PATH: &str = "hub/28261/sync-script-to-git-repo-windmill";
/// Prefix used to identify fork workspaces. A workspace whose id starts with this string is a
/// fork of another workspace.
@@ -33,3 +33,15 @@ pub async fn handle_fork_branch_creation<'c>(
) -> Result<Vec<uuid::Uuid>> {
return Ok(vec![]);
}
#[cfg(not(feature = "private"))]
pub async fn handle_deployment_metadata_batch<'c>(
_email: &str,
_created_by: &str,
_db: &DB,
_w_id: &str,
_objs: Vec<DeployedObject>,
_deployment_message: Option<String>,
) -> Result<()> {
return Ok(());
}
+6 -2
View File
@@ -13,10 +13,14 @@ pub mod git_sync_ee;
pub mod git_sync_oss;
#[cfg(feature = "private")]
pub use git_sync_ee::{handle_deployment_metadata, handle_fork_branch_creation};
pub use git_sync_ee::{
handle_deployment_metadata, handle_deployment_metadata_batch, handle_fork_branch_creation,
};
#[cfg(not(feature = "private"))]
pub use git_sync_oss::{handle_deployment_metadata, handle_fork_branch_creation};
pub use git_sync_oss::{
handle_deployment_metadata, handle_deployment_metadata_batch, handle_fork_branch_creation,
};
#[derive(Clone, Debug)]
pub enum DeployedObject {
+188 -205
View File
@@ -18,9 +18,7 @@ use std::collections::HashMap;
use std::fmt::Debug;
use anyhow::anyhow;
use base64::Engine;
use hmac::Mac;
use itertools::Itertools;
use serde::{de::DeserializeOwned, Deserialize, Serialize};
use sqlx::{Postgres, Transaction};
use tower_cookies::{Cookie, Cookies};
@@ -89,6 +87,76 @@ pub struct OAuthConfig {
pub req_body_auth: Option<bool>,
#[serde(default = "default_grant_types")]
pub grant_types: Vec<String>,
/// Optional URL overrides for the provider's sandbox environment. When
/// present and the admin has configured a `<name>_sandbox` credentials
/// entry, `build_oauth_clients` registers a second client under that key.
#[serde(skip_serializing_if = "Option::is_none")]
pub sandbox: Option<OAuthSandboxOverride>,
}
/// URL overrides for an OAuth provider's sandbox environment. Inherits
/// scopes, extra_params, etc. from the parent [`OAuthConfig`].
#[derive(Clone, Debug, Default, Serialize, Deserialize)]
pub struct OAuthSandboxOverride {
#[serde(skip_serializing_if = "Option::is_none")]
pub auth_url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub token_url: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub userinfo_url: Option<String>,
}
impl OAuthConfig {
/// Returns a copy of this config with sandbox URL overrides applied and
/// the nested `sandbox` field cleared. Returns `None` if no overrides are
/// set.
pub fn as_sandbox(&self) -> Option<OAuthConfig> {
let sb = self.sandbox.as_ref()?;
let mut out = self.clone();
out.sandbox = None;
if let Some(u) = &sb.auth_url {
out.auth_url = u.clone();
}
if let Some(u) = &sb.token_url {
out.token_url = u.clone();
}
if sb.userinfo_url.is_some() {
out.userinfo_url = sb.userinfo_url.clone();
}
Some(out)
}
}
/// Suffix appended to a provider name to identify its sandbox variant in the
/// instance credentials map and in `account.client`.
pub const SANDBOX_SUFFIX: &str = "_sandbox";
/// Strips [`SANDBOX_SUFFIX`] from a client name, returning the canonical
/// provider name. Returns the input unchanged if no suffix is present.
pub fn canonical_provider_name(client_name: &str) -> &str {
client_name
.strip_suffix(SANDBOX_SUFFIX)
.unwrap_or(client_name)
}
/// Resolves a registry [`OAuthConfig`] for `client_name`, transparently
/// applying the `sandbox` override block when the name carries the sandbox
/// suffix (e.g. `docusign_sandbox` resolves to `docusign` with sandbox URLs
/// applied). Used so callers don't need to know whether a name is a sandbox
/// variant before looking it up.
pub fn resolve_registry_config(
static_configs: &HashMap<String, OAuthConfig>,
client_name: &str,
) -> Option<OAuthConfig> {
if let Some(cfg) = static_configs.get(client_name) {
return Some(cfg.clone());
}
if client_name.ends_with(SANDBOX_SUFFIX) {
return static_configs
.get(canonical_provider_name(client_name))
.and_then(|cfg| cfg.as_sandbox());
}
None
}
/// OAuth client credentials
@@ -181,181 +249,6 @@ pub struct OAuthCallback {
pub state: String,
}
/// Build all OAuth clients from configuration
pub async fn build_oauth_clients(
base_url: &str,
oauths_from_config: Option<HashMap<String, OAuthClient>>,
connect_configs_json: &str,
login_configs_json: &str,
) -> anyhow::Result<AllClients> {
let connect_configs =
serde_json::from_str::<HashMap<String, OAuthConfig>>(connect_configs_json)?;
let login_configs = serde_json::from_str::<HashMap<String, OAuthConfig>>(login_configs_json)?;
let oauths = if let Some(oauths) = oauths_from_config {
tracing::info!("Using OAuth clients from config: {oauths:?}");
oauths
} else {
let path = "./oauth.json";
let content: String = if let Ok(e) = std::env::var("OAUTH_JSON_AS_BASE64") {
std::str::from_utf8(
&base64::engine::general_purpose::STANDARD
.decode(e)
.map_err(to_anyhow)?,
)?
.to_string()
} else if std::path::Path::new(path).exists() {
std::fs::read_to_string(path).map_err(to_anyhow)?
} else {
tracing::warn!("oauth.json not found, no OAuth clients loaded");
return Ok(AllClients {
logins: HashMap::new(),
connects: HashMap::new(),
slack: None,
});
};
if content.is_empty() {
tracing::warn!("oauth.json is empty, no OAuth clients loaded");
return Ok(AllClients {
logins: HashMap::new(),
connects: HashMap::new(),
slack: None,
});
};
match serde_json::from_str::<HashMap<String, OAuthClient>>(&content) {
Ok(clients) => clients,
Err(e) => {
tracing::error!("deserializing oauth.json: {e}");
HashMap::new()
}
}
.into_iter()
.collect()
};
tracing::info!("OAuth loaded clients: {}", oauths.keys().join(", "));
let logins = login_configs
.into_iter()
.filter_map(|x| oauths.get(&x.0).map(|c| (x.0, (c, x.1))))
.chain(oauths.iter().filter_map(|x| {
x.1.login_config
.as_ref()
.map(|c| (x.0.clone(), (x.1, c.clone())))
}))
.filter_map(|(k, (client_params, config))| {
let named_client = build_basic_client(
k.clone(),
config.clone(),
client_params.clone(),
true,
base_url,
None,
);
named_client
.map(|named_client| {
(
named_client.0,
ClientWithScopes {
client: named_client.1,
scopes: config.scopes.unwrap_or(vec![]),
extra_params: config.extra_params,
extra_params_callback: config.extra_params_callback,
allowed_domains: client_params.allowed_domains.clone(),
userinfo_url: config.userinfo_url,
display_name: client_params.display_name.clone(),
grant_types: client_params.grant_types.clone(),
},
)
})
.map_err(|e| {
tracing::error!("Error building oauth client {k}: {e}");
e
})
.ok()
})
.collect();
let connects = connect_configs
.into_iter()
.filter_map(|x| oauths.get(&x.0).map(|c| (x.0, (c, x.1))))
.chain(oauths.iter().filter_map(|x| {
x.1.connect_config
.as_ref()
.map(|c| (x.0.clone(), (x.1, c.clone())))
}))
.filter_map(|(k, (client_params, config))| {
let named_client = build_basic_client(
k.clone(),
config.clone(),
client_params.clone(),
false,
base_url,
if k == "supabase_wizard" {
Some(format!("{base_url}/oauth/callback_supabase"))
} else {
None
},
);
named_client
.map(|named_client| {
(
named_client.0,
ClientWithScopes {
client: named_client.1,
scopes: config.scopes.unwrap_or(vec![]),
extra_params: config.extra_params,
extra_params_callback: config.extra_params_callback,
allowed_domains: None,
userinfo_url: None,
display_name: client_params.display_name.clone(),
grant_types: client_params.grant_types.clone(),
},
)
})
.map_err(|e| {
tracing::error!("Error building oauth client {k}: {e}");
e
})
.ok()
})
.collect();
let slack = oauths
.get("slack")
.map(|v| {
build_basic_client(
"slack".to_string(),
OAuthConfig {
auth_url: "https://slack.com/oauth/v2/authorize".to_string(),
token_url: "https://slack.com/api/oauth.v2.access".to_string(),
userinfo_url: None,
scopes: None,
extra_params: None,
extra_params_callback: None,
req_body_auth: None,
grant_types: vec!["authorization_code".to_string()],
},
v.clone(),
false,
base_url,
Some(format!("{base_url}/oauth/callback_slack")),
)
.map(|x| x.1)
.map_err(|e| {
tracing::error!("Error building oauth slack client: {e}");
e
})
.ok()
})
.flatten();
let all_clients = AllClients { logins, connects, slack };
tracing::debug!("Final oauth config: {all_clients:#?}");
Ok(all_clients)
}
/// Build a basic OAuth client from configuration
pub fn build_basic_client(
name: String,
@@ -433,38 +326,29 @@ pub async fn build_client_credentials_oauth_client(
let oauth_client_config: OAuthClient = serde_json::from_value(oauth_config.clone())
.map_err(|e| error::Error::BadRequest(format!("Invalid OAuth config: {}", e)))?;
let mut connect_config = if let Some(ref config) = oauth_client_config.connect_config {
if !config.auth_url.is_empty() && !config.token_url.is_empty() {
config.clone()
} else {
let static_configs =
serde_json::from_str::<HashMap<String, OAuthConfig>>(connect_configs_json)
.map_err(|e| {
error::Error::InternalErr(format!(
"Failed to parse oauth_connect.json: {}",
e
))
})?;
static_configs.get(client_name).cloned().ok_or_else(|| {
error::Error::BadRequest(format!(
"OAuth configuration not found for '{}' in either global settings or static config",
client_name
))
})?
}
} else {
let static_configs =
serde_json::from_str::<HashMap<String, OAuthConfig>>(connect_configs_json).map_err(
|e| error::Error::InternalErr(format!("Failed to parse oauth_connect.json: {}", e)),
)?;
static_configs.get(client_name).cloned().ok_or_else(|| {
let parse_static_configs = || {
serde_json::from_str::<HashMap<String, OAuthConfig>>(connect_configs_json).map_err(|e| {
error::Error::InternalErr(format!("Failed to parse oauth_connect.json: {}", e))
})
};
let resolve_from_registry = |client_name: &str| -> error::Result<OAuthConfig> {
let static_configs = parse_static_configs()?;
resolve_registry_config(&static_configs, client_name).ok_or_else(|| {
error::Error::BadRequest(format!(
"OAuth configuration not found for '{}' in either global settings or static config",
client_name
))
})?
})
};
let mut connect_config = if let Some(ref config) = oauth_client_config.connect_config {
if !config.auth_url.is_empty() && !config.token_url.is_empty() {
config.clone()
} else {
resolve_from_registry(client_name)?
}
} else {
resolve_from_registry(client_name)?
};
if let Some(override_url) = cc_token_url_override {
@@ -905,4 +789,103 @@ mod tests {
let verifier = SlackVerifier::new("test_secret").unwrap();
assert!(verifier.verify("123", "body", "wrong_sig").is_err());
}
#[test]
fn canonical_provider_name_strips_sandbox_suffix() {
assert_eq!(canonical_provider_name("docusign_sandbox"), "docusign");
assert_eq!(canonical_provider_name("docusign"), "docusign");
assert_eq!(canonical_provider_name(""), "");
// Only strips the suffix once; trailing suffix on already-canonical name.
assert_eq!(
canonical_provider_name("foo_sandbox_sandbox"),
"foo_sandbox"
);
}
fn sample_oauth_config(with_sandbox: bool) -> OAuthConfig {
OAuthConfig {
auth_url: "https://account.example.com/oauth/auth".to_string(),
token_url: "https://account.example.com/oauth/token".to_string(),
userinfo_url: Some("https://account.example.com/userinfo".to_string()),
scopes: Some(vec!["signature".to_string()]),
extra_params: None,
extra_params_callback: None,
req_body_auth: None,
grant_types: default_grant_types(),
sandbox: with_sandbox.then(|| OAuthSandboxOverride {
auth_url: Some("https://account-d.example.com/oauth/auth".to_string()),
token_url: Some("https://account-d.example.com/oauth/token".to_string()),
userinfo_url: None,
}),
}
}
#[test]
fn as_sandbox_returns_none_when_no_override() {
assert!(sample_oauth_config(false).as_sandbox().is_none());
}
#[test]
fn as_sandbox_overlays_urls_and_inherits_rest() {
let resolved = sample_oauth_config(true).as_sandbox().unwrap();
// URLs overridden by sandbox block
assert_eq!(
resolved.auth_url,
"https://account-d.example.com/oauth/auth"
);
assert_eq!(
resolved.token_url,
"https://account-d.example.com/oauth/token"
);
// userinfo_url not in override → inherits from parent
assert_eq!(
resolved.userinfo_url,
Some("https://account.example.com/userinfo".to_string())
);
// Scopes/grant_types inherited from parent
assert_eq!(resolved.scopes, Some(vec!["signature".to_string()]));
assert_eq!(resolved.grant_types, default_grant_types());
// Nested sandbox field cleared on the resolved config
assert!(resolved.sandbox.is_none());
}
#[test]
fn resolve_registry_config_direct_lookup() {
let mut registry = HashMap::new();
registry.insert("docusign".to_string(), sample_oauth_config(true));
let resolved = resolve_registry_config(&registry, "docusign").unwrap();
assert_eq!(resolved.auth_url, "https://account.example.com/oauth/auth");
// Direct lookup returns the entry as-is (sandbox block still attached).
assert!(resolved.sandbox.is_some());
}
#[test]
fn resolve_registry_config_sandbox_fallback() {
let mut registry = HashMap::new();
registry.insert("docusign".to_string(), sample_oauth_config(true));
let resolved = resolve_registry_config(&registry, "docusign_sandbox").unwrap();
// Sandbox-suffixed lookup resolves to parent's sandbox-overlaid config.
assert_eq!(
resolved.auth_url,
"https://account-d.example.com/oauth/auth"
);
assert!(resolved.sandbox.is_none());
}
#[test]
fn resolve_registry_config_missing_returns_none() {
let registry: HashMap<String, OAuthConfig> = HashMap::new();
assert!(resolve_registry_config(&registry, "docusign").is_none());
assert!(resolve_registry_config(&registry, "docusign_sandbox").is_none());
}
#[test]
fn resolve_registry_config_sandbox_without_block_returns_none() {
let mut registry = HashMap::new();
// Parent exists but has no sandbox override.
registry.insert("docusign".to_string(), sample_oauth_config(false));
assert!(resolve_registry_config(&registry, "docusign_sandbox").is_none());
}
}
+1
View File
@@ -5387,6 +5387,7 @@ async fn push_inner<'c, 'd>(
cache_ttl: cache_ttl.map(|val| val as u32),
cache_ignore_s3_path: cache_ignore_s3_path,
same_worker: false,
preserve_step_tags: false,
early_return: None,
skip_expr: None,
preprocessor_module: None,
+248 -10
View File
@@ -1,15 +1,253 @@
//! Per-workspace fairness for the shared worker pool (Enterprise feature).
//! # Per-workspace fairness for the shared worker pool (Enterprise feature)
//!
//! The real algorithm — overloaded-set aggregation, coordinated refresh on
//! `background_task_state`, audit emission, stochastic admission decision —
//! lives in [`crate::workspace_fairness_ee`] and only compiles when the
//! `private` feature is on. This module is the public surface used by the
//! pull dispatch in `jobs.rs` and the integration tests; when EE is on it
//! transparently re-exports the EE implementation, when EE is off it
//! provides no-op stubs so the OSS build stays bit-identical to the
//! pre-fairness pull path.
//! On multi-tenant deployments (notably `app.windmill.dev`, and any EE
//! cluster with a single shared worker group) a single workspace flooding
//! the queue with jobs can degrade quality of service for everyone else.
//! This module computes the set of "overloaded" workspaces whose share of
//! the worker pool must be capped, and the dispatch in `jobs.rs` uses a
//! **duration-weighted stochastic admission rule** at pull time to enforce
//! the cap as a *worker-second* share, not a pull-count share.
//!
//! See [`crate::workspace_fairness_ee`] for design notes and SQL details.
//! The full algorithm lives in [`crate::workspace_fairness_ee`] behind the
//! `private` feature — this OSS-facing module is the public surface that
//! the pull dispatch and integration tests call. When EE is on, the symbols
//! here transparently re-export the EE implementation. When EE is off, they
//! are no-ops: `maybe_refresh_overloaded` does nothing, `should_admit_capped`
//! always returns `true`, and the pull path is bit-identical to its
//! pre-fairness shape. **Both runtime correctness and the entire reasoning
//! below assume the EE module is compiled in**; the OSS build is a stub.
//!
//! Every numerical default mentioned below (`MAX_PERCENT = 50`,
//! `DURATION_SECS = 10`, `MIN_TOTAL = 4`, `WORKER_PING_LIVE_SECS = 60`,
//! `ADMISSION_EPSILON_PERCENT = 5`) is tunable via global settings or
//! constants; the values here are the as-shipped defaults at the time of
//! writing and what the design discussion below was calibrated against.
//!
//! ## 1. What "overloaded" means — worker-seconds, not jobs
//!
//! A workspace is overloaded when, over a rolling
//! `WORKSPACE_FAIRNESS_DURATION_SECS = 10s` window, it has consumed at least
//! `WORKSPACE_FAIRNESS_MAX_PERCENT = 50%` of cluster worker-time. Activity
//! is measured in **worker-seconds**: each job contributes the wall-clock
//! time it actually held a worker, intersected with the window. A
//! count-based signal — "what fraction of jobs in the window are from this
//! workspace" — gets badly fooled by job-duration heterogeneity: 600
//! short (100ms) jobs and one long (60s) job consume the same worker-time
//! but the count-based form attributes 600× more weight to the spammy
//! workspace. Worker-seconds put both patterns on the same scale.
//!
//! Two sources contribute to a workspace's worker-second total:
//!
//! - **Running** (live, currently-on-a-worker): driven from `v2_job_runtime`
//! filtered on `ping > now() - WORKER_PING_LIVE_SECS` (60s, ≈ 2× worker
//! heartbeat interval), then PK-joined to `v2_job` for the `kind` filter
//! and `v2_job_queue` for `started_at` / `suspend_until`. Contribution is
//! `clamp(min(now, ping) max(started_at, window_start), 0, window)`.
//! End-of-interval is the per-job `ping`, which both (a) implements the
//! zombie defense — a worker that stopped pinging stops accruing
//! worker-seconds at its last heartbeat, so a backlog of stuck
//! `running = true` rows can't dominate the denominator — and (b) matches
//! the semantic of "worker-seconds the worker has confirmed". `v2_job_runtime`
//! is small (rows deleted on completion), so driving the scan from there
//! keeps the per-refresh cost bounded by the *in-flight* count rather
//! than by the queue size, even when one workspace has thousands of
//! `running = true` rows.
//!
//! - **Completed** (recently finished): pulled by an index scan over
//! `v2_job_completed (completed_at)`, then PK-joined to `v2_job`. The
//! index hit is critical — see "Why no `WITH params AS (...)` CTE" below.
//! Contribution is `clamp(min(completed_at, now) max(started_at,
//! completed_at duration_ms, window_start), 0, window)`. Clamping
//! start-of-interval by `completed_at duration_ms` defends against
//! zombie rows that `zombie_monitor` force-failed: `started_at` may be
//! far in the past, but `duration_ms` reflects the actual measured worker
//! time, so the row only contributes its real runtime, not the idle wait
//! before force-fail.
//!
//! Both halves exclude **flow-orchestration kinds**
//! (`flow, flowpreview, flownode, singlestepflow`) and **concurrency-
//! suspended rows** (`suspend_until IS NOT NULL`) — these hold
//! `running = true` but consume no worker slot. Same predicate as
//! `handle_zombie_jobs` in `monitor.rs`.
//!
//! `WORKSPACE_FAIRNESS_MIN_TOTAL = 4` is also in worker-seconds (≈ 40 %
//! utilization of one worker over a 10s window) — below the floor, the
//! cluster is too quiet to bother capping anyone.
//!
//! ## 2. The cap is enforced stochastically, weighted by duration
//!
//! The pull dispatch in `jobs.rs` flips a coin on every pull: with
//! probability `p_c` it uses the standard pull query (capped workspaces
//! are admissible — FIFO will pick them if they're at the head), and with
//! probability `1 p_c` it uses the *fairness pull query* which excludes
//! the overloaded workspaces. Doing it as a probabilistic split rather
//! than a binary cap/uncap gate keeps victim latency flat instead of
//! breathing in/out with each refresh cycle.
//!
//! The key design choice is how `p_c` is set. The natural first try is
//! `p_c = (MAX_PERCENT + ε) / 100` — a constant. That converges the
//! *pull-count* ratio to `MAX_PERCENT`, but only matches the worker-second
//! ratio when capped and uncapped workspaces share the same mean job
//! duration. The steady-state share equation is:
//!
//! `share = p_c · D_c / (p_c · D_c + (1 p_c) · D_u)`
//!
//! where `D_c` and `D_u` are the per-job mean durations of capped and
//! uncapped workspaces respectively. With `D_c = 34s` and `D_u = 1s` (the
//! exact numbers observed during the lancom01-prod / jps-internal cloud
//! incident), a constant `p_c = 0.65` (60 % + 5 % ε) yields
//!
//! `share = 0.65 · 34 / (0.65 · 34 + 0.35 · 1) = 22.1 / 22.45 ≈ 98%`
//!
//! — i.e., the "60 % cap" was in practice giving capped workspaces 98 %
//! of worker-seconds. Victims were observed waiting 15s+ for pickup
//! despite the cap firing on every pull.
//!
//! Inverting the equation for the desired share `t = (MAX_PERCENT + ε) / 100`:
//!
//! `p_c = t · D_u / ((1 t) · D_c + t · D_u)`
//!
//! Same numbers, target 0.65: `p_c ≈ 0.054` — about 12× tighter than the
//! count-based form. The refresh computes `p_c` and stores it in
//! [`WORKSPACE_FAIRNESS_ADMISSION_PPM`] (parts-per-10_000, fits in an
//! `AtomicU32`). The pull-time check is one atomic load plus one
//! `rand::rng().random_range(0..10_000)` draw — same hot-path cost as the
//! count-based form.
//!
//! ### `D_c`/`D_u` come from a separate, longer service-time window
//!
//! Crucially, `D_c` and `D_u` must be **true mean service times**, because
//! the share equation above is Little's-law-based
//! (`occupancy = arrival_rate × mean_service_time`). They are **not** taken
//! from the occupancy aggregation: that aggregation clamps each job's
//! contribution to the short occupancy window (`DURATION_SECS`, 10s), so a
//! job longer than the window contributes at most 10s — fine for measuring
//! *share*, but it would truncate `D_c` to ≤ 10s and systematically
//! under-admit the skew exactly when capped jobs are long (the case the cap
//! exists for: e.g. true `D_c = 34s` clamped to 10s gives `p_c ≈ 0.157`, an
//! 86 % effective share instead of 65 %). Instead, the refresh samples true
//! unclamped `duration_ms` of completed jobs over a longer, decoupled
//! service-time window (`DURATION_SAMPLE_SECS`, 60s) — long enough to avoid
//! truncation and to keep the mean stable when few jobs complete within the
//! 10s occupancy window. So the refresh emits two per-workspace signals:
//! windowed occupancy worker-seconds (for classification) and a 60s
//! service-time `(Σ duration_ms, count)` (for admission), merged per
//! workspace.
//!
//! ### Why we kept the fallback when the fairness pull returns empty
//!
//! The 100 `p_c` % of pulls that try the fairness query (excluding
//! capped workspaces) fall back to the standard query if the fairness
//! query returns no row. The alternative — idle the worker, holding the
//! slot open in case a victim shows up — was considered but rejected for
//! the first iteration: with `p_c` correctly tightened, victims do get the
//! slot they need *when they exist*, and absent victims, falling back to
//! the capped pool is the right behaviour (otherwise the cluster
//! under-utilises itself for no benefit). Adding a reserve-capacity skip
//! is a fine-tuning lever for bursty victim arrival patterns and is left
//! as a follow-up.
//!
//! ### Degenerate cases
//!
//! If either bucket is empty — no capped jobs, no uncapped jobs, or a
//! capped workspace with zero completions in the 60s service-time window
//! (all its jobs still running) — the formula is undefined. The refresh
//! falls back to the count-based `p_c = t` in those cases — it matches
//! the pre-refactor behaviour and is the safest thing to do when there's
//! no service-time signal yet to weight on.
//!
//! ## 3. Coordinated refresh — exactly once per cycle, cluster-wide
//!
//! The aggregation is too expensive to run on every worker process every
//! pull (and would produce no new information on the sub-second
//! timescale). It runs **at most once every `refresh_interval` seconds
//! across the entire fleet**, gated by both a per-process CAS and a
//! DB-side row lock:
//!
//! 1. **Per-process gate** — `maybe_refresh_overloaded` (called from the
//! pull path) does `LAST_REFRESH_MICROS.compare_exchange` to ensure at
//! most one in-flight refresh per process per interval. If the CAS
//! fails or the interval hasn't elapsed yet, the call is a no-op.
//! Cost on the hot path: one atomic load, optionally one CAS.
//!
//! 2. **DB-side claim** — `refresh_overloaded` first does a cheap upsert
//! (`INSERT ... ON CONFLICT ON background_task_state ... WHERE
//! updated_at < NOW() refresh_interval RETURNING true`). The `VALUES`
//! clause is all constants, so Postgres has no expensive work to do
//! even for losers. Only the unique winner per cycle gets `Some(true)`;
//! losers get `None` and skip the aggregation entirely.
//!
//! 3. **Winner-only aggregation** — the winner runs the
//! `v2_job_runtime v2_job_completed` worker-second aggregation
//! returning per-workspace `(workspace_id, worker_seconds, jobs)`,
//! classifies into overloaded/uncapped, computes `p_c`, and writes the
//! new payload `{"overloaded": [...], "admission_ppm": N}` back to
//! `background_task_state.workspace_fairness`.
//!
//! 4. **Everyone reads** — winner and losers alike then `SELECT` the
//! current value, parse it, and update their in-process
//! `WORKSPACE_FAIRNESS_OVERLOADED` and `WORKSPACE_FAIRNESS_ADMISSION_PPM`
//! atomics. This is what makes losers eventually see the winner's
//! decision; they just don't pay the aggregation cost.
//!
//! The refresh interval is `ACTIVE_REFRESH_SECS = 2s` when the cluster
//! currently has a capped workspace (faster — we want the cap to lift
//! promptly once load drops) and `IDLE_REFRESH_SECS = 5s` otherwise
//! (slower — minimise DB load during normal operation). The DB-side guard
//! always uses the tighter `ACTIVE_REFRESH_SECS` to bound the race
//! window; the per-process gate enforces the idle cadence.
//!
//! If a refresh fails (DB error, timeout > 5s), `LAST_REFRESH_MICROS` is
//! left set to the attempt's timestamp so the next attempt has to wait a
//! full interval — exactly the same cooldown as a successful refresh.
//! Resetting to `0` on failure would remove the rate limit entirely
//! precisely when DB load is highest, which is the wrong direction.
//!
//! ## 4. Audit logging
//!
//! Workspaces entering or leaving the capped set produce
//! `workspace_fairness.capped` / `workspace_fairness.uncapped` audit
//! entries scoped to the `admins` workspace, with the affected workspace
//! as the `resource` field. Emitted by the refresh winner only, so a
//! transition produces exactly one audit row regardless of fleet size.
//! The "previous list" diffed against is the DB value (not the per-process
//! cache) so a freshly-restarted worker that happens to win the first
//! claim doesn't emit spurious "newly capped" entries for workspaces that
//! were already capped before it started.
//!
//! ## 5. Notable SQL performance constraints
//!
//! - **No `WITH params AS (...)` CTE for `window_start`.** A natural
//! refactor would be to compute `NOW() - make_interval(secs => N)` once
//! in a CTE and reference it in both halves of the UNION. But Postgres
//! *materialises* the CTE and the optimiser can no longer push the
//! `completed_at > window_start` predicate down to the
//! `ix_job_completed_completed_at` index. On the production cloud DB
//! (~12M `v2_job_completed` rows), that turns a 10 ms index scan into a
//! ~47s full table scan. The query intentionally inlines `NOW()` and
//! `NOW() - make_interval(...)` at every callsite.
//!
//! - **Drive running side from `v2_job_runtime`, not `v2_job_queue`.**
//! Naive ordering ("scan v2_job_queue for `running = true`, join v2_job
//! for the kind filter") does a Seq Scan over ~thousands of running-or-
//! bookkeeping rows and does a PK lookup into `v2_job` for every one of
//! them — ~10 ms in prod, but worse: bounded by *queue size*. Pivoting
//! to drive the scan from `v2_job_runtime` filtered on
//! `ping > NOW() - 60s` narrows to the in-flight set (small, deletes-
//! on-completion) *before* any PK lookups: 1.3 ms, 9× less I/O,
//! bounded by *live worker count*.
//!
//! ## 6. Enterprise gating
//!
//! The cap is an Enterprise feature. `windmill-api-settings` rejects
//! `workspace_fairness_enabled = true` writes from non-EE builds, and on a
//! single-tenant self-hosted deployment the default
//! `workspace_fairness_enabled = false` keeps the pull path identical to
//! the pre-fairness baseline. At runtime the dispatch checks the atomic
//! only — when fairness is off, `maybe_refresh_overloaded` drains the
//! cached state in one pull cycle (resetting `WORKSPACE_FAIRNESS_OVERLOADED`
//! to empty and `WORKSPACE_FAIRNESS_ADMISSION_PPM` to 10_000 = "admit all"),
//! so toggling the feature off without restarting workers is safe.
#[cfg(feature = "private")]
#[allow(unused)]
+12
View File
@@ -110,6 +110,11 @@ pub struct NewFlow {
pub ws_error_handler_muted: Option<bool>,
#[serde(default)]
pub labels: Option<Vec<String>>,
/// Caller-intent flag (set by the CLI / git sync): when true, deploying
/// this flow must NOT delete an existing user draft at the same path.
/// Transient — never persisted.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub skip_draft_deletion: Option<bool>,
}
impl NewFlow {
@@ -169,6 +174,13 @@ pub struct FlowValue {
#[serde(default)]
#[serde(skip_serializing_if = "is_default")]
pub same_worker: bool,
// When the flow runs on a custom worker tag, by default that tag is propagated to
// (and overrides) every step, script and nested sub-flow. Set this to true to instead
// let steps that declare their own non-empty tag run on it; steps without their own tag
// still inherit the flow's tag. Defaults to false to preserve the historical behavior.
#[serde(default)]
#[serde(skip_serializing_if = "is_default")]
pub preserve_step_tags: bool,
#[serde(flatten)]
pub concurrency_settings: ConcurrencySettings,
#[serde(flatten)]
+10
View File
@@ -540,9 +540,19 @@ pub struct NewScript {
pub auto_parent: Option<bool>,
#[serde(default)]
pub labels: Option<Vec<String>>,
/// Caller-intent flag (set by the CLI / git sync): when true, deploying
/// this script must NOT delete an existing user draft at the same path.
/// Transient — never persisted. Deliberately excluded from `impl Hash`
/// below (it must not affect the version hash) and from the no-op
/// comparison in the deploy handler (it isn't part of what the script
/// *is*). See `is_noop_deploy_against_parent`.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub skip_draft_deletion: Option<bool>,
}
// IMPORTANT: update this Hash impl when adding fields to NewScript
// (exception: caller-intent flags like `skip_draft_deletion` are intentionally
// omitted — they must not influence the computed version hash)
impl Hash for NewScript {
fn hash<H: Hasher>(&self, state: &mut H) {
self.path.hash(state);
@@ -31,6 +31,7 @@ $PY_PATH
$INDEX_URL_ARG $EXTRA_INDEX_URL_ARG $TRUSTED_HOST_ARG
--system
--reinstall
--compile-bytecode
"
echo $CMD
+1 -1
View File
@@ -1,6 +1,6 @@
// AI executor module structure
// This module will contain all AI-related execution logic
pub mod query_builder;
pub mod stream_event_processor;
pub mod tools;
pub mod utils;
+1 -1
View File
@@ -1,4 +1,4 @@
use crate::ai::query_builder::StreamEventProcessor;
use crate::ai::stream_event_processor::StreamEventProcessor;
use crate::ai::utils::{
add_message_to_conversation, execute_mcp_tool, get_step_name_from_flow,
is_completed_input_transform, update_flow_status_module_with_actions,
+1 -1
View File
@@ -45,7 +45,7 @@ use windmill_common::{
use windmill_queue::{cancel_single_job, CanceledBy, MiniPulledJob};
use crate::{
ai::query_builder::StreamEventProcessor,
ai::stream_event_processor::StreamEventProcessor,
common::{build_args_map, resolve_job_timeout, OccupancyMetrics, StreamNotifier},
handle_child::{run_future_with_polling_update_job_poller_graceful, GracefulPollOutcome},
};
@@ -2095,6 +2095,9 @@ async fn spawn_uv_install(
"--no-cache",
// If we invoke uv pip install, then we want to overwrite existing data
"--reinstall",
// Compile .py to .pyc at install time so imports are fast even
// through read-only nsjail mounts (no in-memory compilation per job).
"--compile-bytecode",
];
if let Some(py_path) = py_path.as_ref() {
+101 -8
View File
@@ -2985,6 +2985,87 @@ struct PushNextFlowJobRec {
// #[async_recursion]
// #[instrument(level = "trace", skip_all)]
/// Resolve the worker tag for a flow's child job (step, nested sub-flow, preprocessor).
///
/// A child normally inherits the parent flow job's tag so the whole flow runs on one worker
/// group. The exceptions, in order:
/// - the preprocessor step, or a flow running on the generic `flow` / `flow-{workspace}` tag,
/// always uses the child's own tag (`step_tag`);
/// - when the flow opts into `preserve_step_tags` and the child declares its own non-empty tag,
/// that tag is honored instead of being overridden by the flow tag;
/// - otherwise the child inherits the parent flow job's tag.
fn resolve_flow_step_tag(
is_preprocessor_step: bool,
flow_tag: &str,
workspace_id: &str,
preserve_step_tags: bool,
step_tag: Option<&str>,
) -> Option<String> {
if is_preprocessor_step || flow_tag == "flow" || flow_tag == format!("flow-{}", workspace_id) {
step_tag.map(str::to_string)
} else if preserve_step_tags && step_tag.is_some_and(|t| !t.is_empty()) {
step_tag.map(str::to_string)
} else {
Some(flow_tag.to_string())
}
}
#[cfg(test)]
mod tag_resolution_tests {
use super::resolve_flow_step_tag;
#[test]
fn step_inherits_custom_flow_tag_by_default() {
// Parent flow on a custom tag, step declares its own tag, preserve disabled:
// the step inherits the flow tag (historical behavior).
assert_eq!(
resolve_flow_step_tag(false, "worker-group-A", "w1", false, Some("worker-group-B")),
Some("worker-group-A".to_string())
);
}
#[test]
fn step_keeps_own_tag_when_preserve_enabled() {
// The exact customer scenario: a sub-flow tagged worker-group-B run as a step of a
// flow tagged worker-group-A now runs on worker-group-B when preserve_step_tags is on.
assert_eq!(
resolve_flow_step_tag(false, "worker-group-A", "w1", true, Some("worker-group-B")),
Some("worker-group-B".to_string())
);
}
#[test]
fn untagged_step_inherits_flow_tag_even_when_preserve_enabled() {
assert_eq!(
resolve_flow_step_tag(false, "worker-group-A", "w1", true, None),
Some("worker-group-A".to_string())
);
// An empty tag counts as "no tag" and still inherits.
assert_eq!(
resolve_flow_step_tag(false, "worker-group-A", "w1", true, Some("")),
Some("worker-group-A".to_string())
);
}
#[test]
fn generic_flow_tag_always_uses_step_tag() {
for flow_tag in ["flow", "flow-w1"] {
assert_eq!(
resolve_flow_step_tag(false, flow_tag, "w1", false, Some("worker-group-B")),
Some("worker-group-B".to_string())
);
}
}
#[test]
fn preprocessor_step_uses_step_tag() {
assert_eq!(
resolve_flow_step_tag(true, "worker-group-A", "w1", false, Some("worker-group-B")),
Some("worker-group-B".to_string())
);
}
}
async fn push_next_flow_job(
flow_job: Arc<MiniPulledJob>,
mut status: FlowStatus,
@@ -4118,13 +4199,13 @@ async fn push_next_flow_job(
.map(|x| x.into());
tracing::debug!(id = %flow_job.id, root_id = %job_root, "computed perms for job {i} of {len}");
let tag = if step.is_preprocessor_step()
|| (flow_job.tag == "flow" || flow_job.tag == format!("flow-{}", flow_job.workspace_id))
{
payload_tag.tag.clone()
} else {
Some(flow_job.tag.clone())
};
let tag = resolve_flow_step_tag(
step.is_preprocessor_step(),
&flow_job.tag,
&flow_job.workspace_id,
flow.preserve_step_tags,
payload_tag.tag.as_deref(),
);
let (email, permissioned_as) = if let Some(on_behalf_of) = payload_tag.on_behalf_of.as_ref()
{
@@ -4822,6 +4903,7 @@ fn payload_from_modules<'a>(
modules_node: Option<FlowNodeId>,
failure_module: Option<&Box<FlowModule>>,
same_worker: bool,
preserve_step_tags: bool,
id: impl FnOnce() -> String,
path: impl FnOnce() -> String,
opt_empty_inner_flows: bool,
@@ -4842,7 +4924,13 @@ fn payload_from_modules<'a>(
}
Some(JobPayload::RawFlow {
value: FlowValue { modules, failure_module, same_worker, ..Default::default() },
value: FlowValue {
modules,
failure_module,
same_worker,
preserve_step_tags,
..Default::default()
},
path: Some(path()),
restarted_from: None,
})
@@ -5144,6 +5232,7 @@ async fn compute_next_flow_transform(
modules_node,
flow.failure_module.as_ref(),
flow.same_worker,
flow.preserve_step_tags,
|| format!("{}-{i}", status.step),
|| format!("{}/forloop-{i}", flow_job.runnable_path()),
true,
@@ -5280,6 +5369,7 @@ async fn compute_next_flow_transform(
modules_node,
flow.failure_module.as_ref(),
flow.same_worker,
flow.preserve_step_tags,
|| status.step.to_string(),
|| format!("{}/branchone-{}", flow_job.runnable_path(), branch_idx),
true,
@@ -5321,6 +5411,7 @@ async fn compute_next_flow_transform(
modules_node,
flow.failure_module.as_ref(),
flow.same_worker,
flow.preserve_step_tags,
|| format!("{}-{i}", status.step),
|| format!("{}/branchall-{}", flow_job.runnable_path(), i),
false,
@@ -5391,6 +5482,7 @@ async fn compute_next_flow_transform(
modules_node,
flow.failure_module.as_ref(),
flow.same_worker,
flow.preserve_step_tags,
|| format!("{}-{}", status.step, branch_status.branch),
|| {
format!(
@@ -5473,6 +5565,7 @@ async fn next_loop_iteration(
modules_node,
flow.failure_module.as_ref(),
flow.same_worker,
flow.preserve_step_tags,
|| format!("{}-{}", status.step, ns.index),
inner_path,
true,
+1 -1
View File
@@ -2,7 +2,7 @@ import { sleep } from "https://deno.land/x/sleep@v1.2.1/mod.ts";
import * as windmill from "https://deno.land/x/windmill@v1.174.0/mod.ts";
import * as api from "https://deno.land/x/windmill@v1.174.0/windmill-api/index.ts";
export const VERSION = "v1.711.0";
export const VERSION = "v1.714.0";
export async function login(email: string, password: string): Promise<string> {
return await windmill.UserService.login({
+6
View File
@@ -192,6 +192,8 @@ export async function pushApp(
deployment_message: message,
...localAppBody,
...preserveFields,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
},
});
}
@@ -205,6 +207,8 @@ export async function pushApp(
deployment_message: message,
...localAppBody,
...preserveFields,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
},
});
}
@@ -480,6 +484,8 @@ const command = new Command()
on_behalf_of_email: email,
} as any,
preserve_on_behalf_of: true,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
},
});
log.info(colors.green(`Updated permissioned_as for app ${appPath} to ${email}`));
+4
View File
@@ -466,6 +466,8 @@ export async function pushRawApp(
summary: localApp.summary,
policy: appForPolicy.policy,
deployment_message: message,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
...(localApp.custom_path
? { custom_path: localApp.custom_path }
: {}),
@@ -486,6 +488,8 @@ export async function pushRawApp(
summary: localApp.summary,
policy: appForPolicy.policy,
deployment_message: message,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
...(localApp.custom_path
? { custom_path: localApp.custom_path }
: {}),
+6
View File
@@ -224,6 +224,8 @@ export async function pushFlow(
deployment_message: message,
...localFlowBody,
...preserveFields,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
},
});
}
@@ -237,6 +239,8 @@ export async function pushFlow(
deployment_message: message,
...localFlowBody,
...preserveFields,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
},
});
} catch (e) {
@@ -1159,6 +1163,8 @@ const command = new Command()
path: flowPath,
on_behalf_of_email: email,
preserve_on_behalf_of: true,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
} as any,
});
log.info(colors.green(`Updated permissioned_as for flow ${flowPath} to ${email}`));
+5
View File
@@ -758,6 +758,9 @@ async function createScript(
workspace: Workspace
): Promise<number> {
const start = performance.now();
// Preserve any user draft at this path: a CLI / git-sync deploy must not wipe
// an in-progress draft the way a UI "deploy from draft" intentionally does.
body = { ...body, skip_draft_deletion: true };
// skip_if_noop asks the backend to treat deploys identical to the parent
// (same content, lockfile, and metadata) as a no-op, so the CLI does not
// produce phantom git-sync / promotion commits on re-pushes.
@@ -1796,6 +1799,8 @@ async function setPermissionedAs(
parent_hash: remote.hash,
on_behalf_of_email: email,
preserve_on_behalf_of: true,
// Preserve any user draft at this path (see backend skip_draft_deletion).
skip_draft_deletion: true,
},
});
log.info(colors.green(`Updated permissioned_as for script ${scriptPath} to ${email}`));
+42 -10
View File
@@ -22,6 +22,7 @@ import {
showConflict,
showDiff,
extractNativeTriggerInfo,
redactEncryptionKey,
} from "../../types.ts";
import { downloadZip } from "./pull.ts";
import { runLint, printReport, checkMissingLocks } from "../lint/lint.ts";
@@ -2503,8 +2504,20 @@ export async function pull(
}
if (opts.onlyCreateBranch) {
// Branch is checked out locally; the caller pushes it. Symmetric with
// the non-onlyCreateBranch path: CLI does branch + pull, never push.
// Branch-only publish: there is no commit here, so the GPG-cache-warmth
// invariant that motivated moving commit+push to the hub script (WIN-1974,
// #9284) does not apply — a bare `git push` of the (empty) branch ref needs
// no signing. The hub script only runs its in-process commit+push for the
// non-onlyCreateBranch path (`if (!only_create_branch) git_push(...)`), so
// the CLI MUST publish the fork branch here or it is never pushed at all.
gitSyncDeployPush({
items: deployItems,
authorName: process.env["WM_USERNAME"] || "windmill",
authorEmail: process.env["WM_EMAIL"] || "windmill@windmill.dev",
committerName: opts.gitCommitterName,
committerEmail: opts.gitCommitterEmail,
onlyCreateBranch: true,
});
return;
}
}
@@ -3040,12 +3053,13 @@ export async function gitDeploy(
...(opts.extraIncludes ?? []),
...includes.extraIncludes,
],
includeSchedules: opts.includeSchedules || includes.includeSchedules,
includeGroups: opts.includeGroups || includes.includeGroups,
includeUsers: opts.includeUsers || includes.includeUsers,
includeTriggers: opts.includeTriggers || includes.includeTriggers,
includeSettings: opts.includeSettings || includes.includeSettings,
includeKey: opts.includeKey || includes.includeKey,
// Workspace-wide mode force-includes the deployed default-excluded kinds
// (full mirror). Individual-branch/promotion mode forces nothing — these
// keys stay ABSENT so pull resolves them from the promotion target's
// effective wmill.yaml filters. Spreading (not setting `false`) is what
// makes the deferral work: an explicit `false` would clobber the effective
// config in pull's Object.assign-based option merge.
...includes.forcedIncludes,
promotion,
} as any);
}
@@ -3095,16 +3109,22 @@ function prettyChanges(
),
);
} else if (change.name === "edited") {
const changeType = getTypeStrFromPath(change.path);
log.info(
colors.yellow(
`~ ${getTypeStrFromPath(change.path)} ` +
`~ ${changeType} ` +
displayPath +
colors.gray(wsNote) +
(change.codebase ? ` (codebase changed)` : ""),
),
);
if (change.before != change.after) {
if (change.path.endsWith(".yaml")) {
if (changeType === "encryption_key") {
showDiff(
redactEncryptionKey(change.before),
redactEncryptionKey(change.after),
);
} else if (change.path.endsWith(".yaml")) {
try {
showDiff(
yamlStringify(
@@ -4022,6 +4042,10 @@ export async function push(
originalWorkspaceSpecificPath,
permissionedAsContext,
isWsSpecific ? true : undefined,
{
noninteractive: (opts.yes ?? false) || !process.stdin.isTTY,
skipReencrypt: opts.skipReencryptOnKeyChange,
},
);
if (stateTarget) {
@@ -4107,6 +4131,10 @@ export async function push(
localFilePath, // Pass the actual local file path
permissionedAsContext,
isAddedWsSpecific ? true : undefined,
{
noninteractive: (opts.yes ?? false) || !process.stdin.isTTY,
skipReencrypt: opts.skipReencryptOnKeyChange,
},
);
if (stateTarget) {
@@ -4663,6 +4691,10 @@ const command = new Command()
.option("--include-groups", "Include syncing groups")
.option("--include-settings", "Include syncing workspace settings")
.option("--include-key", "Include workspace encryption key")
.option(
"--skip-reencrypt-on-key-change",
"When the pushed encryption key differs from the remote, do NOT re-encrypt existing remote secrets. Only safe if they are already encrypted with the new key (e.g. workspace/instance migration). Default is to re-encrypt.",
)
.option("--skip-branch-validation", "Skip git branch validation and prompts")
.option("--json-output", "Output results in JSON format")
.option(
+1
View File
@@ -88,6 +88,7 @@ export interface SyncOptions {
includeGroups?: boolean;
includeSettings?: boolean;
includeKey?: boolean;
skipReencryptOnKeyChange?: boolean;
skipBranchValidation?: boolean;
message?: string;
includes?: string[];
+48 -7
View File
@@ -445,11 +445,23 @@ export async function pushWorkspaceSettings(
}
}
export interface PushWorkspaceKeyOptions {
// True when no prompt may be shown (e.g. `--yes` was passed or stdin is not a
// TTY). In that case the re-encryption decision is taken from `skipReencrypt`
// / the WMILL_NO_REENCRYPT_ON_KEY_CHANGE env var instead of an interactive
// confirmation.
noninteractive?: boolean;
// Explicit re-encryption decision from `--skip-reencrypt-on-key-change`.
// When set it takes precedence over the prompt and the env var.
skipReencrypt?: boolean;
}
export async function pushWorkspaceKey(
workspace: string,
_path: string,
key: string | undefined,
localKey: string
localKey: string,
opts?: PushWorkspaceKeyOptions
) {
try {
key = await wmill
@@ -461,17 +473,46 @@ export async function pushWorkspaceKey(
throw new Error(`Failed to get workspace encryption key: ${err}`);
}
if (localKey && key !== localKey) {
const confirm = await Confirm.prompt({
message:
"The local workspace encryption key does not match the remote. Do you want to reencrypt all your secrets on the remote with the new key?\nSay 'no' if your local secrets are already encrypted with the new key (e.g. workspace/instance migration)\nOtherwise, say 'yes' and pull the secrets after the reencryption.\n",
default: true,
});
// Changing the key on the remote means the existing ciphertexts (encrypted
// with the old key) become unreadable unless they are re-encrypted. By
// default we ask the backend to re-encrypt every secret variable with the
// new key, which preserves their plaintext values. The only reason to skip
// re-encryption is when the stored ciphertexts are *already* encrypted with
// the new key (e.g. a workspace/instance migration).
let reencrypt: boolean;
// Explicit choice via `--skip-reencrypt-on-key-change` or the env var wins
// over everything, regardless of interactivity.
const explicitSkip =
opts?.skipReencrypt ||
(process.env.WMILL_NO_REENCRYPT_ON_KEY_CHANGE ?? "").toLowerCase() ===
"true";
if (explicitSkip) {
reencrypt = false;
log.info(
"Workspace encryption key changed; leaving remote ciphertexts untouched (skip re-encryption requested)."
);
} else if (opts?.noninteractive) {
// No TTY (or --yes) and no explicit skip: we can't prompt, so default to
// re-encrypting (matches the interactive default) to preserve secret
// values. Pass --skip-reencrypt-on-key-change (or set
// WMILL_NO_REENCRYPT_ON_KEY_CHANGE=true) to opt out.
reencrypt = true;
log.info(
"Workspace encryption key changed; re-encrypting all remote secrets with the new key (non-interactive)."
);
} else {
reencrypt = await Confirm.prompt({
message:
"The local workspace encryption key does not match the remote. Do you want to reencrypt all your secrets on the remote with the new key?\nSay 'no' if your local secrets are already encrypted with the new key (e.g. workspace/instance migration)\nOtherwise, say 'yes' and pull the secrets after the reencryption.\n",
default: true,
});
}
log.debug(`Updating workspace encryption key...`);
await wmill.setWorkspaceEncryptionKey({
workspace,
requestBody: {
new_key: localKey,
skip_reencrypt: !confirm,
skip_reencrypt: !reencrypt,
},
});
} else {
File diff suppressed because one or more lines are too long
+12 -6
View File
@@ -207,13 +207,19 @@ async function reconcileIncludingFile(options: {
}
function referencesIncludeLine(content: string, includeLine: string): boolean {
// Match only when the include sits on a line by itself (allowing leading
// and trailing whitespace). Earlier we split on `\s+`, but that
// false-positives on commented-out includes like `<!-- @AGENTS.cli.md -->`
// where the middle token equals the include. CRLF is handled by the
// `\r?\n` split.
// Match when the include appears as a whitespace-separated token on any
// line that isn't an HTML comment. We can't require the include to be on a
// line by itself: our own CLAUDE.md default is `Instructions are in
// @AGENTS.md` (one sentence), and a strict equality check made `wmill
// refresh prompts` re-prompt every run on files wmill itself wrote.
// Skipping comment-bearing lines keeps `<!-- @AGENTS.cli.md -->` from
// false-positiving.
for (const line of content.split(/\r?\n/)) {
if (line.trim() === includeLine) {
const trimmed = line.trim();
if (trimmed.startsWith("<!--") || trimmed.endsWith("-->")) {
continue;
}
if (trimmed.split(/\s+/).includes(includeLine)) {
return true;
}
}
+1 -1
View File
@@ -89,7 +89,7 @@ export {
token,
};
export const VERSION = "1.711.0";
export const VERSION = "1.714.0";
// Re-exported from constants.ts to maintain backwards compatibility
export { WM_FORK_PREFIX } from "./core/constants.ts";
+44 -3
View File
@@ -18,7 +18,11 @@ import { pushSchedule } from "./commands/schedule/schedule.ts";
import { pushWorkspaceUser } from "./commands/user/user.ts";
import { pushGroup } from "./commands/user/user.ts";
import { pushWorkspaceDependencies } from "./commands/dependencies/dependencies.ts";
import { pushWorkspaceSettings, pushWorkspaceKey } from "./core/settings.ts";
import {
pushWorkspaceSettings,
pushWorkspaceKey,
PushWorkspaceKeyOptions,
} from "./core/settings.ts";
import { pushTrigger, pushNativeTrigger } from "./commands/trigger/trigger.ts";
import { pushRawApp } from "./commands/app/raw_apps.ts";
import type { PermissionedAsContext } from "./core/permissioned_as.ts";
@@ -129,11 +133,46 @@ export function showDiff(local: string, remote: string) {
export function showConflict(path: string, local: string, remote: string) {
log.info(colors.yellow(`- ${path}`));
showDiff(local, remote);
let isEncryptionKey = false;
try {
isEncryptionKey = getTypeStrFromPath(path) === "encryption_key";
} catch {
// ignore
}
if (isEncryptionKey) {
showDiff(redactEncryptionKey(local), redactEncryptionKey(remote));
} else {
showDiff(local, remote);
}
log.info("\x1b[31mlocal\x1b[31m - \x1b[32mremote\x1b[32m");
log.info("\n");
}
// Reveal only the first 5 chars of the key so a rotation is still visible in
// the diff (different prefixes), without leaking the whole secret to stdout.
// The remaining chars are replaced with `*`, preserving length so the diff
// keeps showing whether the key length changed.
export function redactEncryptionKey(content: string): string {
if (!content) return content;
// The encryption_key payload is JSON-encoded (a quoted string). Parse it so
// we redact the key value itself, then re-serialize to JSON to preserve the
// file's shape; fall back to raw redaction if parsing fails.
try {
const parsed = JSON.parse(content);
if (typeof parsed === "string") {
return JSON.stringify(redactString(parsed));
}
} catch {
// not JSON — treat content as the raw key
}
return redactString(content);
}
function redactString(s: string): string {
if (s.length <= 5) return s;
return s.slice(0, 5) + "*".repeat(s.length - 5);
}
/**
* Pushes an object to the workspace server based on its type
* @param workspace - The workspace ID to push to
@@ -144,6 +183,7 @@ export function showConflict(path: string, local: string, remote: string) {
* @param alreadySynced - Array to track already synced items
* @param message - Optional commit/update message
* @param originalLocalPath - The original local file path (used for branch-specific resource file resolution)
* @param keyPushOpts - Options for the encryption_key push: non-interactive flag and explicit re-encryption choice
*/
export async function pushObj(
workspace: string,
@@ -156,6 +196,7 @@ export async function pushObj(
originalLocalPath?: string,
permissionedAsContext?: PermissionedAsContext,
wsSpecific?: boolean,
keyPushOpts?: PushWorkspaceKeyOptions,
) {
const typeEnding = getTypeStrFromPath(p);
@@ -221,7 +262,7 @@ export async function pushObj(
} else if (typeEnding === "settings") {
await pushWorkspaceSettings(workspace, p, befObj, newObj);
} else if (typeEnding === "encryption_key") {
await pushWorkspaceKey(workspace, p, befObj, newObj);
await pushWorkspaceKey(workspace, p, befObj, newObj, keyPushOpts);
} else {
throw new Error(
`The item ${p} has an unrecognized type ending ${typeEnding}`
+40 -16
View File
@@ -252,21 +252,44 @@ export function gitSyncIncludePattern(
}
}
export interface GitSyncDeployIncludes {
extraIncludes: string[];
// `forcedIncludes` carries ONLY the include-* flags that must be force-set to
// true (overriding the repo's wmill.yaml). Kinds not present are intentionally
// omitted (never set to false) so the caller can spread this object and let
// the repo's effective config govern the rest — see deriveGitSyncDeployIncludes.
export type GitSyncForcedIncludes = Partial<{
includeSchedules: boolean;
includeGroups: boolean;
includeUsers: boolean;
includeTriggers: boolean;
includeSettings: boolean;
includeKey: boolean;
}>;
export interface GitSyncDeployIncludes {
extraIncludes: string[];
forcedIncludes: GitSyncForcedIncludes;
}
// Mirrors the hub script's wmill_sync_pull include-derivation: build the
// --extra-includes set from the deployed items, and (only in workspace-wide
// mode — never with --use-individual-branch) opt object kinds that are
// excluded by default back in. Replaces the script's regexFromPath +
// --extra-includes set from the deployed items, and decide which default-
// excluded object kinds (triggers, schedules, groups, users, settings, key)
// must be force-included in the pull. Replaces the script's regexFromPath +
// per-kind --include-* construction so the hub script can drop both.
//
// Branch-mode distinction (this is load-bearing — see the trigger-promotion
// bug it fixes):
// - Workspace-wide mode: the repo is a full mirror of the workspace, so a
// deployed object of a default-excluded kind MUST be re-included, even if
// wmill.yaml would otherwise skip it. We force the flag on.
// - Individual-branch (promotion) mode: the repo is a filtered prod surface
// whose own wmill.yaml filters decide what gets promoted. We force NOTHING
// here and the keys stay absent, so the caller's pull resolves them from
// the target's effective config (a deployed trigger lands iff the target
// includes triggers). Forcing `false` (the original behavior) did NOT
// defer — it CLOBBERED the effective config via Object.assign in pull's
// option merge, silently dropping kinds the target actually wanted (e.g. a
// deployed trigger when the target has includeTriggers: true), and the
// server then omitted the object from the tarball entirely.
export function deriveGitSyncDeployIncludes(
items: GitSyncDeployItem[],
useIndividualBranch: boolean,
@@ -283,18 +306,19 @@ export function deriveGitSyncDeployIncludes(
}
}
const has = (pred: (t: string) => boolean) =>
!useIndividualBranch && items.some((i) => pred(i.path_type));
const forcedIncludes: GitSyncForcedIncludes = {};
if (!useIndividualBranch) {
const has = (pred: (t: string) => boolean) =>
items.some((i) => pred(i.path_type));
if (has((t) => t === "schedule")) forcedIncludes.includeSchedules = true;
if (has((t) => t === "group")) forcedIncludes.includeGroups = true;
if (has((t) => t === "user")) forcedIncludes.includeUsers = true;
if (has((t) => t.includes("trigger"))) forcedIncludes.includeTriggers = true;
if (has((t) => t === "settings")) forcedIncludes.includeSettings = true;
if (has((t) => t === "key")) forcedIncludes.includeKey = true;
}
return {
extraIncludes,
includeSchedules: has((t) => t === "schedule"),
includeGroups: has((t) => t === "group"),
includeUsers: has((t) => t === "user"),
includeTriggers: has((t) => t.includes("trigger")),
includeSettings: has((t) => t === "settings"),
includeKey: has((t) => t === "key"),
};
return { extraIncludes, forcedIncludes };
}
function git(
+47 -11
View File
@@ -254,7 +254,7 @@ describe("deriveGitSyncDeployIncludes", () => {
]);
});
test("workspace-wide mode opts excluded kinds back in", () => {
test("workspace-wide mode force-includes deployed default-excluded kinds", () => {
const r = deriveGitSyncDeployIncludes(
[
{ path_type: "schedule", path: "f/s" },
@@ -266,15 +266,36 @@ describe("deriveGitSyncDeployIncludes", () => {
],
false
);
expect(r.includeSchedules).toBe(true);
expect(r.includeGroups).toBe(true);
expect(r.includeTriggers).toBe(true);
expect(r.includeSettings).toBe(true);
expect(r.includeKey).toBe(true);
expect(r.includeUsers).toBe(true);
// Full-mirror repo: a deployed object of a default-excluded kind must be
// re-included even if wmill.yaml would skip it, so the flag is forced on.
expect(r.forcedIncludes).toEqual({
includeSchedules: true,
includeGroups: true,
includeTriggers: true,
includeSettings: true,
includeKey: true,
includeUsers: true,
});
});
test("individual-branch mode NEVER sets include flags (matches hub script)", () => {
test("workspace-wide mode only forces the kinds actually deployed", () => {
const r = deriveGitSyncDeployIncludes(
[{ path_type: "script", path: "f/s" }],
false
);
// Scripts are included by default — nothing to force.
expect(r.forcedIncludes).toEqual({});
});
test("individual-branch (promotion) mode forces NOTHING — defers to wmill.yaml", () => {
// Regression: these flags used to be force-disabled (set to false) in
// promotion mode, which CLOBBERED the promotion target's effective
// wmill.yaml config (an explicit false wins in pull's Object.assign merge).
// The server then stripped the object from the tarball, the pull wrote
// nothing, and `git add '<path>**'` failed with "pathspec did not match
// any files". Forcing nothing leaves the keys absent so the target's
// effective filters govern; extraIncludes still scopes the pull to the
// changed object.
const r = deriveGitSyncDeployIncludes(
[
{ path_type: "schedule", path: "f/s" },
@@ -282,12 +303,27 @@ describe("deriveGitSyncDeployIncludes", () => {
],
true
);
expect(r.includeSchedules).toBe(false);
expect(r.includeTriggers).toBe(false);
// extra-includes are still derived regardless of branch mode
expect(r.forcedIncludes).toEqual({});
expect(r.extraIncludes).toContain("f/s.schedule.*");
expect(r.extraIncludes).toContain("f/t.kafka_trigger.*");
});
test("regression: http_trigger promotion deploy does not clobber the target's includeTriggers", () => {
// Brad's scenario: an HTTP trigger is deployed and the promotion repo uses
// individual branches. path_type is "httptrigger" (the no-underscore value
// the backend puts on item.path_type — see git_sync_ee.rs
// insert_path_type_and_return_message). includeTriggers must NOT be forced
// false here, so the target's effective includeTriggers (true in Brad's
// config) is honored and the trigger file is pulled and committed.
const r = deriveGitSyncDeployIncludes(
[{ path_type: "httptrigger", path: "f/platform/on_call_chat_http_route" }],
true
);
expect(r.forcedIncludes.includeTriggers).toBeUndefined();
expect(r.extraIncludes).toContain(
"f/platform/on_call_chat_http_route.http_trigger.*"
);
});
});
// =============================================================================
+446
View File
@@ -22,6 +22,18 @@ import { withTestBackend } from "./test_backend.ts";
import { shouldSkipOnCI } from "./cargo_backend.ts";
import { addWorkspace } from "../workspace.ts";
// The HTTP-trigger promotion test creates an http_trigger, whose API routes are
// behind the `http_trigger` cargo feature — NOT in the default EE test feature
// set. The shared test backend reads TEST_FEATURES at construction (first
// `withTestBackend` call), so appending here at module load enables it. Guarded
// on shouldSkipOnCI() so we only widen the build when these EE tests actually
// run (i.e. EE_LICENSE_KEY present); minimal CI builds stay untouched.
if (!shouldSkipOnCI()) {
process.env["TEST_FEATURES"] = [process.env["TEST_FEATURES"], "http_trigger"]
.filter(Boolean)
.join(",");
}
function git(cwd: string, ...args: string[]): string {
return execFileSync("git", args, { cwd, encoding: "utf8" }).trim();
}
@@ -44,6 +56,24 @@ function remoteHead(bareDir: string, branch: string): string {
).trim();
}
// True if `filePath` exists in the tree of `branch` on the bare remote.
function fileExistsOnBranch(
bareDir: string,
branch: string,
filePath: string,
): boolean {
try {
execFileSync(
"git",
["--git-dir", bareDir, "cat-file", "-e", `refs/heads/${branch}:${filePath}`],
{ stdio: "ignore" },
);
return true;
} catch {
return false;
}
}
test.skipIf(shouldSkipOnCI())(
"git-sync promotion: use_individual_branch pushes to wm_deploy branch, not main",
async () => {
@@ -200,3 +230,419 @@ test.skipIf(shouldSkipOnCI())(
});
},
);
/**
* Regression test for the promotion trigger-include bug (fix/gitsync-promotion-
* trigger-export): deploying a trigger (or any excluded-by-default kind:
* schedule, group, user, settings, key) with `use_individual_branch` must still
* land the object file on the `wm_deploy` branch.
*
* Root cause: `deriveGitSyncDeployIncludes` used to force the per-kind include
* flags (`includeTriggers` etc.) to false in individual-branch mode. The
* server-side tarball export STRIPS those object kinds entirely when their
* include flag is false (`if include_triggers { … }` in workspaces_export.rs),
* and `extraIncludes` is only a client-side filter over what the tarball
* already contains it can't recover a file the server never sent. So the
* pull wrote no trigger file, the wm_deploy branch was created empty of the
* trigger, and production's `git add '<path>**'` failed with "pathspec did not
* match any files". A script (always-included kind) never hit this hence the
* dedicated trigger case here.
*
* Without the fix this test fails: the branch exists but the
* `*.http_trigger.yaml` file is absent from it.
*/
test.skipIf(shouldSkipOnCI())(
"git-sync promotion: use_individual_branch lands a trigger file on the wm_deploy branch",
async () => {
await withTestBackend(async (backend) => {
const ws = backend.workspace; // "test"
await addWorkspace(
{
remote: backend.baseUrl,
workspaceId: ws,
name: ws,
token: backend.token,
} as any,
{ force: true, configDir: backend.testConfigDir },
);
// --- 1. Bare "remote" seeded with an initial `main` commit ---
const bareDir = await mkdtemp(join(tmpdir(), "wmill_promo_trig_bare_"));
execFileSync("git", ["init", "--bare", "--initial-branch=main", bareDir]);
const seedDir = await mkdtemp(join(tmpdir(), "wmill_promo_trig_seed_"));
git(seedDir, "init", "--initial-branch=main");
git(seedDir, "config", "user.email", "seed@windmill.dev");
git(seedDir, "config", "user.name", "seed");
await writeFile(join(seedDir, "README.md"), "# promo trigger test\n");
git(seedDir, "add", "-A");
git(seedDir, "commit", "-m", "seed");
git(seedDir, "remote", "add", "origin", `file://${bareDir}`);
git(seedDir, "push", "-u", "origin", "main");
const seedMain = remoteHead(bareDir, "main");
// --- 2. Workspace content: a script + an HTTP trigger pointing at it ---
await backend.apiRequest!(`/api/w/${ws}/folders/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ name: "promo", owners: [], extra_perms: {} }),
});
await backend.apiRequest!(`/api/w/${ws}/scripts/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
path: "f/promo/foo",
summary: "",
description: "",
content: "export async function main() { return 1 }",
language: "bun",
}),
});
const trigRes = await backend.apiRequest!(`/api/w/${ws}/http_triggers/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
path: "f/promo/hook",
script_path: "f/promo/foo",
route_path: "promo_hook",
is_flow: false,
http_method: "post",
authentication_method: "none",
is_static_website: false,
request_type: "sync",
}),
});
// Guard against the route silently 404ing (the http_trigger cargo feature
// not being built) — otherwise the pull below would find nothing to sync
// and the real assertion would fail with a confusing message.
expect(trigRes.status).toBe(201);
// --- 3. git_repository resource + git-sync config (triggers included) ---
await backend.apiRequest!(`/api/w/${ws}/resources/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
path: "u/test/promo_repo",
resource_type: "git_repository",
value: { url: `file://${bareDir}`, branch: "main", token: "" },
}),
});
await backend.updateGitSyncConfig!({
git_sync_settings: {
repositories: [
{
git_repo_resource_path: "u/test/promo_repo",
script_path: "f/**",
use_individual_branch: true,
group_by_folder: false,
settings: {
include_path: ["f/**"],
include_type: ["script", "trigger"],
},
},
],
},
});
// The backend sets path_type "httptrigger" (no underscore) on the deploy
// item — see DeployedObject::HttpTrigger => "httptrigger" in git_sync_ee.rs.
const deployItems = JSON.stringify([
{
path_type: "httptrigger",
path: "f/promo/hook",
commit_msg: "deploy hook",
},
]);
const work = await mkdtemp(join(tmpdir(), "wmill_promo_trig_work_"));
git(work, "clone", `file://${bareDir}`, ".");
// Option B semantic: in promotion mode the deploy forces NOTHING — the
// trigger lands only because THIS target's effective wmill.yaml opts
// triggers in. (Reverting the source fix re-introduces the force-`false`
// that clobbers this `includeTriggers: true`, so the file is dropped.)
await writeFile(
join(work, "wmill.yaml"),
"defaultTs: bun\nincludes:\n - f/**\nexcludes: []\nincludeTriggers: true\n",
);
const res = await backend.runCLICommand(
[
"sync",
"git-deploy",
"--repository",
"u/test/promo_repo",
"--use-individual-branch",
"--git-deploy-items",
deployItems,
],
work,
);
expect(res.code).toBe(0);
// Caller-half (mirrors the hub script): stage what the pull wrote, commit
// on the checked-out wm_deploy branch, push.
git(work, "config", "user.email", "test@windmill.dev");
git(work, "config", "user.name", "test");
git(work, "add", "-A");
try {
git(work, "diff", "--cached", "--quiet");
} catch {
git(work, "commit", "-m", "deploy hook");
}
git(work, "push", "--porcelain", "-u", "origin", "HEAD");
const expectedBranch = `refs/heads/wm_deploy/${ws}/httptrigger/f__promo__hook`;
expect(remoteBranches(bareDir)).toContain(expectedBranch);
// The regression: the trigger file MUST be present on the branch. Without
// the fix the include flag is false, the server strips the trigger from
// the tarball, the pull writes nothing, and this file is absent.
expect(
fileExistsOnBranch(
bareDir,
`wm_deploy/${ws}/httptrigger/f__promo__hook`,
"f/promo/hook.http_trigger.yaml",
),
).toBe(true);
// Base branch untouched (individual-branch never pushes to the base).
expect(remoteHead(bareDir, "main")).toBe(seedMain);
await rm(bareDir, { recursive: true, force: true });
await rm(seedDir, { recursive: true, force: true });
await rm(work, { recursive: true, force: true });
});
},
);
/**
* Same regression as the HTTP-trigger case above, for a `schedule` a
* different excluded-by-default kind that exercises a DISTINCT path: its own
* include flag (`includeSchedules`), its own server-side `if include_schedules`
* tarball-strip branch, and its own `.schedule.yaml` extension. Unlike triggers
* it needs no extra cargo feature, so it guards the fix even where the
* trigger-specific features aren't built.
*
* Without the fix this test fails: the branch exists but the
* `*.schedule.yaml` file is absent from it.
*/
test.skipIf(shouldSkipOnCI())(
"git-sync promotion: use_individual_branch lands a schedule file on the wm_deploy branch",
async () => {
await withTestBackend(async (backend) => {
const ws = backend.workspace; // "test"
await addWorkspace(
{
remote: backend.baseUrl,
workspaceId: ws,
name: ws,
token: backend.token,
} as any,
{ force: true, configDir: backend.testConfigDir },
);
// --- 1. Bare "remote" seeded with an initial `main` commit ---
const bareDir = await mkdtemp(join(tmpdir(), "wmill_promo_sched_bare_"));
execFileSync("git", ["init", "--bare", "--initial-branch=main", bareDir]);
const seedDir = await mkdtemp(join(tmpdir(), "wmill_promo_sched_seed_"));
git(seedDir, "init", "--initial-branch=main");
git(seedDir, "config", "user.email", "seed@windmill.dev");
git(seedDir, "config", "user.name", "seed");
await writeFile(join(seedDir, "README.md"), "# promo schedule test\n");
git(seedDir, "add", "-A");
git(seedDir, "commit", "-m", "seed");
git(seedDir, "remote", "add", "origin", `file://${bareDir}`);
git(seedDir, "push", "-u", "origin", "main");
const seedMain = remoteHead(bareDir, "main");
// --- 2. Workspace content: a script + a (disabled) schedule for it ---
await backend.apiRequest!(`/api/w/${ws}/folders/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ name: "promo", owners: [], extra_perms: {} }),
});
await backend.apiRequest!(`/api/w/${ws}/scripts/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
path: "f/promo/foo",
summary: "",
description: "",
content: "export async function main() { return 1 }",
language: "bun",
}),
});
const schedRes = await backend.apiRequest!(`/api/w/${ws}/schedules/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
path: "f/promo/sched",
schedule: "0 0 12 * * *",
timezone: "UTC",
script_path: "f/promo/foo",
is_flow: false,
args: {},
enabled: false,
}),
});
expect(schedRes.status).toBe(200);
// --- 3. git_repository resource + git-sync config (schedules included) ---
await backend.apiRequest!(`/api/w/${ws}/resources/create`, {
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
path: "u/test/promo_repo",
resource_type: "git_repository",
value: { url: `file://${bareDir}`, branch: "main", token: "" },
}),
});
await backend.updateGitSyncConfig!({
git_sync_settings: {
repositories: [
{
git_repo_resource_path: "u/test/promo_repo",
script_path: "f/**",
use_individual_branch: true,
group_by_folder: false,
settings: {
include_path: ["f/**"],
include_type: ["script", "schedule"],
},
},
],
},
});
const deployItems = JSON.stringify([
{
path_type: "schedule",
path: "f/promo/sched",
commit_msg: "deploy sched",
},
]);
const work = await mkdtemp(join(tmpdir(), "wmill_promo_sched_work_"));
git(work, "clone", `file://${bareDir}`, ".");
// Option B semantic: in promotion mode the deploy forces NOTHING — the
// schedule lands only because THIS target's effective wmill.yaml opts
// schedules in. (Reverting the source fix re-introduces the force-`false`
// that clobbers this `includeSchedules: true`, so the file is dropped.)
await writeFile(
join(work, "wmill.yaml"),
"defaultTs: bun\nincludes:\n - f/**\nexcludes: []\nincludeSchedules: true\n",
);
const res = await backend.runCLICommand(
[
"sync",
"git-deploy",
"--repository",
"u/test/promo_repo",
"--use-individual-branch",
"--git-deploy-items",
deployItems,
],
work,
);
expect(res.code).toBe(0);
// Caller-half (mirrors the hub script): stage what the pull wrote, commit
// on the checked-out wm_deploy branch, push.
git(work, "config", "user.email", "test@windmill.dev");
git(work, "config", "user.name", "test");
git(work, "add", "-A");
try {
git(work, "diff", "--cached", "--quiet");
} catch {
git(work, "commit", "-m", "deploy sched");
}
git(work, "push", "--porcelain", "-u", "origin", "HEAD");
const expectedBranch = `refs/heads/wm_deploy/${ws}/schedule/f__promo__sched`;
expect(remoteBranches(bareDir)).toContain(expectedBranch);
// The regression: the schedule file MUST be present on the branch. Without
// the fix the include flag is false, the server strips the schedule from
// the tarball, the pull writes nothing, and this file is absent.
expect(
fileExistsOnBranch(
bareDir,
`wm_deploy/${ws}/schedule/f__promo__sched`,
"f/promo/sched.schedule.yaml",
),
).toBe(true);
// Base branch untouched (individual-branch never pushes to the base).
expect(remoteHead(bareDir, "main")).toBe(seedMain);
await rm(bareDir, { recursive: true, force: true });
await rm(seedDir, { recursive: true, force: true });
await rm(work, { recursive: true, force: true });
});
},
);
/**
* Regression test for WIN-1997: forking a workspace with git sync configured
* must publish a `wm-fork/<branch>/<id>` branch to the remote.
*
* The fork-branch callback runs the sync script with `only_create_branch:
* true` and no items. The hub script delegates branch checkout + push of that
* empty ref to `wmill sync git-deploy --only-create-branch` its own
* in-process commit+push runs ONLY for the `!only_create_branch` path. So if
* the CLI doesn't push the freshly checked-out branch here, nothing does and
* the fork branch never reaches the remote (the symptom that broke the e2e
* test after #9284 moved commit+push to the caller). This guards that the CLI
* owns the push for the branch-only case.
*/
test.skipIf(shouldSkipOnCI())(
"git-sync fork: only_create_branch publishes the wm-fork branch (CLI owns the push)",
async () => {
await withTestBackend(async (backend) => {
// Bare "remote" seeded with an initial `main` commit.
const bareDir = await mkdtemp(join(tmpdir(), "wmill_fork_bare_"));
execFileSync("git", ["init", "--bare", "--initial-branch=main", bareDir]);
const seedDir = await mkdtemp(join(tmpdir(), "wmill_fork_seed_"));
git(seedDir, "init", "--initial-branch=main");
git(seedDir, "config", "user.email", "seed@windmill.dev");
git(seedDir, "config", "user.name", "seed");
await writeFile(join(seedDir, "README.md"), "# fork test\n");
git(seedDir, "add", "-A");
git(seedDir, "commit", "-m", "seed");
git(seedDir, "remote", "add", "origin", `file://${bareDir}`);
git(seedDir, "push", "-u", "origin", "main");
const seedMain = remoteHead(bareDir, "main");
// The CWD the hub script runs git-deploy in: a clone of the repo on main.
const work = await mkdtemp(join(tmpdir(), "wmill_fork_work_"));
git(work, "clone", `file://${bareDir}`, ".");
await writeFile(
join(work, "wmill.yaml"),
"defaultTs: bun\nincludes:\n - f/**\nexcludes: []\n",
);
// Branch creation happens BEFORE the fork workspace exists (step 1 of the
// fork flow), so we pass the fork workspace id straight through — whoami
// returns synthetic superadmin info for it. No items, only_create_branch.
const forkWs = "wm-fork-clitest";
const res = await backend.runCLICommand(
[
"sync",
"git-deploy",
"--repository",
"u/test/unused_on_branch_only_path",
"--git-deploy-items",
"[]",
"--only-create-branch",
],
work,
{ workspace: forkWs },
);
expect(res.code).toBe(0);
// The regression: with NO caller-side commit/push, the fork branch must
// already be on the remote because the CLI pushed it.
expect(remoteBranches(bareDir)).toContain("refs/heads/wm-fork/main/clitest");
// Base branch untouched — branch-only publish creates no commit.
expect(remoteHead(bareDir, "main")).toBe(seedMain);
await rm(bareDir, { recursive: true, force: true });
await rm(seedDir, { recursive: true, force: true });
await rm(work, { recursive: true, force: true });
});
},
);
+7 -1
View File
@@ -391,6 +391,13 @@ describe("writeAiGuidanceFiles — referencesAgentsCli (via reconciliation)", ()
["between blank lines", "before\n\n@AGENTS.cli.md\n\nafter"],
["leading whitespace then include", " @AGENTS.cli.md\n"],
["CRLF line endings", "line one\r\n@AGENTS.cli.md\r\nline three"],
// Mid-sentence include: this is how our own CLAUDE.md default looks
// ("Instructions are in @AGENTS.md"). A strict line-equality check made
// `wmill refresh prompts` re-prompt every run on files wmill wrote.
["mid-sentence include", "Instructions are in @AGENTS.cli.md\n"],
// `>` blockquote prefix doesn't disable Claude's `@`-import expansion,
// so we treat it as a reference too.
["blockquoted include", "> @AGENTS.cli.md"],
])("treats %s as a reference (no append)", async (_label, content) => {
await withTempDir(async (tempDir) => {
await writeFile(join(tempDir, "AGENTS.md"), content, "utf8");
@@ -406,7 +413,6 @@ describe("writeAiGuidanceFiles — referencesAgentsCli (via reconciliation)", ()
["@AGENTS-cli-md (lookalike)", "@AGENTS-cli-md"],
["@AGENTS.cli.md without surrounding whitespace", "foo@AGENTS.cli.md"],
["commented-out include", "<!-- @AGENTS.cli.md -->"],
["blockquoted include", "> @AGENTS.cli.md"],
])("does not treat %s as a reference (append happens)", async (_label, content) => {
await withTempDir(async (tempDir) => {
await writeFile(join(tempDir, "AGENTS.md"), content, "utf8");
+93
View File
@@ -0,0 +1,93 @@
/**
* Unit tests for pushWorkspaceKey in settings.ts.
*
* Covers WIN-2005: changing the encryption key in encryption_key.yaml and
* pushing it must (by default) re-encrypt the remote secrets with the new key.
*
* Verifies that:
* - an unchanged key is a no-op (no setWorkspaceEncryptionKey call)
* - a changed key in non-interactive mode re-encrypts by default
* (skip_reencrypt = false), so secret plaintext values are preserved
* - the --skip-reencrypt-on-key-change flag keeps the remote ciphertexts
* untouched (skip_reencrypt = true)
* - WMILL_NO_REENCRYPT_ON_KEY_CHANGE=true does the same via env var
*/
import { expect, test, describe, beforeEach, afterEach, mock } from "bun:test";
// Track calls to mocked wmill functions
let remoteKey = "";
let setEncryptionKeyCalls: {
workspace: string;
requestBody: { new_key: string; skip_reencrypt?: boolean };
}[] = [];
// Mock the wmill module before importing settings.ts
mock.module("../gen/services.gen.ts", () => ({
getWorkspaceEncryptionKey: async (_args: { workspace: string }) => ({
key: remoteKey,
}),
setWorkspaceEncryptionKey: async (args: {
workspace: string;
requestBody: { new_key: string; skip_reencrypt?: boolean };
}) => {
setEncryptionKeyCalls.push(args);
},
}));
import { pushWorkspaceKey } from "../src/core/settings.ts";
describe("pushWorkspaceKey", () => {
const ws = "test-workspace";
beforeEach(() => {
remoteKey = "";
setEncryptionKeyCalls = [];
delete process.env.WMILL_NO_REENCRYPT_ON_KEY_CHANGE;
});
afterEach(() => {
delete process.env.WMILL_NO_REENCRYPT_ON_KEY_CHANGE;
});
test("no-op when local key matches the remote key", async () => {
remoteKey = "samekey";
await pushWorkspaceKey(ws, "encryption_key", undefined, "samekey", {
noninteractive: true,
});
expect(setEncryptionKeyCalls.length).toBe(0);
});
test("changed key re-encrypts by default in non-interactive mode", async () => {
remoteKey = "oldkey";
await pushWorkspaceKey(ws, "encryption_key", undefined, "newkey", {
noninteractive: true,
});
expect(setEncryptionKeyCalls.length).toBe(1);
expect(setEncryptionKeyCalls[0].requestBody.new_key).toBe("newkey");
// skip_reencrypt false => backend re-encrypts existing secrets with new key
expect(setEncryptionKeyCalls[0].requestBody.skip_reencrypt).toBe(false);
});
test("--skip-reencrypt-on-key-change skips re-encryption", async () => {
remoteKey = "oldkey";
await pushWorkspaceKey(ws, "encryption_key", undefined, "newkey", {
noninteractive: true,
skipReencrypt: true,
});
expect(setEncryptionKeyCalls.length).toBe(1);
expect(setEncryptionKeyCalls[0].requestBody.new_key).toBe("newkey");
expect(setEncryptionKeyCalls[0].requestBody.skip_reencrypt).toBe(true);
});
test("WMILL_NO_REENCRYPT_ON_KEY_CHANGE=true skips re-encryption non-interactively", async () => {
remoteKey = "oldkey";
process.env.WMILL_NO_REENCRYPT_ON_KEY_CHANGE = "true";
await pushWorkspaceKey(ws, "encryption_key", undefined, "newkey", {
noninteractive: true,
});
expect(setEncryptionKeyCalls.length).toBe(1);
expect(setEncryptionKeyCalls[0].requestBody.new_key).toBe("newkey");
expect(setEncryptionKeyCalls[0].requestBody.skip_reencrypt).toBe(true);
});
});

Some files were not shown because too many files have changed in this diff Show More