From 17448c97d33ddd2e2c6b5aef54699a12226e150f Mon Sep 17 00:00:00 2001 From: hugocasa Date: Tue, 29 Sep 2026 12:26:11 +0200 Subject: [PATCH] count trigger suspend, resume and discard (#11405) * feat(telemetry): count trigger suspend, resume and discard Co-Authored-By: Claude Opus 5.5 * refactor: shorten the telemetry disclosure to one line per category Co-Authored-By: Claude Opus 5.5 * docs: correct the resume branch comments and note the pre-commit fire count Co-Authored-By: Claude Opus 5.5 * fix: name feature adoption in the telemetry disclosure Co-Authored-By: Claude Opus 5.5 * chore: update ee-repo-ref to 9855e1b7a43a0a33e04f8accf1c497af3fd9b139 This commit updates the EE repository reference after PR #834 was merged in windmill-ee-private. Previous ee-repo-ref: 1d5b128ec956c156fe549cf099ba0dbc1b6467bf New ee-repo-ref: 9855e1b7a43a0a33e04f8accf1c497af3fd9b139 Automated by sync-ee-ref workflow. --------- Co-authored-by: Claude Opus 5.5 Co-authored-by: windmill-internal-app[bot] --- .agents/skills/svelte-frontend/SKILL.md | 4 +- ...3500a39efa20e4915ffa56fd40537597db36e.json | 26 ----- ...a5c6acbf5468c190fa72b6b6707d950267752.json | 32 ++++++ backend/ee-repo-ref.txt | 2 +- backend/windmill-queue/src/jobs.rs | 12 ++- .../windmill-trigger/src/global_handler.rs | 22 +++- docs/feature-telemetry.md | 11 +- .../lib/components/InstanceSettings.svelte | 101 +++--------------- 8 files changed, 88 insertions(+), 122 deletions(-) delete mode 100644 backend/.sqlx/query-1d346a14ad5586af347b8e7ac413500a39efa20e4915ffa56fd40537597db36e.json create mode 100644 backend/.sqlx/query-872c06b3098cf59797bcaa90d6da5c6acbf5468c190fa72b6b6707d950267752.json diff --git a/.agents/skills/svelte-frontend/SKILL.md b/.agents/skills/svelte-frontend/SKILL.md index 6aceedc25c..ddeb8e59a1 100644 --- a/.agents/skills/svelte-frontend/SKILL.md +++ b/.agents/skills/svelte-frontend/SKILL.md @@ -119,8 +119,8 @@ Form components (TextInput, Toggle, Select, etc.) should use the unified size sy New user-facing UX is the main source of `feature_usage` counters — propose them in the plan, not as a separate question, and read `docs/feature-telemetry.md` first. `logFeatureUsage()` from `$lib/utils/featureUsage` is only half the change: the `(feature, kind)` pair must also be -registered in the backend allowlist or every event is silently discarded, and the disclosure copy -in `InstanceSettings.svelte` must name what you added. +registered in the backend allowlist or every event is silently discarded. The disclosure copy in +`InstanceSettings.svelte` stays at the category level; see the recipe's step 3. ## Svelte MCP Server diff --git a/backend/.sqlx/query-1d346a14ad5586af347b8e7ac413500a39efa20e4915ffa56fd40537597db36e.json b/backend/.sqlx/query-1d346a14ad5586af347b8e7ac413500a39efa20e4915ffa56fd40537597db36e.json deleted file mode 100644 index e8ffcfb95b..0000000000 --- a/backend/.sqlx/query-1d346a14ad5586af347b8e7ac413500a39efa20e4915ffa56fd40537597db36e.json +++ /dev/null @@ -1,26 +0,0 @@ -{ - "db_name": "PostgreSQL", - "query": "\n SELECT 'schedule' AS \"kind!\", COUNT(*)::BIGINT AS \"count!\" FROM schedule\n UNION ALL SELECT 'http', COUNT(*)::BIGINT FROM http_trigger\n UNION ALL SELECT 'websocket', COUNT(*)::BIGINT FROM websocket_trigger\n UNION ALL SELECT 'kafka', COUNT(*)::BIGINT FROM kafka_trigger\n UNION ALL SELECT 'nats', COUNT(*)::BIGINT FROM nats_trigger\n UNION ALL SELECT 'postgres', COUNT(*)::BIGINT FROM postgres_trigger\n UNION ALL SELECT 'mqtt', COUNT(*)::BIGINT FROM mqtt_trigger\n UNION ALL SELECT 'sqs', COUNT(*)::BIGINT FROM sqs_trigger\n UNION ALL SELECT 'gcp', COUNT(*)::BIGINT FROM gcp_trigger\n UNION ALL SELECT 'azure', COUNT(*)::BIGINT FROM azure_trigger\n UNION ALL SELECT 'amqp', COUNT(*)::BIGINT FROM amqp_trigger\n UNION ALL SELECT 'email', COUNT(*)::BIGINT FROM email_trigger\n -- Grouped, not a single 'native' key: these fire as nextcloud/google/github,\n -- so a lone key would not line up with the `trigger`/`fired` series.\n UNION ALL SELECT service_name::text, COUNT(*)::BIGINT FROM native_trigger GROUP BY service_name\n ", - "describe": { - "columns": [ - { - "ordinal": 0, - "name": "kind!", - "type_info": "Text" - }, - { - "ordinal": 1, - "name": "count!", - "type_info": "Int8" - } - ], - "parameters": { - "Left": [] - }, - "nullable": [ - null, - null - ] - }, - "hash": "1d346a14ad5586af347b8e7ac413500a39efa20e4915ffa56fd40537597db36e" -} diff --git a/backend/.sqlx/query-872c06b3098cf59797bcaa90d6da5c6acbf5468c190fa72b6b6707d950267752.json b/backend/.sqlx/query-872c06b3098cf59797bcaa90d6da5c6acbf5468c190fa72b6b6707d950267752.json new file mode 100644 index 0000000000..3bdcb5e48c --- /dev/null +++ b/backend/.sqlx/query-872c06b3098cf59797bcaa90d6da5c6acbf5468c190fa72b6b6707d950267752.json @@ -0,0 +1,32 @@ +{ + "db_name": "PostgreSQL", + "query": "\n SELECT 'schedule' AS \"kind!\", COUNT(*)::BIGINT AS \"count!\", 0::BIGINT AS \"suspended!\" FROM schedule\n UNION ALL SELECT 'http', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM http_trigger\n UNION ALL SELECT 'websocket', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM websocket_trigger\n UNION ALL SELECT 'kafka', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM kafka_trigger\n UNION ALL SELECT 'nats', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM nats_trigger\n UNION ALL SELECT 'postgres', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM postgres_trigger\n UNION ALL SELECT 'mqtt', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM mqtt_trigger\n UNION ALL SELECT 'sqs', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM sqs_trigger\n UNION ALL SELECT 'gcp', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM gcp_trigger\n UNION ALL SELECT 'azure', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM azure_trigger\n UNION ALL SELECT 'amqp', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM amqp_trigger\n UNION ALL SELECT 'email', COUNT(*)::BIGINT, COUNT(*) FILTER (WHERE mode = 'suspended')::BIGINT FROM email_trigger\n -- Grouped, not a single 'native' key: these fire as nextcloud/google/github,\n -- so a lone key would not line up with the `trigger`/`fired` series.\n UNION ALL SELECT service_name::text, COUNT(*)::BIGINT, 0::BIGINT FROM native_trigger GROUP BY service_name\n ", + "describe": { + "columns": [ + { + "ordinal": 0, + "name": "kind!", + "type_info": "Text" + }, + { + "ordinal": 1, + "name": "count!", + "type_info": "Int8" + }, + { + "ordinal": 2, + "name": "suspended!", + "type_info": "Int8" + } + ], + "parameters": { + "Left": [] + }, + "nullable": [ + null, + null, + null + ] + }, + "hash": "872c06b3098cf59797bcaa90d6da5c6acbf5468c190fa72b6b6707d950267752" +} diff --git a/backend/ee-repo-ref.txt b/backend/ee-repo-ref.txt index 263784f2aa..58f9aaf07d 100644 --- a/backend/ee-repo-ref.txt +++ b/backend/ee-repo-ref.txt @@ -1 +1 @@ -421cf2a8b4f98b421e93c0fc7c1c378314a66e50 +9855e1b7a43a0a33e04f8accf1c497af3fd9b139 diff --git a/backend/windmill-queue/src/jobs.rs b/backend/windmill-queue/src/jobs.rs index 880b72cc0e..7b552fb02c 100644 --- a/backend/windmill-queue/src/jobs.rs +++ b/backend/windmill-queue/src/jobs.rs @@ -6869,9 +6869,19 @@ async fn push_inner<'c, 'd>( // `schedule_path` (see `FlowJob::schedule_path`), so counting per push would // score one run as a fire per step job — a loop pushes two of those per // iteration — burying every other kind, and would sit on the per-step path. + // + // A job a suspended trigger parks is not a fire: it counts as `fired` only + // when `resume_suspended_trigger_jobs` releases it, or never if discarded. + // Both are counted before the queue caps below and the caller's commit, so a + // rejected push still counts; accepted for a telemetry counter. if flow_step_id.is_none() { if let Some(kind) = trigger_kind.as_ref() { - windmill_common::feature_usage::log_feature_usage("trigger", "fired", kind.as_str()); + let action = if suspended_mode.unwrap_or(false) { + "suspended" + } else { + "fired" + }; + windmill_common::feature_usage::log_feature_usage("trigger", action, kind.as_str()); } } diff --git a/backend/windmill-trigger/src/global_handler.rs b/backend/windmill-trigger/src/global_handler.rs index 6e1c92d588..ec74b4f139 100644 --- a/backend/windmill-trigger/src/global_handler.rs +++ b/backend/windmill-trigger/src/global_handler.rs @@ -14,6 +14,7 @@ use windmill_api_jobs::execution::cancel_jobs; use windmill_common::{ db::{UserDB, DB}, error::{self, Error, Result}, + feature_usage::log_feature_usage, jobs::{delete_jobs, JobTriggerKind}, triggers::TriggerMetadata, }; @@ -168,14 +169,16 @@ pub async fn resume_suspended_trigger_jobs( .await? }; - let trigger_metadata = TriggerMetadata::new(Some(trigger_path.clone()), trigger_kind); + let trigger_metadata = TriggerMetadata::new(Some(trigger_path.clone()), trigger_kind.clone()); let l = jobs.len(); + let mut unsuspended_in_place = 0; for job in jobs { - // If job was created before trigger was edited, simply update it to unsuspend + // If job was created after trigger was edited, simply update it to unsuspend // instead of deleting and repushing if job.created_at > trigger.edited_at { + unsuspended_in_place += 1; // Map the placeholder unassigned kind back to its assigned counterpart so // singlestepflow wrappers (retry/error_handler/skip_handler) keep their // flow-orchestrator identity. @@ -209,7 +212,7 @@ pub async fn resume_suspended_trigger_jobs( .execute(&mut *tx) .await?; } else { - // Job was created after trigger edit - delete and repush with new configuration + // Job was created before trigger edit - delete and repush with new configuration // Pass the transaction to trigger_runnable_inner so everything is in the same transaction let (_uuid, _delete_after_use, _early_return, _has_failure_module, tx_o) = trigger_runnable_inner( @@ -274,6 +277,16 @@ pub async fn resume_suspended_trigger_jobs( tx.commit().await?; + // A repushed job is counted as `fired` inside `push`, before this commit, so + // a resume that rolls back still counts it; accepted for a telemetry counter. + // One unsuspended in place never goes through `push`, so it is counted here. + for _ in 0..unsuspended_in_place { + log_feature_usage("trigger", "fired", trigger_kind.as_str()); + } + for _ in 0..l { + log_feature_usage("trigger", "resumed", trigger_kind.as_str()); + } + Ok(Json(format!("Reassigned {} jobs", l))) } @@ -334,6 +347,9 @@ pub async fn cancel_suspended_trigger_jobs( true, ) .await?; + for _ in 0..cancelled_jobs.0.len() { + log_feature_usage("trigger", "discarded", trigger_kind.as_str()); + } Ok(Json(format!("Canceled {} jobs", cancelled_jobs.0.len()))) } else { Ok(Json(format!("No jobs to cancel"))) diff --git a/docs/feature-telemetry.md b/docs/feature-telemetry.md index 9dacfc1cbd..c0d7b9c027 100644 --- a/docs/feature-telemetry.md +++ b/docs/feature-telemetry.md @@ -4,7 +4,7 @@ anonymous usage-stats payload. It answers "does anyone use this, and which variant do they pick" without any identifying data leaving the instance. -It currently carries 58 registered actions across twenty features (`ai_session`, `ai_chat`, +It currently carries 61 registered actions across twenty features (`ai_session`, `ai_chat`, `ai_fix`, `ai_agent`, `ai_agent_eval`, `app_sandbox`, `datatable`, `db_manager`, `flow_editor`, `flow_run`, `flow_step`, `home`, `run_form`, `debugger`, `trigger`, `command_script`, `hub_script`, `usage_meter`, `sso_groups_claim`, `cloud_trial_offer`). Nearly all of the @@ -48,7 +48,7 @@ vocabulary closed and small — enumerate the values in a TS union next to the c ## The recipe -Four steps. Skipping step 1 or 3 fails quietly. +Four steps. Skipping step 1 fails quietly. **1. Register the pair** in `FEATURE_USAGE_KINDS` (`backend/windmill-common/src/feature_usage_ee.rs`, tracked in `windmill-ee-private`). An @@ -68,9 +68,10 @@ Fire-and-forget. Events sum locally per `(workspace, feature, kind, key, entityI every 30s, on `visibilitychange` → hidden, and on `pagehide`; 50 events per request, and a failed batch is dropped rather than retried. -**3. Update the disclosure.** `InstanceSettings.svelte` lists what a non-minimal payload contains -(two places — the copy appears twice). A new counter that isn't named there means the instance -under-discloses what it sends. This has already drifted once. +**3. Check the disclosure.** `InstanceSettings.svelte` describes a non-minimal payload by +category, not counter by counter (two places — the copy appears twice); keep it short. A plain +counter needs no edit. Only a key that sends a new kind of value beyond a count — a name or an +identifier, like the AI model and hub names it already lists — must be added there. **4. Verify a row lands.** The silent-drop path means "no error" proves nothing: diff --git a/frontend/src/lib/components/InstanceSettings.svelte b/frontend/src/lib/components/InstanceSettings.svelte index 7cb8ced225..0c27a3c7c4 100644 --- a/frontend/src/lib/components/InstanceSettings.svelte +++ b/frontend/src/lib/components/InstanceSettings.svelte @@ -1065,54 +1065,19 @@ Telemetry is required on Enterprise Edition for license compliance. When minimal telemetry is enabled, only the following data is sent:
    -
  • version of your instance
  • -
  • instance base URL
  • -
  • login type usage (login type, count)
  • -
  • worker usage (worker, worker instance, vCPUs, memory)
  • -
  • user usage (author count, operator count, the distinct guests of the last 30 days, - the seats they add past the free allowance, and the workspaces that allow guests)
  • -
  • superadmin email addresses
  • -
  • development instance status
  • +
  • Instance version and base URL
  • +
  • Login types, user and seat counts
  • +
  • Worker count, vCPUs and memory
  • +
  • Superadmin email addresses
  • +
  • Development instance status

When minimal telemetry is disabled, the following is also collected:
    -
  • job usage (language, total duration, count)
  • -
  • git sync repo count (sync vs promotion mode)
  • -
  • feature usage (counts of which product features are used, including AI provider and - model identifiers, the names of public hub scripts used, the languages debug sessions - are started for, whether AI chat skills are turned on or off and how often one is - loaded, whether an AI agent run narrows the tools it may call and whether that leaves - it with none, whether SSO logins evaluate an IdP groups claim (SAML or OIDC) and - change a membership, the plan tier and quota shown when the execution meter is opened, - whether app sandbox isolation is turned on, whether a step's workspace script is - edited from the flow editor, which skin approval steps are given, how many AI sessions - are brought back from the workspace object storage backup, how data tables and their - migrations are set up and used, how often the database manager is switched between its - data and schema diagram views, how often an empty workspace home is seen, how often - the home page’s create menu and hub-project picker are opened and from which entry - point, the name of any public hub project imported from the home page and how far that - import got, whether a pre-approved trial offer was opened, whether data tables are put - under roles and whether callers name a role or take the default, which kinds of access - change (grant, revoke, ownership, default privileges) are applied to data tables, and - whether a data table under roles is cloned into a fork with its schema only or with - its data, last 30 days)
  • -
  • feature adoption (counts of which flow, script, trigger, worker and data table - features your deployed items use, including how many apps run sandboxed, how many data - tables exist per database kind, how many use migrations, and what references them)
  • -
  • resource counts (workspaces, scripts per language, flows, workflows as code, low-code - apps, raw apps)
  • -
  • infrastructure info (container runtime, managed database provider, database version, - size and cluster size, max and active connections, object storage backend)
  • +
  • Job counts and durations per language
  • +
  • Git sync repository counts
  • +
  • Feature usage and adoption counts, including AI model and public Hub names
  • +
  • Counts of workspaces, scripts, flows and apps
  • +
  • Database, runtime and object storage details

For air-gapped instances, you can download the telemetry data and send it manually. @@ -1141,45 +1106,13 @@ Anonymous usage data is collected to help improve Windmill.
The following information is collected:
    -
  • version of your instance
  • -
  • instance base URL
  • -
  • job usage (language, total duration, count)
  • -
  • login type usage (login type, count)
  • -
  • worker usage (worker, worker instance, vCPUs, memory)
  • -
  • user usage (author count, operator count, the distinct guests of the last 30 days, - the seats they add past the free allowance, and the workspaces that allow guests)
  • -
  • development instance status
  • -
  • feature usage (counts of which product features are used, including AI provider and - model identifiers, the names of public hub scripts used, the languages debug sessions - are started for, whether AI chat skills are turned on or off and how often one is - loaded, whether an AI agent run narrows the tools it may call and whether that leaves - it with none, whether SSO logins evaluate an IdP groups claim (SAML or OIDC) and - change a membership, the plan tier and quota shown when the execution meter is opened, - whether app sandbox isolation is turned on, whether a step's workspace script is - edited from the flow editor, which skin approval steps are given, how many AI sessions - are brought back from the workspace object storage backup, how data tables and their - migrations are set up and used, how often the database manager is switched between its - data and schema diagram views, how often an empty workspace home is seen, how often - the home page’s create menu and hub-project picker are opened and from which entry - point, the name of any public hub project imported from the home page and how far that - import got, whether a pre-approved trial offer was opened, whether data tables are put - under roles and whether callers name a role or take the default, which kinds of access - change (grant, revoke, ownership, default privileges) are applied to data tables, and - whether a data table under roles is cloned into a fork with its schema only or with - its data, last 30 days)
  • -
  • feature adoption (counts of which flow, script, trigger, worker and data table - features your deployed items use, including how many apps run sandboxed, how many data - tables exist per database kind, how many use migrations, and what references them)
  • -
  • resource counts (workspaces, scripts per language, flows, workflows as code, low-code - apps, raw apps)
  • +
  • Instance version and base URL
  • +
  • Job counts and durations per language
  • +
  • Login types, user and seat counts
  • +
  • Worker count, vCPUs and memory
  • +
  • Development instance status
  • +
  • Feature usage and adoption counts, including AI model and public Hub names
  • +
  • Counts of workspaces, scripts, flows and apps
{/if}