name: Agent Observability Benchmark on: workflow_dispatch: inputs: run_greptimedb: description: Run GreptimeDB type: boolean default: true required: false run_clickhouse: description: Run ClickHouse type: boolean default: false required: false run_victorialogs: description: Run VictoriaLogs type: boolean default: false required: false dataset: description: 'Dataset size: S=5K, P=10M, M=100M' type: choice options: - S - P - M default: S required: true greptimedb_tag: description: Docker Hub greptime/greptimedb tag type: string default: latest required: true clickhouse_tag: description: Docker Hub clickhouse/clickhouse-server tag type: string default: 26.6.1.1193 required: true victorialogs_tag: description: Docker Hub victoriametrics/victoria-logs tag or tag@sha256 digest type: string default: v1.52.0@sha256:47b820890d64c4575a2a0a46415dcd8a4fd59a0f1fcd6a377693d7aea639442e required: true o11ybench_ref: description: Optional full commit SHA override (default uses runtime image revision) type: string required: false runtime_image: description: Aliyun runtime image (tag or digest reference) type: string default: greptime-registry.cn-hangzhou.cr.aliyuncs.com/tools/o11ybench-runtime:20260917-8111034e@sha256:ce9dee2abb1e70f907a09ea01854904a91cc97748b0c4ef6f33d9ec34044bec0 required: true provider: description: Cloud provider for the benchmark runner type: choice options: [Aliyun, AWS] default: Aliyun required: true instance_type: description: 'Instance type; auto = Aliyun: ecs.c9i.2xlarge, AWS: c7i.2xlarge (both 8 vCPU / 16 GiB). Enter a specific type to override.' type: string default: auto required: true system_disk_gib: description: 'System disk GiB: auto (or 0) = S 80, P 100, M 500; enter 20..2048 to override' type: string default: auto required: true benchmark_timeout_minutes: description: Benchmark job timeout in minutes (1..360) type: number default: 360 required: true janitor_ttl_hours: description: Instance lifetime before janitor eligibility (1..168 hours; at least timeout + 2h) type: number default: 8 required: true db_cpus: description: CPU limit per DB container type: string default: '8' required: true db_memory: description: Memory limit per DB container type: string default: 16g required: true permissions: contents: read env: PROFILE: ${{ inputs.dataset }} GREPTIMEDB_TAG: ${{ inputs.greptimedb_tag }} CLICKHOUSE_TAG: ${{ inputs.clickhouse_tag }} VICTORIALOGS_TAG: ${{ inputs.victorialogs_tag }} O11YBENCH_REF: ${{ inputs.o11ybench_ref }} RUNTIME_IMAGE: ${{ inputs.runtime_image }} PROVIDER: ${{ inputs.provider }} REQUESTED_INSTANCE_TYPE: ${{ inputs.instance_type }} SYSTEM_DISK_GIB: ${{ inputs.system_disk_gib }} BENCHMARK_TIMEOUT_MINUTES: ${{ inputs.benchmark_timeout_minutes }} JANITOR_TTL_HOURS: ${{ inputs.janitor_ttl_hours }} DB_CPUS: ${{ inputs.db_cpus }} DB_MEMORY: ${{ inputs.db_memory }} jobs: validate: runs-on: ubuntu-latest timeout-minutes: 10 outputs: instance_type: ${{ steps.runner.outputs.instance_type }} targets: ${{ steps.inputs.outputs.targets }} system_disk_gib: ${{ steps.inputs.outputs.system_disk_gib }} greptimedb: ${{ steps.inputs.outputs.greptimedb }} clickhouse: ${{ steps.inputs.outputs.clickhouse }} victorialogs: ${{ steps.inputs.outputs.victorialogs }} steps: - uses: actions/checkout@v4 with: path: greptimedb persist-credentials: false - name: Resolve CI runner id: runner run: python3 greptimedb/.github/scripts/ci-runner-config.py --provider "$PROVIDER" --instance-type "$REQUESTED_INSTANCE_TYPE" - name: Test CI runner helpers run: | python3 greptimedb/tests/perf/test_aliyun_ecs_runner_scripts.py python3 greptimedb/tests/perf/test_ci_runner_scripts.py python3 greptimedb/tests/perf/test_agent_observability_summary.py - name: Validate requested benchmark id: inputs shell: bash env: greptimedb: ${{ inputs.run_greptimedb }} clickhouse: ${{ inputs.run_clickhouse }} victorialogs: ${{ inputs.run_victorialogs }} run: | set -euo pipefail # Validate cloud lifecycle budgets before provisioning; runtime owns workload validation. case "$PROFILE" in S) auto_disk_gib=80 ;; P) auto_disk_gib=100 ;; M) auto_disk_gib=500 ;; *) echo "Unknown dataset: $PROFILE" >&2; exit 1 ;; esac if [[ "$SYSTEM_DISK_GIB" == auto || "$SYSTEM_DISK_GIB" == 0 ]]; then SYSTEM_DISK_GIB=$auto_disk_gib fi for value in "$SYSTEM_DISK_GIB" "$BENCHMARK_TIMEOUT_MINUTES" "$JANITOR_TTL_HOURS"; do [[ "$value" =~ ^[1-9][0-9]{0,3}$ ]] || { echo 'Resource budgets must be positive integers' >&2; exit 1; } done ((SYSTEM_DISK_GIB >= 20 && SYSTEM_DISK_GIB <= 2048)) printf 'system_disk_gib=%s\n' "$SYSTEM_DISK_GIB" >> "$GITHUB_OUTPUT" printf 'Dataset %s: system disk %s GiB\n' "$PROFILE" "$SYSTEM_DISK_GIB" ((BENCHMARK_TIMEOUT_MINUTES <= 360 && JANITOR_TTL_HOURS <= 168)) # Reserve 45m provision + 15m teardown + 60m queue headroom before expiry. ((JANITOR_TTL_HOURS * 60 >= BENCHMARK_TIMEOUT_MINUTES + 120)) || { echo 'Janitor TTL must cover benchmark timeout plus 2 hours' >&2; exit 1; } [[ -z "$O11YBENCH_REF" || "$O11YBENCH_REF" =~ ^[0-9a-f]{40}$ ]] [[ -n "$RUNTIME_IMAGE" ]] || { echo 'runtime_image must not be empty' >&2; exit 1; } selected= for target in greptimedb clickhouse victorialogs; do enabled=${!target} printf '%s=%s\n' "$target" "$enabled" >> "$GITHUB_OUTPUT" if [[ "$enabled" == true ]]; then selected="${selected:+$selected,}$target" fi done [[ -n "$selected" ]] || { echo 'Select at least one database' >&2; exit 1; } printf 'targets=%s\n' "$selected" >> "$GITHUB_OUTPUT" provision: needs: validate runs-on: ubuntu-latest timeout-minutes: 45 outputs: label: ${{ steps.ecs.outputs.label || steps.aws.outputs.label }} instance_id: ${{ steps.ecs.outputs.instance_id || steps.aws.outputs.instance_id }} runner_name: ${{ steps.ecs.outputs.runner_name || steps.aws.outputs.runner_name }} steps: - uses: actions/checkout@v4 with: path: greptimedb persist-credentials: false - name: Create ephemeral ECS runner if: inputs.provider == 'Aliyun' id: ecs uses: ./greptimedb/.github/actions/aliyun-ecs-create with: access-key-id: ${{ secrets.ALICLOUD_ECS_ACCESS_KEY_ID }} access-key-secret: ${{ secrets.ALICLOUD_ECS_ACCESS_KEY_SECRET }} github-token: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }} region-id: ${{ vars.ALIYUN_ECS_REGION_ID }} vswitch-id: ${{ vars.ALIYUN_ECS_VSWITCH_ID }} security-group-id: ${{ vars.ALIYUN_ECS_SECURITY_GROUP_ID }} image-id: ${{ vars.QUERY_REGRESSION_ECS_IMAGE_ID }} instance-type: ${{ needs.validate.outputs.instance_type }} enable-docker: 'true' system-disk-gib: ${{ needs.validate.outputs.system_disk_gib }} ttl-hours: ${{ inputs.janitor_ttl_hours }} resource-group-id: ${{ vars.ALIYUN_ECS_RESOURCE_GROUP_ID }} runner-uid: ${{ vars.QUERY_REGRESSION_RUNNER_UID || '1001' }} runner-gid: ${{ vars.QUERY_REGRESSION_RUNNER_GID || '1001' }} - uses: aws-actions/configure-aws-credentials@v4 if: inputs.provider == 'AWS' with: aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }} aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }} aws-region: ${{ vars.EC2_RUNNER_REGION }} - uses: astral-sh/setup-uv@v6 if: inputs.provider == 'AWS' - name: Create ephemeral EC2 runner if: inputs.provider == 'AWS' id: aws env: GH_PERSONAL_ACCESS_TOKEN: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }} AWS_REGION: ${{ vars.EC2_RUNNER_REGION }} AWS_EC2_IMAGE_ID: ${{ vars.BENCHMARK_EC2_IMAGE_ID || 'auto' }} AWS_EC2_INSTANCE_TYPE: ${{ needs.validate.outputs.instance_type }} AWS_EC2_SUBNET_ID: ${{ vars.EC2_RUNNER_SUBNET_ID }} AWS_EC2_SECURITY_GROUP_ID: ${{ vars.EC2_RUNNER_SECURITY_GROUP_ID }} AWS_EC2_SYSTEM_DISK_GIB: ${{ needs.validate.outputs.system_disk_gib }} AWS_EC2_TTL_HOURS: ${{ inputs.janitor_ttl_hours }} AWS_EC2_RUN_ID: ${{ github.run_id }}-${{ github.run_attempt }} run: uv run greptimedb/.github/scripts/aws-ec2-runner-provision.py benchmark: needs: - validate - provision runs-on: ${{ needs.provision.outputs.label }} timeout-minutes: ${{ fromJSON(format('{0}', inputs.benchmark_timeout_minutes)) }} env: INSTANCE_TYPE: ${{ needs.validate.outputs.instance_type }} TARGETS: ${{ needs.validate.outputs.targets }} SYSTEM_DISK_GIB: ${{ needs.validate.outputs.system_disk_gib }} OWNER: o11ybench-${{ github.run_id }}-${{ github.run_attempt }} ARTIFACT_ROOT: ${{ github.workspace }}/benchmark-data/${{ github.run_id }}-${{ github.run_attempt }} MANIFEST_PATH: ${{ github.workspace }}/run-manifest.json O11YBENCH_DIR: ${{ github.workspace }}/o11ybench GREPTIMEDB_HTTP_PORT: '14000' GREPTIMEDB_GRPC_PORT: '14001' CLICKHOUSE_HTTP_PORT: '18123' CLICKHOUSE_NATIVE_PORT: '19000' VICTORIALOGS_HTTP_PORT: '19428' steps: - uses: actions/checkout@v4 with: path: greptimedb persist-credentials: false - name: Set up benchmark dependencies uses: ./greptimedb/.github/actions/setup-benchmark - name: Check Docker on ephemeral Linux host shell: bash run: | set -euo pipefail [[ "$(uname -m)" == x86_64 ]] jq --version docker info # The workspace (and Docker state) must not use a small /tmp filesystem. df -h "$GITHUB_WORKSPACE" /var/lib/docker - name: Resolve image tags and verify runtime revision shell: bash run: | set -euo pipefail actual= [[ "$RUNTIME_IMAGE" == *.cr.aliyuncs.com/* ]] images='{}' resolve_image() { local target=$1 reference=$2 variable=$3 metadata resolved revision docker pull "$reference" metadata=$(docker image inspect "$reference") resolved=$(jq -er '.[0].RepoDigests[0]' <<< "$metadata") if [[ "$target" == runtime ]]; then revision=$(jq -er '.[0].Config.Labels["org.opencontainers.image.revision"]' <<< "$metadata") [[ "$revision" =~ ^[0-9a-f]{40}$ ]] || { echo 'Runtime lacks a full commit revision' >&2; return 1; } [[ -z "$O11YBENCH_REF" || "$revision" == "$O11YBENCH_REF" ]] || { echo 'Runtime and requested o11ybench revisions differ' >&2; return 1; } actual=$revision printf 'RESOLVED_O11YBENCH_REF=%s\n' "$revision" >> "$GITHUB_ENV" fi printf '%s=%s\n' "$variable" "$resolved" >> "$GITHUB_ENV" images=$(jq --arg target "$target" --arg requested "$reference" --argjson metadata "$metadata" \ '. + {($target): {requested: $requested, id: $metadata[0].Id, digests: ($metadata[0].RepoDigests // [])}}' <<< "$images") } resolve_image runtime "$RUNTIME_IMAGE" RESOLVED_RUNTIME_IMAGE for target in greptimedb clickhouse victorialogs; do [[ "$TARGETS" == all || ",$TARGETS," == *",$target,"* ]] || continue case "$target" in greptimedb) resolve_image "$target" "greptime/greptimedb:$GREPTIMEDB_TAG" GREPTIMEDB_IMAGE ;; clickhouse) resolve_image "$target" "clickhouse/clickhouse-server:$CLICKHOUSE_TAG" CLICKHOUSE_IMAGE ;; victorialogs) resolve_image "$target" "victoriametrics/victoria-logs:$VICTORIALOGS_TAG" VICTORIALOGS_IMAGE ;; esac done jq -n --arg sha "$actual" --arg targets "$TARGETS" --arg profile "$PROFILE" \ --arg cpus "$DB_CPUS" --arg memory "$DB_MEMORY" --argjson images "$images" \ --arg provider "$PROVIDER" --arg instance_type "$INSTANCE_TYPE" --argjson disk_gib "$SYSTEM_DISK_GIB" \ --argjson timeout_minutes "$BENCHMARK_TIMEOUT_MINUTES" --argjson ttl_hours "$JANITOR_TTL_HOURS" \ '{o11ybench_sha:$sha, targets:$targets, profile:$profile, db_cpus:$cpus, db_memory:$memory, images:$images, ($provider | if . == "AWS" then "ec2" else "ecs" end):{instance_type:$instance_type, system_disk_gib:$disk_gib, benchmark_timeout_minutes:$timeout_minutes, janitor_ttl_hours:$ttl_hours}}' > "$MANIFEST_PATH" - name: Checkout benchmark tools uses: actions/checkout@v4 with: repository: GreptimeTeam/o11ybench token: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }} ref: ${{ env.RESOLVED_O11YBENCH_REF }} path: o11ybench persist-credentials: false - name: Verify benchmark checkout revision shell: bash run: test "$(git -C "$O11YBENCH_DIR" rev-parse HEAD)" = "$RESOLVED_O11YBENCH_REF" - name: Generate dataset once id: generate run: |- GITHUB_STEP_SUMMARY= bash o11ybench/scripts/run-agent-observability.sh --stage generate --execute \ --owner "$OWNER" --runtime-image "$RESOLVED_RUNTIME_IMAGE" \ --root "$ARTIFACT_ROOT" --profile "$PROFILE" --db-cpus "$DB_CPUS" --db-memory "$DB_MEMORY" cp "$MANIFEST_PATH" "$ARTIFACT_ROOT/run-manifest.json" - name: Load and benchmark greptimedb id: greptimedb if: ${{ !cancelled() && steps.generate.outcome == 'success' && needs.validate.outputs.greptimedb == 'true' }} run: |- GITHUB_STEP_SUMMARY= bash o11ybench/scripts/run-agent-observability.sh --stage target --execute \ --owner "$OWNER" --runtime-image "$RESOLVED_RUNTIME_IMAGE" --root "$ARTIFACT_ROOT" \ --profile "$PROFILE" --db-cpus "$DB_CPUS" --db-memory "$DB_MEMORY" \ --target greptimedb --image "$GREPTIMEDB_IMAGE" \ --greptimedb-http-port "$GREPTIMEDB_HTTP_PORT" --greptimedb-grpc-port "$GREPTIMEDB_GRPC_PORT" - name: Load and benchmark clickhouse id: clickhouse if: ${{ !cancelled() && steps.generate.outcome == 'success' && needs.validate.outputs.clickhouse == 'true' }} run: |- GITHUB_STEP_SUMMARY= bash o11ybench/scripts/run-agent-observability.sh --stage target --execute \ --owner "$OWNER" --runtime-image "$RESOLVED_RUNTIME_IMAGE" --root "$ARTIFACT_ROOT" \ --profile "$PROFILE" --db-cpus "$DB_CPUS" --db-memory "$DB_MEMORY" \ --target clickhouse --image "$CLICKHOUSE_IMAGE" \ --clickhouse-http-port "$CLICKHOUSE_HTTP_PORT" --clickhouse-native-port "$CLICKHOUSE_NATIVE_PORT" - name: Load and benchmark victorialogs id: victorialogs if: ${{ !cancelled() && steps.generate.outcome == 'success' && needs.validate.outputs.victorialogs == 'true' }} run: |- GITHUB_STEP_SUMMARY= bash o11ybench/scripts/run-agent-observability.sh --stage target --execute \ --owner "$OWNER" --runtime-image "$RESOLVED_RUNTIME_IMAGE" --root "$ARTIFACT_ROOT" \ --profile "$PROFILE" --db-cpus "$DB_CPUS" --db-memory "$DB_MEMORY" \ --target victorialogs --image "$VICTORIALOGS_IMAGE" \ --victorialogs-http-port "$VICTORIALOGS_HTTP_PORT" - name: Compare all three targets if: ${{ !cancelled() && steps.greptimedb.outcome == 'success' && steps.clickhouse.outcome == 'success' && steps.victorialogs.outcome == 'success' }} run: |- docker run --rm --user "$(id -u):$(id -g)" --label "o11ybench.owner=$OWNER" \ --mount "type=bind,src=$ARTIFACT_ROOT,dst=/results" "$RESOLVED_RUNTIME_IMAGE" \ scripts/finalize_logbench_agent_observability_measured.py \ --greptimedb-summary /results/greptimedb-1/greptimedb-1/measured-summary.json \ --clickhouse-summary /results/clickhouse-1/clickhouse-1/measured-summary.json \ --victorialogs-summary /results/victorialogs-1/victorialogs-1/measured-summary.json \ --output /results/comparison.json - name: Clean up owned containers if: always() run: |- if [[ -f o11ybench/scripts/run-agent-observability.sh ]]; then GITHUB_STEP_SUMMARY= bash o11ybench/scripts/run-agent-observability.sh --stage cleanup --execute \ --owner "$OWNER" --root "$ARTIFACT_ROOT" fi - name: Benchmark summary if: always() run: | python3 greptimedb/.github/scripts/agent-observability-summary.py \ --root "$ARTIFACT_ROOT" --targets "$TARGETS" --section benchmark >> "$GITHUB_STEP_SUMMARY" - name: Load / lifecycle summary if: always() run: | python3 greptimedb/.github/scripts/agent-observability-summary.py \ --root "$ARTIFACT_ROOT" --targets "$TARGETS" --section lifecycle >> "$GITHUB_STEP_SUMMARY" - name: Upload results (no dataset) if: always() uses: actions/upload-artifact@v4 with: name: agent-observability-${{ github.run_id }}-${{ github.run_attempt }} path: |- ${{ env.MANIFEST_PATH }} ${{ env.ARTIFACT_ROOT }} !${{ env.ARTIFACT_ROOT }}/corpus/agent_observations.jsonl if-no-files-found: warn retention-days: 14 teardown: needs: - provision - benchmark if: ${{ always() && (needs.provision.outputs.instance_id != '' || (inputs.provider == 'AWS' && needs.provision.result != 'skipped')) }} runs-on: ubuntu-latest timeout-minutes: 15 steps: - uses: actions/checkout@v4 with: path: greptimedb persist-credentials: false - name: Delete ECS and unregister runner if: inputs.provider == 'Aliyun' uses: ./greptimedb/.github/actions/aliyun-ecs-delete with: access-key-id: ${{ secrets.ALICLOUD_ECS_ACCESS_KEY_ID }} access-key-secret: ${{ secrets.ALICLOUD_ECS_ACCESS_KEY_SECRET }} github-token: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }} region-id: ${{ vars.ALIYUN_ECS_REGION_ID }} instance-id: ${{ needs.provision.outputs.instance_id }} runner-name: ${{ needs.provision.outputs.runner_name }} - uses: aws-actions/configure-aws-credentials@v4 if: inputs.provider == 'AWS' with: aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }} aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }} aws-region: ${{ vars.EC2_RUNNER_REGION }} - uses: astral-sh/setup-uv@v6 if: inputs.provider == 'AWS' - name: Delete EC2 and unregister runner if: inputs.provider == 'AWS' env: GH_PERSONAL_ACCESS_TOKEN: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }} AWS_REGION: ${{ vars.EC2_RUNNER_REGION }} AWS_EC2_INSTANCE_ID: ${{ needs.provision.outputs.instance_id }} AWS_EC2_RUNNER_NAME: ${{ needs.provision.outputs.runner_name }} AWS_EC2_RUN_ID: ${{ github.run_id }}-${{ github.run_attempt }} run: uv run greptimedb/.github/scripts/aws-ec2-runner-teardown.py