name: Spider Bench on: pull_request: types: [opened, synchronize, reopened] workflow_dispatch: inputs: base_ref: description: Base commit or ref required: false default: main head_ref: description: Head commit or ref (defaults to the dispatched commit) required: false default: "" concurrency: group: spider-bench-${{ github.event.pull_request.number || github.ref }} cancel-in-progress: true # PR code builds and runs without a token capable of writing back to GitHub. # The separate Spider Bench Comment workflow consumes the resulting artifact. permissions: contents: read env: CARGO_TERM_COLOR: always CARGO_INCREMENTAL: "0" jobs: compare: name: Base vs HEAD if: >- github.event_name != 'pull_request' || (github.event.pull_request.head.repo.full_name == github.repository && github.actor != 'dependabot[bot]') runs-on: [self-hosted, moli-ci] timeout-minutes: 90 container: rust:1.96.1-bookworm env: BENCHMARK_OUTPUT: spider-bench-results BINARY_DIR: /tmp/moli-spider-bench-binaries TARGET_BASE_REF: ${{ github.event.pull_request.base.sha || inputs.base_ref || github.event.repository.default_branch }} HEAD_REF: ${{ github.event.pull_request.head.sha || inputs.head_ref || github.sha }} steps: # The benchmark harness is infrastructure, not part of the candidate # being measured. Keep it on the trusted target base even when the PR # branch predates the harness itself. - name: Checkout trusted target-base harness uses: actions/checkout@v4 with: fetch-depth: 0 persist-credentials: false ref: ${{ github.event.pull_request.base.sha || inputs.base_ref || github.event.repository.default_branch }} - name: Resolve commits and initialize evidence id: refs shell: bash run: | set -euo pipefail # actions/checkout writes safe.directory through a temporary HOME. # Re-register it in the container user's persistent HOME before the # first git command run by this job. git config --global --add safe.directory "$(pwd -P)" mkdir -p "$BENCHMARK_OUTPUT" printf '%s\n' '{"schema":"moli.browser-spider.run-metadata.v1","status":"initializing"}' \ > "$BENCHMARK_OUTPUT/run-metadata.json" target_base_sha=$(git rev-parse "${TARGET_BASE_REF}^{commit}") head_sha=$(git rev-parse "${HEAD_REF}^{commit}") base_sha=$(git merge-base "$target_base_sha" "$head_sha") git merge-base --is-ancestor "$base_sha" "$target_base_sha" git merge-base --is-ancestor "$base_sha" "$head_sha" echo "target_base_sha=$target_base_sha" >> "$GITHUB_OUTPUT" echo "base_sha=$base_sha" >> "$GITHUB_OUTPUT" echo "head_sha=$head_sha" >> "$GITHUB_OUTPUT" if (( GITHUB_RUN_NUMBER % 2 == 0 )); then execution_order=head-first else execution_order=base-first fi echo "execution_order=$execution_order" >> "$GITHUB_OUTPUT" printf '{\n "schema": "moli.browser-spider.run-metadata.v1",\n "status": "resolved",\n "target_base": { "sha": "%s" },\n "base": { "sha": "%s", "role": "common-ancestor" },\n "head": { "sha": "%s" },\n "harness": { "sha": "%s" },\n "execution_order": "%s"\n}\n' \ "$target_base_sha" "$base_sha" "$head_sha" "$target_base_sha" \ "$execution_order" > "$BENCHMARK_OUTPUT/run-metadata.json" - name: Build common-ancestor base and exact HEAD release binaries shell: bash env: TARGET_BASE_SHA: ${{ steps.refs.outputs.target_base_sha }} BASE_SHA: ${{ steps.refs.outputs.base_sha }} HEAD_SHA: ${{ steps.refs.outputs.head_sha }} run: | set -euo pipefail mkdir -p "$BINARY_DIR" restore_harness() { git checkout --detach "$TARGET_BASE_SHA" } trap restore_harness EXIT git checkout --detach "$BASE_SHA" cargo build --release --locked -p moli install -m 755 target/release/moli "$BINARY_DIR/moli-base" git checkout --detach "$HEAD_SHA" cargo build --release --locked -p moli install -m 755 target/release/moli "$BINARY_DIR/moli-head" restore_harness trap - EXIT - name: Install Node.js uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 with: node-version: 24.15.0 cache: npm cache-dependency-path: moli-benchmark/browser-spider-local/package-lock.json - name: Install Spider Bench dependencies run: npm ci --prefix moli-benchmark/browser-spider-local - name: Run deterministic and public-web A/B env: EXECUTION_ORDER: ${{ steps.refs.outputs.execution_order }} run: >- node moli-benchmark/browser-spider-local/compare.mjs run --base-bin "$BINARY_DIR/moli-base" --head-bin "$BINARY_DIR/moli-head" --base-sha "${{ steps.refs.outputs.base_sha }}" --head-sha "${{ steps.refs.outputs.head_sha }}" --execution-order "$EXECUTION_ORDER" --output-dir "$BENCHMARK_OUTPUT" --run-url "https://github.com/${{ github.repository }}/actions/runs/${{ github.run_id }}" - name: Render Actions job summary if: ${{ always() && !cancelled() }} continue-on-error: true shell: bash env: COMMENT_PATH: ${{ env.BENCHMARK_OUTPUT }}/comment.md run: | if [[ -f "$COMMENT_PATH" ]]; then cat "$COMMENT_PATH" >> "$GITHUB_STEP_SUMMARY" fi - name: Upload full benchmark evidence if: ${{ always() && !cancelled() }} uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4 with: name: spider-bench-results path: ${{ env.BENCHMARK_OUTPUT }} if-no-files-found: error retention-days: 14