From bc5569f537c7050cb2d03972e0814559e5984b3e Mon Sep 17 00:00:00 2001 From: ldm0 Date: Thu, 24 Sep 2026 17:36:57 +0800 Subject: [PATCH] ci: include WebMainBench in the PR regression report Add completion, failure, panic, crash, and timeout counts to the existing aggregate PR comment. Validate summary counts, bound failure details, and keep missing or incomplete evidence visible with a link to the source run. --- .../scripts/render-ci-regression-comment.cjs | 84 ++++++++++++- .../render-ci-regression-comment.test.cjs | 110 +++++++++++++++++- .github/workflows/ci-regression-comment.yml | 13 ++- moli-benchmark/README.md | 9 +- 4 files changed, 206 insertions(+), 10 deletions(-) diff --git a/.github/scripts/render-ci-regression-comment.cjs b/.github/scripts/render-ci-regression-comment.cjs index 5e06dcf1cb..e4cfbd6b3f 100644 --- a/.github/scripts/render-ci-regression-comment.cjs +++ b/.github/scripts/render-ci-regression-comment.cjs @@ -10,6 +10,8 @@ const MAX_RESULT_ROWS = 2_000; const MAX_DETAIL_ROWS = 10; const MAX_MATRIX_ROWS = 5_000; const MAX_CDP_GROUPS = 100; +const WEBMAINBENCH_CASES = 545; +const WEBMAINBENCH_STATUSES = ['success', 'expected_failure', 'failure', 'panic', 'crash', 'timeout', 'empty_output']; const MIB = 1024 * 1024; function isObject(value) { @@ -639,6 +641,79 @@ function renderCdp(section) { return lines; } +function loadWebMainBench(root) { + const summary = readJson(root, 'summary.json'); + if ( + !isObject(summary) || + summary.schema_version !== 1 || + summary.expected_cases !== WEBMAINBENCH_CASES || + count(summary.completed) === null || + summary.completed > WEBMAINBENCH_CASES || + typeof summary.passed !== 'boolean' || + !isObject(summary.counts) || + !Array.isArray(summary.issues) || + summary.issues.length > WEBMAINBENCH_CASES + 2 || + summary.issues.some((issue) => typeof issue !== 'string') || + Object.entries(summary.counts).some(([status, value]) => + !WEBMAINBENCH_STATUSES.includes(status) || count(value) === null || value > WEBMAINBENCH_CASES + ) + ) { + throw new Error('invalid WebMainBench artifact'); + } + const counts = Object.fromEntries(WEBMAINBENCH_STATUSES.map((status) => [status, summary.counts[status] ?? 0])); + if (sum(Object.values(counts)) !== summary.completed || counts.expected_failure > 1) { + throw new Error('inconsistent WebMainBench counts'); + } + return { ...summary, counts }; +} + +function webMainBenchOverview(section) { + if (!section.available) { + return { status: '⚪', signal: 'artifact unavailable or invalid' }; + } + const { completed, counts, passed, issues } = section.data; + const unexpected = completed - counts.success - counts.expected_failure; + const ok = passed && completed === WEBMAINBENCH_CASES && unexpected === 0 && issues.length === 0; + return { + status: ok ? '✅' : '❌', + signal: `${formatInteger(completed)}/${WEBMAINBENCH_CASES} completed; ${formatInteger(unexpected)} unexpected failures`, + }; +} + +function renderWebMainBench(section, runLink) { + const overview = webMainBenchOverview(section); + const lines = [`
WebMainBench · 545 pages — ${overview.status} ${overview.signal}`, '']; + if (!section.available) { + lines.push('Artifact unavailable or invalid. See the source CI run for infrastructure details.', '', '
'); + return lines; + } + const { completed, counts, issues } = section.data; + const unexpected = completed - counts.success - counts.expected_failure; + lines.push( + '| Completed | Successful | Expected DNS failure | Unexpected failures | Panics | Crashes | Timeouts | Empty output |', + '| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |', + `| ${formatInteger(completed)}/${WEBMAINBENCH_CASES} | ${formatInteger(counts.success)} | ${formatInteger(counts.expected_failure)} | ${formatInteger(unexpected)} | ${formatInteger(counts.panic)} | ${formatInteger(counts.crash)} | ${formatInteger(counts.timeout)} | ${formatInteger(counts.empty_output)} |`, + '', + 'Frozen HTML · Linux · `--wait done` · 45 seconds per page · no retries. Content quality is evaluated separately.' + ); + if (issues.length !== 0) { + lines.push('', '**Failures and incomplete evidence**', ''); + for (const issue of issues.slice(0, MAX_DETAIL_ROWS)) { + lines.push(`- ${code(issue, 200)}`); + } + if (issues.length > MAX_DETAIL_ROWS) { + lines.push(`- ${formatInteger(issues.length - MAX_DETAIL_ROWS)} more issues in the artifact.`); + } + } + lines.push( + '', + `Diagnostics: \`webmainbench-results\` — every page's Markdown, stderr, and exit status.${runLink ? ` [Source run and artifacts](${runLink})` : ''}`, + '', + '' + ); + return lines; +} + function trustedRunUrl(value) { try { const url = new URL(value); @@ -651,13 +726,14 @@ function trustedRunUrl(value) { return null; } -function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, runUrl, conclusion }) { +function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, webmainbenchRoot, runUrl, conclusion }) { const sections = { release: loadSection(() => loadRelease(releaseRoot)), frontend: loadSection(() => loadFrontend(frontendRoot)), agent: loadSection(() => loadAgent(agentRoot)), runtime: loadSection(() => loadRuntime(runtimeRoot)), cdp: loadSection(() => loadCdp(cdpRoot)), + webmainbench: loadSection(() => loadWebMainBench(webmainbenchRoot)), }; const overviews = [ ['Release regression', releaseOverview(sections.release)], @@ -665,6 +741,7 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo ['Agent episodes', agentOverview(sections.agent)], ['Runtime/CDP contracts', runtimeOverview(sections.runtime)], ['CDP smoke', cdpOverview(sections.cdp)], + ['WebMainBench', webMainBenchOverview(sections.webmainbench)], ]; const available = Object.values(sections).filter((section) => section.available).length; const sourceConclusion = ['success', 'failure', 'cancelled', 'timed_out', 'in_progress'].includes(conclusion) @@ -675,7 +752,7 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo COMMENT_MARKER, '## CI Regression Report', '', - `${link ? `[Source CI run](${link})` : 'Source CI run'} · source state at render: \`${sourceConclusion}\` · artifacts: \`${available}/5\``, + `${link ? `[Source CI run](${link})` : 'Source CI run'} · source state at render: \`${sourceConclusion}\` · artifacts: \`${available}/${overviews.length}\``, '', '| Check | Status | Signal |', '| --- | :---: | --- |', @@ -695,6 +772,8 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo '', ...renderCdp(sections.cdp), '', + ...renderWebMainBench(sections.webmainbench, link), + '', '_All artifact fields are parsed by the trusted default-branch renderer; missing or invalid inputs remain visible as unavailable._', '' ); @@ -729,6 +808,7 @@ function main(argv = process.argv.slice(2)) { agentRoot: args.agent, runtimeRoot: args.runtime, cdpRoot: args.cdp, + webmainbenchRoot: args.webmainbench, runUrl: args['run-url'], conclusion: args.conclusion, }); diff --git a/.github/scripts/render-ci-regression-comment.test.cjs b/.github/scripts/render-ci-regression-comment.test.cjs index 4ef1765c35..c2269331b8 100644 --- a/.github/scripts/render-ci-regression-comment.test.cjs +++ b/.github/scripts/render-ci-regression-comment.test.cjs @@ -104,6 +104,18 @@ function agentTarget({ chrome = false, failure = false }) { }; } +function webMainBenchSummary(overrides = {}) { + return { + schema_version: 1, + expected_cases: 545, + completed: 545, + counts: { success: 544, expected_failure: 1 }, + passed: true, + issues: [], + ...overrides, + }; +} + function createArtifacts(root) { const releaseRoot = path.join(root, 'release'); writeJson(releaseRoot, 'base-startup/startup/summary.json', startupSummary(0)); @@ -190,10 +202,13 @@ function createArtifacts(root) { ], }); - return { releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot }; + const webmainbenchRoot = path.join(root, 'webmainbench'); + writeJson(webmainbenchRoot, 'summary.json', webMainBenchSummary()); + + return { releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, webmainbenchRoot }; } -test('renders all five trusted artifact sections into one bounded report', (t) => { +test('renders all six trusted artifact sections into one bounded report', (t) => { const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-report-')); t.after(() => fs.rmSync(root, { recursive: true, force: true })); const roots = createArtifacts(root); @@ -204,7 +219,7 @@ test('renders all five trusted artifact sections into one bounded report', (t) = }); assert.ok(report.startsWith(COMMENT_MARKER)); - assert.match(report, /artifacts: `5\/5`/); + assert.match(report, /artifacts: `6\/6`/); assert.match(report, /Raw binary \| 100 B \| 110 B \| \+10 B \| \+10\.000000%/); assert.match(report, /Frontend differential/); assert.match(report, /1 Chromium reference recoveries/); @@ -218,6 +233,10 @@ test('renders all five trusted artifact sections into one bounded report', (t) = assert.match(report, /PSS partial/); assert.match(report, /Runtime and CDP session contracts/); assert.match(report, /`bad\\\|group<`/); + assert.match(report, /\| WebMainBench \| ✅ \| 545\/545 completed; 0 unexpected failures \|/); + assert.match(report, /\| 545\/545 \| 544 \| 1 \| 0 \| 0 \| 0 \| 0 \| 0 \|/); + assert.match(report, /\[Source run and artifacts\]\(https:\/\/github\.com\/lexmount\/moli\/actions\/runs\/123\)/); + assert.match(report, /Diagnostics: `webmainbench-results`/); assert.ok(Buffer.byteLength(report, 'utf8') < 32 * 1024); }); @@ -227,8 +246,8 @@ test('renders missing artifacts as unavailable without throwing', () => { conclusion: 'timed_out', }); - assert.match(report, /artifacts: `0\/5`/); - assert.equal((report.match(/Artifact unavailable or invalid\./g) || []).length, 5); + assert.match(report, /artifacts: `0\/6`/); + assert.equal((report.match(/Artifact unavailable or invalid\./g) || []).length, 6); assert.doesNotMatch(report, /\]\(not-a-trusted-url\)/); }); @@ -246,5 +265,84 @@ test('rejects an unbounded frontend result list as unavailable', (t) => { }); const report = renderReport({ frontendRoot, conclusion: 'failure' }); - assert.match(report, /artifacts: `0\/5`/); + assert.match(report, /artifacts: `0\/6`/); +}); + +test('shows WebMainBench failures and escapes issue text', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-failures-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + writeJson(root, 'summary.json', webMainBenchSummary({ + passed: false, + counts: { success: 540, expected_failure: 1, panic: 1, crash: 1, timeout: 1, empty_output: 1 }, + issues: ['first-case: panic', 'second-case: timeout', 'bad|case<`\n: empty_output'], + })); + + const report = renderReport({ webmainbenchRoot: root, conclusion: 'failure' }); + + assert.match(report, /\| WebMainBench \| ❌ \| 545\/545 completed; 4 unexpected failures \|/); + assert.match(report, /\| 545\/545 \| 540 \| 1 \| 4 \| 1 \| 1 \| 1 \| 1 \|/); + assert.match(report, /`first-case: panic`/); + assert.match(report, /bad\\\|case<' <img src=x>/); + assert.doesNotMatch(report, //); +}); + +test('does not display a green WebMainBench result for incomplete or failed evidence', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-incomplete-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + for (const summary of [ + webMainBenchSummary({ completed: 200, counts: { success: 200 } }), + webMainBenchSummary({ passed: false }), + webMainBenchSummary({ issues: ['Infrastructure error: fixture failed'] }), + ]) { + writeJson(root, 'summary.json', summary); + const report = renderReport({ webmainbenchRoot: root, conclusion: 'success' }); + assert.match(report, /artifacts: `1\/6`/); + assert.match(report, /\| WebMainBench \| ❌ \|/); + assert.doesNotMatch(report, /\| WebMainBench \| ✅ \|/); + } +}); + +test('rejects malformed WebMainBench counts and oversized issue lists', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-invalid-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + for (const overrides of [ + { schema_version: 2 }, + { expected_cases: 1 }, + { completed: 546 }, + { counts: { success: '544', expected_failure: 1 } }, + { counts: { success: -1 } }, + { counts: { success: 544, unknown_status: 1 } }, + { counts: { success: 543, expected_failure: 2 } }, + { counts: {} }, + { issues: [{}] }, + { issues: Array.from({ length: 548 }, () => 'failure') }, + ]) { + writeJson(root, 'summary.json', webMainBenchSummary(overrides)); + const report = renderReport({ webmainbenchRoot: root, conclusion: 'success' }); + assert.match(report, /artifacts: `0\/6`/, JSON.stringify(overrides)); + assert.match(report, /\| WebMainBench \| ⚪ \| artifact unavailable or invalid \|/); + } +}); + +test('bounds WebMainBench issue details and does not use an untrusted diagnostics URL', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-bounded-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + writeJson(root, 'summary.json', webMainBenchSummary({ + counts: { failure: 545 }, + passed: false, + issues: Array.from({ length: 545 }, (_, index) => `case-${index}: ${'x'.repeat(1_000)}`), + diagnostics_url: 'https://attacker.example/report', + })); + + const report = renderReport({ + webmainbenchRoot: root, + runUrl: 'javascript:alert(1)', + conclusion: 'failure', + }); + + assert.match(report, /535 more issues in the artifact/); + assert.match(report, /`case-9:/); + assert.doesNotMatch(report, /`case-10:/); + assert.doesNotMatch(report, /attacker\.example|javascript:|\[Source run and artifacts\]/); + assert.ok(Buffer.byteLength(report, 'utf8') < 32 * 1024); }); diff --git a/.github/workflows/ci-regression-comment.yml b/.github/workflows/ci-regression-comment.yml index 6f96690b6f..266be2a457 100644 --- a/.github/workflows/ci-regression-comment.yml +++ b/.github/workflows/ci-regression-comment.yml @@ -37,7 +37,7 @@ jobs: uses: actions/github-script@f28e40c7f34bde8b3046d885e986cb6290c5673b # v7 env: ARTIFACT_NAMES: >- - ["release-regression-results","frontend-differential-results","agent-episode-results","runtime-contract-results","cdp-smoke-diagnostics"] + ["release-regression-results","frontend-differential-results","agent-episode-results","runtime-contract-results","cdp-smoke-diagnostics","webmainbench-results"] with: retries: 3 script: | @@ -99,6 +99,16 @@ jobs: github-token: ${{ secrets.GITHUB_TOKEN }} run-id: ${{ github.event.workflow_run.id }} + - name: Download WebMainBench evidence + if: contains(fromJSON(steps.wait.outputs.available_artifacts), 'webmainbench-results') + continue-on-error: true + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4 + with: + name: webmainbench-results + path: ci-regression-artifacts/webmainbench + github-token: ${{ secrets.GITHUB_TOKEN }} + run-id: ${{ github.event.workflow_run.id }} + - name: Install Node.js if: steps.wait.outputs.conclusion != 'cancelled' uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 @@ -117,6 +127,7 @@ jobs: --agent ci-regression-artifacts/agent --runtime ci-regression-artifacts/runtime --cdp ci-regression-artifacts/cdp + --webmainbench ci-regression-artifacts/webmainbench --run-url "$RUN_URL" --conclusion "$CONCLUSION" --output ci-regression-comment.md diff --git a/moli-benchmark/README.md b/moli-benchmark/README.md index ecbe3d9160..b9c76d2ed5 100644 --- a/moli-benchmark/README.md +++ b/moli-benchmark/README.md @@ -88,7 +88,14 @@ Use a new output directory for each run. On systems that disallow unprivileged user namespaces, use `sudo unshare --net -- env PYTHONPATH="$PYTHONPATH" python3 ...` as the CI job does. The runner refuses to run against an ordinary network. -The job summary reports the verdict and failure IDs. +The job summary and the existing PR `CI Regression Report` comment report the +verdict, completion/success counts, expected DNS failures, unexpected failures, +panics, crashes, timeouts, and empty outputs. The comment includes bounded +failure details and a link to the source run and artifacts. Missing or malformed +reports remain visibly unavailable. As with the other aggregate checks, comments +are updated for same-repository PRs by the trusted default-branch workflow; +this integration takes effect after these workflow and renderer changes reach +the default branch. The `webmainbench-results` artifact retains every page's Markdown, stderr, exit status, elapsed time, and HTML/output hashes for seven days, including when the