diff --git a/.github/scripts/render-ci-regression-comment.cjs b/.github/scripts/render-ci-regression-comment.cjs index 5e06dcf1cb..e4cfbd6b3f 100644 --- a/.github/scripts/render-ci-regression-comment.cjs +++ b/.github/scripts/render-ci-regression-comment.cjs @@ -10,6 +10,8 @@ const MAX_RESULT_ROWS = 2_000; const MAX_DETAIL_ROWS = 10; const MAX_MATRIX_ROWS = 5_000; const MAX_CDP_GROUPS = 100; +const WEBMAINBENCH_CASES = 545; +const WEBMAINBENCH_STATUSES = ['success', 'expected_failure', 'failure', 'panic', 'crash', 'timeout', 'empty_output']; const MIB = 1024 * 1024; function isObject(value) { @@ -639,6 +641,79 @@ function renderCdp(section) { return lines; } +function loadWebMainBench(root) { + const summary = readJson(root, 'summary.json'); + if ( + !isObject(summary) || + summary.schema_version !== 1 || + summary.expected_cases !== WEBMAINBENCH_CASES || + count(summary.completed) === null || + summary.completed > WEBMAINBENCH_CASES || + typeof summary.passed !== 'boolean' || + !isObject(summary.counts) || + !Array.isArray(summary.issues) || + summary.issues.length > WEBMAINBENCH_CASES + 2 || + summary.issues.some((issue) => typeof issue !== 'string') || + Object.entries(summary.counts).some(([status, value]) => + !WEBMAINBENCH_STATUSES.includes(status) || count(value) === null || value > WEBMAINBENCH_CASES + ) + ) { + throw new Error('invalid WebMainBench artifact'); + } + const counts = Object.fromEntries(WEBMAINBENCH_STATUSES.map((status) => [status, summary.counts[status] ?? 0])); + if (sum(Object.values(counts)) !== summary.completed || counts.expected_failure > 1) { + throw new Error('inconsistent WebMainBench counts'); + } + return { ...summary, counts }; +} + +function webMainBenchOverview(section) { + if (!section.available) { + return { status: '⚪', signal: 'artifact unavailable or invalid' }; + } + const { completed, counts, passed, issues } = section.data; + const unexpected = completed - counts.success - counts.expected_failure; + const ok = passed && completed === WEBMAINBENCH_CASES && unexpected === 0 && issues.length === 0; + return { + status: ok ? '✅' : '❌', + signal: `${formatInteger(completed)}/${WEBMAINBENCH_CASES} completed; ${formatInteger(unexpected)} unexpected failures`, + }; +} + +function renderWebMainBench(section, runLink) { + const overview = webMainBenchOverview(section); + const lines = [`
WebMainBench · 545 pages — ${overview.status} ${overview.signal}`, '']; + if (!section.available) { + lines.push('Artifact unavailable or invalid. See the source CI run for infrastructure details.', '', '
'); + return lines; + } + const { completed, counts, issues } = section.data; + const unexpected = completed - counts.success - counts.expected_failure; + lines.push( + '| Completed | Successful | Expected DNS failure | Unexpected failures | Panics | Crashes | Timeouts | Empty output |', + '| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |', + `| ${formatInteger(completed)}/${WEBMAINBENCH_CASES} | ${formatInteger(counts.success)} | ${formatInteger(counts.expected_failure)} | ${formatInteger(unexpected)} | ${formatInteger(counts.panic)} | ${formatInteger(counts.crash)} | ${formatInteger(counts.timeout)} | ${formatInteger(counts.empty_output)} |`, + '', + 'Frozen HTML · Linux · `--wait done` · 45 seconds per page · no retries. Content quality is evaluated separately.' + ); + if (issues.length !== 0) { + lines.push('', '**Failures and incomplete evidence**', ''); + for (const issue of issues.slice(0, MAX_DETAIL_ROWS)) { + lines.push(`- ${code(issue, 200)}`); + } + if (issues.length > MAX_DETAIL_ROWS) { + lines.push(`- ${formatInteger(issues.length - MAX_DETAIL_ROWS)} more issues in the artifact.`); + } + } + lines.push( + '', + `Diagnostics: \`webmainbench-results\` — every page's Markdown, stderr, and exit status.${runLink ? ` [Source run and artifacts](${runLink})` : ''}`, + '', + '' + ); + return lines; +} + function trustedRunUrl(value) { try { const url = new URL(value); @@ -651,13 +726,14 @@ function trustedRunUrl(value) { return null; } -function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, runUrl, conclusion }) { +function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, webmainbenchRoot, runUrl, conclusion }) { const sections = { release: loadSection(() => loadRelease(releaseRoot)), frontend: loadSection(() => loadFrontend(frontendRoot)), agent: loadSection(() => loadAgent(agentRoot)), runtime: loadSection(() => loadRuntime(runtimeRoot)), cdp: loadSection(() => loadCdp(cdpRoot)), + webmainbench: loadSection(() => loadWebMainBench(webmainbenchRoot)), }; const overviews = [ ['Release regression', releaseOverview(sections.release)], @@ -665,6 +741,7 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo ['Agent episodes', agentOverview(sections.agent)], ['Runtime/CDP contracts', runtimeOverview(sections.runtime)], ['CDP smoke', cdpOverview(sections.cdp)], + ['WebMainBench', webMainBenchOverview(sections.webmainbench)], ]; const available = Object.values(sections).filter((section) => section.available).length; const sourceConclusion = ['success', 'failure', 'cancelled', 'timed_out', 'in_progress'].includes(conclusion) @@ -675,7 +752,7 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo COMMENT_MARKER, '## CI Regression Report', '', - `${link ? `[Source CI run](${link})` : 'Source CI run'} · source state at render: \`${sourceConclusion}\` · artifacts: \`${available}/5\``, + `${link ? `[Source CI run](${link})` : 'Source CI run'} · source state at render: \`${sourceConclusion}\` · artifacts: \`${available}/${overviews.length}\``, '', '| Check | Status | Signal |', '| --- | :---: | --- |', @@ -695,6 +772,8 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo '', ...renderCdp(sections.cdp), '', + ...renderWebMainBench(sections.webmainbench, link), + '', '_All artifact fields are parsed by the trusted default-branch renderer; missing or invalid inputs remain visible as unavailable._', '' ); @@ -729,6 +808,7 @@ function main(argv = process.argv.slice(2)) { agentRoot: args.agent, runtimeRoot: args.runtime, cdpRoot: args.cdp, + webmainbenchRoot: args.webmainbench, runUrl: args['run-url'], conclusion: args.conclusion, }); diff --git a/.github/scripts/render-ci-regression-comment.test.cjs b/.github/scripts/render-ci-regression-comment.test.cjs index 4ef1765c35..c2269331b8 100644 --- a/.github/scripts/render-ci-regression-comment.test.cjs +++ b/.github/scripts/render-ci-regression-comment.test.cjs @@ -104,6 +104,18 @@ function agentTarget({ chrome = false, failure = false }) { }; } +function webMainBenchSummary(overrides = {}) { + return { + schema_version: 1, + expected_cases: 545, + completed: 545, + counts: { success: 544, expected_failure: 1 }, + passed: true, + issues: [], + ...overrides, + }; +} + function createArtifacts(root) { const releaseRoot = path.join(root, 'release'); writeJson(releaseRoot, 'base-startup/startup/summary.json', startupSummary(0)); @@ -190,10 +202,13 @@ function createArtifacts(root) { ], }); - return { releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot }; + const webmainbenchRoot = path.join(root, 'webmainbench'); + writeJson(webmainbenchRoot, 'summary.json', webMainBenchSummary()); + + return { releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, webmainbenchRoot }; } -test('renders all five trusted artifact sections into one bounded report', (t) => { +test('renders all six trusted artifact sections into one bounded report', (t) => { const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-report-')); t.after(() => fs.rmSync(root, { recursive: true, force: true })); const roots = createArtifacts(root); @@ -204,7 +219,7 @@ test('renders all five trusted artifact sections into one bounded report', (t) = }); assert.ok(report.startsWith(COMMENT_MARKER)); - assert.match(report, /artifacts: `5\/5`/); + assert.match(report, /artifacts: `6\/6`/); assert.match(report, /Raw binary \| 100 B \| 110 B \| \+10 B \| \+10\.000000%/); assert.match(report, /Frontend differential/); assert.match(report, /1 Chromium reference recoveries/); @@ -218,6 +233,10 @@ test('renders all five trusted artifact sections into one bounded report', (t) = assert.match(report, /PSS partial/); assert.match(report, /Runtime and CDP session contracts/); assert.match(report, /`bad\\\|group<`/); + assert.match(report, /\| WebMainBench \| ✅ \| 545\/545 completed; 0 unexpected failures \|/); + assert.match(report, /\| 545\/545 \| 544 \| 1 \| 0 \| 0 \| 0 \| 0 \| 0 \|/); + assert.match(report, /\[Source run and artifacts\]\(https:\/\/github\.com\/lexmount\/moli\/actions\/runs\/123\)/); + assert.match(report, /Diagnostics: `webmainbench-results`/); assert.ok(Buffer.byteLength(report, 'utf8') < 32 * 1024); }); @@ -227,8 +246,8 @@ test('renders missing artifacts as unavailable without throwing', () => { conclusion: 'timed_out', }); - assert.match(report, /artifacts: `0\/5`/); - assert.equal((report.match(/Artifact unavailable or invalid\./g) || []).length, 5); + assert.match(report, /artifacts: `0\/6`/); + assert.equal((report.match(/Artifact unavailable or invalid\./g) || []).length, 6); assert.doesNotMatch(report, /\]\(not-a-trusted-url\)/); }); @@ -246,5 +265,84 @@ test('rejects an unbounded frontend result list as unavailable', (t) => { }); const report = renderReport({ frontendRoot, conclusion: 'failure' }); - assert.match(report, /artifacts: `0\/5`/); + assert.match(report, /artifacts: `0\/6`/); +}); + +test('shows WebMainBench failures and escapes issue text', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-failures-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + writeJson(root, 'summary.json', webMainBenchSummary({ + passed: false, + counts: { success: 540, expected_failure: 1, panic: 1, crash: 1, timeout: 1, empty_output: 1 }, + issues: ['first-case: panic', 'second-case: timeout', 'bad|case<`\n: empty_output'], + })); + + const report = renderReport({ webmainbenchRoot: root, conclusion: 'failure' }); + + assert.match(report, /\| WebMainBench \| ❌ \| 545\/545 completed; 4 unexpected failures \|/); + assert.match(report, /\| 545\/545 \| 540 \| 1 \| 4 \| 1 \| 1 \| 1 \| 1 \|/); + assert.match(report, /`first-case: panic`/); + assert.match(report, /bad\\\|case<' <img src=x>/); + assert.doesNotMatch(report, //); +}); + +test('does not display a green WebMainBench result for incomplete or failed evidence', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-incomplete-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + for (const summary of [ + webMainBenchSummary({ completed: 200, counts: { success: 200 } }), + webMainBenchSummary({ passed: false }), + webMainBenchSummary({ issues: ['Infrastructure error: fixture failed'] }), + ]) { + writeJson(root, 'summary.json', summary); + const report = renderReport({ webmainbenchRoot: root, conclusion: 'success' }); + assert.match(report, /artifacts: `1\/6`/); + assert.match(report, /\| WebMainBench \| ❌ \|/); + assert.doesNotMatch(report, /\| WebMainBench \| ✅ \|/); + } +}); + +test('rejects malformed WebMainBench counts and oversized issue lists', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-invalid-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + for (const overrides of [ + { schema_version: 2 }, + { expected_cases: 1 }, + { completed: 546 }, + { counts: { success: '544', expected_failure: 1 } }, + { counts: { success: -1 } }, + { counts: { success: 544, unknown_status: 1 } }, + { counts: { success: 543, expected_failure: 2 } }, + { counts: {} }, + { issues: [{}] }, + { issues: Array.from({ length: 548 }, () => 'failure') }, + ]) { + writeJson(root, 'summary.json', webMainBenchSummary(overrides)); + const report = renderReport({ webmainbenchRoot: root, conclusion: 'success' }); + assert.match(report, /artifacts: `0\/6`/, JSON.stringify(overrides)); + assert.match(report, /\| WebMainBench \| ⚪ \| artifact unavailable or invalid \|/); + } +}); + +test('bounds WebMainBench issue details and does not use an untrusted diagnostics URL', (t) => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-bounded-')); + t.after(() => fs.rmSync(root, { recursive: true, force: true })); + writeJson(root, 'summary.json', webMainBenchSummary({ + counts: { failure: 545 }, + passed: false, + issues: Array.from({ length: 545 }, (_, index) => `case-${index}: ${'x'.repeat(1_000)}`), + diagnostics_url: 'https://attacker.example/report', + })); + + const report = renderReport({ + webmainbenchRoot: root, + runUrl: 'javascript:alert(1)', + conclusion: 'failure', + }); + + assert.match(report, /535 more issues in the artifact/); + assert.match(report, /`case-9:/); + assert.doesNotMatch(report, /`case-10:/); + assert.doesNotMatch(report, /attacker\.example|javascript:|\[Source run and artifacts\]/); + assert.ok(Buffer.byteLength(report, 'utf8') < 32 * 1024); }); diff --git a/.github/workflows/ci-regression-comment.yml b/.github/workflows/ci-regression-comment.yml index 6f96690b6f..266be2a457 100644 --- a/.github/workflows/ci-regression-comment.yml +++ b/.github/workflows/ci-regression-comment.yml @@ -37,7 +37,7 @@ jobs: uses: actions/github-script@f28e40c7f34bde8b3046d885e986cb6290c5673b # v7 env: ARTIFACT_NAMES: >- - ["release-regression-results","frontend-differential-results","agent-episode-results","runtime-contract-results","cdp-smoke-diagnostics"] + ["release-regression-results","frontend-differential-results","agent-episode-results","runtime-contract-results","cdp-smoke-diagnostics","webmainbench-results"] with: retries: 3 script: | @@ -99,6 +99,16 @@ jobs: github-token: ${{ secrets.GITHUB_TOKEN }} run-id: ${{ github.event.workflow_run.id }} + - name: Download WebMainBench evidence + if: contains(fromJSON(steps.wait.outputs.available_artifacts), 'webmainbench-results') + continue-on-error: true + uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4 + with: + name: webmainbench-results + path: ci-regression-artifacts/webmainbench + github-token: ${{ secrets.GITHUB_TOKEN }} + run-id: ${{ github.event.workflow_run.id }} + - name: Install Node.js if: steps.wait.outputs.conclusion != 'cancelled' uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 @@ -117,6 +127,7 @@ jobs: --agent ci-regression-artifacts/agent --runtime ci-regression-artifacts/runtime --cdp ci-regression-artifacts/cdp + --webmainbench ci-regression-artifacts/webmainbench --run-url "$RUN_URL" --conclusion "$CONCLUSION" --output ci-regression-comment.md diff --git a/moli-benchmark/README.md b/moli-benchmark/README.md index ecbe3d9160..b9c76d2ed5 100644 --- a/moli-benchmark/README.md +++ b/moli-benchmark/README.md @@ -88,7 +88,14 @@ Use a new output directory for each run. On systems that disallow unprivileged user namespaces, use `sudo unshare --net -- env PYTHONPATH="$PYTHONPATH" python3 ...` as the CI job does. The runner refuses to run against an ordinary network. -The job summary reports the verdict and failure IDs. +The job summary and the existing PR `CI Regression Report` comment report the +verdict, completion/success counts, expected DNS failures, unexpected failures, +panics, crashes, timeouts, and empty outputs. The comment includes bounded +failure details and a link to the source run and artifacts. Missing or malformed +reports remain visibly unavailable. As with the other aggregate checks, comments +are updated for same-repository PRs by the trusted default-branch workflow; +this integration takes effect after these workflow and renderer changes reach +the default branch. The `webmainbench-results` artifact retains every page's Markdown, stderr, exit status, elapsed time, and HTML/output hashes for seven days, including when the