diff --git a/.github/scripts/render-ci-regression-comment.cjs b/.github/scripts/render-ci-regression-comment.cjs
index 5e06dcf1cb..e4cfbd6b3f 100644
--- a/.github/scripts/render-ci-regression-comment.cjs
+++ b/.github/scripts/render-ci-regression-comment.cjs
@@ -10,6 +10,8 @@ const MAX_RESULT_ROWS = 2_000;
const MAX_DETAIL_ROWS = 10;
const MAX_MATRIX_ROWS = 5_000;
const MAX_CDP_GROUPS = 100;
+const WEBMAINBENCH_CASES = 545;
+const WEBMAINBENCH_STATUSES = ['success', 'expected_failure', 'failure', 'panic', 'crash', 'timeout', 'empty_output'];
const MIB = 1024 * 1024;
function isObject(value) {
@@ -639,6 +641,79 @@ function renderCdp(section) {
return lines;
}
+function loadWebMainBench(root) {
+ const summary = readJson(root, 'summary.json');
+ if (
+ !isObject(summary) ||
+ summary.schema_version !== 1 ||
+ summary.expected_cases !== WEBMAINBENCH_CASES ||
+ count(summary.completed) === null ||
+ summary.completed > WEBMAINBENCH_CASES ||
+ typeof summary.passed !== 'boolean' ||
+ !isObject(summary.counts) ||
+ !Array.isArray(summary.issues) ||
+ summary.issues.length > WEBMAINBENCH_CASES + 2 ||
+ summary.issues.some((issue) => typeof issue !== 'string') ||
+ Object.entries(summary.counts).some(([status, value]) =>
+ !WEBMAINBENCH_STATUSES.includes(status) || count(value) === null || value > WEBMAINBENCH_CASES
+ )
+ ) {
+ throw new Error('invalid WebMainBench artifact');
+ }
+ const counts = Object.fromEntries(WEBMAINBENCH_STATUSES.map((status) => [status, summary.counts[status] ?? 0]));
+ if (sum(Object.values(counts)) !== summary.completed || counts.expected_failure > 1) {
+ throw new Error('inconsistent WebMainBench counts');
+ }
+ return { ...summary, counts };
+}
+
+function webMainBenchOverview(section) {
+ if (!section.available) {
+ return { status: '⚪', signal: 'artifact unavailable or invalid' };
+ }
+ const { completed, counts, passed, issues } = section.data;
+ const unexpected = completed - counts.success - counts.expected_failure;
+ const ok = passed && completed === WEBMAINBENCH_CASES && unexpected === 0 && issues.length === 0;
+ return {
+ status: ok ? '✅' : '❌',
+ signal: `${formatInteger(completed)}/${WEBMAINBENCH_CASES} completed; ${formatInteger(unexpected)} unexpected failures`,
+ };
+}
+
+function renderWebMainBench(section, runLink) {
+ const overview = webMainBenchOverview(section);
+ const lines = [`WebMainBench · 545 pages — ${overview.status} ${overview.signal}
`, ''];
+ if (!section.available) {
+ lines.push('Artifact unavailable or invalid. See the source CI run for infrastructure details.', '', ' ');
+ return lines;
+ }
+ const { completed, counts, issues } = section.data;
+ const unexpected = completed - counts.success - counts.expected_failure;
+ lines.push(
+ '| Completed | Successful | Expected DNS failure | Unexpected failures | Panics | Crashes | Timeouts | Empty output |',
+ '| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |',
+ `| ${formatInteger(completed)}/${WEBMAINBENCH_CASES} | ${formatInteger(counts.success)} | ${formatInteger(counts.expected_failure)} | ${formatInteger(unexpected)} | ${formatInteger(counts.panic)} | ${formatInteger(counts.crash)} | ${formatInteger(counts.timeout)} | ${formatInteger(counts.empty_output)} |`,
+ '',
+ 'Frozen HTML · Linux · `--wait done` · 45 seconds per page · no retries. Content quality is evaluated separately.'
+ );
+ if (issues.length !== 0) {
+ lines.push('', '**Failures and incomplete evidence**', '');
+ for (const issue of issues.slice(0, MAX_DETAIL_ROWS)) {
+ lines.push(`- ${code(issue, 200)}`);
+ }
+ if (issues.length > MAX_DETAIL_ROWS) {
+ lines.push(`- ${formatInteger(issues.length - MAX_DETAIL_ROWS)} more issues in the artifact.`);
+ }
+ }
+ lines.push(
+ '',
+ `Diagnostics: \`webmainbench-results\` — every page's Markdown, stderr, and exit status.${runLink ? ` [Source run and artifacts](${runLink})` : ''}`,
+ '',
+ ''
+ );
+ return lines;
+}
+
function trustedRunUrl(value) {
try {
const url = new URL(value);
@@ -651,13 +726,14 @@ function trustedRunUrl(value) {
return null;
}
-function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, runUrl, conclusion }) {
+function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, webmainbenchRoot, runUrl, conclusion }) {
const sections = {
release: loadSection(() => loadRelease(releaseRoot)),
frontend: loadSection(() => loadFrontend(frontendRoot)),
agent: loadSection(() => loadAgent(agentRoot)),
runtime: loadSection(() => loadRuntime(runtimeRoot)),
cdp: loadSection(() => loadCdp(cdpRoot)),
+ webmainbench: loadSection(() => loadWebMainBench(webmainbenchRoot)),
};
const overviews = [
['Release regression', releaseOverview(sections.release)],
@@ -665,6 +741,7 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo
['Agent episodes', agentOverview(sections.agent)],
['Runtime/CDP contracts', runtimeOverview(sections.runtime)],
['CDP smoke', cdpOverview(sections.cdp)],
+ ['WebMainBench', webMainBenchOverview(sections.webmainbench)],
];
const available = Object.values(sections).filter((section) => section.available).length;
const sourceConclusion = ['success', 'failure', 'cancelled', 'timed_out', 'in_progress'].includes(conclusion)
@@ -675,7 +752,7 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo
COMMENT_MARKER,
'## CI Regression Report',
'',
- `${link ? `[Source CI run](${link})` : 'Source CI run'} · source state at render: \`${sourceConclusion}\` · artifacts: \`${available}/5\``,
+ `${link ? `[Source CI run](${link})` : 'Source CI run'} · source state at render: \`${sourceConclusion}\` · artifacts: \`${available}/${overviews.length}\``,
'',
'| Check | Status | Signal |',
'| --- | :---: | --- |',
@@ -695,6 +772,8 @@ function renderReport({ releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRo
'',
...renderCdp(sections.cdp),
'',
+ ...renderWebMainBench(sections.webmainbench, link),
+ '',
'_All artifact fields are parsed by the trusted default-branch renderer; missing or invalid inputs remain visible as unavailable._',
''
);
@@ -729,6 +808,7 @@ function main(argv = process.argv.slice(2)) {
agentRoot: args.agent,
runtimeRoot: args.runtime,
cdpRoot: args.cdp,
+ webmainbenchRoot: args.webmainbench,
runUrl: args['run-url'],
conclusion: args.conclusion,
});
diff --git a/.github/scripts/render-ci-regression-comment.test.cjs b/.github/scripts/render-ci-regression-comment.test.cjs
index 4ef1765c35..c2269331b8 100644
--- a/.github/scripts/render-ci-regression-comment.test.cjs
+++ b/.github/scripts/render-ci-regression-comment.test.cjs
@@ -104,6 +104,18 @@ function agentTarget({ chrome = false, failure = false }) {
};
}
+function webMainBenchSummary(overrides = {}) {
+ return {
+ schema_version: 1,
+ expected_cases: 545,
+ completed: 545,
+ counts: { success: 544, expected_failure: 1 },
+ passed: true,
+ issues: [],
+ ...overrides,
+ };
+}
+
function createArtifacts(root) {
const releaseRoot = path.join(root, 'release');
writeJson(releaseRoot, 'base-startup/startup/summary.json', startupSummary(0));
@@ -190,10 +202,13 @@ function createArtifacts(root) {
],
});
- return { releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot };
+ const webmainbenchRoot = path.join(root, 'webmainbench');
+ writeJson(webmainbenchRoot, 'summary.json', webMainBenchSummary());
+
+ return { releaseRoot, frontendRoot, agentRoot, runtimeRoot, cdpRoot, webmainbenchRoot };
}
-test('renders all five trusted artifact sections into one bounded report', (t) => {
+test('renders all six trusted artifact sections into one bounded report', (t) => {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-report-'));
t.after(() => fs.rmSync(root, { recursive: true, force: true }));
const roots = createArtifacts(root);
@@ -204,7 +219,7 @@ test('renders all five trusted artifact sections into one bounded report', (t) =
});
assert.ok(report.startsWith(COMMENT_MARKER));
- assert.match(report, /artifacts: `5\/5`/);
+ assert.match(report, /artifacts: `6\/6`/);
assert.match(report, /Raw binary \| 100 B \| 110 B \| \+10 B \| \+10\.000000%/);
assert.match(report, /Frontend differential/);
assert.match(report, /1 Chromium reference recoveries/);
@@ -218,6 +233,10 @@ test('renders all five trusted artifact sections into one bounded report', (t) =
assert.match(report, /PSS partial/);
assert.match(report, /Runtime and CDP session contracts/);
assert.match(report, /`bad\\\|group<`/);
+ assert.match(report, /\| WebMainBench \| ✅ \| 545\/545 completed; 0 unexpected failures \|/);
+ assert.match(report, /\| 545\/545 \| 544 \| 1 \| 0 \| 0 \| 0 \| 0 \| 0 \|/);
+ assert.match(report, /\[Source run and artifacts\]\(https:\/\/github\.com\/lexmount\/moli\/actions\/runs\/123\)/);
+ assert.match(report, /Diagnostics: `webmainbench-results`/);
assert.ok(Buffer.byteLength(report, 'utf8') < 32 * 1024);
});
@@ -227,8 +246,8 @@ test('renders missing artifacts as unavailable without throwing', () => {
conclusion: 'timed_out',
});
- assert.match(report, /artifacts: `0\/5`/);
- assert.equal((report.match(/Artifact unavailable or invalid\./g) || []).length, 5);
+ assert.match(report, /artifacts: `0\/6`/);
+ assert.equal((report.match(/Artifact unavailable or invalid\./g) || []).length, 6);
assert.doesNotMatch(report, /\]\(not-a-trusted-url\)/);
});
@@ -246,5 +265,84 @@ test('rejects an unbounded frontend result list as unavailable', (t) => {
});
const report = renderReport({ frontendRoot, conclusion: 'failure' });
- assert.match(report, /artifacts: `0\/5`/);
+ assert.match(report, /artifacts: `0\/6`/);
+});
+
+test('shows WebMainBench failures and escapes issue text', (t) => {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-failures-'));
+ t.after(() => fs.rmSync(root, { recursive: true, force: true }));
+ writeJson(root, 'summary.json', webMainBenchSummary({
+ passed: false,
+ counts: { success: 540, expected_failure: 1, panic: 1, crash: 1, timeout: 1, empty_output: 1 },
+ issues: ['first-case: panic', 'second-case: timeout', 'bad|case<`\n
: empty_output'],
+ }));
+
+ const report = renderReport({ webmainbenchRoot: root, conclusion: 'failure' });
+
+ assert.match(report, /\| WebMainBench \| ❌ \| 545\/545 completed; 4 unexpected failures \|/);
+ assert.match(report, /\| 545\/545 \| 540 \| 1 \| 4 \| 1 \| 1 \| 1 \| 1 \|/);
+ assert.match(report, /`first-case: panic`/);
+ assert.match(report, /bad\\\|case<' <img src=x>/);
+ assert.doesNotMatch(report, /
/);
+});
+
+test('does not display a green WebMainBench result for incomplete or failed evidence', (t) => {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-incomplete-'));
+ t.after(() => fs.rmSync(root, { recursive: true, force: true }));
+ for (const summary of [
+ webMainBenchSummary({ completed: 200, counts: { success: 200 } }),
+ webMainBenchSummary({ passed: false }),
+ webMainBenchSummary({ issues: ['Infrastructure error: fixture failed'] }),
+ ]) {
+ writeJson(root, 'summary.json', summary);
+ const report = renderReport({ webmainbenchRoot: root, conclusion: 'success' });
+ assert.match(report, /artifacts: `1\/6`/);
+ assert.match(report, /\| WebMainBench \| ❌ \|/);
+ assert.doesNotMatch(report, /\| WebMainBench \| ✅ \|/);
+ }
+});
+
+test('rejects malformed WebMainBench counts and oversized issue lists', (t) => {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-invalid-'));
+ t.after(() => fs.rmSync(root, { recursive: true, force: true }));
+ for (const overrides of [
+ { schema_version: 2 },
+ { expected_cases: 1 },
+ { completed: 546 },
+ { counts: { success: '544', expected_failure: 1 } },
+ { counts: { success: -1 } },
+ { counts: { success: 544, unknown_status: 1 } },
+ { counts: { success: 543, expected_failure: 2 } },
+ { counts: {} },
+ { issues: [{}] },
+ { issues: Array.from({ length: 548 }, () => 'failure') },
+ ]) {
+ writeJson(root, 'summary.json', webMainBenchSummary(overrides));
+ const report = renderReport({ webmainbenchRoot: root, conclusion: 'success' });
+ assert.match(report, /artifacts: `0\/6`/, JSON.stringify(overrides));
+ assert.match(report, /\| WebMainBench \| ⚪ \| artifact unavailable or invalid \|/);
+ }
+});
+
+test('bounds WebMainBench issue details and does not use an untrusted diagnostics URL', (t) => {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'moli-ci-webmainbench-bounded-'));
+ t.after(() => fs.rmSync(root, { recursive: true, force: true }));
+ writeJson(root, 'summary.json', webMainBenchSummary({
+ counts: { failure: 545 },
+ passed: false,
+ issues: Array.from({ length: 545 }, (_, index) => `case-${index}: ${'x'.repeat(1_000)}`),
+ diagnostics_url: 'https://attacker.example/report',
+ }));
+
+ const report = renderReport({
+ webmainbenchRoot: root,
+ runUrl: 'javascript:alert(1)',
+ conclusion: 'failure',
+ });
+
+ assert.match(report, /535 more issues in the artifact/);
+ assert.match(report, /`case-9:/);
+ assert.doesNotMatch(report, /`case-10:/);
+ assert.doesNotMatch(report, /attacker\.example|javascript:|\[Source run and artifacts\]/);
+ assert.ok(Buffer.byteLength(report, 'utf8') < 32 * 1024);
});
diff --git a/.github/workflows/ci-regression-comment.yml b/.github/workflows/ci-regression-comment.yml
index 6f96690b6f..266be2a457 100644
--- a/.github/workflows/ci-regression-comment.yml
+++ b/.github/workflows/ci-regression-comment.yml
@@ -37,7 +37,7 @@ jobs:
uses: actions/github-script@f28e40c7f34bde8b3046d885e986cb6290c5673b # v7
env:
ARTIFACT_NAMES: >-
- ["release-regression-results","frontend-differential-results","agent-episode-results","runtime-contract-results","cdp-smoke-diagnostics"]
+ ["release-regression-results","frontend-differential-results","agent-episode-results","runtime-contract-results","cdp-smoke-diagnostics","webmainbench-results"]
with:
retries: 3
script: |
@@ -99,6 +99,16 @@ jobs:
github-token: ${{ secrets.GITHUB_TOKEN }}
run-id: ${{ github.event.workflow_run.id }}
+ - name: Download WebMainBench evidence
+ if: contains(fromJSON(steps.wait.outputs.available_artifacts), 'webmainbench-results')
+ continue-on-error: true
+ uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
+ with:
+ name: webmainbench-results
+ path: ci-regression-artifacts/webmainbench
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+ run-id: ${{ github.event.workflow_run.id }}
+
- name: Install Node.js
if: steps.wait.outputs.conclusion != 'cancelled'
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
@@ -117,6 +127,7 @@ jobs:
--agent ci-regression-artifacts/agent
--runtime ci-regression-artifacts/runtime
--cdp ci-regression-artifacts/cdp
+ --webmainbench ci-regression-artifacts/webmainbench
--run-url "$RUN_URL"
--conclusion "$CONCLUSION"
--output ci-regression-comment.md
diff --git a/moli-benchmark/README.md b/moli-benchmark/README.md
index ecbe3d9160..b9c76d2ed5 100644
--- a/moli-benchmark/README.md
+++ b/moli-benchmark/README.md
@@ -88,7 +88,14 @@ Use a new output directory for each run. On systems that disallow unprivileged
user namespaces, use `sudo unshare --net -- env PYTHONPATH="$PYTHONPATH" python3 ...`
as the CI job does. The runner refuses to run against an ordinary network.
-The job summary reports the verdict and failure IDs.
+The job summary and the existing PR `CI Regression Report` comment report the
+verdict, completion/success counts, expected DNS failures, unexpected failures,
+panics, crashes, timeouts, and empty outputs. The comment includes bounded
+failure details and a link to the source run and artifacts. Missing or malformed
+reports remain visibly unavailable. As with the other aggregate checks, comments
+are updated for same-repository PRs by the trusted default-branch workflow;
+this integration takes effect after these workflow and renderer changes reach
+the default branch.
The `webmainbench-results` artifact retains every page's Markdown, stderr, exit
status, elapsed time, and HTML/output hashes for seven days, including when the