From 8a347002db46a29f5de5ac612a8df6000240b2df Mon Sep 17 00:00:00 2001 From: ldm0 Date: Fri, 18 Sep 2026 06:03:36 +0800 Subject: [PATCH] fix(dom): return standard XML parser error documents Replace partial XML trees with a single parsererror root in the namespace specified by HTML's DOMParser algorithm. Keep diagnostics as text and retain the requested content type and document metadata. XHR parsing continues to return no document on XML errors. Replace Chromium-specific error-wrapper assertions with coverage for empty, malformed and namespace-invalid input, prologues, doctypes, cross-realm documents, serialization and valid author-created parsererror elements. Validation: workspace fmt and strict all-feature Clippy; full nextest 19258 passed, 13 skipped; 22 WPT cases / 344 subtests; 1776 JS assertions. --- .../wpt-cross-current/failed-cases.txt | 1 - .../wpt-cross-current/passed-cases.txt | 1 + moli-renderer-v8/src/dom_parser.rs | 131 ++++++------------ .../script_vm/tests/dom_elements/detached.rs | 64 +++------ .../src/script_vm/tests/dom_xhr/dom.rs | 44 ++---- .../tests/fixtures/domparser-xml-errors.js | 90 ++++++++++++ 6 files changed, 159 insertions(+), 172 deletions(-) create mode 100644 moli-renderer-v8/tests/fixtures/domparser-xml-errors.js diff --git a/moli-benchmark/wpt-cross-current/failed-cases.txt b/moli-benchmark/wpt-cross-current/failed-cases.txt index 41c4a693a9..c2014dc280 100644 --- a/moli-benchmark/wpt-cross-current/failed-cases.txt +++ b/moli-benchmark/wpt-cross-current/failed-cases.txt @@ -2505,7 +2505,6 @@ dom/ranges/tentative/OpaqueRange-range-updates.html dom/ranges/tentative/OpaqueRange-supported-elements.html dom/ranges/tentative/OpaqueRange-unsupported-elements.html dom/ranges/tentative/OpaqueRange-validation.html -domparsing/DOMParser-parseFromString-xml.html domparsing/tentative/all-stream-methods-with-trusted-types-no-policy.html domparsing/tentative/positional-methods-with-trusted-types.html domparsing/tentative/positional-methods.html diff --git a/moli-benchmark/wpt-cross-current/passed-cases.txt b/moli-benchmark/wpt-cross-current/passed-cases.txt index ee306df2df..f982ddc629 100644 --- a/moli-benchmark/wpt-cross-current/passed-cases.txt +++ b/moli-benchmark/wpt-cross-current/passed-cases.txt @@ -4602,6 +4602,7 @@ domparsing/DOMParser-parseFromString-url.html domparsing/DOMParser-parseFromString-xml-doctype.html domparsing/DOMParser-parseFromString-xml-internal-subset.html domparsing/DOMParser-parseFromString-xml-parsererror.html +domparsing/DOMParser-parseFromString-xml.html domparsing/XMLSerializer-serializeToString.html domparsing/createContextualFragment-in-detached-xml-document-crash.html domparsing/createContextualFragment.html diff --git a/moli-renderer-v8/src/dom_parser.rs b/moli-renderer-v8/src/dom_parser.rs index a6fbcfb70a..f2f8b8fd68 100644 --- a/moli-renderer-v8/src/dom_parser.rs +++ b/moli-renderer-v8/src/dom_parser.rs @@ -6,7 +6,7 @@ use url::Url; use crate::{ document_runtime::DomHandle, - dom::native::{DomHost, NativeDom, NativeNodeId}, + dom::native::{DomHost, NativeDom}, parser::{HtmlParser, XmlParser}, webidl, }; @@ -27,9 +27,7 @@ use super::{ pub(crate) const DOM_PARSER_FOREIGN_NODE_SLOT: &str = "__moliDomParserForeignNode"; const DOM_PARSER_DOCUMENT_HANDLE_SLOT: &str = "__moliDomParserDocumentHandle"; -const HTML_NAMESPACE: &str = "http://www.w3.org/1999/xhtml"; -const PARSER_ERROR_STYLE: &str = "display: block; white-space: pre; border: 2px solid #c77; padding: 0 1em 0 1em; margin: 1em; background-color: #fdd; color: black"; -const PARSER_ERROR_DETAIL_STYLE: &str = "font-family:monospace;font-size:12px"; +const XML_PARSER_ERROR_NAMESPACE: &str = "http://www.mozilla.org/newlayout/xml/parsererror.xml"; #[derive(Clone, Copy)] pub(super) enum XmlParseErrorBehavior { @@ -309,102 +307,21 @@ fn materialize_xml_parser_error_document(parsed: NativeDom) -> NativeDom { .unwrap_or_else(|| "XML document has no document element".to_owned()); let mut host = DomHost::from_dom(parsed); let document = host.document_handle(); - let document_element = host - .child_handles(document) - .find(|handle| host.node(*handle).is_some_and(|node| node.is_element())); - - let parser_error = create_dom_parser_error_element(&mut host, document, &error_detail); - if let Some(document_element) = document_element { - let first_child = host - .node(document_element) - .and_then(|node| node.first_child()); - let _ = host.insert_before(document_element, parser_error, first_child); - return host.snapshot_document(); - } - + // DOMParser's XML error document contains only the error root, even if + // the underlying parser recovered a partial tree, doctype or prologue. for child in host.child_handles(document).collect::>() { let _ = host.remove_child(document, child); } - let html = host.create_parser_element_without_attributes_for_document( - document, - "html".to_owned(), - HTML_NAMESPACE.to_owned(), - None, - ); - let body = host.create_parser_element_without_attributes_for_document( - document, - "body".to_owned(), - HTML_NAMESPACE.to_owned(), - None, - ); - let _ = host.append_child(document, html); - let _ = host.append_child(html, body); - let _ = host.append_child(body, parser_error); - host.snapshot_document() -} - -fn create_dom_parser_error_element( - host: &mut DomHost, - document: NativeNodeId, - error_detail: &str, -) -> NativeNodeId { let parser_error = host.create_parser_element_without_attributes_for_document( document, "parsererror".to_owned(), - HTML_NAMESPACE.to_owned(), + XML_PARSER_ERROR_NAMESPACE.to_owned(), None, ); - let _ = host.set_attribute(parser_error, "style", PARSER_ERROR_STYLE); - - let heading = create_dom_parser_error_child(host, document, "h3", None); - append_dom_parser_error_text( - host, - document, - heading, - "This page contains the following errors:", - ); - let detail = - create_dom_parser_error_child(host, document, "div", Some(PARSER_ERROR_DETAIL_STYLE)); - append_dom_parser_error_text(host, document, detail, error_detail); - let footer = create_dom_parser_error_child(host, document, "h3", None); - append_dom_parser_error_text( - host, - document, - footer, - "Below is a rendering of the page up to the first error.", - ); - let _ = host.append_child(parser_error, heading); + let detail = host.create_text_node_for_document(document, &error_detail); let _ = host.append_child(parser_error, detail); - let _ = host.append_child(parser_error, footer); - parser_error -} - -fn create_dom_parser_error_child( - host: &mut DomHost, - document: NativeNodeId, - local_name: &str, - style: Option<&str>, -) -> NativeNodeId { - let element = host.create_parser_element_without_attributes_for_document( - document, - local_name.to_owned(), - HTML_NAMESPACE.to_owned(), - None, - ); - if let Some(style) = style { - let _ = host.set_attribute(element, "style", style); - } - element -} - -fn append_dom_parser_error_text( - host: &mut DomHost, - document: NativeNodeId, - parent: NativeNodeId, - text: &str, -) { - let text = host.create_text_node_for_document(document, text); - let _ = host.append_child(parent, text); + let _ = host.append_child(document, parser_error); + host.snapshot_document() } /// Builds a detached HTML document wrapper from raw markup and an explicit document URL. @@ -628,6 +545,38 @@ mod tests { .expect("document element") } + #[test] + fn xml_parser_error_document_discards_the_tree_and_preserves_error_text() { + let mut parsed = XmlParser.parse( + Url::parse("https://example.test/source.xml").unwrap(), + "" + .to_owned(), + ); + assert!(parsed.parse_errors().is_empty()); + let detail = "Unexpected '], + ['namespace', ''], + ['encoding-declaration', ''], + ]; + function metadata(doc, mime, label, realm) { + check(label + '/Document', doc instanceof realm.Document, true); + check(label + '/not-XMLDocument', doc instanceof realm.XMLDocument, false); + check(label + '/prototype', Object.getPrototypeOf(doc) === realm.Document.prototype, true); + check(label + '/URL', doc.URL, realm.document.URL); + check(label + '/documentURI', doc.documentURI, realm.document.URL); + check(label + '/baseURI', doc.baseURI, realm.document.URL); + check(label + '/contentType', doc.contentType, mime); + for (const property of ['characterSet', 'charset', 'inputEncoding']) check(label + '/' + property, doc[property], 'UTF-8'); + check(label + '/readyState', doc.readyState, 'complete'); + check(label + '/defaultView', doc.defaultView, null); + check(label + '/location', doc.location, null); + check(label + '/hidden', doc.hidden, true); + check(label + '/visibilityState', doc.visibilityState, 'hidden'); + check(label + '/compatMode', doc.compatMode, 'CSS1Compat'); + } + function errorDocument(realm, parser, source, mime, label) { + const doc = parser.parseFromString(source, mime); + metadata(doc, mime, label, realm); + const root = doc.documentElement; + check(label + '/namespace', root.namespaceURI, namespace); + check(label + '/localName', root.localName, 'parsererror'); + check(label + '/tagName', root.tagName, 'parsererror'); + check(label + '/prefix', root.prefix, null); + check(label + '/only-child', doc.childNodes.length, 1); + check(label + '/first-child', doc.firstChild === root, true); + check(label + '/parent', root.parentNode === doc, true); + check(label + '/owner', root.ownerDocument === doc, true); + check(label + '/doctype-removed', doc.doctype, null); + check(label + '/partial-tree-removed', doc.getElementById('original'), null); + check(label + '/error-count', doc.getElementsByTagName('parsererror').length, 1); + check(label + '/error-by-namespace', doc.getElementsByTagNameNS(namespace, 'parsererror')[0] === root, true); + check(label + '/error-description', root.textContent.length > 0, true); + check(label + '/body', doc.body, null); + check(label + '/head', doc.head, null); + check(label + '/script-inert', realm.domParserErrorScriptRan, undefined); + const serialized = new realm.XMLSerializer().serializeToString(doc); + check(label + '/serialized-root', serialized.startsWith('kept < & 💡', mime); + metadata(doc, mime, prefix, realm); + check(prefix + '/namespace', doc.documentElement.namespaceURI, authorNamespace); + check(prefix + '/marker', doc.documentElement.getAttribute('data-author'), 'yes'); + check(prefix + '/doctype', doc.doctype?.name, 'parsererror'); + check(prefix + '/child-namespace', doc.documentElement.firstElementChild.namespaceURI, 'urn:kept'); + check(prefix + '/text', doc.documentElement.textContent, 'kept < & 💡'); + } catch (error) { check(prefix + '/exception', String(error.stack || error), 'success'); } + } + } + } + run(globalThis, invalid, 'main'); + let frame; + try { + frame = document.body.appendChild(document.createElement('iframe')); + run(frame.contentWindow, [['mismatched', '']], 'child'); + } catch (error) { check('child/exception', String(error.stack || error), 'success'); } + finally { frame?.remove(); } + return {state: checks.every(item => item.pass) ? 'pass' : 'fail', checks}; +}