fix(html): materialize plain-text child documents

This commit is contained in:
ldm0
2026-09-09 06:52:20 +08:00
parent 6e92015fa2
commit 33ced89d7e
4 changed files with 134 additions and 0 deletions
+62
View File
@@ -1,3 +1,4 @@
use html5ever::tree_builder::QuirksMode;
use moli_web_mime::{is_dom_parser_xml_mime, is_html_document_mime};
use moli_webapi_declare::WebApiFunctionTemplate;
use url::Url;
@@ -483,12 +484,35 @@ fn parse_browsing_context_document_snapshot(
DetachedDocumentKind::Xml,
);
}
if content_type.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain")) {
let mut document =
html_parser.parse_dom_host(document_url, plain_text_document_parser_input(source));
// Text documents are HTML Documents whose mode is explicitly no-quirks,
// despite having no doctype that would select that mode through parsing.
document.set_html_quirks_mode_for_parser(QuirksMode::NoQuirks);
return (document, DetachedDocumentKind::Html);
}
(
html_parser.parse_dom_host(document_url, source.to_owned()),
DetachedDocumentKind::Html,
)
}
pub(crate) fn plain_text_document_parser_input(source: &str) -> String {
let mut input = String::with_capacity(source.len().saturating_add(64));
input.push_str("<html><head></head><body><pre>");
for character in source.chars() {
match character {
'&' => input.push_str("&amp;"),
'<' => input.push_str("&lt;"),
'\0' => input.push('\u{fffd}'),
_ => input.push(character),
}
}
input.push_str("</pre></body></html>");
input
}
fn child_document_url_is_xml_like(url: &Url) -> bool {
let path = url.path().to_ascii_lowercase();
path.ends_with(".xml") || path.ends_with(".xhtml") || path.ends_with(".svg")
@@ -656,4 +680,42 @@ mod tests {
let disabled = parse(HtmlParser::SCRIPTING_DISABLED);
assert!(disabled.element_handle_by_id("fallback").is_some());
}
#[test]
fn child_plain_text_document_uses_pre_and_no_quirks_mode() {
let (document, kind) = parse_browsing_context_document_snapshot(
Url::parse("https://example.test/sample.txt").expect("test URL"),
"alpha<&amp;\r\nbeta\rgamma\0",
Some("text/plain"),
HtmlParser::SCRIPTING_ENABLED,
);
let document_handle = document.document_handle();
let document_children = document.child_handles(document_handle).collect::<Vec<_>>();
assert_eq!(kind, DetachedDocumentKind::Html);
assert_eq!(document_children.len(), 1);
assert!(document_children.iter().all(|child| {
document
.node(*child)
.is_none_or(|node| node.as_document_type().is_none())
}));
assert_eq!(
document.document_quirks_mode_for_handle(document_handle),
Some(selectors::matching::QuirksMode::NoQuirks)
);
let html = document_children[0];
let html_children = document.child_handles(html).collect::<Vec<_>>();
assert_eq!(html_children.len(), 2);
assert!(document.is_html_element_named(html_children[0], "head"));
assert!(document.is_html_element_named(html_children[1], "body"));
let body_children = document.child_handles(html_children[1]).collect::<Vec<_>>();
assert_eq!(body_children.len(), 1);
assert!(document.is_html_element_named(body_children[0], "pre"));
assert_eq!(
document.text_content(body_children[0]).as_deref(),
Some("alpha<&amp;\nbeta\ngamma\u{fffd}")
);
}
}
@@ -170,8 +170,16 @@ impl JsContextHost {
navigation_loader: Option<crate::network::navigation::NavigationResourceLoader>,
is_xml_document: bool,
) -> Option<ChildDocumentInstallResult> {
let is_plain_text_document = snapshot
.content_type
.as_deref()
.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain"));
let source = if is_xml_document {
std::borrow::Cow::Borrowed(snapshot.markup.as_str())
} else if is_plain_text_document {
std::borrow::Cow::Owned(crate::dom_parser::plain_text_document_parser_input(
&snapshot.markup,
))
} else {
crate::dom_parser::preserve_decoded_bom_only_browsing_context_body(
&snapshot.markup,
@@ -917,6 +917,14 @@ impl JsContextHost {
let _ = parser.admit_delayed_finish_at_local_owner_boundary();
let finish_signals =
self.with_live_child_parser_step(scope, document_handle, |owner| parser.finish(owner));
if self
.dom_host()
.document_content_type_for_handle(document_handle)
.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain"))
{
self.dom_host_mut()
.set_html_quirks_mode_for_parser_document(document_handle, QuirksMode::NoQuirks);
}
let host_ptr = self as *mut JsContextHost;
self.run_pending_child_parser_post_step_runtime_work(scope, host_ptr);
self.queue_live_child_parser_discovery_signals(
@@ -610,6 +610,62 @@ fn child_document_content_type_uses_resource_mime() {
assert_eq!(result, "text/html|image/png");
}
#[test]
fn child_plain_text_document_uses_pre_and_standards_mode() {
let mut vm = new_storage_test_vm("https://child-plain-text-document.test/");
vm.eval(
r#"
(() => {
const root = document.documentElement || document.appendChild(document.createElement('html'));
const body = document.body || root.appendChild(document.createElement('body'));
const frame = document.createElement('iframe');
frame.id = 'plain';
frame.src = 'data:text/plain,alpha%3Cb%3E%26amp%3B%3C%2Fb%3E%0D%0Abeta';
body.appendChild(frame);
})()
"#,
)
.expect("plain-text child document setup should evaluate");
vm.drain_pending_child_frame_work_for_test();
let child_context_id = vm
.live_child_default_runtime_realm_inventory()
.into_iter()
.map(|realm| realm.context_id)
.next()
.expect("plain-text child default realm should materialize");
let result = vm
.eval_in_child_default_context(
child_context_id,
r#"
(() => {
const doc = document;
const pre = doc.body.firstChild;
return [
doc.compatMode,
doc.contentType,
doc.doctype === null,
doc.childNodes.length,
doc.documentElement.children.length,
doc.head.tagName,
doc.body.tagName,
doc.body.childNodes.length,
pre.tagName,
pre.children.length,
pre.firstChild.data
].join('|');
})()
"#,
)
.expect("plain-text child document should evaluate");
assert_eq!(
result,
"CSS1Compat|text/plain|true|1|2|HEAD|BODY|1|PRE|0|alpha<b>&amp;</b>\nbeta"
);
}
#[test]
fn node_move_before_parent_node_surface_and_validation() {
let mut vm = new_storage_test_vm("https://node-move-before.test/");