fix(document): share literal text parsing across child document paths

This commit is contained in:
lanyue-llk
2026-09-29 18:01:12 +08:00
committed by Donough Liu
parent 9d236a20ef
commit 345ba4cbc4
6 changed files with 124 additions and 43 deletions
+38
View File
@@ -1937,6 +1937,44 @@ mod tests {
}
}
#[test]
fn text_document_shell_has_no_doctype_and_preserves_normalized_literal_text() {
for mime in [
"text/plain",
"application/json",
"application/problem+json",
"text/javascript",
] {
let stream = HtmlParser::SCRIPTING_ENABLED
.start_text_document(Url::parse("https://example.test/data").unwrap(), mime);
for chunk in [
"\n<&amp;",
"\r",
"\nbeta\rgamma\0",
"</pre><script>bad()</script>",
] {
stream.feed(chunk);
}
let document = stream.finish_dom_host();
let root = document.document_handle();
let children = document.child_handles(root).collect::<Vec<_>>();
assert_eq!(children.len(), 1, "no synthetic doctype: {mime}");
assert!(document.is_html_element_named(children[0], "html"));
let expected_mode = HtmlParser::SCRIPTING_ENABLED.parse_dom_host(
Url::parse("https://example.test/control").unwrap(),
"<!doctype html><p>control".to_owned(),
);
assert_eq!(
document.document_quirks_mode_for_handle(root),
expected_mode.document_quirks_mode_for_handle(expected_mode.document_handle())
);
assert_eq!(
document.text_content(root).as_deref(),
Some("\n<&amp;\nbeta\ngamma\u{fffd}</pre><script>bad()</script>")
);
}
}
#[test]
fn html_document_still_parses_markup_and_entities() {
let document = parse_test_document("<b>Gülçek&amp;</b>");
+4 -1
View File
@@ -236,9 +236,12 @@ impl HtmlParserSession {
// Parse only the browser-owned shell. Response bytes enter the tokenizer
// after it has switched to plaintext, so tags and entities stay literal.
self.process(StrTendril::from(concat!(
"<!doctype html><html><head></head><body>",
"<html><head></head><body>",
"<pre style=\"word-wrap: break-word; white-space: pre-wrap;\">\n"
)));
// Text documents have no doctype but always use no-quirks mode.
self.sink()
.set_quirks_mode(html5ever::tree_builder::QuirksMode::NoQuirks);
self.tokenizer.set_plaintext_state();
}
@@ -12,6 +12,20 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
true,
PAYLOAD,
),
(
"plain.xml",
"text/plain; charset=utf-8",
"text/plain",
true,
"\n<&amp;\r\nbeta\rgamma\0</pre><script>window.executed=42</script>",
),
(
"json.xml",
"application/problem+json",
"application/problem+json",
true,
PAYLOAD,
),
(
"plain-bom",
"text/plain; charset=utf-8",
@@ -79,7 +93,12 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
).await;
let expected = if literal {
json!([
payload.strip_prefix('\u{feff}').unwrap_or(payload),
payload
.strip_prefix('\u{feff}')
.unwrap_or(payload)
.replace("\r\n", "\n")
.replace('\r', "\n")
.replace('\0', "\u{fffd}"),
null,
0,
content_type
@@ -88,6 +107,17 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
json!(["literal&Gülçek", 42, 1, content_type])
};
assert_eq!(observed["value"], expected, "{prefix}{name}");
if literal {
let structure = classic_request_json_with_body(
app.clone(), Method::POST, &format!("/session/{session_id}/execute/sync"),
json!({"script": "const d=(document.querySelector('iframe')?.contentWindow ?? window).document; return [d.doctype===null,d.compatMode,d.body.children.length,d.body.firstElementChild.localName,d.querySelectorAll('pre').length];", "args": []}),
).await;
assert_eq!(
structure["value"],
json!([true, "CSS1Compat", 1, "pre", 1]),
"{prefix}{name}"
);
}
}
}
classic_request_json(app, Method::DELETE, &format!("/session/{session_id}")).await;
+51 -25
View File
@@ -1,5 +1,4 @@
use crate::web_api_interfaces;
use html5ever::tree_builder::QuirksMode;
use moli_web_mime::{is_dom_parser_xml_mime, is_html_document_mime};
use moli_webapi_declare::WebApiFunctionTemplate;
use url::Url;
@@ -396,7 +395,7 @@ fn parse_browsing_context_document_snapshot(
html_parser: HtmlParser,
) -> (DomHost, DetachedDocumentKind) {
if content_type.is_some_and(is_dom_parser_xml_mime)
|| child_document_url_is_xml_like(&document_url)
|| (content_type.is_none() && child_document_url_is_xml_like(&document_url))
{
let parser = XmlParser;
return (
@@ -404,13 +403,12 @@ fn parse_browsing_context_document_snapshot(
DetachedDocumentKind::Xml,
);
}
if content_type.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain")) {
let mut document =
html_parser.parse_dom_host(document_url, plain_text_document_parser_input(source));
// Text documents are HTML Documents whose mode is explicitly no-quirks,
// despite having no doctype that would select that mode through parsing.
document.set_html_quirks_mode_for_parser(QuirksMode::NoQuirks);
return (document, DetachedDocumentKind::Html);
if let Some(content_type) =
content_type.filter(|mime| moli_web_mime::is_text_document_mime(mime))
{
let stream = HtmlParser::SCRIPTING_DISABLED.start_text_document(document_url, content_type);
stream.feed(source);
return (stream.finish_dom_host(), DetachedDocumentKind::Html);
}
(
html_parser.parse_dom_host(document_url, source.to_owned()),
@@ -418,22 +416,6 @@ fn parse_browsing_context_document_snapshot(
)
}
pub(crate) fn plain_text_document_parser_input(source: &str) -> String {
let mut input = String::with_capacity(source.len().saturating_add(64));
// The HTML parser discards the first LF after <pre>; preserve any source LF.
input.push_str("<html><head></head><body><pre>\n");
for character in source.chars() {
match character {
'&' => input.push_str("&amp;"),
'<' => input.push_str("&lt;"),
'\0' => input.push('\u{fffd}'),
_ => input.push(character),
}
}
input.push_str("</pre></body></html>");
input
}
fn child_document_url_is_xml_like(url: &Url) -> bool {
let path = url.path().to_ascii_lowercase();
path.ends_with(".xml") || path.ends_with(".xhtml") || path.ends_with(".svg")
@@ -602,6 +584,50 @@ mod tests {
assert!(disabled.element_handle_by_id("fallback").is_some());
}
#[test]
fn child_projection_explicit_mime_overrides_xml_extension() {
let source = "\n<&amp;</pre><script>window.executed=1</script>";
for mime in [
"text/plain",
"application/json",
"application/problem+json",
"text/javascript",
] {
let (document, kind) = parse_browsing_context_document_snapshot(
Url::parse("https://example.test/data.xml").unwrap(),
source,
Some(mime),
HtmlParser::SCRIPTING_ENABLED,
);
let root = document.document_handle();
assert_eq!(kind, DetachedDocumentKind::Html);
assert_eq!(
document.text_content(root).as_deref(),
Some(source),
"{mime}"
);
let children = document.child_handles(root).collect::<Vec<_>>();
assert_eq!(children.len(), 1);
assert!(document.is_html_element_named(children[0], "html"));
assert_eq!(
document.document_quirks_mode_for_handle(root),
Some(selectors::matching::QuirksMode::NoQuirks)
);
}
let (document, kind) = parse_browsing_context_document_snapshot(
Url::parse("https://example.test/page.svg").unwrap(),
"<!doctype html><b id='parsed'>&amp;</b>",
Some("text/html"),
HtmlParser::SCRIPTING_ENABLED,
);
assert_eq!(kind, DetachedDocumentKind::Html);
assert!(document.element_handle_by_id("parsed").is_some());
assert_eq!(
document.text_content(document.document_handle()).as_deref(),
Some("&")
);
}
#[test]
fn child_plain_text_document_uses_pre_and_no_quirks_mode() {
let (document, kind) = parse_browsing_context_document_snapshot(
@@ -170,16 +170,8 @@ impl JsContextHost {
navigation_loader: Option<crate::network::navigation::NavigationResourceLoader>,
is_xml_document: bool,
) -> Option<ChildDocumentInstallResult> {
let is_plain_text_document = snapshot
.content_type
.as_deref()
.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain"));
let source = if is_xml_document {
std::borrow::Cow::Borrowed(snapshot.markup.as_str())
} else if is_plain_text_document {
std::borrow::Cow::Owned(crate::dom_parser::plain_text_document_parser_input(
&snapshot.markup,
))
} else {
crate::dom_parser::preserve_decoded_bom_only_browsing_context_body(
&snapshot.markup,
@@ -1012,14 +1012,6 @@ impl JsContextHost {
let mut owner = ChildFrameLiveParserOwner::new(self, scope, document_handle);
parser.finish(&mut owner)
};
if self
.dom_host()
.document_content_type_for_handle(document_handle)
.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain"))
{
self.dom_host_mut()
.set_html_quirks_mode_for_parser_document(document_handle, QuirksMode::NoQuirks);
}
self.queue_live_child_parser_discovery_signals(
child_handle,
document_handle,