mirror of
https://github.com/lexmount/moli.git
synced 2026-10-01 08:00:38 +00:00
fix(document): share literal text parsing across child document paths
This commit is contained in:
@@ -1937,6 +1937,44 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn text_document_shell_has_no_doctype_and_preserves_normalized_literal_text() {
|
||||
for mime in [
|
||||
"text/plain",
|
||||
"application/json",
|
||||
"application/problem+json",
|
||||
"text/javascript",
|
||||
] {
|
||||
let stream = HtmlParser::SCRIPTING_ENABLED
|
||||
.start_text_document(Url::parse("https://example.test/data").unwrap(), mime);
|
||||
for chunk in [
|
||||
"\n<&",
|
||||
"\r",
|
||||
"\nbeta\rgamma\0",
|
||||
"</pre><script>bad()</script>",
|
||||
] {
|
||||
stream.feed(chunk);
|
||||
}
|
||||
let document = stream.finish_dom_host();
|
||||
let root = document.document_handle();
|
||||
let children = document.child_handles(root).collect::<Vec<_>>();
|
||||
assert_eq!(children.len(), 1, "no synthetic doctype: {mime}");
|
||||
assert!(document.is_html_element_named(children[0], "html"));
|
||||
let expected_mode = HtmlParser::SCRIPTING_ENABLED.parse_dom_host(
|
||||
Url::parse("https://example.test/control").unwrap(),
|
||||
"<!doctype html><p>control".to_owned(),
|
||||
);
|
||||
assert_eq!(
|
||||
document.document_quirks_mode_for_handle(root),
|
||||
expected_mode.document_quirks_mode_for_handle(expected_mode.document_handle())
|
||||
);
|
||||
assert_eq!(
|
||||
document.text_content(root).as_deref(),
|
||||
Some("\n<&\nbeta\ngamma\u{fffd}</pre><script>bad()</script>")
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn html_document_still_parses_markup_and_entities() {
|
||||
let document = parse_test_document("<b>Gülçek&</b>");
|
||||
|
||||
@@ -236,9 +236,12 @@ impl HtmlParserSession {
|
||||
// Parse only the browser-owned shell. Response bytes enter the tokenizer
|
||||
// after it has switched to plaintext, so tags and entities stay literal.
|
||||
self.process(StrTendril::from(concat!(
|
||||
"<!doctype html><html><head></head><body>",
|
||||
"<html><head></head><body>",
|
||||
"<pre style=\"word-wrap: break-word; white-space: pre-wrap;\">\n"
|
||||
)));
|
||||
// Text documents have no doctype but always use no-quirks mode.
|
||||
self.sink()
|
||||
.set_quirks_mode(html5ever::tree_builder::QuirksMode::NoQuirks);
|
||||
self.tokenizer.set_plaintext_state();
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,20 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
|
||||
true,
|
||||
PAYLOAD,
|
||||
),
|
||||
(
|
||||
"plain.xml",
|
||||
"text/plain; charset=utf-8",
|
||||
"text/plain",
|
||||
true,
|
||||
"\n<&\r\nbeta\rgamma\0</pre><script>window.executed=42</script>",
|
||||
),
|
||||
(
|
||||
"json.xml",
|
||||
"application/problem+json",
|
||||
"application/problem+json",
|
||||
true,
|
||||
PAYLOAD,
|
||||
),
|
||||
(
|
||||
"plain-bom",
|
||||
"text/plain; charset=utf-8",
|
||||
@@ -79,7 +93,12 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
|
||||
).await;
|
||||
let expected = if literal {
|
||||
json!([
|
||||
payload.strip_prefix('\u{feff}').unwrap_or(payload),
|
||||
payload
|
||||
.strip_prefix('\u{feff}')
|
||||
.unwrap_or(payload)
|
||||
.replace("\r\n", "\n")
|
||||
.replace('\r', "\n")
|
||||
.replace('\0', "\u{fffd}"),
|
||||
null,
|
||||
0,
|
||||
content_type
|
||||
@@ -88,6 +107,17 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
|
||||
json!(["literal&Gülçek", 42, 1, content_type])
|
||||
};
|
||||
assert_eq!(observed["value"], expected, "{prefix}{name}");
|
||||
if literal {
|
||||
let structure = classic_request_json_with_body(
|
||||
app.clone(), Method::POST, &format!("/session/{session_id}/execute/sync"),
|
||||
json!({"script": "const d=(document.querySelector('iframe')?.contentWindow ?? window).document; return [d.doctype===null,d.compatMode,d.body.children.length,d.body.firstElementChild.localName,d.querySelectorAll('pre').length];", "args": []}),
|
||||
).await;
|
||||
assert_eq!(
|
||||
structure["value"],
|
||||
json!([true, "CSS1Compat", 1, "pre", 1]),
|
||||
"{prefix}{name}"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
classic_request_json(app, Method::DELETE, &format!("/session/{session_id}")).await;
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
use crate::web_api_interfaces;
|
||||
use html5ever::tree_builder::QuirksMode;
|
||||
use moli_web_mime::{is_dom_parser_xml_mime, is_html_document_mime};
|
||||
use moli_webapi_declare::WebApiFunctionTemplate;
|
||||
use url::Url;
|
||||
@@ -396,7 +395,7 @@ fn parse_browsing_context_document_snapshot(
|
||||
html_parser: HtmlParser,
|
||||
) -> (DomHost, DetachedDocumentKind) {
|
||||
if content_type.is_some_and(is_dom_parser_xml_mime)
|
||||
|| child_document_url_is_xml_like(&document_url)
|
||||
|| (content_type.is_none() && child_document_url_is_xml_like(&document_url))
|
||||
{
|
||||
let parser = XmlParser;
|
||||
return (
|
||||
@@ -404,13 +403,12 @@ fn parse_browsing_context_document_snapshot(
|
||||
DetachedDocumentKind::Xml,
|
||||
);
|
||||
}
|
||||
if content_type.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain")) {
|
||||
let mut document =
|
||||
html_parser.parse_dom_host(document_url, plain_text_document_parser_input(source));
|
||||
// Text documents are HTML Documents whose mode is explicitly no-quirks,
|
||||
// despite having no doctype that would select that mode through parsing.
|
||||
document.set_html_quirks_mode_for_parser(QuirksMode::NoQuirks);
|
||||
return (document, DetachedDocumentKind::Html);
|
||||
if let Some(content_type) =
|
||||
content_type.filter(|mime| moli_web_mime::is_text_document_mime(mime))
|
||||
{
|
||||
let stream = HtmlParser::SCRIPTING_DISABLED.start_text_document(document_url, content_type);
|
||||
stream.feed(source);
|
||||
return (stream.finish_dom_host(), DetachedDocumentKind::Html);
|
||||
}
|
||||
(
|
||||
html_parser.parse_dom_host(document_url, source.to_owned()),
|
||||
@@ -418,22 +416,6 @@ fn parse_browsing_context_document_snapshot(
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn plain_text_document_parser_input(source: &str) -> String {
|
||||
let mut input = String::with_capacity(source.len().saturating_add(64));
|
||||
// The HTML parser discards the first LF after <pre>; preserve any source LF.
|
||||
input.push_str("<html><head></head><body><pre>\n");
|
||||
for character in source.chars() {
|
||||
match character {
|
||||
'&' => input.push_str("&"),
|
||||
'<' => input.push_str("<"),
|
||||
'\0' => input.push('\u{fffd}'),
|
||||
_ => input.push(character),
|
||||
}
|
||||
}
|
||||
input.push_str("</pre></body></html>");
|
||||
input
|
||||
}
|
||||
|
||||
fn child_document_url_is_xml_like(url: &Url) -> bool {
|
||||
let path = url.path().to_ascii_lowercase();
|
||||
path.ends_with(".xml") || path.ends_with(".xhtml") || path.ends_with(".svg")
|
||||
@@ -602,6 +584,50 @@ mod tests {
|
||||
assert!(disabled.element_handle_by_id("fallback").is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn child_projection_explicit_mime_overrides_xml_extension() {
|
||||
let source = "\n<&</pre><script>window.executed=1</script>";
|
||||
for mime in [
|
||||
"text/plain",
|
||||
"application/json",
|
||||
"application/problem+json",
|
||||
"text/javascript",
|
||||
] {
|
||||
let (document, kind) = parse_browsing_context_document_snapshot(
|
||||
Url::parse("https://example.test/data.xml").unwrap(),
|
||||
source,
|
||||
Some(mime),
|
||||
HtmlParser::SCRIPTING_ENABLED,
|
||||
);
|
||||
let root = document.document_handle();
|
||||
assert_eq!(kind, DetachedDocumentKind::Html);
|
||||
assert_eq!(
|
||||
document.text_content(root).as_deref(),
|
||||
Some(source),
|
||||
"{mime}"
|
||||
);
|
||||
let children = document.child_handles(root).collect::<Vec<_>>();
|
||||
assert_eq!(children.len(), 1);
|
||||
assert!(document.is_html_element_named(children[0], "html"));
|
||||
assert_eq!(
|
||||
document.document_quirks_mode_for_handle(root),
|
||||
Some(selectors::matching::QuirksMode::NoQuirks)
|
||||
);
|
||||
}
|
||||
let (document, kind) = parse_browsing_context_document_snapshot(
|
||||
Url::parse("https://example.test/page.svg").unwrap(),
|
||||
"<!doctype html><b id='parsed'>&</b>",
|
||||
Some("text/html"),
|
||||
HtmlParser::SCRIPTING_ENABLED,
|
||||
);
|
||||
assert_eq!(kind, DetachedDocumentKind::Html);
|
||||
assert!(document.element_handle_by_id("parsed").is_some());
|
||||
assert_eq!(
|
||||
document.text_content(document.document_handle()).as_deref(),
|
||||
Some("&")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn child_plain_text_document_uses_pre_and_no_quirks_mode() {
|
||||
let (document, kind) = parse_browsing_context_document_snapshot(
|
||||
|
||||
@@ -170,16 +170,8 @@ impl JsContextHost {
|
||||
navigation_loader: Option<crate::network::navigation::NavigationResourceLoader>,
|
||||
is_xml_document: bool,
|
||||
) -> Option<ChildDocumentInstallResult> {
|
||||
let is_plain_text_document = snapshot
|
||||
.content_type
|
||||
.as_deref()
|
||||
.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain"));
|
||||
let source = if is_xml_document {
|
||||
std::borrow::Cow::Borrowed(snapshot.markup.as_str())
|
||||
} else if is_plain_text_document {
|
||||
std::borrow::Cow::Owned(crate::dom_parser::plain_text_document_parser_input(
|
||||
&snapshot.markup,
|
||||
))
|
||||
} else {
|
||||
crate::dom_parser::preserve_decoded_bom_only_browsing_context_body(
|
||||
&snapshot.markup,
|
||||
|
||||
@@ -1012,14 +1012,6 @@ impl JsContextHost {
|
||||
let mut owner = ChildFrameLiveParserOwner::new(self, scope, document_handle);
|
||||
parser.finish(&mut owner)
|
||||
};
|
||||
if self
|
||||
.dom_host()
|
||||
.document_content_type_for_handle(document_handle)
|
||||
.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain"))
|
||||
{
|
||||
self.dom_host_mut()
|
||||
.set_html_quirks_mode_for_parser_document(document_handle, QuirksMode::NoQuirks);
|
||||
}
|
||||
self.queue_live_child_parser_discovery_signals(
|
||||
child_handle,
|
||||
document_handle,
|
||||
|
||||
Reference in New Issue
Block a user