control".to_owned(), + ); + assert_eq!( + document.document_quirks_mode_for_handle(root), + expected_mode.document_quirks_mode_for_handle(expected_mode.document_handle()) + ); + assert_eq!( + document.text_content(root).as_deref(), + Some("\n<&\nbeta\ngamma\u{fffd}") + ); + } + } + #[test] fn html_document_still_parses_markup_and_entities() { let document = parse_test_document("Gülçek&"); diff --git a/moli-parser/src/session.rs b/moli-parser/src/session.rs index a38de08e5b..9b8e3581fb 100644 --- a/moli-parser/src/session.rs +++ b/moli-parser/src/session.rs @@ -236,9 +236,12 @@ impl HtmlParserSession { // Parse only the browser-owned shell. Response bytes enter the tokenizer // after it has switched to plaintext, so tags and entities stay literal. self.process(StrTendril::from(concat!( - "
", + "", "\n"
)));
+ // Text documents have no doctype but always use no-quirks mode.
+ self.sink()
+ .set_quirks_mode(html5ever::tree_builder::QuirksMode::NoQuirks);
self.tokenizer.set_plaintext_state();
}
diff --git a/moli-protocol-server/src/protocol_server/tests/classic/extracted/navigation.rs b/moli-protocol-server/src/protocol_server/tests/classic/extracted/navigation.rs
index 57c4abc15e..abf196d998 100644
--- a/moli-protocol-server/src/protocol_server/tests/classic/extracted/navigation.rs
+++ b/moli-protocol-server/src/protocol_server/tests/classic/extracted/navigation.rs
@@ -12,6 +12,20 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
true,
PAYLOAD,
),
+ (
+ "plain.xml",
+ "text/plain; charset=utf-8",
+ "text/plain",
+ true,
+ "\n<&\r\nbeta\rgamma\0",
+ ),
+ (
+ "json.xml",
+ "application/problem+json",
+ "application/problem+json",
+ true,
+ PAYLOAD,
+ ),
(
"plain-bom",
"text/plain; charset=utf-8",
@@ -79,7 +93,12 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
).await;
let expected = if literal {
json!([
- payload.strip_prefix('\u{feff}').unwrap_or(payload),
+ payload
+ .strip_prefix('\u{feff}')
+ .unwrap_or(payload)
+ .replace("\r\n", "\n")
+ .replace('\r', "\n")
+ .replace('\0', "\u{fffd}"),
null,
0,
content_type
@@ -88,6 +107,17 @@ async fn webdriver_classic_document_mime_is_shared_by_main_and_child_documents()
json!(["literal&Gülçek", 42, 1, content_type])
};
assert_eq!(observed["value"], expected, "{prefix}{name}");
+ if literal {
+ let structure = classic_request_json_with_body(
+ app.clone(), Method::POST, &format!("/session/{session_id}/execute/sync"),
+ json!({"script": "const d=(document.querySelector('iframe')?.contentWindow ?? window).document; return [d.doctype===null,d.compatMode,d.body.children.length,d.body.firstElementChild.localName,d.querySelectorAll('pre').length];", "args": []}),
+ ).await;
+ assert_eq!(
+ structure["value"],
+ json!([true, "CSS1Compat", 1, "pre", 1]),
+ "{prefix}{name}"
+ );
+ }
}
}
classic_request_json(app, Method::DELETE, &format!("/session/{session_id}")).await;
diff --git a/moli-renderer-v8/src/dom_parser.rs b/moli-renderer-v8/src/dom_parser.rs
index e72deebc20..43f1e5b210 100644
--- a/moli-renderer-v8/src/dom_parser.rs
+++ b/moli-renderer-v8/src/dom_parser.rs
@@ -1,5 +1,4 @@
use crate::web_api_interfaces;
-use html5ever::tree_builder::QuirksMode;
use moli_web_mime::{is_dom_parser_xml_mime, is_html_document_mime};
use moli_webapi_declare::WebApiFunctionTemplate;
use url::Url;
@@ -396,7 +395,7 @@ fn parse_browsing_context_document_snapshot(
html_parser: HtmlParser,
) -> (DomHost, DetachedDocumentKind) {
if content_type.is_some_and(is_dom_parser_xml_mime)
- || child_document_url_is_xml_like(&document_url)
+ || (content_type.is_none() && child_document_url_is_xml_like(&document_url))
{
let parser = XmlParser;
return (
@@ -404,13 +403,12 @@ fn parse_browsing_context_document_snapshot(
DetachedDocumentKind::Xml,
);
}
- if content_type.is_some_and(|mime| mime.eq_ignore_ascii_case("text/plain")) {
- let mut document =
- html_parser.parse_dom_host(document_url, plain_text_document_parser_input(source));
- // Text documents are HTML Documents whose mode is explicitly no-quirks,
- // despite having no doctype that would select that mode through parsing.
- document.set_html_quirks_mode_for_parser(QuirksMode::NoQuirks);
- return (document, DetachedDocumentKind::Html);
+ if let Some(content_type) =
+ content_type.filter(|mime| moli_web_mime::is_text_document_mime(mime))
+ {
+ let stream = HtmlParser::SCRIPTING_DISABLED.start_text_document(document_url, content_type);
+ stream.feed(source);
+ return (stream.finish_dom_host(), DetachedDocumentKind::Html);
}
(
html_parser.parse_dom_host(document_url, source.to_owned()),
@@ -418,22 +416,6 @@ fn parse_browsing_context_document_snapshot(
)
}
-pub(crate) fn plain_text_document_parser_input(source: &str) -> String {
- let mut input = String::with_capacity(source.len().saturating_add(64));
- // The HTML parser discards the first LF after ; preserve any source LF.
- input.push_str("\n");
- for character in source.chars() {
- match character {
- '&' => input.push_str("&"),
- '<' => input.push_str("<"),
- '\0' => input.push('\u{fffd}'),
- _ => input.push(character),
- }
- }
- input.push_str("");
- input
-}
-
fn child_document_url_is_xml_like(url: &Url) -> bool {
let path = url.path().to_ascii_lowercase();
path.ends_with(".xml") || path.ends_with(".xhtml") || path.ends_with(".svg")
@@ -602,6 +584,50 @@ mod tests {
assert!(disabled.element_handle_by_id("fallback").is_some());
}
+ #[test]
+ fn child_projection_explicit_mime_overrides_xml_extension() {
+ let source = "\n<&";
+ for mime in [
+ "text/plain",
+ "application/json",
+ "application/problem+json",
+ "text/javascript",
+ ] {
+ let (document, kind) = parse_browsing_context_document_snapshot(
+ Url::parse("https://example.test/data.xml").unwrap(),
+ source,
+ Some(mime),
+ HtmlParser::SCRIPTING_ENABLED,
+ );
+ let root = document.document_handle();
+ assert_eq!(kind, DetachedDocumentKind::Html);
+ assert_eq!(
+ document.text_content(root).as_deref(),
+ Some(source),
+ "{mime}"
+ );
+ let children = document.child_handles(root).collect::