From 5d4de26f6eb2374a70df2db86799b2ffd2b1f863 Mon Sep 17 00:00:00 2001 From: ldm0 Date: Wed, 26 Aug 2026 17:28:24 +0800 Subject: [PATCH] refactor(content-type): centralize parsing semantics --- Cargo.lock | 10 +- Cargo.toml | 1 + moli-content-type/Cargo.toml | 11 ++ moli-content-type/src/lib.rs | 18 +++ moli-content-type/src/response.rs | 147 ++++++++++++++++++ moli-content-type/src/tests.rs | 233 ++++++++++++++++++++++++++++ moli-content-type/src/whatwg.rs | 73 +++++++++ moli-encoding/Cargo.toml | 2 +- moli-encoding/src/labels.rs | 37 +---- moli-encoding/src/tests.rs | 42 ++++- moli-header-field/src/parameters.rs | 36 +++-- moli-header-field/src/tests.rs | 12 ++ moli-web-mime/Cargo.toml | 1 + moli-web-mime/src/parse.rs | 40 +---- 14 files changed, 581 insertions(+), 82 deletions(-) create mode 100644 moli-content-type/Cargo.toml create mode 100644 moli-content-type/src/lib.rs create mode 100644 moli-content-type/src/response.rs create mode 100644 moli-content-type/src/tests.rs create mode 100644 moli-content-type/src/whatwg.rs diff --git a/Cargo.lock b/Cargo.lock index 5971a53fcb..8af9149d50 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2013,6 +2013,13 @@ dependencies = [ "html5ever", ] +[[package]] +name = "moli-content-type" +version = "0.1.0" +dependencies = [ + "data-url", +] + [[package]] name = "moli-cookie-cache" version = "0.1.0" @@ -2139,7 +2146,7 @@ version = "0.1.0" dependencies = [ "encoding_rs", "moli-charset-parser", - "moli-header-field", + "moli-content-type", ] [[package]] @@ -2797,6 +2804,7 @@ dependencies = [ "data-url", "http", "mime", + "moli-content-type", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index 26ae9e0d4b..a1769b0043 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,6 +12,7 @@ members = [ "moli-cookie-cache", "moli-protocol-cdp", "moli-cookie-store", + "moli-content-type", "moli-css-parse", "moli-crypto", "moli-curl", diff --git a/moli-content-type/Cargo.toml b/moli-content-type/Cargo.toml new file mode 100644 index 0000000000..ebe61659b0 --- /dev/null +++ b/moli-content-type/Cargo.toml @@ -0,0 +1,11 @@ +[package] +license.workspace = true +name = "moli-content-type" +version = "0.1.0" +edition = "2024" + +[dependencies] +data-url = "0.3.2" + +[lints] +workspace = true diff --git a/moli-content-type/src/lib.rs b/moli-content-type/src/lib.rs new file mode 100644 index 0000000000..22a4b8d022 --- /dev/null +++ b/moli-content-type/src/lib.rs @@ -0,0 +1,18 @@ +//! Shared parsing for MIME values and HTTP response `Content-Type` fields. +//! +//! Web-facing MIME operations use the WHATWG parser, while transport response +//! metadata uses Chromium's deliberately more tolerant network semantics. The +//! two entry points are named separately so callers cannot select the wrong +//! behavior through a generic leniency flag. + +mod response; +mod whatwg; + +pub use response::{ResponseContentType, parse_response_content_type}; +pub use whatwg::{ + MimeType, mime_charset, mime_essence, mime_parameter, normalize_web_api_mime_type, + parse_mime_type, +}; + +#[cfg(test)] +mod tests; diff --git a/moli-content-type/src/response.rs b/moli-content-type/src/response.rs new file mode 100644 index 0000000000..3a230ec1bb --- /dev/null +++ b/moli-content-type/src/response.rs @@ -0,0 +1,147 @@ +/// A response `Content-Type` parsed with Chromium network-layer tolerances. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct ResponseContentType { + mime_type: String, + parameters: Vec<(String, String)>, +} + +impl ResponseContentType { + /// The lower-case media type Chromium accepted from the response. + pub fn mime_type(&self) -> &str { + &self.mime_type + } + + /// Parsed parameters in header order. + pub fn parameters(&self) -> &[(String, String)] { + &self.parameters + } + + /// The first parameter with the given ASCII-case-insensitive name. + pub fn parameter(&self, name: &str) -> Option<&str> { + self.parameters + .iter() + .find(|(candidate, _)| candidate.eq_ignore_ascii_case(name)) + .map(|(_, value)| value.as_str()) + } + + /// The first `charset` value, with HTTP linear whitespace trimmed. + pub fn charset(&self) -> Option<&str> { + self.parameter("charset").map(trim_http_lws) + } + + /// The first `boundary` value, with HTTP linear whitespace trimmed. + pub fn boundary(&self) -> Option<&str> { + self.parameter("boundary").map(trim_http_lws) + } +} + +/// Parses an HTTP response `Content-Type` like Chromium's +/// `net::HttpUtil::ParseContentType` and `net::ParseMimeType` path. +/// +/// This intentionally accepts malformed values seen on the network, including +/// an unterminated quoted parameter and an empty subtype. Use +/// [`crate::parse_mime_type`] for standards-facing Web API behavior. +pub fn parse_response_content_type(input: &str) -> Option { + // Chromium treats an exact bare wildcard as meaningless, while retaining + // wildcard values that have parameters or even trailing whitespace. + if input == "*/*" { + return None; + } + + let bytes = input.as_bytes(); + let input_len = bytes.len(); + let type_start = skip_http_lws(bytes, 0); + let type_end = bytes[type_start..] + .iter() + .position(|byte| is_http_lws(*byte) || matches!(*byte, b';' | b'(')) + .map_or(input_len, |offset| type_start + offset); + let slash = bytes.iter().position(|byte| *byte == b'/')?; + if slash > type_end { + return None; + } + + let mime_type = input[type_start..type_end].to_ascii_lowercase(); + let mut parameters = Vec::new(); + let mut offset = find_from(bytes, type_end, |byte| byte == b';').unwrap_or(input_len); + + while offset < input_len { + offset = skip_http_lws(bytes, offset + 1); + let name_start = offset; + let Some(delimiter) = find_from(bytes, offset, |byte| matches!(byte, b';' | b'=')) else { + break; + }; + offset = delimiter; + if bytes[offset] == b';' { + continue; + } + + let name = input[name_start..offset].to_owned(); + offset = skip_http_lws(bytes, offset + 1); + if offset >= input_len || bytes[offset] == b';' { + continue; + } + + let value = if bytes[offset] == b'"' { + let (value, next_offset) = parse_quoted_value(input, offset + 1); + offset = find_from(bytes, next_offset, |byte| byte == b';').unwrap_or(input_len); + value + } else { + let value_start = offset; + offset = find_from(bytes, offset, |byte| byte == b';').unwrap_or(input_len); + let value_end = trim_trailing_http_lws(bytes, value_start, offset); + input[value_start..value_end].to_owned() + }; + parameters.push((name, value)); + } + + Some(ResponseContentType { + mime_type, + parameters, + }) +} + +fn parse_quoted_value(input: &str, mut offset: usize) -> (String, usize) { + let bytes = input.as_bytes(); + let mut value = String::new(); + while offset < bytes.len() && bytes[offset] != b'"' { + if bytes[offset] == b'\\' && offset + 1 < bytes.len() { + offset += 1; + } + let character = input[offset..] + .chars() + .next() + .expect("offset is inside the input"); + value.push(character); + offset += character.len_utf8(); + } + (value, offset) +} + +fn find_from(bytes: &[u8], start: usize, predicate: impl Fn(u8) -> bool) -> Option { + bytes[start..] + .iter() + .position(|byte| predicate(*byte)) + .map(|offset| start + offset) +} + +fn skip_http_lws(bytes: &[u8], mut offset: usize) -> usize { + while bytes.get(offset).is_some_and(|byte| is_http_lws(*byte)) { + offset += 1; + } + offset +} + +fn trim_trailing_http_lws(bytes: &[u8], start: usize, mut end: usize) -> usize { + while end > start && is_http_lws(bytes[end - 1]) { + end -= 1; + } + end +} + +fn trim_http_lws(value: &str) -> &str { + value.trim_matches(|character: char| matches!(character, ' ' | '\t' | '\r' | '\n')) +} + +fn is_http_lws(byte: u8) -> bool { + matches!(byte, b' ' | b'\t' | b'\r' | b'\n') +} diff --git a/moli-content-type/src/tests.rs b/moli-content-type/src/tests.rs new file mode 100644 index 0000000000..929bfa5611 --- /dev/null +++ b/moli-content-type/src/tests.rs @@ -0,0 +1,233 @@ +use super::*; + +fn response_charset(input: &str) -> Option { + parse_response_content_type(input)? + .charset() + .map(str::to_owned) +} + +fn assert_response_content_type( + input: &str, + expected_mime_type: Option<&str>, + expected_charset: Option<&str>, +) { + let parsed = parse_response_content_type(input); + assert_eq!( + parsed.as_ref().map(ResponseContentType::mime_type), + expected_mime_type, + "mime type for {input:?}" + ); + assert_eq!( + parsed.as_ref().and_then(|value| value.charset()), + expected_charset, + "charset for {input:?}" + ); +} + +#[test] +fn whatwg_parser_validates_the_mime_essence() { + let parsed = parse_mime_type(" Text/Plain ; Charset=\"UTF-8\" ").unwrap(); + assert_eq!(parsed.type_(), "text"); + assert_eq!(parsed.subtype(), "plain"); + assert_eq!(parsed.essence(), "text/plain"); + assert_eq!(parsed.parameter("CHARSET"), Some("UTF-8")); + assert_eq!(parsed.to_string(), "text/plain;charset=UTF-8"); + + assert!(parse_mime_type("garbage; charset=utf-8").is_none()); + assert!(parse_mime_type("text/").is_none()); +} + +#[test] +fn whatwg_parser_recovers_valid_parameters() { + assert_eq!( + mime_parameter("text/plain; title=\"alpha;beta\"", "title").as_deref(), + Some("alpha;beta") + ); + assert_eq!( + mime_charset("text/plain; charset=; charset=gbk").as_deref(), + Some("gbk") + ); + assert_eq!( + mime_charset("text/plain; charset=\"\"; charset=gbk").as_deref(), + Some("") + ); + assert_eq!( + mime_charset("text/plain; charset=\"utf-8").as_deref(), + Some("utf-8") + ); +} + +#[test] +fn response_parser_matches_chromium_quoted_parameter_tolerances() { + assert_eq!( + response_charset("text/html; boundary=\"; charset=gbk\""), + None + ); + assert_eq!( + response_charset("text/html; name=\"a\\\"; charset=gbk\"; charset=utf-8").as_deref(), + Some("utf-8") + ); + assert_eq!( + response_charset("text/html; charset=\"\\utf\\-\\8\"").as_deref(), + Some("utf-8") + ); + assert_eq!( + response_charset("text/html; charset=\"utf-8").as_deref(), + Some("utf-8") + ); + assert_eq!( + response_charset("text/html; charset=\"\\\\\\\"\\").as_deref(), + Some("\\\"\\") + ); +} + +#[test] +fn response_parser_does_not_reopen_quotes_after_a_value_started() { + assert_eq!( + response_charset("text/html; x=a=\"unterminated; charset=gbk").as_deref(), + Some("gbk") + ); + assert_eq!( + response_charset("text/html; x=\"ok\"junk=\"unterminated; charset=gbk").as_deref(), + Some("gbk") + ); +} + +#[test] +fn response_parser_preserves_chromium_empty_and_duplicate_rules() { + assert_eq!( + response_charset("text/html; charset=; charset=gbk").as_deref(), + Some("gbk") + ); + assert_eq!( + response_charset("text/html; charset=\"\"; charset=gbk").as_deref(), + Some("") + ); + assert_eq!( + response_charset("text/html; charset=foo; charset=utf-8").as_deref(), + Some("foo") + ); +} + +#[test] +fn response_parser_preserves_parameter_name_and_quote_semantics() { + assert_eq!(response_charset("text/html; charset =utf-8"), None); + assert_eq!( + response_charset("text/html; charset='utf-8'").as_deref(), + Some("'utf-8'") + ); + assert_eq!( + response_charset("text/html; \"; \"\"; charset=utf-8").as_deref(), + Some("utf-8") + ); + assert_eq!( + response_charset("text/html; charset=u\"tf-8\"").as_deref(), + Some("u\"tf-8\"") + ); +} + +#[test] +fn response_parser_has_a_distinct_network_level_validity_boundary() { + assert!(parse_response_content_type("garbage; charset=utf-8").is_none()); + assert!(parse_response_content_type("*/*").is_none()); + assert_eq!( + parse_response_content_type("text/") + .map(|parsed| parsed.mime_type().to_owned()) + .as_deref(), + Some("text/") + ); + assert_eq!( + response_charset("*/*; charset=utf-8").as_deref(), + Some("utf-8") + ); +} + +#[test] +fn response_parser_matches_chromium_http_util_regression_matrix() { + for (input, mime_type, charset) in [ + ("text/html", Some("text/html"), None), + ("text/html;", Some("text/html"), None), + ("text/html; charset=utf-8", Some("text/html"), Some("utf-8")), + ("text/html; charset =utf-8", Some("text/html"), None), + ( + "text/html; charset= utf-8", + Some("text/html"), + Some("utf-8"), + ), + ( + "text/html; charset=utf-8 ", + Some("text/html"), + Some("utf-8"), + ), + ("text/html; charset", Some("text/html"), None), + ("text/html; charset=", Some("text/html"), None), + ("text/html; charset= ", Some("text/html"), None), + ("text/html; charset= ;", Some("text/html"), None), + ("text/html; charset=\"\"", Some("text/html"), Some("")), + ("text/html; charset=\" \"", Some("text/html"), Some("")), + ( + "text/html; charset=\" foo \"", + Some("text/html"), + Some("foo"), + ), + ( + "text/html; charset=foo; charset=utf-8", + Some("text/html"), + Some("foo"), + ), + ( + "text/html; charset; charset=; charset=utf-8", + Some("text/html"), + Some("utf-8"), + ), + ( + "text/html; charset=utf-8; charset=; charset", + Some("text/html"), + Some("utf-8"), + ), + ( + "text/html; \"; \"\"; charset=utf-8", + Some("text/html"), + Some("utf-8"), + ), + ( + "text/html; charset=u\"tf-8\"", + Some("text/html"), + Some("u\"tf-8\""), + ), + ( + "text/html; charset=\"utf-8", + Some("text/html"), + Some("utf-8"), + ), + ( + "text/html; charset=\";charset=utf-8;\"", + Some("text/html"), + Some(";charset=utf-8;"), + ), + ( + "text/html; charset='utf-8'", + Some("text/html"), + Some("'utf-8'"), + ), + ("text/", Some("text/"), None), + ("*/*", None, None), + ("*/*; charset=utf-8", Some("*/*"), Some("utf-8")), + ("*/* ", Some("*/*"), None), + ("teXT/html", Some("text/html"), None), + ] { + assert_response_content_type(input, mime_type, charset); + } + + let boundary = + parse_response_content_type("text/html; boundary=\"WebKit-ada-df-dsf-adsfadsfs \"") + .unwrap(); + assert_eq!(boundary.boundary(), Some("WebKit-ada-df-dsf-adsfadsfs")); +} + +#[test] +fn normalizing_web_api_mime_types_rejects_non_http_bytes() { + assert_eq!(normalize_web_api_mime_type("Text/Plain"), "text/plain"); + assert_eq!(normalize_web_api_mime_type("text/\nplain"), ""); + assert_eq!(normalize_web_api_mime_type("text/你好"), ""); +} diff --git a/moli-content-type/src/whatwg.rs b/moli-content-type/src/whatwg.rs new file mode 100644 index 0000000000..eeb4343dae --- /dev/null +++ b/moli-content-type/src/whatwg.rs @@ -0,0 +1,73 @@ +use data_url::mime::Mime as WhatwgMime; +use std::fmt; + +/// A MIME type parsed according to the WHATWG MIME Sniffing Standard. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct MimeType { + inner: WhatwgMime, +} + +impl MimeType { + /// The lower-case MIME top-level type. + pub fn type_(&self) -> &str { + &self.inner.type_ + } + + /// The lower-case MIME subtype. + pub fn subtype(&self) -> &str { + &self.inner.subtype + } + + /// The lower-case `type/subtype` MIME essence. + pub fn essence(&self) -> String { + format!("{}/{}", self.type_(), self.subtype()) + } + + /// Parsed parameters in header order. + pub fn parameters(&self) -> &[(String, String)] { + &self.inner.parameters + } + + /// The first parameter with the given ASCII-case-insensitive name. + pub fn parameter(&self, name: &str) -> Option<&str> { + self.inner + .parameters + .iter() + .find(|(candidate, _)| candidate.eq_ignore_ascii_case(name)) + .map(|(_, value)| value.as_str()) + } +} + +impl fmt::Display for MimeType { + fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { + self.inner.fmt(formatter) + } +} + +/// Parses a MIME value using WHATWG validation and parameter recovery rules. +pub fn parse_mime_type(input: &str) -> Option { + input.parse().ok().map(|inner| MimeType { inner }) +} + +/// Returns the MIME essence of a valid WHATWG MIME value. +pub fn mime_essence(input: &str) -> Option { + parse_mime_type(input).map(|mime| mime.essence()) +} + +/// Returns one parsed MIME parameter by ASCII-case-insensitive name. +pub fn mime_parameter(input: &str, name: &str) -> Option { + parse_mime_type(input)?.parameter(name).map(str::to_owned) +} + +/// Returns the parsed `charset` MIME parameter. +pub fn mime_charset(input: &str) -> Option { + mime_parameter(input, "charset") +} + +/// Normalizes a Web API MIME string without parsing its structure. +pub fn normalize_web_api_mime_type(raw: &str) -> String { + if raw.is_empty() || raw.bytes().any(|byte| !(0x20..=0x7e).contains(&byte)) { + return String::new(); + } + raw.to_ascii_lowercase() +} diff --git a/moli-encoding/Cargo.toml b/moli-encoding/Cargo.toml index 2a49e0f201..908d897762 100644 --- a/moli-encoding/Cargo.toml +++ b/moli-encoding/Cargo.toml @@ -7,7 +7,7 @@ edition = "2024" [dependencies] encoding_rs = "0.8" moli-charset-parser = { path = "../moli-charset-parser" } -moli-header-field = { path = "../moli-header-field" } +moli-content-type = { path = "../moli-content-type" } [lints] workspace = true diff --git a/moli-encoding/src/labels.rs b/moli-encoding/src/labels.rs index 99973bac13..133d4ce540 100644 --- a/moli-encoding/src/labels.rs +++ b/moli-encoding/src/labels.rs @@ -1,5 +1,5 @@ use encoding_rs::Encoding; -use moli_header_field::{split_outside_quoted_strings, unquote_parameter_value}; +use moli_content_type::parse_response_content_type; pub fn encoding_for_label(label: &str) -> Option<&'static Encoding> { Encoding::for_label(label.trim().as_bytes()) @@ -7,36 +7,13 @@ pub fn encoding_for_label(label: &str) -> Option<&'static Encoding> { /// The `charset` parameter of a `Content-Type` value. /// -/// Parameters are separated by the `;` characters outside a quoted string, and -/// a quoted value has its quoting backslashes removed, so a `charset` written -/// inside another parameter's quoted string is not read as a parameter here. +/// Response metadata is parsed with Chromium network-layer tolerances before +/// the label is handed to `encoding_rs`. In particular, quoted separators stay +/// inside their parameter and unterminated quoted values remain recoverable. pub fn charset_from_content_type(value: &str) -> Option { - // The first segment is the media type itself, not a parameter. - for parameter in split_outside_quoted_strings(value, ';').into_iter().skip(1) { - let Some((name, parameter_value)) = parameter.split_once('=') else { - continue; - }; - if !name.trim().eq_ignore_ascii_case("charset") { - continue; - } - let parameter_value = parameter_value.trim(); - let charset = if parameter_value.starts_with('"') { - // Delimiters are already gone, so any `"` left is data produced by - // an escaped quote and must not be trimmed away. - unquote_parameter_value(parameter_value).into_owned() - } else { - // Apostrophe delimiters are not a quoted string, but receivers - // have long tolerated them here, so keep stripping them. - parameter_value - .trim_matches(|ch| ch == '"' || ch == '\'') - .trim() - .to_owned() - }; - if !charset.is_empty() { - return Some(charset); - } - } - None + let parsed = parse_response_content_type(value)?; + let charset = parsed.charset()?; + (!charset.is_empty()).then(|| charset.to_owned()) } pub fn charset_from_headers(headers: &[(String, String)]) -> Option { diff --git a/moli-encoding/src/tests.rs b/moli-encoding/src/tests.rs index d305a0b29d..2cf1f56f85 100644 --- a/moli-encoding/src/tests.rs +++ b/moli-encoding/src/tests.rs @@ -705,15 +705,13 @@ fn header_charset_removes_quoting_backslashes() { } #[test] -fn header_charset_keeps_its_existing_tolerances() { +fn header_charset_matches_chromium_network_tolerances() { for header in [ "text/html; charset=utf-8", "text/html;charset=utf-8", "TEXT/HTML; CHARSET=UTF-8", - "text/html; charset = utf-8 ", "text/html; charset=\"utf-8\"", "text/html;charset=utf-8;", - "text/html; charset='utf-8'", "text/html; charset=\"utf-8", ] { assert_eq!( @@ -728,12 +726,50 @@ fn header_charset_keeps_its_existing_tolerances() { assert_eq!(charset_from_content_type("text/html"), None); assert_eq!(charset_from_content_type("text/html; charset="), None); assert_eq!(charset_from_content_type("charset=utf-8"), None); + assert_eq!( + charset_from_content_type("text/html; charset = utf-8"), + None + ); + assert_eq!( + charset_from_content_type("text/html; charset='utf-8'").as_deref(), + Some("'utf-8'") + ); + assert!( + charset_from_content_type("text/html; charset='utf-8'") + .as_deref() + .and_then(encoding_for_label) + .is_none() + ); assert_eq!( charset_from_content_type("text/html; charset=gbk; boundary=x").as_deref(), Some("gbk") ); } +#[test] +fn header_charset_preserves_chromium_empty_parameter_precedence() { + assert_eq!( + charset_from_content_type("text/html; charset=; charset=gbk").as_deref(), + Some("gbk") + ); + assert_eq!( + charset_from_content_type("text/html; charset=\"\"; charset=gbk"), + None + ); +} + +#[test] +fn header_charset_does_not_reopen_quotes_after_a_value_started() { + assert_eq!( + charset_from_content_type("text/html; x=a=\"unterminated; charset=gbk").as_deref(), + Some("gbk") + ); + assert_eq!( + charset_from_content_type("text/html; x=\"ok\"junk=\"unterminated; charset=gbk").as_deref(), + Some("gbk") + ); +} + #[test] fn header_charset_recovers_after_a_stray_quote() { // WPT MIME case: the `"` does not open a parameter value, so the following diff --git a/moli-header-field/src/parameters.rs b/moli-header-field/src/parameters.rs index 436955dcba..129b55c3b3 100644 --- a/moli-header-field/src/parameters.rs +++ b/moli-header-field/src/parameters.rs @@ -1,5 +1,15 @@ use std::borrow::Cow; +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +enum ParameterScanState { + #[default] + BeforeValue, + AtValueStart, + UnquotedValue, + QuotedValue, + AfterQuotedValue, +} + /// Splits `value` on every `separator` that is not inside a quoted parameter /// value. /// @@ -25,14 +35,13 @@ pub fn split_outside_quoted_strings(value: &str, separator: char) -> Vec<&str> { let bytes = value.as_bytes(); let mut segments = Vec::new(); let mut segment_start = 0; - let mut at_value_start = false; - let mut inside_quotes = false; + let mut state = ParameterScanState::BeforeValue; let mut index = 0; while index < bytes.len() { let byte = bytes[index]; - if inside_quotes { + if state == ParameterScanState::QuotedValue { if byte == b'\\' { // Step over the escaped byte so a quoted `\"` does not close // the value. @@ -40,7 +49,7 @@ pub fn split_outside_quoted_strings(value: &str, separator: char) -> Vec<&str> { continue; } if byte == b'"' { - inside_quotes = false; + state = ParameterScanState::AfterQuotedValue; } index += 1; continue; @@ -49,16 +58,15 @@ pub fn split_outside_quoted_strings(value: &str, separator: char) -> Vec<&str> { if byte == separator { segments.push(&value[segment_start..index]); segment_start = index + 1; - at_value_start = false; - } else if byte == b'=' { - at_value_start = true; - } else if byte == b'"' && at_value_start { - inside_quotes = true; - at_value_start = false; - } else if !matches!(byte, b' ' | b'\t') || !at_value_start { - // Whitespace between `=` and the value keeps the value still to - // come; anything else means the value was not quoted. - at_value_start = false; + state = ParameterScanState::BeforeValue; + } else { + state = match state { + ParameterScanState::BeforeValue if byte == b'=' => ParameterScanState::AtValueStart, + ParameterScanState::AtValueStart if matches!(byte, b' ' | b'\t') => state, + ParameterScanState::AtValueStart if byte == b'"' => ParameterScanState::QuotedValue, + ParameterScanState::AtValueStart => ParameterScanState::UnquotedValue, + _ => state, + }; } index += 1; } diff --git a/moli-header-field/src/tests.rs b/moli-header-field/src/tests.rs index 75fd6ec7e1..151135fd45 100644 --- a/moli-header-field/src/tests.rs +++ b/moli-header-field/src/tests.rs @@ -224,3 +224,15 @@ fn a_trailing_backslash_is_kept_rather_than_dropped() { fn an_escaped_quote_survives_as_data() { assert_eq!(unquote_parameter_value("\"utf-8\\\"\""), "utf-8\""); } + +#[test] +fn equals_inside_a_value_does_not_reopen_quoted_string_mode() { + assert_eq!( + split_outside_quoted_strings("token=a=\"unterminated, next=value", ','), + vec!["token=a=\"unterminated", " next=value"] + ); + assert_eq!( + split_outside_quoted_strings("token=\"ok\"junk=\"unterminated, next=value", ','), + vec!["token=\"ok\"junk=\"unterminated", " next=value"] + ); +} diff --git a/moli-web-mime/Cargo.toml b/moli-web-mime/Cargo.toml index 33729ccbb4..8208d6599c 100644 --- a/moli-web-mime/Cargo.toml +++ b/moli-web-mime/Cargo.toml @@ -9,6 +9,7 @@ content_disposition = "0.4.0" data-url = "0.3.2" http = "1" mime = "0.3" +moli-content-type = { path = "../moli-content-type" } [lints] workspace = true diff --git a/moli-web-mime/src/parse.rs b/moli-web-mime/src/parse.rs index 72db8f5d13..5def66406b 100644 --- a/moli-web-mime/src/parse.rs +++ b/moli-web-mime/src/parse.rs @@ -1,42 +1,16 @@ -use data_url::mime::Mime as WebMime; +use moli_content_type::parse_mime_type; + +pub use moli_content_type::{ + mime_charset, mime_essence, mime_parameter, normalize_web_api_mime_type, +}; pub fn parse_mime(input: &str) -> Option { - parse_web_mime(input).and_then(|mime| mime.to_string().parse().ok()) -} - -pub fn mime_essence(input: &str) -> Option { - parse_web_mime(input).map(|mime| mime_essence_from_parsed(&mime)) + parse_mime_type(input).and_then(|mime| mime.to_string().parse().ok()) } pub fn request_header_content_type_essence(input: &str) -> Option { if input.contains(',') { return None; } - parse_web_mime(input).map(|mime| mime_essence_from_parsed(&mime)) -} - -pub fn mime_charset(input: &str) -> Option { - mime_parameter(input, "charset") -} - -pub fn mime_parameter(input: &str, name: &str) -> Option { - let name = name.to_ascii_lowercase(); - parse_web_mime(input)? - .get_parameter(&name) - .map(str::to_owned) -} - -pub fn normalize_web_api_mime_type(raw: &str) -> String { - if raw.is_empty() || raw.bytes().any(|byte| !(0x20..=0x7e).contains(&byte)) { - return String::new(); - } - raw.to_ascii_lowercase() -} - -fn parse_web_mime(input: &str) -> Option { - input.parse().ok() -} - -fn mime_essence_from_parsed(mime: &WebMime) -> String { - format!("{}/{}", mime.type_, mime.subtype) + mime_essence(input) }