mirror of
https://github.com/lexmount/moli.git
synced 2026-10-05 16:00:54 +00:00
refactor(content-type): centralize parsing semantics
This commit is contained in:
Generated
+9
-1
@@ -2013,6 +2013,13 @@ dependencies = [
|
||||
"html5ever",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "moli-content-type"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"data-url",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "moli-cookie-cache"
|
||||
version = "0.1.0"
|
||||
@@ -2139,7 +2146,7 @@ version = "0.1.0"
|
||||
dependencies = [
|
||||
"encoding_rs",
|
||||
"moli-charset-parser",
|
||||
"moli-header-field",
|
||||
"moli-content-type",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2797,6 +2804,7 @@ dependencies = [
|
||||
"data-url",
|
||||
"http",
|
||||
"mime",
|
||||
"moli-content-type",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
@@ -12,6 +12,7 @@ members = [
|
||||
"moli-cookie-cache",
|
||||
"moli-protocol-cdp",
|
||||
"moli-cookie-store",
|
||||
"moli-content-type",
|
||||
"moli-css-parse",
|
||||
"moli-crypto",
|
||||
"moli-curl",
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
[package]
|
||||
license.workspace = true
|
||||
name = "moli-content-type"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
|
||||
[dependencies]
|
||||
data-url = "0.3.2"
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
@@ -0,0 +1,18 @@
|
||||
//! Shared parsing for MIME values and HTTP response `Content-Type` fields.
|
||||
//!
|
||||
//! Web-facing MIME operations use the WHATWG parser, while transport response
|
||||
//! metadata uses Chromium's deliberately more tolerant network semantics. The
|
||||
//! two entry points are named separately so callers cannot select the wrong
|
||||
//! behavior through a generic leniency flag.
|
||||
|
||||
mod response;
|
||||
mod whatwg;
|
||||
|
||||
pub use response::{ResponseContentType, parse_response_content_type};
|
||||
pub use whatwg::{
|
||||
MimeType, mime_charset, mime_essence, mime_parameter, normalize_web_api_mime_type,
|
||||
parse_mime_type,
|
||||
};
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests;
|
||||
@@ -0,0 +1,147 @@
|
||||
/// A response `Content-Type` parsed with Chromium network-layer tolerances.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct ResponseContentType {
|
||||
mime_type: String,
|
||||
parameters: Vec<(String, String)>,
|
||||
}
|
||||
|
||||
impl ResponseContentType {
|
||||
/// The lower-case media type Chromium accepted from the response.
|
||||
pub fn mime_type(&self) -> &str {
|
||||
&self.mime_type
|
||||
}
|
||||
|
||||
/// Parsed parameters in header order.
|
||||
pub fn parameters(&self) -> &[(String, String)] {
|
||||
&self.parameters
|
||||
}
|
||||
|
||||
/// The first parameter with the given ASCII-case-insensitive name.
|
||||
pub fn parameter(&self, name: &str) -> Option<&str> {
|
||||
self.parameters
|
||||
.iter()
|
||||
.find(|(candidate, _)| candidate.eq_ignore_ascii_case(name))
|
||||
.map(|(_, value)| value.as_str())
|
||||
}
|
||||
|
||||
/// The first `charset` value, with HTTP linear whitespace trimmed.
|
||||
pub fn charset(&self) -> Option<&str> {
|
||||
self.parameter("charset").map(trim_http_lws)
|
||||
}
|
||||
|
||||
/// The first `boundary` value, with HTTP linear whitespace trimmed.
|
||||
pub fn boundary(&self) -> Option<&str> {
|
||||
self.parameter("boundary").map(trim_http_lws)
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses an HTTP response `Content-Type` like Chromium's
|
||||
/// `net::HttpUtil::ParseContentType` and `net::ParseMimeType` path.
|
||||
///
|
||||
/// This intentionally accepts malformed values seen on the network, including
|
||||
/// an unterminated quoted parameter and an empty subtype. Use
|
||||
/// [`crate::parse_mime_type`] for standards-facing Web API behavior.
|
||||
pub fn parse_response_content_type(input: &str) -> Option<ResponseContentType> {
|
||||
// Chromium treats an exact bare wildcard as meaningless, while retaining
|
||||
// wildcard values that have parameters or even trailing whitespace.
|
||||
if input == "*/*" {
|
||||
return None;
|
||||
}
|
||||
|
||||
let bytes = input.as_bytes();
|
||||
let input_len = bytes.len();
|
||||
let type_start = skip_http_lws(bytes, 0);
|
||||
let type_end = bytes[type_start..]
|
||||
.iter()
|
||||
.position(|byte| is_http_lws(*byte) || matches!(*byte, b';' | b'('))
|
||||
.map_or(input_len, |offset| type_start + offset);
|
||||
let slash = bytes.iter().position(|byte| *byte == b'/')?;
|
||||
if slash > type_end {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mime_type = input[type_start..type_end].to_ascii_lowercase();
|
||||
let mut parameters = Vec::new();
|
||||
let mut offset = find_from(bytes, type_end, |byte| byte == b';').unwrap_or(input_len);
|
||||
|
||||
while offset < input_len {
|
||||
offset = skip_http_lws(bytes, offset + 1);
|
||||
let name_start = offset;
|
||||
let Some(delimiter) = find_from(bytes, offset, |byte| matches!(byte, b';' | b'=')) else {
|
||||
break;
|
||||
};
|
||||
offset = delimiter;
|
||||
if bytes[offset] == b';' {
|
||||
continue;
|
||||
}
|
||||
|
||||
let name = input[name_start..offset].to_owned();
|
||||
offset = skip_http_lws(bytes, offset + 1);
|
||||
if offset >= input_len || bytes[offset] == b';' {
|
||||
continue;
|
||||
}
|
||||
|
||||
let value = if bytes[offset] == b'"' {
|
||||
let (value, next_offset) = parse_quoted_value(input, offset + 1);
|
||||
offset = find_from(bytes, next_offset, |byte| byte == b';').unwrap_or(input_len);
|
||||
value
|
||||
} else {
|
||||
let value_start = offset;
|
||||
offset = find_from(bytes, offset, |byte| byte == b';').unwrap_or(input_len);
|
||||
let value_end = trim_trailing_http_lws(bytes, value_start, offset);
|
||||
input[value_start..value_end].to_owned()
|
||||
};
|
||||
parameters.push((name, value));
|
||||
}
|
||||
|
||||
Some(ResponseContentType {
|
||||
mime_type,
|
||||
parameters,
|
||||
})
|
||||
}
|
||||
|
||||
fn parse_quoted_value(input: &str, mut offset: usize) -> (String, usize) {
|
||||
let bytes = input.as_bytes();
|
||||
let mut value = String::new();
|
||||
while offset < bytes.len() && bytes[offset] != b'"' {
|
||||
if bytes[offset] == b'\\' && offset + 1 < bytes.len() {
|
||||
offset += 1;
|
||||
}
|
||||
let character = input[offset..]
|
||||
.chars()
|
||||
.next()
|
||||
.expect("offset is inside the input");
|
||||
value.push(character);
|
||||
offset += character.len_utf8();
|
||||
}
|
||||
(value, offset)
|
||||
}
|
||||
|
||||
fn find_from(bytes: &[u8], start: usize, predicate: impl Fn(u8) -> bool) -> Option<usize> {
|
||||
bytes[start..]
|
||||
.iter()
|
||||
.position(|byte| predicate(*byte))
|
||||
.map(|offset| start + offset)
|
||||
}
|
||||
|
||||
fn skip_http_lws(bytes: &[u8], mut offset: usize) -> usize {
|
||||
while bytes.get(offset).is_some_and(|byte| is_http_lws(*byte)) {
|
||||
offset += 1;
|
||||
}
|
||||
offset
|
||||
}
|
||||
|
||||
fn trim_trailing_http_lws(bytes: &[u8], start: usize, mut end: usize) -> usize {
|
||||
while end > start && is_http_lws(bytes[end - 1]) {
|
||||
end -= 1;
|
||||
}
|
||||
end
|
||||
}
|
||||
|
||||
fn trim_http_lws(value: &str) -> &str {
|
||||
value.trim_matches(|character: char| matches!(character, ' ' | '\t' | '\r' | '\n'))
|
||||
}
|
||||
|
||||
fn is_http_lws(byte: u8) -> bool {
|
||||
matches!(byte, b' ' | b'\t' | b'\r' | b'\n')
|
||||
}
|
||||
@@ -0,0 +1,233 @@
|
||||
use super::*;
|
||||
|
||||
fn response_charset(input: &str) -> Option<String> {
|
||||
parse_response_content_type(input)?
|
||||
.charset()
|
||||
.map(str::to_owned)
|
||||
}
|
||||
|
||||
fn assert_response_content_type(
|
||||
input: &str,
|
||||
expected_mime_type: Option<&str>,
|
||||
expected_charset: Option<&str>,
|
||||
) {
|
||||
let parsed = parse_response_content_type(input);
|
||||
assert_eq!(
|
||||
parsed.as_ref().map(ResponseContentType::mime_type),
|
||||
expected_mime_type,
|
||||
"mime type for {input:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
parsed.as_ref().and_then(|value| value.charset()),
|
||||
expected_charset,
|
||||
"charset for {input:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn whatwg_parser_validates_the_mime_essence() {
|
||||
let parsed = parse_mime_type(" Text/Plain ; Charset=\"UTF-8\" ").unwrap();
|
||||
assert_eq!(parsed.type_(), "text");
|
||||
assert_eq!(parsed.subtype(), "plain");
|
||||
assert_eq!(parsed.essence(), "text/plain");
|
||||
assert_eq!(parsed.parameter("CHARSET"), Some("UTF-8"));
|
||||
assert_eq!(parsed.to_string(), "text/plain;charset=UTF-8");
|
||||
|
||||
assert!(parse_mime_type("garbage; charset=utf-8").is_none());
|
||||
assert!(parse_mime_type("text/").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn whatwg_parser_recovers_valid_parameters() {
|
||||
assert_eq!(
|
||||
mime_parameter("text/plain; title=\"alpha;beta\"", "title").as_deref(),
|
||||
Some("alpha;beta")
|
||||
);
|
||||
assert_eq!(
|
||||
mime_charset("text/plain; charset=; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
assert_eq!(
|
||||
mime_charset("text/plain; charset=\"\"; charset=gbk").as_deref(),
|
||||
Some("")
|
||||
);
|
||||
assert_eq!(
|
||||
mime_charset("text/plain; charset=\"utf-8").as_deref(),
|
||||
Some("utf-8")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_parser_matches_chromium_quoted_parameter_tolerances() {
|
||||
assert_eq!(
|
||||
response_charset("text/html; boundary=\"; charset=gbk\""),
|
||||
None
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; name=\"a\\\"; charset=gbk\"; charset=utf-8").as_deref(),
|
||||
Some("utf-8")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=\"\\utf\\-\\8\"").as_deref(),
|
||||
Some("utf-8")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=\"utf-8").as_deref(),
|
||||
Some("utf-8")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=\"\\\\\\\"\\").as_deref(),
|
||||
Some("\\\"\\")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_parser_does_not_reopen_quotes_after_a_value_started() {
|
||||
assert_eq!(
|
||||
response_charset("text/html; x=a=\"unterminated; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; x=\"ok\"junk=\"unterminated; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_parser_preserves_chromium_empty_and_duplicate_rules() {
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=\"\"; charset=gbk").as_deref(),
|
||||
Some("")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=foo; charset=utf-8").as_deref(),
|
||||
Some("foo")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_parser_preserves_parameter_name_and_quote_semantics() {
|
||||
assert_eq!(response_charset("text/html; charset =utf-8"), None);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset='utf-8'").as_deref(),
|
||||
Some("'utf-8'")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; \"; \"\"; charset=utf-8").as_deref(),
|
||||
Some("utf-8")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("text/html; charset=u\"tf-8\"").as_deref(),
|
||||
Some("u\"tf-8\"")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_parser_has_a_distinct_network_level_validity_boundary() {
|
||||
assert!(parse_response_content_type("garbage; charset=utf-8").is_none());
|
||||
assert!(parse_response_content_type("*/*").is_none());
|
||||
assert_eq!(
|
||||
parse_response_content_type("text/")
|
||||
.map(|parsed| parsed.mime_type().to_owned())
|
||||
.as_deref(),
|
||||
Some("text/")
|
||||
);
|
||||
assert_eq!(
|
||||
response_charset("*/*; charset=utf-8").as_deref(),
|
||||
Some("utf-8")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn response_parser_matches_chromium_http_util_regression_matrix() {
|
||||
for (input, mime_type, charset) in [
|
||||
("text/html", Some("text/html"), None),
|
||||
("text/html;", Some("text/html"), None),
|
||||
("text/html; charset=utf-8", Some("text/html"), Some("utf-8")),
|
||||
("text/html; charset =utf-8", Some("text/html"), None),
|
||||
(
|
||||
"text/html; charset= utf-8",
|
||||
Some("text/html"),
|
||||
Some("utf-8"),
|
||||
),
|
||||
(
|
||||
"text/html; charset=utf-8 ",
|
||||
Some("text/html"),
|
||||
Some("utf-8"),
|
||||
),
|
||||
("text/html; charset", Some("text/html"), None),
|
||||
("text/html; charset=", Some("text/html"), None),
|
||||
("text/html; charset= ", Some("text/html"), None),
|
||||
("text/html; charset= ;", Some("text/html"), None),
|
||||
("text/html; charset=\"\"", Some("text/html"), Some("")),
|
||||
("text/html; charset=\" \"", Some("text/html"), Some("")),
|
||||
(
|
||||
"text/html; charset=\" foo \"",
|
||||
Some("text/html"),
|
||||
Some("foo"),
|
||||
),
|
||||
(
|
||||
"text/html; charset=foo; charset=utf-8",
|
||||
Some("text/html"),
|
||||
Some("foo"),
|
||||
),
|
||||
(
|
||||
"text/html; charset; charset=; charset=utf-8",
|
||||
Some("text/html"),
|
||||
Some("utf-8"),
|
||||
),
|
||||
(
|
||||
"text/html; charset=utf-8; charset=; charset",
|
||||
Some("text/html"),
|
||||
Some("utf-8"),
|
||||
),
|
||||
(
|
||||
"text/html; \"; \"\"; charset=utf-8",
|
||||
Some("text/html"),
|
||||
Some("utf-8"),
|
||||
),
|
||||
(
|
||||
"text/html; charset=u\"tf-8\"",
|
||||
Some("text/html"),
|
||||
Some("u\"tf-8\""),
|
||||
),
|
||||
(
|
||||
"text/html; charset=\"utf-8",
|
||||
Some("text/html"),
|
||||
Some("utf-8"),
|
||||
),
|
||||
(
|
||||
"text/html; charset=\";charset=utf-8;\"",
|
||||
Some("text/html"),
|
||||
Some(";charset=utf-8;"),
|
||||
),
|
||||
(
|
||||
"text/html; charset='utf-8'",
|
||||
Some("text/html"),
|
||||
Some("'utf-8'"),
|
||||
),
|
||||
("text/", Some("text/"), None),
|
||||
("*/*", None, None),
|
||||
("*/*; charset=utf-8", Some("*/*"), Some("utf-8")),
|
||||
("*/* ", Some("*/*"), None),
|
||||
("teXT/html", Some("text/html"), None),
|
||||
] {
|
||||
assert_response_content_type(input, mime_type, charset);
|
||||
}
|
||||
|
||||
let boundary =
|
||||
parse_response_content_type("text/html; boundary=\"WebKit-ada-df-dsf-adsfadsfs \"")
|
||||
.unwrap();
|
||||
assert_eq!(boundary.boundary(), Some("WebKit-ada-df-dsf-adsfadsfs"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn normalizing_web_api_mime_types_rejects_non_http_bytes() {
|
||||
assert_eq!(normalize_web_api_mime_type("Text/Plain"), "text/plain");
|
||||
assert_eq!(normalize_web_api_mime_type("text/\nplain"), "");
|
||||
assert_eq!(normalize_web_api_mime_type("text/你好"), "");
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
use data_url::mime::Mime as WhatwgMime;
|
||||
use std::fmt;
|
||||
|
||||
/// A MIME type parsed according to the WHATWG MIME Sniffing Standard.
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct MimeType {
|
||||
inner: WhatwgMime,
|
||||
}
|
||||
|
||||
impl MimeType {
|
||||
/// The lower-case MIME top-level type.
|
||||
pub fn type_(&self) -> &str {
|
||||
&self.inner.type_
|
||||
}
|
||||
|
||||
/// The lower-case MIME subtype.
|
||||
pub fn subtype(&self) -> &str {
|
||||
&self.inner.subtype
|
||||
}
|
||||
|
||||
/// The lower-case `type/subtype` MIME essence.
|
||||
pub fn essence(&self) -> String {
|
||||
format!("{}/{}", self.type_(), self.subtype())
|
||||
}
|
||||
|
||||
/// Parsed parameters in header order.
|
||||
pub fn parameters(&self) -> &[(String, String)] {
|
||||
&self.inner.parameters
|
||||
}
|
||||
|
||||
/// The first parameter with the given ASCII-case-insensitive name.
|
||||
pub fn parameter(&self, name: &str) -> Option<&str> {
|
||||
self.inner
|
||||
.parameters
|
||||
.iter()
|
||||
.find(|(candidate, _)| candidate.eq_ignore_ascii_case(name))
|
||||
.map(|(_, value)| value.as_str())
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Display for MimeType {
|
||||
fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
self.inner.fmt(formatter)
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses a MIME value using WHATWG validation and parameter recovery rules.
|
||||
pub fn parse_mime_type(input: &str) -> Option<MimeType> {
|
||||
input.parse().ok().map(|inner| MimeType { inner })
|
||||
}
|
||||
|
||||
/// Returns the MIME essence of a valid WHATWG MIME value.
|
||||
pub fn mime_essence(input: &str) -> Option<String> {
|
||||
parse_mime_type(input).map(|mime| mime.essence())
|
||||
}
|
||||
|
||||
/// Returns one parsed MIME parameter by ASCII-case-insensitive name.
|
||||
pub fn mime_parameter(input: &str, name: &str) -> Option<String> {
|
||||
parse_mime_type(input)?.parameter(name).map(str::to_owned)
|
||||
}
|
||||
|
||||
/// Returns the parsed `charset` MIME parameter.
|
||||
pub fn mime_charset(input: &str) -> Option<String> {
|
||||
mime_parameter(input, "charset")
|
||||
}
|
||||
|
||||
/// Normalizes a Web API MIME string without parsing its structure.
|
||||
pub fn normalize_web_api_mime_type(raw: &str) -> String {
|
||||
if raw.is_empty() || raw.bytes().any(|byte| !(0x20..=0x7e).contains(&byte)) {
|
||||
return String::new();
|
||||
}
|
||||
raw.to_ascii_lowercase()
|
||||
}
|
||||
@@ -7,7 +7,7 @@ edition = "2024"
|
||||
[dependencies]
|
||||
encoding_rs = "0.8"
|
||||
moli-charset-parser = { path = "../moli-charset-parser" }
|
||||
moli-header-field = { path = "../moli-header-field" }
|
||||
moli-content-type = { path = "../moli-content-type" }
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
use encoding_rs::Encoding;
|
||||
use moli_header_field::{split_outside_quoted_strings, unquote_parameter_value};
|
||||
use moli_content_type::parse_response_content_type;
|
||||
|
||||
pub fn encoding_for_label(label: &str) -> Option<&'static Encoding> {
|
||||
Encoding::for_label(label.trim().as_bytes())
|
||||
@@ -7,36 +7,13 @@ pub fn encoding_for_label(label: &str) -> Option<&'static Encoding> {
|
||||
|
||||
/// The `charset` parameter of a `Content-Type` value.
|
||||
///
|
||||
/// Parameters are separated by the `;` characters outside a quoted string, and
|
||||
/// a quoted value has its quoting backslashes removed, so a `charset` written
|
||||
/// inside another parameter's quoted string is not read as a parameter here.
|
||||
/// Response metadata is parsed with Chromium network-layer tolerances before
|
||||
/// the label is handed to `encoding_rs`. In particular, quoted separators stay
|
||||
/// inside their parameter and unterminated quoted values remain recoverable.
|
||||
pub fn charset_from_content_type(value: &str) -> Option<String> {
|
||||
// The first segment is the media type itself, not a parameter.
|
||||
for parameter in split_outside_quoted_strings(value, ';').into_iter().skip(1) {
|
||||
let Some((name, parameter_value)) = parameter.split_once('=') else {
|
||||
continue;
|
||||
};
|
||||
if !name.trim().eq_ignore_ascii_case("charset") {
|
||||
continue;
|
||||
}
|
||||
let parameter_value = parameter_value.trim();
|
||||
let charset = if parameter_value.starts_with('"') {
|
||||
// Delimiters are already gone, so any `"` left is data produced by
|
||||
// an escaped quote and must not be trimmed away.
|
||||
unquote_parameter_value(parameter_value).into_owned()
|
||||
} else {
|
||||
// Apostrophe delimiters are not a quoted string, but receivers
|
||||
// have long tolerated them here, so keep stripping them.
|
||||
parameter_value
|
||||
.trim_matches(|ch| ch == '"' || ch == '\'')
|
||||
.trim()
|
||||
.to_owned()
|
||||
};
|
||||
if !charset.is_empty() {
|
||||
return Some(charset);
|
||||
}
|
||||
}
|
||||
None
|
||||
let parsed = parse_response_content_type(value)?;
|
||||
let charset = parsed.charset()?;
|
||||
(!charset.is_empty()).then(|| charset.to_owned())
|
||||
}
|
||||
|
||||
pub fn charset_from_headers(headers: &[(String, String)]) -> Option<String> {
|
||||
|
||||
@@ -705,15 +705,13 @@ fn header_charset_removes_quoting_backslashes() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_charset_keeps_its_existing_tolerances() {
|
||||
fn header_charset_matches_chromium_network_tolerances() {
|
||||
for header in [
|
||||
"text/html; charset=utf-8",
|
||||
"text/html;charset=utf-8",
|
||||
"TEXT/HTML; CHARSET=UTF-8",
|
||||
"text/html; charset = utf-8 ",
|
||||
"text/html; charset=\"utf-8\"",
|
||||
"text/html;charset=utf-8;",
|
||||
"text/html; charset='utf-8'",
|
||||
"text/html; charset=\"utf-8",
|
||||
] {
|
||||
assert_eq!(
|
||||
@@ -728,12 +726,50 @@ fn header_charset_keeps_its_existing_tolerances() {
|
||||
assert_eq!(charset_from_content_type("text/html"), None);
|
||||
assert_eq!(charset_from_content_type("text/html; charset="), None);
|
||||
assert_eq!(charset_from_content_type("charset=utf-8"), None);
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; charset = utf-8"),
|
||||
None
|
||||
);
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; charset='utf-8'").as_deref(),
|
||||
Some("'utf-8'")
|
||||
);
|
||||
assert!(
|
||||
charset_from_content_type("text/html; charset='utf-8'")
|
||||
.as_deref()
|
||||
.and_then(encoding_for_label)
|
||||
.is_none()
|
||||
);
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; charset=gbk; boundary=x").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_charset_preserves_chromium_empty_parameter_precedence() {
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; charset=; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; charset=\"\"; charset=gbk"),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_charset_does_not_reopen_quotes_after_a_value_started() {
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; x=a=\"unterminated; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
assert_eq!(
|
||||
charset_from_content_type("text/html; x=\"ok\"junk=\"unterminated; charset=gbk").as_deref(),
|
||||
Some("gbk")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_charset_recovers_after_a_stray_quote() {
|
||||
// WPT MIME case: the `"` does not open a parameter value, so the following
|
||||
|
||||
@@ -1,5 +1,15 @@
|
||||
use std::borrow::Cow;
|
||||
|
||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||
enum ParameterScanState {
|
||||
#[default]
|
||||
BeforeValue,
|
||||
AtValueStart,
|
||||
UnquotedValue,
|
||||
QuotedValue,
|
||||
AfterQuotedValue,
|
||||
}
|
||||
|
||||
/// Splits `value` on every `separator` that is not inside a quoted parameter
|
||||
/// value.
|
||||
///
|
||||
@@ -25,14 +35,13 @@ pub fn split_outside_quoted_strings(value: &str, separator: char) -> Vec<&str> {
|
||||
let bytes = value.as_bytes();
|
||||
let mut segments = Vec::new();
|
||||
let mut segment_start = 0;
|
||||
let mut at_value_start = false;
|
||||
let mut inside_quotes = false;
|
||||
let mut state = ParameterScanState::BeforeValue;
|
||||
let mut index = 0;
|
||||
|
||||
while index < bytes.len() {
|
||||
let byte = bytes[index];
|
||||
|
||||
if inside_quotes {
|
||||
if state == ParameterScanState::QuotedValue {
|
||||
if byte == b'\\' {
|
||||
// Step over the escaped byte so a quoted `\"` does not close
|
||||
// the value.
|
||||
@@ -40,7 +49,7 @@ pub fn split_outside_quoted_strings(value: &str, separator: char) -> Vec<&str> {
|
||||
continue;
|
||||
}
|
||||
if byte == b'"' {
|
||||
inside_quotes = false;
|
||||
state = ParameterScanState::AfterQuotedValue;
|
||||
}
|
||||
index += 1;
|
||||
continue;
|
||||
@@ -49,16 +58,15 @@ pub fn split_outside_quoted_strings(value: &str, separator: char) -> Vec<&str> {
|
||||
if byte == separator {
|
||||
segments.push(&value[segment_start..index]);
|
||||
segment_start = index + 1;
|
||||
at_value_start = false;
|
||||
} else if byte == b'=' {
|
||||
at_value_start = true;
|
||||
} else if byte == b'"' && at_value_start {
|
||||
inside_quotes = true;
|
||||
at_value_start = false;
|
||||
} else if !matches!(byte, b' ' | b'\t') || !at_value_start {
|
||||
// Whitespace between `=` and the value keeps the value still to
|
||||
// come; anything else means the value was not quoted.
|
||||
at_value_start = false;
|
||||
state = ParameterScanState::BeforeValue;
|
||||
} else {
|
||||
state = match state {
|
||||
ParameterScanState::BeforeValue if byte == b'=' => ParameterScanState::AtValueStart,
|
||||
ParameterScanState::AtValueStart if matches!(byte, b' ' | b'\t') => state,
|
||||
ParameterScanState::AtValueStart if byte == b'"' => ParameterScanState::QuotedValue,
|
||||
ParameterScanState::AtValueStart => ParameterScanState::UnquotedValue,
|
||||
_ => state,
|
||||
};
|
||||
}
|
||||
index += 1;
|
||||
}
|
||||
|
||||
@@ -224,3 +224,15 @@ fn a_trailing_backslash_is_kept_rather_than_dropped() {
|
||||
fn an_escaped_quote_survives_as_data() {
|
||||
assert_eq!(unquote_parameter_value("\"utf-8\\\"\""), "utf-8\"");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn equals_inside_a_value_does_not_reopen_quoted_string_mode() {
|
||||
assert_eq!(
|
||||
split_outside_quoted_strings("token=a=\"unterminated, next=value", ','),
|
||||
vec!["token=a=\"unterminated", " next=value"]
|
||||
);
|
||||
assert_eq!(
|
||||
split_outside_quoted_strings("token=\"ok\"junk=\"unterminated, next=value", ','),
|
||||
vec!["token=\"ok\"junk=\"unterminated", " next=value"]
|
||||
);
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@ content_disposition = "0.4.0"
|
||||
data-url = "0.3.2"
|
||||
http = "1"
|
||||
mime = "0.3"
|
||||
moli-content-type = { path = "../moli-content-type" }
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -1,42 +1,16 @@
|
||||
use data_url::mime::Mime as WebMime;
|
||||
use moli_content_type::parse_mime_type;
|
||||
|
||||
pub use moli_content_type::{
|
||||
mime_charset, mime_essence, mime_parameter, normalize_web_api_mime_type,
|
||||
};
|
||||
|
||||
pub fn parse_mime(input: &str) -> Option<mime::Mime> {
|
||||
parse_web_mime(input).and_then(|mime| mime.to_string().parse().ok())
|
||||
}
|
||||
|
||||
pub fn mime_essence(input: &str) -> Option<String> {
|
||||
parse_web_mime(input).map(|mime| mime_essence_from_parsed(&mime))
|
||||
parse_mime_type(input).and_then(|mime| mime.to_string().parse().ok())
|
||||
}
|
||||
|
||||
pub fn request_header_content_type_essence(input: &str) -> Option<String> {
|
||||
if input.contains(',') {
|
||||
return None;
|
||||
}
|
||||
parse_web_mime(input).map(|mime| mime_essence_from_parsed(&mime))
|
||||
}
|
||||
|
||||
pub fn mime_charset(input: &str) -> Option<String> {
|
||||
mime_parameter(input, "charset")
|
||||
}
|
||||
|
||||
pub fn mime_parameter(input: &str, name: &str) -> Option<String> {
|
||||
let name = name.to_ascii_lowercase();
|
||||
parse_web_mime(input)?
|
||||
.get_parameter(&name)
|
||||
.map(str::to_owned)
|
||||
}
|
||||
|
||||
pub fn normalize_web_api_mime_type(raw: &str) -> String {
|
||||
if raw.is_empty() || raw.bytes().any(|byte| !(0x20..=0x7e).contains(&byte)) {
|
||||
return String::new();
|
||||
}
|
||||
raw.to_ascii_lowercase()
|
||||
}
|
||||
|
||||
fn parse_web_mime(input: &str) -> Option<WebMime> {
|
||||
input.parse().ok()
|
||||
}
|
||||
|
||||
fn mime_essence_from_parsed(mime: &WebMime) -> String {
|
||||
format!("{}/{}", mime.type_, mime.subtype)
|
||||
mime_essence(input)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user