From 337e4fbedab2f75bada6f374d86187f418ee7ec0 Mon Sep 17 00:00:00 2001 From: ldm0 Date: Wed, 26 Aug 2026 17:43:07 +0800 Subject: [PATCH] fix(encoding): match Blink exact charset lookup --- moli-content-type/src/lib.rs | 21 ++++++++++++++----- moli-content-type/src/response.rs | 8 +++++--- moli-content-type/src/whatwg.rs | 15 ++++++++++++-- moli-encoding/src/labels.rs | 15 +++++++++++++- moli-encoding/src/tests.rs | 34 +++++++++++++++++++++++++++++++ 5 files changed, 82 insertions(+), 11 deletions(-) diff --git a/moli-content-type/src/lib.rs b/moli-content-type/src/lib.rs index 22a4b8d022..bf53e96d72 100644 --- a/moli-content-type/src/lib.rs +++ b/moli-content-type/src/lib.rs @@ -1,9 +1,20 @@ -//! Shared parsing for MIME values and HTTP response `Content-Type` fields. +//! MIME parsing for two distinct browser contexts. //! -//! Web-facing MIME operations use the WHATWG parser, while transport response -//! metadata uses Chromium's deliberately more tolerant network semantics. The -//! two entry points are named separately so callers cannot select the wrong -//! behavior through a generic leniency flag. +//! Choose the entry point from the source of the value: +//! +//! - [`parse_mime_type`] applies the WHATWG MIME parsing and serialization +//! rules. Use it for standards-facing MIME operations such as Blob/File type +//! handling, MIME classification, and interpreting an already selected MIME +//! value. +//! - [`parse_response_content_type`] matches Chromium's network-layer parsing +//! of a raw HTTP response `Content-Type` field. Use it when transport +//! metadata must retain Chromium behavior for values such as `charset` and +//! multipart `boundary`. +//! +//! These parsers intentionally have different validity and recovery rules. A +//! value accepted by one is not necessarily accepted by the other, so callers +//! must select the parser from the value's origin rather than treating the two +//! result types as interchangeable. mod response; mod whatwg; diff --git a/moli-content-type/src/response.rs b/moli-content-type/src/response.rs index 3a230ec1bb..718a650eae 100644 --- a/moli-content-type/src/response.rs +++ b/moli-content-type/src/response.rs @@ -35,12 +35,14 @@ impl ResponseContentType { } } -/// Parses an HTTP response `Content-Type` like Chromium's +/// Parses a raw HTTP response `Content-Type` field like Chromium's /// `net::HttpUtil::ParseContentType` and `net::ParseMimeType` path. /// /// This intentionally accepts malformed values seen on the network, including -/// an unterminated quoted parameter and an empty subtype. Use -/// [`crate::parse_mime_type`] for standards-facing Web API behavior. +/// an unterminated quoted parameter and an empty subtype. Use this only for +/// response transport metadata, such as extracting `charset` or multipart +/// `boundary`; standards-facing MIME operations must use +/// [`crate::parse_mime_type`]. pub fn parse_response_content_type(input: &str) -> Option { // Chromium treats an exact bare wildcard as meaningless, while retaining // wildcard values that have parameters or even trailing whitespace. diff --git a/moli-content-type/src/whatwg.rs b/moli-content-type/src/whatwg.rs index eeb4343dae..860bb16934 100644 --- a/moli-content-type/src/whatwg.rs +++ b/moli-content-type/src/whatwg.rs @@ -44,7 +44,13 @@ impl fmt::Display for MimeType { } } -/// Parses a MIME value using WHATWG validation and parameter recovery rules. +/// Parses a standards-facing MIME value using WHATWG validation, +/// normalization, and parameter recovery rules. +/// +/// Use this for Web API MIME operations and for interpreting an already +/// selected MIME value. Raw HTTP response `Content-Type` metadata must instead +/// use [`crate::parse_response_content_type`] so its Chromium network behavior +/// is preserved. pub fn parse_mime_type(input: &str) -> Option { input.parse().ok().map(|inner| MimeType { inner }) } @@ -64,7 +70,12 @@ pub fn mime_charset(input: &str) -> Option { mime_parameter(input, "charset") } -/// Normalizes a Web API MIME string without parsing its structure. +/// Normalizes a Blob/File-style Web API MIME string without parsing its +/// structure. +/// +/// This only validates the permitted byte range and lowercases ASCII. It does +/// not produce a [`MimeType`] and is not a substitute for +/// [`parse_mime_type`]. pub fn normalize_web_api_mime_type(raw: &str) -> String { if raw.is_empty() || raw.bytes().any(|byte| !(0x20..=0x7e).contains(&byte)) { return String::new(); diff --git a/moli-encoding/src/labels.rs b/moli-encoding/src/labels.rs index 133d4ce540..d59538d824 100644 --- a/moli-encoding/src/labels.rs +++ b/moli-encoding/src/labels.rs @@ -1,8 +1,21 @@ use encoding_rs::Encoding; use moli_content_type::parse_response_content_type; +/// Looks up an encoding label after the caller has applied its own whitespace +/// rules, matching Blink's exact `TextEncoding` registry lookup. pub fn encoding_for_label(label: &str) -> Option<&'static Encoding> { - Encoding::for_label(label.trim().as_bytes()) + let bytes = label.as_bytes(); + // `encoding_rs` implements the Encoding Standard's preprocessing and + // trims ASCII whitespace itself. Blink's response path has already + // removed HTTP LWS, so accepting anything still present here (notably VT + // and FF) would turn an invalid response charset into a valid label. + if bytes.first().is_some_and(|byte| byte.is_ascii_whitespace()) + || bytes.last().is_some_and(|byte| byte.is_ascii_whitespace()) + { + return None; + } + + Encoding::for_label(bytes) } /// The `charset` parameter of a `Content-Type` value. diff --git a/moli-encoding/src/tests.rs b/moli-encoding/src/tests.rs index 2cf1f56f85..4b639069e7 100644 --- a/moli-encoding/src/tests.rs +++ b/moli-encoding/src/tests.rs @@ -746,6 +746,40 @@ fn header_charset_matches_chromium_network_tolerances() { ); } +#[test] +fn header_charset_does_not_trim_non_http_ascii_whitespace() { + for whitespace in ['\u{000b}', '\u{000c}'] { + let headers = vec![( + "Content-Type".to_owned(), + format!("text/html; charset={whitespace}gbk"), + )]; + + assert_eq!( + charset_from_headers(&headers).as_deref(), + Some(format!("{whitespace}gbk").as_str()) + ); + assert_eq!( + decode_html_document(&gbk_bytes("太平洋"), &headers).1, + "windows-1252" + ); + } +} + +#[test] +fn encoding_label_lookup_requires_callers_to_preprocess_whitespace() { + assert_eq!(encoding_for_label("gbk"), Some(encoding_rs::GBK)); + for label in [ + " gbk", + "gbk ", + "\tgbk", + "gbk\n", + "\u{000b}gbk", + "gbk\u{000c}", + ] { + assert_eq!(encoding_for_label(label), None, "label={label:?}"); + } +} + #[test] fn header_charset_preserves_chromium_empty_parameter_precedence() { assert_eq!(