fix(url): align hierarchical and opaque path encoding

Encode carets in hierarchical paths and encode only the final opaque-path space before query or fragment delimiters. Preserve the encoded path across suffix mutations and remove the obsolete parser trimming and V8 query-removal workaround.

Add five shared-parser and three VM regressions covering constructors, setters, URLSearchParams, DOM links, and cross-realm behavior. Synchronize 11 stale vendored WPT expected records with pinned current WPT data; inputs, the WPT driver, and the remaining expected-failure entry are unchanged.

Validation: cargo fmt --all; cargo clippy --workspace --all-targets --all-features -- -D warnings; cargo nextest run --no-fail-fast (17632 passed, 13 skipped). Vendored all-features unit tests, WPT driver, and 67 doctests pass.

On the same 50 URL/Location WPT cases, CLI and CDP both improve from 42 to 48 passing cases and from 7036 to 7085 passing subtests (+49), with no new failures. The two remaining failures are the existing webkitURL alias and srcdoc scrolling cases. Move the verified non-file anchor case to passed.

CDP smoke: 42/43 groups pass. Puppeteer cannot run because node is absent, reproduced with the before binary.

Artifacts: target/wpt-url-encoding-20260908-AP6CI1. WPT revision: db95fafd1fcef8428805e41eb5705d444e8c67ce. Release SHA256: f132f0afe7f49d627ed9b5d397b1d5b71d80432180b08b1d076aa67637efdd65.
This commit is contained in:
ldm0
2026-09-14 15:06:20 +08:00
parent 6a41ddc1af
commit 223f93ccbc
13 changed files with 279 additions and 112 deletions
@@ -3316,7 +3316,6 @@ trusted-types/trusted-types-reporting-for-SharedWorker-ServiceWorkerContainer-re
trusted-types/trusted-types-secondary-document.html
uievents/interface/click-event.htm
uievents/order-of-events/focus-events/focus-automated-blink-webkit.html
url/a-element.html?exclude=(file|javascript|mailto)
viewport/viewport-segments.html
wasm/core/memory64/table64.wast.js.html
wasm/jsapi/proto-from-ctor-realm.html
@@ -8480,6 +8480,7 @@ uievents/legacy/Event-subclasses-init.html
uievents/textInput/api.html
uievents/ui_event_pseudo_target.html
url/a-element-origin.html
url/a-element.html?exclude=(file|javascript|mailto)
url/a-element.html?include=file
url/a-element.html?include=javascript
url/a-element.html?include=mailto
@@ -1,6 +1,5 @@
use super::helpers::{
can_parse_url_input, constructor_url_href, require_url_receiver, resolve_url_constructor_input,
url_href_slot,
can_parse_url_input, require_url_receiver, resolve_url_constructor_input, url_href_slot,
};
use super::*;
use crate::util::get_private_value;
@@ -73,7 +72,7 @@ pub(super) fn url_constructor_callback<'s>(
};
let this = args.this();
let href = constructor_url_href(&parsed.input, &url);
let href = url.as_str().to_owned();
let has_search_params = get_private_value(scope, this, URL_SEARCH_PARAMS_SLOT)
.is_some_and(|value| !value.is_undefined());
let search_params = if has_search_params {
@@ -74,47 +74,7 @@ pub(in crate::context_bootstrap) fn apply_url_update<'s>(
object: v8::Local<'s, v8::Object>,
url: &url::Url,
) {
let href = url_href_slot(scope, object)
.and_then(|old_href| href_after_opaque_query_removal(&old_href, url))
.unwrap_or_else(|| url.as_str().to_owned());
if let Some(href) = v8_string(scope, &href) {
if let Some(href) = v8_string(scope, url.as_str()) {
set_private_value(scope, object, URL_HREF_SLOT, href.into());
}
}
pub(super) fn constructor_url_href(input: &str, url: &url::Url) -> String {
if let Some(href) = href_after_opaque_query_removal(input, url) {
return href;
}
url.as_str().to_owned()
}
fn href_after_opaque_query_removal(old_href: &str, url: &url::Url) -> Option<String> {
if !url.cannot_be_a_base() || url.query().is_some() {
return None;
}
let old_url = url::Url::parse(old_href).ok()?;
old_url.query()?;
let old_path = raw_opaque_path(old_href)?;
let encoded_path = encode_final_trailing_space(old_path)?;
let mut href = format!("{}:{encoded_path}", url.scheme());
if let Some(fragment) = url.fragment() {
href.push('#');
href.push_str(fragment);
}
Some(href)
}
fn raw_opaque_path(href: &str) -> Option<&str> {
let (_, after_scheme) = href.split_once(':')?;
let end = after_scheme.find(['?', '#']).unwrap_or(after_scheme.len());
Some(&after_scheme[..end])
}
fn encode_final_trailing_space(path: &str) -> Option<String> {
let prefix = path.strip_suffix(' ')?;
let mut encoded = String::with_capacity(path.len() + 2);
encoded.push_str(prefix);
encoded.push_str("%20");
Some(encoded)
}
@@ -50,6 +50,88 @@ return 'ok';
assert_eq!(result, "ok");
}
#[test]
fn url_components_encode_carets_only_in_hierarchical_paths() {
assert_url_components(
r#"
for (const [name, create] of factories) {
for (const prefix of ['https://host', 'file://', 'custom:', 'custom://host']) {
const object = create(prefix + '/a^b/%5e?^#^');
assert(object.href === prefix + '/a%5Eb/%5e?^#^', name + ': path caret encoded once');
assert(object.pathname === '/a%5Eb/%5e', name + ': pathname uses the path encode set');
object.pathname = '/^/%5e';
assert(object.href === prefix + '/%5E/%5e?^#^', name + ': pathname setter shares encoding');
object.search = '^';
object.hash = '^';
assert(object.search === '?^' && object.hash === '#^', name + ': suffix carets are literal');
}
const opaque = create('data:a b^c?^#^');
assert(opaque.href === 'data:a b^c?^#^', name + ': opaque path uses its own encode set');
opaque.pathname = '/replacement^';
assert(opaque.pathname === 'a b^c', name + ': opaque pathname is not mutable');
}
"#,
);
}
#[test]
fn url_components_encode_the_final_opaque_space_during_parsing() {
assert_url_components(
r#"
for (const [name, create] of factories) {
for (const [input, path] of [
['payload ', 'payload%20'], ['payload ', 'payload %20'],
['payload \t\n\r', 'payload%20'], ['payload \t \n \r', 'payload %20'],
['payload%20', 'payload%20'], ['payload %3F', 'payload %3F'],
]) {
for (const suffix of ['?', '#', '?q', '#h', '?q#h', '?#']) {
const object = create('data:' + input + suffix);
const expected = 'data:' + path + suffix;
assert(object.href === expected, name + ': canonical href at construction: ' + expected);
assert(object.pathname === path, name + ': canonical path before any setter');
assert(new URL(object.href).href === expected, name + ': serialization is stable');
object.href = 'custom:' + input + suffix;
assert(object.href === 'custom:' + path + suffix, name + ': href replacement shares parsing');
}
}
}
"#,
);
}
#[test]
fn url_components_opaque_path_is_stable_across_suffix_and_search_params_mutations() {
assert_url_components(
r#"
const base = 'data:payload %20';
for (const [name, create] of factories) {
for (const first of ['search', 'hash']) {
const object = create('data:payload ?q#h');
const params = object.searchParams;
object[first] = '';
assert(object.href === base + (first === 'search' ? '#h' : '?q'), name + ': clearing one suffix');
object[first === 'search' ? 'hash' : 'search'] = '';
assert(object.href === base, name + ': clearing both suffixes preserves the path');
object.search = '?';
object.hash = '#';
assert(object.href === base + '?#', name + ': empty suffixes keep their delimiters');
if (params) {
assert(object.searchParams === params, name + ': setters retain searchParams identity');
params.append('q', 'value');
assert(object.href === base + '?q=value#', name + ': append does not modify path');
params.delete('q');
assert(object.href === base + '#', name + ': delete removes only the query');
params.sort();
assert(object.href === base + '#', name + ': sorting empty params preserves the path');
object.hash = '';
assert(object.href === base, name + ': removing the final suffix preserves the path');
}
}
}
"#,
);
}
#[test]
fn url_components_empty_getters_preserve_href() {
assert_url_components(
+2
View File
@@ -6,6 +6,8 @@ pub mod search_params;
mod file_url;
#[cfg(test)]
mod hierarchical_path;
#[cfg(test)]
mod path_encoding;
pub use origin::{
WebOrigin, is_about_blank, is_opaque_origin, is_potentially_trustworthy_url,
+118
View File
@@ -0,0 +1,118 @@
use url::Url;
fn assert_url(url: &Url, expected: &str, path: &str) {
assert_eq!(url.as_str(), expected);
assert_eq!(url.path(), path);
url.check_invariants().unwrap();
assert_eq!(Url::parse(url.as_str()).unwrap(), *url);
}
#[test]
fn hierarchical_paths_encode_carets_without_changing_other_components() {
for prefix in [
"http://host",
"https://host",
"ws://host",
"wss://host",
"ftp://host",
"file://",
"custom:",
"custom://",
"custom://host",
] {
let url = Url::parse(&format!("{prefix}/a^b/%5e?q=^#^")).unwrap();
assert_url(&url, &format!("{prefix}/a%5Eb/%5e?q=^#^"), "/a%5Eb/%5e");
assert_eq!(url.query(), Some("q=^"));
assert_eq!(url.fragment(), Some("^"));
}
let url = Url::parse("https://u^:p^@host/^?^#^").unwrap();
assert_url(&url, "https://u%5E:p%5E@host/%5E?^#^", "/%5E");
}
#[test]
fn hierarchical_path_mutations_share_caret_encoding() {
for prefix in ["https://host", "file://", "custom:", "custom://host"] {
let base = Url::parse(&format!("{prefix}/old?^#^")).unwrap();
assert_url(
&base.join("^/%5e?^#^").unwrap(),
&format!("{prefix}/%5E/%5e?^#^"),
"/%5E/%5e",
);
let mut url = base.clone();
url.set_path("/a^b/%5e");
assert_url(&url, &format!("{prefix}/a%5Eb/%5e?^#^"), "/a%5Eb/%5e");
crate::components::set_pathname(&mut url, "/^/%5e");
assert_url(&url, &format!("{prefix}/%5E/%5e?^#^"), "/%5E/%5e");
url.path_segments_mut()
.unwrap()
.clear()
.push("^")
.push("%5e");
// Unlike set_path, path_segments_mut accepts unencoded segments.
assert_url(&url, &format!("{prefix}/%5E/%255e?^#^"), "/%5E/%255e");
}
}
#[test]
fn opaque_paths_encode_only_the_last_space_before_a_query_or_fragment() {
for scheme in ["data", "mailto", "non-special"] {
for (input, path) in [
("payload ", "payload%20"),
("payload ", "payload %20"),
("payload \t\r\n", "payload%20"),
("payload \t \n \r", "payload %20"),
(" ", "%20"),
("payload%20", "payload%20"),
("payload \u{a0}", "payload %C2%A0"),
("payload %3F", "payload %3F"),
] {
for suffix in ["?", "#", "?q", "#h", "?q#h", "?#"] {
let url = Url::parse(&format!("{scheme}:{input}{suffix}")).unwrap();
assert_url(&url, &format!("{scheme}:{path}{suffix}"), path);
assert!(url.cannot_be_a_base());
}
}
}
}
#[test]
fn opaque_paths_preserve_interior_spaces_carets_and_encoded_delimiters() {
for (input, expected, path) in [
("data:a b^c?^#^", "data:a b^c?^#^", "a b^c"),
("data:a %3F %23?^#^", "data:a %3F %23?^#^", "a %3F %23"),
("data:a%20?^#^", "data:a%20?^#^", "a%20"),
(" \tdata:a b^c \r\n", "data:a b^c", "a b^c"),
("data:space ?a #b ", "data:space%20?a%20#b", "space%20"),
] {
assert_url(&Url::parse(input).unwrap(), expected, path);
}
}
#[test]
fn clearing_query_and_fragment_preserves_the_encoded_opaque_path_in_either_order() {
for input in ["data:payload ?q#h", "data:payload %20?q#h"] {
for query_first in [false, true] {
let mut url = Url::parse(input).unwrap();
if query_first {
url.set_query(None);
assert_url(&url, "data:payload %20#h", "payload %20");
url.set_fragment(None);
} else {
url.set_fragment(None);
assert_url(&url, "data:payload %20?q", "payload %20");
url.set_query(None);
}
assert_url(&url, "data:payload %20", "payload %20");
assert_url(
&url.join("#next").unwrap(),
"data:payload %20#next",
"payload %20",
);
url.set_query(Some(""));
url.set_fragment(Some(""));
assert_url(&url, "data:payload %20?#", "payload %20");
url.query_pairs_mut().clear().append_pair("next", "value");
assert_url(&url, "data:payload %20?next=value#", "payload %20");
}
}
}
+34 -8
View File
@@ -48,11 +48,37 @@ References:
<https://url.spec.whatwg.org/#concept-url-serializer>, and
<https://url.spec.whatwg.org/#dom-url-pathname>.
The upstream WPT data is unchanged. The 39 file parsing/setter tests and seven
hierarchical setter tests fixed by these patches were removed from
`tests/expected_failures.txt`; the one remaining upstream expected failure is
retained. The obsolete unit-test expectation of a host/drive-letter syntax
violation is replaced with an explicit preservation and no-violation assertion.
Path percent-encoding also follows the current URL Standard:
- Include "^" in the hierarchical path encode set, including path setters and
path-segment mutation, without changing opaque paths, queries, or fragments.
- When parsing an opaque path, encode only the last space immediately before
"?" or "#". Lookahead ignores ASCII tabs and newlines just as parsing does.
- Preserve that encoded path when removing a query or fragment. The obsolete
trailing-space stripping algorithm and the V8-only query-removal workaround
are no longer needed.
References:
<https://url.spec.whatwg.org/#path-percent-encode-set>,
<https://url.spec.whatwg.org/#cannot-be-a-base-url-path-state>,
<https://url.spec.whatwg.org/#dom-url-search>, and
<https://url.spec.whatwg.org/#dom-url-hash>.
The normative snapshot used for this update is
<https://url.spec.whatwg.org/commit-snapshots/55d6699373ba68a16ec182f34222a74ed8bc3dac/>.
The 39 file parsing/setter tests and seven hierarchical setter tests fixed by
these patches were removed from `tests/expected_failures.txt`; the one remaining
upstream expected failure is retained. The obsolete unit-test expectations of
a host/drive-letter syntax violation and opaque-path space stripping are
replaced with explicit preservation assertions.
Eleven vendored WPT records (two parsing cases and nine setter cases) had
expectations predating the current caret and opaque-space rules. Their expected
results are synchronized with the corresponding records in
<https://github.com/web-platform-tests/wpt/blob/258f285de043b79e44324228c0fd800b38d21879/url/resources/urltestdata.json>
and
<https://github.com/web-platform-tests/wpt/blob/258f285de043b79e44324228c0fd800b38d21879/url/resources/setters_tests.json>.
Their inputs, all other expected results, and the WPT driver are unchanged.
Packaging adjustment: `debug_metadata/url.natvis` is copied from the same pinned
upstream revision, and its include path is made package-local so the
@@ -66,7 +92,7 @@ cargo test --manifest-path vendor/url-2.5.8/Cargo.toml --all-features
Workspace regressions live in `moli-url/src/file_url.rs`,
`moli-url/src/hierarchical_path.rs`,
`moli-url/src/path_encoding.rs`,
`moli-renderer-v8/src/script_vm/tests/url_components.rs`, and
`moli-url-policy/src/tests.rs`. Remaining opaque-path and percent-encoding
conformance issues are separate work; these patches do not change resource
access permissions.
`moli-url-policy/src/tests.rs`. These patches do not change resource access
permissions.
-30
View File
@@ -384,32 +384,6 @@ impl Url {
url
}
/// https://url.spec.whatwg.org/#potentially-strip-trailing-spaces-from-an-opaque-path
fn strip_trailing_spaces_from_opaque_path(&mut self) {
if !self.cannot_be_a_base() {
return;
}
if self.fragment_start.is_some() {
return;
}
if self.query_start.is_some() {
return;
}
let trailing_space_count = self
.serialization
.chars()
.rev()
.take_while(|c| *c == ' ')
.count();
let start = self.serialization.len() - trailing_space_count;
self.serialization.truncate(start);
}
/// Parse a string as an URL, with this URL as the base URL.
///
/// The inverse of this is [`make_relative`].
@@ -1592,7 +1566,6 @@ impl Url {
self.mutate(|parser| parser.parse_fragment(parser::Input::new_no_trim(input)))
} else {
self.fragment_start = None;
self.strip_trailing_spaces_from_opaque_path();
}
}
@@ -1658,9 +1631,6 @@ impl Url {
});
} else {
self.query_start = None;
if fragment.is_none() {
self.strip_trailing_spaces_from_opaque_path();
}
}
self.restore_already_parsed_fragment(fragment);
+12 -4
View File
@@ -20,7 +20,7 @@ use percent_encoding::{percent_encode, utf8_percent_encode, AsciiSet, CONTROLS};
const FRAGMENT: &AsciiSet = &CONTROLS.add(b' ').add(b'"').add(b'<').add(b'>').add(b'`');
/// https://url.spec.whatwg.org/#path-percent-encode-set
const PATH: &AsciiSet = &FRAGMENT.add(b'#').add(b'?').add(b'{').add(b'}');
const PATH: &AsciiSet = &FRAGMENT.add(b'#').add(b'?').add(b'^').add(b'{').add(b'}');
/// https://url.spec.whatwg.org/#userinfo-percent-encode-set
pub(crate) const USERINFO: &AsciiSet = &PATH
@@ -32,7 +32,6 @@ pub(crate) const USERINFO: &AsciiSet = &PATH
.add(b'[')
.add(b'\\')
.add(b']')
.add(b'^')
.add(b'|');
pub(crate) const PATH_SEGMENT: &AsciiSet = &PATH.add(b'/').add(b'%');
@@ -1341,8 +1340,17 @@ impl Parser<'_> {
}
Some((c, utf8_c)) => {
self.check_url_code_point(c, &input);
self.serialization
.extend(utf8_percent_encode(utf8_c, CONTROLS));
// Encode only the last space before a suffix delimiter.
// Input lookahead, like iteration, skips ASCII tabs/newlines.
if c == ' '
&& self.context == Context::UrlParser
&& (input.starts_with('?') || input.starts_with('#'))
{
self.serialization.push_str("%20");
} else {
self.serialization
.extend(utf8_percent_encode(utf8_c, CONTROLS));
}
}
None => return input,
}
+18 -18
View File
@@ -1878,8 +1878,8 @@
"href": "a:/",
"new_value": "\u0000\u0001\t\n\r\u001f !\"#$%&'()*+,-./09:;<=>?@AZ[\\]^_`az{|}~\u007f\u0080\u0081Éé",
"expected": {
"href": "a:/%00%01%1F%20!%22%23$%&'()*+,-./09:;%3C=%3E%3F@AZ[\\]^_%60az%7B|%7D~%7F%C2%80%C2%81%C3%89%C3%A9",
"pathname": "/%00%01%1F%20!%22%23$%&'()*+,-./09:;%3C=%3E%3F@AZ[\\]^_%60az%7B|%7D~%7F%C2%80%C2%81%C3%89%C3%A9"
"href": "a:/%00%01%1F%20!%22%23$%&'()*+,-./09:;%3C=%3E%3F@AZ[\\]%5E_%60az%7B|%7D~%7F%C2%80%C2%81%C3%89%C3%A9",
"pathname": "/%00%01%1F%20!%22%23$%&'()*+,-./09:;%3C=%3E%3F@AZ[\\]%5E_%60az%7B|%7D~%7F%C2%80%C2%81%C3%89%C3%A9"
}
},
{
@@ -2126,12 +2126,12 @@
}
},
{
"comment": "Drop trailing spaces from trailing opaque paths",
"comment": "Trailing spaces in opaque paths are encoded during parsing",
"href": "data:space ?query",
"new_value": "",
"expected": {
"href": "data:space",
"pathname": "space",
"href": "data:space%20",
"pathname": "space%20",
"search": ""
}
},
@@ -2139,17 +2139,17 @@
"href": "sc:space ?query",
"new_value": "",
"expected": {
"href": "sc:space",
"pathname": "space",
"href": "sc:space%20",
"pathname": "space%20",
"search": ""
}
},
{
"comment": "Do not drop trailing spaces from non-trailing opaque paths",
"comment": "Trailing spaces in opaque paths are encoded during parsing",
"href": "data:space ?query#fragment",
"new_value": "",
"expected": {
"href": "data:space #fragment",
"href": "data:space %20#fragment",
"search": ""
}
},
@@ -2157,7 +2157,7 @@
"href": "sc:space ?query#fragment",
"new_value": "",
"expected": {
"href": "sc:space #fragment",
"href": "sc:space %20#fragment",
"search": ""
}
},
@@ -2314,12 +2314,12 @@
}
},
{
"comment": "Drop trailing spaces from trailing opaque paths",
"comment": "Trailing spaces in opaque paths are encoded during parsing",
"href": "data:space #fragment",
"new_value": "",
"expected": {
"href": "data:space",
"pathname": "space",
"href": "data:space %20",
"pathname": "space %20",
"hash": ""
}
},
@@ -2327,17 +2327,17 @@
"href": "sc:space #fragment",
"new_value": "",
"expected": {
"href": "sc:space",
"pathname": "space",
"href": "sc:space %20",
"pathname": "space %20",
"hash": ""
}
},
{
"comment": "Do not drop trailing spaces from non-trailing opaque paths",
"comment": "Trailing spaces in opaque paths are encoded during parsing",
"href": "data:space ?query#fragment",
"new_value": "",
"expected": {
"href": "data:space ?query",
"href": "data:space %20?query",
"hash": ""
}
},
@@ -2345,7 +2345,7 @@
"href": "sc:space ?query#fragment",
"new_value": "",
"expected": {
"href": "sc:space ?query",
"href": "sc:space %20?query",
"hash": ""
}
},
+5 -3
View File
@@ -68,14 +68,16 @@ fn test_relative_empty() {
}
#[test]
fn test_strip_trailing_spaces_from_opaque_path() {
fn test_preserve_encoded_trailing_space_in_opaque_path() {
let mut url: Url = "data:space ?query".parse().unwrap();
assert_eq!(url.as_str(), "data:space %20?query");
url.set_query(None);
assert_eq!(url.as_str(), "data:space");
assert_eq!(url.as_str(), "data:space %20");
let mut url: Url = "data:space #hash".parse().unwrap();
assert_eq!(url.as_str(), "data:space %20#hash");
url.set_fragment(None);
assert_eq!(url.as_str(), "data:space");
assert_eq!(url.as_str(), "data:space %20");
}
#[test]
+4 -4
View File
@@ -8492,10 +8492,10 @@
"hash": "",
"host": "host",
"hostname": "host",
"href": "foo://host/%20!%22$%&'()*+,-./:;%3C=%3E@[\\]^_%60%7B|%7D~",
"href": "foo://host/%20!%22$%&'()*+,-./:;%3C=%3E@[\\]%5E_%60%7B|%7D~",
"origin": "null",
"password": "",
"pathname": "/%20!%22$%&'()*+,-./:;%3C=%3E@[\\]^_%60%7B|%7D~",
"pathname": "/%20!%22$%&'()*+,-./:;%3C=%3E@[\\]%5E_%60%7B|%7D~",
"port":"",
"protocol": "foo:",
"search": "",
@@ -8507,10 +8507,10 @@
"hash": "",
"host": "host",
"hostname": "host",
"href": "wss://host/%20!%22$%&'()*+,-./:;%3C=%3E@[/]^_%60%7B|%7D~",
"href": "wss://host/%20!%22$%&'()*+,-./:;%3C=%3E@[/]%5E_%60%7B|%7D~",
"origin": "wss://host",
"password": "",
"pathname": "/%20!%22$%&'()*+,-./:;%3C=%3E@[/]^_%60%7B|%7D~",
"pathname": "/%20!%22$%&'()*+,-./:;%3C=%3E@[/]%5E_%60%7B|%7D~",
"port":"",
"protocol": "wss:",
"search": "",