fix(url): honor base URLs in parsing APIs

Validate supplied base URLs before parsing the input, including absolute inputs. Use the base for same-scheme relative parsing and share the result with URL.canParse so all three parsing APIs agree.

Add regression coverage for invalid bases, same-scheme HTTP/FTP/file inputs, opaque and absent bases, and WebIDL conversion order. Mark url/failure.html passing after CLI and CDP confirmation.

Validation: cargo fmt --all; full workspace clippy with warnings denied; nextest 17593 passed, 13 skipped. URL WPT slice improves from 12/22 to 15/22 passing in both CLI and CDP, with 237 additional passing subtests and no newly failing cases. CDP smoke remains 42/43, with Puppeteer unavailable because node is missing.
This commit is contained in:
ldm0
2026-09-09 06:59:50 +08:00
parent 432796c42c
commit 0b024cbe7d
4 changed files with 119 additions and 12 deletions
@@ -3302,7 +3302,6 @@ trusted-types/trusted-types-secondary-document.html
uievents/order-of-events/focus-events/focus-automated-blink-webkit.html
url/a-element.html?exclude=(file|javascript|mailto)
url/a-element.html?include=file
url/failure.html
viewport/viewport-segments.html
wasm/core/memory64/table64.wast.js.html
wasm/jsapi/proto-from-ctor-realm.html
@@ -8398,6 +8398,7 @@ url/a-element-origin.html
url/a-element.html?include=javascript
url/a-element.html?include=mailto
url/data-uri-fragment.html
url/failure.html
user-timing/clearMarks.html
user-timing/clearMeasures.html
user-timing/invoke_with_timing_attributes.html
@@ -41,21 +41,16 @@ pub(super) fn resolve_url_constructor_input(
input: &str,
base: Option<&str>,
) -> std::result::Result<url::Url, url::ParseError> {
if let Ok(url) = url::Url::parse(input) {
return Ok(url);
// A supplied base must be validated even for an absolute input. It also
// participates in parsing same-scheme inputs such as `https:child`.
match base {
Some(base) => url::Url::parse(base)?.join(input),
None => url::Url::parse(input),
}
let Some(base) = base else {
return Err(url::ParseError::RelativeUrlWithoutBase);
};
let base = url::Url::parse(base)?;
base.join(input)
}
pub(super) fn can_parse_url_input(input: &str, base: Option<&str>) -> bool {
match base {
Some(base) => url::Url::parse(base).is_ok_and(|base| base.join(input).is_ok()),
None => url::Url::parse(input).is_ok(),
}
resolve_url_constructor_input(input, base).is_ok()
}
pub(in crate::context_bootstrap) fn url_href_slot<'s>(
@@ -8530,6 +8530,118 @@ fn url_and_search_params_declared_slots_ignore_prototype_spoofing() {
);
}
#[test]
fn url_parsing_apis_reject_invalid_bases_for_absolute_inputs() {
let mut vm = new_storage_test_vm("https://url-invalid-base.test/");
let result = vm
.eval(
r#"
(() => {
const assert = (condition, message) => { if (!condition) throw new Error(message); };
const invalidBases = [
'', '/relative', 'relative', 'http://', 'https://example.test:bogus/',
'http://[::1', 'file://example:1/', null, false,
];
for (const input of ['about:blank', 'https://example.test/', 'data:text/plain,x', 'child']) {
for (const base of invalidBases) {
let error;
try { new URL(input, base); } catch (caught) { error = caught; }
assert(error instanceof TypeError, `constructor rejects invalid base ${base} for ${input}`);
assert(URL.parse(input, base) === null, `parse rejects invalid base ${base} for ${input}`);
assert(URL.canParse(input, base) === false, `canParse rejects invalid base ${base} for ${input}`);
}
}
return 'ok';
})()
"#,
)
.expect("all URL parsing APIs must validate a supplied base");
assert_eq!(result, "ok");
}
#[test]
fn url_parsing_apis_resolve_same_scheme_inputs_against_the_base() {
let mut vm = new_storage_test_vm("https://url-same-scheme-base.test/");
let result = vm
.eval(
r#"
(() => {
const assert = (condition, message) => { if (!condition) throw new Error(message); };
const cases = [
['https:child', 'https://u:p@example.test:8443/dir/base?old#old', 'https://u:p@example.test:8443/dir/child'],
['HTTPS:../next', 'https://example.test/dir/base', 'https://example.test/next'],
['https:?next', 'https://example.test/dir/base?old#old', 'https://example.test/dir/base?next'],
['https:#next', 'https://example.test/dir/base?old#old', 'https://example.test/dir/base?old#next'],
['ftp:child', 'ftp://example.test/dir/base', 'ftp://example.test/dir/child'],
['file:child', 'file:///dir/base', 'file:///dir/child'],
['https:child', 'http://example.test/dir/base', 'https://child/'],
['https:child', undefined, 'https://child/'],
['about:blank', 'https://example.test/', 'about:blank'],
['data:text/plain,x', 'about:blank', 'data:text/plain,x'],
['#new', 'about:blank?old#old', 'about:blank?old#new'],
['https://other.test/x', 'https://example.test/dir/base', 'https://other.test/x'],
];
for (const [input, base, expected] of cases) {
assert(new URL(input, base).href === expected, `constructor resolves ${input} against ${base}`);
const parsed = URL.parse(input, base);
assert(parsed instanceof URL && parsed.href === expected, `parse resolves ${input} against ${base}`);
assert(URL.canParse(input, base), `canParse accepts ${input} against ${base}`);
assert(parsed.searchParams.toString() === new URL(expected).searchParams.toString(), 'resolved query initializes URLSearchParams');
}
return 'ok';
})()
"#,
)
.expect("URL parsing must use the base even when the input includes a scheme");
assert_eq!(result, "ok");
}
#[test]
fn url_parsing_apis_convert_arguments_before_parsing() {
let mut vm = new_storage_test_vm("https://url-base-conversion-order.test/");
let result = vm
.eval(
r#"
(() => {
const assert = (condition, message) => { if (!condition) throw new Error(message); };
for (const parse of [(...args) => new URL(...args), URL.parse, URL.canParse]) {
const order = [];
const value = (name, text) => ({
[Symbol.toPrimitive](hint) {
assert(hint === 'string', 'URL arguments use string conversion');
order.push(name);
return text;
},
});
parse(value('input', 'https://example.test/'), value('base', 'https://base.test/'));
assert(order.join(',') === 'input,base', 'input then base are each converted once');
order.length = 0;
const marker = new RangeError('conversion marker');
let error;
try {
parse(value('input', 'http://['), {
toString() { order.push('base'); throw marker; },
});
} catch (caught) { error = caught; }
assert(error === marker && order.join(',') === 'input,base', 'base conversion errors precede URL parsing');
order.length = 0;
error = undefined;
try {
parse({ toString() { throw marker; } }, value('base', 'https://base.test/'));
} catch (caught) { error = caught; }
assert(error === marker && order.length === 0, 'failed input conversion does not touch the base');
const withoutBase = parse('https://example.test/');
const undefinedBase = parse('https://example.test/', undefined);
assert(String(withoutBase) === String(undefinedBase), 'undefined base is equivalent to omission');
}
return 'ok';
})()
"#,
)
.expect("URL argument conversion order must be preserved");
assert_eq!(result, "ok");
}
#[test]
fn url_static_parse_and_can_parse_stringify_undefined_input() {
let mut vm = new_storage_test_vm("https://url-static-stringification.test/");