From cf5e6764f9d9730a400e2a9358bb60d2a4628092 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Mon, 6 Jul 2026 11:00:48 +0000 Subject: [PATCH 1/6] url: stop running url.parse() hostnames through the WHATWG host parser Legacy url.parse() mapped every hostname through `new URL("http://" + hostname)` for IDNA support. That ran the WHATWG host parser, which canonicalizes IPv4 and IPv6 hosts and rejects hosts the legacy grammar accepts. Node applies a pure IDNA ToASCII to the hostname instead, then guards against the mapping introducing a character that changes how the host is read. Add a `toASCII` binding that does the UTS #46 mapping without host parsing, and port node's forbidden-character guards. Also throw on an invalid port, which node does as of v23 in place of the DEP0170 warning. --- src/js/node/url.ts | 56 +++++++++------ src/jsc/bindings/NodeURL.cpp | 71 ++++++++++++++----- .../test/parallel/test-url-parse-format.js | 16 ----- .../parallel/test-url-parse-invalid-input.js | 20 ++---- test/js/node/url/url-parse-format.test.js | 15 ---- .../node/url/url-parse-invalid-input.test.js | 18 ++--- test/js/node/url/url-parse-ipv6.test.ts | 39 +++++----- test/js/node/url/url.test.ts | 67 +++++++++++++++++ 8 files changed, 184 insertions(+), 118 deletions(-) diff --git a/src/js/node/url.ts b/src/js/node/url.ts index f8809450309e..d852c2830a35 100644 --- a/src/js/node/url.ts +++ b/src/js/node/url.ts @@ -26,7 +26,7 @@ "use strict"; const { URL, URLSearchParams } = globalThis; -const [domainToASCII, domainToUnicode] = $cpp("NodeURL.cpp", "Bun::createNodeURLBinding"); +const [domainToASCII, domainToUnicode, toASCII] = $cpp("NodeURL.cpp", "Bun::createNodeURLBinding"); const { urlToHttpOptions } = require("internal/url"); const { validateString } = require("internal/validators"); const ObjectSetPrototypeOf = Object.setPrototypeOf; @@ -75,6 +75,14 @@ var protocolPattern = /^([a-z0-9.+-]+:)/i, nonHostChars = ["%", "/", "?", ";", "#"].concat(autoEscape), hostEndingChars = ["/", "?", "#"], hostnameMaxLen = 255, + /* + * Prevents spoofing bugs caused by IDNA toASCII mapping a character into one + * that changes how the host is interpreted. ':' spoofs the protocol, '@' the + * auth, and '[' / ']' make a non-IPv6 host look like IPv6. + */ + forbiddenHostChars = /[\0\t\n\r #%/:<>?@[\\\]^|]/, + // For IPv6, permit '[', ']', and ':'. + forbiddenHostCharsIpv6 = /[\0\t\n\r #%/<>?@\\^|]/, // protocols that can allow "unsafe" and "unwise" chars. unsafeProtocol = { __proto__: null, @@ -110,12 +118,7 @@ function urlParse( if ($isObject(url) && url instanceof Url) return url; var u = new Url(); - try { - u.parse(url, parseQueryString, slashesDenoteHost); - } catch (e) { - $putByIdDirect(e, "input", url); - throw e; - } + u.parse(url, parseQueryString, slashesDenoteHost); return u; } @@ -340,14 +343,28 @@ Url.prototype.parse = function parse(url: string, parseQueryString?: boolean, sl this.hostname = this.hostname.toLowerCase(); } - /* - * IDNA Support: Returns a punycoded representation of "domain". - * It only converts parts of the domain name that - * have non-ASCII characters, i.e. it doesn't matter if - * you call it with a domain that already is ASCII-only. - */ - if (this.hostname) { - this.hostname = new URL("http://" + this.hostname).hostname; + if (this.hostname !== "") { + if (ipv6Hostname) { + if (forbiddenHostCharsIpv6.test(this.hostname)) { + throw $ERR_INVALID_URL(url); + } + } else { + /* + * IDNA Support: Returns a punycoded representation of "domain". + * It only converts parts of the domain name that + * have non-ASCII characters, i.e. it doesn't matter if + * you call it with a domain that already is ASCII-only. + */ + this.hostname = toASCII(this.hostname); + + /* + * An empty hostname or a forbidden character can only have been + * introduced by toASCII, since getHostname filters them out otherwise. + */ + if (this.hostname === "" || forbiddenHostChars.test(this.hostname)) { + throw $ERR_INVALID_URL(url); + } + } } var p = this.port ? ":" + this.port : ""; @@ -437,7 +454,6 @@ function isIpv6Hostname(hostname: string) { ); } -let warnInvalidPort = true; function getHostname(self, rest, hostname: string, url) { for (let i = 0; i < hostname.length; ++i) { const code = hostname.$charCodeAt(i); @@ -450,12 +466,8 @@ function getHostname(self, rest, hostname: string, url) { if (!isValid) { // If leftover starts with :, then it represents an invalid port. - // But url.parse() is lenient about it for now. - // Issue a warning and continue. - if (warnInvalidPort && code === Char.COLON) { - const detail = `The URL ${url} is invalid. Future versions of Node.js will throw an error.`; - process.emitWarning(detail, "DeprecationWarning", "DEP0170"); - warnInvalidPort = false; + if (code === Char.COLON) { + throw $ERR_INVALID_ARG_VALUE("url", "Invalid port in url", url); } self.hostname = hostname.slice(0, i); return `/${hostname.slice(i)}${rest}`; diff --git a/src/jsc/bindings/NodeURL.cpp b/src/jsc/bindings/NodeURL.cpp index 09af0674a4e9..3000ae502e50 100644 --- a/src/jsc/bindings/NodeURL.cpp +++ b/src/jsc/bindings/NodeURL.cpp @@ -4,6 +4,47 @@ namespace Bun { +static constexpr int allowedNameToASCIIErrors = UIDNA_ERROR_EMPTY_LABEL | UIDNA_ERROR_LABEL_TOO_LONG | UIDNA_ERROR_DOMAIN_NAME_TOO_LONG | UIDNA_ERROR_LEADING_HYPHEN | UIDNA_ERROR_TRAILING_HYPHEN | UIDNA_ERROR_HYPHEN_3_4; +static constexpr size_t hostnameBufferLength = 2048; + +// UTS #46 ToASCII, with the relaxations the WHATWG URL Standard applies +// (CheckHyphens and VerifyDnsLength disabled). Returns a null string on failure. +static WTF::String nameToASCII(const WTF::String& input) +{ + if (input.isEmpty()) + return {}; + + WTF::String domain = input; + domain.convertTo16Bit(); + + auto* encoder = &WTF::URLParser::internationalDomainNameTranscoder(); + char16_t hostnameBuffer[hostnameBufferLength]; + UErrorCode error = U_ZERO_ERROR; + UIDNAInfo processingDetails = UIDNA_INFO_INITIALIZER; + const auto span = domain.span16(); + int32_t numCharactersConverted = uidna_nameToASCII(encoder, span.data(), span.size(), hostnameBuffer, hostnameBufferLength, &processingDetails, &error); + + if (U_SUCCESS(error) && !(processingDetails.errors & ~allowedNameToASCIIErrors) && numCharactersConverted) + return WTF::String(std::span { hostnameBuffer, static_cast(numCharactersConverted) }); + return {}; +} + +// The IDNA mapping `url.parse()` applies to a hostname. Unlike `domainToASCII` +// this performs no host parsing, so IPv4 and IPv6 hosts are left untouched. +JSC_DEFINE_HOST_FUNCTION(jsToASCII, (JSC::JSGlobalObject * globalObject, JSC::CallFrame* callFrame)) +{ + auto& vm = JSC::getVM(globalObject); + auto scope = DECLARE_THROW_SCOPE(vm); + + auto domain = callFrame->argument(0).toWTFString(globalObject); + RETURN_IF_EXCEPTION(scope, {}); + + auto ascii = nameToASCII(domain); + if (ascii.isNull()) + return JSC::JSValue::encode(jsEmptyString(vm)); + return JSC::JSValue::encode(JSC::jsString(vm, WTF::move(ascii))); +} + JSC_DEFINE_HOST_FUNCTION(jsDomainToASCII, (JSC::JSGlobalObject * globalObject, JSC::CallFrame* callFrame)) { auto& vm = JSC::getVM(globalObject); @@ -52,23 +93,11 @@ JSC_DEFINE_HOST_FUNCTION(jsDomainToASCII, (JSC::JSGlobalObject * globalObject, J if (domain.containsOnlyASCII()) return JSC::JSValue::encode(arg0); - if (domain.is8Bit()) - domain.convertTo16Bit(); - - constexpr static int allowedNameToASCIIErrors = UIDNA_ERROR_EMPTY_LABEL | UIDNA_ERROR_LABEL_TOO_LONG | UIDNA_ERROR_DOMAIN_NAME_TOO_LONG | UIDNA_ERROR_LEADING_HYPHEN | UIDNA_ERROR_TRAILING_HYPHEN | UIDNA_ERROR_HYPHEN_3_4; - constexpr static size_t hostnameBufferLength = 2048; - auto encoder = &WTF::URLParser::internationalDomainNameTranscoder(); - char16_t hostnameBuffer[hostnameBufferLength]; - UErrorCode error = U_ZERO_ERROR; - UIDNAInfo processingDetails = UIDNA_INFO_INITIALIZER; - const auto span = domain.span16(); - int32_t numCharactersConverted = uidna_nameToASCII(encoder, span.data(), span.size(), hostnameBuffer, hostnameBufferLength, &processingDetails, &error); - - if (U_SUCCESS(error) && !(processingDetails.errors & ~allowedNameToASCIIErrors) && numCharactersConverted) { - return JSC::JSValue::encode(JSC::jsString(vm, WTF::String(std::span { hostnameBuffer, static_cast(numCharactersConverted) }))); - } - return JSC::JSValue::encode(jsEmptyString(vm)); + auto ascii = nameToASCII(domain); + if (ascii.isNull()) + return JSC::JSValue::encode(jsEmptyString(vm)); + return JSC::JSValue::encode(JSC::jsString(vm, WTF::move(ascii))); } JSC_DEFINE_HOST_FUNCTION(jsDomainToUnicode, (JSC::JSGlobalObject * globalObject, JSC::CallFrame* callFrame)) @@ -124,7 +153,6 @@ JSC_DEFINE_HOST_FUNCTION(jsDomainToUnicode, (JSC::JSGlobalObject * globalObject, domain.convertTo16Bit(); constexpr static int allowedNameToUnicodeErrors = UIDNA_ERROR_EMPTY_LABEL | UIDNA_ERROR_LABEL_TOO_LONG | UIDNA_ERROR_DOMAIN_NAME_TOO_LONG | UIDNA_ERROR_LEADING_HYPHEN | UIDNA_ERROR_TRAILING_HYPHEN | UIDNA_ERROR_HYPHEN_3_4; - constexpr static int hostnameBufferLength = 2048; auto encoder = &WTF::URLParser::internationalDomainNameTranscoder(); char16_t hostnameBuffer[hostnameBufferLength]; @@ -145,13 +173,15 @@ JSC::JSValue createNodeURLBinding(Zig::GlobalObject* globalObject) { VM& vm = globalObject->vm(); auto scope = DECLARE_THROW_SCOPE(vm); - auto binding = constructEmptyArray(globalObject, nullptr, 2); + auto binding = constructEmptyArray(globalObject, nullptr, 3); RETURN_IF_EXCEPTION(scope, {}); ASSERT(binding); auto domainToAsciiFunction = JSC::JSFunction::create(vm, globalObject, 1, "domainToAscii"_s, jsDomainToASCII, ImplementationVisibility::Public); ASSERT(domainToAsciiFunction); auto domainToUnicodeFunction = JSC::JSFunction::create(vm, globalObject, 1, "domainToUnicode"_s, jsDomainToUnicode, ImplementationVisibility::Public); ASSERT(domainToUnicodeFunction); + auto toAsciiFunction = JSC::JSFunction::create(vm, globalObject, 1, "toASCII"_s, jsToASCII, ImplementationVisibility::Public); + ASSERT(toAsciiFunction); binding->putByIndexInline( globalObject, (unsigned)0, @@ -162,6 +192,11 @@ JSC::JSValue createNodeURLBinding(Zig::GlobalObject* globalObject) (unsigned)1, domainToUnicodeFunction, false); + binding->putByIndexInline( + globalObject, + (unsigned)2, + toAsciiFunction, + false); return binding; } diff --git a/test/js/node/test/parallel/test-url-parse-format.js b/test/js/node/test/parallel/test-url-parse-format.js index f8761514a30b..c22d409d0f41 100644 --- a/test/js/node/test/parallel/test-url-parse-format.js +++ b/test/js/node/test/parallel/test-url-parse-format.js @@ -865,22 +865,6 @@ const parseTests = { href: 'http://a%22%20%3C\'b:b@cd/e?f' }, - // Git urls used by npm - 'git+ssh://git@github.com:npm/npm': { - protocol: 'git+ssh:', - slashes: true, - auth: 'git', - host: 'github.com', - port: null, - hostname: 'github.com', - hash: null, - search: null, - query: null, - pathname: '/:npm/npm', - path: '/:npm/npm', - href: 'git+ssh://git@github.com/:npm/npm' - }, - 'https://*': { protocol: 'https:', slashes: true, diff --git a/test/js/node/test/parallel/test-url-parse-invalid-input.js b/test/js/node/test/parallel/test-url-parse-invalid-input.js index 8f8a4d40f8c0..368ebff451c4 100644 --- a/test/js/node/test/parallel/test-url-parse-invalid-input.js +++ b/test/js/node/test/parallel/test-url-parse-invalid-input.js @@ -83,25 +83,13 @@ if (common.hasIntl) { badURLs.forEach((badURL) => { common.spawnPromisified(process.execPath, ['-e', `url.parse(${JSON.stringify(badURL)})`]) .then(common.mustCall(({ code, stdout, stderr }) => { - assert.strictEqual(code, 0); - assert.strictEqual(stdout, ''); - // NOTE: bun formats errors slightly differently from node, but we're - // printing the same deprecation message. - // assert.match(stderr, /\[DEP0170\] DeprecationWarning:/); - assert.match(stderr, /\DEP0170/); - assert.match(stderr, /\DeprecationWarning/); + assert.strictEqual(code, 1); })); }); - // Warning should only happen once per process. - common.expectWarning({ - DeprecationWarning: { - // NOTE: this warning is noisy and annoying. We've disabled it intentionally. - // DEP0169: '`url.parse()` behavior is not standardized and prone to errors that have security implications. Use the WHATWG URL API instead. CVEs are not issued for `url.parse()` vulnerabilities.', - DEP0170: `The URL ${badURLs[0]} is invalid. Future versions of Node.js will throw an error.`, - }, - }); badURLs.forEach((badURL) => { - url.parse(badURL); + assert.throws(() => url.parse(badURL), { + code: 'ERR_INVALID_ARG_VALUE', + }); }); } diff --git a/test/js/node/url/url-parse-format.test.js b/test/js/node/url/url-parse-format.test.js index bec764981954..2c2d7178d87e 100644 --- a/test/js/node/url/url-parse-format.test.js +++ b/test/js/node/url/url-parse-format.test.js @@ -861,21 +861,6 @@ describe("url.parse then url.format", () => { // href: "http://a%22%20%3C'b:b@cd/e?f", // }, - // Git urls used by npm - "git+ssh://git@github.com:npm/npm": { - protocol: "git+ssh:", - slashes: true, - auth: "git", - host: "github.com", - port: null, - hostname: "github.com", - hash: null, - search: null, - query: null, - pathname: "/:npm/npm", - path: "/:npm/npm", - href: "git+ssh://git@github.com/:npm/npm", - }, // TODO: Support parsing these. // // "https://*": { diff --git a/test/js/node/url/url-parse-invalid-input.test.js b/test/js/node/url/url-parse-invalid-input.test.js index e8b9b1831c1c..2539c5554369 100644 --- a/test/js/node/url/url-parse-invalid-input.test.js +++ b/test/js/node/url/url-parse-invalid-input.test.js @@ -97,24 +97,16 @@ describe("url.parse", () => { const badURLs = ["https://evil.com:.example.com", "git+ssh://git@github.com:npm/npm"]; badURLs.forEach(badURL => { common.spawnPromisified(process.execPath, ["-e", `url.parse(${JSON.stringify(badURL)})`]).then( - common.mustCall(({ code, stdout, stderr }) => { - assert.strictEqual(code, 0); - assert.strictEqual(stdout, ""); - assert.match(stderr, /\[DEP0170\] DeprecationWarning:/); + common.mustCall(({ code }) => { + assert.strictEqual(code, 1); }), ); }); - // Warning should only happen once per process. - const expectedWarning = [ - `The URL ${badURLs[0]} is invalid. Future versions of Node.js will throw an error.`, - "DEP0170", - ]; - common.expectWarning({ - DeprecationWarning: expectedWarning, - }); badURLs.forEach(badURL => { - url.parse(badURL); + assert.throws(() => url.parse(badURL), { + code: "ERR_INVALID_ARG_VALUE", + }); }); } }); diff --git a/test/js/node/url/url-parse-ipv6.test.ts b/test/js/node/url/url-parse-ipv6.test.ts index 2a79c83c931c..6d83378f61e2 100644 --- a/test/js/node/url/url-parse-ipv6.test.ts +++ b/test/js/node/url/url-parse-ipv6.test.ts @@ -6,12 +6,9 @@ import url from "node:url"; process.emitWarning = () => {}; describe("Invalid IPv6 addresses", () => { - it.each(["https://[::1", "https://[:::1]", "https://[\n::1]", "http://[::banana]"])( - "Invalid hostnames - parsing '%s' fails", - input => { - expect(() => url.parse(input)).toThrowError(TypeError); - }, - ); + it.each(["https://[::1", "https://[\n::1]"])("Invalid hostnames - parsing '%s' fails", input => { + expect(() => url.parse(input)).toThrowError(TypeError); + }); it.each(["https://[::1]::", "https://[::1]:foo"])("Invalid ports - parsing '%s' fails", input => { expect(() => url.parse(input)).toThrowError(TypeError); @@ -22,22 +19,28 @@ describe("Valid spot checks", () => { it.each([ // ports ["http://[::1]:", { host: "[::1]", hostname: "::1", port: null, path: "/", href: "http://[::1]/" }], // trailing colons are ignored - ["http://[::1]:1", { host: "[::1]", hostname: "::1", port: "1", path: "/", href: "http://[::1]/" }], + ["http://[::1]:1", { host: "[::1]:1", hostname: "::1", port: "1", path: "/", href: "http://[::1]:1/" }], // unicast - ["http://[::0]", { host: "[::0]", path: "/" }], - ["http://[::f]", { host: "[::f]", path: "/" }], - ["http://[::F]", { host: "[::F]", path: "/" }], - // these are technically invalid unicast addresses but url.parse allows them - ["http://[::7]", { host: "[::7]", path: "/" }], - // ["http://[::z]", { host: "[::7]", path: "/" }], - // ["http://[::😩]", { host: "[::😩]", path: "/" }], + ["http://[::0]", { host: "[::0]", hostname: "::0", path: "/", href: "http://[::0]/" }], + ["http://[::f]", { host: "[::f]", hostname: "::f", path: "/", href: "http://[::f]/" }], + ["http://[::F]", { host: "[::f]", hostname: "::f", path: "/", href: "http://[::f]/" }], // hostnames are lower cased + + // url.parse never validates the contents of an IPv6 host, so these parse + // even though they are not addresses. + ["http://[::7]", { host: "[::7]", hostname: "::7", path: "/", href: "http://[::7]/" }], + ["https://[:::1]", { host: "[:::1]", hostname: ":::1", path: "/", href: "https://[:::1]/" }], + ["http://[::banana]", { host: "[::banana]", hostname: "::banana", path: "/", href: "http://[::banana]/" }], // full form-ish - ["https://[::1:2:3:4:5]", { host: "[::1:2:3:4:5]", path: "/" }], - ["[0:0:0:1:2:3:4:5]", { host: "[0:0:0:1:2:3:4:5]", path: "/" }], + [ + "https://[::1:2:3:4:5]", + { host: "[::1:2:3:4:5]", hostname: "::1:2:3:4:5", path: "/", href: "https://[::1:2:3:4:5]/" }, + ], + // w/o a protocol, it's treated as a path + ["[0:0:0:1:2:3:4:5]", { host: null, hostname: null, path: "[0:0:0:1:2:3:4:5]", href: "[0:0:0:1:2:3:4:5]" }], ])("Parsing '%s' succeeds", (input, expected) => { - expect(url.parse(input)).toMatchObject(expect.objectContaining(expected)); + expect(url.parse(input)).toMatchObject(expected); }); }); // @@ -175,7 +178,7 @@ describe.each([ it("parses to the expected object", () => { const { query, ...rest } = expected; - expect(parsed).toMatchObject(expect.objectContaining(rest)); + expect(parsed).toMatchObject(rest); }); it("parses the query", () => { diff --git a/test/js/node/url/url.test.ts b/test/js/node/url/url.test.ts index 2fd8fa6277a0..2e6fc7887ce2 100644 --- a/test/js/node/url/url.test.ts +++ b/test/js/node/url/url.test.ts @@ -67,6 +67,73 @@ describe("Url.prototype.parse", () => { }); }); +/* + * url.parse() uses node's legacy host grammar, not the WHATWG host parser: the + * hostname is IDNA mapped, but never canonicalized or validated as an IP. + */ +describe("Url.prototype.parse host handling", () => { + function parseError(input: string): Error & { code?: string; input?: string } { + try { + parse(input); + } catch (e) { + return e as Error & { code?: string }; + } + throw new Error(`expected url.parse(${JSON.stringify(input)}) to throw`); + } + + it("does not canonicalize IPv4 or IPv6 hosts", () => { + const hostnames = [ + "http://0x7f.1/", + "http://0300.0250.0.01/", + "http://2130706433/", + "http://127.1/", + "http://[::ffff:1.2.3.4]/", + "http://[0:0:0:0:0:0:0:1]/", + ].map(input => parse(input).hostname); + + expect(hostnames).toEqual(["0x7f.1", "0300.0250.0.01", "2130706433", "127.1", "::ffff:1.2.3.4", "0:0:0:0:0:0:0:1"]); + }); + + it("accepts hosts the WHATWG host parser rejects", () => { + const hostnames = ["http://192.168.1.256/", "http://1.2.3.4.5/", "http://0x100000000/"].map( + input => parse(input).hostname, + ); + + expect(hostnames).toEqual(["192.168.1.256", "1.2.3.4.5", "0x100000000"]); + }); + + it("keeps the unparsed host in host and href", () => { + expect(parse("http://0x7f.1:8080/p")).toMatchObject({ + host: "0x7f.1:8080", + hostname: "0x7f.1", + port: "8080", + href: "http://0x7f.1:8080/p", + }); + }); + + it("still IDNA maps non-ASCII hosts", () => { + expect(parse("http://Ünicode.com/").hostname).toBe("xn--nicode-2ya.com"); + }); + + it.each([ + ["http://h:8a/x", "ERR_INVALID_ARG_VALUE"], + ["https://evil.com:.example.com", "ERR_INVALID_ARG_VALUE"], + ["git+ssh://git@github.com:npm/npm", "ERR_INVALID_ARG_VALUE"], + ["http://fail\uFF20fail.com/", "ERR_INVALID_URL"], + ["http://fail\u2100fail.com/", "ERR_INVALID_URL"], + ["http://\u00AD/bad.com/", "ERR_INVALID_URL"], + ["http://[127.0.0.1\0c8763]:8000/", "ERR_INVALID_URL"], + ])("parsing %j throws %s", (input, code) => { + const err = parseError(input); + expect(err).toBeInstanceOf(TypeError); + expect(err.code).toBe(code); + }); + + it("reports the whole url as the input of an ERR_INVALID_URL", () => { + expect(parseError("http://[127.0.0.1\0c8763]:8000/").input).toBe("http://[127.0.0.1\0c8763]:8000/"); + }); +}); + it("URL constructor throws ERR_MISSING_ARGS", () => { var err; try { From 0e0192020aae757aef9f6bc010580954fba8d459 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Mon, 6 Jul 2026 11:23:06 +0000 Subject: [PATCH 2/6] test: cover the multi-host connection string from #24812 --- test/js/node/url/url.test.ts | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/test/js/node/url/url.test.ts b/test/js/node/url/url.test.ts index 2e6fc7887ce2..4ad8b02904ad 100644 --- a/test/js/node/url/url.test.ts +++ b/test/js/node/url/url.test.ts @@ -115,6 +115,19 @@ describe("Url.prototype.parse host handling", () => { expect(parse("http://Ünicode.com/").hostname).toBe("xn--nicode-2ya.com"); }); + // https://github.com/oven-sh/bun/issues/24812 + it("parses a multi-host connection string", () => { + expect(parse("mongodb://user:password@[fd34:b871:e6a7::1],[fd34:b871:e6a7::2]:27017/db")).toMatchObject({ + protocol: "mongodb:", + auth: "user:password", + host: "[fd34:b871:e6a7::1],[fd34:b871:e6a7::2]:27017", + port: "27017", + hostname: "fd34:b871:e6a7::1],[fd34:b871:e6a7::2", + pathname: "/db", + href: "mongodb://user:password@[fd34:b871:e6a7::1],[fd34:b871:e6a7::2]:27017/db", + }); + }); + it.each([ ["http://h:8a/x", "ERR_INVALID_ARG_VALUE"], ["https://evil.com:.example.com", "ERR_INVALID_ARG_VALUE"], From 7912e8abefacc50c89225d8de7ecc4186f3b20be Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Mon, 6 Jul 2026 11:33:57 +0000 Subject: [PATCH 3/6] test: drop the now-dead process.emitWarning override --- test/js/node/url/url-parse-ipv6.test.ts | 3 --- 1 file changed, 3 deletions(-) diff --git a/test/js/node/url/url-parse-ipv6.test.ts b/test/js/node/url/url-parse-ipv6.test.ts index 6d83378f61e2..30007819a5a0 100644 --- a/test/js/node/url/url-parse-ipv6.test.ts +++ b/test/js/node/url/url-parse-ipv6.test.ts @@ -2,9 +2,6 @@ import { beforeAll,describe,expect,it } from "bun:test"; import url from "node:url"; -// url.parse is deprecated. -process.emitWarning = () => {}; - describe("Invalid IPv6 addresses", () => { it.each(["https://[::1", "https://[\n::1]"])("Invalid hostnames - parsing '%s' fails", input => { expect(() => url.parse(input)).toThrowError(TypeError); From f04b62624788b8abad2c5d903b69c6d27f3afde3 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Mon, 6 Jul 2026 12:07:54 +0000 Subject: [PATCH 4/6] url: consolidate the IDNA error mask, un-todo the invalid-input test Both ICU call sites ignore the same error set, for the same reason: the WHATWG URL Standard turns off CheckHyphens and VerifyDnsLength. url-parse-invalid-input.test.js was entirely test.todo and referenced an undefined `common`, so it never ran. Its "TODO: Support error code" is what this branch implements, so rewrite it to run. The full badIDNA sweep stays in the parallel/ copy: it walks the whole code point range and takes minutes under a debug build. --- src/jsc/bindings/NodeURL.cpp | 13 +- .../node/url/url-parse-invalid-input.test.js | 156 ++++++++---------- 2 files changed, 71 insertions(+), 98 deletions(-) diff --git a/src/jsc/bindings/NodeURL.cpp b/src/jsc/bindings/NodeURL.cpp index 3000ae502e50..a7b5cb3bcf6b 100644 --- a/src/jsc/bindings/NodeURL.cpp +++ b/src/jsc/bindings/NodeURL.cpp @@ -4,11 +4,12 @@ namespace Bun { -static constexpr int allowedNameToASCIIErrors = UIDNA_ERROR_EMPTY_LABEL | UIDNA_ERROR_LABEL_TOO_LONG | UIDNA_ERROR_DOMAIN_NAME_TOO_LONG | UIDNA_ERROR_LEADING_HYPHEN | UIDNA_ERROR_TRAILING_HYPHEN | UIDNA_ERROR_HYPHEN_3_4; +// The errors UTS #46 reports that the WHATWG URL Standard ignores, by turning +// off CheckHyphens and VerifyDnsLength. ICU has no option for either. +static constexpr int allowedIDNAErrors = UIDNA_ERROR_EMPTY_LABEL | UIDNA_ERROR_LABEL_TOO_LONG | UIDNA_ERROR_DOMAIN_NAME_TOO_LONG | UIDNA_ERROR_LEADING_HYPHEN | UIDNA_ERROR_TRAILING_HYPHEN | UIDNA_ERROR_HYPHEN_3_4; static constexpr size_t hostnameBufferLength = 2048; -// UTS #46 ToASCII, with the relaxations the WHATWG URL Standard applies -// (CheckHyphens and VerifyDnsLength disabled). Returns a null string on failure. +// UTS #46 ToASCII. Returns a null string when the domain is not valid. static WTF::String nameToASCII(const WTF::String& input) { if (input.isEmpty()) @@ -24,7 +25,7 @@ static WTF::String nameToASCII(const WTF::String& input) const auto span = domain.span16(); int32_t numCharactersConverted = uidna_nameToASCII(encoder, span.data(), span.size(), hostnameBuffer, hostnameBufferLength, &processingDetails, &error); - if (U_SUCCESS(error) && !(processingDetails.errors & ~allowedNameToASCIIErrors) && numCharactersConverted) + if (U_SUCCESS(error) && !(processingDetails.errors & ~allowedIDNAErrors) && numCharactersConverted) return WTF::String(std::span { hostnameBuffer, static_cast(numCharactersConverted) }); return {}; } @@ -152,8 +153,6 @@ JSC_DEFINE_HOST_FUNCTION(jsDomainToUnicode, (JSC::JSGlobalObject * globalObject, domain.convertTo16Bit(); - constexpr static int allowedNameToUnicodeErrors = UIDNA_ERROR_EMPTY_LABEL | UIDNA_ERROR_LABEL_TOO_LONG | UIDNA_ERROR_DOMAIN_NAME_TOO_LONG | UIDNA_ERROR_LEADING_HYPHEN | UIDNA_ERROR_TRAILING_HYPHEN | UIDNA_ERROR_HYPHEN_3_4; - auto encoder = &WTF::URLParser::internationalDomainNameTranscoder(); char16_t hostnameBuffer[hostnameBufferLength]; UErrorCode error = U_ZERO_ERROR; @@ -163,7 +162,7 @@ JSC_DEFINE_HOST_FUNCTION(jsDomainToUnicode, (JSC::JSGlobalObject * globalObject, int32_t numCharactersConverted = uidna_nameToUnicode(encoder, span.data(), span.size(), hostnameBuffer, hostnameBufferLength, &processingDetails, &error); - if (U_SUCCESS(error) && !(processingDetails.errors & ~allowedNameToUnicodeErrors) && numCharactersConverted) { + if (U_SUCCESS(error) && !(processingDetails.errors & ~allowedIDNAErrors) && numCharactersConverted) { return JSC::JSValue::encode(JSC::jsString(vm, WTF::String(std::span { hostnameBuffer, static_cast(numCharactersConverted) }))); } return JSC::JSValue::encode(jsEmptyString(vm)); diff --git a/test/js/node/url/url-parse-invalid-input.test.js b/test/js/node/url/url-parse-invalid-input.test.js index 2539c5554369..57619c89721a 100644 --- a/test/js/node/url/url-parse-invalid-input.test.js +++ b/test/js/node/url/url-parse-invalid-input.test.js @@ -1,113 +1,87 @@ -import { describe, test } from "bun:test"; +import { describe, expect, test } from "bun:test"; +import { bunEnv, bunExe } from "harness"; import assert from "node:assert"; import url from "node:url"; describe("url.parse", () => { - // TODO: Support error code. - test.todo("invalid input", () => { + test("rejects a non-string url", () => { // https://github.com/joyent/node/issues/568 [ [undefined, "undefined"], - [null, "object"], - [true, "boolean"], - [false, "boolean"], - [0.0, "number"], - [0, "number"], - [[], "object"], - [{}, "object"], - [() => {}, "function"], - [Symbol("foo"), "symbol"], - ].forEach(([val, type]) => { - assert.throws( - () => { - url.parse(val); - }, - { - code: "ERR_INVALID_ARG_TYPE", - name: "TypeError", - message: 'The "url" argument must be of type string.', - }, - ); + [null, "null"], + [true, "type boolean (true)"], + [false, "type boolean (false)"], + [0.0, "type number (0)"], + [0, "type number (0)"], + [[], "an instance of Array"], + [{}, "an instance of Object"], + [() => {}, "function "], + [Symbol("foo"), "type symbol (Symbol(foo))"], + ].forEach(([val, received]) => { + assert.throws(() => url.parse(val), { + code: "ERR_INVALID_ARG_TYPE", + name: "TypeError", + message: `The "url" argument must be of type string. Received ${received}`, + }); }); + }); + test("surfaces the JS engine's URIError for a malformed escape", () => { assert.throws( - () => { - url.parse("http://%E0%A4%A@fail"); - }, - e => { - // The error should be a URIError. - if (!(e instanceof URIError)) return false; - - // The error should be from the JS engine and not from Node.js. - // JS engine errors do not have the `code` property. - return e.code === undefined; - }, - ); - - assert.throws( - () => { - url.parse("http://[127.0.0.1\x00c8763]:8000/"); - }, - { code: "ERR_INVALID_URL", input: "http://[127.0.0.1\x00c8763]:8000/" }, + () => url.parse("http://%E0%A4%A@fail"), + // The error comes from the JS engine, not from us, so it has no `code`. + e => e instanceof URIError && e.code === undefined, ); + }); - if (common.hasIntl) { - // An array of Unicode code points whose Unicode NFKD contains a "bad - // character". - const badIDNA = (() => { - const BAD_CHARS = "#%/:?@[\\]^|"; - const out = []; - for (let i = 0x80; i < 0x110000; i++) { - const cp = String.fromCodePoint(i); - for (const badChar of BAD_CHARS) { - if (cp.normalize("NFKD").includes(badChar)) { - out.push(cp); - } - } - } - return out; - })(); - - // The generation logic above should at a minimum produce these two - // characters. - assert(badIDNA.includes("℀")); - assert(badIDNA.includes("@")); - - for (const badCodePoint of badIDNA) { - const badURL = `http://fail${badCodePoint}fail.com/`; - assert.throws( - () => { - url.parse(badURL); - }, - e => e.code === "ERR_INVALID_URL", - `parsing ${badURL}`, - ); - } + test("rejects a forbidden character in an IPv6 host", () => { + assert.throws(() => url.parse("http://[127.0.0.1\x00c8763]:8000/"), { + code: "ERR_INVALID_URL", + input: "http://[127.0.0.1\x00c8763]:8000/", + }); + }); + test("rejects a hostname that IDNA maps to a forbidden character", () => { + /* + * A slice of the code points whose NFKD contains one of `#%/:?@[\]^|`. + * test/js/node/test/parallel/test-url-parse-invalid-input.js sweeps the + * whole range, which is far too slow to repeat under the test runner. + */ + for (const badCodePoint of ["\u2100", "\uFF20", "\uFF1A", "\uFF0F", "\uFF03", "\uFF1F"]) { + const badURL = `http://fail${badCodePoint}fail.com/`; assert.throws( - () => { - url.parse("http://\u00AD/bad.com/"); - }, + () => url.parse(badURL), e => e.code === "ERR_INVALID_URL", - "parsing http://\u00AD/bad.com/", + `parsing ${badURL}`, ); } - { - const badURLs = ["https://evil.com:.example.com", "git+ssh://git@github.com:npm/npm"]; - badURLs.forEach(badURL => { - common.spawnPromisified(process.execPath, ["-e", `url.parse(${JSON.stringify(badURL)})`]).then( - common.mustCall(({ code }) => { - assert.strictEqual(code, 1); - }), - ); - }); + // A hostname that IDNA maps to nothing at all. + assert.throws( + () => url.parse("http://\u00AD/bad.com/"), + e => e.code === "ERR_INVALID_URL", + "parsing http://\u00AD/bad.com/", + ); + }); - badURLs.forEach(badURL => { - assert.throws(() => url.parse(badURL), { - code: "ERR_INVALID_ARG_VALUE", - }); - }); + test("rejects an invalid port", () => { + for (const badURL of ["https://evil.com:.example.com", "git+ssh://git@github.com:npm/npm"]) { + assert.throws(() => url.parse(badURL), { code: "ERR_INVALID_ARG_VALUE" }); } }); + + test("an invalid port is fatal to a script that does not catch it", async () => { + await using proc = Bun.spawn({ + cmd: [bunExe(), "-e", `url.parse("https://evil.com:.example.com")`], + env: bunEnv, + stdout: "pipe", + stderr: "pipe", + }); + + const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]); + + expect(stdout).toBe(""); + expect(stderr).toContain("ERR_INVALID_ARG_VALUE"); + expect(exitCode).toBe(1); + }); }); From 55a8401e51d6a614e11579f055f29dcff59fdfa6 Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Mon, 6 Jul 2026 12:19:36 +0000 Subject: [PATCH 5/6] test: pin the invalid-port error message node passes the reason where ERR_INVALID_ARG_VALUE expects the value, so the message it prints reads oddly. Assert it verbatim, and say so at the throw, to keep the argument order from being "corrected" into a message node never emits. --- src/js/node/url.ts | 2 ++ test/js/node/url/url.test.ts | 11 +++++++++++ 2 files changed, 13 insertions(+) diff --git a/src/js/node/url.ts b/src/js/node/url.ts index d852c2830a35..d4f375f7ab18 100644 --- a/src/js/node/url.ts +++ b/src/js/node/url.ts @@ -467,6 +467,8 @@ function getHostname(self, rest, hostname: string, url) { if (!isValid) { // If leftover starts with :, then it represents an invalid port. if (code === Char.COLON) { + // node passes the reason where ERR_INVALID_ARG_VALUE expects the value, + // which reads oddly. Kept so the message matches node's exactly. throw $ERR_INVALID_ARG_VALUE("url", "Invalid port in url", url); } self.hostname = hostname.slice(0, i); diff --git a/test/js/node/url/url.test.ts b/test/js/node/url/url.test.ts index 4ad8b02904ad..a5b4068021dd 100644 --- a/test/js/node/url/url.test.ts +++ b/test/js/node/url/url.test.ts @@ -145,6 +145,17 @@ describe("Url.prototype.parse host handling", () => { it("reports the whole url as the input of an ERR_INVALID_URL", () => { expect(parseError("http://[127.0.0.1\0c8763]:8000/").input).toBe("http://[127.0.0.1\0c8763]:8000/"); }); + + /* + * This message reads oddly because node passes the reason where + * ERR_INVALID_ARG_VALUE expects the value. It is verbatim what node v26.3.0 + * prints, so don't "fix" the argument order in getHostname. + */ + it("words the invalid port error exactly like node", () => { + expect(parseError("http://h:8a/x").message).toBe( + "The argument 'url' http://h:8a/x. Received 'Invalid port in url'", + ); + }); }); it("URL constructor throws ERR_MISSING_ARGS", () => { From 34eef26062eb1aed192f3c3ca4776b1af013035b Mon Sep 17 00:00:00 2001 From: robobun <117481402+robobun@users.noreply.github.com> Date: Mon, 6 Jul 2026 17:48:03 +0000 Subject: [PATCH 6/6] ci: retrigger Build 69089 went red on two infrastructure failures with no test or compile failure behind them: linux x64-baseline build-cpp expired without ever being assigned an agent (cascading 42 dependents), and darwin 26 aarch64 test-bun aborted on a 120s buildkite-agent artifact download timeout before running a single test. 238 jobs passed.