diff --git a/scripts/build/deps/webkit.ts b/scripts/build/deps/webkit.ts index 359f8a466e32..df8637be0759 100644 --- a/scripts/build/deps/webkit.ts +++ b/scripts/build/deps/webkit.ts @@ -3,7 +3,7 @@ * for local mode. Override via `--webkit-version=` to test a branch. * From https://github.com/oven-sh/WebKit releases. */ -export const WEBKIT_VERSION = "cf1b36ec8703d8e87436094d21d478d358c7d886"; +export const WEBKIT_VERSION = "autobuild-preview-pr-643-d4914100"; /** * WebKit (JavaScriptCore) — the JS engine. diff --git a/src/jsc/URLSearchParams.rs b/src/jsc/URLSearchParams.rs index d66044d09172..eed2b45c3fb5 100644 --- a/src/jsc/URLSearchParams.rs +++ b/src/jsc/URLSearchParams.rs @@ -18,7 +18,7 @@ unsafe extern "C" { self_: &mut URLSearchParams, ctx: *mut c_void, callback: extern "C" fn(ctx: *mut c_void, str: *const EncodedSlice), - ); + ) -> bool; } impl URLSearchParams { @@ -28,11 +28,13 @@ impl URLSearchParams { URLSearchParams__fromJS(value) } + /// Returns false, without a call of `callback`, when the result does not fit in a `WTF::String`. + #[must_use] pub fn to_string( &mut self, ctx: &mut Ctx, callback: fn(ctx: &mut Ctx, str: EncodedSlice), - ) { + ) -> bool { // A fn pointer cannot be a const generic, so pack (ctx, callback) on the // stack and pass the pair through the C trampoline's void* context. struct Wrap<'a, Ctx> { @@ -53,6 +55,6 @@ impl URLSearchParams { let mut w = Wrap { ctx, callback }; // `w` lives for the duration of the call (URLSearchParams__toString invokes // the callback synchronously, does not retain it). - URLSearchParams__toString(self, (&raw mut w).cast::(), cb::); + URLSearchParams__toString(self, (&raw mut w).cast::(), cb::) } } diff --git a/src/jsc/VirtualMachine.rs b/src/jsc/VirtualMachine.rs index 6155a2b59163..2360cd7f7eee 100644 --- a/src/jsc/VirtualMachine.rs +++ b/src/jsc/VirtualMachine.rs @@ -52,6 +52,18 @@ pub fn synthetic_allocation_limit() -> usize { // `Bun__stringSyntheticAllocationLimit`. pub use bun_core::STRING_ALLOCATION_LIMIT; +unsafe extern "C" { + safe fn Bun__setURLMaximumLengthForTesting(limit: usize); +} + +/// Sets the limit for Rust, Bun's C++ and WTF's URL parser. Returns the previous limit. +pub(crate) fn set_synthetic_allocation_limit(limit: usize) -> usize { + let previous = SYNTHETIC_ALLOCATION_LIMIT.swap(limit, core::sync::atomic::Ordering::Relaxed); + STRING_ALLOCATION_LIMIT.store(limit, core::sync::atomic::Ordering::Relaxed); + Bun__setURLMaximumLengthForTesting(limit); + previous +} + // ────────────────────────────────────────────────────────────────────────── // Type aliases // ────────────────────────────────────────────────────────────────────────── @@ -3812,9 +3824,7 @@ impl VirtualMachine { if let Some(value) = map.get(b"BUN_FEATURE_FLAG_SYNTHETIC_MEMORY_LIMIT") { match bun_core::fmt::parse_int::(value, 10).ok() { Some(limit) => { - SYNTHETIC_ALLOCATION_LIMIT - .store(limit, core::sync::atomic::Ordering::Relaxed); - STRING_ALLOCATION_LIMIT.store(limit, core::sync::atomic::Ordering::Relaxed); + set_synthetic_allocation_limit(limit); } None => bun_core::Output::panic(format_args!( "BUN_FEATURE_FLAG_SYNTHETIC_MEMORY_LIMIT must be a positive integer" @@ -5310,8 +5320,7 @@ impl VirtualMachine { /// Put the startup value back so a file's limit stays with that file. fn undo_synthetic_allocation_limit(&mut self) { if let Some(limit) = self.test_isolation_state.synthetic_allocation_limit { - SYNTHETIC_ALLOCATION_LIMIT.store(limit, core::sync::atomic::Ordering::Relaxed); - STRING_ALLOCATION_LIMIT.store(limit, core::sync::atomic::Ordering::Relaxed); + set_synthetic_allocation_limit(limit); } } diff --git a/src/jsc/bindings/DOMFormData.cpp b/src/jsc/bindings/DOMFormData.cpp index 4f3bca439c39..c2f85f430e41 100644 --- a/src/jsc/bindings/DOMFormData.cpp +++ b/src/jsc/bindings/DOMFormData.cpp @@ -57,18 +57,6 @@ Ref DOMFormData::create(ScriptExecutionContext* context, const Stri return newFormData; } -String DOMFormData::toURLEncodedString() -{ - WTF::URLParser::URLEncodedForm form; - form.reserveInitialCapacity(m_items.size()); - for (auto& item : m_items) { - if (auto value = std::get_if(&item.data)) - form.append({ item.name, *value }); - } - - return WTF::URLParser::serialize(form); -} - extern "C" void DOMFormData__forEach(DOMFormData* form, void* context, void (*callback)(void* context, EncodedSlice*, void*, EncodedSlice*, uint8_t)) { for (auto& item : form->items()) { diff --git a/src/jsc/bindings/DOMFormData.h b/src/jsc/bindings/DOMFormData.h index c969b503bdd2..6f4029414b76 100644 --- a/src/jsc/bindings/DOMFormData.h +++ b/src/jsc/bindings/DOMFormData.h @@ -77,8 +77,6 @@ class DOMFormData : public RefCounted, public ContextDestructionObs size_t count() const { return m_items.size(); } size_t memoryCost() const; - String toURLEncodedString(); - class Iterator { public: explicit Iterator(DOMFormData&); diff --git a/src/jsc/bindings/DOMURL.cpp b/src/jsc/bindings/DOMURL.cpp index 23668383be86..bf736ecbd675 100644 --- a/src/jsc/bindings/DOMURL.cpp +++ b/src/jsc/bindings/DOMURL.cpp @@ -28,10 +28,25 @@ #include "NodeURLHelpers.h" #include "URLSearchParams.h" +#include #include +extern "C" size_t Bun__stringSyntheticAllocationLimit; + +// WTF's URL parser keeps its own copy of the limit that setSyntheticAllocationLimitForTesting lowers. +extern "C" void Bun__setURLMaximumLengthForTesting(size_t limit) +{ + WTF::URLParser::setMaximumLengthForTesting(static_cast(std::min(limit, String::MaxLength))); +} + namespace WebCore { +// A null input parses to the null URL too. Input that is not a URL gives an invalid URL that keeps the input. +static bool isTooLong(const URL& parsed, const String& input) +{ + return doesNotFitInString(parsed) && !input.isNull(); +} + // The WHATWG parser (WebKit) fast-paths all-ASCII hosts without validating // xn-- labels; Node's ada rejects invalid punycode in special-scheme hosts. // `input` is the string the host was parsed from (a base URL's host was checked when the base was parsed). @@ -94,6 +109,8 @@ inline DOMURL::DOMURL(URL&& completeURL) ExceptionOr> DOMURL::create(const String& url) { URL completeURL { url }; + if (isTooLong(completeURL, url)) [[unlikely]] + return Exception { OutOfMemoryError }; if (!completeURL.isValid() || !hasValidParsedHost(completeURL, url)) return Exception { InvalidURLError, url }; return adoptRef(*new DOMURL(WTF::move(completeURL))); @@ -103,17 +120,20 @@ ExceptionOr> DOMURL::create(const String& url, const URL& base, cons { ASSERT(base.isValid() || base.isNull()); URL completeURL { base, url }; + if (isTooLong(completeURL, url)) [[unlikely]] + return Exception { OutOfMemoryError }; if (!completeURL.isValid() || !hasValidParsedHost(completeURL, url)) return Exception { InvalidURLError, url, baseInput }; return adoptRef(*new DOMURL(WTF::move(completeURL))); } -// A null URL means the base did not parse or has an invalid host. -static URL parseBase(const String& base, DOMURL::BaseURLCache* cache) +// A null URL means the base did not parse or has an invalid host. tooLong tells a base that does not fit in a String from those. +static URL parseBase(const String& base, DOMURL::BaseURLCache* cache, bool& tooLong) { if (cache && cache->input == base) [[likely]] return cache->url; URL baseURL { base }; + tooLong = isTooLong(baseURL, base); if (!baseURL.isValid() || !hasValidParsedHost(baseURL, base)) return {}; if (cache) { @@ -125,7 +145,10 @@ static URL parseBase(const String& base, DOMURL::BaseURLCache* cache) ExceptionOr> DOMURL::create(const String& url, const String& base, BaseURLCache* cache) { - URL baseURL = base.isNull() ? URL {} : parseBase(base, cache); + bool baseIsTooLong = false; + URL baseURL = base.isNull() ? URL {} : parseBase(base, cache, baseIsTooLong); + if (baseIsTooLong) [[unlikely]] + return Exception { OutOfMemoryError }; if (!base.isNull() && !baseURL.isValid()) return Exception { InvalidURLError, url, base }; return create(url, baseURL, base); @@ -135,7 +158,8 @@ DOMURL::~DOMURL() = default; static URL parseInternal(const String& url, const String& base, DOMURL::BaseURLCache* cache) { - URL baseURL = base.isNull() ? URL {} : parseBase(base, cache); + bool baseIsTooLong = false; + URL baseURL = base.isNull() ? URL {} : parseBase(base, cache, baseIsTooLong); if (!base.isNull() && !baseURL.isValid()) return {}; URL result { baseURL, url }; @@ -160,32 +184,82 @@ bool DOMURL::canParse(const String& url, const String& base, BaseURLCache* cache ExceptionOr DOMURL::setHref(const String& url) { URL completeURL { URL {}, url }; + if (isTooLong(completeURL, url)) [[unlikely]] + return Exception { OutOfMemoryError }; if (!completeURL.isValid() || !hasValidParsedHost(completeURL, url)) return Exception { InvalidURLError, url }; m_url = WTF::move(completeURL); m_searchParamsDirty = false; + m_pendingSearchParamsLength = 0; if (m_searchParams) m_searchParams->updateFromAssociatedURL(); return {}; } +// Per the URL spec the setters ignore a value that is not valid. They only report a URL that does not fit in a String. +ExceptionOr DOMURL::setFullURL(const URL& fullURL) +{ + if (doesNotFitInString(fullURL)) [[unlikely]] + return Exception { OutOfMemoryError }; + auto result = setHref(fullURL.string()); + if (result.hasException() && result.exception().code() != OutOfMemoryError) + return {}; + return result; +} + +static size_t maximumURLLength() +{ + return std::min(String::MaxLength, Bun__stringSyntheticAllocationLimit); +} + +// "(" in the query of the URL is "%28=" when the params serialize it. +static constexpr uint64_t maximumGrowthOfSerializedQuery = 4; + +bool DOMURL::deferSearchParamsUpdate(uint64_t addedLength) +{ + uint64_t lengthBound = maximumGrowthOfSerializedQuery * m_url.string().length() + m_pendingSearchParamsLength + addedLength; + // Half of the limit leaves WTF::URLParser the room it reserves. Past it the URL takes the pairs at once, which is exact. + if (lengthBound > maximumURLLength() / 2) [[unlikely]] + return false; + m_pendingSearchParamsLength += static_cast(addedLength); + m_searchParamsDirty = true; + return true; +} + +bool DOMURL::updateFromSearchParams() +{ + bool wasDirty = std::exchange(m_searchParamsDirty, true); + if (flushPendingSearchParamsUpdate()) + return true; + // URLSearchParams puts its pairs back, so the URL is as much behind them as it was. + m_searchParamsDirty = wasDirty; + return false; +} + // The update steps invoked on URLSearchParams::{append,set,delete,sort} set // m_searchParamsDirty instead of eagerly re-serializing m_url on every call so // that N appends through url.searchParams stay O(N) instead of O(N^2). All // reads of m_url (href/fullURL) call this first to reconcile. -void DOMURL::flushPendingSearchParamsUpdate() const +// False when the URL does not fit in a String with the new query. It then keeps the query it has. +bool DOMURL::flushPendingSearchParamsUpdate() const { if (!m_searchParamsDirty) [[likely]] - return; - m_searchParamsDirty = false; + return true; auto* self = const_cast(this); - if (!self->m_searchParams) - return; - auto serialized = self->m_searchParams->toString(); - if (serialized.isEmpty()) - self->m_url.setQuery({}); - else - self->m_url.setQuery(WTF::move(serialized)); + if (self->m_searchParams) { + auto serialized = self->m_searchParams->toString(); + if (serialized.hasException()) [[unlikely]] + return false; + String query = serialized.releaseReturnValue(); + URL url = m_url; + url.setQuery(query.isEmpty() ? StringView() : StringView(query)); + if (!url.isValid()) [[unlikely]] + return false; + self->m_url = WTF::move(url); + } + m_searchParamsDirty = false; + m_pendingSearchParamsLength = 0; + return true; } URLSearchParams& DOMURL::searchParams() diff --git a/src/jsc/bindings/DOMURL.h b/src/jsc/bindings/DOMURL.h index 6928f44f3052..70657f4d9db4 100644 --- a/src/jsc/bindings/DOMURL.h +++ b/src/jsc/bindings/DOMURL.h @@ -57,7 +57,10 @@ class DOMURL final : public RefCounted, public CanMakeWeakPtr, p ExceptionOr setHref(const String&); URLSearchParams& searchParams(); - void markSearchParamsDirty() { m_searchParamsDirty = true; } + // For a change of the searchParams that adds at most addedLength characters. True: the URL takes the pairs at its next read. + bool deferSearchParamsUpdate(uint64_t addedLength); + // Takes the pairs now. False if the URL does not fit in a String with them. It is then as it was. + bool updateFromSearchParams(); size_t memoryCost() const { @@ -77,11 +80,13 @@ class DOMURL final : public RefCounted, public CanMakeWeakPtr, p flushPendingSearchParamsUpdate(); return m_url; } - void setFullURL(const URL& fullURL) final { setHref(fullURL.string()); } - void flushPendingSearchParamsUpdate() const; + ExceptionOr setFullURL(const URL&) final; + bool flushPendingSearchParamsUpdate() const; URL m_url; RefPtr m_searchParams; + // At least what the changes since the last flush add to the query. deferSearchParamsUpdate() keeps it under 2^30. + mutable uint32_t m_pendingSearchParamsLength { 0 }; uint16_t m_initialURLCostForGC { 0 }; mutable bool m_searchParamsDirty { false }; }; diff --git a/src/jsc/bindings/ImportMetaObject.cpp b/src/jsc/bindings/ImportMetaObject.cpp index dd998bb947f7..49d23b77dc5c 100644 --- a/src/jsc/bindings/ImportMetaObject.cpp +++ b/src/jsc/bindings/ImportMetaObject.cpp @@ -410,6 +410,11 @@ JSC_DEFINE_HOST_FUNCTION(functionImportMeta__resolve, } WTF::URL url(fromURL, specifier); + // The null URL is a URL that does not fit in a String. + if (url.isNull()) [[unlikely]] { + throwOutOfMemoryError(globalObject, scope); + return {}; + } RELEASE_AND_RETURN(scope, JSValue::encode(jsString(vm, url.string()))); } diff --git a/src/jsc/bindings/URLDecomposition.cpp b/src/jsc/bindings/URLDecomposition.cpp index a59c4747e242..2d2bd13dc2fb 100644 --- a/src/jsc/bindings/URLDecomposition.cpp +++ b/src/jsc/bindings/URLDecomposition.cpp @@ -63,11 +63,11 @@ String URLDecomposition::protocol() const return makeString(fullURL.protocol(), ':'); } -void URLDecomposition::setProtocol(StringView value) +ExceptionOr URLDecomposition::setProtocol(StringView value) { URL copy = fullURL(); copy.setProtocol(value); - setFullURL(copy); + return setFullURL(copy); } String URLDecomposition::username() const @@ -75,13 +75,13 @@ String URLDecomposition::username() const return fullURL().encodedUser().toString(); } -void URLDecomposition::setUsername(StringView user) +ExceptionOr URLDecomposition::setUsername(StringView user) { auto fullURL = this->fullURL(); if (fullURL.host().isEmpty() || fullURL.protocolIsFile()) - return; + return {}; fullURL.setUser(user); - setFullURL(fullURL); + return setFullURL(fullURL); } String URLDecomposition::password() const @@ -89,13 +89,13 @@ String URLDecomposition::password() const return fullURL().encodedPassword().toString(); } -void URLDecomposition::setPassword(StringView password) +ExceptionOr URLDecomposition::setPassword(StringView password) { auto fullURL = this->fullURL(); if (fullURL.host().isEmpty() || fullURL.protocolIsFile()) - return; + return {}; fullURL.setPassword(password); - setFullURL(fullURL); + return setFullURL(fullURL); } String URLDecomposition::host() const @@ -113,18 +113,18 @@ static unsigned countASCIIDigits(StringView string) return length; } -void URLDecomposition::setHost(StringView value) +ExceptionOr URLDecomposition::setHost(StringView value) { auto fullURL = this->fullURL(); if (value.isEmpty() && !fullURL.protocolIsFile() && fullURL.hasSpecialScheme()) - return; + return {}; size_t separator = value.reverseFind(':'); if (!separator) - return; + return {}; if (fullURL.hasOpaquePath()) - return; + return {}; // No port if no colon or rightmost colon is within the IPv6 section. size_t ipv6Separator = value.reverseFind(']'); @@ -133,7 +133,7 @@ void URLDecomposition::setHost(StringView value) else { // Multiple colons are acceptable only in case of IPv6. if (value.find(':') != separator && ipv6Separator == notFound) - return; + return {}; unsigned portLength = countASCIIDigits(value.substring(separator + 1)); if (!portLength) { fullURL.setHost(value.left(separator)); @@ -145,8 +145,9 @@ void URLDecomposition::setHost(StringView value) fullURL.setHostAndPort(value.left(separator + 1 + portLength)); } } - if (fullURL.isValid() && hasAcceptableHost(fullURL)) - setFullURL(fullURL); + if (doesNotFitInString(fullURL) || (fullURL.isValid() && hasAcceptableHost(fullURL))) + return setFullURL(fullURL); + return {}; } String URLDecomposition::hostname() const @@ -154,16 +155,17 @@ String URLDecomposition::hostname() const return fullURL().host().toString(); } -void URLDecomposition::setHostname(StringView host) +ExceptionOr URLDecomposition::setHostname(StringView host) { auto fullURL = this->fullURL(); if (host.isEmpty() && !fullURL.protocolIsFile() && fullURL.hasSpecialScheme()) - return; + return {}; if (fullURL.hasOpaquePath()) - return; + return {}; fullURL.setHost(host); - if (fullURL.isValid() && hasAcceptableHost(fullURL)) - setFullURL(fullURL); + if (doesNotFitInString(fullURL) || (fullURL.isValid() && hasAcceptableHost(fullURL))) + return setFullURL(fullURL); + return {}; } String URLDecomposition::port() const @@ -201,16 +203,16 @@ std::optional> URLDecomposition::parsePort(StringView st return { { static_cast(port) } }; } -void URLDecomposition::setPort(StringView value) +ExceptionOr URLDecomposition::setPort(StringView value) { auto fullURL = this->fullURL(); if (fullURL.host().isEmpty() || fullURL.protocolIsFile()) - return; + return {}; auto port = parsePort(value, fullURL.protocol()); if (!port) - return; + return {}; fullURL.setPort(*port); - setFullURL(fullURL); + return setFullURL(fullURL); } String URLDecomposition::pathname() const @@ -218,13 +220,13 @@ String URLDecomposition::pathname() const return fullURL().path().toString(); } -void URLDecomposition::setPathname(StringView value) +ExceptionOr URLDecomposition::setPathname(StringView value) { auto fullURL = this->fullURL(); if (fullURL.hasOpaquePath()) - return; + return {}; fullURL.setPath(value); - setFullURL(fullURL); + return setFullURL(fullURL); } String URLDecomposition::search() const @@ -233,7 +235,22 @@ String URLDecomposition::search() const return fullURL.query().isEmpty() ? emptyString() : fullURL.queryWithLeadingQuestionMark().toString(); } -void URLDecomposition::setSearch(const String& value) +// makeStringByReplacingAll() crashes when its result does not fit in a String. +static bool fitsInStringWithNumberSignsEscaped(const String& value) +{ + if (value.length() <= String::MaxLength / 3) [[likely]] + return true; + auto countNumberSigns = [](auto characters) { + size_t count = 0; + for (auto character : characters) + count += character == '#'; + return count; + }; + size_t numberSigns = value.is8Bit() ? countNumberSigns(value.span8()) : countNumberSigns(value.span16()); + return value.length() + 2 * numberSigns <= String::MaxLength; +} + +ExceptionOr URLDecomposition::setSearch(const String& value) { auto fullURL = this->fullURL(); if (value.isEmpty()) { @@ -241,9 +258,11 @@ void URLDecomposition::setSearch(const String& value) fullURL.setQuery({}); } else { // Make sure that '#' in the query does not leak to the hash. + if (!fitsInStringWithNumberSignsEscaped(value)) [[unlikely]] + return Exception { OutOfMemoryError }; fullURL.setQuery(makeStringByReplacingAll(value, '#', "%23"_s)); } - setFullURL(fullURL); + return setFullURL(fullURL); } String URLDecomposition::hash() const @@ -252,14 +271,14 @@ String URLDecomposition::hash() const return fullURL.fragmentIdentifier().isEmpty() ? emptyString() : fullURL.fragmentIdentifierWithLeadingNumberSign().toString(); } -void URLDecomposition::setHash(StringView value) +ExceptionOr URLDecomposition::setHash(StringView value) { auto fullURL = this->fullURL(); if (value.isEmpty()) fullURL.removeFragmentIdentifier(); else fullURL.setFragmentIdentifier(value.startsWith('#') ? value.substring(1) : value); - setFullURL(fullURL); + return setFullURL(fullURL); } } diff --git a/src/jsc/bindings/URLDecomposition.h b/src/jsc/bindings/URLDecomposition.h index 47f88b5f41dd..a9ecdc7e16c3 100644 --- a/src/jsc/bindings/URLDecomposition.h +++ b/src/jsc/bindings/URLDecomposition.h @@ -27,12 +27,16 @@ #include "root.h" +#include "ExceptionOr.h" #include #include namespace WebCore { +// WTF::URL is the null URL after a parse or a setter whose result does not fit in a String. +inline bool doesNotFitInString(const URL& url) { return url.isNull(); } + class URLDecomposition { public: // Parse a port string with optional protocol for default port detection @@ -42,38 +46,39 @@ class URLDecomposition { String origin() const; WEBCORE_EXPORT String protocol() const; - void setProtocol(StringView); + ExceptionOr setProtocol(StringView); String username() const; - void setUsername(StringView); + ExceptionOr setUsername(StringView); String password() const; - void setPassword(StringView); + ExceptionOr setPassword(StringView); WEBCORE_EXPORT String host() const; - void setHost(StringView); + ExceptionOr setHost(StringView); WEBCORE_EXPORT String hostname() const; - void setHostname(StringView); + ExceptionOr setHostname(StringView); WEBCORE_EXPORT String port() const; - void setPort(StringView); + ExceptionOr setPort(StringView); WEBCORE_EXPORT String pathname() const; - void setPathname(StringView); + ExceptionOr setPathname(StringView); WEBCORE_EXPORT String search() const; - void setSearch(const String&); + ExceptionOr setSearch(const String&); WEBCORE_EXPORT String hash() const; - void setHash(StringView); + ExceptionOr setHash(StringView); protected: virtual ~URLDecomposition() = default; private: virtual URL fullURL() const = 0; - virtual void setFullURL(const URL&) = 0; + // The setters forward the result. A URL that does not fit in a String is the one failure. + virtual ExceptionOr setFullURL(const URL&) = 0; }; } // namespace WebCore diff --git a/src/jsc/bindings/URLSearchParams.cpp b/src/jsc/bindings/URLSearchParams.cpp index 9f7b1efb543c..fa45a167a8cf 100644 --- a/src/jsc/bindings/URLSearchParams.cpp +++ b/src/jsc/bindings/URLSearchParams.cpp @@ -39,11 +39,22 @@ extern "C" WebCore::URLSearchParams* URLSearchParams__fromJS(JSC::EncodedJSValue // callback accepting a void* and a const EncodedSlice*, returning void typedef void (*URLSearchParams__toStringCallback)(void* ctx, const EncodedSlice* str); -extern "C" void URLSearchParams__toString(WebCore::URLSearchParams* urlSearchParams, void* ctx, URLSearchParams__toStringCallback callback) +// False, and no call of the callback, when the params do not fit in a String. +extern "C" bool URLSearchParams__toString(WebCore::URLSearchParams* urlSearchParams, void* ctx, URLSearchParams__toStringCallback callback) { - String str = urlSearchParams->toString(); + auto result = urlSearchParams->toString(); + if (result.hasException()) [[unlikely]] + return false; + String str = result.releaseReturnValue(); auto slice = Zig::toEncodedSlice(str); callback(ctx, &slice); + return true; +} + +// A code unit is at most 3 UTF-8 bytes (2 in an 8-bit string), and toString() gives at most 3 characters for a byte. +static uint64_t serializedLengthBound(const String& string) +{ + return static_cast(string.length()) * (string.is8Bit() ? 6 : 9); } URLSearchParams::URLSearchParams(const String& init, DOMURL* associatedURL) @@ -90,45 +101,75 @@ bool URLSearchParams::has(const StringView name, const String& value) const return false; } -void URLSearchParams::sort() +// Applies the change and tells the URL. If the URL cannot take it, which needs gigabytes, the pairs go back to what they were. +template +ALWAYS_INLINE ExceptionOr URLSearchParams::changePairs(uint64_t addedLength, const Change& change) { - std::stable_sort(m_pairs.begin(), m_pairs.end(), [](const auto& a, const auto& b) { - return WTF::codePointCompareLessThan(a.key, b.key); - }); - updateURL(); - needsSorting = false; + auto* url = m_associatedURL.get(); + if (url && !url->deferSearchParamsUpdate(addedLength)) [[unlikely]] + return changePairsAndUpdateURL(*url, change); + change(); + return {}; } -void URLSearchParams::set(const String& name, const String& value) +template +NEVER_INLINE ExceptionOr URLSearchParams::changePairsAndUpdateURL(DOMURL& url, const Change& change) { - for (auto& pair : m_pairs) { - if (pair.key != name) - continue; - if (pair.value != value) - pair.value = value; - bool skippedFirstMatch = false; - m_pairs.removeAllMatching([&](const auto& pair) { - if (pair.key == name) { - if (skippedFirstMatch) - return true; - skippedFirstMatch = true; - } - return false; + auto pairsBefore = m_pairs; + change(); + if (!url.updateFromSearchParams()) { + m_pairs = WTF::move(pairsBefore); + return Exception { OutOfMemoryError }; + } + return {}; +} + +ExceptionOr URLSearchParams::sort() +{ + auto result = changePairs(0, [&] { + std::stable_sort(m_pairs.begin(), m_pairs.end(), [](const auto& a, const auto& b) { + return WTF::codePointCompareLessThan(a.key, b.key); }); - updateURL(); + }); + if (!result.hasException()) + needsSorting = false; + return result; +} + +ExceptionOr URLSearchParams::set(const String& name, const String& value) +{ + auto result = changePairs(serializedLengthBound(name) + serializedLengthBound(value) + 2, [&] { + for (auto& pair : m_pairs) { + if (pair.key != name) + continue; + if (pair.value != value) + pair.value = value; + bool skippedFirstMatch = false; + m_pairs.removeAllMatching([&](const auto& pair) { + if (pair.key == name) { + if (skippedFirstMatch) + return true; + skippedFirstMatch = true; + } + return false; + }); + return; + } + m_pairs.append({ name, value }); + }); + if (!result.hasException()) needsSorting = true; - return; - } - m_pairs.append({ name, value }); - needsSorting = true; - updateURL(); + return result; } -void URLSearchParams::append(const String& name, const String& value) +ExceptionOr URLSearchParams::append(const String& name, const String& value) { - m_pairs.append({ name, value }); - updateURL(); - needsSorting = true; + auto result = changePairs(serializedLengthBound(name) + serializedLengthBound(value) + 2, [&] { + m_pairs.append({ name, value }); + }); + if (!result.hasException()) + needsSorting = true; + return result; } Vector URLSearchParams::getAll(const StringView name) const @@ -143,24 +184,24 @@ Vector URLSearchParams::getAll(const StringView name) const return values; } -void URLSearchParams::remove(const StringView name, const String& value) +ExceptionOr URLSearchParams::remove(const StringView name, const String& value) { - m_pairs.removeAllMatching([&](const auto& pair) { - return pair.key == name && (value.isNull() || pair.value == value); + auto result = changePairs(0, [&] { + m_pairs.removeAllMatching([&](const auto& pair) { + return pair.key == name && (value.isNull() || pair.value == value); + }); }); - updateURL(); - needsSorting = true; -} - -String URLSearchParams::toString() const -{ - return WTF::URLParser::serialize(m_pairs); + if (!result.hasException()) + needsSorting = true; + return result; } -void URLSearchParams::updateURL() +ExceptionOr URLSearchParams::toString() const { - if (m_associatedURL) - m_associatedURL->markSearchParamsDirty(); + auto serialized = WTF::URLParser::trySerialize(m_pairs); + if (!serialized) [[unlikely]] + return Exception { OutOfMemoryError }; + return WTF::move(*serialized); } void URLSearchParams::updateFromAssociatedURL() diff --git a/src/jsc/bindings/URLSearchParams.h b/src/jsc/bindings/URLSearchParams.h index 2bc00cb59087..f28cbab40790 100644 --- a/src/jsc/bindings/URLSearchParams.h +++ b/src/jsc/bindings/URLSearchParams.h @@ -46,15 +46,17 @@ class URLSearchParams : public RefCounted { return adoptRef(*new URLSearchParams(string, associatedURL)); } - void append(const String& name, const String& value); - void remove(const StringView name, const String& value = {}); + // append, remove, set and sort throw when the params belong to a URL that does not fit in a String with the change. + ExceptionOr append(const String& name, const String& value); + ExceptionOr remove(const StringView name, const String& value = {}); String get(const StringView name) const; Vector getAll(const StringView name) const; bool has(const StringView name, const String& value = {}) const; - void set(const String& name, const String& value); - String toString() const; + ExceptionOr set(const String& name, const String& value); + // Throws when the result does not fit in a String. + ExceptionOr toString() const; void updateFromAssociatedURL(); - void sort(); + ExceptionOr sort(); size_t size() const { return m_pairs.size(); } size_t memoryCost() const; @@ -74,7 +76,8 @@ class URLSearchParams : public RefCounted { const Vector>& pairs() const { return m_pairs; } URLSearchParams(const String&, DOMURL*); URLSearchParams(const Vector>&); - void updateURL(); + template ExceptionOr changePairs(uint64_t addedLength, const Change&); + template ExceptionOr changePairsAndUpdateURL(DOMURL&, const Change&); WeakPtr m_associatedURL; Vector> m_pairs; diff --git a/src/jsc/bindings/bindings.cpp b/src/jsc/bindings/bindings.cpp index 890fea174a97..e9487a419aea 100644 --- a/src/jsc/bindings/bindings.cpp +++ b/src/jsc/bindings/bindings.cpp @@ -6273,16 +6273,6 @@ CPP_DECL size_t WebCore__DOMFormData__count(WebCore::DOMFormData* arg0) return arg0->count(); } -extern "C" void DOMFormData__toQueryString( - DOMFormData* formData, - void* ctx, - void (*callback)(void* ctx, EncodedSlice* encoded)) -{ - auto str = formData->toURLEncodedString(); - EncodedSlice encoded = toEncodedSlice(str); - callback(ctx, &encoded); -} - CPP_DECL JSC::EncodedJSValue WebCore__DOMFormData__createFromURLQuery(JSC::JSGlobalObject* arg0, const EncodedSlice* arg1) { Zig::GlobalObject* globalObject = static_cast(arg0); diff --git a/src/jsc/bindings/webcore/URLPattern.cpp b/src/jsc/bindings/webcore/URLPattern.cpp index e6ca32fff62e..b60d1b9cae4e 100644 --- a/src/jsc/bindings/webcore/URLPattern.cpp +++ b/src/jsc/bindings/webcore/URLPattern.cpp @@ -146,11 +146,23 @@ static ExceptionOr processInit(URLPatternInit&& init, BaseURLStr result.protocol = protocolResult.releaseReturnValue(); } - if (!init.username.isNull()) - result.username = canonicalizeUsername(init.username, type); + if (!init.username.isNull()) { + auto usernameResult = canonicalizeUsername(init.username, type); - if (!init.password.isNull()) - result.password = canonicalizePassword(init.password, type); + if (usernameResult.hasException()) + return usernameResult.releaseException(); + + result.username = usernameResult.releaseReturnValue(); + } + + if (!init.password.isNull()) { + auto passwordResult = canonicalizePassword(init.password, type); + + if (passwordResult.hasException()) + return passwordResult.releaseException(); + + result.password = passwordResult.releaseReturnValue(); + } if (!init.hostname.isNull()) { auto hostResult = canonicalizeHostname(init.hostname, type); diff --git a/src/jsc/bindings/webcore/URLPatternCanonical.cpp b/src/jsc/bindings/webcore/URLPatternCanonical.cpp index 35a2f56f7c59..0ae581775af6 100644 --- a/src/jsc/bindings/webcore/URLPatternCanonical.cpp +++ b/src/jsc/bindings/webcore/URLPatternCanonical.cpp @@ -87,7 +87,7 @@ ExceptionOr canonicalizeProtocol(StringView value, BaseURLStringType val } // https://urlpattern.spec.whatwg.org/#canonicalize-a-username, combined with https://urlpattern.spec.whatwg.org/#process-username-for-init -String canonicalizeUsername(StringView value, BaseURLStringType valueType) +ExceptionOr canonicalizeUsername(StringView value, BaseURLStringType valueType) { if (value.isEmpty()) return value.toString(); @@ -97,12 +97,14 @@ String canonicalizeUsername(StringView value, BaseURLStringType valueType) URL dummyURL(dummyURLCharacters); dummyURL.setUser(value); + if (doesNotFitInString(dummyURL)) [[unlikely]] + return Exception { ExceptionCode::OutOfMemoryError }; return dummyURL.encodedUser().toString(); } // https://urlpattern.spec.whatwg.org/#canonicalize-a-password, combined with https://urlpattern.spec.whatwg.org/#process-password-for-init -String canonicalizePassword(StringView value, BaseURLStringType valueType) +ExceptionOr canonicalizePassword(StringView value, BaseURLStringType valueType) { if (value.isEmpty()) return value.toString(); @@ -112,6 +114,8 @@ String canonicalizePassword(StringView value, BaseURLStringType valueType) URL dummyURL(dummyURLCharacters); dummyURL.setPassword(value); + if (doesNotFitInString(dummyURL)) [[unlikely]] + return Exception { ExceptionCode::OutOfMemoryError }; return dummyURL.encodedPassword().toString(); } @@ -197,6 +201,8 @@ ExceptionOr canonicalizePathname(StringView pathnameValue) // FIXME: Set state override to State::PathStart after URLParser supports state override. URL dummyURL(dummyURLCharacters); dummyURL.setPath(maybeAddSlashPrefix); + if (doesNotFitInString(dummyURL)) [[unlikely]] + return Exception { ExceptionCode::OutOfMemoryError }; ASSERT(dummyURL.isValid()); auto result = dummyURL.path(); @@ -234,6 +240,8 @@ ExceptionOr canonicalizeSearch(StringView value, BaseURLStringType value URL dummyURL(dummyURLCharacters); dummyURL.setQuery(strippedValue); + if (doesNotFitInString(dummyURL)) [[unlikely]] + return Exception { ExceptionCode::OutOfMemoryError }; ASSERT(dummyURL.isValid()); return dummyURL.query().toString(); @@ -252,6 +260,8 @@ ExceptionOr canonicalizeHash(StringView value, BaseURLStringType valueTy URL dummyURL(dummyURLCharacters); dummyURL.setFragmentIdentifier(strippedValue); + if (doesNotFitInString(dummyURL)) [[unlikely]] + return Exception { ExceptionCode::OutOfMemoryError }; ASSERT(dummyURL.isValid()); return dummyURL.fragmentIdentifier().toString(); diff --git a/src/jsc/bindings/webcore/URLPatternCanonical.h b/src/jsc/bindings/webcore/URLPatternCanonical.h index 690fc5b6bf0a..86d5ab33d825 100644 --- a/src/jsc/bindings/webcore/URLPatternCanonical.h +++ b/src/jsc/bindings/webcore/URLPatternCanonical.h @@ -46,8 +46,8 @@ enum class EncodingCallbackType : uint8_t { Protocol, bool isAbsolutePathname(StringView input, BaseURLStringType inputType); ExceptionOr canonicalizeProtocol(StringView, BaseURLStringType valueType); -String canonicalizeUsername(StringView value, BaseURLStringType valueType); -String canonicalizePassword(StringView value, BaseURLStringType valueType); +ExceptionOr canonicalizeUsername(StringView value, BaseURLStringType valueType); +ExceptionOr canonicalizePassword(StringView value, BaseURLStringType valueType); ExceptionOr canonicalizeHostname(StringView value, BaseURLStringType valueType); ExceptionOr canonicalizeIPv6Hostname(StringView value, BaseURLStringType valueType); ExceptionOr canonicalizePort(StringView portValue, StringView protocolValue, BaseURLStringType portValueType); diff --git a/src/jsc/virtual_machine_exports.rs b/src/jsc/virtual_machine_exports.rs index 74c1fb6ecbc3..8a9d308a59cc 100644 --- a/src/jsc/virtual_machine_exports.rs +++ b/src/jsc/virtual_machine_exports.rs @@ -316,9 +316,6 @@ pub fn Bun__setSyntheticAllocationLimitForTesting( let limit: usize = usize::try_from(arg.coerce_to_int64(global)?.max(1024 * 1024)).expect("int cast"); - let prev = crate::virtual_machine::SYNTHETIC_ALLOCATION_LIMIT - .swap(limit, core::sync::atomic::Ordering::Relaxed); - crate::virtual_machine::STRING_ALLOCATION_LIMIT - .store(limit, core::sync::atomic::Ordering::Relaxed); + let prev = crate::virtual_machine::set_synthetic_allocation_limit(limit); Ok(JSValue::js_number(prev as f64)) } diff --git a/src/runtime/webcore/Blob.rs b/src/runtime/webcore/Blob.rs index d5574473c709..3b3bd461ac2b 100644 --- a/src/runtime/webcore/Blob.rs +++ b/src/runtime/webcore/Blob.rs @@ -194,7 +194,7 @@ pub trait BlobExt { fn from_url_search_params( global_this: &JSGlobalObject, search_params: &mut jsc::URLSearchParams, - ) -> Blob + ) -> JsResult where Self: Sized; fn from_dom_form_data(global_this: &JSGlobalObject, form_data: &mut jsc::DOMFormData) -> Blob @@ -839,9 +839,11 @@ impl BlobExt for Blob { fn from_url_search_params( global_this: &JSGlobalObject, search_params: &mut jsc::URLSearchParams, - ) -> Blob { + ) -> JsResult { let mut converter = URLSearchParamsConverter { buf: Vec::new() }; - search_params.to_string(&mut converter, URLSearchParamsConverter::convert); + if !search_params.to_string(&mut converter, URLSearchParamsConverter::convert) { + return Err(global_this.throw_out_of_memory()); + } let store = Store::init(converter.buf); // SAFETY: `store` is the sole +1 on this freshly-allocated Store. unsafe { @@ -859,7 +861,7 @@ impl BlobExt for Blob { let blob = Blob::init_with_store(store, global_this); blob.content_type.set(content_type); blob.content_type_was_set.set(true); - blob + Ok(blob) } fn from_dom_form_data(global_this: &JSGlobalObject, form_data: &mut jsc::DOMFormData) -> Blob { diff --git a/src/runtime/webcore/Body.rs b/src/runtime/webcore/Body.rs index 8e3ef3b70497..33af7edab200 100644 --- a/src/runtime/webcore/Body.rs +++ b/src/runtime/webcore/Body.rs @@ -997,7 +997,7 @@ impl Value { return Ok(Value::Blob(Blob::from_url_search_params( global_this, unsafe { &mut *search_params }, - ))); + )?)); } if js_type == jsc::JSType::DOMWrapper { diff --git a/test/js/bun/resolve/import-meta-resolve.test.mjs b/test/js/bun/resolve/import-meta-resolve.test.mjs index 30042db23626..67efdaa95cc7 100644 --- a/test/js/bun/resolve/import-meta-resolve.test.mjs +++ b/test/js/bun/resolve/import-meta-resolve.test.mjs @@ -97,6 +97,19 @@ exact(() => import.meta.resolve("node:doesnotexist"), "node:doesnotexist"); if (process?.versions?.bun) { exact(() => import.meta.resolve("bun:sqlite"), "bun:sqlite"); exact(() => import.meta.resolve("bun:doesnotexist"), "bun:doesnotexist"); + + // A relative specifier is percent-encoded into a URL, which can be longer than a string can be (2 ** 31 - 1 + // characters). That returned "". U+00E9 is one character and "%C3%A9" is six, and 1 MiB stands in for 2 ** 31 - 1. + const { setSyntheticAllocationLimitForTesting } = await import("bun:internal-for-testing"); + wrapped("a specifier whose URL does not fit in a string", () => { + const specifier = "./" + Buffer.alloc(176_000, 0xe9).toString("latin1"); + const previous = setSyntheticAllocationLimitForTesting(1024 * 1024); + try { + assert.throws(() => import.meta.resolve(specifier), { name: "RangeError", message: "Out of memory" }); + } finally { + setSyntheticAllocationLimitForTesting(previous); + } + }); } fileUrlRelTo(() => import.meta.resolve("./something.node"), "./something.node"); diff --git a/test/js/web/html/URLSearchParams.test.ts b/test/js/web/html/URLSearchParams.test.ts index 28f38c3d245d..55e9bc75d13d 100644 --- a/test/js/web/html/URLSearchParams.test.ts +++ b/test/js/web/html/URLSearchParams.test.ts @@ -1,4 +1,7 @@ -import { describe, expect, it } from "bun:test"; +import { setSyntheticAllocationLimitForTesting } from "bun:internal-for-testing"; +import { describe, expect, it, test } from "bun:test"; +import { bunEnv, bunExe, isASAN, isDebug } from "harness"; +import os from "node:os"; describe("URLSearchParams", () => { it("does not crash when calling .toJSON() on a URLSearchParams object with a large number of properties", () => { @@ -306,3 +309,136 @@ it(".has second argument", () => { expect(params.has("b", 3)).toBe(true); expect(params.has("b", 4)).toBe(false); }); + +// toString() percent-encodes, so its result can be longer than a string can be (2 ** 31 - 1 characters) when every name and +// value fits. That used to abort the process. It throws the RangeError that JSC throws for a string that is too long. +describe("params that do not fit in a string when serialized", () => { + const MiB = 1024 * 1024; + const outOfMemory = { name: "RangeError", message: "Out of memory" }; + // A string of one character. The default is U+00E9. + const repeated = (count: number, character = "\u00e9") => Buffer.alloc(count, character, "latin1").toString("latin1"); + // U+00E9 is one character, and "%C3%A9" is six: 1.2 M characters when serialized. + const tooLong = repeated(200_000); + + // 1 MiB stands in for 2 ** 31 - 1. The limit is process-wide, so each test puts it back. + function withStringLimit(limit: number, fn: () => void) { + const previous = setSyntheticAllocationLimitForTesting(limit); + try { + fn(); + } finally { + setSyntheticAllocationLimitForTesting(previous); + } + } + + // The error, or the length of what was returned. Never the value: under the limit the test runner cannot print it. + function outcome(fn: () => { length: number } | undefined | void) { + try { + return fn()?.length; + } catch (e: any) { + return { name: e.name, message: e.message }; + } + } + + it("toString() throws a RangeError", () => { + withStringLimit(MiB, () => { + const params = new URLSearchParams(); + params.set("a", tooLong); + const twoByteValue = Buffer.alloc(2 * 5_000, "\u4e2d", "utf16le").toString("utf16le"); + const manyPairs = new URLSearchParams(Array.from({ length: 30 }, (_, i) => ["k" + i, twoByteValue])); + expect({ + toString: outcome(() => params.toString()), + string: outcome(() => String(params)), + template: outcome(() => `${params}`), + concatenation: outcome(() => params + ""), + twoByteStrings: outcome(() => manyPairs.toString()), + size: params.size, + value: params.get("a") === tooLong, + }).toEqual({ + toString: outOfMemory, + string: outOfMemory, + template: outOfMemory, + concatenation: outOfMemory, + twoByteStrings: outOfMemory, + size: 1, + value: true, + }); + }); + }); + + it("toString() returns a string of exactly the limit", () => { + withStringLimit(MiB, () => { + const params = new URLSearchParams(); + params.set("a", repeated(MiB - 2, "x")); + const atTheLimit = outcome(() => params.toString()); + params.set("a", repeated(MiB - 1, "x")); + const oneMore = outcome(() => params.toString()); + params.set("a", repeated(100_000)); + const encoded = params.toString(); + expect({ + atTheLimit, + oneMore, + encodedLength: encoded.length, + encodedStart: encoded.slice(0, 14), + roundTrip: new URLSearchParams(encoded).get("a") === repeated(100_000), + }).toEqual({ + atTheLimit: MiB, + oneMore: outOfMemory, + encodedLength: 2 + 600_000, + encodedStart: "a=%C3%A9%C3%A9", + roundTrip: true, + }); + }); + }); + + it("a Response or Request body made from the params throws a RangeError", () => { + withStringLimit(MiB, () => { + const params = new URLSearchParams(); + params.set("a", tooLong); + expect({ + response: outcome(() => void new Response(params)), + request: outcome(() => void new Request("http://example.com/", { method: "POST", body: params })), + }).toEqual({ response: outOfMemory, request: outOfMemory }); + }); + }); + + // A string of 2 ** 30 Latin-1 characters used to abort even when all of them are ASCII and the result fits. The child + // commits about 6 GB, and a debug or ASAN build needs minutes for it. + const memory = Math.min(os.totalmem(), process.constrainedMemory() || Infinity); + test.skipIf(isDebug || isASAN || memory < 16 * 1024 ** 3)( + "at the real limit", + async () => { + await using proc = Bun.spawn({ + cmd: [ + bunExe(), + "-e", + ` + const outcome = fn => { try { return fn()?.length; } catch (e) { return e.name + ": " + e.message; } }; + const params = new URLSearchParams(); + params.set("a", Buffer.alloc(2 ** 29, 0xe9).toString("latin1")); + const encoded = outcome(() => params.toString()); + const response = outcome(() => void new Response(params)); + params.set("a", Buffer.alloc(2 ** 30, "x").toString("latin1")); + const ascii = outcome(() => params.toString()); + console.log(JSON.stringify({ encoded, response, ascii })); + `, + ], + env: bunEnv, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]); + expect({ stdout, stderr, exitCode, signalCode: proc.signalCode }).toEqual({ + stdout: + JSON.stringify({ + encoded: "RangeError: Out of memory", + response: "RangeError: Out of memory", + ascii: 2 + 2 ** 30, + }) + "\n", + stderr: "", + exitCode: 0, + signalCode: null, + }); + }, + 120_000, + ); +}); diff --git a/test/js/web/url/url.test.ts b/test/js/web/url/url.test.ts index 41ba8ec9ee37..f8b7fe37d629 100755 --- a/test/js/web/url/url.test.ts +++ b/test/js/web/url/url.test.ts @@ -1,6 +1,8 @@ +import { setSyntheticAllocationLimitForTesting } from "bun:internal-for-testing"; import { describe, expect, it, test } from "bun:test"; -import { bunEnv, bunExe } from "harness"; +import { bunEnv, bunExe, isASAN, isDebug } from "harness"; import { resolveObjectURL } from "node:buffer"; +import os from "node:os"; import util from "node:util"; describe("url", () => { @@ -686,3 +688,285 @@ describe("object URL prefix check", () => { }); }, 60_000); }); + +// Percent-encoding makes a URL longer than its input: U+00E9 is one character and "%C3%A9" is six. A URL longer than a +// string can be (2 ** 31 - 1 characters) used to abort the process. It throws the RangeError that JSC throws for a string +// that is too long. +describe("a URL that does not fit in a string", () => { + const MiB = 1024 * 1024; + const outOfMemory = { name: "RangeError", message: "Out of memory" }; + // A string of one character. The default is U+00E9. + const repeated = (count: number, character = "\u00e9") => Buffer.alloc(count, character, "latin1").toString("latin1"); + // A little over 1 Mi characters when percent-encoded, and 0.6 M. + const tooLong = repeated(176_000); + const fits = repeated(100_000); + + // 1 MiB stands in for 2 ** 31 - 1. The limit is process-wide, so each test puts it back. + function withStringLimit(limit: number, fn: () => void) { + const previous = setSyntheticAllocationLimitForTesting(limit); + try { + fn(); + } finally { + setSyntheticAllocationLimitForTesting(previous); + } + } + + // The error, or the length of what was returned. Never the value: under the limit the test runner cannot print it. + function outcome(fn: () => { length: number } | undefined | void) { + try { + return fn()?.length; + } catch (e: any) { + return { name: e.name, message: e.message }; + } + } + + it.each([ + ["the query", () => new URL("http://a/?" + tooLong)], + ["the path", () => new URL("http://a/" + tooLong)], + ["the fragment", () => new URL("http://a/#" + tooLong)], + ["the username", () => new URL("http://" + tooLong + "@a/")], + ["an opaque path", () => new URL("foo:" + tooLong)], + [ + "a two-byte string", + () => new URL("http://a/?" + Buffer.alloc(2 * 117_000, "\u4e2d", "utf16le").toString("utf16le")), + ], + ["ASCII that is escaped", () => new URL("http://a/" + repeated(350_000, " ") + "x")], + ["a relative URL", () => new URL("?" + tooLong, "http://a/b")], + ["the base URL", () => new URL("c", "http://a/?" + tooLong)], + ])("the constructor throws a RangeError when %s makes the URL too long", (_, construct) => { + withStringLimit(MiB, () => { + expect(outcome(() => construct().href)).toEqual(outOfMemory); + }); + }); + + it("URL.canParse() returns false and URL.parse() returns null", () => { + withStringLimit(MiB, () => { + expect({ + canParse: URL.canParse("http://a/?" + tooLong), + parse: URL.parse("?" + tooLong, "http://a/"), + }).toEqual({ canParse: false, parse: null }); + }); + }); + + it("a URL that fits parses, and input that is not a URL is a TypeError", () => { + withStringLimit(MiB, () => { + expect({ + query: outcome(() => new URL("http://a/?" + fits).href), + relative: outcome(() => new URL("?" + fits, "http://a/b").href), + setter: outcome(() => { + const url = new URL("http://a/"); + url.hash = fits; + return url.href; + }), + notAURL: outcome(() => new URL("http://a b/?" + tooLong).href), + notAURLWithBase: outcome(() => new URL("//a b/?" + tooLong, "http://c/").href), + }).toEqual({ + query: "http://a/?".length + 6 * fits.length, + relative: "http://a/b?".length + 6 * fits.length, + setter: "http://a/#".length + 6 * fits.length, + notAURL: { name: "TypeError", message: "Invalid URL" }, + notAURLWithBase: { name: "TypeError", message: "Invalid URL" }, + }); + }); + }); + + it.each(["href", "search", "hash", "pathname", "username", "password"] as const)( + "the %s setter throws a RangeError and leaves the URL as it was", + setter => { + withStringLimit(MiB, () => { + const href = "http://u:p@h:8/p?q#f"; + const url = new URL(href); + const params = url.searchParams; + expect({ + error: outcome(() => void (url[setter] = setter === "href" ? "http://a/?" + tooLong : tooLong)), + href: url.href, + params: [...params], + }).toEqual({ error: outOfMemory, href, params: [["q", ""]] }); + }); + }, + ); + + it("every setter throws a RangeError on a URL that is as long as a URL can be", () => { + withStringLimit(MiB, () => { + // The parser refuses input that is too long before it reads it, so input that is not a URL finds the longest + // input fast: a TypeError is input that it read. + let length = 0; + for (let step = MiB / 2; step >= 1; step >>= 1) { + const error = outcome(() => void new URL(repeated(length + step, "1"))); + if (typeof error === "object" && error.name === "TypeError") length += step; + } + const url = new URL("http://a/" + repeated(length - "http://a/".length, "x")); + expect({ + length: url.href.length, + oneMore: outcome(() => new URL(url.href + "x").href), + port: outcome(() => void (url.port = "8080")), + host: outcome(() => void (url.host = "bb")), + hostname: outcome(() => void (url.hostname = "bb")), + protocol: outcome(() => void (url.protocol = "https")), + username: outcome(() => void (url.username = "u")), + password: outcome(() => void (url.password = "p")), + pathname: outcome(() => void (url.pathname = url.pathname + "x")), + search: outcome(() => void (url.search = "q")), + hash: outcome(() => void (url.hash = "f")), + lengthAfter: url.href.length, + // A value that does not make the URL longer is fine. + sameLength: outcome(() => void (url.hostname = "b")), + hostnameAfter: url.hostname, + }).toEqual({ + length, + oneMore: outOfMemory, + port: outOfMemory, + host: outOfMemory, + hostname: outOfMemory, + protocol: outOfMemory, + username: outOfMemory, + password: outOfMemory, + pathname: outOfMemory, + search: outOfMemory, + hash: outOfMemory, + lengthAfter: length, + sameLength: undefined, + hostnameAfter: "b", + }); + }); + }); + + it("url.searchParams.append() and set() throw a RangeError and leave the URL and the params as they were", () => { + withStringLimit(MiB, () => { + const url = new URL("http://a/?x=1#f"); + const params = url.searchParams; + expect({ + append: outcome(() => params.append("a", tooLong)), + setExisting: outcome(() => params.set("x", tooLong)), + setNew: outcome(() => params.set("b", tooLong)), + href: url.href, + entries: [...params], + }).toEqual({ + append: outOfMemory, + setExisting: outOfMemory, + setNew: outOfMemory, + href: "http://a/?x=1#f", + entries: [["x", "1"]], + }); + + // Values that fit one by one. The URL takes them at its next read, until one more may be too long. From then on + // each append serializes, and the one that does not fit throws. + const value = repeated(10_000); + let appended = 0; + const error = outcome(() => { + while (appended < 100) { + params.append("k", value); + appended++; + } + }); + expect({ + error, + appended: appended > 10 && appended < 20, + size: params.size, + hrefLength: url.href.length, + lastValueLength: params.getAll("k").at(-1)!.length, + }).toEqual({ + error: outOfMemory, + appended: true, + size: 1 + appended, + hrefLength: "http://a/?x=1#f".length + appended * ("&k=".length + 6 * value.length), + lastValueLength: value.length, + }); + + // delete() and sort() still work on a URL this long. + params.delete("k"); + params.append("a", "2"); + params.sort(); + expect(url.href).toBe("http://a/?a=2&x=1#f"); + + // A query that the URL keeps as it is can be 3 times as long once the params serialize it. Then no change fits, + // not even one that removes a pair. + const parentheses = new URL("http://a/?z=1&" + repeated(400_000, "(") + "&x=1"); + expect({ + append: outcome(() => parentheses.searchParams.append("a", "b")), + delete: outcome(() => parentheses.searchParams.delete("x")), + sort: outcome(() => parentheses.searchParams.sort()), + keys: [...parentheses.searchParams.keys()].map(key => (key.length > 1 ? key.length : key)), + hrefLength: parentheses.href.length, + hrefEnd: parentheses.href.slice(-5), + }).toEqual({ + append: outOfMemory, + delete: outOfMemory, + sort: outOfMemory, + keys: ["z", 400_000, "x"], + hrefLength: "http://a/?z=1&&x=1".length + 400_000, + hrefEnd: "(&x=1", + }); + + // A change that throws does not make the URL serialize the pairs it had. "(" would become "%28=". + const parenthesis = new URL("http://a/?("); + expect({ + append: outcome(() => parenthesis.searchParams.append("a", tooLong)), + href: parenthesis.href, + }).toEqual({ append: outOfMemory, href: "http://a/?(" }); + }); + }); + + // The cases that only the real limit reaches: the parser's buffer past the size where a doubling Vector gives up, the + // concatenation in a setter, and the UTF-8 copy a setter makes of a string of 2 ** 30 characters. Each child commits + // 4 to 8 GB, and a debug or ASAN build needs minutes for them. + const memory = Math.min(os.totalmem(), process.constrainedMemory() || Infinity); + describe.skipIf(isDebug || isASAN || memory < 16 * 1024 ** 3)("at the real limit", () => { + async function runChild(source: string) { + await using proc = Bun.spawn({ + cmd: [bunExe(), "-e", source], + env: bunEnv, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([proc.stdout.text(), proc.stderr.text(), proc.exited]); + return { stdout, stderr, exitCode, signalCode: proc.signalCode }; + } + // The ASCII strings come from repeat(): it needs half the memory of a Buffer, and it is fast in a release build. + const prelude = ` + const outcome = fn => { try { return fn()?.length; } catch (e) { return e.name + ": " + e.message; } }; + const latin1 = n => Buffer.alloc(n, 0xe9).toString("latin1"); + `; + const printed = (value: unknown) => ({ + stdout: JSON.stringify(value) + "\n", + stderr: "", + exitCode: 0, + signalCode: null, + }); + + test("the constructor throws a RangeError", async () => { + const result = await runChild(`${prelude} + console.log(JSON.stringify({ constructor: outcome(() => new URL("http://a/?" + latin1(2 ** 29)).href) })); + `); + expect(result).toEqual(printed({ constructor: "RangeError: Out of memory" })); + }, 120_000); + + test("a URL of nearly 2 ** 31 characters parses", async () => { + const result = await runChild(`${prelude} + // 6 * 357_000_000 + 10 is 2_142_000_010 characters. + console.log(JSON.stringify({ fits: outcome(() => new URL("http://a/?" + latin1(357_000_000)).href) })); + `); + expect(result).toEqual(printed({ fits: 2_142_000_010 })); + }, 120_000); + + test("setters whose value is too long before any encoding", async () => { + const result = await runChild(`${prelude} + const url = new URL("http://a/p"); + console.log(JSON.stringify({ + pathnameOf2To30: outcome(() => void (url.pathname = latin1(2 ** 30))), + searchOfNumberSigns: outcome(() => void (url.search = "#".repeat(2 ** 30))), + searchNearTheLimit: outcome(() => void (url.search = "a".repeat(2 ** 31 - 5))), + href: url.href, + })); + `); + expect(result).toEqual( + printed({ + pathnameOf2To30: "RangeError: Out of memory", + searchOfNumberSigns: "RangeError: Out of memory", + searchNearTheLimit: "RangeError: Out of memory", + href: "http://a/p", + }), + ); + }, 120_000); + }); +}); diff --git a/test/js/web/urlpattern/urlpattern.test.ts b/test/js/web/urlpattern/urlpattern.test.ts index c2a727ab5b15..a8cbe43aafe2 100644 --- a/test/js/web/urlpattern/urlpattern.test.ts +++ b/test/js/web/urlpattern/urlpattern.test.ts @@ -1,5 +1,6 @@ // Test data from Web Platform Tests // https://github.com/web-platform-tests/wpt/blob/master/LICENSE.md +import { setSyntheticAllocationLimitForTesting } from "bun:internal-for-testing"; import { describe, expect, test } from "bun:test"; import testData from "./urlpatterntestdata.json"; @@ -206,4 +207,24 @@ describe("URLPattern", () => { expect(new URLPattern({ pathname: "/a/:foo/:baz([a-z]+)?/b/*" }).hasRegExpGroups).toBe(true); }); }); + + // test() and exec() percent-encode each component of their input through a URL. That URL can be longer than a string + // can be (2 ** 31 - 1 characters). The component then read as empty, so it matched an empty pattern. + describe("an input component that does not fit in a string when it is percent-encoded", () => { + // U+00E9 is one character and "%C3%A9" is six: a little over 1 Mi characters. 1 MiB stands in for 2 ** 31 - 1. + const tooLong = Buffer.alloc(176_000, 0xe9).toString("latin1"); + + test.each(["pathname", "search", "hash", "username", "password"] as const)("%s does not match", component => { + const previous = setSyntheticAllocationLimitForTesting(1024 * 1024); + try { + expect({ + empty: new URLPattern({ [component]: "" }).test({ [component]: tooLong }), + wildcard: new URLPattern({ [component]: "*" }).exec({ [component]: tooLong }), + inAURL: new URLPattern({ [component]: "" }).test("https://u:p@example.com/p?q#h"), + }).toEqual({ empty: false, wildcard: null, inAURL: false }); + } finally { + setSyntheticAllocationLimitForTesting(previous); + } + }); + }); });