File
Blob: src/workerd/api/url.c++
| 1 | // Copyright (c) 2017-2022 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | |
| 5 | #include "url.h" |
| 6 | |
| 7 | #include "util.h" |
| 8 | |
| 9 | #include <kj/encoding.h> |
| 10 | #include <kj/parse/char.h> |
| 11 | #include <kj/string-tree.h> |
| 12 | |
| 13 | #include <algorithm> |
| 14 | #include <map> |
| 15 | #include <set> |
| 16 | |
| 17 | namespace workerd::api { |
| 18 | |
| 19 | namespace { |
| 20 | |
| 21 | // Helper functions for the origin, pathname, and search getters and setters. |
| 22 | |
| 23 | // The folowing two lists needs to be kept in sync since the length and the order |
| 24 | // of them are needed to properly calculate the hash/index. |
| 25 | constexpr kj::StringPtr is_special_list[] = { |
| 26 | "http"_kj, " "_kj, "https"_kj, "ws"_kj, "ftp"_kj, "wss"_kj, "file"_kj, " "_kj}; |
| 27 | constexpr kj::StringPtr special_ports[] = { |
| 28 | "80"_kj, ""_kj, "443"_kj, "80"_kj, "21"_kj, "443"_kj, ""_kj, ""_kj}; |
| 29 | |
| 30 | // Taken from Ada URL library. |
| 31 | // Ref: https://github.com/ada-url/ada/blob/b431670699cf4f3ebb2e2c394c23a89850bb6f3f/include/ada/scheme-inl.h#L49 |
| 32 | bool isSpecialScheme(kj::StringPtr scheme) noexcept { |
| 33 | if (scheme.size() == 0) { |
| 34 | return false; |
| 35 | } |
| 36 | // Depending on the first character and the size of the input, this line calculates |
| 37 | // the index from the list above. |
| 38 | // |
| 39 | // Generate a simple hash value that will always be between 0 and 7 (inclusive), |
| 40 | // regardless of the input. This is because the bitwise AND with 7 ensures that |
| 41 | // only the last 3 bits of the result are kept. |
| 42 | int hash_value = (2 * scheme.size() + static_cast<unsigned>(scheme[0])) & 7; |
| 43 | const auto target = is_special_list[hash_value]; |
| 44 | return (target[0] == scheme[0]) && (target.slice(1) == scheme.slice(1)); |
| 45 | } |
| 46 | |
| 47 | // Taken from Ada URL library. |
| 48 | // Ref: https://github.com/ada-url/ada/blob/b431670699cf4f3ebb2e2c394c23a89850bb6f3f/include/ada/scheme-inl.h#L57 |
| 49 | kj::Maybe<kj::StringPtr> defaultPortForScheme(kj::StringPtr scheme) noexcept { |
| 50 | if (scheme.size() == 0) { |
| 51 | return kj::none; |
| 52 | } |
| 53 | int hash_value = (2 * scheme.size() + static_cast<unsigned>(scheme[0])) & 7; |
| 54 | const auto target = is_special_list[hash_value]; |
| 55 | if ((target[0] == scheme[0]) && (target.slice(1) == scheme.slice(1))) { |
| 56 | auto port = special_ports[hash_value]; |
| 57 | if (port.size() == 0) { |
| 58 | return kj::none; |
| 59 | } |
| 60 | return port; |
| 61 | } |
| 62 | |
| 63 | return kj::none; |
| 64 | } |
| 65 | |
| 66 | void normalizePort(kj::Url& url) { |
| 67 | // Remove trailing ':', and remove ':xxx' if xxx is the scheme-default port. |
| 68 | |
| 69 | KJ_IF_SOME(colon, url.host.findFirst(':')) { |
| 70 | if (url.host.size() == colon + 1) { |
| 71 | // Remove trailing ':'. |
| 72 | url.host = kj::str(url.host.first(colon)); |
| 73 | } else KJ_IF_SOME(defaultPort, defaultPortForScheme(url.scheme)) { |
| 74 | if (defaultPort == url.host.slice(colon + 1)) { |
| 75 | // Remove scheme-default port. |
| 76 | url.host = kj::str(url.host.first(colon)); |
| 77 | } |
| 78 | } |
| 79 | } |
| 80 | } |
| 81 | |
| 82 | kj::Maybe<kj::ArrayPtr<const char>> trySplit(kj::ArrayPtr<const char>& text, char c) { |
| 83 | // TODO(cleanup): Code duplication with kj/compat/url.c++. |
| 84 | |
| 85 | for (auto i: kj::indices(text)) { |
| 86 | if (text[i] == c) { |
| 87 | kj::ArrayPtr<const char> result = text.first(i); |
| 88 | text = text.slice(i + 1, text.size()); |
| 89 | return result; |
| 90 | } |
| 91 | } |
| 92 | return kj::none; |
| 93 | } |
| 94 | |
| 95 | kj::ArrayPtr<const char> split(kj::StringPtr& text, const kj::parse::CharGroup_& chars) { |
| 96 | // TODO(cleanup): Code duplication with kj/compat/url.c++. |
| 97 | |
| 98 | for (auto i: kj::indices(text)) { |
| 99 | if (chars.contains(text[i])) { |
| 100 | kj::ArrayPtr<const char> result = text.first(i); |
| 101 | text = text.slice(i); |
| 102 | return result; |
| 103 | } |
| 104 | } |
| 105 | auto result = text.asArray(); |
| 106 | text = ""; |
| 107 | return result; |
| 108 | } |
| 109 | |
| 110 | kj::String percentDecode(kj::ArrayPtr<const char> text, bool& hadErrors) { |
| 111 | // TODO(cleanup): Code duplication with kj/compat/url.c++. |
| 112 | |
| 113 | auto result = kj::decodeUriComponent(text); |
| 114 | if (result.hadErrors) hadErrors = true; |
| 115 | return kj::mv(result); |
| 116 | } |
| 117 | |
| 118 | kj::String percentDecodeQuery(kj::ArrayPtr<const char> text, bool& hadErrors) { |
| 119 | // TODO(cleanup): Code duplication with kj/compat/url.c++. |
| 120 | |
| 121 | auto result = kj::decodeWwwForm(text); |
| 122 | if (result.hadErrors) hadErrors = true; |
| 123 | return kj::mv(result); |
| 124 | } |
| 125 | |
| 126 | // Use this instead of calling kj::Url::toString() directly. |
| 127 | kj::String kjUrlToString(const kj::Url& url) { |
| 128 | kj::String result; |
| 129 | KJ_IF_SOME(exception, kj::runCatchingExceptions([&]() { |
| 130 | result = url.toString(); |
| 131 | // TODO(soon): This stringifier does not append trailing slashes to the pathname conformantly. |
| 132 | // For example, this equality currently does not hold true: |
| 133 | // |
| 134 | // new URL('https://capnproto.org?query').href === 'https://capnproto.org/?query' |
| 135 | // |
| 136 | // Fixing this bug would enable a plurality of the W3C test cases which currently fail. I.e., |
| 137 | // it's the lowest hanging fruit. ;) |
| 138 | })) { |
| 139 | // TODO(conform): toString() really shouldn't be throwing anything, because it shouldn't be |
| 140 | // possible to get the URL object in a state where it has any invalid component. However, a |
| 141 | // variety of bugs conspire to make it possible (notably, EW-962 and EW-1731), and we're stuck |
| 142 | // with the situation for now. Rather than expose these errors to the user as opaque internal |
| 143 | // errors (and nag us via Sentry), we get our hands dirty with some string matching, in the |
| 144 | // hopes of helping users work around the bugs. |
| 145 | KJ_IF_SOME(e, |
| 146 | translateKjException(exception, |
| 147 | { |
| 148 | {"invalid hostname when stringifying URL"_kj, |
| 149 | "Invalid hostname when stringifying URL."_kj}, |
| 150 | {"invalid name in URL path"_kj, "Invalid pathname when stringifying URL."_kj}, |
| 151 | })) { |
| 152 | kj::throwFatalException(kj::mv(e)); |
| 153 | } |
| 154 | |
| 155 | // This is either an error we should know about and expect, or an "internal error". Either way, |
| 156 | // squawk about it. |
| 157 | KJ_LOG(ERROR, exception); |
| 158 | JSG_FAIL_REQUIRE(TypeError, "Error stringifying URL."); |
| 159 | } |
| 160 | |
| 161 | return kj::mv(result); |
| 162 | } |
| 163 | |
| 164 | } // namespace |
| 165 | |
| 166 | // ======================================================================================= |
| 167 | // URL |
| 168 | |
| 169 | jsg::Ref<URL> URL::constructor(jsg::Lock& js, kj::String url, jsg::Optional<kj::String> base) { |
| 170 | KJ_IF_SOME(b, base) { |
| 171 | auto baseUrl = |
| 172 | JSG_REQUIRE_NONNULL(kj::Url::tryParse(kj::mv(b)), TypeError, "Invalid base URL string."); |
| 173 | return js.alloc<URL>(JSG_REQUIRE_NONNULL( |
| 174 | baseUrl.tryParseRelative(kj::mv(url)), TypeError, "Invalid relative URL string.")); |
| 175 | } |
| 176 | return js.alloc<URL>( |
| 177 | JSG_REQUIRE_NONNULL(kj::Url::tryParse(kj::mv(url)), TypeError, "Invalid URL string.")); |
| 178 | } |
| 179 | |
| 180 | URL::URL(kj::Url&& u): url(kj::refcounted<RefcountedUrl>(kj::mv(u))) { |
| 181 | normalizePort(*url); |
| 182 | } |
| 183 | |
| 184 | // Setters and Getters |
| 185 | // |
| 186 | // When possible, getters just pull out the corresponding attribute from kj::Url and return it. |
| 187 | // Sometimes we need to modify the output a bit, e.g. to get the hostname and port separately. |
| 188 | // |
| 189 | // Setters need to set and validate new input. To accomplish this without reimplementing validation |
| 190 | // code that ought to live in kj::Url, I have implemented setters using the following general |
| 191 | // strategy: |
| 192 | // |
| 193 | // 1. Pre-process the input, if necessary. E.g., we drop anything after a ':' when setting protocol. |
| 194 | // 2. Clone the kj::Url object. |
| 195 | // 3. Replace the cloned component in question with the new value. |
| 196 | // 4. Stringify and parse the clone. If this succeeds, the clone is the new kj::Url object. |
| 197 | // |
| 198 | // Notably, we do little to no validation in this wrapper class. As validation checks are added to |
| 199 | // kj::Url's parser, more and more unit tests for this wrapper class should start passing without |
| 200 | // modification. |
| 201 | // |
| 202 | // TODO(perf): Pre-processing input, cloning, stringifying, and parsing the cloned URL is an awfully |
| 203 | // heavyweight operation when all we want to do is validly replace a URL component. A couple |
| 204 | // attributes, pathname and search, are able to take advantage of the kj::Url parser's context |
| 205 | // argument: we can parse a pathname using the HTTP_REQUEST context, for instance. The WHATWG URL |
| 206 | // spec defines a parser state machine allowing for the state to be overridden to parse only |
| 207 | // specific components of a URL. This is more or less a generalization of kj::Url's parser |
| 208 | // context, and offers an obvious path forward to both conformance and performance. |
| 209 | |
| 210 | kj::String URL::getHref() { |
| 211 | return toString(); |
| 212 | } |
| 213 | void URL::setHref(jsg::Lock& js, kj::String value) { |
| 214 | KJ_IF_SOME(u, kj::Url::tryParse(kj::mv(value))) { |
| 215 | url->kj::Url::operator=(kj::mv(u)); |
| 216 | } else { |
| 217 | auto context = jsg::TypeErrorContext::setterArgument(typeid(URL), "href"); |
| 218 | jsg::throwTypeError(js.v8Isolate, context, "valid URL string"); |
| 219 | // href's is the only setter which is allowed to throw on invalid input, according to the spec. |
| 220 | } |
| 221 | } |
| 222 | |
| 223 | kj::String URL::getOrigin() { |
| 224 | // TODO(cleanup): Move this logic into kj::Url. |
| 225 | |
| 226 | if (isSpecialScheme(url->scheme) && url->scheme != "file") { |
| 227 | return kj::str(url->scheme, "://", url->host); |
| 228 | } else if (url->scheme == "file") { |
| 229 | return kj::str("null"); |
| 230 | } else if (url->scheme == "blob") { |
| 231 | // TODO(soon): Parse url->path[0] and return that if it has an origin. |
| 232 | return kj::str("null"); |
| 233 | } |
| 234 | return kj::str("null"); |
| 235 | } |
| 236 | |
| 237 | kj::String URL::getProtocol() { |
| 238 | return kj::str(url->scheme, ':'); |
| 239 | } |
| 240 | void URL::setProtocol(kj::String value) { |
| 241 | KJ_IF_SOME(colon, value.findFirst(':')) { |
| 242 | value = kj::str(value.first(colon)); |
| 243 | } |
| 244 | |
| 245 | auto copy = url->clone(); |
| 246 | copy.scheme = kj::mv(value); |
| 247 | |
| 248 | KJ_IF_SOME(u, kj::Url::tryParse(kjUrlToString(copy))) { |
| 249 | url->kj::Url::operator=(kj::mv(u)); |
| 250 | } |
| 251 | |
| 252 | normalizePort(*url); |
| 253 | } |
| 254 | |
| 255 | kj::String URL::getUsername() { |
| 256 | KJ_IF_SOME(userInfo, url->userInfo) { |
| 257 | return kj::encodeUriUserInfo(userInfo.username); |
| 258 | } |
| 259 | return {}; |
| 260 | } |
| 261 | void URL::setUsername(kj::String value) { |
| 262 | auto copy = url->clone(); |
| 263 | KJ_IF_SOME(ui, copy.userInfo) { |
| 264 | ui.username = kj::mv(value); |
| 265 | } else { |
| 266 | copy.userInfo = kj::Url::UserInfo{.username = kj::mv(value)}; |
| 267 | } |
| 268 | |
| 269 | KJ_IF_SOME(u, kj::Url::tryParse(kjUrlToString(copy))) { |
| 270 | url->kj::Url::operator=(kj::mv(u)); |
| 271 | } |
| 272 | } |
| 273 | |
| 274 | kj::String URL::getPassword() { |
| 275 | KJ_IF_SOME(userInfo, url->userInfo) { |
| 276 | KJ_IF_SOME(password, userInfo.password) { |
| 277 | return kj::encodeUriUserInfo(password); |
| 278 | } |
| 279 | } |
| 280 | return {}; |
| 281 | } |
| 282 | void URL::setPassword(kj::String value) { |
| 283 | auto copy = url->clone(); |
| 284 | KJ_IF_SOME(ui, copy.userInfo) { |
| 285 | // We already have userInfo. Set the password if we were given a non-empty string, otherwise |
| 286 | // reset the password Maybe. |
| 287 | if (value.size() > 0) { |
| 288 | ui.password = kj::mv(value); |
| 289 | } else { |
| 290 | ui.password = kj::none; |
| 291 | } |
| 292 | } else if (value.size() > 0) { |
| 293 | copy.userInfo = kj::Url::UserInfo{.password = kj::mv(value)}; |
| 294 | } |
| 295 | |
| 296 | KJ_IF_SOME(u, kj::Url::tryParse(kjUrlToString(copy))) { |
| 297 | url->kj::Url::operator=(kj::mv(u)); |
| 298 | } |
| 299 | } |
| 300 | |
| 301 | kj::String URL::getHost() { |
| 302 | return kj::str(url->host); |
| 303 | } |
| 304 | void URL::setHost(kj::String value) { |
| 305 | // The spec provides the following helpful note: |
| 306 | // |
| 307 | // If the given value for the host attribute’s setter lacks a port, context object’s url’s port |
| 308 | // will not change. This can be unexpected as host attribute’s getter does return a URL-port |
| 309 | // string so one might have assumed the setter to always "reset" both. |
| 310 | |
| 311 | // If the new host value lacks a port, copy the current one over to the new value, if any. We can |
| 312 | // assume that if the current one has a port, it must not be the default port for this URL's |
| 313 | // scheme. |
| 314 | KJ_IF_SOME(colon, url->host.findFirst(':')) { |
| 315 | KJ_IF_SOME(newHostColon, value.findFirst(':')) { |
| 316 | if (value.size() == newHostColon + 1) { |
| 317 | // The new host has a colon but nothing after it. Adopt the current port. |
| 318 | value = kj::str(kj::mv(value), url->host.slice(colon + 1)); |
| 319 | } else { |
| 320 | // The new host has a port, so we don't copy the current one over. |
| 321 | } |
| 322 | } else { |
| 323 | // The new host has no port. Adopt the current port. |
| 324 | value = kj::str(kj::mv(value), url->host.slice(colon)); |
| 325 | } |
| 326 | } |
| 327 | |
| 328 | // TODO(soon): Validate the new host string. kj::Url::isValidHost(value)? |
| 329 | url->host = kj::mv(value); |
| 330 | |
| 331 | normalizePort(*url); |
| 332 | } |
| 333 | |
| 334 | kj::String URL::getHostname() { |
| 335 | KJ_IF_SOME(colon, url->host.findFirst(':')) { |
| 336 | return kj::str(url->host.first(colon)); |
| 337 | } |
| 338 | return kj::str(url->host); |
| 339 | } |
| 340 | void URL::setHostname(kj::String value) { |
| 341 | // In contrast to the host setter, the hostname setter explicitly ignores any new port. We take |
| 342 | // the hostname from the new value and the port from the old value. |
| 343 | auto hostnameString = value.first(value.findFirst(':').orDefault(value.size())); |
| 344 | auto portString = url->host.slice(url->host.findFirst(':').orDefault(url->host.size())); |
| 345 | |
| 346 | url->host = kj::str(hostnameString, portString); |
| 347 | } |
| 348 | |
| 349 | kj::String URL::getPort() { |
| 350 | KJ_IF_SOME(colon, url->host.findFirst(':')) { |
| 351 | return kj::str(url->host.slice(colon + 1)); |
| 352 | } |
| 353 | return {}; |
| 354 | } |
| 355 | void URL::setPort(kj::String value) { |
| 356 | KJ_IF_SOME(colon, url->host.findFirst(':')) { |
| 357 | // Our url's host already has a port. Replace it. |
| 358 | value = kj::str(url->host.first(colon + 1), kj::mv(value)); |
| 359 | } else { |
| 360 | value = kj::str(url->host, ':', kj::mv(value)); |
| 361 | } |
| 362 | |
| 363 | url->host = kj::mv(value); |
| 364 | |
| 365 | normalizePort(*url); |
| 366 | } |
| 367 | |
| 368 | kj::String URL::getPathname() { |
| 369 | if (!url->path.empty()) { |
| 370 | auto components = |
| 371 | KJ_MAP(component, url->path) { return kj::str('/', kj::encodeUriPath(component)); }; |
| 372 | return kj::str(kj::strArray(components, ""), url->hasTrailingSlash ? "/" : ""); |
| 373 | } else if (url->hasTrailingSlash || isSpecialScheme(url->scheme)) { |
| 374 | // Special URLs have non-empty paths by definition, regardless of the value of hasTrailingSlash. |
| 375 | return kj::str('/'); |
| 376 | } |
| 377 | return {}; |
| 378 | } |
| 379 | void URL::setPathname(kj::String value) { |
| 380 | decltype(url->path) newPath; |
| 381 | bool newHasTrailingSlash; |
| 382 | |
| 383 | auto text = value.slice(0); |
| 384 | bool err = false; |
| 385 | |
| 386 | // TODO(cleanup): Code duplication with kj/compat/url.c++. |
| 387 | |
| 388 | auto addPart = [&]() { |
| 389 | // We only look for / to end path components in this setter, not ? and # like in |
| 390 | // kj::Url::tryParse(). |
| 391 | constexpr auto END_PATH_PART = kj::parse::anyOfChars("/"); |
| 392 | auto part = split(text, END_PATH_PART); |
| 393 | if (part.size() == 2 && part[0] == '.' && part[1] == '.') { |
| 394 | if (!newPath.empty()) { |
| 395 | newPath.removeLast(); |
| 396 | } |
| 397 | newHasTrailingSlash = true; |
| 398 | } else if (part.size() == 0 || (part.size() == 1 && part[0] == '.')) { |
| 399 | // Collapse consecutive slashes and "/./". |
| 400 | newHasTrailingSlash = true; |
| 401 | } else { |
| 402 | newPath.add(percentDecode(part, err)); |
| 403 | newHasTrailingSlash = false; |
| 404 | } |
| 405 | }; |
| 406 | |
| 407 | // Unlike kj::Url::tryParse(), the pathname being set doesn't have to begin with a slash. |
| 408 | if (!text.startsWith("/")) { |
| 409 | addPart(); |
| 410 | } |
| 411 | |
| 412 | while (text.startsWith("/")) { |
| 413 | text = text.slice(1); |
| 414 | addPart(); |
| 415 | } |
| 416 | |
| 417 | if (!err) { |
| 418 | url->hasTrailingSlash = newHasTrailingSlash; |
| 419 | url->path = newPath.releaseAsArray(); |
| 420 | } |
| 421 | } |
| 422 | |
| 423 | kj::String URL::getSearch() { |
| 424 | auto query = KJ_MAP(q, url->query) { |
| 425 | // TODO(soon): We shouldn't be performing any encoding here, because our setSearch() (and URL |
| 426 | // constructor) shouldn't be performing application/x-www-form-urlencoded decoding on the |
| 427 | // query string themselves -- that's for URLSearchParams to do. |
| 428 | if (q.value.begin() != nullptr) { |
| 429 | return kj::str(kj::encodeWwwForm(q.name), '=', kj::encodeWwwForm(q.value)); |
| 430 | } |
| 431 | return kj::str(kj::encodeWwwForm(q.name)); |
| 432 | }; |
| 433 | |
| 434 | if (query.size() > 0) { |
| 435 | return kj::str('?', kj::strArray(query, "&")); |
| 436 | } |
| 437 | return {}; |
| 438 | } |
| 439 | void URL::setSearch(kj::String value) { |
| 440 | decltype(url->query) newQuery; |
| 441 | |
| 442 | auto text = value.slice(value.startsWith("?") ? 1 : 0); |
| 443 | bool err = false; |
| 444 | |
| 445 | // TODO(cleanup): Code duplication with kj/compat/url.c++. |
| 446 | |
| 447 | for (;;) { |
| 448 | // We only look for & to end path components in this setter, not # like in kj::Url::tryParse(). |
| 449 | constexpr auto END_QUERY_PART = kj::parse::anyOfChars("&"); |
| 450 | auto part = split(text, END_QUERY_PART); |
| 451 | |
| 452 | if (part.size() > 0) { |
| 453 | // TODO(soon): We shouldn't be performing any decoding here. Rather, the spec dictates that we |
| 454 | // should actually be percent-*encoding* with a very specific character set. Note that this |
| 455 | // also applies to URL's constructor as well. |
| 456 | // |
| 457 | // See step 1.3.1 of https://url.spec.whatwg.org/#query-state |
| 458 | KJ_IF_SOME(key, trySplit(part, '=')) { |
| 459 | newQuery.add( |
| 460 | kj::Url::QueryParam{percentDecodeQuery(key, err), percentDecodeQuery(part, err)}); |
| 461 | } else { |
| 462 | newQuery.add(kj::Url::QueryParam{percentDecodeQuery(part, err), nullptr}); |
| 463 | } |
| 464 | } |
| 465 | |
| 466 | if (!text.startsWith("&")) break; |
| 467 | text = text.slice(1); |
| 468 | } |
| 469 | |
| 470 | if (!err) { |
| 471 | url->query = newQuery.releaseAsArray(); |
| 472 | } |
| 473 | } |
| 474 | |
| 475 | jsg::Ref<URLSearchParams> URL::getSearchParams(jsg::Lock& js) { |
| 476 | KJ_IF_SOME(usp, searchParams) { |
| 477 | return usp.addRef(); |
| 478 | } else { |
| 479 | searchParams.emplace(js.alloc<URLSearchParams>(kj::addRef(*url))); |
| 480 | return KJ_ASSERT_NONNULL(searchParams).addRef(); |
| 481 | } |
| 482 | } |
| 483 | |
| 484 | kj::String URL::getHash() { |
| 485 | KJ_IF_SOME(fragment, url->fragment) { |
| 486 | if (fragment.size() > 0) { |
| 487 | return kj::str('#', kj::encodeUriFragment(fragment)); |
| 488 | } |
| 489 | } |
| 490 | return {}; |
| 491 | } |
| 492 | void URL::setHash(kj::String value) { |
| 493 | // Omit any starting '#'. |
| 494 | url->fragment = kj::decodeUriComponent(value.slice(value.startsWith("#") ? 1 : 0)); |
| 495 | } |
| 496 | |
| 497 | kj::String URL::toString() { |
| 498 | return kjUrlToString(*url); |
| 499 | } |
| 500 | kj::String URL::toJSON() { |
| 501 | return toString(); |
| 502 | } |
| 503 | |
| 504 | // ======================================================================================= |
| 505 | // URLSearchParams |
| 506 | |
| 507 | URLSearchParams::URLSearchParams(kj::Own<URL::RefcountedUrl> url): url(kj::mv(url)) {} |
| 508 | |
| 509 | jsg::Ref<URLSearchParams> URLSearchParams::constructor( |
| 510 | jsg::Lock& js, jsg::Optional<URLSearchParams::Initializer> init) { |
| 511 | auto searchParams = js.alloc<URLSearchParams>(kj::refcounted<URL::RefcountedUrl>()); |
| 512 | |
| 513 | KJ_IF_SOME(i, init) { |
| 514 | KJ_SWITCH_ONEOF(i) { |
| 515 | KJ_CASE_ONEOF(usp, jsg::Ref<URLSearchParams>) { |
| 516 | searchParams->url->kj::Url::operator=(usp->url->clone()); |
| 517 | } |
| 518 | KJ_CASE_ONEOF(queryString, kj::String) { |
| 519 | parseQueryString(searchParams->url->query, kj::mv(queryString), true); |
| 520 | } |
| 521 | KJ_CASE_ONEOF(dict, jsg::Dict<kj::String>) { |
| 522 | searchParams->url->query = KJ_MAP(entry, dict.fields) { |
| 523 | return kj::Url::QueryParam{kj::mv(entry.name), kj::mv(entry.value)}; |
| 524 | }; |
| 525 | } |
| 526 | KJ_CASE_ONEOF(arrayOfArrays, kj::Array<kj::Array<kj::String>>) { |
| 527 | searchParams->url->query = KJ_MAP(entry, arrayOfArrays) { |
| 528 | JSG_REQUIRE(entry.size() == 2, TypeError, |
| 529 | "To initialize a URLSearchParams object " |
| 530 | "from an array-of-arrays, each inner array must have exactly two elements."); |
| 531 | return kj::Url::QueryParam{kj::mv(entry[0]), kj::mv(entry[1])}; |
| 532 | }; |
| 533 | } |
| 534 | } |
| 535 | } |
| 536 | |
| 537 | return searchParams; |
| 538 | } |
| 539 | |
| 540 | void URLSearchParams::append(kj::String name, kj::String value) { |
| 541 | url->query.add(kj::Url::QueryParam{kj::mv(name), kj::mv(value)}); |
| 542 | } |
| 543 | |
| 544 | void URLSearchParams::delete_(kj::String name) { |
| 545 | auto pivot = std::remove_if( |
| 546 | url->query.begin(), url->query.end(), [&name](const auto& kv) { return kv.name == name; }); |
| 547 | url->query.truncate(pivot - url->query.begin()); |
| 548 | } |
| 549 | |
| 550 | kj::Maybe<kj::String> URLSearchParams::get(kj::String name) { |
| 551 | for (auto& [k, v]: url->query) { |
| 552 | if (k == name) { |
| 553 | return kj::str(v); |
| 554 | } |
| 555 | } |
| 556 | return kj::none; |
| 557 | } |
| 558 | |
| 559 | kj::Array<kj::String> URLSearchParams::getAll(kj::String name) { |
| 560 | kj::Vector<kj::String> result; |
| 561 | for (auto& [k, v]: url->query) { |
| 562 | if (k == name) { |
| 563 | result.add(kj::str(v)); |
| 564 | } |
| 565 | } |
| 566 | return result.releaseAsArray(); |
| 567 | } |
| 568 | |
| 569 | bool URLSearchParams::has(kj::String name) { |
| 570 | for (auto& [k, v]: url->query) { |
| 571 | if (k == name) { |
| 572 | return true; |
| 573 | } |
| 574 | } |
| 575 | return false; |
| 576 | } |
| 577 | |
| 578 | // Set the first element named `name` to `value`, then remove all the rest matching that name. |
| 579 | void URLSearchParams::set(kj::String name, kj::String value) { |
| 580 | const auto predicate = [name = name.slice(0)](const auto& kv) { return kv.name == name; }; |
| 581 | auto firstFound = std::find_if(url->query.begin(), url->query.end(), predicate); |
| 582 | if (firstFound != url->query.end()) { |
| 583 | firstFound->value = kj::mv(value); |
| 584 | auto pivot = std::remove_if(++firstFound, url->query.end(), predicate); |
| 585 | url->query.truncate(pivot - url->query.begin()); |
| 586 | } else { |
| 587 | append(kj::mv(name), kj::mv(value)); |
| 588 | } |
| 589 | } |
| 590 | |
| 591 | // Sort by UTF-16 code unit, preserving order of equal elements. |
| 592 | void URLSearchParams::sort() { |
| 593 | // TODO(perf): This UTF-16 business is sad. The WPT points out the specific example 🌈 < ffi, |
| 594 | // because the rainbow is lexicographically less than the ligature in UTF-16 code units. In |
| 595 | // UTF-8 code units, their order is the opposite. |
| 596 | // |
| 597 | // UTF-8 | UTF-16 |
| 598 | // ffi ef ac 83 | fb03 |
| 599 | // 🌈 f0 9f 8c 88 | d83c df08 |
| 600 | |
| 601 | std::stable_sort(url->query.begin(), url->query.end(), [](const auto& left, const auto& right) { |
| 602 | auto leftUtf16 = fastEncodeUtf16(left.name.asArray()); |
| 603 | auto rightUtf16 = fastEncodeUtf16(right.name.asArray()); |
| 604 | return std::lexicographical_compare( |
| 605 | leftUtf16.begin(), leftUtf16.end(), rightUtf16.begin(), rightUtf16.end()); |
| 606 | }); |
| 607 | } |
| 608 | |
| 609 | void URLSearchParams::forEach(jsg::Lock& js, |
| 610 | jsg::Function<void(kj::StringPtr, kj::StringPtr, jsg::Ref<URLSearchParams>)> callback, |
| 611 | jsg::Optional<jsg::Value> thisArg) { |
| 612 | auto receiver = js.v8Undefined(); |
| 613 | KJ_IF_SOME(arg, thisArg) { |
| 614 | auto handle = arg.getHandle(js); |
| 615 | if (!handle->IsNullOrUndefined()) { |
| 616 | receiver = handle; |
| 617 | } |
| 618 | } |
| 619 | callback.setReceiver(js.v8Ref(receiver)); |
| 620 | |
| 621 | // On each iteration of the for loop, a JavaScript callback is invoked. If a new |
| 622 | // item is appended to the this->url->query within that function, the loop must pick |
| 623 | // it up. Using the classic for (;;) syntax here allows for that. However, this does |
| 624 | // mean that it's possible for a user to trigger an infinite loop here if new items |
| 625 | // are added to the search params unconditionally on each iteration. |
| 626 | // Silence clang-tidy warning, using an iterator would not work correctly if callback |
| 627 | // increases the size of data. |
| 628 | // NOLINTNEXTLINE(modernize-loop-convert) |
| 629 | for (size_t i = 0; i < this->url->query.size(); i++) { |
| 630 | auto& [key, value] = this->url->query[i]; |
| 631 | callback(js, value, key, JSG_THIS); |
| 632 | } |
| 633 | } |
| 634 | |
| 635 | jsg::Ref<URLSearchParams::EntryIterator> URLSearchParams::entries(jsg::Lock& js) { |
| 636 | return js.alloc<EntryIterator>(IteratorState{JSG_THIS}); |
| 637 | } |
| 638 | |
| 639 | jsg::Ref<URLSearchParams::KeyIterator> URLSearchParams::keys(jsg::Lock& js) { |
| 640 | return js.alloc<KeyIterator>(IteratorState{JSG_THIS}); |
| 641 | } |
| 642 | |
| 643 | jsg::Ref<URLSearchParams::ValueIterator> URLSearchParams::values(jsg::Lock& js) { |
| 644 | return js.alloc<ValueIterator>(IteratorState{JSG_THIS}); |
| 645 | } |
| 646 | |
| 647 | kj::String URLSearchParams::toString() { |
| 648 | kj::Vector<char> chars(128); |
| 649 | |
| 650 | bool first = true; |
| 651 | for (auto& param: url->query) { |
| 652 | if (!first) chars.add('&'); |
| 653 | first = false; |
| 654 | chars.addAll(kj::encodeWwwForm(param.name)); |
| 655 | // This *intentionally* differs from the behavior in URL::getSearch() and kj::Url::toString()! |
| 656 | // URLSearchParams has no concept of "null-valued" query parameters -- they get coerced to |
| 657 | // empty-valued query parameters, so we unconditionally add the '=' sign. |
| 658 | chars.add('='); |
| 659 | chars.addAll(kj::encodeWwwForm(param.value)); |
| 660 | } |
| 661 | |
| 662 | chars.add('\0'); |
| 663 | return kj::String(chars.releaseAsArray()); |
| 664 | } |
| 665 | |
| 666 | } // namespace workerd::api |