#pragma once #include #include #include #include namespace workerd::jsg { // A WHATWG-compliant URL implementation provided by ada-url. class Url final { public: // Keep in sync with ada::scheme:type enum class SchemeType { HTTP = 0, NOT_SPECIAL = 1, HTTPS = 2, WS = 3, FTP = 4, WSS = 5, FILE = 6 }; // Keep in sync with ada::url_host_type enum class HostType { DEFAULT = 0, IPV4 = 1, IPV6 = 2, }; Url(decltype(nullptr)) {} Url(Url&& other) = default; KJ_DISALLOW_COPY(Url); Url& operator=(Url&& other) = default; bool operator==(const Url& other) const KJ_WARN_UNUSED_RESULT; enum class EquivalenceOption { DEFAULT = 0, // When set, the fragment/hash portion of the URL will be ignored when comparing or // cloning URLs. IGNORE_FRAGMENTS = 1 << 0, // When set, the search portion of the URL will be ignored when comparing or cloning URLs. IGNORE_SEARCH = 1 << 1, // When set, the pathname portion of the URL will be normalized by percent-decoding // then re-encoding the pathname. This is useful when comparing URLs that may have // different, but equivalent percent-encoded paths. e.g. %66oo and foo are equivalent. NORMALIZE_PATH = 1 << 2, }; bool equal(const Url& other, EquivalenceOption option = EquivalenceOption::DEFAULT) const KJ_WARN_UNUSED_RESULT; // Returns true if the given input can be successfully parsed as a URL. This is generally // more performant than using tryParse and checking for a kj::none result if all you want // to do is verify that the input is parseable. If you actually want to parse and use the // result, use tryParse instead. static bool canParse(kj::ArrayPtr input, kj::Maybe> base = kj::none) KJ_WARN_UNUSED_RESULT; static bool canParse( kj::StringPtr input, kj::Maybe base = kj::none) KJ_WARN_UNUSED_RESULT; static kj::Maybe tryParse(kj::ArrayPtr input, kj::Maybe> base = kj::none) KJ_WARN_UNUSED_RESULT; static kj::Maybe tryParse( kj::StringPtr input, kj::Maybe base = kj::none) KJ_WARN_UNUSED_RESULT; kj::Array getOrigin() const KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getProtocol() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getHref() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getPathname() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getUsername() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getPassword() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getPort() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getHash() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getHost() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getHostname() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::ArrayPtr getSearch() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; bool setHref(kj::ArrayPtr value); bool setHost(kj::ArrayPtr value); bool setHostname(kj::ArrayPtr value); bool setProtocol(kj::ArrayPtr value); bool setUsername(kj::ArrayPtr value); bool setPassword(kj::ArrayPtr value); bool setPort(kj::Maybe> value); bool setPathname(kj::ArrayPtr value); void setSearch(kj::Maybe> value); void setHash(kj::Maybe> value); kj::uint hashCode() const; kj::Maybe resolve(kj::ArrayPtr input) KJ_WARN_UNUSED_RESULT; // Copies this Url. If the option is set of EquivalenceOption::IGNORE_FRAGMENTS, the // copied Url will clear any fragment/hash that exists. Url clone(EquivalenceOption option = EquivalenceOption::DEFAULT) const KJ_WARN_UNUSED_RESULT; // Resolve the input relative to this URL kj::Maybe tryResolve(kj::ArrayPtr input) const KJ_WARN_UNUSED_RESULT; enum class RelativeOption { DEFAULT, // If the URL ends with a trailing slash, remove it before determining the basename. STRIP_TAILING_SLASHES, }; struct Relative; // Given this URL, returns a struct that is a basename and a base Url pair // such that base.tryResolve(basename) is equivalent to this URL. Query // parameters and fragments are not preserved. Relative getRelative(RelativeOption option = RelativeOption::DEFAULT) const; // Returns the parent URL of this URL, which is the URL with the last path component // removed and the trailing slash removed if it exists. For instance, if the URL is // "https://example.com/foo/bar/baz", the parent URL will be "https://example.com/foo/bar". // If the URL has no parent (e.g. "https://example.com/") kj::none is returned. kj::Maybe getParent() const KJ_WARN_UNUSED_RESULT; HostType getHostType() const; SchemeType getSchemeType() const; // Convert an ASCII hostname to Unicode. static kj::Array idnToUnicode(kj::ArrayPtr value) KJ_WARN_UNUSED_RESULT; // Convert a Unicode hostname to ASCII. static kj::Array idnToAscii(kj::ArrayPtr value) KJ_WARN_UNUSED_RESULT; static bool isSpecialScheme(kj::StringPtr protocol); static bool isSpecialSchemeDefaultPort(kj::StringPtr protocol, kj::StringPtr port); JSG_MEMORY_INFO(Url) { tracker.trackFieldWithSize("inner", getProtocol().size() + getUsername().size() + getPassword().size() + getHost().size() + getPathname().size() + getHash().size() + getSearch().size()); } static kj::Array percentDecode(kj::ArrayPtr input); private: Url(kj::Own inner); kj::Own inner; }; struct Url::Relative { Url base; kj::String name; }; constexpr Url::EquivalenceOption operator|(Url::EquivalenceOption a, Url::EquivalenceOption b) { return static_cast(static_cast(a) | static_cast(b)); } constexpr Url::EquivalenceOption operator&(Url::EquivalenceOption a, Url::EquivalenceOption b) { return static_cast(static_cast(a) & static_cast(b)); } class UrlSearchParams final { public: class KeyIterator final { public: bool hasNext() const; kj::Maybe> next() const; private: KeyIterator(kj::Own inner); kj::Own inner; friend class UrlSearchParams; }; class ValueIterator final { public: bool hasNext() const; kj::Maybe> next() const; private: ValueIterator(kj::Own inner); kj::Own inner; friend class UrlSearchParams; }; class EntryIterator final { public: struct Entry { kj::ArrayPtr key; kj::ArrayPtr value; }; bool hasNext() const; kj::Maybe next() const; private: EntryIterator(kj::Own inner); kj::Own inner; friend class UrlSearchParams; }; UrlSearchParams(); UrlSearchParams(UrlSearchParams&& other) = default; KJ_DISALLOW_COPY(UrlSearchParams); UrlSearchParams& operator=(UrlSearchParams&& other) = default; bool operator==(const UrlSearchParams& other) const KJ_WARN_UNUSED_RESULT; static kj::Maybe tryParse(kj::ArrayPtr input) KJ_WARN_UNUSED_RESULT; size_t size() const; void append(kj::ArrayPtr key, kj::ArrayPtr value); void set(kj::ArrayPtr key, kj::ArrayPtr value); void delete_( kj::ArrayPtr key, kj::Maybe> maybeValue = kj::none); bool has(kj::ArrayPtr key, kj::Maybe> maybeValue = kj::none) const KJ_WARN_UNUSED_RESULT; kj::Maybe> get( kj::ArrayPtr key) const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::Array> getAll( kj::ArrayPtr key) const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; void sort(); KeyIterator getKeys() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; ValueIterator getValues() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; EntryIterator getEntries() const KJ_LIFETIMEBOUND KJ_WARN_UNUSED_RESULT; kj::Array toStr() const KJ_WARN_UNUSED_RESULT; JSG_MEMORY_INFO(Url) { tracker.trackField("inner", toStr()); } void reset(kj::Maybe> input); private: UrlSearchParams(kj::Own inner); kj::Own inner; }; inline kj::String KJ_STRINGIFY(const Url& url) { return kj::str(url.getHref()); } inline kj::String KJ_STRINGIFY(const UrlSearchParams& searchParams) { return kj::str(searchParams.toStr()); } // ====================================================================================== // Encapsulates a parsed URLPattern. // @see https://wicg.github.io/urlpattern class UrlPattern final { public: // If the value is T, the operation is successful. // If the value is kj::String, that's an Error message. template using Result = kj::OneOf; // An individual, compiled component of a URLPattern. class Component final { public: Component(kj::String pattern, kj::String regex, kj::Array names); Component(Component&&) = default; Component& operator=(Component&&) = default; KJ_DISALLOW_COPY(Component); inline kj::StringPtr getPattern() const KJ_LIFETIMEBOUND { return pattern; } inline kj::StringPtr getRegex() const KJ_LIFETIMEBOUND { return regex; } inline kj::ArrayPtr getNames() const KJ_LIFETIMEBOUND { return names.asPtr(); } JSG_MEMORY_INFO(Component) { tracker.trackField("pattern", pattern); tracker.trackField("regex", regex); for (const auto& name: names) { tracker.trackField("name", name); } } private: // The normalized pattern for this component. kj::String pattern = nullptr; // The generated JavaScript regular expression for this component. kj::String regex = nullptr; // The list of sub-component names extracted for this component. kj::Array names = nullptr; }; // A structure providing matching patterns for individual components of a URL. struct Init { kj::Maybe protocol; kj::Maybe username; kj::Maybe password; kj::Maybe hostname; kj::Maybe port; kj::Maybe pathname; kj::Maybe search; kj::Maybe hash; kj::Maybe baseUrl; }; struct ProcessInitOptions { enum class Mode { PATTERN, URL, }; Mode mode = Mode::PATTERN; kj::Maybe protocol = kj::none; kj::Maybe username = kj::none; kj::Maybe password = kj::none; kj::Maybe hostname = kj::none; kj::Maybe port = kj::none; kj::Maybe pathname = kj::none; kj::Maybe search = kj::none; kj::Maybe hash = kj::none; }; // Processes the given init according to the specified mode and options. // If a kj::String is returned, then processing failed and the string // is the description to include in the error message (if any). static Result processInit( Init init, kj::Maybe options = kj::none) KJ_WARN_UNUSED_RESULT; struct CompileOptions { // The base URL to use. Only used in the compile(kj::StringPtr, ...) variant. kj::Maybe baseUrl; bool ignoreCase = false; }; static Result tryCompile( kj::StringPtr, kj::Maybe = kj::none) KJ_WARN_UNUSED_RESULT; static Result tryCompile( Init init, kj::Maybe = kj::none) KJ_WARN_UNUSED_RESULT; UrlPattern(UrlPattern&&) = default; UrlPattern& operator=(UrlPattern&&) = default; KJ_DISALLOW_COPY(UrlPattern); inline const Component& getProtocol() const KJ_LIFETIMEBOUND { return protocol; } inline const Component& getUsername() const KJ_LIFETIMEBOUND { return username; } inline const Component& getPassword() const KJ_LIFETIMEBOUND { return password; } inline const Component& getHostname() const KJ_LIFETIMEBOUND { return hostname; } inline const Component& getPort() const KJ_LIFETIMEBOUND { return port; } inline const Component& getPathname() const KJ_LIFETIMEBOUND { return pathname; } inline const Component& getSearch() const KJ_LIFETIMEBOUND { return search; } inline const Component& getHash() const KJ_LIFETIMEBOUND { return hash; } // If ignoreCase is true, the JavaScript regular expression created for each pattern // must use the `vi` flag. Otherwise, they must use the `v` flag. inline bool getIgnoreCase() const { return ignoreCase; } JSG_MEMORY_INFO(UrlPattern) { tracker.trackField("protocol", protocol); tracker.trackField("username", username); tracker.trackField("password", password); tracker.trackField("hostname", hostname); tracker.trackField("port", port); tracker.trackField("pathname", pathname); tracker.trackField("search", search); tracker.trackField("hash", hash); } private: UrlPattern(kj::Array components, bool ignoreCase); Component protocol; Component username; Component password; Component hostname; Component port; Component pathname; Component search; Component hash; bool ignoreCase; static Result tryCompileInit(UrlPattern::Init init, const CompileOptions& options); }; } // namespace workerd::jsg // Append _url to a string literal to create a parsed URL. An assert will be triggered // if the value cannot be parsed successfully. const workerd::jsg::Url operator""_url(const char* str, size_t size) KJ_WARN_UNUSED_RESULT;