File
Blob: src/workerd/util/mimetype.c++
| 1 | // Copyright (c) 2023 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | #include "mimetype.h" |
| 5 | |
| 6 | #include "strings.h" |
| 7 | |
| 8 | #include <workerd/util/string-buffer.h> |
| 9 | |
| 10 | #include <kj/debug.h> |
| 11 | |
| 12 | namespace workerd { |
| 13 | |
| 14 | namespace { |
| 15 | |
| 16 | constexpr bool isWhitespace(const char c) noexcept { |
| 17 | return (c == '\r' || c == '\n' || c == '\t' || c == ' '); |
| 18 | } |
| 19 | |
| 20 | static constexpr kj::FixedArray<uint8_t, 256> token_table = []() consteval { |
| 21 | kj::FixedArray<uint8_t, 256> result{}; |
| 22 | |
| 23 | for (uint8_t c: |
| 24 | {'!', '#', '$', '%', '&', '\'', '*', '+', '\\', '-', '.', '^', '_', '`', '|', '~'}) { |
| 25 | result[c] = true; |
| 26 | } |
| 27 | |
| 28 | // (c >= 'A' && c <= 'Z') |
| 29 | for (uint8_t c = 'A'; c <= 'Z'; c++) { |
| 30 | result[c] = true; |
| 31 | } |
| 32 | |
| 33 | // (c >= 'a' && c <= 'z') |
| 34 | for (uint8_t c = 'a'; c <= 'z'; c++) { |
| 35 | result[c] = true; |
| 36 | } |
| 37 | |
| 38 | // (c >= '0' && c <= '9') |
| 39 | for (uint8_t c = '0'; c <= '9'; c++) { |
| 40 | result[c] = true; |
| 41 | } |
| 42 | |
| 43 | return result; |
| 44 | }(); |
| 45 | |
| 46 | constexpr bool isTokenChar(const uint8_t c) noexcept { |
| 47 | return token_table[c]; |
| 48 | } |
| 49 | |
| 50 | static constexpr kj::FixedArray<uint8_t, 256> quoted_string_token_table = []() consteval { |
| 51 | kj::FixedArray<uint8_t, 256> result{}; |
| 52 | result['\t'] = true; |
| 53 | |
| 54 | for (uint8_t c = 0x20; c <= 0x7e; c++) { |
| 55 | result[c] = true; |
| 56 | } |
| 57 | |
| 58 | for (uint8_t c = 0x80; c < 255; c++) { |
| 59 | result[c] = true; |
| 60 | } |
| 61 | |
| 62 | return result; |
| 63 | }(); |
| 64 | |
| 65 | constexpr bool isQuotedStringTokenChar(const uint8_t c) noexcept { |
| 66 | return quoted_string_token_table[c]; |
| 67 | } |
| 68 | |
| 69 | kj::ArrayPtr<const char> skipWhitespace(kj::ArrayPtr<const char> str) { |
| 70 | auto ptr = str.begin(); |
| 71 | auto end = str.end(); |
| 72 | while (ptr != end && isWhitespace(*ptr)) { |
| 73 | ptr++; |
| 74 | } |
| 75 | return str.slice(ptr - str.begin()); |
| 76 | } |
| 77 | |
| 78 | kj::ArrayPtr<const char> trimWhitespace(kj::ArrayPtr<const char> str) { |
| 79 | auto ptr = str.end(); |
| 80 | while (ptr > str.begin() && isWhitespace(*(ptr - 1))) --ptr; |
| 81 | return str.first(ptr - str.begin()); |
| 82 | } |
| 83 | |
| 84 | constexpr bool hasInvalidCodepoints(kj::ArrayPtr<const char> str, auto predicate) { |
| 85 | bool has_invalid_codepoints = false; |
| 86 | for (const char c: str) { |
| 87 | has_invalid_codepoints |= !predicate(static_cast<uint8_t>(c)); |
| 88 | } |
| 89 | return has_invalid_codepoints; |
| 90 | } |
| 91 | |
| 92 | kj::Maybe<size_t> findParamDelimiter(kj::ArrayPtr<const char> str) { |
| 93 | auto ptr = str.begin(); |
| 94 | while (ptr != str.end()) { |
| 95 | if (*ptr == ';' || *ptr == '=') return ptr - str.begin(); |
| 96 | ++ptr; |
| 97 | } |
| 98 | return kj::none; |
| 99 | } |
| 100 | |
| 101 | } // namespace |
| 102 | |
| 103 | MimeType MimeType::parse(kj::StringPtr input, ParseOptions options) { |
| 104 | return KJ_ASSERT_NONNULL(tryParse(input, options)); |
| 105 | } |
| 106 | |
| 107 | kj::Maybe<MimeType> MimeType::tryParse(kj::ArrayPtr<const char> input, ParseOptions options) { |
| 108 | // Skip leading whitespace from start |
| 109 | input = skipWhitespace(input); |
| 110 | if (input.size() == 0) return kj::none; |
| 111 | |
| 112 | kj::Maybe<kj::String> maybeType; |
| 113 | // Let's try to find the solidus that separates the type and subtype |
| 114 | KJ_IF_SOME(n, input.findFirst('/')) { |
| 115 | auto typeCandidate = input.first(n); |
| 116 | if (typeCandidate.size() == 0 || hasInvalidCodepoints(typeCandidate, isTokenChar)) { |
| 117 | return kj::none; |
| 118 | } |
| 119 | maybeType = toLower(typeCandidate); |
| 120 | input = input.slice(n + 1); |
| 121 | } else { |
| 122 | // If the solidus is not found, then it's not a valid mime type |
| 123 | return kj::none; |
| 124 | } |
| 125 | |
| 126 | // If there's nothing else to parse at this point, it's not a valid mime type. |
| 127 | if (input.size() == 0) return kj::none; |
| 128 | |
| 129 | kj::Maybe<kj::String> maybeSubtype; |
| 130 | KJ_IF_SOME(n, input.findFirst(';')) { |
| 131 | // If a semi-colon is found, the subtype is everything up to that point |
| 132 | // minus trailing whitespace. |
| 133 | auto subtypeCandidate = trimWhitespace(input.first(n)); |
| 134 | if (subtypeCandidate.size() == 0 || hasInvalidCodepoints(subtypeCandidate, isTokenChar)) { |
| 135 | return kj::none; |
| 136 | } |
| 137 | maybeSubtype = toLower(subtypeCandidate); |
| 138 | input = input.slice(n + 1); |
| 139 | } else { |
| 140 | auto subtypeCandidate = trimWhitespace(input); |
| 141 | if (subtypeCandidate.size() == 0 || hasInvalidCodepoints(subtypeCandidate, isTokenChar)) { |
| 142 | return kj::none; |
| 143 | } |
| 144 | maybeSubtype = toLower(subtypeCandidate); |
| 145 | input = {}; |
| 146 | } |
| 147 | |
| 148 | MimeType result(kj::mv(KJ_ASSERT_NONNULL(maybeType)), kj::mv(KJ_ASSERT_NONNULL(maybeSubtype))); |
| 149 | |
| 150 | if (!(options & ParseOptions::IGNORE_PARAMS)) { |
| 151 | // Parse the parameters... |
| 152 | while (input.size() > 0) { |
| 153 | input = skipWhitespace(input); |
| 154 | if (input.size() == 0) break; |
| 155 | KJ_IF_SOME(n, findParamDelimiter(input)) { |
| 156 | // If the delimiter found is a ; then the parameter is invalid here, and |
| 157 | // we will ignore it. |
| 158 | if (input[n] == ';') { |
| 159 | input = input.slice(n + 1); |
| 160 | continue; |
| 161 | } |
| 162 | KJ_ASSERT(input[n] == '='); |
| 163 | auto nameCandidate = input.first(n); |
| 164 | input = input.slice(n + 1); |
| 165 | if (nameCandidate.size() == 0 || hasInvalidCodepoints(nameCandidate, isTokenChar)) { |
| 166 | // The name is invalid, try skipping to the next... |
| 167 | KJ_IF_SOME(p, input.findFirst(';')) { |
| 168 | input = input.slice(p + 1); |
| 169 | continue; |
| 170 | } else { |
| 171 | break; |
| 172 | } |
| 173 | } |
| 174 | if (input.size() == 0) break; |
| 175 | |
| 176 | // Check to see if the value starts off quoted or not. |
| 177 | if (*input.begin() == '"') { |
| 178 | // Collect an HTTP quoted string per Fetch spec §2.6, with extract-value=true. |
| 179 | // Process character-by-character to correctly handle backslash escapes. |
| 180 | input = input.slice(1); // Skip opening quote |
| 181 | auto valueBuf = kj::heapString(input.size()); |
| 182 | char* out = valueBuf.begin(); |
| 183 | while (input.size() > 0) { |
| 184 | char c = input[0]; |
| 185 | if (c == '"') { |
| 186 | // Closing quote found |
| 187 | input = input.slice(1); |
| 188 | break; |
| 189 | } else if (c == '\\') { |
| 190 | input = input.slice(1); |
| 191 | if (input.size() == 0) { |
| 192 | // Trailing backslash at end of input — append literal backslash per spec |
| 193 | *out++ = '\\'; |
| 194 | break; |
| 195 | } |
| 196 | *out++ = input[0]; |
| 197 | input = input.slice(1); |
| 198 | } else { |
| 199 | *out++ = c; |
| 200 | input = input.slice(1); |
| 201 | } |
| 202 | } |
| 203 | auto valueCandidate = |
| 204 | kj::heapString(kj::arrayPtr(valueBuf.begin(), out - valueBuf.begin())); |
| 205 | // Spec step 11.8.2: skip any trailing content to the next ';' |
| 206 | KJ_IF_SOME(p, input.findFirst(';')) { |
| 207 | result.addParam(nameCandidate, valueCandidate); |
| 208 | input = input.slice(p + 1); |
| 209 | continue; |
| 210 | } |
| 211 | result.addParam(nameCandidate, valueCandidate); |
| 212 | break; |
| 213 | } else { |
| 214 | // The parameter is not quoted. Let's scan ahead for the next semi-colon. |
| 215 | KJ_IF_SOME(p, input.findFirst(';')) { |
| 216 | auto valueCandidate = trimWhitespace(input.first(p)); |
| 217 | input = input.slice(p + 1); |
| 218 | if (valueCandidate.size() > 0 && |
| 219 | !hasInvalidCodepoints(valueCandidate, isQuotedStringTokenChar)) { |
| 220 | result.addParam(nameCandidate, valueCandidate); |
| 221 | } |
| 222 | continue; |
| 223 | } else { |
| 224 | auto valueCandidate = trimWhitespace(input); |
| 225 | if (valueCandidate.size() > 0 && |
| 226 | !hasInvalidCodepoints(valueCandidate, isQuotedStringTokenChar)) { |
| 227 | result.addParam(nameCandidate, valueCandidate); |
| 228 | } |
| 229 | } |
| 230 | break; |
| 231 | } |
| 232 | } else { |
| 233 | // If we got here, we scanned input and did not find a semi-colon or equal |
| 234 | // sign before hitting the end of the input. We treat the remaining bits as |
| 235 | // invalid and ignore them. |
| 236 | break; |
| 237 | } |
| 238 | } |
| 239 | } |
| 240 | |
| 241 | return kj::mv(result); |
| 242 | } |
| 243 | |
| 244 | MimeType::MimeType(kj::StringPtr type, kj::StringPtr subtype, kj::Maybe<MimeParams> params) |
| 245 | : MimeType(toLower(type), toLower(subtype), kj::mv(params)) {} |
| 246 | |
| 247 | MimeType::MimeType(kj::String type, kj::String subtype, kj::Maybe<MimeParams> params) |
| 248 | : type_(kj::mv(type)), |
| 249 | subtype_(kj::mv(subtype)) { |
| 250 | KJ_IF_SOME(p, params) { |
| 251 | params_ = kj::mv(p); |
| 252 | } |
| 253 | } |
| 254 | |
| 255 | kj::StringPtr MimeType::type() const { |
| 256 | return type_; |
| 257 | } |
| 258 | |
| 259 | bool MimeType::setType(kj::StringPtr type) { |
| 260 | if (type.size() == 0 || hasInvalidCodepoints(type, isTokenChar)) return false; |
| 261 | type_ = toLower(type); |
| 262 | return true; |
| 263 | } |
| 264 | |
| 265 | kj::StringPtr MimeType::subtype() const { |
| 266 | return subtype_; |
| 267 | } |
| 268 | |
| 269 | bool MimeType::setSubtype(kj::StringPtr type) { |
| 270 | if (type.size() == 0 || hasInvalidCodepoints(type, isTokenChar)) return false; |
| 271 | subtype_ = toLower(type); |
| 272 | return true; |
| 273 | } |
| 274 | |
| 275 | const MimeType::MimeParams& MimeType::params() const { |
| 276 | return params_; |
| 277 | } |
| 278 | |
| 279 | bool MimeType::addParam(kj::ArrayPtr<const char> name, kj::ArrayPtr<const char> value) { |
| 280 | if (name.size() == 0 || hasInvalidCodepoints(name, isTokenChar) || |
| 281 | hasInvalidCodepoints(value, isQuotedStringTokenChar)) { |
| 282 | return false; |
| 283 | } |
| 284 | params_.upsert(toLower(name), kj::str(value), [](auto&, auto&&) {}); |
| 285 | return true; |
| 286 | } |
| 287 | |
| 288 | void MimeType::eraseParam(kj::StringPtr name) { |
| 289 | params_.erase(toLower(name)); |
| 290 | } |
| 291 | |
| 292 | kj::String MimeType::essence() const { |
| 293 | return kj::str(type(), "/", subtype()); |
| 294 | } |
| 295 | |
| 296 | kj::String MimeType::paramsToString() const { |
| 297 | ToStringBuffer buffer(512); |
| 298 | paramsToString(buffer); |
| 299 | return buffer.toString(); |
| 300 | } |
| 301 | |
| 302 | void MimeType::paramsToString(MimeType::ToStringBuffer& buffer) const { |
| 303 | bool first = true; |
| 304 | for (auto& param: params()) { |
| 305 | buffer.append(first ? "" : ";"); |
| 306 | if (param.key == "boundary") { |
| 307 | // This is to pass a pedantic WPT test that expects a space before only the boundary parameter |
| 308 | // [1]: https://html.spec.whatwg.org/#submit-body |
| 309 | // [2]: https://github.com/web-platform-tests/wpt/pull/29554 |
| 310 | buffer.append(" "); |
| 311 | } |
| 312 | |
| 313 | buffer.append(param.key, "="); |
| 314 | first = false; |
| 315 | if (param.value.size() == 0) { |
| 316 | buffer.append("\"\""); |
| 317 | } else if (hasInvalidCodepoints(param.value, isTokenChar)) { |
| 318 | auto view = param.value.asPtr(); |
| 319 | buffer.append("\""); |
| 320 | while (view.size() > 0) { |
| 321 | // Find the next character that needs escaping (per MIME Sniffing §4.1 step 4.4.1: |
| 322 | // precede each occurrence of U+0022 (") or U+005C (\) with U+005C (\)). |
| 323 | size_t i = 0; |
| 324 | while (i < view.size() && view[i] != '"' && view[i] != '\\') ++i; |
| 325 | buffer.append(view.first(i)); |
| 326 | if (i < view.size()) { |
| 327 | buffer.append("\\", view.slice(i, i + 1)); |
| 328 | view = view.slice(i + 1); |
| 329 | } else { |
| 330 | break; |
| 331 | } |
| 332 | } |
| 333 | buffer.append("\""); |
| 334 | } else { |
| 335 | buffer.append(param.value); |
| 336 | } |
| 337 | } |
| 338 | } |
| 339 | |
| 340 | kj::String MimeType::toString() const { |
| 341 | ToStringBuffer buffer(512); |
| 342 | buffer.append(type(), "/", subtype()); |
| 343 | if (params_.size() > 0) { |
| 344 | buffer.append(";"); |
| 345 | paramsToString(buffer); |
| 346 | } |
| 347 | return buffer.toString(); |
| 348 | } |
| 349 | |
| 350 | MimeType MimeType::clone(ParseOptions options) const { |
| 351 | MimeParams copy; |
| 352 | if (!(options & ParseOptions::IGNORE_PARAMS)) { |
| 353 | for (const auto& entry: params_) { |
| 354 | copy.insert(kj::str(entry.key), kj::str(entry.value)); |
| 355 | } |
| 356 | } |
| 357 | return MimeType(kj::str(type_), kj::str(subtype_), kj::mv(copy)); |
| 358 | } |
| 359 | |
| 360 | bool MimeType::operator==(const MimeType& other) const { |
| 361 | return this == &other || (type_ == other.type_ && subtype_ == other.subtype_); |
| 362 | } |
| 363 | |
| 364 | MimeType::operator kj::String() const { |
| 365 | return toString(); |
| 366 | } |
| 367 | |
| 368 | kj::String KJ_STRINGIFY(const MimeType& mimeType) { |
| 369 | return mimeType.toString(); |
| 370 | } |
| 371 | |
| 372 | kj::String KJ_STRINGIFY(const ConstMimeType& state) { |
| 373 | return state.toString(); |
| 374 | } |
| 375 | |
| 376 | const MimeType MimeType::PLAINTEXT = MimeType::parse(PLAINTEXT_STRING); |
| 377 | const MimeType MimeType::PLAINTEXT_ASCII = MimeType::parse(PLAINTEXT_ASCII_STRING); |
| 378 | |
| 379 | kj::Maybe<MimeType> MimeType::extract(kj::StringPtr input) { |
| 380 | kj::Maybe<MimeType> mimeType; |
| 381 | |
| 382 | constexpr static auto findNextSeparator = [](auto& input) -> kj::Maybe<size_t> { |
| 383 | // Scans input to find the next comma (,) that is not contained within |
| 384 | // a quoted section, returning the position of the comma or kj::none |
| 385 | // if not found. |
| 386 | for (size_t i = 0; i < input.size(); ++i) { |
| 387 | if (input[i] == '"' && (i == 0 || input[i - 1] != '\\')) { |
| 388 | // Skip to the end of the quoted section |
| 389 | while (++i < input.size() && (input[i] != '"' || input[i - 1] == '\\')) {} |
| 390 | } else if (input[i] == ',' && (i == 0 || input[i - 1] != '\\')) { |
| 391 | return i; |
| 392 | } |
| 393 | } |
| 394 | return kj::none; |
| 395 | }; |
| 396 | |
| 397 | constexpr static auto processPart = [](auto& mimeType, auto& part) -> kj::Maybe<MimeType> { |
| 398 | KJ_IF_SOME(parsed, tryParse(part)) { |
| 399 | if (parsed == MimeType::WILDCARD) return kj::none; |
| 400 | |
| 401 | KJ_IF_SOME(current, mimeType) { |
| 402 | if (current == parsed) { |
| 403 | // mimeType will be set to parsed, but if parsed does not |
| 404 | // have a charset, we will set the charset from current, if any. |
| 405 | if (parsed.params().find("charset"_kj) == kj::none) { |
| 406 | KJ_IF_SOME(charset, current.params().find("charset"_kj)) { |
| 407 | parsed.addParam("charset"_kj, charset); |
| 408 | } |
| 409 | } |
| 410 | } |
| 411 | } |
| 412 | return kj::mv(parsed); |
| 413 | } |
| 414 | |
| 415 | return kj::none; |
| 416 | }; |
| 417 | |
| 418 | while (input.size() > 0) { |
| 419 | KJ_IF_SOME(pos, findNextSeparator(input)) { |
| 420 | auto part = input.first(pos); |
| 421 | input = input.slice(pos + 1); |
| 422 | KJ_IF_SOME(parsed, processPart(mimeType, part)) { |
| 423 | mimeType = kj::mv(parsed); |
| 424 | } else { |
| 425 | continue; |
| 426 | } |
| 427 | } else { |
| 428 | KJ_IF_SOME(parsed, processPart(mimeType, input)) { |
| 429 | mimeType = kj::mv(parsed); |
| 430 | } |
| 431 | break; |
| 432 | } |
| 433 | } |
| 434 | |
| 435 | return kj::mv(mimeType); |
| 436 | } |
| 437 | |
| 438 | kj::String MimeType::formDataWithBoundary(kj::StringPtr boundary) { |
| 439 | // Note that the expectation is that the boundary is already properly formed |
| 440 | // and does not need any additional quoting or escaping. |
| 441 | return kj::str("multipart/form-data; boundary=", boundary); |
| 442 | } |
| 443 | |
| 444 | kj::String MimeType::formUrlEncodedWithCharset(kj::StringPtr charset) { |
| 445 | return kj::str("application/x-www-form-urlencoded;charset=", charset); |
| 446 | } |
| 447 | |
| 448 | } // namespace workerd |