Skip to content
File

Blob: src/workerd/util/mimetype.c++

13.6 KB
1// Copyright (c) 2023 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4#include "mimetype.h"
5 
6#include "strings.h"
7 
8#include <workerd/util/string-buffer.h>
9 
10#include <kj/debug.h>
11 
12namespace workerd {
13 
14namespace {
15 
16constexpr bool isWhitespace(const char c) noexcept {
17 return (c == '\r' || c == '\n' || c == '\t' || c == ' ');
18}
19 
20static constexpr kj::FixedArray<uint8_t, 256> token_table = []() consteval {
21 kj::FixedArray<uint8_t, 256> result{};
22 
23 for (uint8_t c:
24 {'!', '#', '$', '%', '&', '\'', '*', '+', '\\', '-', '.', '^', '_', '`', '|', '~'}) {
25 result[c] = true;
26 }
27 
28 // (c >= 'A' && c <= 'Z')
29 for (uint8_t c = 'A'; c <= 'Z'; c++) {
30 result[c] = true;
31 }
32 
33 // (c >= 'a' && c <= 'z')
34 for (uint8_t c = 'a'; c <= 'z'; c++) {
35 result[c] = true;
36 }
37 
38 // (c >= '0' && c <= '9')
39 for (uint8_t c = '0'; c <= '9'; c++) {
40 result[c] = true;
41 }
42 
43 return result;
44}();
45 
46constexpr bool isTokenChar(const uint8_t c) noexcept {
47 return token_table[c];
48}
49 
50static constexpr kj::FixedArray<uint8_t, 256> quoted_string_token_table = []() consteval {
51 kj::FixedArray<uint8_t, 256> result{};
52 result['\t'] = true;
53 
54 for (uint8_t c = 0x20; c <= 0x7e; c++) {
55 result[c] = true;
56 }
57 
58 for (uint8_t c = 0x80; c < 255; c++) {
59 result[c] = true;
60 }
61 
62 return result;
63}();
64 
65constexpr bool isQuotedStringTokenChar(const uint8_t c) noexcept {
66 return quoted_string_token_table[c];
67}
68 
69kj::ArrayPtr<const char> skipWhitespace(kj::ArrayPtr<const char> str) {
70 auto ptr = str.begin();
71 auto end = str.end();
72 while (ptr != end && isWhitespace(*ptr)) {
73 ptr++;
74 }
75 return str.slice(ptr - str.begin());
76}
77 
78kj::ArrayPtr<const char> trimWhitespace(kj::ArrayPtr<const char> str) {
79 auto ptr = str.end();
80 while (ptr > str.begin() && isWhitespace(*(ptr - 1))) --ptr;
81 return str.first(ptr - str.begin());
82}
83 
84constexpr bool hasInvalidCodepoints(kj::ArrayPtr<const char> str, auto predicate) {
85 bool has_invalid_codepoints = false;
86 for (const char c: str) {
87 has_invalid_codepoints |= !predicate(static_cast<uint8_t>(c));
88 }
89 return has_invalid_codepoints;
90}
91 
92kj::Maybe<size_t> findParamDelimiter(kj::ArrayPtr<const char> str) {
93 auto ptr = str.begin();
94 while (ptr != str.end()) {
95 if (*ptr == ';' || *ptr == '=') return ptr - str.begin();
96 ++ptr;
97 }
98 return kj::none;
99}
100 
101} // namespace
102 
103MimeType MimeType::parse(kj::StringPtr input, ParseOptions options) {
104 return KJ_ASSERT_NONNULL(tryParse(input, options));
105}
106 
107kj::Maybe<MimeType> MimeType::tryParse(kj::ArrayPtr<const char> input, ParseOptions options) {
108 // Skip leading whitespace from start
109 input = skipWhitespace(input);
110 if (input.size() == 0) return kj::none;
111 
112 kj::Maybe<kj::String> maybeType;
113 // Let's try to find the solidus that separates the type and subtype
114 KJ_IF_SOME(n, input.findFirst('/')) {
115 auto typeCandidate = input.first(n);
116 if (typeCandidate.size() == 0 || hasInvalidCodepoints(typeCandidate, isTokenChar)) {
117 return kj::none;
118 }
119 maybeType = toLower(typeCandidate);
120 input = input.slice(n + 1);
121 } else {
122 // If the solidus is not found, then it's not a valid mime type
123 return kj::none;
124 }
125 
126 // If there's nothing else to parse at this point, it's not a valid mime type.
127 if (input.size() == 0) return kj::none;
128 
129 kj::Maybe<kj::String> maybeSubtype;
130 KJ_IF_SOME(n, input.findFirst(';')) {
131 // If a semi-colon is found, the subtype is everything up to that point
132 // minus trailing whitespace.
133 auto subtypeCandidate = trimWhitespace(input.first(n));
134 if (subtypeCandidate.size() == 0 || hasInvalidCodepoints(subtypeCandidate, isTokenChar)) {
135 return kj::none;
136 }
137 maybeSubtype = toLower(subtypeCandidate);
138 input = input.slice(n + 1);
139 } else {
140 auto subtypeCandidate = trimWhitespace(input);
141 if (subtypeCandidate.size() == 0 || hasInvalidCodepoints(subtypeCandidate, isTokenChar)) {
142 return kj::none;
143 }
144 maybeSubtype = toLower(subtypeCandidate);
145 input = {};
146 }
147 
148 MimeType result(kj::mv(KJ_ASSERT_NONNULL(maybeType)), kj::mv(KJ_ASSERT_NONNULL(maybeSubtype)));
149 
150 if (!(options & ParseOptions::IGNORE_PARAMS)) {
151 // Parse the parameters...
152 while (input.size() > 0) {
153 input = skipWhitespace(input);
154 if (input.size() == 0) break;
155 KJ_IF_SOME(n, findParamDelimiter(input)) {
156 // If the delimiter found is a ; then the parameter is invalid here, and
157 // we will ignore it.
158 if (input[n] == ';') {
159 input = input.slice(n + 1);
160 continue;
161 }
162 KJ_ASSERT(input[n] == '=');
163 auto nameCandidate = input.first(n);
164 input = input.slice(n + 1);
165 if (nameCandidate.size() == 0 || hasInvalidCodepoints(nameCandidate, isTokenChar)) {
166 // The name is invalid, try skipping to the next...
167 KJ_IF_SOME(p, input.findFirst(';')) {
168 input = input.slice(p + 1);
169 continue;
170 } else {
171 break;
172 }
173 }
174 if (input.size() == 0) break;
175 
176 // Check to see if the value starts off quoted or not.
177 if (*input.begin() == '"') {
178 // Collect an HTTP quoted string per Fetch spec §2.6, with extract-value=true.
179 // Process character-by-character to correctly handle backslash escapes.
180 input = input.slice(1); // Skip opening quote
181 auto valueBuf = kj::heapString(input.size());
182 char* out = valueBuf.begin();
183 while (input.size() > 0) {
184 char c = input[0];
185 if (c == '"') {
186 // Closing quote found
187 input = input.slice(1);
188 break;
189 } else if (c == '\\') {
190 input = input.slice(1);
191 if (input.size() == 0) {
192 // Trailing backslash at end of input — append literal backslash per spec
193 *out++ = '\\';
194 break;
195 }
196 *out++ = input[0];
197 input = input.slice(1);
198 } else {
199 *out++ = c;
200 input = input.slice(1);
201 }
202 }
203 auto valueCandidate =
204 kj::heapString(kj::arrayPtr(valueBuf.begin(), out - valueBuf.begin()));
205 // Spec step 11.8.2: skip any trailing content to the next ';'
206 KJ_IF_SOME(p, input.findFirst(';')) {
207 result.addParam(nameCandidate, valueCandidate);
208 input = input.slice(p + 1);
209 continue;
210 }
211 result.addParam(nameCandidate, valueCandidate);
212 break;
213 } else {
214 // The parameter is not quoted. Let's scan ahead for the next semi-colon.
215 KJ_IF_SOME(p, input.findFirst(';')) {
216 auto valueCandidate = trimWhitespace(input.first(p));
217 input = input.slice(p + 1);
218 if (valueCandidate.size() > 0 &&
219 !hasInvalidCodepoints(valueCandidate, isQuotedStringTokenChar)) {
220 result.addParam(nameCandidate, valueCandidate);
221 }
222 continue;
223 } else {
224 auto valueCandidate = trimWhitespace(input);
225 if (valueCandidate.size() > 0 &&
226 !hasInvalidCodepoints(valueCandidate, isQuotedStringTokenChar)) {
227 result.addParam(nameCandidate, valueCandidate);
228 }
229 }
230 break;
231 }
232 } else {
233 // If we got here, we scanned input and did not find a semi-colon or equal
234 // sign before hitting the end of the input. We treat the remaining bits as
235 // invalid and ignore them.
236 break;
237 }
238 }
239 }
240 
241 return kj::mv(result);
242}
243 
244MimeType::MimeType(kj::StringPtr type, kj::StringPtr subtype, kj::Maybe<MimeParams> params)
245 : MimeType(toLower(type), toLower(subtype), kj::mv(params)) {}
246 
247MimeType::MimeType(kj::String type, kj::String subtype, kj::Maybe<MimeParams> params)
248 : type_(kj::mv(type)),
249 subtype_(kj::mv(subtype)) {
250 KJ_IF_SOME(p, params) {
251 params_ = kj::mv(p);
252 }
253}
254 
255kj::StringPtr MimeType::type() const {
256 return type_;
257}
258 
259bool MimeType::setType(kj::StringPtr type) {
260 if (type.size() == 0 || hasInvalidCodepoints(type, isTokenChar)) return false;
261 type_ = toLower(type);
262 return true;
263}
264 
265kj::StringPtr MimeType::subtype() const {
266 return subtype_;
267}
268 
269bool MimeType::setSubtype(kj::StringPtr type) {
270 if (type.size() == 0 || hasInvalidCodepoints(type, isTokenChar)) return false;
271 subtype_ = toLower(type);
272 return true;
273}
274 
275const MimeType::MimeParams& MimeType::params() const {
276 return params_;
277}
278 
279bool MimeType::addParam(kj::ArrayPtr<const char> name, kj::ArrayPtr<const char> value) {
280 if (name.size() == 0 || hasInvalidCodepoints(name, isTokenChar) ||
281 hasInvalidCodepoints(value, isQuotedStringTokenChar)) {
282 return false;
283 }
284 params_.upsert(toLower(name), kj::str(value), [](auto&, auto&&) {});
285 return true;
286}
287 
288void MimeType::eraseParam(kj::StringPtr name) {
289 params_.erase(toLower(name));
290}
291 
292kj::String MimeType::essence() const {
293 return kj::str(type(), "/", subtype());
294}
295 
296kj::String MimeType::paramsToString() const {
297 ToStringBuffer buffer(512);
298 paramsToString(buffer);
299 return buffer.toString();
300}
301 
302void MimeType::paramsToString(MimeType::ToStringBuffer& buffer) const {
303 bool first = true;
304 for (auto& param: params()) {
305 buffer.append(first ? "" : ";");
306 if (param.key == "boundary") {
307 // This is to pass a pedantic WPT test that expects a space before only the boundary parameter
308 // [1]: https://html.spec.whatwg.org/#submit-body
309 // [2]: https://github.com/web-platform-tests/wpt/pull/29554
310 buffer.append(" ");
311 }
312 
313 buffer.append(param.key, "=");
314 first = false;
315 if (param.value.size() == 0) {
316 buffer.append("\"\"");
317 } else if (hasInvalidCodepoints(param.value, isTokenChar)) {
318 auto view = param.value.asPtr();
319 buffer.append("\"");
320 while (view.size() > 0) {
321 // Find the next character that needs escaping (per MIME Sniffing §4.1 step 4.4.1:
322 // precede each occurrence of U+0022 (") or U+005C (\) with U+005C (\)).
323 size_t i = 0;
324 while (i < view.size() && view[i] != '"' && view[i] != '\\') ++i;
325 buffer.append(view.first(i));
326 if (i < view.size()) {
327 buffer.append("\\", view.slice(i, i + 1));
328 view = view.slice(i + 1);
329 } else {
330 break;
331 }
332 }
333 buffer.append("\"");
334 } else {
335 buffer.append(param.value);
336 }
337 }
338}
339 
340kj::String MimeType::toString() const {
341 ToStringBuffer buffer(512);
342 buffer.append(type(), "/", subtype());
343 if (params_.size() > 0) {
344 buffer.append(";");
345 paramsToString(buffer);
346 }
347 return buffer.toString();
348}
349 
350MimeType MimeType::clone(ParseOptions options) const {
351 MimeParams copy;
352 if (!(options & ParseOptions::IGNORE_PARAMS)) {
353 for (const auto& entry: params_) {
354 copy.insert(kj::str(entry.key), kj::str(entry.value));
355 }
356 }
357 return MimeType(kj::str(type_), kj::str(subtype_), kj::mv(copy));
358}
359 
360bool MimeType::operator==(const MimeType& other) const {
361 return this == &other || (type_ == other.type_ && subtype_ == other.subtype_);
362}
363 
364MimeType::operator kj::String() const {
365 return toString();
366}
367 
368kj::String KJ_STRINGIFY(const MimeType& mimeType) {
369 return mimeType.toString();
370}
371 
372kj::String KJ_STRINGIFY(const ConstMimeType& state) {
373 return state.toString();
374}
375 
376const MimeType MimeType::PLAINTEXT = MimeType::parse(PLAINTEXT_STRING);
377const MimeType MimeType::PLAINTEXT_ASCII = MimeType::parse(PLAINTEXT_ASCII_STRING);
378 
379kj::Maybe<MimeType> MimeType::extract(kj::StringPtr input) {
380 kj::Maybe<MimeType> mimeType;
381 
382 constexpr static auto findNextSeparator = [](auto& input) -> kj::Maybe<size_t> {
383 // Scans input to find the next comma (,) that is not contained within
384 // a quoted section, returning the position of the comma or kj::none
385 // if not found.
386 for (size_t i = 0; i < input.size(); ++i) {
387 if (input[i] == '"' && (i == 0 || input[i - 1] != '\\')) {
388 // Skip to the end of the quoted section
389 while (++i < input.size() && (input[i] != '"' || input[i - 1] == '\\')) {}
390 } else if (input[i] == ',' && (i == 0 || input[i - 1] != '\\')) {
391 return i;
392 }
393 }
394 return kj::none;
395 };
396 
397 constexpr static auto processPart = [](auto& mimeType, auto& part) -> kj::Maybe<MimeType> {
398 KJ_IF_SOME(parsed, tryParse(part)) {
399 if (parsed == MimeType::WILDCARD) return kj::none;
400 
401 KJ_IF_SOME(current, mimeType) {
402 if (current == parsed) {
403 // mimeType will be set to parsed, but if parsed does not
404 // have a charset, we will set the charset from current, if any.
405 if (parsed.params().find("charset"_kj) == kj::none) {
406 KJ_IF_SOME(charset, current.params().find("charset"_kj)) {
407 parsed.addParam("charset"_kj, charset);
408 }
409 }
410 }
411 }
412 return kj::mv(parsed);
413 }
414 
415 return kj::none;
416 };
417 
418 while (input.size() > 0) {
419 KJ_IF_SOME(pos, findNextSeparator(input)) {
420 auto part = input.first(pos);
421 input = input.slice(pos + 1);
422 KJ_IF_SOME(parsed, processPart(mimeType, part)) {
423 mimeType = kj::mv(parsed);
424 } else {
425 continue;
426 }
427 } else {
428 KJ_IF_SOME(parsed, processPart(mimeType, input)) {
429 mimeType = kj::mv(parsed);
430 }
431 break;
432 }
433 }
434 
435 return kj::mv(mimeType);
436}
437 
438kj::String MimeType::formDataWithBoundary(kj::StringPtr boundary) {
439 // Note that the expectation is that the boundary is already properly formed
440 // and does not need any additional quoting or escaping.
441 return kj::str("multipart/form-data; boundary=", boundary);
442}
443 
444kj::String MimeType::formUrlEncodedWithCharset(kj::StringPtr charset) {
445 return kj::str("application/x-www-form-urlencoded;charset=", charset);
446}
447 
448} // namespace workerd