File
Blob: src/workerd/api/blob.c++
| 1 | // Copyright (c) 2017-2022 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | |
| 5 | #include "blob.h" |
| 6 | |
| 7 | #include <workerd/api/streams/readable-source.h> |
| 8 | #include <workerd/api/streams/readable.h> |
| 9 | #include <workerd/io/features.h> |
| 10 | #include <workerd/io/observer.h> |
| 11 | #include <workerd/util/mimetype.h> |
| 12 | #include <workerd/util/stream-utils.h> |
| 13 | |
| 14 | namespace workerd::api { |
| 15 | |
| 16 | namespace { |
| 17 | // Concatenate an array of segments (parameter to Blob constructor). |
| 18 | kj::Maybe<jsg::JsBufferSource> concat(jsg::Lock& js, jsg::Optional<Blob::Bits> maybeBits) { |
| 19 | auto bits = kj::mv(maybeBits).orDefault(nullptr); |
| 20 | if (bits.size() == 0) { |
| 21 | return kj::none; |
| 22 | } |
| 23 | |
| 24 | auto rejectResizable = FeatureFlags::get(js).getNoResizableArrayBufferInBlob(); |
| 25 | auto maxBlobSize = Worker::Isolate::from(js).getLimitEnforcer().getBlobSizeLimit(); |
| 26 | static constexpr int kMaxInt KJ_UNUSED = kj::maxValue; |
| 27 | KJ_DASSERT(maxBlobSize <= kMaxInt, "Blob size limit exceeds int range"); |
| 28 | size_t size = 0; |
| 29 | kj::SmallArray<size_t, 8> cachedPartSizes(bits.size()); |
| 30 | size_t index = 0; |
| 31 | for (auto& part: bits) { |
| 32 | size_t partSize = 0; |
| 33 | KJ_SWITCH_ONEOF(part) { |
| 34 | KJ_CASE_ONEOF(bytes, jsg::JsBufferSource) { |
| 35 | if (rejectResizable) { |
| 36 | JSG_REQUIRE( |
| 37 | !bytes.isResizable(), TypeError, "Cannot create a Blob with a resizable ArrayBuffer"); |
| 38 | } |
| 39 | partSize = bytes.size(); |
| 40 | } |
| 41 | KJ_CASE_ONEOF(text, kj::String) { |
| 42 | partSize = text.asBytes().size(); |
| 43 | } |
| 44 | KJ_CASE_ONEOF(blob, jsg::Ref<Blob>) { |
| 45 | partSize = blob->getData().size(); |
| 46 | } |
| 47 | } |
| 48 | cachedPartSizes[index++] = partSize; |
| 49 | |
| 50 | // We can skip the remaining checks if the part is empty. |
| 51 | if (partSize == 0) continue; |
| 52 | |
| 53 | // While overflow is *extremely* unlikely to ever be a problem here, let's |
| 54 | // be extra cautious and check for it anyway. |
| 55 | static constexpr size_t kOverflowLimit = kj::maxValue; |
| 56 | // Upper limit the max number of bytes we can add to size to avoid overflow. |
| 57 | // partSize must be less than or equal to upperLimit. Practically speaking, |
| 58 | // however, it is practically impossible to reach this limit in any real-world |
| 59 | // scenario given the size limit check below. |
| 60 | size_t upperLimit = kOverflowLimit - size; |
| 61 | JSG_REQUIRE( |
| 62 | partSize <= upperLimit, RangeError, kj::str("Blob part too large: ", partSize, " bytes")); |
| 63 | |
| 64 | // Checks for oversize |
| 65 | JSG_REQUIRE(size + partSize <= maxBlobSize, RangeError, |
| 66 | kj::str("Blob size ", size + partSize, " exceeds limit ", maxBlobSize)); |
| 67 | size += partSize; |
| 68 | } |
| 69 | |
| 70 | if (size == 0) { |
| 71 | return kj::none; |
| 72 | } |
| 73 | |
| 74 | auto u8 = jsg::JsUint8Array::create(js, size); |
| 75 | |
| 76 | auto view = u8.asArrayPtr(); |
| 77 | |
| 78 | index = 0; |
| 79 | for (auto& part: bits) { |
| 80 | KJ_SWITCH_ONEOF(part) { |
| 81 | KJ_CASE_ONEOF(bytes, jsg::JsBufferSource) { |
| 82 | size_t cachedSize = cachedPartSizes[index++]; |
| 83 | // If the ArrayBuffer was resized larger, we'll ignore the additional bytes. |
| 84 | // If the ArrayBuffer was resized smaller, we'll copy up to the current size. |
| 85 | // In either case, data is packed tightly — any unused space from shrunk |
| 86 | // buffers ends up as zeros at the end of the output rather than as gaps |
| 87 | // in the middle. |
| 88 | size_t toCopy = kj::min(bytes.size(), cachedSize); |
| 89 | if (toCopy > 0) { |
| 90 | KJ_ASSERT(view.size() >= toCopy); |
| 91 | view.first(toCopy).copyFrom(bytes.asArrayPtr().first(toCopy)); |
| 92 | } |
| 93 | view = view.slice(toCopy); |
| 94 | } |
| 95 | KJ_CASE_ONEOF(text, kj::String) { |
| 96 | auto byteLength = text.asBytes().size(); |
| 97 | KJ_ASSERT(byteLength == cachedPartSizes[index++]); |
| 98 | if (byteLength == 0) continue; |
| 99 | KJ_ASSERT(view.size() >= byteLength); |
| 100 | view.first(byteLength).copyFrom(text.asBytes()); |
| 101 | view = view.slice(byteLength); |
| 102 | } |
| 103 | KJ_CASE_ONEOF(blob, jsg::Ref<Blob>) { |
| 104 | auto data = blob->getData(); |
| 105 | KJ_ASSERT(data.size() == cachedPartSizes[index++]); |
| 106 | if (data.size() == 0) continue; |
| 107 | KJ_ASSERT(view.size() >= data.size()); |
| 108 | view.first(data.size()).copyFrom(data); |
| 109 | view = view.slice(data.size()); |
| 110 | } |
| 111 | } |
| 112 | } |
| 113 | |
| 114 | // view.size() will be non-zero if one or more resizable ArrayBuffers were shrunk |
| 115 | // between the size-computation pass and the copy pass. In that case, create a |
| 116 | // trimmed view over just the bytes that were actually written. |
| 117 | size_t bytesWritten = size - view.size(); |
| 118 | if (bytesWritten == 0) { |
| 119 | return kj::none; |
| 120 | } |
| 121 | if (bytesWritten < size) { |
| 122 | return jsg::JsBufferSource(u8.slice(js, bytesWritten)); |
| 123 | } |
| 124 | |
| 125 | KJ_ASSERT(view == nullptr); |
| 126 | return jsg::JsBufferSource(u8); |
| 127 | } |
| 128 | |
| 129 | kj::String normalizeType(kj::String type) { |
| 130 | // This does not properly parse mime types. We have the new workerd::MimeType impl |
| 131 | // but that handles mime types a bit more strictly than this. Ideally we'd be able to |
| 132 | // switch over to it but there's a non-zero risk of breaking running code. We might need |
| 133 | // a compat flag to switch at some point but for now we'll keep this as it is. |
| 134 | |
| 135 | // https://www.w3.org/TR/FileAPI/#constructorBlob step 3 inexplicably insists that if the |
| 136 | // type contains non-printable-ASCII characters we should discard it, and otherwise we should |
| 137 | // lower-case it. |
| 138 | for (char& c: type) { |
| 139 | if (static_cast<signed char>(c) < 0x20) { |
| 140 | // Throw it away. |
| 141 | return nullptr; |
| 142 | } else if ('A' <= c && c <= 'Z') { |
| 143 | c = c - 'A' + 'a'; |
| 144 | } |
| 145 | } |
| 146 | |
| 147 | return kj::mv(type); |
| 148 | } |
| 149 | |
| 150 | } // namespace |
| 151 | |
| 152 | Blob::Blob(kj::String type): ownData(Empty{}), data(nullptr), type(kj::mv(type)) {} |
| 153 | |
| 154 | Blob::Blob(jsg::Lock& js, jsg::JsBufferSource data, kj::String type) |
| 155 | : ownData(data.addRef(js)), |
| 156 | data(data.asArrayPtr()), |
| 157 | type(kj::mv(type)) { |
| 158 | if (FeatureFlags::get(js).getNoResizableArrayBufferInBlob()) { |
| 159 | JSG_REQUIRE( |
| 160 | !data.isResizable(), TypeError, "Cannot create a Blob with a resizable ArrayBuffer"); |
| 161 | } |
| 162 | } |
| 163 | |
| 164 | Blob::Blob(jsg::Ref<Blob> parent, kj::ArrayPtr<const byte> data, kj::String type) |
| 165 | : ownData(kj::mv(parent)), |
| 166 | data(data), |
| 167 | type(kj::mv(type)) {} |
| 168 | |
| 169 | jsg::Ref<Blob> Blob::constructor( |
| 170 | jsg::Lock& js, jsg::Optional<Bits> bits, jsg::Optional<Options> options) { |
| 171 | kj::String type; // note: default value is intentionally empty string |
| 172 | KJ_IF_SOME(o, options) { |
| 173 | KJ_IF_SOME(t, o.type) { |
| 174 | type = normalizeType(kj::mv(t)); |
| 175 | } |
| 176 | } |
| 177 | |
| 178 | KJ_IF_SOME(b, bits) { |
| 179 | // Optimize for the case where the input is a single Blob, where we can just |
| 180 | // return a new view on the existing data without copying. |
| 181 | if (b.size() == 1) { |
| 182 | KJ_IF_SOME(parent, b[0].template tryGet<jsg::Ref<Blob>>()) { |
| 183 | if (parent->getSize() == 0) { |
| 184 | return js.alloc<Blob>(kj::mv(type)); |
| 185 | } |
| 186 | auto ptr = parent->data; |
| 187 | KJ_IF_SOME(root, parent->ownData.template tryGet<jsg::Ref<Blob>>()) { |
| 188 | parent = root.addRef(); |
| 189 | } |
| 190 | return js.alloc<Blob>(kj::mv(parent), ptr, kj::mv(type)); |
| 191 | } |
| 192 | } |
| 193 | } |
| 194 | |
| 195 | KJ_IF_SOME(data, concat(js, kj::mv(bits))) { |
| 196 | return js.alloc<Blob>(js, data, kj::mv(type)); |
| 197 | } |
| 198 | return js.alloc<Blob>(kj::mv(type)); |
| 199 | } |
| 200 | |
| 201 | kj::ArrayPtr<const byte> Blob::getData() const { |
| 202 | return data; |
| 203 | } |
| 204 | |
| 205 | jsg::Ref<Blob> Blob::slice(jsg::Lock& js, |
| 206 | jsg::Optional<int> maybeStart, |
| 207 | jsg::Optional<int> maybeEnd, |
| 208 | jsg::Optional<kj::String> type) { |
| 209 | |
| 210 | auto normalizedType = normalizeType(kj::mv(type).orDefault(nullptr)); |
| 211 | if (data.size() == 0) { |
| 212 | // Blob is empty, there's nothing to slice. |
| 213 | return js.alloc<Blob>(kj::mv(normalizedType)); |
| 214 | } |
| 215 | |
| 216 | int start = maybeStart.orDefault(0); |
| 217 | int end = maybeEnd.orDefault(data.size()); |
| 218 | |
| 219 | if (start < 0) { |
| 220 | // Negative value interpreted as offset from end. |
| 221 | start += data.size(); |
| 222 | } |
| 223 | if (end < 0) { |
| 224 | // Negative value interpreted as offset from end. |
| 225 | end += data.size(); |
| 226 | } |
| 227 | |
| 228 | // Clamp start and end to range. |
| 229 | start = kj::max(0, kj::min(start, static_cast<int>(data.size()))); |
| 230 | end = kj::max(start, kj::min(end, static_cast<int>(data.size()))); |
| 231 | |
| 232 | // We run with KJ_IREQUIRE checks enabled in production, which will catch |
| 233 | // out of bounds start/end ... but since we're clamping them above, this |
| 234 | // should never actually be a problem. |
| 235 | auto slicedData = data.slice(start, end); |
| 236 | |
| 237 | // If the slice is empty, we can just return a new empty Blob without worrying about |
| 238 | // referencing the original data at all. Super minor optimization that avoids an |
| 239 | // unnecessary refcount. |
| 240 | if (slicedData.size() == 0) { |
| 241 | return js.alloc<Blob>(kj::mv(normalizedType)); |
| 242 | } |
| 243 | |
| 244 | KJ_SWITCH_ONEOF(ownData) { |
| 245 | KJ_CASE_ONEOF(_, Empty) { |
| 246 | // Handled at the beginning of the function with the zero-length check. |
| 247 | KJ_FAIL_ASSERT("Empty blob should have been handled at the beginning of the function"); |
| 248 | } |
| 249 | KJ_CASE_ONEOF(parent, jsg::Ref<Blob>) { |
| 250 | // If this blob is itself a slice (backed by a Ref<Blob>), reference the |
| 251 | // root data-owning blob directly. This prevents unbounded chain depth — |
| 252 | // every slice always points to the root, so depth is always ≤ 1. |
| 253 | return js.alloc<Blob>(parent.addRef(), slicedData, kj::mv(normalizedType)); |
| 254 | } |
| 255 | KJ_CASE_ONEOF(_, jsg::JsRef<jsg::JsBufferSource>) { |
| 256 | return js.alloc<Blob>(JSG_THIS, slicedData, kj::mv(normalizedType)); |
| 257 | } |
| 258 | } |
| 259 | KJ_UNREACHABLE; |
| 260 | } |
| 261 | |
| 262 | jsg::Promise<jsg::JsRef<jsg::JsArrayBuffer>> Blob::arrayBuffer(jsg::Lock& js) { |
| 263 | FeatureObserver::maybeRecordUse(FeatureObserver::Feature::BLOB_AS_ARRAY_BUFFER); |
| 264 | auto ret = jsg::JsArrayBuffer::create(js, data); |
| 265 | return js.resolvedPromise(ret.addRef(js)); |
| 266 | } |
| 267 | |
| 268 | jsg::Promise<jsg::JsRef<jsg::JsUint8Array>> Blob::bytes(jsg::Lock& js) { |
| 269 | FeatureObserver::maybeRecordUse(FeatureObserver::Feature::BLOB_AS_ARRAY_BUFFER); |
| 270 | auto ret = jsg::JsUint8Array::create(js, data); |
| 271 | return js.resolvedPromise(ret.addRef(js)); |
| 272 | } |
| 273 | |
| 274 | jsg::Promise<jsg::JsRef<jsg::JsString>> Blob::text(jsg::Lock& js) { |
| 275 | FeatureObserver::maybeRecordUse(FeatureObserver::Feature::BLOB_AS_TEXT); |
| 276 | // Using js.str here instead of returning kj::String avoids an additional |
| 277 | // intermediate allocation and copy of the string data. |
| 278 | return js.resolvedPromise(js.str(data.asChars()).addRef(js)); |
| 279 | } |
| 280 | |
| 281 | jsg::Ref<ReadableStream> Blob::stream(jsg::Lock& js) { |
| 282 | FeatureObserver::maybeRecordUse(FeatureObserver::Feature::BLOB_AS_STREAM); |
| 283 | return js.alloc<ReadableStream>( |
| 284 | IoContext::current(), streams::newMemorySource(data, kj::heap(JSG_THIS))); |
| 285 | } |
| 286 | |
| 287 | // ======================================================================================= |
| 288 | |
| 289 | File::File(kj::String name, kj::String type, double lastModified) |
| 290 | : Blob(kj::mv(type)), |
| 291 | name(kj::mv(name)), |
| 292 | lastModified(lastModified) {} |
| 293 | |
| 294 | File::File( |
| 295 | jsg::Lock& js, jsg::JsBufferSource data, kj::String name, kj::String type, double lastModified) |
| 296 | : Blob(js, kj::mv(data), kj::mv(type)), |
| 297 | name(kj::mv(name)), |
| 298 | lastModified(lastModified) {} |
| 299 | |
| 300 | File::File(jsg::Ref<Blob> parent, |
| 301 | kj::ArrayPtr<const byte> data, |
| 302 | kj::String name, |
| 303 | kj::String type, |
| 304 | double lastModified) |
| 305 | : Blob(kj::mv(parent), data, kj::mv(type)), |
| 306 | name(kj::mv(name)), |
| 307 | lastModified(lastModified) {} |
| 308 | |
| 309 | jsg::Ref<File> File::constructor( |
| 310 | jsg::Lock& js, jsg::Optional<Bits> bits, kj::String name, jsg::Optional<Options> options) { |
| 311 | kj::String type; // note: default value is intentionally empty string |
| 312 | kj::Maybe<double> maybeLastModified; |
| 313 | KJ_IF_SOME(o, options) { |
| 314 | KJ_IF_SOME(t, o.type) { |
| 315 | type = normalizeType(kj::mv(t)); |
| 316 | } |
| 317 | maybeLastModified = o.lastModified; |
| 318 | } |
| 319 | |
| 320 | double lastModified; |
| 321 | KJ_IF_SOME(m, maybeLastModified) { |
| 322 | lastModified = kj::isNaN(m) ? 0 : m; |
| 323 | } else { |
| 324 | lastModified = dateNow(); |
| 325 | } |
| 326 | |
| 327 | KJ_IF_SOME(b, bits) { |
| 328 | // Optimize for the case where the input is a single Blob, where we can just |
| 329 | // return a new view on the existing data without copying. |
| 330 | if (b.size() == 1) { |
| 331 | KJ_IF_SOME(parent, b[0].template tryGet<jsg::Ref<Blob>>()) { |
| 332 | if (parent->getSize() == 0) { |
| 333 | return js.alloc<File>(kj::mv(name), kj::mv(type), lastModified); |
| 334 | } |
| 335 | auto ptr = parent->data; |
| 336 | KJ_IF_SOME(root, parent->ownData.template tryGet<jsg::Ref<Blob>>()) { |
| 337 | parent = root.addRef(); |
| 338 | } |
| 339 | return js.alloc<File>(kj::mv(parent), ptr, kj::mv(name), kj::mv(type), lastModified); |
| 340 | } |
| 341 | } |
| 342 | } |
| 343 | |
| 344 | KJ_IF_SOME(data, concat(js, kj::mv(bits))) { |
| 345 | return js.alloc<File>(js, data, kj::mv(name), kj::mv(type), lastModified); |
| 346 | } |
| 347 | return js.alloc<File>(kj::mv(name), kj::mv(type), lastModified); |
| 348 | } |
| 349 | |
| 350 | } // namespace workerd::api |