File
Blob: src/workerd/api/node/buffer.c++
| 1 | // Copyright (c) 2017-2022 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | // Copyright Joyent and Node contributors. All rights reserved. MIT license. |
| 5 | |
| 6 | #include "buffer.h" |
| 7 | |
| 8 | #include "buffer-string-search.h" |
| 9 | #include "nbytes.h" |
| 10 | #include "simdutf.h" |
| 11 | |
| 12 | #include <workerd/jsg/jsg.h> |
| 13 | |
| 14 | #include <kj/array.h> |
| 15 | #include <kj/encoding.h> |
| 16 | |
| 17 | #include <algorithm> |
| 18 | |
| 19 | namespace workerd::api::node { |
| 20 | |
| 21 | namespace { |
| 22 | |
| 23 | kj::Maybe<uint> tryFromHexDigit(char c) { |
| 24 | if ('0' <= c && c <= '9') { |
| 25 | return c - '0'; |
| 26 | } else if ('a' <= c && c <= 'f') { |
| 27 | return c - ('a' - 10); |
| 28 | } else if ('A' <= c && c <= 'F') { |
| 29 | return c - ('A' - 10); |
| 30 | } |
| 31 | return kj::none; |
| 32 | } |
| 33 | |
| 34 | jsg::JsUint8Array decodeHexTruncated( |
| 35 | jsg::Lock& js, kj::ArrayPtr<kj::byte> text, bool strict = false) { |
| 36 | // We do not use kj::decodeHex because we need to match Node.js' |
| 37 | // behavior of truncating the response at the first invalid hex |
| 38 | // pair as opposed to just marking that an error happened and |
| 39 | // trying to continue with the decode. |
| 40 | if (text.size() % 2 != 0) { |
| 41 | JSG_REQUIRE(!strict, TypeError, "The text is not valid hex"); |
| 42 | text = text.first(text.size() - 1); |
| 43 | } |
| 44 | auto vec = jsg::JsUint8Array::create(js, text.size() / 2); |
| 45 | auto ptr = vec.asArrayPtr(); |
| 46 | size_t len = 0; |
| 47 | |
| 48 | for (size_t i = 0; i < text.size(); i += 2) { |
| 49 | kj::byte b = 0; |
| 50 | KJ_IF_SOME(d1, tryFromHexDigit(text[i])) { |
| 51 | b = d1 << 4; |
| 52 | } else { |
| 53 | JSG_REQUIRE(!strict, TypeError, "The text is not valid hex"); |
| 54 | break; |
| 55 | } |
| 56 | KJ_IF_SOME(d2, tryFromHexDigit(text[i + 1])) { |
| 57 | b |= d2; |
| 58 | } else { |
| 59 | JSG_REQUIRE(!strict, TypeError, "The text is not valid hex"); |
| 60 | break; |
| 61 | } |
| 62 | ptr[len++] = b; |
| 63 | } |
| 64 | |
| 65 | if (len == vec.size()) { |
| 66 | return vec; |
| 67 | } |
| 68 | |
| 69 | return vec.slice(js, len); |
| 70 | } |
| 71 | |
| 72 | uint32_t writeInto(jsg::Lock& js, |
| 73 | kj::ArrayPtr<kj::byte> buffer, |
| 74 | jsg::JsString string, |
| 75 | uint32_t offset, |
| 76 | uint32_t length, |
| 77 | Encoding encoding) { |
| 78 | KJ_ASSERT(offset <= buffer.size()); |
| 79 | KJ_ASSERT(length <= buffer.size() - offset); |
| 80 | auto dest = buffer.slice(offset, kj::min(offset + length, buffer.size())); |
| 81 | if (dest.size() == 0 || string.length(js) == 0) { |
| 82 | return 0; |
| 83 | } |
| 84 | |
| 85 | static constexpr jsg::JsString::WriteFlags flags = |
| 86 | jsg::JsString::WriteFlags::REPLACE_INVALID_UTF8; |
| 87 | |
| 88 | switch (encoding) { |
| 89 | case Encoding::ASCII: |
| 90 | // Fall-through |
| 91 | case Encoding::LATIN1: { |
| 92 | auto result = string.writeInto(js, dest, flags); |
| 93 | return result.written; |
| 94 | } |
| 95 | case Encoding::UTF8: { |
| 96 | auto result = string.writeInto(js, dest.asChars(), flags); |
| 97 | return result.written; |
| 98 | } |
| 99 | case Encoding::UTF16LE: { |
| 100 | #if __has_feature(undefined_behavior_sanitizer) |
| 101 | // UBSan warns about unaligned writes, but this can be hard to avoid if dest is unaligned. |
| 102 | // Use temp variable to perform aligned write instead. |
| 103 | kj::Array<uint16_t> tmpBuf = kj::heapArray<uint16_t>(dest.size() / sizeof(uint16_t)); |
| 104 | auto result = string.writeInto(js, tmpBuf, flags); |
| 105 | kj::ArrayPtr<uint16_t> buf(reinterpret_cast<uint16_t*>(dest.begin()), result.written); |
| 106 | buf.copyFrom(tmpBuf.first(result.written)); |
| 107 | #else |
| 108 | kj::ArrayPtr<uint16_t> buf( |
| 109 | reinterpret_cast<uint16_t*>(dest.begin()), dest.size() / sizeof(uint16_t)); |
| 110 | auto result = string.writeInto(js, buf, flags); |
| 111 | #endif |
| 112 | return result.written * sizeof(uint16_t); |
| 113 | } |
| 114 | case Encoding::BASE64: |
| 115 | // Fall-through |
| 116 | case Encoding::BASE64URL: { |
| 117 | auto str = string.toString(js); |
| 118 | return nbytes::Base64Decode(dest.asChars().begin(), dest.size(), str.begin(), str.size()); |
| 119 | } |
| 120 | case Encoding::HEX: { |
| 121 | KJ_STACK_ARRAY(kj::byte, buf, string.length(js), 1024, 536870888); |
| 122 | string.writeInto(js, buf, flags); |
| 123 | auto backing = decodeHexTruncated(js, buf, false); |
| 124 | auto bytes = backing.asArrayPtr(); |
| 125 | auto amountToCopy = kj::min(bytes.size(), dest.size()); |
| 126 | dest.first(amountToCopy).copyFrom(bytes.first(amountToCopy)); |
| 127 | return amountToCopy; |
| 128 | } |
| 129 | default: |
| 130 | KJ_UNREACHABLE; |
| 131 | } |
| 132 | } |
| 133 | |
| 134 | jsg::JsUint8Array decodeStringImpl( |
| 135 | jsg::Lock& js, const jsg::JsString& string, Encoding encoding, bool strict = false) { |
| 136 | auto length = string.length(js); |
| 137 | if (length == 0) { |
| 138 | return jsg::JsUint8Array::create(js, 0); |
| 139 | } |
| 140 | |
| 141 | static constexpr jsg::JsString::WriteFlags options = jsg::JsString::REPLACE_INVALID_UTF8; |
| 142 | |
| 143 | switch (encoding) { |
| 144 | case Encoding::ASCII: |
| 145 | // Fall-through |
| 146 | case Encoding::LATIN1: { |
| 147 | auto dest = jsg::JsUint8Array::create(js, length); |
| 148 | writeInto(js, dest.asArrayPtr(), string, 0, dest.size(), Encoding::LATIN1); |
| 149 | return kj::mv(dest); |
| 150 | } |
| 151 | case Encoding::UTF8: { |
| 152 | auto dest = jsg::JsUint8Array::create(js, string.utf8Length(js)); |
| 153 | writeInto(js, dest.asArrayPtr(), string, 0, dest.size(), Encoding::UTF8); |
| 154 | return kj::mv(dest); |
| 155 | } |
| 156 | case Encoding::UTF16LE: { |
| 157 | auto dest = jsg::JsUint8Array::create(js, length * sizeof(uint16_t)); |
| 158 | writeInto(js, dest.asArrayPtr(), string, 0, dest.size(), Encoding::UTF16LE); |
| 159 | return kj::mv(dest); |
| 160 | } |
| 161 | case Encoding::BASE64: |
| 162 | // Fall-through |
| 163 | case Encoding::BASE64URL: { |
| 164 | // We do not use the kj::String conversion here because inline null-characters |
| 165 | // need to be ignored. |
| 166 | KJ_STACK_ARRAY(kj::byte, buf, length, 1024, 536870888); |
| 167 | auto result = string.writeInto(js, buf, options); |
| 168 | auto len = result.written; |
| 169 | auto dest = jsg::JsUint8Array::create( |
| 170 | js, simdutf::maximal_binary_length_from_base64(buf.asChars().begin(), len)); |
| 171 | auto dec_result = simdutf::base64_to_binary(buf.asChars().begin(), len, |
| 172 | dest.asArrayPtr().asChars().begin(), simdutf::base64_default_or_url_accept_garbage); |
| 173 | if (dec_result.count < dest.size()) { |
| 174 | return dest.slice(js, dec_result.count); |
| 175 | } |
| 176 | return dest; |
| 177 | } |
| 178 | case Encoding::HEX: { |
| 179 | KJ_STACK_ARRAY(kj::byte, buf, length, 1024, 536870888); |
| 180 | string.writeInto(js, buf, options); |
| 181 | return decodeHexTruncated(js, buf, strict); |
| 182 | } |
| 183 | default: |
| 184 | KJ_UNREACHABLE; |
| 185 | } |
| 186 | } |
| 187 | } // namespace |
| 188 | |
| 189 | uint32_t BufferUtil::byteLength(jsg::Lock& js, jsg::JsString str) { |
| 190 | return str.utf8Length(js); |
| 191 | } |
| 192 | |
| 193 | int BufferUtil::compare(jsg::Lock& js, |
| 194 | jsg::JsUint8Array one, |
| 195 | jsg::JsUint8Array two, |
| 196 | jsg::Optional<CompareOptions> maybeOptions) { |
| 197 | kj::ArrayPtr<kj::byte> ptrOne = one.asArrayPtr(); |
| 198 | kj::ArrayPtr<kj::byte> ptrTwo = two.asArrayPtr(); |
| 199 | |
| 200 | // The options allow comparing subranges within the two inputs. |
| 201 | KJ_IF_SOME(options, maybeOptions) { |
| 202 | auto end = options.aEnd.orDefault(ptrOne.size()); |
| 203 | end = kj::min(end, ptrOne.size()); |
| 204 | auto start = kj::min(end, options.aStart.orDefault(0)); |
| 205 | ptrOne = ptrOne.slice(start, end); |
| 206 | end = options.bEnd.orDefault(ptrTwo.size()); |
| 207 | end = kj::min(end, ptrTwo.size()); |
| 208 | start = kj::min(end, options.bStart.orDefault(0)); |
| 209 | ptrTwo = ptrTwo.slice(start, end); |
| 210 | } |
| 211 | |
| 212 | size_t toCompare = kj::min(ptrOne.size(), ptrTwo.size()); |
| 213 | auto result = toCompare > 0 ? memcmp(ptrOne.begin(), ptrTwo.begin(), toCompare) : 0; |
| 214 | |
| 215 | if (result == 0) { |
| 216 | if (ptrOne.size() > ptrTwo.size()) |
| 217 | return 1; |
| 218 | else if (ptrOne.size() < ptrTwo.size()) |
| 219 | return -1; |
| 220 | else |
| 221 | return 0; |
| 222 | } |
| 223 | |
| 224 | return result > 0 ? 1 : -1; |
| 225 | } |
| 226 | |
| 227 | jsg::JsUint8Array BufferUtil::concat( |
| 228 | jsg::Lock& js, kj::Array<jsg::JsUint8Array> list, uint32_t length) { |
| 229 | // The Node.js Buffer.concat is interesting in that it doesn't just append |
| 230 | // the buffers together as is. The length parameter is used to determine the |
| 231 | // length of the result which can be lesser or greater than the actual |
| 232 | // combined lengths of the inputs. If the length is lesser, the result will |
| 233 | // be a truncated version of the combined buffers. If the length is greater, |
| 234 | // the result will be the combined buffers with the remaining space filled |
| 235 | // with zeroes. |
| 236 | |
| 237 | JSG_REQUIRE(length <= v8::ArrayBuffer::kMaxByteLength, RangeError, "The length is too large"); |
| 238 | |
| 239 | auto dest = jsg::JsUint8Array::create(js, length); |
| 240 | if (length > 0) { |
| 241 | auto view = dest.asArrayPtr(); |
| 242 | |
| 243 | for (auto& src: list) { |
| 244 | // JsUint8Array is a view directly on the v8::Uint8Array, so we don't |
| 245 | // really need to worry about whether the underlying ArrayBuffer is |
| 246 | // detached or resized, etc. The length is not cached. |
| 247 | auto ptr = src.asArrayPtr(); |
| 248 | if (ptr.size() == 0) continue; |
| 249 | // The amount to copy is the lesser of the remaining space in the destination or |
| 250 | // the size of the chunk we're copying. |
| 251 | auto amountToCopy = kj::min(ptr.size(), view.size()); |
| 252 | view.first(amountToCopy).copyFrom(ptr.first(amountToCopy)); |
| 253 | view = view.slice(amountToCopy); |
| 254 | // If there's no more space in the destination, we're done. |
| 255 | if (view == nullptr) { |
| 256 | break; |
| 257 | } |
| 258 | } |
| 259 | } |
| 260 | |
| 261 | return dest; |
| 262 | } |
| 263 | |
| 264 | jsg::JsUint8Array BufferUtil::decodeString( |
| 265 | jsg::Lock& js, jsg::JsString string, EncodingValue encoding) { |
| 266 | return decodeStringImpl(js, string, static_cast<Encoding>(encoding)); |
| 267 | } |
| 268 | |
| 269 | void BufferUtil::fillImpl(jsg::Lock& js, |
| 270 | jsg::JsUint8Array buffer, |
| 271 | kj::OneOf<jsg::JsString, jsg::JsUint8Array> value, |
| 272 | uint32_t start, |
| 273 | uint32_t end, |
| 274 | jsg::Optional<EncodingValue> encoding) { |
| 275 | end = kj::min(end, buffer.size()); |
| 276 | if (end <= start) return; |
| 277 | |
| 278 | auto ptr = buffer.asArrayPtr().slice(start, end); |
| 279 | KJ_SWITCH_ONEOF(value) { |
| 280 | KJ_CASE_ONEOF(string, jsg::JsString) { |
| 281 | auto enc = encoding.orDefault(Encoding::UTF8); |
| 282 | auto decoded = decodeStringImpl(js, string, static_cast<Encoding>(enc), true /* strict */); |
| 283 | if (decoded.size() == 0) { |
| 284 | ptr.fill(0); |
| 285 | return; |
| 286 | } |
| 287 | ptr.fill(decoded.asArrayPtr()); |
| 288 | } |
| 289 | KJ_CASE_ONEOF(source, jsg::JsUint8Array) { |
| 290 | if (source.size() == 0) { |
| 291 | ptr.fill(0); |
| 292 | return; |
| 293 | } |
| 294 | ptr.fill(source.asArrayPtr()); |
| 295 | } |
| 296 | } |
| 297 | } |
| 298 | |
| 299 | namespace { |
| 300 | |
| 301 | // Computes the offset for starting an indexOf or lastIndexOf search. |
| 302 | // Returns either a valid offset in [0...<length - 1>], ie inside the Buffer, |
| 303 | // or -1 to signal that there is no possible match. |
| 304 | int32_t indexOfOffset(size_t length, int32_t offset, int32_t needle_length, bool isForward) { |
| 305 | int32_t len = static_cast<int32_t>(length); |
| 306 | if (offset < 0) { |
| 307 | if (offset + len >= 0) { |
| 308 | // Negative offsets count backwards from the end of the buffer. |
| 309 | return len + offset; |
| 310 | } else if (isForward || needle_length == 0) { |
| 311 | // indexOf from before the start of the buffer: search the whole buffer. |
| 312 | return 0; |
| 313 | } else { |
| 314 | // lastIndexOf from before the start of the buffer: no match. |
| 315 | return -1; |
| 316 | } |
| 317 | } else { |
| 318 | // cast to int64_t to avoid overflow. |
| 319 | if (static_cast<int64_t>(offset) + needle_length <= len) { |
| 320 | // Valid positive offset. |
| 321 | return offset; |
| 322 | } else if (needle_length == 0) { |
| 323 | // Out of buffer bounds, but empty needle: point to end of buffer. |
| 324 | return len; |
| 325 | } else if (isForward) { |
| 326 | // indexOf from past the end of the buffer: no match. |
| 327 | return -1; |
| 328 | } else { |
| 329 | // lastIndexOf from past the end of the buffer: search the whole buffer. |
| 330 | return len - 1; |
| 331 | } |
| 332 | } |
| 333 | } |
| 334 | |
| 335 | jsg::Optional<uint32_t> indexOfBuffer(jsg::Lock& js, |
| 336 | kj::ArrayPtr<kj::byte> hayStack, |
| 337 | jsg::JsUint8Array needle, |
| 338 | int32_t byteOffset, |
| 339 | EncodingValue encoding, |
| 340 | bool isForward) { |
| 341 | auto enc = static_cast<Encoding>(encoding); |
| 342 | // Round down to the nearest multiple of 2 in case of UCS2. |
| 343 | auto hayStackLength = enc == Encoding::UTF16LE ? hayStack.size() & ~1 : hayStack.size(); |
| 344 | auto optOffset = indexOfOffset(hayStackLength, byteOffset, needle.size(), isForward); |
| 345 | |
| 346 | if (needle.size() == 0) return optOffset; |
| 347 | if (hayStackLength == 0 || optOffset <= -1 || |
| 348 | (isForward && needle.size() + optOffset > hayStackLength) || needle.size() > hayStackLength) { |
| 349 | return kj::none; |
| 350 | } |
| 351 | auto result = hayStackLength; |
| 352 | if (enc == Encoding::UTF16LE) { |
| 353 | if (hayStackLength < 2 || needle.size() < 2) { |
| 354 | return kj::none; |
| 355 | } |
| 356 | // Copy haystack and needle to aligned buffers to avoid undefined behavior |
| 357 | // from unaligned uint16_t access (the data pointer may have an odd byte offset). |
| 358 | auto hayStackU16Len = hayStackLength / 2; |
| 359 | kj::SmallArray<uint16_t, 1024> alignedHayStack(hayStackU16Len); |
| 360 | alignedHayStack.asBytes().copyFrom(hayStack.first(hayStackLength)); |
| 361 | |
| 362 | auto needleLen = needle.size() & ~1; |
| 363 | auto needleU16Len = needleLen / 2; |
| 364 | kj::SmallArray<uint16_t, 1024> alignedNeedle(needleU16Len); |
| 365 | alignedNeedle.asBytes().copyFrom(needle.asArrayPtr().first(needleLen)); |
| 366 | |
| 367 | result = SearchString(alignedHayStack.begin(), hayStackU16Len, alignedNeedle.begin(), |
| 368 | needleU16Len, optOffset / 2, isForward); |
| 369 | result *= 2; |
| 370 | } else { |
| 371 | result = SearchString(hayStack.asBytes().begin(), hayStack.size(), |
| 372 | needle.asArrayPtr().asBytes().begin(), needle.size(), optOffset, isForward); |
| 373 | } |
| 374 | |
| 375 | if (result == hayStackLength) return kj::none; |
| 376 | |
| 377 | return result; |
| 378 | } |
| 379 | |
| 380 | jsg::Optional<uint32_t> indexOfString(jsg::Lock& js, |
| 381 | kj::ArrayPtr<kj::byte> hayStack, |
| 382 | const jsg::JsString& needle, |
| 383 | int32_t byteOffset, |
| 384 | EncodingValue encoding, |
| 385 | bool isForward) { |
| 386 | |
| 387 | auto enc = static_cast<Encoding>(encoding); |
| 388 | auto decodedNeedle = decodeStringImpl(js, needle, enc); |
| 389 | |
| 390 | // Round down to the nearest multiple of 2 in case of UCS2 |
| 391 | auto hayStackLength = enc == Encoding::UTF16LE ? hayStack.size() & ~1 : hayStack.size(); |
| 392 | auto optOffset = indexOfOffset(hayStackLength, byteOffset, decodedNeedle.size(), isForward); |
| 393 | |
| 394 | if (decodedNeedle.size() == 0) { |
| 395 | return optOffset; |
| 396 | } |
| 397 | |
| 398 | if (hayStackLength == 0 || optOffset <= -1 || |
| 399 | (isForward && decodedNeedle.size() + optOffset > hayStackLength) || |
| 400 | decodedNeedle.size() > hayStackLength) { |
| 401 | return kj::none; |
| 402 | } |
| 403 | |
| 404 | auto result = hayStackLength; |
| 405 | |
| 406 | if (enc == Encoding::UTF16LE) { |
| 407 | if (hayStackLength < 2 || decodedNeedle.size() < 2) { |
| 408 | return kj::none; |
| 409 | } |
| 410 | // Copy haystack to an aligned buffer to avoid undefined behavior from |
| 411 | // unaligned uint16_t access (the data pointer may have an odd byte offset). |
| 412 | auto hayStackU16Len = hayStackLength / 2; |
| 413 | kj::SmallArray<uint16_t, 1024> alignedHayStack(hayStackU16Len); |
| 414 | alignedHayStack.asBytes().copyFrom(hayStack.first(hayStackLength)); |
| 415 | |
| 416 | result = SearchString(alignedHayStack.begin(), hayStackU16Len, |
| 417 | reinterpret_cast<const uint16_t*>(decodedNeedle.asArrayPtr<char>().begin()), |
| 418 | decodedNeedle.size() / 2, optOffset / 2, isForward); |
| 419 | result *= 2; |
| 420 | } else { |
| 421 | result = SearchString(hayStack.asBytes().begin(), hayStack.size(), |
| 422 | decodedNeedle.asArrayPtr().begin(), decodedNeedle.size(), optOffset, isForward); |
| 423 | } |
| 424 | |
| 425 | if (result == hayStackLength) return kj::none; |
| 426 | |
| 427 | return result; |
| 428 | } |
| 429 | |
| 430 | jsg::JsString toStringImpl( |
| 431 | jsg::Lock& js, kj::ArrayPtr<kj::byte> bytes, uint32_t start, uint32_t end, Encoding encoding) { |
| 432 | KJ_ASSERT(end <= bytes.size()); |
| 433 | if (end < start) end = start; |
| 434 | auto slice = bytes.slice(start, end); |
| 435 | if (slice.size() == 0) return js.str(); |
| 436 | switch (encoding) { |
| 437 | case Encoding::ASCII: { |
| 438 | // TODO(perf): We can look at making this more performant later. |
| 439 | // Essentially we have to modify the buffer such that every byte |
| 440 | // has the highest bit turned off. Whee! Node.js has a faster |
| 441 | // algorithm that it implements so we can likely adopt that. |
| 442 | kj::Array<kj::byte> copy = KJ_MAP(b, slice) -> kj::byte { return b & 0x7f; }; |
| 443 | return js.str(copy); |
| 444 | } |
| 445 | case Encoding::LATIN1: { |
| 446 | return js.str(slice); |
| 447 | } |
| 448 | case Encoding::UTF8: { |
| 449 | return js.str(slice.asChars()); |
| 450 | } |
| 451 | case Encoding::UTF16LE: { |
| 452 | // Why are we copying here? Good question! There's really not much of a |
| 453 | // good reason why we should be copying here except that V8 doesn't seem |
| 454 | // to like it! We end up with an asan buffer over-read error if we pass |
| 455 | // in the slice directly... looks to have something to do with alignment |
| 456 | // issues. Copying here sidesteps the problem and avoids the asan issue. |
| 457 | kj::ArrayPtr<uint16_t> view(reinterpret_cast<uint16_t*>(slice.begin()), slice.size() / 2); |
| 458 | KJ_STACK_ARRAY(uint16_t, data, view.size(), 1024, 4096); |
| 459 | data.copyFrom(view); |
| 460 | return js.str(data); |
| 461 | } |
| 462 | case Encoding::BASE64: { |
| 463 | size_t length = simdutf::base64_length_from_binary(slice.size()); |
| 464 | KJ_STACK_ARRAY(kj::byte, out, length, 1024, 4096); |
| 465 | simdutf::binary_to_base64(reinterpret_cast<const char*>(slice.begin()), slice.size(), |
| 466 | reinterpret_cast<char*>(out.begin())); |
| 467 | return js.str(out); |
| 468 | } |
| 469 | case Encoding::BASE64URL: { |
| 470 | auto options = simdutf::base64_url; |
| 471 | size_t length = simdutf::base64_length_from_binary(slice.size(), options); |
| 472 | KJ_STACK_ARRAY(kj::byte, out, length, 1024, 4096); |
| 473 | simdutf::binary_to_base64(reinterpret_cast<const char*>(slice.begin()), slice.size(), |
| 474 | reinterpret_cast<char*>(out.begin()), options); |
| 475 | return js.str(out); |
| 476 | } |
| 477 | case Encoding::HEX: { |
| 478 | return js.str(kj::encodeHex(slice)); |
| 479 | } |
| 480 | default: |
| 481 | KJ_UNREACHABLE; |
| 482 | } |
| 483 | } |
| 484 | |
| 485 | } // namespace |
| 486 | |
| 487 | jsg::Optional<uint32_t> BufferUtil::indexOf(jsg::Lock& js, |
| 488 | jsg::JsUint8Array buffer, |
| 489 | kj::OneOf<jsg::JsString, jsg::JsUint8Array> value, |
| 490 | int32_t byteOffset, |
| 491 | EncodingValue encoding, |
| 492 | bool isForward) { |
| 493 | |
| 494 | KJ_SWITCH_ONEOF(value) { |
| 495 | KJ_CASE_ONEOF(string, jsg::JsString) { |
| 496 | return indexOfString(js, buffer.asArrayPtr(), string, byteOffset, encoding, isForward); |
| 497 | } |
| 498 | KJ_CASE_ONEOF(source, jsg::JsUint8Array) { |
| 499 | return indexOfBuffer( |
| 500 | js, buffer.asArrayPtr(), kj::mv(source), byteOffset, encoding, isForward); |
| 501 | } |
| 502 | } |
| 503 | KJ_UNREACHABLE; |
| 504 | } |
| 505 | |
| 506 | void BufferUtil::swap(jsg::Lock& js, jsg::JsUint8Array buffer, int size) { |
| 507 | if (buffer.size() <= 1) return; |
| 508 | switch (size) { |
| 509 | case 16: { |
| 510 | JSG_REQUIRE(nbytes::SwapBytes16(buffer.asArrayPtr().asChars().begin(), buffer.size()), Error, |
| 511 | "Swap bytes failed"); |
| 512 | break; |
| 513 | } |
| 514 | case 32: { |
| 515 | JSG_REQUIRE(nbytes::SwapBytes32(buffer.asArrayPtr().asChars().begin(), buffer.size()), Error, |
| 516 | "Swap bytes failed"); |
| 517 | break; |
| 518 | } |
| 519 | case 64: { |
| 520 | JSG_REQUIRE(nbytes::SwapBytes64(buffer.asArrayPtr().asChars().begin(), buffer.size()), Error, |
| 521 | "Swap bytes failed"); |
| 522 | break; |
| 523 | } |
| 524 | default: |
| 525 | JSG_FAIL_REQUIRE(Error, "Unreachable"); |
| 526 | } |
| 527 | } |
| 528 | |
| 529 | jsg::JsString BufferUtil::toString( |
| 530 | jsg::Lock& js, jsg::JsUint8Array bytes, uint32_t start, uint32_t end, EncodingValue encoding) { |
| 531 | end = kj::min(bytes.size(), end); |
| 532 | if (end <= start) return js.str(); |
| 533 | return toStringImpl(js, bytes.asArrayPtr(), start, end, static_cast<Encoding>(encoding)); |
| 534 | } |
| 535 | |
| 536 | uint32_t BufferUtil::write(jsg::Lock& js, |
| 537 | jsg::JsUint8Array buffer, |
| 538 | jsg::JsString string, |
| 539 | uint32_t offset, |
| 540 | uint32_t length, |
| 541 | EncodingValue encoding) { |
| 542 | length = kj::min(length, buffer.size() - offset); |
| 543 | if (length == 0) return 0; |
| 544 | return writeInto( |
| 545 | js, buffer.asArrayPtr(), string, offset, length, static_cast<Encoding>(encoding)); |
| 546 | } |
| 547 | |
| 548 | // ====================================================================================== |
| 549 | // StringDecoder |
| 550 | // |
| 551 | // It's helpful to review a bit about how the implementation works here. |
| 552 | // |
| 553 | // StringDecoder is a streaming decoder that ensures that multi-byte characters are correctly |
| 554 | // handled. So, for instance, let's suppose I have the utf8 bytes for a euro symbol (0xe2, 0x82, |
| 555 | // 0xac), but I only get those one at a time... StringDecoder will ensure that those are correctly |
| 556 | // handled over multiple calls to write(...)... |
| 557 | // |
| 558 | // const sd = new StringDecoder(); |
| 559 | // let results = ''; |
| 560 | // results += sd.write(new Uint8Array([0xe2])); // results.length === 0 |
| 561 | // results += sd.write(new Uint8Array([0x82])); // results.length === 0 |
| 562 | // results += sd.write(new Uint8Array([0xac])); // results.length === 1 |
| 563 | // results += sd.end(); |
| 564 | // |
| 565 | // Internally, the decoder allocates a small 7 byte buffer (the state) argument below. |
| 566 | // |
| 567 | // The first four bytes of the state are used to hold partial bytes received on the previous |
| 568 | // write. The fifth byte in state is a count of the number of missing bytes we need to complete |
| 569 | // the character. The sixth byte in state is the number of bytes that have been encoded into the |
| 570 | // first four. The seventh byte in state identifies the Encoding and matches the values of the |
| 571 | // Encoding enum. |
| 572 | // |
| 573 | // So, in our example above, initially the first six bytes of the state are [0x00, 0x00, 0x00, |
| 574 | // 0x00, 0x00, 0x00] |
| 575 | // |
| 576 | // After the first call to write above, the state is updated to: [0xe2, 0x00, 0x00, 0x00, 0x02, |
| 577 | // 0x01] |
| 578 | // |
| 579 | // After the second call to write, the state is updated to: [0xe2, 0x82, 0x00, 0x00, 0x01, 0x02] |
| 580 | // |
| 581 | // After the third call to write, the pending multibyte character is completed, the state becomes: |
| 582 | // [0xe2, 0x82, 0xac, 0x00, 0x00, 0x00] ... while the bytes are still in state, the buffered bytes |
| 583 | // and bytes needed are zeroed out. Since the character is completed on that third write, it is |
| 584 | // included in the returned string. |
| 585 | // |
| 586 | // The implementation here is taken nearly verbatim from Node.js with a few adaptations. The code |
| 587 | // from Node.js has remained largely unchanged for years and is well-proven. |
| 588 | |
| 589 | namespace { |
| 590 | inline kj::byte getMissingBytes(kj::ArrayPtr<kj::byte> state) { |
| 591 | JSG_REQUIRE(state[BufferUtil::kMissingBytes] <= BufferUtil::kIncompleteCharactersEnd, Error, |
| 592 | "Missing bytes cannot exceed 4"); |
| 593 | return state[BufferUtil::kMissingBytes]; |
| 594 | } |
| 595 | |
| 596 | inline kj::byte getBufferedBytes(kj::ArrayPtr<kj::byte> state) { |
| 597 | JSG_REQUIRE(state[BufferUtil::kBufferedBytes] <= BufferUtil::kIncompleteCharactersEnd, Error, |
| 598 | "Buffered bytes cannot exceed 4"); |
| 599 | return state[BufferUtil::kBufferedBytes]; |
| 600 | } |
| 601 | |
| 602 | inline kj::byte* getIncompleteCharacterBuffer(kj::ArrayPtr<kj::byte> state) { |
| 603 | return state.begin() + BufferUtil::kIncompleteCharactersStart; |
| 604 | } |
| 605 | |
| 606 | inline Encoding getEncoding(kj::ArrayPtr<kj::byte> state) { |
| 607 | JSG_REQUIRE(state[BufferUtil::kEncoding] <= static_cast<kj::byte>(Encoding::HEX), Error, |
| 608 | "Invalid StringDecoder state"); |
| 609 | return static_cast<Encoding>(state[BufferUtil::kEncoding]); |
| 610 | } |
| 611 | |
| 612 | jsg::JsString getBufferedString(jsg::Lock& js, kj::ArrayPtr<kj::byte> state) { |
| 613 | JSG_REQUIRE(getBufferedBytes(state) <= BufferUtil::kIncompleteCharactersEnd, Error, |
| 614 | "Invalid StringDecoder state"); |
| 615 | auto ret = toStringImpl(js, state, BufferUtil::kIncompleteCharactersStart, |
| 616 | BufferUtil::kIncompleteCharactersStart + getBufferedBytes(state), getEncoding(state)); |
| 617 | state[BufferUtil::kBufferedBytes] = 0; |
| 618 | return ret; |
| 619 | } |
| 620 | } // namespace |
| 621 | |
| 622 | jsg::JsString BufferUtil::decode(jsg::Lock& js, jsg::JsUint8Array bytes, jsg::JsUint8Array state) { |
| 623 | JSG_REQUIRE(state.size() == BufferUtil::kSize, TypeError, "Invalid StringDecoder"); |
| 624 | auto bytesPtr = bytes.asArrayPtr(); |
| 625 | auto statePtr = state.asArrayPtr(); |
| 626 | auto enc = getEncoding(statePtr); |
| 627 | if (enc == Encoding::ASCII || enc == Encoding::LATIN1 || enc == Encoding::HEX) { |
| 628 | // For ascii, latin1, and hex, we can just use the regular |
| 629 | // toString option since there will never be a case where |
| 630 | // these have left-over characters. |
| 631 | return toStringImpl(js, bytesPtr, 0, bytesPtr.size(), enc); |
| 632 | } |
| 633 | |
| 634 | jsg::JsString prepend = js.str(); |
| 635 | jsg::JsString body = js.str(); |
| 636 | auto nread = bytesPtr.size(); |
| 637 | |
| 638 | // If bytes is empty there's nothing to decode. |
| 639 | if (bytesPtr.size() == 0) return js.str(); |
| 640 | |
| 641 | auto data = bytesPtr.begin(); |
| 642 | |
| 643 | if (getMissingBytes(statePtr) > 0) { |
| 644 | JSG_REQUIRE(getMissingBytes(statePtr) + getBufferedBytes(statePtr) <= |
| 645 | BufferUtil::kIncompleteCharactersEnd, |
| 646 | Error, "Invalid StringDecoder state"); |
| 647 | if (enc == Encoding::UTF8) { |
| 648 | // For UTF-8, we need special treatment to align with the V8 decoder: |
| 649 | // If an incomplete character is found at a chunk boundary, we use |
| 650 | // its remainder and pass it to V8 as-is. |
| 651 | for (size_t i = 0; i < nread && i < getMissingBytes(statePtr); ++i) { |
| 652 | if ((data[i] & 0xC0) != 0x80) { |
| 653 | // This byte is not a continuation byte even though it should have |
| 654 | // been one. We stop decoding of the incomplete character at this |
| 655 | // point (but still use the rest of the incomplete bytes from this |
| 656 | // chunk) and assume that the new, unexpected byte starts a new one. |
| 657 | statePtr[kMissingBytes] = 0; |
| 658 | // Use memmove: data may point into the incomplete character buffer |
| 659 | // (e.g. when the caller passes decoder.lastChar as input). |
| 660 | memmove(getIncompleteCharacterBuffer(statePtr) + getBufferedBytes(statePtr), data, i); |
| 661 | statePtr[kBufferedBytes] += i; |
| 662 | data += i; |
| 663 | nread -= i; |
| 664 | break; |
| 665 | } |
| 666 | } |
| 667 | } |
| 668 | |
| 669 | size_t found_bytes = std::min(nread, static_cast<size_t>(getMissingBytes(statePtr))); |
| 670 | KJ_ASSERT(data != nullptr); |
| 671 | // Use memmove: data may point into the incomplete character buffer |
| 672 | // (e.g. when the caller passes decoder.lastChar as input). |
| 673 | memmove(getIncompleteCharacterBuffer(statePtr) + getBufferedBytes(statePtr), data, found_bytes); |
| 674 | // Adjust the two buffers. |
| 675 | data += found_bytes; |
| 676 | nread -= found_bytes; |
| 677 | |
| 678 | statePtr[kMissingBytes] -= found_bytes; |
| 679 | statePtr[kBufferedBytes] += found_bytes; |
| 680 | |
| 681 | if (getMissingBytes(statePtr) == 0) { |
| 682 | // If no more bytes are missing, create a small string that we will later prepend. |
| 683 | prepend = getBufferedString(js, statePtr); |
| 684 | } |
| 685 | } |
| 686 | |
| 687 | if (nread == 0) { |
| 688 | body = prepend.length(js) ? prepend : js.str(); |
| 689 | prepend = js.str(); |
| 690 | } else { |
| 691 | JSG_REQUIRE(getMissingBytes(statePtr) == 0, Error, "Invalid StringDecoder state"); |
| 692 | JSG_REQUIRE(getBufferedBytes(statePtr) == 0, Error, "Invalid StringDecoder state"); |
| 693 | |
| 694 | // See whether there is a character that we may have to cut off and |
| 695 | // finish when receiving the next chunk. |
| 696 | if (enc == Encoding::UTF8 && data[nread - 1] & 0x80) { |
| 697 | // This is UTF-8 encoded data and we ended on a non-ASCII UTF-8 byte. |
| 698 | // This means we'll need to figure out where the character to which |
| 699 | // the byte belongs begins. |
| 700 | for (size_t i = nread - 1;; --i) { |
| 701 | JSG_REQUIRE(i < nread, Error, "Invalid StringDecoder state"); |
| 702 | statePtr[kBufferedBytes]++; |
| 703 | if ((data[i] & 0xC0) == 0x80) { |
| 704 | // This byte does not start a character (a "trailing" byte). |
| 705 | if (statePtr[kBufferedBytes] >= 4 || i == 0) { |
| 706 | // We either have more then 4 trailing bytes (which means |
| 707 | // the current character would not be inside the range for |
| 708 | // valid Unicode, and in particular cannot be represented |
| 709 | // through JavaScript's UTF-16-based approach to strings), or the |
| 710 | // current buffer does not contain the start of an UTF-8 character |
| 711 | // at all. Either way, this is invalid UTF8 and we can just |
| 712 | // let the engine's decoder handle it. |
| 713 | statePtr[kBufferedBytes] = 0; |
| 714 | break; |
| 715 | } |
| 716 | } else { |
| 717 | // Found the first byte of a UTF-8 character. By looking at the |
| 718 | // upper bits we can tell how long the character *should* be. |
| 719 | if ((data[i] & 0xE0) == 0xC0) { |
| 720 | statePtr[kMissingBytes] = 2; |
| 721 | } else if ((data[i] & 0xF0) == 0xE0) { |
| 722 | statePtr[kMissingBytes] = 3; |
| 723 | } else if ((data[i] & 0xF8) == 0xF0) { |
| 724 | statePtr[kMissingBytes] = 4; |
| 725 | } else { |
| 726 | // This lead byte would indicate a character outside of the |
| 727 | // representable range. |
| 728 | statePtr[kBufferedBytes] = 0; |
| 729 | break; |
| 730 | } |
| 731 | |
| 732 | if (getBufferedBytes(statePtr) >= getMissingBytes(statePtr)) { |
| 733 | // Received more or exactly as many trailing bytes than the lead |
| 734 | // character would indicate. In the "==" case, we have valid |
| 735 | // data and don't need to slice anything off; |
| 736 | // in the ">" case, this is invalid UTF-8 anyway. |
| 737 | statePtr[kMissingBytes] = 0; |
| 738 | statePtr[kBufferedBytes] = 0; |
| 739 | } |
| 740 | |
| 741 | statePtr[kMissingBytes] -= statePtr[kBufferedBytes]; |
| 742 | break; |
| 743 | } |
| 744 | } |
| 745 | } else if (enc == Encoding::UTF16LE) { |
| 746 | if ((nread % 2) == 1) { |
| 747 | // WePtr got half a codepoint, and need the second byte of it. |
| 748 | statePtr[kBufferedBytes] = 1; |
| 749 | statePtr[kMissingBytes] = 1; |
| 750 | } else if ((data[nread - 1] & 0xFC) == 0xD8) { |
| 751 | // Half a split UTF-16 character. |
| 752 | statePtr[kBufferedBytes] = 2; |
| 753 | statePtr[kMissingBytes] = 2; |
| 754 | } |
| 755 | } else if (enc == Encoding::BASE64 || enc == Encoding::BASE64URL) { |
| 756 | statePtr[kBufferedBytes] = nread % 3; |
| 757 | if (statePtr[kBufferedBytes] > 0) statePtr[kMissingBytes] = 3 - getBufferedBytes(statePtr); |
| 758 | } |
| 759 | |
| 760 | if (getBufferedBytes(statePtr) > 0) { |
| 761 | // Copy the requested number of buffered bytes from the end of the |
| 762 | // input into the incomplete character buffer. |
| 763 | nread -= getBufferedBytes(statePtr); |
| 764 | // Use memmove: data may point into the incomplete character buffer |
| 765 | // (e.g. when the caller passes decoder.lastChar as input). |
| 766 | memmove(getIncompleteCharacterBuffer(statePtr), data + nread, getBufferedBytes(statePtr)); |
| 767 | } |
| 768 | |
| 769 | if (nread > 0) { |
| 770 | body = toStringImpl(js, kj::ArrayPtr<kj::byte>(data, data + nread), 0, nread, enc); |
| 771 | } else { |
| 772 | body = js.str(); |
| 773 | } |
| 774 | } |
| 775 | |
| 776 | if (prepend.length(js) == 0) { |
| 777 | return body; |
| 778 | } else { |
| 779 | return jsg::JsString::concat(js, prepend, body); |
| 780 | } |
| 781 | |
| 782 | return js.str(); |
| 783 | } |
| 784 | |
| 785 | jsg::JsString BufferUtil::flush(jsg::Lock& js, jsg::JsUint8Array state) { |
| 786 | JSG_REQUIRE(state.size() == BufferUtil::kSize, TypeError, "Invalid StringDecoder"); |
| 787 | auto statePtr = state.asArrayPtr(); |
| 788 | auto enc = getEncoding(statePtr); |
| 789 | if (enc == Encoding::ASCII || enc == Encoding::HEX || enc == Encoding::LATIN1) { |
| 790 | JSG_REQUIRE(getMissingBytes(statePtr) == 0, Error, "Invalid StringDecoder state"); |
| 791 | JSG_REQUIRE(getBufferedBytes(statePtr) == 0, Error, "Invalid StringDecoder state"); |
| 792 | } |
| 793 | |
| 794 | if (enc == Encoding::UTF16LE && getBufferedBytes(statePtr) % 2 == 1) { |
| 795 | // Ignore a single trailing byte, like the JS decoder does. |
| 796 | statePtr[kMissingBytes]--; |
| 797 | statePtr[kBufferedBytes]--; |
| 798 | } |
| 799 | |
| 800 | if (getBufferedBytes(statePtr) == 0) { |
| 801 | return js.str(); |
| 802 | } |
| 803 | |
| 804 | auto ret = getBufferedString(js, statePtr); |
| 805 | statePtr[kMissingBytes] = 0; |
| 806 | |
| 807 | return ret; |
| 808 | } |
| 809 | |
| 810 | bool BufferUtil::isAscii(jsg::JsUint8Array buffer) { |
| 811 | if (buffer.size() == 0) return true; |
| 812 | return simdutf::validate_ascii(buffer.asArrayPtr().asChars().begin(), buffer.size()); |
| 813 | } |
| 814 | |
| 815 | bool BufferUtil::isUtf8(jsg::JsUint8Array buffer) { |
| 816 | if (buffer.size() == 0) return true; |
| 817 | return simdutf::validate_utf8(buffer.asArrayPtr().asChars().begin(), buffer.size()); |
| 818 | } |
| 819 | |
| 820 | jsg::JsUint8Array BufferUtil::transcode(jsg::Lock& js, |
| 821 | jsg::JsUint8Array source, |
| 822 | EncodingValue rawFromEncoding, |
| 823 | EncodingValue rawToEncoding) { |
| 824 | auto fromEncoding = static_cast<Encoding>(rawFromEncoding); |
| 825 | auto toEncoding = static_cast<Encoding>(rawToEncoding); |
| 826 | |
| 827 | JSG_REQUIRE(i18n::canBeTranscoded(fromEncoding) && i18n::canBeTranscoded(toEncoding), Error, |
| 828 | "Unable to transcode buffer due to unsupported encoding"); |
| 829 | |
| 830 | return i18n::transcode(js, source.asArrayPtr(), fromEncoding, toEncoding); |
| 831 | } |
| 832 | |
| 833 | } // namespace workerd::api::node |