Skip to content
File

Blob: src/workerd/api/node/buffer.c++

30.5 KB
1// Copyright (c) 2017-2022 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4// Copyright Joyent and Node contributors. All rights reserved. MIT license.
5 
6#include "buffer.h"
7 
8#include "buffer-string-search.h"
9#include "nbytes.h"
10#include "simdutf.h"
11 
12#include <workerd/jsg/jsg.h>
13 
14#include <kj/array.h>
15#include <kj/encoding.h>
16 
17#include <algorithm>
18 
19namespace workerd::api::node {
20 
21namespace {
22 
23kj::Maybe<uint> tryFromHexDigit(char c) {
24 if ('0' <= c && c <= '9') {
25 return c - '0';
26 } else if ('a' <= c && c <= 'f') {
27 return c - ('a' - 10);
28 } else if ('A' <= c && c <= 'F') {
29 return c - ('A' - 10);
30 }
31 return kj::none;
32}
33 
34jsg::JsUint8Array decodeHexTruncated(
35 jsg::Lock& js, kj::ArrayPtr<kj::byte> text, bool strict = false) {
36 // We do not use kj::decodeHex because we need to match Node.js'
37 // behavior of truncating the response at the first invalid hex
38 // pair as opposed to just marking that an error happened and
39 // trying to continue with the decode.
40 if (text.size() % 2 != 0) {
41 JSG_REQUIRE(!strict, TypeError, "The text is not valid hex");
42 text = text.first(text.size() - 1);
43 }
44 auto vec = jsg::JsUint8Array::create(js, text.size() / 2);
45 auto ptr = vec.asArrayPtr();
46 size_t len = 0;
47 
48 for (size_t i = 0; i < text.size(); i += 2) {
49 kj::byte b = 0;
50 KJ_IF_SOME(d1, tryFromHexDigit(text[i])) {
51 b = d1 << 4;
52 } else {
53 JSG_REQUIRE(!strict, TypeError, "The text is not valid hex");
54 break;
55 }
56 KJ_IF_SOME(d2, tryFromHexDigit(text[i + 1])) {
57 b |= d2;
58 } else {
59 JSG_REQUIRE(!strict, TypeError, "The text is not valid hex");
60 break;
61 }
62 ptr[len++] = b;
63 }
64 
65 if (len == vec.size()) {
66 return vec;
67 }
68 
69 return vec.slice(js, len);
70}
71 
72uint32_t writeInto(jsg::Lock& js,
73 kj::ArrayPtr<kj::byte> buffer,
74 jsg::JsString string,
75 uint32_t offset,
76 uint32_t length,
77 Encoding encoding) {
78 KJ_ASSERT(offset <= buffer.size());
79 KJ_ASSERT(length <= buffer.size() - offset);
80 auto dest = buffer.slice(offset, kj::min(offset + length, buffer.size()));
81 if (dest.size() == 0 || string.length(js) == 0) {
82 return 0;
83 }
84 
85 static constexpr jsg::JsString::WriteFlags flags =
86 jsg::JsString::WriteFlags::REPLACE_INVALID_UTF8;
87 
88 switch (encoding) {
89 case Encoding::ASCII:
90 // Fall-through
91 case Encoding::LATIN1: {
92 auto result = string.writeInto(js, dest, flags);
93 return result.written;
94 }
95 case Encoding::UTF8: {
96 auto result = string.writeInto(js, dest.asChars(), flags);
97 return result.written;
98 }
99 case Encoding::UTF16LE: {
100#if __has_feature(undefined_behavior_sanitizer)
101 // UBSan warns about unaligned writes, but this can be hard to avoid if dest is unaligned.
102 // Use temp variable to perform aligned write instead.
103 kj::Array<uint16_t> tmpBuf = kj::heapArray<uint16_t>(dest.size() / sizeof(uint16_t));
104 auto result = string.writeInto(js, tmpBuf, flags);
105 kj::ArrayPtr<uint16_t> buf(reinterpret_cast<uint16_t*>(dest.begin()), result.written);
106 buf.copyFrom(tmpBuf.first(result.written));
107#else
108 kj::ArrayPtr<uint16_t> buf(
109 reinterpret_cast<uint16_t*>(dest.begin()), dest.size() / sizeof(uint16_t));
110 auto result = string.writeInto(js, buf, flags);
111#endif
112 return result.written * sizeof(uint16_t);
113 }
114 case Encoding::BASE64:
115 // Fall-through
116 case Encoding::BASE64URL: {
117 auto str = string.toString(js);
118 return nbytes::Base64Decode(dest.asChars().begin(), dest.size(), str.begin(), str.size());
119 }
120 case Encoding::HEX: {
121 KJ_STACK_ARRAY(kj::byte, buf, string.length(js), 1024, 536870888);
122 string.writeInto(js, buf, flags);
123 auto backing = decodeHexTruncated(js, buf, false);
124 auto bytes = backing.asArrayPtr();
125 auto amountToCopy = kj::min(bytes.size(), dest.size());
126 dest.first(amountToCopy).copyFrom(bytes.first(amountToCopy));
127 return amountToCopy;
128 }
129 default:
130 KJ_UNREACHABLE;
131 }
132}
133 
134jsg::JsUint8Array decodeStringImpl(
135 jsg::Lock& js, const jsg::JsString& string, Encoding encoding, bool strict = false) {
136 auto length = string.length(js);
137 if (length == 0) {
138 return jsg::JsUint8Array::create(js, 0);
139 }
140 
141 static constexpr jsg::JsString::WriteFlags options = jsg::JsString::REPLACE_INVALID_UTF8;
142 
143 switch (encoding) {
144 case Encoding::ASCII:
145 // Fall-through
146 case Encoding::LATIN1: {
147 auto dest = jsg::JsUint8Array::create(js, length);
148 writeInto(js, dest.asArrayPtr(), string, 0, dest.size(), Encoding::LATIN1);
149 return kj::mv(dest);
150 }
151 case Encoding::UTF8: {
152 auto dest = jsg::JsUint8Array::create(js, string.utf8Length(js));
153 writeInto(js, dest.asArrayPtr(), string, 0, dest.size(), Encoding::UTF8);
154 return kj::mv(dest);
155 }
156 case Encoding::UTF16LE: {
157 auto dest = jsg::JsUint8Array::create(js, length * sizeof(uint16_t));
158 writeInto(js, dest.asArrayPtr(), string, 0, dest.size(), Encoding::UTF16LE);
159 return kj::mv(dest);
160 }
161 case Encoding::BASE64:
162 // Fall-through
163 case Encoding::BASE64URL: {
164 // We do not use the kj::String conversion here because inline null-characters
165 // need to be ignored.
166 KJ_STACK_ARRAY(kj::byte, buf, length, 1024, 536870888);
167 auto result = string.writeInto(js, buf, options);
168 auto len = result.written;
169 auto dest = jsg::JsUint8Array::create(
170 js, simdutf::maximal_binary_length_from_base64(buf.asChars().begin(), len));
171 auto dec_result = simdutf::base64_to_binary(buf.asChars().begin(), len,
172 dest.asArrayPtr().asChars().begin(), simdutf::base64_default_or_url_accept_garbage);
173 if (dec_result.count < dest.size()) {
174 return dest.slice(js, dec_result.count);
175 }
176 return dest;
177 }
178 case Encoding::HEX: {
179 KJ_STACK_ARRAY(kj::byte, buf, length, 1024, 536870888);
180 string.writeInto(js, buf, options);
181 return decodeHexTruncated(js, buf, strict);
182 }
183 default:
184 KJ_UNREACHABLE;
185 }
186}
187} // namespace
188 
189uint32_t BufferUtil::byteLength(jsg::Lock& js, jsg::JsString str) {
190 return str.utf8Length(js);
191}
192 
193int BufferUtil::compare(jsg::Lock& js,
194 jsg::JsUint8Array one,
195 jsg::JsUint8Array two,
196 jsg::Optional<CompareOptions> maybeOptions) {
197 kj::ArrayPtr<kj::byte> ptrOne = one.asArrayPtr();
198 kj::ArrayPtr<kj::byte> ptrTwo = two.asArrayPtr();
199 
200 // The options allow comparing subranges within the two inputs.
201 KJ_IF_SOME(options, maybeOptions) {
202 auto end = options.aEnd.orDefault(ptrOne.size());
203 end = kj::min(end, ptrOne.size());
204 auto start = kj::min(end, options.aStart.orDefault(0));
205 ptrOne = ptrOne.slice(start, end);
206 end = options.bEnd.orDefault(ptrTwo.size());
207 end = kj::min(end, ptrTwo.size());
208 start = kj::min(end, options.bStart.orDefault(0));
209 ptrTwo = ptrTwo.slice(start, end);
210 }
211 
212 size_t toCompare = kj::min(ptrOne.size(), ptrTwo.size());
213 auto result = toCompare > 0 ? memcmp(ptrOne.begin(), ptrTwo.begin(), toCompare) : 0;
214 
215 if (result == 0) {
216 if (ptrOne.size() > ptrTwo.size())
217 return 1;
218 else if (ptrOne.size() < ptrTwo.size())
219 return -1;
220 else
221 return 0;
222 }
223 
224 return result > 0 ? 1 : -1;
225}
226 
227jsg::JsUint8Array BufferUtil::concat(
228 jsg::Lock& js, kj::Array<jsg::JsUint8Array> list, uint32_t length) {
229 // The Node.js Buffer.concat is interesting in that it doesn't just append
230 // the buffers together as is. The length parameter is used to determine the
231 // length of the result which can be lesser or greater than the actual
232 // combined lengths of the inputs. If the length is lesser, the result will
233 // be a truncated version of the combined buffers. If the length is greater,
234 // the result will be the combined buffers with the remaining space filled
235 // with zeroes.
236 
237 JSG_REQUIRE(length <= v8::ArrayBuffer::kMaxByteLength, RangeError, "The length is too large");
238 
239 auto dest = jsg::JsUint8Array::create(js, length);
240 if (length > 0) {
241 auto view = dest.asArrayPtr();
242 
243 for (auto& src: list) {
244 // JsUint8Array is a view directly on the v8::Uint8Array, so we don't
245 // really need to worry about whether the underlying ArrayBuffer is
246 // detached or resized, etc. The length is not cached.
247 auto ptr = src.asArrayPtr();
248 if (ptr.size() == 0) continue;
249 // The amount to copy is the lesser of the remaining space in the destination or
250 // the size of the chunk we're copying.
251 auto amountToCopy = kj::min(ptr.size(), view.size());
252 view.first(amountToCopy).copyFrom(ptr.first(amountToCopy));
253 view = view.slice(amountToCopy);
254 // If there's no more space in the destination, we're done.
255 if (view == nullptr) {
256 break;
257 }
258 }
259 }
260 
261 return dest;
262}
263 
264jsg::JsUint8Array BufferUtil::decodeString(
265 jsg::Lock& js, jsg::JsString string, EncodingValue encoding) {
266 return decodeStringImpl(js, string, static_cast<Encoding>(encoding));
267}
268 
269void BufferUtil::fillImpl(jsg::Lock& js,
270 jsg::JsUint8Array buffer,
271 kj::OneOf<jsg::JsString, jsg::JsUint8Array> value,
272 uint32_t start,
273 uint32_t end,
274 jsg::Optional<EncodingValue> encoding) {
275 end = kj::min(end, buffer.size());
276 if (end <= start) return;
277 
278 auto ptr = buffer.asArrayPtr().slice(start, end);
279 KJ_SWITCH_ONEOF(value) {
280 KJ_CASE_ONEOF(string, jsg::JsString) {
281 auto enc = encoding.orDefault(Encoding::UTF8);
282 auto decoded = decodeStringImpl(js, string, static_cast<Encoding>(enc), true /* strict */);
283 if (decoded.size() == 0) {
284 ptr.fill(0);
285 return;
286 }
287 ptr.fill(decoded.asArrayPtr());
288 }
289 KJ_CASE_ONEOF(source, jsg::JsUint8Array) {
290 if (source.size() == 0) {
291 ptr.fill(0);
292 return;
293 }
294 ptr.fill(source.asArrayPtr());
295 }
296 }
297}
298 
299namespace {
300 
301// Computes the offset for starting an indexOf or lastIndexOf search.
302// Returns either a valid offset in [0...<length - 1>], ie inside the Buffer,
303// or -1 to signal that there is no possible match.
304int32_t indexOfOffset(size_t length, int32_t offset, int32_t needle_length, bool isForward) {
305 int32_t len = static_cast<int32_t>(length);
306 if (offset < 0) {
307 if (offset + len >= 0) {
308 // Negative offsets count backwards from the end of the buffer.
309 return len + offset;
310 } else if (isForward || needle_length == 0) {
311 // indexOf from before the start of the buffer: search the whole buffer.
312 return 0;
313 } else {
314 // lastIndexOf from before the start of the buffer: no match.
315 return -1;
316 }
317 } else {
318 // cast to int64_t to avoid overflow.
319 if (static_cast<int64_t>(offset) + needle_length <= len) {
320 // Valid positive offset.
321 return offset;
322 } else if (needle_length == 0) {
323 // Out of buffer bounds, but empty needle: point to end of buffer.
324 return len;
325 } else if (isForward) {
326 // indexOf from past the end of the buffer: no match.
327 return -1;
328 } else {
329 // lastIndexOf from past the end of the buffer: search the whole buffer.
330 return len - 1;
331 }
332 }
333}
334 
335jsg::Optional<uint32_t> indexOfBuffer(jsg::Lock& js,
336 kj::ArrayPtr<kj::byte> hayStack,
337 jsg::JsUint8Array needle,
338 int32_t byteOffset,
339 EncodingValue encoding,
340 bool isForward) {
341 auto enc = static_cast<Encoding>(encoding);
342 // Round down to the nearest multiple of 2 in case of UCS2.
343 auto hayStackLength = enc == Encoding::UTF16LE ? hayStack.size() & ~1 : hayStack.size();
344 auto optOffset = indexOfOffset(hayStackLength, byteOffset, needle.size(), isForward);
345 
346 if (needle.size() == 0) return optOffset;
347 if (hayStackLength == 0 || optOffset <= -1 ||
348 (isForward && needle.size() + optOffset > hayStackLength) || needle.size() > hayStackLength) {
349 return kj::none;
350 }
351 auto result = hayStackLength;
352 if (enc == Encoding::UTF16LE) {
353 if (hayStackLength < 2 || needle.size() < 2) {
354 return kj::none;
355 }
356 // Copy haystack and needle to aligned buffers to avoid undefined behavior
357 // from unaligned uint16_t access (the data pointer may have an odd byte offset).
358 auto hayStackU16Len = hayStackLength / 2;
359 kj::SmallArray<uint16_t, 1024> alignedHayStack(hayStackU16Len);
360 alignedHayStack.asBytes().copyFrom(hayStack.first(hayStackLength));
361 
362 auto needleLen = needle.size() & ~1;
363 auto needleU16Len = needleLen / 2;
364 kj::SmallArray<uint16_t, 1024> alignedNeedle(needleU16Len);
365 alignedNeedle.asBytes().copyFrom(needle.asArrayPtr().first(needleLen));
366 
367 result = SearchString(alignedHayStack.begin(), hayStackU16Len, alignedNeedle.begin(),
368 needleU16Len, optOffset / 2, isForward);
369 result *= 2;
370 } else {
371 result = SearchString(hayStack.asBytes().begin(), hayStack.size(),
372 needle.asArrayPtr().asBytes().begin(), needle.size(), optOffset, isForward);
373 }
374 
375 if (result == hayStackLength) return kj::none;
376 
377 return result;
378}
379 
380jsg::Optional<uint32_t> indexOfString(jsg::Lock& js,
381 kj::ArrayPtr<kj::byte> hayStack,
382 const jsg::JsString& needle,
383 int32_t byteOffset,
384 EncodingValue encoding,
385 bool isForward) {
386 
387 auto enc = static_cast<Encoding>(encoding);
388 auto decodedNeedle = decodeStringImpl(js, needle, enc);
389 
390 // Round down to the nearest multiple of 2 in case of UCS2
391 auto hayStackLength = enc == Encoding::UTF16LE ? hayStack.size() & ~1 : hayStack.size();
392 auto optOffset = indexOfOffset(hayStackLength, byteOffset, decodedNeedle.size(), isForward);
393 
394 if (decodedNeedle.size() == 0) {
395 return optOffset;
396 }
397 
398 if (hayStackLength == 0 || optOffset <= -1 ||
399 (isForward && decodedNeedle.size() + optOffset > hayStackLength) ||
400 decodedNeedle.size() > hayStackLength) {
401 return kj::none;
402 }
403 
404 auto result = hayStackLength;
405 
406 if (enc == Encoding::UTF16LE) {
407 if (hayStackLength < 2 || decodedNeedle.size() < 2) {
408 return kj::none;
409 }
410 // Copy haystack to an aligned buffer to avoid undefined behavior from
411 // unaligned uint16_t access (the data pointer may have an odd byte offset).
412 auto hayStackU16Len = hayStackLength / 2;
413 kj::SmallArray<uint16_t, 1024> alignedHayStack(hayStackU16Len);
414 alignedHayStack.asBytes().copyFrom(hayStack.first(hayStackLength));
415 
416 result = SearchString(alignedHayStack.begin(), hayStackU16Len,
417 reinterpret_cast<const uint16_t*>(decodedNeedle.asArrayPtr<char>().begin()),
418 decodedNeedle.size() / 2, optOffset / 2, isForward);
419 result *= 2;
420 } else {
421 result = SearchString(hayStack.asBytes().begin(), hayStack.size(),
422 decodedNeedle.asArrayPtr().begin(), decodedNeedle.size(), optOffset, isForward);
423 }
424 
425 if (result == hayStackLength) return kj::none;
426 
427 return result;
428}
429 
430jsg::JsString toStringImpl(
431 jsg::Lock& js, kj::ArrayPtr<kj::byte> bytes, uint32_t start, uint32_t end, Encoding encoding) {
432 KJ_ASSERT(end <= bytes.size());
433 if (end < start) end = start;
434 auto slice = bytes.slice(start, end);
435 if (slice.size() == 0) return js.str();
436 switch (encoding) {
437 case Encoding::ASCII: {
438 // TODO(perf): We can look at making this more performant later.
439 // Essentially we have to modify the buffer such that every byte
440 // has the highest bit turned off. Whee! Node.js has a faster
441 // algorithm that it implements so we can likely adopt that.
442 kj::Array<kj::byte> copy = KJ_MAP(b, slice) -> kj::byte { return b & 0x7f; };
443 return js.str(copy);
444 }
445 case Encoding::LATIN1: {
446 return js.str(slice);
447 }
448 case Encoding::UTF8: {
449 return js.str(slice.asChars());
450 }
451 case Encoding::UTF16LE: {
452 // Why are we copying here? Good question! There's really not much of a
453 // good reason why we should be copying here except that V8 doesn't seem
454 // to like it! We end up with an asan buffer over-read error if we pass
455 // in the slice directly... looks to have something to do with alignment
456 // issues. Copying here sidesteps the problem and avoids the asan issue.
457 kj::ArrayPtr<uint16_t> view(reinterpret_cast<uint16_t*>(slice.begin()), slice.size() / 2);
458 KJ_STACK_ARRAY(uint16_t, data, view.size(), 1024, 4096);
459 data.copyFrom(view);
460 return js.str(data);
461 }
462 case Encoding::BASE64: {
463 size_t length = simdutf::base64_length_from_binary(slice.size());
464 KJ_STACK_ARRAY(kj::byte, out, length, 1024, 4096);
465 simdutf::binary_to_base64(reinterpret_cast<const char*>(slice.begin()), slice.size(),
466 reinterpret_cast<char*>(out.begin()));
467 return js.str(out);
468 }
469 case Encoding::BASE64URL: {
470 auto options = simdutf::base64_url;
471 size_t length = simdutf::base64_length_from_binary(slice.size(), options);
472 KJ_STACK_ARRAY(kj::byte, out, length, 1024, 4096);
473 simdutf::binary_to_base64(reinterpret_cast<const char*>(slice.begin()), slice.size(),
474 reinterpret_cast<char*>(out.begin()), options);
475 return js.str(out);
476 }
477 case Encoding::HEX: {
478 return js.str(kj::encodeHex(slice));
479 }
480 default:
481 KJ_UNREACHABLE;
482 }
483}
484 
485} // namespace
486 
487jsg::Optional<uint32_t> BufferUtil::indexOf(jsg::Lock& js,
488 jsg::JsUint8Array buffer,
489 kj::OneOf<jsg::JsString, jsg::JsUint8Array> value,
490 int32_t byteOffset,
491 EncodingValue encoding,
492 bool isForward) {
493 
494 KJ_SWITCH_ONEOF(value) {
495 KJ_CASE_ONEOF(string, jsg::JsString) {
496 return indexOfString(js, buffer.asArrayPtr(), string, byteOffset, encoding, isForward);
497 }
498 KJ_CASE_ONEOF(source, jsg::JsUint8Array) {
499 return indexOfBuffer(
500 js, buffer.asArrayPtr(), kj::mv(source), byteOffset, encoding, isForward);
501 }
502 }
503 KJ_UNREACHABLE;
504}
505 
506void BufferUtil::swap(jsg::Lock& js, jsg::JsUint8Array buffer, int size) {
507 if (buffer.size() <= 1) return;
508 switch (size) {
509 case 16: {
510 JSG_REQUIRE(nbytes::SwapBytes16(buffer.asArrayPtr().asChars().begin(), buffer.size()), Error,
511 "Swap bytes failed");
512 break;
513 }
514 case 32: {
515 JSG_REQUIRE(nbytes::SwapBytes32(buffer.asArrayPtr().asChars().begin(), buffer.size()), Error,
516 "Swap bytes failed");
517 break;
518 }
519 case 64: {
520 JSG_REQUIRE(nbytes::SwapBytes64(buffer.asArrayPtr().asChars().begin(), buffer.size()), Error,
521 "Swap bytes failed");
522 break;
523 }
524 default:
525 JSG_FAIL_REQUIRE(Error, "Unreachable");
526 }
527}
528 
529jsg::JsString BufferUtil::toString(
530 jsg::Lock& js, jsg::JsUint8Array bytes, uint32_t start, uint32_t end, EncodingValue encoding) {
531 end = kj::min(bytes.size(), end);
532 if (end <= start) return js.str();
533 return toStringImpl(js, bytes.asArrayPtr(), start, end, static_cast<Encoding>(encoding));
534}
535 
536uint32_t BufferUtil::write(jsg::Lock& js,
537 jsg::JsUint8Array buffer,
538 jsg::JsString string,
539 uint32_t offset,
540 uint32_t length,
541 EncodingValue encoding) {
542 length = kj::min(length, buffer.size() - offset);
543 if (length == 0) return 0;
544 return writeInto(
545 js, buffer.asArrayPtr(), string, offset, length, static_cast<Encoding>(encoding));
546}
547 
548// ======================================================================================
549// StringDecoder
550//
551// It's helpful to review a bit about how the implementation works here.
552//
553// StringDecoder is a streaming decoder that ensures that multi-byte characters are correctly
554// handled. So, for instance, let's suppose I have the utf8 bytes for a euro symbol (0xe2, 0x82,
555// 0xac), but I only get those one at a time... StringDecoder will ensure that those are correctly
556// handled over multiple calls to write(...)...
557//
558// const sd = new StringDecoder();
559// let results = '';
560// results += sd.write(new Uint8Array([0xe2])); // results.length === 0
561// results += sd.write(new Uint8Array([0x82])); // results.length === 0
562// results += sd.write(new Uint8Array([0xac])); // results.length === 1
563// results += sd.end();
564//
565// Internally, the decoder allocates a small 7 byte buffer (the state) argument below.
566//
567// The first four bytes of the state are used to hold partial bytes received on the previous
568// write. The fifth byte in state is a count of the number of missing bytes we need to complete
569// the character. The sixth byte in state is the number of bytes that have been encoded into the
570// first four. The seventh byte in state identifies the Encoding and matches the values of the
571// Encoding enum.
572//
573// So, in our example above, initially the first six bytes of the state are [0x00, 0x00, 0x00,
574// 0x00, 0x00, 0x00]
575//
576// After the first call to write above, the state is updated to: [0xe2, 0x00, 0x00, 0x00, 0x02,
577// 0x01]
578//
579// After the second call to write, the state is updated to: [0xe2, 0x82, 0x00, 0x00, 0x01, 0x02]
580//
581// After the third call to write, the pending multibyte character is completed, the state becomes:
582// [0xe2, 0x82, 0xac, 0x00, 0x00, 0x00] ... while the bytes are still in state, the buffered bytes
583// and bytes needed are zeroed out. Since the character is completed on that third write, it is
584// included in the returned string.
585//
586// The implementation here is taken nearly verbatim from Node.js with a few adaptations. The code
587// from Node.js has remained largely unchanged for years and is well-proven.
588 
589namespace {
590inline kj::byte getMissingBytes(kj::ArrayPtr<kj::byte> state) {
591 JSG_REQUIRE(state[BufferUtil::kMissingBytes] <= BufferUtil::kIncompleteCharactersEnd, Error,
592 "Missing bytes cannot exceed 4");
593 return state[BufferUtil::kMissingBytes];
594}
595 
596inline kj::byte getBufferedBytes(kj::ArrayPtr<kj::byte> state) {
597 JSG_REQUIRE(state[BufferUtil::kBufferedBytes] <= BufferUtil::kIncompleteCharactersEnd, Error,
598 "Buffered bytes cannot exceed 4");
599 return state[BufferUtil::kBufferedBytes];
600}
601 
602inline kj::byte* getIncompleteCharacterBuffer(kj::ArrayPtr<kj::byte> state) {
603 return state.begin() + BufferUtil::kIncompleteCharactersStart;
604}
605 
606inline Encoding getEncoding(kj::ArrayPtr<kj::byte> state) {
607 JSG_REQUIRE(state[BufferUtil::kEncoding] <= static_cast<kj::byte>(Encoding::HEX), Error,
608 "Invalid StringDecoder state");
609 return static_cast<Encoding>(state[BufferUtil::kEncoding]);
610}
611 
612jsg::JsString getBufferedString(jsg::Lock& js, kj::ArrayPtr<kj::byte> state) {
613 JSG_REQUIRE(getBufferedBytes(state) <= BufferUtil::kIncompleteCharactersEnd, Error,
614 "Invalid StringDecoder state");
615 auto ret = toStringImpl(js, state, BufferUtil::kIncompleteCharactersStart,
616 BufferUtil::kIncompleteCharactersStart + getBufferedBytes(state), getEncoding(state));
617 state[BufferUtil::kBufferedBytes] = 0;
618 return ret;
619}
620} // namespace
621 
622jsg::JsString BufferUtil::decode(jsg::Lock& js, jsg::JsUint8Array bytes, jsg::JsUint8Array state) {
623 JSG_REQUIRE(state.size() == BufferUtil::kSize, TypeError, "Invalid StringDecoder");
624 auto bytesPtr = bytes.asArrayPtr();
625 auto statePtr = state.asArrayPtr();
626 auto enc = getEncoding(statePtr);
627 if (enc == Encoding::ASCII || enc == Encoding::LATIN1 || enc == Encoding::HEX) {
628 // For ascii, latin1, and hex, we can just use the regular
629 // toString option since there will never be a case where
630 // these have left-over characters.
631 return toStringImpl(js, bytesPtr, 0, bytesPtr.size(), enc);
632 }
633 
634 jsg::JsString prepend = js.str();
635 jsg::JsString body = js.str();
636 auto nread = bytesPtr.size();
637 
638 // If bytes is empty there's nothing to decode.
639 if (bytesPtr.size() == 0) return js.str();
640 
641 auto data = bytesPtr.begin();
642 
643 if (getMissingBytes(statePtr) > 0) {
644 JSG_REQUIRE(getMissingBytes(statePtr) + getBufferedBytes(statePtr) <=
645 BufferUtil::kIncompleteCharactersEnd,
646 Error, "Invalid StringDecoder state");
647 if (enc == Encoding::UTF8) {
648 // For UTF-8, we need special treatment to align with the V8 decoder:
649 // If an incomplete character is found at a chunk boundary, we use
650 // its remainder and pass it to V8 as-is.
651 for (size_t i = 0; i < nread && i < getMissingBytes(statePtr); ++i) {
652 if ((data[i] & 0xC0) != 0x80) {
653 // This byte is not a continuation byte even though it should have
654 // been one. We stop decoding of the incomplete character at this
655 // point (but still use the rest of the incomplete bytes from this
656 // chunk) and assume that the new, unexpected byte starts a new one.
657 statePtr[kMissingBytes] = 0;
658 // Use memmove: data may point into the incomplete character buffer
659 // (e.g. when the caller passes decoder.lastChar as input).
660 memmove(getIncompleteCharacterBuffer(statePtr) + getBufferedBytes(statePtr), data, i);
661 statePtr[kBufferedBytes] += i;
662 data += i;
663 nread -= i;
664 break;
665 }
666 }
667 }
668 
669 size_t found_bytes = std::min(nread, static_cast<size_t>(getMissingBytes(statePtr)));
670 KJ_ASSERT(data != nullptr);
671 // Use memmove: data may point into the incomplete character buffer
672 // (e.g. when the caller passes decoder.lastChar as input).
673 memmove(getIncompleteCharacterBuffer(statePtr) + getBufferedBytes(statePtr), data, found_bytes);
674 // Adjust the two buffers.
675 data += found_bytes;
676 nread -= found_bytes;
677 
678 statePtr[kMissingBytes] -= found_bytes;
679 statePtr[kBufferedBytes] += found_bytes;
680 
681 if (getMissingBytes(statePtr) == 0) {
682 // If no more bytes are missing, create a small string that we will later prepend.
683 prepend = getBufferedString(js, statePtr);
684 }
685 }
686 
687 if (nread == 0) {
688 body = prepend.length(js) ? prepend : js.str();
689 prepend = js.str();
690 } else {
691 JSG_REQUIRE(getMissingBytes(statePtr) == 0, Error, "Invalid StringDecoder state");
692 JSG_REQUIRE(getBufferedBytes(statePtr) == 0, Error, "Invalid StringDecoder state");
693 
694 // See whether there is a character that we may have to cut off and
695 // finish when receiving the next chunk.
696 if (enc == Encoding::UTF8 && data[nread - 1] & 0x80) {
697 // This is UTF-8 encoded data and we ended on a non-ASCII UTF-8 byte.
698 // This means we'll need to figure out where the character to which
699 // the byte belongs begins.
700 for (size_t i = nread - 1;; --i) {
701 JSG_REQUIRE(i < nread, Error, "Invalid StringDecoder state");
702 statePtr[kBufferedBytes]++;
703 if ((data[i] & 0xC0) == 0x80) {
704 // This byte does not start a character (a "trailing" byte).
705 if (statePtr[kBufferedBytes] >= 4 || i == 0) {
706 // We either have more then 4 trailing bytes (which means
707 // the current character would not be inside the range for
708 // valid Unicode, and in particular cannot be represented
709 // through JavaScript's UTF-16-based approach to strings), or the
710 // current buffer does not contain the start of an UTF-8 character
711 // at all. Either way, this is invalid UTF8 and we can just
712 // let the engine's decoder handle it.
713 statePtr[kBufferedBytes] = 0;
714 break;
715 }
716 } else {
717 // Found the first byte of a UTF-8 character. By looking at the
718 // upper bits we can tell how long the character *should* be.
719 if ((data[i] & 0xE0) == 0xC0) {
720 statePtr[kMissingBytes] = 2;
721 } else if ((data[i] & 0xF0) == 0xE0) {
722 statePtr[kMissingBytes] = 3;
723 } else if ((data[i] & 0xF8) == 0xF0) {
724 statePtr[kMissingBytes] = 4;
725 } else {
726 // This lead byte would indicate a character outside of the
727 // representable range.
728 statePtr[kBufferedBytes] = 0;
729 break;
730 }
731 
732 if (getBufferedBytes(statePtr) >= getMissingBytes(statePtr)) {
733 // Received more or exactly as many trailing bytes than the lead
734 // character would indicate. In the "==" case, we have valid
735 // data and don't need to slice anything off;
736 // in the ">" case, this is invalid UTF-8 anyway.
737 statePtr[kMissingBytes] = 0;
738 statePtr[kBufferedBytes] = 0;
739 }
740 
741 statePtr[kMissingBytes] -= statePtr[kBufferedBytes];
742 break;
743 }
744 }
745 } else if (enc == Encoding::UTF16LE) {
746 if ((nread % 2) == 1) {
747 // WePtr got half a codepoint, and need the second byte of it.
748 statePtr[kBufferedBytes] = 1;
749 statePtr[kMissingBytes] = 1;
750 } else if ((data[nread - 1] & 0xFC) == 0xD8) {
751 // Half a split UTF-16 character.
752 statePtr[kBufferedBytes] = 2;
753 statePtr[kMissingBytes] = 2;
754 }
755 } else if (enc == Encoding::BASE64 || enc == Encoding::BASE64URL) {
756 statePtr[kBufferedBytes] = nread % 3;
757 if (statePtr[kBufferedBytes] > 0) statePtr[kMissingBytes] = 3 - getBufferedBytes(statePtr);
758 }
759 
760 if (getBufferedBytes(statePtr) > 0) {
761 // Copy the requested number of buffered bytes from the end of the
762 // input into the incomplete character buffer.
763 nread -= getBufferedBytes(statePtr);
764 // Use memmove: data may point into the incomplete character buffer
765 // (e.g. when the caller passes decoder.lastChar as input).
766 memmove(getIncompleteCharacterBuffer(statePtr), data + nread, getBufferedBytes(statePtr));
767 }
768 
769 if (nread > 0) {
770 body = toStringImpl(js, kj::ArrayPtr<kj::byte>(data, data + nread), 0, nread, enc);
771 } else {
772 body = js.str();
773 }
774 }
775 
776 if (prepend.length(js) == 0) {
777 return body;
778 } else {
779 return jsg::JsString::concat(js, prepend, body);
780 }
781 
782 return js.str();
783}
784 
785jsg::JsString BufferUtil::flush(jsg::Lock& js, jsg::JsUint8Array state) {
786 JSG_REQUIRE(state.size() == BufferUtil::kSize, TypeError, "Invalid StringDecoder");
787 auto statePtr = state.asArrayPtr();
788 auto enc = getEncoding(statePtr);
789 if (enc == Encoding::ASCII || enc == Encoding::HEX || enc == Encoding::LATIN1) {
790 JSG_REQUIRE(getMissingBytes(statePtr) == 0, Error, "Invalid StringDecoder state");
791 JSG_REQUIRE(getBufferedBytes(statePtr) == 0, Error, "Invalid StringDecoder state");
792 }
793 
794 if (enc == Encoding::UTF16LE && getBufferedBytes(statePtr) % 2 == 1) {
795 // Ignore a single trailing byte, like the JS decoder does.
796 statePtr[kMissingBytes]--;
797 statePtr[kBufferedBytes]--;
798 }
799 
800 if (getBufferedBytes(statePtr) == 0) {
801 return js.str();
802 }
803 
804 auto ret = getBufferedString(js, statePtr);
805 statePtr[kMissingBytes] = 0;
806 
807 return ret;
808}
809 
810bool BufferUtil::isAscii(jsg::JsUint8Array buffer) {
811 if (buffer.size() == 0) return true;
812 return simdutf::validate_ascii(buffer.asArrayPtr().asChars().begin(), buffer.size());
813}
814 
815bool BufferUtil::isUtf8(jsg::JsUint8Array buffer) {
816 if (buffer.size() == 0) return true;
817 return simdutf::validate_utf8(buffer.asArrayPtr().asChars().begin(), buffer.size());
818}
819 
820jsg::JsUint8Array BufferUtil::transcode(jsg::Lock& js,
821 jsg::JsUint8Array source,
822 EncodingValue rawFromEncoding,
823 EncodingValue rawToEncoding) {
824 auto fromEncoding = static_cast<Encoding>(rawFromEncoding);
825 auto toEncoding = static_cast<Encoding>(rawToEncoding);
826 
827 JSG_REQUIRE(i18n::canBeTranscoded(fromEncoding) && i18n::canBeTranscoded(toEncoding), Error,
828 "Unable to transcode buffer due to unsupported encoding");
829 
830 return i18n::transcode(js, source.asArrayPtr(), fromEncoding, toEncoding);
831}
832 
833} // namespace workerd::api::node