Skip to content
File

Blob: src/workerd/api/encoding.c++

43.0 KB
1// Copyright (c) 2017-2022 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4 
5#include "encoding.h"
6 
7#include "simdutf.h"
8#include "util.h"
9 
10#include <workerd/io/features.h>
11#include <workerd/jsg/jsg.h>
12#include <workerd/util/autogate.h>
13#include <workerd/util/strings.h>
14 
15#include <unicode/ucnv.h>
16#include <unicode/utf8.h>
17#include <v8.h>
18 
19#include <kj/array.h>
20#include <kj/string.h>
21 
22namespace workerd::api {
23 
24// =======================================================================================
25// TextDecoder implementation
26 
27namespace {
28#define EW_ENCODING_LABELS(V) \
29 V("unicode-1-1-utf-8", Utf8) \
30 V("unicode11utf8", Utf8) \
31 V("unicode20utf8", Utf8) \
32 V("utf-8", Utf8) \
33 V("utf8", Utf8) \
34 V("x-unicode20utf8", Utf8) \
35 V("866", Ibm866) \
36 V("cp866", Ibm866) \
37 V("csibm866", Ibm866) \
38 V("ibm866", Ibm866) \
39 V("csisolatin2", Iso8859_2) \
40 V("iso-8859-2", Iso8859_2) \
41 V("iso-ir-101", Iso8859_2) \
42 V("iso8859-2", Iso8859_2) \
43 V("iso88592", Iso8859_2) \
44 V("iso_8859-2", Iso8859_2) \
45 V("iso_8859-2:1987", Iso8859_2) \
46 V("l2", Iso8859_2) \
47 V("latin2", Iso8859_2) \
48 V("csisolatin3", Iso8859_3) \
49 V("iso-8859-3", Iso8859_3) \
50 V("iso-ir-109", Iso8859_3) \
51 V("iso8859-3", Iso8859_3) \
52 V("iso88593", Iso8859_3) \
53 V("iso_8859-3", Iso8859_3) \
54 V("iso_8859-3:1988", Iso8859_3) \
55 V("l3", Iso8859_3) \
56 V("latin3", Iso8859_3) \
57 V("csisolatin4", Iso8859_4) \
58 V("iso-8859-4", Iso8859_4) \
59 V("iso-ir-110", Iso8859_4) \
60 V("iso8859-4", Iso8859_4) \
61 V("iso88594", Iso8859_4) \
62 V("iso_8859-4", Iso8859_4) \
63 V("iso_8859-4:1988", Iso8859_4) \
64 V("l4", Iso8859_4) \
65 V("latin4", Iso8859_4) \
66 V("csisolatincyrillic", Iso8859_5) \
67 V("cyrillic", Iso8859_5) \
68 V("iso-8859-5", Iso8859_5) \
69 V("iso-ir-144", Iso8859_5) \
70 V("iso8859-5", Iso8859_5) \
71 V("iso88595", Iso8859_5) \
72 V("iso_8859-5", Iso8859_5) \
73 V("iso_8859-5:1988", Iso8859_5) \
74 V("arabic", Iso8859_6) \
75 V("asmo-708", Iso8859_6) \
76 V("csiso88596e", Iso8859_6) \
77 V("csiso88596i", Iso8859_6) \
78 V("csisolatinarabic", Iso8859_6) \
79 V("ecma-114", Iso8859_6) \
80 V("iso-8859-6", Iso8859_6) \
81 V("iso-8859-6-e", Iso8859_6) \
82 V("iso-8859-6-i", Iso8859_6) \
83 V("iso-ir-127", Iso8859_6) \
84 V("iso8859-6", Iso8859_6) \
85 V("iso88596", Iso8859_6) \
86 V("iso_8859-6", Iso8859_6) \
87 V("iso_8859-6:1987", Iso8859_6) \
88 V("csisolatingreek", Iso8859_7) \
89 V("ecma-118", Iso8859_7) \
90 V("elot_928", Iso8859_7) \
91 V("greek", Iso8859_7) \
92 V("greek8", Iso8859_7) \
93 V("iso-8859-7", Iso8859_7) \
94 V("iso-ir-126", Iso8859_7) \
95 V("iso8859-7", Iso8859_7) \
96 V("iso88597", Iso8859_7) \
97 V("iso_8859-7", Iso8859_7) \
98 V("iso_8859-7:1987", Iso8859_7) \
99 V("sun_eu_greek", Iso8859_7) \
100 V("csiso88598e", Iso8859_8) \
101 V("csisolatinhebrew", Iso8859_8) \
102 V("hebrew", Iso8859_8) \
103 V("iso-8859-8", Iso8859_8) \
104 V("iso-8859-8-e", Iso8859_8) \
105 V("iso-ir-138", Iso8859_8) \
106 V("iso8859-8", Iso8859_8) \
107 V("iso88598", Iso8859_8) \
108 V("iso_8859-8", Iso8859_8) \
109 V("iso_8859-8:1988", Iso8859_8) \
110 V("visual", Iso8859_8) \
111 V("csiso88598i", Iso8859_8i) \
112 V("iso-8859-8-i", Iso8859_8i) \
113 V("logical", Iso8859_8i) \
114 V("csisolatin6", Iso8859_10) \
115 V("iso-8859-10", Iso8859_10) \
116 V("iso-ir-157", Iso8859_10) \
117 V("iso8859-10", Iso8859_10) \
118 V("iso885910", Iso8859_10) \
119 V("l6", Iso8859_10) \
120 V("latin6", Iso8859_10) \
121 V("iso-8859-13", Iso8859_13) \
122 V("iso8859-13", Iso8859_13) \
123 V("iso885913", Iso8859_13) \
124 V("iso-8859-14", Iso8859_14) \
125 V("iso8859-14", Iso8859_14) \
126 V("iso885914", Iso8859_14) \
127 V("csisolatin9", Iso8859_15) \
128 V("iso-8859-15", Iso8859_15) \
129 V("iso8859-15", Iso8859_15) \
130 V("iso885915", Iso8859_15) \
131 V("iso_8859-15", Iso8859_15) \
132 V("l9", Iso8859_15) \
133 V("iso-8859-16", Iso8859_16) \
134 V("cskoi8r", Ko18_r) \
135 V("koi", Ko18_r) \
136 V("koi8", Ko18_r) \
137 V("koi8-r", Ko18_r) \
138 V("koi8_r", Ko18_r) \
139 V("koi8-ru", Koi8_u) \
140 V("koi8-u", Koi8_u) \
141 V("csmacintosh", Macintosh) \
142 V("mac", Macintosh) \
143 V("macintosh", Macintosh) \
144 V("x-mac-roman", Macintosh) \
145 V("dos-874", Windows_874) \
146 V("iso-8859-11", Windows_874) \
147 V("iso8859-11", Windows_874) \
148 V("iso885911", Windows_874) \
149 V("tis-620", Windows_874) \
150 V("windows-874", Windows_874) \
151 V("cp1250", Windows_1250) \
152 V("windows-1250", Windows_1250) \
153 V("x-cp1250", Windows_1250) \
154 V("cp1251", Windows_1251) \
155 V("windows-1251", Windows_1251) \
156 V("x-cp1251", Windows_1251) \
157 V("ansi_x3.4-1968", Windows_1252) \
158 V("ascii", Windows_1252) \
159 V("cp1252", Windows_1252) \
160 V("cp819", Windows_1252) \
161 V("csisolatin1", Windows_1252) \
162 V("ibm819", Windows_1252) \
163 V("iso-8859-1", Windows_1252) \
164 V("iso-ir-100", Windows_1252) \
165 V("iso8859-1", Windows_1252) \
166 V("iso88591", Windows_1252) \
167 V("iso_8859-1", Windows_1252) \
168 V("iso_8859-1:1987", Windows_1252) \
169 V("l1", Windows_1252) \
170 V("latin1", Windows_1252) \
171 V("us-ascii", Windows_1252) \
172 V("windows-1252", Windows_1252) \
173 V("x-cp1252", Windows_1252) \
174 V("cp1253", Windows_1253) \
175 V("windows-1253", Windows_1253) \
176 V("x-cp1253", Windows_1253) \
177 V("cp1254", Windows_1254) \
178 V("csisolatin5", Windows_1254) \
179 V("iso-8859-9", Windows_1254) \
180 V("iso-ir-148", Windows_1254) \
181 V("iso8859-9", Windows_1254) \
182 V("iso88599", Windows_1254) \
183 V("iso_8859-9", Windows_1254) \
184 V("iso_8859-9:1989", Windows_1254) \
185 V("l5", Windows_1254) \
186 V("latin5", Windows_1254) \
187 V("windows-1254", Windows_1254) \
188 V("x-cp1254", Windows_1254) \
189 V("cp1255", Windows_1255) \
190 V("windows-1255", Windows_1255) \
191 V("x-cp1255", Windows_1255) \
192 V("cp1256", Windows_1256) \
193 V("windows-1256", Windows_1256) \
194 V("x-cp1256", Windows_1256) \
195 V("cp1257", Windows_1257) \
196 V("windows-1257", Windows_1257) \
197 V("x-cp1257", Windows_1257) \
198 V("cp1258", Windows_1258) \
199 V("windows-1258", Windows_1258) \
200 V("x-cp1258", Windows_1258) \
201 V("x-mac-cyrillic", X_Mac_Cyrillic) \
202 V("x-mac-ukrainian", X_Mac_Cyrillic) \
203 V("chinese", Gbk) \
204 V("csgb2312", Gbk) \
205 V("csiso58gb231280", Gbk) \
206 V("gb2312", Gbk) \
207 V("gb_2312", Gbk) \
208 V("gb_2312-80", Gbk) \
209 V("gbk", Gbk) \
210 V("iso-ir-58", Gbk) \
211 V("x-gbk", Gbk) \
212 V("gb18030", Gb18030) \
213 V("big5", Big5) \
214 V("big5-hkscs", Big5) \
215 V("cn-big5", Big5) \
216 V("csbig5", Big5) \
217 V("x-x-big5", Big5) \
218 V("cseucpkdfmtjapanese", Euc_Jp) \
219 V("euc-jp", Euc_Jp) \
220 V("x-euc-jp", Euc_Jp) \
221 V("csiso2022jp", Iso2022_Jp) \
222 V("iso-2022-jp", Iso2022_Jp) \
223 V("csshiftjis", Shift_Jis) \
224 V("ms932", Shift_Jis) \
225 V("ms_kanji", Shift_Jis) \
226 V("shift-jis", Shift_Jis) \
227 V("shift_jis", Shift_Jis) \
228 V("sjis", Shift_Jis) \
229 V("windows-31j", Shift_Jis) \
230 V("x-sjis", Shift_Jis) \
231 V("cseuckr", Euc_Kr) \
232 V("csksc56011987", Euc_Kr) \
233 V("euc-kr", Euc_Kr) \
234 V("iso-ir-149", Euc_Kr) \
235 V("korean", Euc_Kr) \
236 V("ks_c_5601-1987", Euc_Kr) \
237 V("ks_c_5601-1989", Euc_Kr) \
238 V("ksc5601", Euc_Kr) \
239 V("ksc_5601", Euc_Kr) \
240 V("windows-949", Euc_Kr) \
241 V("csiso2022kr", Replacement) \
242 V("hz-gb-2312", Replacement) \
243 V("iso-2022-cn", Replacement) \
244 V("iso-2022-cn-ext", Replacement) \
245 V("iso-2022-kr", Replacement) \
246 V("replacement", Replacement) \
247 V("unicodefffe", Utf16be) \
248 V("utf-16be", Utf16be) \
249 V("csunicode", Utf16le) \
250 V("iso-10646-ucs-2", Utf16le) \
251 V("ucs-2", Utf16le) \
252 V("unicode", Utf16le) \
253 V("unicodefeff", Utf16le) \
254 V("utf-16", Utf16le) \
255 V("utf-16le", Utf16le) \
256 V("x-user-defined", X_User_Defined)
257 
258kj::StringPtr getEncodingId(Encoding encoding) {
259 switch (encoding) {
260 case Encoding::INVALID:
261 return "invalid"_kj;
262#define V(name, id) \
263 case Encoding::name: \
264 return id##_kj;
265 EW_ENCODINGS(V)
266#undef V
267 }
268 KJ_UNREACHABLE;
269}
270 
271Encoding getEncodingForLabel(kj::StringPtr label) {
272 auto lower = toLower(label);
273 auto trimmed = trimLeadingAndTrailingWhitespace(lower);
274#define V(label, key) \
275 if (trimmed == label##_kjb) return Encoding::key;
276 EW_ENCODING_LABELS(V)
277#undef V
278 return Encoding::INVALID;
279}
280 
281constexpr int MAX_SIZE_FOR_STACK_ALLOC = 4096;
282 
283} // namespace
284 
285const kj::Array<const kj::byte> TextDecoder::EMPTY =
286 kj::Array<const kj::byte>(&DUMMY, 0, kj::NullArrayDisposer::instance);
287const TextDecoder::DecodeOptions TextDecoder::DEFAULT_OPTIONS = TextDecoder::DecodeOptions();
288 
289kj::Maybe<IcuDecoder> IcuDecoder::create(Encoding encoding, bool fatal, bool ignoreBom) {
290 UErrorCode status = U_ZERO_ERROR;
291 // Per the WHATWG encoding spec (section 10.1.1), GBK's decoder is gb18030's decoder.
292 // https://encoding.spec.whatwg.org/#gbk-decoder
293 // We can't change getEncodingId() itself because it is also used for the TextDecoder.encoding
294 // getter, which must still return "gbk" for GBK.
295 auto icuEncoding =
296 encoding == Encoding::Gbk ? getEncodingId(Encoding::Gb18030) : getEncodingId(encoding);
297 UConverter* inner = ucnv_open(icuEncoding.cStr(), &status);
298 JSG_REQUIRE(U_SUCCESS(status), RangeError, "Invalid or unsupported encoding");
299 
300 if (fatal) {
301 status = U_ZERO_ERROR;
302 ucnv_setToUCallBack(inner, UCNV_TO_U_CALLBACK_STOP, nullptr, nullptr, nullptr, &status);
303 if (U_FAILURE(status)) return kj::none;
304 }
305 
306 return IcuDecoder(encoding, inner, fatal, ignoreBom);
307}
308 
309kj::Maybe<jsg::JsString> IcuDecoder::decode(
310 jsg::Lock& js, kj::ArrayPtr<const kj::byte> buffer, bool flush) {
311 UErrorCode status = U_ZERO_ERROR;
312 const auto maxCharSize = [this]() { return ucnv_getMaxCharSize(inner.get()); };
313 
314 const auto isUnicode = [this]() {
315 switch (ucnv_getType(inner.get())) {
316 case UCNV_UTF8:
317 case UCNV_UTF16:
318 case UCNV_UTF16_BigEndian:
319 case UCNV_UTF16_LittleEndian:
320 return true;
321 default:
322 return false;
323 }
324 KJ_UNREACHABLE;
325 };
326 
327 KJ_DEFER({
328 if (flush) reset();
329 });
330 
331 // Evaluate fast-path options. These provide shortcuts for common cases with the caveat
332 // that error handling for invalid sequences might be a bit different (because the
333 // conversions are being handled by v8 directly rather than by the ICU converter).
334 if (buffer.size() > 0 && ucnv_toUCountPending(inner.get(), &status) == 0) {
335 KJ_ASSERT(U_SUCCESS(status));
336 if (encoding == Encoding::Utf8 &&
337 simdutf::validate_ascii(buffer.asChars().begin(), buffer.size())) {
338 // This is a fast-path option for UTF-8 that can be taken when there
339 // are no buffered inputs and the non-empty input buffer contains only
340 // codepoints <= 0x7f. This path is safe because with ASCII range codepoints
341 // we know we won't accidentally split a multi-byte encoding. We also don't
342 // have to worry about the BOM here since the BOM bytes are > 0x7f.
343 // Note also that in this case we'll interpret as Latin1 since UTF-8 bytes
344 // within this range are identical to Latin1 and v8 allocates these more
345 // efficiently.
346 return js.str(buffer);
347 }
348 
349 if (encoding == Encoding::Utf16le && buffer.size() % sizeof(char16_t) == 0) {
350 // This is a fast-path option for UTF-16le that can be taken when:
351 // there are no buffered inputs, the non-empty input buffer length is an
352 // even multiple of 2, and either flush is true or the last code unit
353 // is not a Unicode lead surrogate. This is safe because when flush
354 // is true the converter state will be cleared, and if the last code
355 // unit is not a lead surrogate, we won't have to worry about possibly
356 // splitting a valid surrogate pair.
357 
358 // The input buffer may be at an odd byte offset (e.g. a Uint8Array view
359 // at offset 3 into an ArrayBuffer), which makes reinterpret_cast to
360 // char16_t* undefined behavior due to alignment violation. Copy into an
361 // aligned buffer to avoid this.
362 auto bufSize = buffer.size() / 2;
363 kj::SmallArray<char16_t, 256> aligned(bufSize);
364 aligned.asBytes().copyFrom(buffer.first(bufSize * 2));
365 auto data = aligned.asPtr();
366 
367 if (flush || !U_IS_SURROGATE_LEAD(data[data.size() - 1])) {
368 bool omitInitialBom = false;
369 if (!ignoreBom && !bomSeen) {
370 omitInitialBom = data[0] == 0xfeff;
371 bomSeen = true;
372 }
373 
374 auto slice = data.slice(omitInitialBom ? 1 : 0, data.size());
375 
376 // If textDecoderReplaceSurrogates flag is enabled, then we follow the spec
377 // and fix invalid surrogates on the UTF-16 input.
378 if (slice.size() == 0 || !FeatureFlags::get(js).getTextDecoderReplaceSurrogates()) {
379 return js.str(slice);
380 }
381 
382 if (simdutf::validate_utf16(slice.begin(), slice.size())) {
383 return js.str(slice);
384 }
385 
386 if (fatal) {
387 // In fatal mode, return error for invalid surrogates
388 return kj::none;
389 }
390 
391 // In non-fatal mode, replace invalid surrogates with U+FFFD.
392 // Output size equals input size because each invalid surrogate (1 code unit)
393 // is replaced with U+FFFD (also 1 code unit).
394 // Use stack allocation for small strings (up to 256 code units) to avoid
395 // heap allocation overhead.
396 kj::SmallArray<char16_t, 256> fixed(slice.size());
397 simdutf::to_well_formed_utf16(slice.begin(), slice.size(), fixed.begin());
398 return js.str(fixed.asPtr());
399 }
400 }
401 }
402 
403 status = U_ZERO_ERROR;
404 auto limit = 2 * maxCharSize() *
405 (!flush ? buffer.size()
406 : kj::max(buffer.size(),
407 static_cast<size_t>(ucnv_toUCountPending(inner.get(), &status))));
408 
409 KJ_STACK_ARRAY(UChar, result, limit, 512, 4096);
410 
411 auto dest = result.begin();
412 auto source = reinterpret_cast<const char*>(buffer.begin());
413 
414 ucnv_toUnicode(
415 inner.get(), &dest, dest + limit, &source, source + buffer.size(), nullptr, flush, &status);
416 
417 if (U_FAILURE(status)) return kj::none;
418 
419 auto omitInitialBom = false;
420 auto length = std::distance(result.begin(), dest);
421 if (length > 0 && isUnicode() && !ignoreBom && !bomSeen) {
422 omitInitialBom = result[0] == 0xfeff;
423 bomSeen = true;
424 }
425 
426 return js.str(result.slice(omitInitialBom ? 1 : 0, length));
427}
428 
429void IcuDecoder::reset() {
430 bomSeen = false;
431 return ucnv_reset(inner.get());
432}
433 
434Decoder& TextDecoder::getImpl() {
435 KJ_SWITCH_ONEOF(decoder) {
436 KJ_CASE_ONEOF(dec, LegacyDecoder) {
437 return dec;
438 }
439 KJ_CASE_ONEOF(dec, IcuDecoder) {
440 return dec;
441 }
442 }
443 KJ_UNREACHABLE;
444}
445 
446jsg::Ref<TextDecoder> TextDecoder::constructor(jsg::Lock& js,
447 jsg::Optional<kj::String> maybeLabel,
448 jsg::Optional<ConstructorOptions> maybeOptions) {
449 static constexpr ConstructorOptions DEFAULT_OPTIONS;
450 auto options = maybeOptions.orDefault(DEFAULT_OPTIONS);
451 auto encoding = Encoding::Utf8;
452 
453 const auto errorMessage = [](kj::StringPtr label) {
454 return kj::str("\"", label, "\" is not a valid encoding.");
455 };
456 
457 KJ_IF_SOME(label, maybeLabel) {
458 encoding = getEncodingForLabel(label);
459 JSG_REQUIRE(encoding != Encoding::Replacement && encoding != Encoding::INVALID, RangeError,
460 errorMessage(label));
461 }
462 
463 switch (encoding) {
464 case Encoding::Big5:
465 case Encoding::Euc_Jp:
466 case Encoding::Euc_Kr:
467 case Encoding::Gb18030:
468 case Encoding::Gbk:
469 case Encoding::Iso2022_Jp:
470 case Encoding::Shift_Jis: {
471 // If the feature flag is disabled, we use the ICU decoder.
472 if (!FeatureFlags::get(js).getTextDecoderCjkDecoder()) {
473 break;
474 }
475 
476 // We fallthrough to LegacyDecoder in order to avoid breaking changes.
477 [[fallthrough]];
478 }
479 case Encoding::X_User_Defined:
480 case Encoding::Windows_1252:
481 return js.alloc<TextDecoder>(LegacyDecoder(encoding, DecoderFatal(options.fatal)), options);
482 default:
483 break;
484 }
485 
486 return js.alloc<TextDecoder>(
487 JSG_REQUIRE_NONNULL(IcuDecoder::create(encoding, options.fatal, options.ignoreBOM),
488 RangeError, errorMessage(getEncodingId(encoding))),
489 options);
490}
491 
492kj::StringPtr TextDecoder::getEncoding() {
493 return getEncodingId(getImpl().getEncoding());
494}
495 
496jsg::JsString TextDecoder::decode(jsg::Lock& js,
497 jsg::Optional<kj::Array<const kj::byte>> maybeInput,
498 jsg::Optional<DecodeOptions> maybeOptions) {
499 auto options = maybeOptions.orDefault(DEFAULT_OPTIONS);
500 auto& input = maybeInput.orDefault(EMPTY);
501 return JSG_REQUIRE_NONNULL(
502 getImpl().decode(js, input, !options.stream), TypeError, "Failed to decode input.");
503}
504 
505kj::Maybe<jsg::JsString> TextDecoder::decodePtr(
506 jsg::Lock& js, kj::ArrayPtr<const kj::byte> buffer, bool flush) {
507 KJ_SWITCH_ONEOF(decoder) {
508 KJ_CASE_ONEOF(dec, LegacyDecoder) {
509 return dec.decode(js, buffer, flush);
510 }
511 KJ_CASE_ONEOF(dec, IcuDecoder) {
512 return dec.decode(js, buffer, flush);
513 }
514 }
515 KJ_UNREACHABLE;
516}
517 
518// =======================================================================================
519// TextEncoder implementation
520 
521jsg::Ref<TextEncoder> TextEncoder::constructor(jsg::Lock& js) {
522 return js.alloc<TextEncoder>();
523}
524 
525jsg::JsUint8Array TextEncoder::encode(jsg::Lock& js, jsg::Optional<jsg::JsString> input) {
526 if (!workerd::util::Autogate::isEnabled(workerd::util::AutogateKey::ENABLE_FAST_TEXTENCODER)) {
527 auto str = input.orDefault(js.str());
528 auto view = jsg::JsUint8Array::create(js, str.utf8Length(js));
529 [[maybe_unused]] auto result = str.writeInto(
530 js, view.asArrayPtr().asChars(), jsg::JsString::WriteFlags::REPLACE_INVALID_UTF8);
531 KJ_DASSERT(result.written == view.size());
532 return view;
533 }
534 
535 jsg::JsString str = input.orDefault(js.str());
536 
537 size_t utf8_length = 0;
538 auto length = str.length(js);
539 
540#ifdef KJ_DEBUG
541 bool wasAlreadyFlat = str.isFlat();
542 KJ_DEFER({ KJ_ASSERT(wasAlreadyFlat || !str.isFlat()); });
543#endif
544 
545 // Note: writeInto() doesn't flatten the string - it calls writeTo() which chains through
546 // Write2 -> WriteV2 -> WriteHelperV2 -> String::WriteToFlat.
547 // This means we may read from multiple string segments, but that's fine for our use case.
548 
549 if (str.isOneByte(js)) {
550 // Use off-heap allocation for intermediate Latin-1 buffer to avoid wasting V8 heap space
551 // and potentially triggering GC. Stack allocation for small strings, heap for large.
552 kj::SmallArray<kj::byte, MAX_SIZE_FOR_STACK_ALLOC> latin1Buffer(length);
553 
554 [[maybe_unused]] auto writeResult = str.writeInto(js, latin1Buffer.asPtr());
555 KJ_DASSERT(
556 writeResult.written == length, "writeInto must completely overwrite the backing buffer");
557 
558 utf8_length = simdutf::utf8_length_from_latin1(
559 reinterpret_cast<const char*>(latin1Buffer.begin()), length);
560 
561 auto view = jsg::JsUint8Array::create(js, utf8_length);
562 if (utf8_length == length) {
563 // ASCII fast path: no conversion needed, Latin-1 is same as UTF-8 for ASCII
564 view.asArrayPtr().copyFrom(latin1Buffer);
565 } else {
566 [[maybe_unused]] auto written =
567 simdutf::convert_latin1_to_utf8(reinterpret_cast<const char*>(latin1Buffer.begin()),
568 length, view.asArrayPtr().asChars().begin());
569 KJ_DASSERT(utf8_length == written);
570 }
571 return view;
572 }
573 
574 // Use off-heap allocation for intermediate UTF-16 buffer to avoid wasting V8 heap space
575 // and potentially triggering GC. Stack allocation for small strings, heap for large.
576 // Stack allocation for small strings, heap for large.
577 kj::SmallArray<uint16_t, MAX_SIZE_FOR_STACK_ALLOC> utf16Buffer(length);
578 
579 [[maybe_unused]] auto writeResult = str.writeInto(js, utf16Buffer.asPtr());
580 KJ_DASSERT(
581 writeResult.written == length, "writeInto must completely overwrite the backing buffer");
582 
583 auto data = reinterpret_cast<char16_t*>(utf16Buffer.begin());
584 auto lengthResult = simdutf::utf8_length_from_utf16_with_replacement(data, length);
585 utf8_length = lengthResult.count;
586 
587 if (lengthResult.error == simdutf::SURROGATE) {
588 // If there are surrogates there may be unpaired surrogates. Fix them.
589 simdutf::to_well_formed_utf16(data, length, data);
590 } else {
591 KJ_DASSERT(lengthResult.error == simdutf::SUCCESS);
592 }
593 
594 auto view = jsg::JsUint8Array::create(js, utf8_length);
595 [[maybe_unused]] auto written =
596 simdutf::convert_utf16_to_utf8(data, length, view.asArrayPtr().asChars().begin());
597 KJ_DASSERT(written == utf8_length, "Conversion yielded wrong number of UTF-8 bytes");
598 return view;
599}
600 
601namespace {
602 
603constexpr bool isSurrogatePair(uint16_t lead, uint16_t trail) {
604 // We would like to use simdutf::trim_partial_utf16, but it's not guaranteed
605 // to work right on invalid UTF-16. Hence, we need this method to check for
606 // surrogate pairs and correctly trim utf16 chunks.
607 return (lead & 0xfc00) == 0xd800 && (trail & 0xfc00) == 0xdc00;
608}
609 
610// Ignores surrogates conservatively.
611constexpr size_t simpleUtfEncodingLength(uint16_t c) {
612 return 1 + (c >= 0x80) + (c >= 0x400);
613}
614 
615// Find how many UTF-16 or Latin1 code units fit when converted to UTF-8.
616// May conservatively underestimate the largest number of code units we can fit
617// because of undetected surrogate pairs on boundaries.
618// Works even on malformed UTF-16.
619template <typename Char>
620size_t findBestFit(const Char* data, size_t length, size_t bufferSize) {
621 size_t pos = 0;
622 size_t utf8Accumulated = 0;
623 // The SIMD is more efficient with a size that's a little over a multiple of 16.
624 constexpr size_t CHUNK = 257;
625 // The max number of UTF-8 output bytes per input code unit.
626 constexpr bool UTF16 = sizeof(Char) == 2;
627 constexpr size_t MAX_FACTOR = UTF16 ? 3 : 2;
628 
629 // Our initial guess at how much the number of elements expands in the
630 // conversion to UTF-8.
631 double expansion = 1.15;
632 
633 while (pos < length && utf8Accumulated < bufferSize) {
634 size_t remainingInput = length - pos;
635 size_t spaceRemaining = bufferSize - utf8Accumulated;
636 KJ_DASSERT(expansion >= 1.15);
637 
638 // We estimate how many characters are likely to fit in the buffer, but
639 // only try for CHUNK characters at a time to minimize the worst case
640 // waste of time if we guessed too high.
641 size_t guaranteedToFit = spaceRemaining / MAX_FACTOR;
642 if (guaranteedToFit >= remainingInput) {
643 // Don't even bother checking any more, it's all going to fit. Hitting
644 // this halfway through is also a good reason to limit the CHUNK size.
645 return length;
646 }
647 size_t likelyToFit = kj::min(static_cast<size_t>(spaceRemaining / expansion), CHUNK);
648 size_t fitEstimate = kj::max(1, kj::max(guaranteedToFit, likelyToFit));
649 size_t chunkSize = kj::min(remainingInput, fitEstimate);
650 if (chunkSize == 1) break; // Not worth running this complicated stuff one char at a time.
651 // No div-by-zero because remainingInput and fitEstimate are at least 1.
652 KJ_DASSERT(chunkSize >= 1);
653 
654 size_t chunkUtf8Len;
655 if constexpr (UTF16) {
656 chunkUtf8Len = simdutf::utf8_length_from_utf16_with_replacement(data + pos, chunkSize).count;
657 } else {
658 chunkUtf8Len = simdutf::utf8_length_from_latin1(data + pos, chunkSize);
659 }
660 
661 if (utf8Accumulated + chunkUtf8Len > bufferSize) {
662 // Our chosen chunk didn't fit in the rest of the output buffer.
663 KJ_DASSERT(chunkSize > guaranteedToFit);
664 // Since it didn't fit we adjust our expansion guess upwards.
665 expansion = kj::max(expansion * 1.1, (chunkUtf8Len * 1.1) / chunkSize);
666 } else {
667 // Use successful length calculation to adjust our expansion estimate.
668 expansion = kj::max(1.15, (chunkUtf8Len * 1.1) / chunkSize);
669 pos += chunkSize;
670 utf8Accumulated += chunkUtf8Len;
671 }
672 }
673 // Do the last few code units in a simpler way.
674 while (pos < length && utf8Accumulated < bufferSize) {
675 size_t extra = simpleUtfEncodingLength(data[pos]);
676 if (utf8Accumulated + extra > bufferSize) break;
677 pos++;
678 utf8Accumulated += extra;
679 }
680 if (UTF16 && pos != 0 && pos != length && isSurrogatePair(data[pos - 1], data[pos])) {
681 // We ended on a leading surrogate which has a matching trailing surrogate in the next
682 // position. In order to make progress when the bufferSize is tiny we try to include it.
683 if (utf8Accumulated < bufferSize) {
684 pos++; // We had one more byte, so we can include the pair, UTF-8 encoding 3->4.
685 } else {
686 pos--; // Don't chop the pair in half.
687 }
688 }
689 return pos;
690}
691 
692} // namespace
693 
694// Test helpers used by encoding-test.c++ to verify findBestFit behavior.
695namespace test {
696 
697size_t bestFit(const char* str, size_t bufferSize) {
698 return findBestFit(str, strlen(str), bufferSize);
699}
700 
701size_t bestFit(const char16_t* str, size_t bufferSize) {
702 size_t length = 0;
703 while (str[length] != 0) length++;
704 return findBestFit(str, length, bufferSize);
705}
706 
707} // namespace test
708 
709TextEncoder::EncodeIntoResult TextEncoder::encodeInto(
710 jsg::Lock& js, jsg::JsString input, jsg::JsUint8Array buffer) {
711 if (!workerd::util::Autogate::isEnabled(workerd::util::AutogateKey::ENABLE_FAST_TEXTENCODER)) {
712 auto result = input.writeInto(
713 js, buffer.asArrayPtr<char>(), jsg::JsString::WriteFlags::REPLACE_INVALID_UTF8);
714 return TextEncoder::EncodeIntoResult{
715 .read = static_cast<int>(result.read),
716 .written = static_cast<int>(result.written),
717 };
718 }
719 
720 auto outputBuf = buffer.asArrayPtr<char>();
721 size_t bufferSize = outputBuf.size();
722 
723 size_t read = 0;
724 size_t written = 0;
725 {
726 // Scope for the view - we can't do anything that might cause a V8 GC!
727 v8::String::ValueView view(js.v8Isolate, input);
728 size_t length = view.length();
729 
730 if (view.is_one_byte()) {
731 auto data = reinterpret_cast<const char*>(view.data8());
732 simdutf::result result =
733 simdutf::validate_ascii_with_errors(data, kj::min(length, bufferSize));
734 written = read = result.count;
735 auto outAddr = outputBuf.begin();
736 kj::arrayPtr(outAddr, read).copyFrom(kj::arrayPtr(data, read));
737 outAddr += read;
738 data += read;
739 length -= read;
740 bufferSize -= read;
741 if (length != 0 && bufferSize != 0) {
742 size_t rest = findBestFit(data, length, bufferSize);
743 if (rest != 0) {
744 KJ_DASSERT(simdutf::utf8_length_from_latin1(data, rest) <= bufferSize);
745 written += simdutf::convert_latin1_to_utf8(data, rest, outAddr);
746 read += rest;
747 }
748 }
749 } else {
750 auto data = reinterpret_cast<const char16_t*>(view.data16());
751 read = findBestFit(data, length, bufferSize);
752 if (read != 0) {
753 KJ_DASSERT(
754 simdutf::utf8_length_from_utf16_with_replacement(data, read).count <= bufferSize);
755 simdutf::result result =
756 simdutf::convert_utf16_to_utf8_with_errors(data, read, outputBuf.begin());
757 if (result.error == simdutf::SUCCESS) {
758 written = result.count;
759 } else {
760 // Oh, no, there are unpaired surrogates. This is hopefully rare.
761 kj::SmallArray<char16_t, MAX_SIZE_FOR_STACK_ALLOC> conversionBuffer(read);
762 simdutf::to_well_formed_utf16(data, read, conversionBuffer.begin());
763 written =
764 simdutf::convert_utf16_to_utf8(conversionBuffer.begin(), read, outputBuf.begin());
765 }
766 }
767 }
768 }
769 KJ_DASSERT(written <= outputBuf.size());
770 // V8's String::kMaxLenth is a lot less than a maximal int so this is fine.
771 using RInt = decltype(TextEncoder::EncodeIntoResult::read);
772 using WInt = decltype(TextEncoder::EncodeIntoResult::written);
773 KJ_DASSERT(0 <= read && read <= std::numeric_limits<RInt>::max());
774 KJ_DASSERT(0 <= written && written <= std::numeric_limits<WInt>::max());
775 return TextEncoder::EncodeIntoResult{
776 .read = static_cast<RInt>(read),
777 .written = static_cast<WInt>(written),
778 };
779}
780 
781} // namespace workerd::api