File
Blob: src/workerd/api/encoding.c++
| 1 | // Copyright (c) 2017-2022 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | |
| 5 | #include "encoding.h" |
| 6 | |
| 7 | #include "simdutf.h" |
| 8 | #include "util.h" |
| 9 | |
| 10 | #include <workerd/io/features.h> |
| 11 | #include <workerd/jsg/jsg.h> |
| 12 | #include <workerd/util/autogate.h> |
| 13 | #include <workerd/util/strings.h> |
| 14 | |
| 15 | #include <unicode/ucnv.h> |
| 16 | #include <unicode/utf8.h> |
| 17 | #include <v8.h> |
| 18 | |
| 19 | #include <kj/array.h> |
| 20 | #include <kj/string.h> |
| 21 | |
| 22 | namespace workerd::api { |
| 23 | |
| 24 | // ======================================================================================= |
| 25 | // TextDecoder implementation |
| 26 | |
| 27 | namespace { |
| 28 | #define EW_ENCODING_LABELS(V) \ |
| 29 | V("unicode-1-1-utf-8", Utf8) \ |
| 30 | V("unicode11utf8", Utf8) \ |
| 31 | V("unicode20utf8", Utf8) \ |
| 32 | V("utf-8", Utf8) \ |
| 33 | V("utf8", Utf8) \ |
| 34 | V("x-unicode20utf8", Utf8) \ |
| 35 | V("866", Ibm866) \ |
| 36 | V("cp866", Ibm866) \ |
| 37 | V("csibm866", Ibm866) \ |
| 38 | V("ibm866", Ibm866) \ |
| 39 | V("csisolatin2", Iso8859_2) \ |
| 40 | V("iso-8859-2", Iso8859_2) \ |
| 41 | V("iso-ir-101", Iso8859_2) \ |
| 42 | V("iso8859-2", Iso8859_2) \ |
| 43 | V("iso88592", Iso8859_2) \ |
| 44 | V("iso_8859-2", Iso8859_2) \ |
| 45 | V("iso_8859-2:1987", Iso8859_2) \ |
| 46 | V("l2", Iso8859_2) \ |
| 47 | V("latin2", Iso8859_2) \ |
| 48 | V("csisolatin3", Iso8859_3) \ |
| 49 | V("iso-8859-3", Iso8859_3) \ |
| 50 | V("iso-ir-109", Iso8859_3) \ |
| 51 | V("iso8859-3", Iso8859_3) \ |
| 52 | V("iso88593", Iso8859_3) \ |
| 53 | V("iso_8859-3", Iso8859_3) \ |
| 54 | V("iso_8859-3:1988", Iso8859_3) \ |
| 55 | V("l3", Iso8859_3) \ |
| 56 | V("latin3", Iso8859_3) \ |
| 57 | V("csisolatin4", Iso8859_4) \ |
| 58 | V("iso-8859-4", Iso8859_4) \ |
| 59 | V("iso-ir-110", Iso8859_4) \ |
| 60 | V("iso8859-4", Iso8859_4) \ |
| 61 | V("iso88594", Iso8859_4) \ |
| 62 | V("iso_8859-4", Iso8859_4) \ |
| 63 | V("iso_8859-4:1988", Iso8859_4) \ |
| 64 | V("l4", Iso8859_4) \ |
| 65 | V("latin4", Iso8859_4) \ |
| 66 | V("csisolatincyrillic", Iso8859_5) \ |
| 67 | V("cyrillic", Iso8859_5) \ |
| 68 | V("iso-8859-5", Iso8859_5) \ |
| 69 | V("iso-ir-144", Iso8859_5) \ |
| 70 | V("iso8859-5", Iso8859_5) \ |
| 71 | V("iso88595", Iso8859_5) \ |
| 72 | V("iso_8859-5", Iso8859_5) \ |
| 73 | V("iso_8859-5:1988", Iso8859_5) \ |
| 74 | V("arabic", Iso8859_6) \ |
| 75 | V("asmo-708", Iso8859_6) \ |
| 76 | V("csiso88596e", Iso8859_6) \ |
| 77 | V("csiso88596i", Iso8859_6) \ |
| 78 | V("csisolatinarabic", Iso8859_6) \ |
| 79 | V("ecma-114", Iso8859_6) \ |
| 80 | V("iso-8859-6", Iso8859_6) \ |
| 81 | V("iso-8859-6-e", Iso8859_6) \ |
| 82 | V("iso-8859-6-i", Iso8859_6) \ |
| 83 | V("iso-ir-127", Iso8859_6) \ |
| 84 | V("iso8859-6", Iso8859_6) \ |
| 85 | V("iso88596", Iso8859_6) \ |
| 86 | V("iso_8859-6", Iso8859_6) \ |
| 87 | V("iso_8859-6:1987", Iso8859_6) \ |
| 88 | V("csisolatingreek", Iso8859_7) \ |
| 89 | V("ecma-118", Iso8859_7) \ |
| 90 | V("elot_928", Iso8859_7) \ |
| 91 | V("greek", Iso8859_7) \ |
| 92 | V("greek8", Iso8859_7) \ |
| 93 | V("iso-8859-7", Iso8859_7) \ |
| 94 | V("iso-ir-126", Iso8859_7) \ |
| 95 | V("iso8859-7", Iso8859_7) \ |
| 96 | V("iso88597", Iso8859_7) \ |
| 97 | V("iso_8859-7", Iso8859_7) \ |
| 98 | V("iso_8859-7:1987", Iso8859_7) \ |
| 99 | V("sun_eu_greek", Iso8859_7) \ |
| 100 | V("csiso88598e", Iso8859_8) \ |
| 101 | V("csisolatinhebrew", Iso8859_8) \ |
| 102 | V("hebrew", Iso8859_8) \ |
| 103 | V("iso-8859-8", Iso8859_8) \ |
| 104 | V("iso-8859-8-e", Iso8859_8) \ |
| 105 | V("iso-ir-138", Iso8859_8) \ |
| 106 | V("iso8859-8", Iso8859_8) \ |
| 107 | V("iso88598", Iso8859_8) \ |
| 108 | V("iso_8859-8", Iso8859_8) \ |
| 109 | V("iso_8859-8:1988", Iso8859_8) \ |
| 110 | V("visual", Iso8859_8) \ |
| 111 | V("csiso88598i", Iso8859_8i) \ |
| 112 | V("iso-8859-8-i", Iso8859_8i) \ |
| 113 | V("logical", Iso8859_8i) \ |
| 114 | V("csisolatin6", Iso8859_10) \ |
| 115 | V("iso-8859-10", Iso8859_10) \ |
| 116 | V("iso-ir-157", Iso8859_10) \ |
| 117 | V("iso8859-10", Iso8859_10) \ |
| 118 | V("iso885910", Iso8859_10) \ |
| 119 | V("l6", Iso8859_10) \ |
| 120 | V("latin6", Iso8859_10) \ |
| 121 | V("iso-8859-13", Iso8859_13) \ |
| 122 | V("iso8859-13", Iso8859_13) \ |
| 123 | V("iso885913", Iso8859_13) \ |
| 124 | V("iso-8859-14", Iso8859_14) \ |
| 125 | V("iso8859-14", Iso8859_14) \ |
| 126 | V("iso885914", Iso8859_14) \ |
| 127 | V("csisolatin9", Iso8859_15) \ |
| 128 | V("iso-8859-15", Iso8859_15) \ |
| 129 | V("iso8859-15", Iso8859_15) \ |
| 130 | V("iso885915", Iso8859_15) \ |
| 131 | V("iso_8859-15", Iso8859_15) \ |
| 132 | V("l9", Iso8859_15) \ |
| 133 | V("iso-8859-16", Iso8859_16) \ |
| 134 | V("cskoi8r", Ko18_r) \ |
| 135 | V("koi", Ko18_r) \ |
| 136 | V("koi8", Ko18_r) \ |
| 137 | V("koi8-r", Ko18_r) \ |
| 138 | V("koi8_r", Ko18_r) \ |
| 139 | V("koi8-ru", Koi8_u) \ |
| 140 | V("koi8-u", Koi8_u) \ |
| 141 | V("csmacintosh", Macintosh) \ |
| 142 | V("mac", Macintosh) \ |
| 143 | V("macintosh", Macintosh) \ |
| 144 | V("x-mac-roman", Macintosh) \ |
| 145 | V("dos-874", Windows_874) \ |
| 146 | V("iso-8859-11", Windows_874) \ |
| 147 | V("iso8859-11", Windows_874) \ |
| 148 | V("iso885911", Windows_874) \ |
| 149 | V("tis-620", Windows_874) \ |
| 150 | V("windows-874", Windows_874) \ |
| 151 | V("cp1250", Windows_1250) \ |
| 152 | V("windows-1250", Windows_1250) \ |
| 153 | V("x-cp1250", Windows_1250) \ |
| 154 | V("cp1251", Windows_1251) \ |
| 155 | V("windows-1251", Windows_1251) \ |
| 156 | V("x-cp1251", Windows_1251) \ |
| 157 | V("ansi_x3.4-1968", Windows_1252) \ |
| 158 | V("ascii", Windows_1252) \ |
| 159 | V("cp1252", Windows_1252) \ |
| 160 | V("cp819", Windows_1252) \ |
| 161 | V("csisolatin1", Windows_1252) \ |
| 162 | V("ibm819", Windows_1252) \ |
| 163 | V("iso-8859-1", Windows_1252) \ |
| 164 | V("iso-ir-100", Windows_1252) \ |
| 165 | V("iso8859-1", Windows_1252) \ |
| 166 | V("iso88591", Windows_1252) \ |
| 167 | V("iso_8859-1", Windows_1252) \ |
| 168 | V("iso_8859-1:1987", Windows_1252) \ |
| 169 | V("l1", Windows_1252) \ |
| 170 | V("latin1", Windows_1252) \ |
| 171 | V("us-ascii", Windows_1252) \ |
| 172 | V("windows-1252", Windows_1252) \ |
| 173 | V("x-cp1252", Windows_1252) \ |
| 174 | V("cp1253", Windows_1253) \ |
| 175 | V("windows-1253", Windows_1253) \ |
| 176 | V("x-cp1253", Windows_1253) \ |
| 177 | V("cp1254", Windows_1254) \ |
| 178 | V("csisolatin5", Windows_1254) \ |
| 179 | V("iso-8859-9", Windows_1254) \ |
| 180 | V("iso-ir-148", Windows_1254) \ |
| 181 | V("iso8859-9", Windows_1254) \ |
| 182 | V("iso88599", Windows_1254) \ |
| 183 | V("iso_8859-9", Windows_1254) \ |
| 184 | V("iso_8859-9:1989", Windows_1254) \ |
| 185 | V("l5", Windows_1254) \ |
| 186 | V("latin5", Windows_1254) \ |
| 187 | V("windows-1254", Windows_1254) \ |
| 188 | V("x-cp1254", Windows_1254) \ |
| 189 | V("cp1255", Windows_1255) \ |
| 190 | V("windows-1255", Windows_1255) \ |
| 191 | V("x-cp1255", Windows_1255) \ |
| 192 | V("cp1256", Windows_1256) \ |
| 193 | V("windows-1256", Windows_1256) \ |
| 194 | V("x-cp1256", Windows_1256) \ |
| 195 | V("cp1257", Windows_1257) \ |
| 196 | V("windows-1257", Windows_1257) \ |
| 197 | V("x-cp1257", Windows_1257) \ |
| 198 | V("cp1258", Windows_1258) \ |
| 199 | V("windows-1258", Windows_1258) \ |
| 200 | V("x-cp1258", Windows_1258) \ |
| 201 | V("x-mac-cyrillic", X_Mac_Cyrillic) \ |
| 202 | V("x-mac-ukrainian", X_Mac_Cyrillic) \ |
| 203 | V("chinese", Gbk) \ |
| 204 | V("csgb2312", Gbk) \ |
| 205 | V("csiso58gb231280", Gbk) \ |
| 206 | V("gb2312", Gbk) \ |
| 207 | V("gb_2312", Gbk) \ |
| 208 | V("gb_2312-80", Gbk) \ |
| 209 | V("gbk", Gbk) \ |
| 210 | V("iso-ir-58", Gbk) \ |
| 211 | V("x-gbk", Gbk) \ |
| 212 | V("gb18030", Gb18030) \ |
| 213 | V("big5", Big5) \ |
| 214 | V("big5-hkscs", Big5) \ |
| 215 | V("cn-big5", Big5) \ |
| 216 | V("csbig5", Big5) \ |
| 217 | V("x-x-big5", Big5) \ |
| 218 | V("cseucpkdfmtjapanese", Euc_Jp) \ |
| 219 | V("euc-jp", Euc_Jp) \ |
| 220 | V("x-euc-jp", Euc_Jp) \ |
| 221 | V("csiso2022jp", Iso2022_Jp) \ |
| 222 | V("iso-2022-jp", Iso2022_Jp) \ |
| 223 | V("csshiftjis", Shift_Jis) \ |
| 224 | V("ms932", Shift_Jis) \ |
| 225 | V("ms_kanji", Shift_Jis) \ |
| 226 | V("shift-jis", Shift_Jis) \ |
| 227 | V("shift_jis", Shift_Jis) \ |
| 228 | V("sjis", Shift_Jis) \ |
| 229 | V("windows-31j", Shift_Jis) \ |
| 230 | V("x-sjis", Shift_Jis) \ |
| 231 | V("cseuckr", Euc_Kr) \ |
| 232 | V("csksc56011987", Euc_Kr) \ |
| 233 | V("euc-kr", Euc_Kr) \ |
| 234 | V("iso-ir-149", Euc_Kr) \ |
| 235 | V("korean", Euc_Kr) \ |
| 236 | V("ks_c_5601-1987", Euc_Kr) \ |
| 237 | V("ks_c_5601-1989", Euc_Kr) \ |
| 238 | V("ksc5601", Euc_Kr) \ |
| 239 | V("ksc_5601", Euc_Kr) \ |
| 240 | V("windows-949", Euc_Kr) \ |
| 241 | V("csiso2022kr", Replacement) \ |
| 242 | V("hz-gb-2312", Replacement) \ |
| 243 | V("iso-2022-cn", Replacement) \ |
| 244 | V("iso-2022-cn-ext", Replacement) \ |
| 245 | V("iso-2022-kr", Replacement) \ |
| 246 | V("replacement", Replacement) \ |
| 247 | V("unicodefffe", Utf16be) \ |
| 248 | V("utf-16be", Utf16be) \ |
| 249 | V("csunicode", Utf16le) \ |
| 250 | V("iso-10646-ucs-2", Utf16le) \ |
| 251 | V("ucs-2", Utf16le) \ |
| 252 | V("unicode", Utf16le) \ |
| 253 | V("unicodefeff", Utf16le) \ |
| 254 | V("utf-16", Utf16le) \ |
| 255 | V("utf-16le", Utf16le) \ |
| 256 | V("x-user-defined", X_User_Defined) |
| 257 | |
| 258 | kj::StringPtr getEncodingId(Encoding encoding) { |
| 259 | switch (encoding) { |
| 260 | case Encoding::INVALID: |
| 261 | return "invalid"_kj; |
| 262 | #define V(name, id) \ |
| 263 | case Encoding::name: \ |
| 264 | return id##_kj; |
| 265 | EW_ENCODINGS(V) |
| 266 | #undef V |
| 267 | } |
| 268 | KJ_UNREACHABLE; |
| 269 | } |
| 270 | |
| 271 | Encoding getEncodingForLabel(kj::StringPtr label) { |
| 272 | auto lower = toLower(label); |
| 273 | auto trimmed = trimLeadingAndTrailingWhitespace(lower); |
| 274 | #define V(label, key) \ |
| 275 | if (trimmed == label##_kjb) return Encoding::key; |
| 276 | EW_ENCODING_LABELS(V) |
| 277 | #undef V |
| 278 | return Encoding::INVALID; |
| 279 | } |
| 280 | |
| 281 | constexpr int MAX_SIZE_FOR_STACK_ALLOC = 4096; |
| 282 | |
| 283 | } // namespace |
| 284 | |
| 285 | const kj::Array<const kj::byte> TextDecoder::EMPTY = |
| 286 | kj::Array<const kj::byte>(&DUMMY, 0, kj::NullArrayDisposer::instance); |
| 287 | const TextDecoder::DecodeOptions TextDecoder::DEFAULT_OPTIONS = TextDecoder::DecodeOptions(); |
| 288 | |
| 289 | kj::Maybe<IcuDecoder> IcuDecoder::create(Encoding encoding, bool fatal, bool ignoreBom) { |
| 290 | UErrorCode status = U_ZERO_ERROR; |
| 291 | // Per the WHATWG encoding spec (section 10.1.1), GBK's decoder is gb18030's decoder. |
| 292 | // https://encoding.spec.whatwg.org/#gbk-decoder |
| 293 | // We can't change getEncodingId() itself because it is also used for the TextDecoder.encoding |
| 294 | // getter, which must still return "gbk" for GBK. |
| 295 | auto icuEncoding = |
| 296 | encoding == Encoding::Gbk ? getEncodingId(Encoding::Gb18030) : getEncodingId(encoding); |
| 297 | UConverter* inner = ucnv_open(icuEncoding.cStr(), &status); |
| 298 | JSG_REQUIRE(U_SUCCESS(status), RangeError, "Invalid or unsupported encoding"); |
| 299 | |
| 300 | if (fatal) { |
| 301 | status = U_ZERO_ERROR; |
| 302 | ucnv_setToUCallBack(inner, UCNV_TO_U_CALLBACK_STOP, nullptr, nullptr, nullptr, &status); |
| 303 | if (U_FAILURE(status)) return kj::none; |
| 304 | } |
| 305 | |
| 306 | return IcuDecoder(encoding, inner, fatal, ignoreBom); |
| 307 | } |
| 308 | |
| 309 | kj::Maybe<jsg::JsString> IcuDecoder::decode( |
| 310 | jsg::Lock& js, kj::ArrayPtr<const kj::byte> buffer, bool flush) { |
| 311 | UErrorCode status = U_ZERO_ERROR; |
| 312 | const auto maxCharSize = [this]() { return ucnv_getMaxCharSize(inner.get()); }; |
| 313 | |
| 314 | const auto isUnicode = [this]() { |
| 315 | switch (ucnv_getType(inner.get())) { |
| 316 | case UCNV_UTF8: |
| 317 | case UCNV_UTF16: |
| 318 | case UCNV_UTF16_BigEndian: |
| 319 | case UCNV_UTF16_LittleEndian: |
| 320 | return true; |
| 321 | default: |
| 322 | return false; |
| 323 | } |
| 324 | KJ_UNREACHABLE; |
| 325 | }; |
| 326 | |
| 327 | KJ_DEFER({ |
| 328 | if (flush) reset(); |
| 329 | }); |
| 330 | |
| 331 | // Evaluate fast-path options. These provide shortcuts for common cases with the caveat |
| 332 | // that error handling for invalid sequences might be a bit different (because the |
| 333 | // conversions are being handled by v8 directly rather than by the ICU converter). |
| 334 | if (buffer.size() > 0 && ucnv_toUCountPending(inner.get(), &status) == 0) { |
| 335 | KJ_ASSERT(U_SUCCESS(status)); |
| 336 | if (encoding == Encoding::Utf8 && |
| 337 | simdutf::validate_ascii(buffer.asChars().begin(), buffer.size())) { |
| 338 | // This is a fast-path option for UTF-8 that can be taken when there |
| 339 | // are no buffered inputs and the non-empty input buffer contains only |
| 340 | // codepoints <= 0x7f. This path is safe because with ASCII range codepoints |
| 341 | // we know we won't accidentally split a multi-byte encoding. We also don't |
| 342 | // have to worry about the BOM here since the BOM bytes are > 0x7f. |
| 343 | // Note also that in this case we'll interpret as Latin1 since UTF-8 bytes |
| 344 | // within this range are identical to Latin1 and v8 allocates these more |
| 345 | // efficiently. |
| 346 | return js.str(buffer); |
| 347 | } |
| 348 | |
| 349 | if (encoding == Encoding::Utf16le && buffer.size() % sizeof(char16_t) == 0) { |
| 350 | // This is a fast-path option for UTF-16le that can be taken when: |
| 351 | // there are no buffered inputs, the non-empty input buffer length is an |
| 352 | // even multiple of 2, and either flush is true or the last code unit |
| 353 | // is not a Unicode lead surrogate. This is safe because when flush |
| 354 | // is true the converter state will be cleared, and if the last code |
| 355 | // unit is not a lead surrogate, we won't have to worry about possibly |
| 356 | // splitting a valid surrogate pair. |
| 357 | |
| 358 | // The input buffer may be at an odd byte offset (e.g. a Uint8Array view |
| 359 | // at offset 3 into an ArrayBuffer), which makes reinterpret_cast to |
| 360 | // char16_t* undefined behavior due to alignment violation. Copy into an |
| 361 | // aligned buffer to avoid this. |
| 362 | auto bufSize = buffer.size() / 2; |
| 363 | kj::SmallArray<char16_t, 256> aligned(bufSize); |
| 364 | aligned.asBytes().copyFrom(buffer.first(bufSize * 2)); |
| 365 | auto data = aligned.asPtr(); |
| 366 | |
| 367 | if (flush || !U_IS_SURROGATE_LEAD(data[data.size() - 1])) { |
| 368 | bool omitInitialBom = false; |
| 369 | if (!ignoreBom && !bomSeen) { |
| 370 | omitInitialBom = data[0] == 0xfeff; |
| 371 | bomSeen = true; |
| 372 | } |
| 373 | |
| 374 | auto slice = data.slice(omitInitialBom ? 1 : 0, data.size()); |
| 375 | |
| 376 | // If textDecoderReplaceSurrogates flag is enabled, then we follow the spec |
| 377 | // and fix invalid surrogates on the UTF-16 input. |
| 378 | if (slice.size() == 0 || !FeatureFlags::get(js).getTextDecoderReplaceSurrogates()) { |
| 379 | return js.str(slice); |
| 380 | } |
| 381 | |
| 382 | if (simdutf::validate_utf16(slice.begin(), slice.size())) { |
| 383 | return js.str(slice); |
| 384 | } |
| 385 | |
| 386 | if (fatal) { |
| 387 | // In fatal mode, return error for invalid surrogates |
| 388 | return kj::none; |
| 389 | } |
| 390 | |
| 391 | // In non-fatal mode, replace invalid surrogates with U+FFFD. |
| 392 | // Output size equals input size because each invalid surrogate (1 code unit) |
| 393 | // is replaced with U+FFFD (also 1 code unit). |
| 394 | // Use stack allocation for small strings (up to 256 code units) to avoid |
| 395 | // heap allocation overhead. |
| 396 | kj::SmallArray<char16_t, 256> fixed(slice.size()); |
| 397 | simdutf::to_well_formed_utf16(slice.begin(), slice.size(), fixed.begin()); |
| 398 | return js.str(fixed.asPtr()); |
| 399 | } |
| 400 | } |
| 401 | } |
| 402 | |
| 403 | status = U_ZERO_ERROR; |
| 404 | auto limit = 2 * maxCharSize() * |
| 405 | (!flush ? buffer.size() |
| 406 | : kj::max(buffer.size(), |
| 407 | static_cast<size_t>(ucnv_toUCountPending(inner.get(), &status)))); |
| 408 | |
| 409 | KJ_STACK_ARRAY(UChar, result, limit, 512, 4096); |
| 410 | |
| 411 | auto dest = result.begin(); |
| 412 | auto source = reinterpret_cast<const char*>(buffer.begin()); |
| 413 | |
| 414 | ucnv_toUnicode( |
| 415 | inner.get(), &dest, dest + limit, &source, source + buffer.size(), nullptr, flush, &status); |
| 416 | |
| 417 | if (U_FAILURE(status)) return kj::none; |
| 418 | |
| 419 | auto omitInitialBom = false; |
| 420 | auto length = std::distance(result.begin(), dest); |
| 421 | if (length > 0 && isUnicode() && !ignoreBom && !bomSeen) { |
| 422 | omitInitialBom = result[0] == 0xfeff; |
| 423 | bomSeen = true; |
| 424 | } |
| 425 | |
| 426 | return js.str(result.slice(omitInitialBom ? 1 : 0, length)); |
| 427 | } |
| 428 | |
| 429 | void IcuDecoder::reset() { |
| 430 | bomSeen = false; |
| 431 | return ucnv_reset(inner.get()); |
| 432 | } |
| 433 | |
| 434 | Decoder& TextDecoder::getImpl() { |
| 435 | KJ_SWITCH_ONEOF(decoder) { |
| 436 | KJ_CASE_ONEOF(dec, LegacyDecoder) { |
| 437 | return dec; |
| 438 | } |
| 439 | KJ_CASE_ONEOF(dec, IcuDecoder) { |
| 440 | return dec; |
| 441 | } |
| 442 | } |
| 443 | KJ_UNREACHABLE; |
| 444 | } |
| 445 | |
| 446 | jsg::Ref<TextDecoder> TextDecoder::constructor(jsg::Lock& js, |
| 447 | jsg::Optional<kj::String> maybeLabel, |
| 448 | jsg::Optional<ConstructorOptions> maybeOptions) { |
| 449 | static constexpr ConstructorOptions DEFAULT_OPTIONS; |
| 450 | auto options = maybeOptions.orDefault(DEFAULT_OPTIONS); |
| 451 | auto encoding = Encoding::Utf8; |
| 452 | |
| 453 | const auto errorMessage = [](kj::StringPtr label) { |
| 454 | return kj::str("\"", label, "\" is not a valid encoding."); |
| 455 | }; |
| 456 | |
| 457 | KJ_IF_SOME(label, maybeLabel) { |
| 458 | encoding = getEncodingForLabel(label); |
| 459 | JSG_REQUIRE(encoding != Encoding::Replacement && encoding != Encoding::INVALID, RangeError, |
| 460 | errorMessage(label)); |
| 461 | } |
| 462 | |
| 463 | switch (encoding) { |
| 464 | case Encoding::Big5: |
| 465 | case Encoding::Euc_Jp: |
| 466 | case Encoding::Euc_Kr: |
| 467 | case Encoding::Gb18030: |
| 468 | case Encoding::Gbk: |
| 469 | case Encoding::Iso2022_Jp: |
| 470 | case Encoding::Shift_Jis: { |
| 471 | // If the feature flag is disabled, we use the ICU decoder. |
| 472 | if (!FeatureFlags::get(js).getTextDecoderCjkDecoder()) { |
| 473 | break; |
| 474 | } |
| 475 | |
| 476 | // We fallthrough to LegacyDecoder in order to avoid breaking changes. |
| 477 | [[fallthrough]]; |
| 478 | } |
| 479 | case Encoding::X_User_Defined: |
| 480 | case Encoding::Windows_1252: |
| 481 | return js.alloc<TextDecoder>(LegacyDecoder(encoding, DecoderFatal(options.fatal)), options); |
| 482 | default: |
| 483 | break; |
| 484 | } |
| 485 | |
| 486 | return js.alloc<TextDecoder>( |
| 487 | JSG_REQUIRE_NONNULL(IcuDecoder::create(encoding, options.fatal, options.ignoreBOM), |
| 488 | RangeError, errorMessage(getEncodingId(encoding))), |
| 489 | options); |
| 490 | } |
| 491 | |
| 492 | kj::StringPtr TextDecoder::getEncoding() { |
| 493 | return getEncodingId(getImpl().getEncoding()); |
| 494 | } |
| 495 | |
| 496 | jsg::JsString TextDecoder::decode(jsg::Lock& js, |
| 497 | jsg::Optional<kj::Array<const kj::byte>> maybeInput, |
| 498 | jsg::Optional<DecodeOptions> maybeOptions) { |
| 499 | auto options = maybeOptions.orDefault(DEFAULT_OPTIONS); |
| 500 | auto& input = maybeInput.orDefault(EMPTY); |
| 501 | return JSG_REQUIRE_NONNULL( |
| 502 | getImpl().decode(js, input, !options.stream), TypeError, "Failed to decode input."); |
| 503 | } |
| 504 | |
| 505 | kj::Maybe<jsg::JsString> TextDecoder::decodePtr( |
| 506 | jsg::Lock& js, kj::ArrayPtr<const kj::byte> buffer, bool flush) { |
| 507 | KJ_SWITCH_ONEOF(decoder) { |
| 508 | KJ_CASE_ONEOF(dec, LegacyDecoder) { |
| 509 | return dec.decode(js, buffer, flush); |
| 510 | } |
| 511 | KJ_CASE_ONEOF(dec, IcuDecoder) { |
| 512 | return dec.decode(js, buffer, flush); |
| 513 | } |
| 514 | } |
| 515 | KJ_UNREACHABLE; |
| 516 | } |
| 517 | |
| 518 | // ======================================================================================= |
| 519 | // TextEncoder implementation |
| 520 | |
| 521 | jsg::Ref<TextEncoder> TextEncoder::constructor(jsg::Lock& js) { |
| 522 | return js.alloc<TextEncoder>(); |
| 523 | } |
| 524 | |
| 525 | jsg::JsUint8Array TextEncoder::encode(jsg::Lock& js, jsg::Optional<jsg::JsString> input) { |
| 526 | if (!workerd::util::Autogate::isEnabled(workerd::util::AutogateKey::ENABLE_FAST_TEXTENCODER)) { |
| 527 | auto str = input.orDefault(js.str()); |
| 528 | auto view = jsg::JsUint8Array::create(js, str.utf8Length(js)); |
| 529 | [[maybe_unused]] auto result = str.writeInto( |
| 530 | js, view.asArrayPtr().asChars(), jsg::JsString::WriteFlags::REPLACE_INVALID_UTF8); |
| 531 | KJ_DASSERT(result.written == view.size()); |
| 532 | return view; |
| 533 | } |
| 534 | |
| 535 | jsg::JsString str = input.orDefault(js.str()); |
| 536 | |
| 537 | size_t utf8_length = 0; |
| 538 | auto length = str.length(js); |
| 539 | |
| 540 | #ifdef KJ_DEBUG |
| 541 | bool wasAlreadyFlat = str.isFlat(); |
| 542 | KJ_DEFER({ KJ_ASSERT(wasAlreadyFlat || !str.isFlat()); }); |
| 543 | #endif |
| 544 | |
| 545 | // Note: writeInto() doesn't flatten the string - it calls writeTo() which chains through |
| 546 | // Write2 -> WriteV2 -> WriteHelperV2 -> String::WriteToFlat. |
| 547 | // This means we may read from multiple string segments, but that's fine for our use case. |
| 548 | |
| 549 | if (str.isOneByte(js)) { |
| 550 | // Use off-heap allocation for intermediate Latin-1 buffer to avoid wasting V8 heap space |
| 551 | // and potentially triggering GC. Stack allocation for small strings, heap for large. |
| 552 | kj::SmallArray<kj::byte, MAX_SIZE_FOR_STACK_ALLOC> latin1Buffer(length); |
| 553 | |
| 554 | [[maybe_unused]] auto writeResult = str.writeInto(js, latin1Buffer.asPtr()); |
| 555 | KJ_DASSERT( |
| 556 | writeResult.written == length, "writeInto must completely overwrite the backing buffer"); |
| 557 | |
| 558 | utf8_length = simdutf::utf8_length_from_latin1( |
| 559 | reinterpret_cast<const char*>(latin1Buffer.begin()), length); |
| 560 | |
| 561 | auto view = jsg::JsUint8Array::create(js, utf8_length); |
| 562 | if (utf8_length == length) { |
| 563 | // ASCII fast path: no conversion needed, Latin-1 is same as UTF-8 for ASCII |
| 564 | view.asArrayPtr().copyFrom(latin1Buffer); |
| 565 | } else { |
| 566 | [[maybe_unused]] auto written = |
| 567 | simdutf::convert_latin1_to_utf8(reinterpret_cast<const char*>(latin1Buffer.begin()), |
| 568 | length, view.asArrayPtr().asChars().begin()); |
| 569 | KJ_DASSERT(utf8_length == written); |
| 570 | } |
| 571 | return view; |
| 572 | } |
| 573 | |
| 574 | // Use off-heap allocation for intermediate UTF-16 buffer to avoid wasting V8 heap space |
| 575 | // and potentially triggering GC. Stack allocation for small strings, heap for large. |
| 576 | // Stack allocation for small strings, heap for large. |
| 577 | kj::SmallArray<uint16_t, MAX_SIZE_FOR_STACK_ALLOC> utf16Buffer(length); |
| 578 | |
| 579 | [[maybe_unused]] auto writeResult = str.writeInto(js, utf16Buffer.asPtr()); |
| 580 | KJ_DASSERT( |
| 581 | writeResult.written == length, "writeInto must completely overwrite the backing buffer"); |
| 582 | |
| 583 | auto data = reinterpret_cast<char16_t*>(utf16Buffer.begin()); |
| 584 | auto lengthResult = simdutf::utf8_length_from_utf16_with_replacement(data, length); |
| 585 | utf8_length = lengthResult.count; |
| 586 | |
| 587 | if (lengthResult.error == simdutf::SURROGATE) { |
| 588 | // If there are surrogates there may be unpaired surrogates. Fix them. |
| 589 | simdutf::to_well_formed_utf16(data, length, data); |
| 590 | } else { |
| 591 | KJ_DASSERT(lengthResult.error == simdutf::SUCCESS); |
| 592 | } |
| 593 | |
| 594 | auto view = jsg::JsUint8Array::create(js, utf8_length); |
| 595 | [[maybe_unused]] auto written = |
| 596 | simdutf::convert_utf16_to_utf8(data, length, view.asArrayPtr().asChars().begin()); |
| 597 | KJ_DASSERT(written == utf8_length, "Conversion yielded wrong number of UTF-8 bytes"); |
| 598 | return view; |
| 599 | } |
| 600 | |
| 601 | namespace { |
| 602 | |
| 603 | constexpr bool isSurrogatePair(uint16_t lead, uint16_t trail) { |
| 604 | // We would like to use simdutf::trim_partial_utf16, but it's not guaranteed |
| 605 | // to work right on invalid UTF-16. Hence, we need this method to check for |
| 606 | // surrogate pairs and correctly trim utf16 chunks. |
| 607 | return (lead & 0xfc00) == 0xd800 && (trail & 0xfc00) == 0xdc00; |
| 608 | } |
| 609 | |
| 610 | // Ignores surrogates conservatively. |
| 611 | constexpr size_t simpleUtfEncodingLength(uint16_t c) { |
| 612 | return 1 + (c >= 0x80) + (c >= 0x400); |
| 613 | } |
| 614 | |
| 615 | // Find how many UTF-16 or Latin1 code units fit when converted to UTF-8. |
| 616 | // May conservatively underestimate the largest number of code units we can fit |
| 617 | // because of undetected surrogate pairs on boundaries. |
| 618 | // Works even on malformed UTF-16. |
| 619 | template <typename Char> |
| 620 | size_t findBestFit(const Char* data, size_t length, size_t bufferSize) { |
| 621 | size_t pos = 0; |
| 622 | size_t utf8Accumulated = 0; |
| 623 | // The SIMD is more efficient with a size that's a little over a multiple of 16. |
| 624 | constexpr size_t CHUNK = 257; |
| 625 | // The max number of UTF-8 output bytes per input code unit. |
| 626 | constexpr bool UTF16 = sizeof(Char) == 2; |
| 627 | constexpr size_t MAX_FACTOR = UTF16 ? 3 : 2; |
| 628 | |
| 629 | // Our initial guess at how much the number of elements expands in the |
| 630 | // conversion to UTF-8. |
| 631 | double expansion = 1.15; |
| 632 | |
| 633 | while (pos < length && utf8Accumulated < bufferSize) { |
| 634 | size_t remainingInput = length - pos; |
| 635 | size_t spaceRemaining = bufferSize - utf8Accumulated; |
| 636 | KJ_DASSERT(expansion >= 1.15); |
| 637 | |
| 638 | // We estimate how many characters are likely to fit in the buffer, but |
| 639 | // only try for CHUNK characters at a time to minimize the worst case |
| 640 | // waste of time if we guessed too high. |
| 641 | size_t guaranteedToFit = spaceRemaining / MAX_FACTOR; |
| 642 | if (guaranteedToFit >= remainingInput) { |
| 643 | // Don't even bother checking any more, it's all going to fit. Hitting |
| 644 | // this halfway through is also a good reason to limit the CHUNK size. |
| 645 | return length; |
| 646 | } |
| 647 | size_t likelyToFit = kj::min(static_cast<size_t>(spaceRemaining / expansion), CHUNK); |
| 648 | size_t fitEstimate = kj::max(1, kj::max(guaranteedToFit, likelyToFit)); |
| 649 | size_t chunkSize = kj::min(remainingInput, fitEstimate); |
| 650 | if (chunkSize == 1) break; // Not worth running this complicated stuff one char at a time. |
| 651 | // No div-by-zero because remainingInput and fitEstimate are at least 1. |
| 652 | KJ_DASSERT(chunkSize >= 1); |
| 653 | |
| 654 | size_t chunkUtf8Len; |
| 655 | if constexpr (UTF16) { |
| 656 | chunkUtf8Len = simdutf::utf8_length_from_utf16_with_replacement(data + pos, chunkSize).count; |
| 657 | } else { |
| 658 | chunkUtf8Len = simdutf::utf8_length_from_latin1(data + pos, chunkSize); |
| 659 | } |
| 660 | |
| 661 | if (utf8Accumulated + chunkUtf8Len > bufferSize) { |
| 662 | // Our chosen chunk didn't fit in the rest of the output buffer. |
| 663 | KJ_DASSERT(chunkSize > guaranteedToFit); |
| 664 | // Since it didn't fit we adjust our expansion guess upwards. |
| 665 | expansion = kj::max(expansion * 1.1, (chunkUtf8Len * 1.1) / chunkSize); |
| 666 | } else { |
| 667 | // Use successful length calculation to adjust our expansion estimate. |
| 668 | expansion = kj::max(1.15, (chunkUtf8Len * 1.1) / chunkSize); |
| 669 | pos += chunkSize; |
| 670 | utf8Accumulated += chunkUtf8Len; |
| 671 | } |
| 672 | } |
| 673 | // Do the last few code units in a simpler way. |
| 674 | while (pos < length && utf8Accumulated < bufferSize) { |
| 675 | size_t extra = simpleUtfEncodingLength(data[pos]); |
| 676 | if (utf8Accumulated + extra > bufferSize) break; |
| 677 | pos++; |
| 678 | utf8Accumulated += extra; |
| 679 | } |
| 680 | if (UTF16 && pos != 0 && pos != length && isSurrogatePair(data[pos - 1], data[pos])) { |
| 681 | // We ended on a leading surrogate which has a matching trailing surrogate in the next |
| 682 | // position. In order to make progress when the bufferSize is tiny we try to include it. |
| 683 | if (utf8Accumulated < bufferSize) { |
| 684 | pos++; // We had one more byte, so we can include the pair, UTF-8 encoding 3->4. |
| 685 | } else { |
| 686 | pos--; // Don't chop the pair in half. |
| 687 | } |
| 688 | } |
| 689 | return pos; |
| 690 | } |
| 691 | |
| 692 | } // namespace |
| 693 | |
| 694 | // Test helpers used by encoding-test.c++ to verify findBestFit behavior. |
| 695 | namespace test { |
| 696 | |
| 697 | size_t bestFit(const char* str, size_t bufferSize) { |
| 698 | return findBestFit(str, strlen(str), bufferSize); |
| 699 | } |
| 700 | |
| 701 | size_t bestFit(const char16_t* str, size_t bufferSize) { |
| 702 | size_t length = 0; |
| 703 | while (str[length] != 0) length++; |
| 704 | return findBestFit(str, length, bufferSize); |
| 705 | } |
| 706 | |
| 707 | } // namespace test |
| 708 | |
| 709 | TextEncoder::EncodeIntoResult TextEncoder::encodeInto( |
| 710 | jsg::Lock& js, jsg::JsString input, jsg::JsUint8Array buffer) { |
| 711 | if (!workerd::util::Autogate::isEnabled(workerd::util::AutogateKey::ENABLE_FAST_TEXTENCODER)) { |
| 712 | auto result = input.writeInto( |
| 713 | js, buffer.asArrayPtr<char>(), jsg::JsString::WriteFlags::REPLACE_INVALID_UTF8); |
| 714 | return TextEncoder::EncodeIntoResult{ |
| 715 | .read = static_cast<int>(result.read), |
| 716 | .written = static_cast<int>(result.written), |
| 717 | }; |
| 718 | } |
| 719 | |
| 720 | auto outputBuf = buffer.asArrayPtr<char>(); |
| 721 | size_t bufferSize = outputBuf.size(); |
| 722 | |
| 723 | size_t read = 0; |
| 724 | size_t written = 0; |
| 725 | { |
| 726 | // Scope for the view - we can't do anything that might cause a V8 GC! |
| 727 | v8::String::ValueView view(js.v8Isolate, input); |
| 728 | size_t length = view.length(); |
| 729 | |
| 730 | if (view.is_one_byte()) { |
| 731 | auto data = reinterpret_cast<const char*>(view.data8()); |
| 732 | simdutf::result result = |
| 733 | simdutf::validate_ascii_with_errors(data, kj::min(length, bufferSize)); |
| 734 | written = read = result.count; |
| 735 | auto outAddr = outputBuf.begin(); |
| 736 | kj::arrayPtr(outAddr, read).copyFrom(kj::arrayPtr(data, read)); |
| 737 | outAddr += read; |
| 738 | data += read; |
| 739 | length -= read; |
| 740 | bufferSize -= read; |
| 741 | if (length != 0 && bufferSize != 0) { |
| 742 | size_t rest = findBestFit(data, length, bufferSize); |
| 743 | if (rest != 0) { |
| 744 | KJ_DASSERT(simdutf::utf8_length_from_latin1(data, rest) <= bufferSize); |
| 745 | written += simdutf::convert_latin1_to_utf8(data, rest, outAddr); |
| 746 | read += rest; |
| 747 | } |
| 748 | } |
| 749 | } else { |
| 750 | auto data = reinterpret_cast<const char16_t*>(view.data16()); |
| 751 | read = findBestFit(data, length, bufferSize); |
| 752 | if (read != 0) { |
| 753 | KJ_DASSERT( |
| 754 | simdutf::utf8_length_from_utf16_with_replacement(data, read).count <= bufferSize); |
| 755 | simdutf::result result = |
| 756 | simdutf::convert_utf16_to_utf8_with_errors(data, read, outputBuf.begin()); |
| 757 | if (result.error == simdutf::SUCCESS) { |
| 758 | written = result.count; |
| 759 | } else { |
| 760 | // Oh, no, there are unpaired surrogates. This is hopefully rare. |
| 761 | kj::SmallArray<char16_t, MAX_SIZE_FOR_STACK_ALLOC> conversionBuffer(read); |
| 762 | simdutf::to_well_formed_utf16(data, read, conversionBuffer.begin()); |
| 763 | written = |
| 764 | simdutf::convert_utf16_to_utf8(conversionBuffer.begin(), read, outputBuf.begin()); |
| 765 | } |
| 766 | } |
| 767 | } |
| 768 | } |
| 769 | KJ_DASSERT(written <= outputBuf.size()); |
| 770 | // V8's String::kMaxLenth is a lot less than a maximal int so this is fine. |
| 771 | using RInt = decltype(TextEncoder::EncodeIntoResult::read); |
| 772 | using WInt = decltype(TextEncoder::EncodeIntoResult::written); |
| 773 | KJ_DASSERT(0 <= read && read <= std::numeric_limits<RInt>::max()); |
| 774 | KJ_DASSERT(0 <= written && written <= std::numeric_limits<WInt>::max()); |
| 775 | return TextEncoder::EncodeIntoResult{ |
| 776 | .read = static_cast<RInt>(read), |
| 777 | .written = static_cast<WInt>(written), |
| 778 | }; |
| 779 | } |
| 780 | |
| 781 | } // namespace workerd::api |