File
Blob: src/workerd/api/tests/encoding-test.js
| 1 | // Copyright (c) 2023 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | import { deepStrictEqual, strictEqual, throws, ok } from 'node:assert'; |
| 5 | |
| 6 | // Test for the Encoding standard Web API implementation. |
| 7 | // The implementation for these are in api/encoding.{h|c++} |
| 8 | |
| 9 | function decodeStreaming(decoder, input) { |
| 10 | // Test truncation behavior while streaming by feeding the decoder a single byte at a time. |
| 11 | // Note we don't try-catch here because we don't expect this ever to fail, because we're |
| 12 | // streaming text. |
| 13 | let x = ''; |
| 14 | for (let i = 0; i < input.length; ++i) { |
| 15 | x += decoder.decode(input.slice(i, i + 1), { stream: true }); |
| 16 | } |
| 17 | x += decoder.decode(); |
| 18 | return x; |
| 19 | } |
| 20 | |
| 21 | // From https://developer.mozilla.org/en-US/docs/Web/API/Encoding_API/Encodings |
| 22 | const windows1252Labels = [ |
| 23 | 'ansi_x3.4-1968', |
| 24 | 'ascii', |
| 25 | 'cp1252', |
| 26 | 'cp819', |
| 27 | 'csisolatin1', |
| 28 | 'ibm819', |
| 29 | 'iso-8859-1', |
| 30 | 'iso-ir-100', |
| 31 | 'iso8859-1', |
| 32 | 'iso88591', |
| 33 | 'iso_8859-1', |
| 34 | 'iso_8859-1:1987', |
| 35 | 'l1', |
| 36 | 'latin1', |
| 37 | 'us-ascii', |
| 38 | 'windows-1252', |
| 39 | 'x-cp1252', |
| 40 | ]; |
| 41 | |
| 42 | const utf8Labels = ['unicode-1-1-utf-8', 'utf-8', 'utf8']; |
| 43 | |
| 44 | export const decodeStreamingTest = { |
| 45 | test() { |
| 46 | let _results = []; |
| 47 | |
| 48 | for (const label of windows1252Labels) { |
| 49 | ok( |
| 50 | new TextDecoder(`${label}`).encoding === 'windows-1252', |
| 51 | `TextDecoder constructed with '${label}' label to have 'windows-1252' encoding.` |
| 52 | ); |
| 53 | } |
| 54 | for (const label of utf8Labels) { |
| 55 | ok( |
| 56 | new TextDecoder(`${label}`).encoding === 'utf-8', |
| 57 | `TextDecoder constructed with '${label}' label to have 'utf-8' encoding.` |
| 58 | ); |
| 59 | } |
| 60 | |
| 61 | const decoder = new TextDecoder(); |
| 62 | const fatalDecoder = new TextDecoder('utf-8', { fatal: true }); |
| 63 | const fatalIgnoreBomDecoder = new TextDecoder('utf-8', { |
| 64 | fatal: true, |
| 65 | ignoreBOM: true, |
| 66 | }); |
| 67 | |
| 68 | ok(decoder.encoding === 'utf-8', "default encoding property to be 'utf-8'"); |
| 69 | ok(decoder.fatal === false, 'default fatal property to be false'); |
| 70 | ok(decoder.ignoreBOM === false, 'default ignoreBOM property to be false'); |
| 71 | |
| 72 | const fooCat = new Uint8Array([102, 111, 111, 32, 240, 159, 152, 186]); |
| 73 | |
| 74 | ok(decoder.decode().length === 0, 'decoded undefined array length to be 0'); |
| 75 | ok( |
| 76 | decoder.decode(fooCat) === 'foo 😺', |
| 77 | 'foo-cat from Uint8Buffer to be foo 😺' |
| 78 | ); |
| 79 | ok( |
| 80 | decoder.decode(fooCat.buffer) === 'foo 😺', |
| 81 | 'foo-cat from ArrayBuffer to be foo 😺' |
| 82 | ); |
| 83 | |
| 84 | const twoByteCodePoint = new Uint8Array([0xc2, 0xa2]); // cent sign |
| 85 | const threeByteCodePoint = new Uint8Array([0xe2, 0x82, 0xac]); // euro sign |
| 86 | const fourByteCodePoint = new Uint8Array([240, 159, 152, 186]); // cat emoji |
| 87 | |
| 88 | [twoByteCodePoint, threeByteCodePoint, fourByteCodePoint].forEach( |
| 89 | (input) => { |
| 90 | // For each input sequence of code units, try decoding each subsequence of code units except the |
| 91 | // full code point itself. |
| 92 | for (let i = 1; i < input.length; ++i) { |
| 93 | const head = input.slice(0, input.length - i); |
| 94 | const tail = input.slice(input.length - i); |
| 95 | |
| 96 | ok( |
| 97 | decoder.decode(head) === '�', |
| 98 | 'code point fragment (head) to be replaced with replacement character' |
| 99 | ); |
| 100 | ok( |
| 101 | decoder.decode(tail) === '�'.repeat(tail.length), |
| 102 | 'code point fragment (tail) to be replaced with replacement character' |
| 103 | ); |
| 104 | |
| 105 | const _errMsg = 'Failed to decode input.'; |
| 106 | |
| 107 | // Exception to be thrown decoding code point fragment (tail) in fatal mode |
| 108 | throws(() => fatalDecoder.decode(head)); |
| 109 | throws(() => fatalDecoder.decode(tail)); |
| 110 | } |
| 111 | } |
| 112 | ); |
| 113 | |
| 114 | // Test ASCII |
| 115 | const asciiDecoder = new TextDecoder('ascii'); |
| 116 | ok( |
| 117 | asciiDecoder.decode().length === 0, |
| 118 | 'decoded undefined array length to be 0' |
| 119 | ); |
| 120 | ok( |
| 121 | asciiDecoder.decode(new Uint8Array([162, 174, 255])) === '¢®ÿ', |
| 122 | 'decoded extended ascii correctly' |
| 123 | ); |
| 124 | |
| 125 | // Test streaming |
| 126 | |
| 127 | ok( |
| 128 | decodeStreaming(fatalDecoder, twoByteCodePoint) === '¢', |
| 129 | '2-byte code point (cent sign) to be decoded correctly' |
| 130 | ); |
| 131 | ok( |
| 132 | decodeStreaming(fatalDecoder, threeByteCodePoint) === '€', |
| 133 | '3-byte code point (euro sign) to be decoded correctly' |
| 134 | ); |
| 135 | ok( |
| 136 | decodeStreaming(fatalDecoder, fourByteCodePoint) === '😺', |
| 137 | '4-byte code point (cat emoji) to be decoded correctly' |
| 138 | ); |
| 139 | |
| 140 | const bom = new Uint8Array([0xef, 0xbb, 0xbf]); |
| 141 | const bomBom = new Uint8Array([0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]); |
| 142 | |
| 143 | ok( |
| 144 | decodeStreaming(fatalDecoder, bom) === '', |
| 145 | 'BOM to be stripped by TextDecoder without ignoreBOM set' |
| 146 | ); |
| 147 | ok( |
| 148 | decodeStreaming(fatalDecoder, bomBom) === '\ufeff', |
| 149 | 'first BOM to be stripped by TextDecoder without ignoreBOM set' |
| 150 | ); |
| 151 | |
| 152 | ok( |
| 153 | decodeStreaming(fatalIgnoreBomDecoder, bom) === '\ufeff', |
| 154 | 'BOM not to be stripped by TextDecoder with ignoreBOM set' |
| 155 | ); |
| 156 | ok( |
| 157 | decodeStreaming(fatalIgnoreBomDecoder, bomBom) === '\ufeff\ufeff', |
| 158 | 'first BOM not to be stripped by TextDecoder with ignoreBOM set' |
| 159 | ); |
| 160 | |
| 161 | const shiftJisDecoder = new TextDecoder('shift_jis'); |
| 162 | let shiftJisResult = ''; |
| 163 | shiftJisResult += shiftJisDecoder.decode(new Uint8Array([0x82]), { |
| 164 | stream: true, |
| 165 | }); |
| 166 | shiftJisResult += shiftJisDecoder.decode(new Uint8Array(), { |
| 167 | stream: true, |
| 168 | }); |
| 169 | shiftJisResult += shiftJisDecoder.decode(new Uint8Array([0xa0]), { |
| 170 | stream: true, |
| 171 | }); |
| 172 | shiftJisResult += shiftJisDecoder.decode(); |
| 173 | ok( |
| 174 | shiftJisResult === 'あ', |
| 175 | 'streaming Shift_JIS should tolerate empty chunks without flushing state' |
| 176 | ); |
| 177 | }, |
| 178 | }; |
| 179 | |
| 180 | export const textEncoderTest = { |
| 181 | test() { |
| 182 | const encoder = new TextEncoder(); |
| 183 | strictEqual(encoder.encoding, 'utf-8'); |
| 184 | strictEqual(encoder.encode().length, 0); |
| 185 | deepStrictEqual( |
| 186 | encoder.encode('foo 😺'), |
| 187 | new Uint8Array([102, 111, 111, 32, 240, 159, 152, 186]) |
| 188 | ); |
| 189 | }, |
| 190 | }; |
| 191 | |
| 192 | export const encodeWptTest = { |
| 193 | test() { |
| 194 | w3cTestEncode(); |
| 195 | w3cTestEncodeInto(); |
| 196 | |
| 197 | function w3cTestEncode() { |
| 198 | const bad = [ |
| 199 | { |
| 200 | input: '\uD800', |
| 201 | expected: '\uFFFD', |
| 202 | name: 'lone surrogate lead', |
| 203 | }, |
| 204 | { |
| 205 | input: '\uDC00', |
| 206 | expected: '\uFFFD', |
| 207 | name: 'lone surrogate trail', |
| 208 | }, |
| 209 | { |
| 210 | input: '\uD800\u0000', |
| 211 | expected: '\uFFFD\u0000', |
| 212 | name: 'unmatched surrogate lead', |
| 213 | }, |
| 214 | { |
| 215 | input: '\uDC00\u0000', |
| 216 | expected: '\uFFFD\u0000', |
| 217 | name: 'unmatched surrogate trail', |
| 218 | }, |
| 219 | { |
| 220 | input: '\uDC00\uD800', |
| 221 | expected: '\uFFFD\uFFFD', |
| 222 | name: 'swapped surrogate pair', |
| 223 | }, |
| 224 | { |
| 225 | input: '\uD834\uDD1E', |
| 226 | expected: '\uD834\uDD1E', |
| 227 | name: 'properly encoded MUSICAL SYMBOL G CLEF (U+1D11E)', |
| 228 | }, |
| 229 | ]; |
| 230 | |
| 231 | bad.forEach(function (t) { |
| 232 | const encoded = new TextEncoder().encode(t.input); |
| 233 | const decoded = new TextDecoder().decode(encoded); |
| 234 | strictEqual(decoded, t.expected); |
| 235 | }); |
| 236 | |
| 237 | strictEqual(new TextEncoder().encode().length, 0); |
| 238 | } |
| 239 | |
| 240 | function w3cTestEncodeInto() { |
| 241 | [ |
| 242 | { |
| 243 | input: 'Hi', |
| 244 | read: 0, |
| 245 | destinationLength: 0, |
| 246 | written: [], |
| 247 | }, |
| 248 | { |
| 249 | input: 'A', |
| 250 | read: 1, |
| 251 | destinationLength: 10, |
| 252 | written: [0x41], |
| 253 | }, |
| 254 | { |
| 255 | input: '\u{1D306}', // "\uD834\uDF06" |
| 256 | read: 2, |
| 257 | destinationLength: 4, |
| 258 | written: [0xf0, 0x9d, 0x8c, 0x86], |
| 259 | }, |
| 260 | { |
| 261 | input: '\u{1D306}A', |
| 262 | read: 0, |
| 263 | destinationLength: 3, |
| 264 | written: [], |
| 265 | }, |
| 266 | { |
| 267 | input: '\uD834A\uDF06A¥Hi', |
| 268 | read: 5, |
| 269 | destinationLength: 10, |
| 270 | written: [0xef, 0xbf, 0xbd, 0x41, 0xef, 0xbf, 0xbd, 0x41, 0xc2, 0xa5], |
| 271 | }, |
| 272 | { |
| 273 | input: 'A\uDF06', |
| 274 | read: 2, |
| 275 | destinationLength: 4, |
| 276 | written: [0x41, 0xef, 0xbf, 0xbd], |
| 277 | }, |
| 278 | { |
| 279 | input: '¥¥', |
| 280 | read: 2, |
| 281 | destinationLength: 4, |
| 282 | written: [0xc2, 0xa5, 0xc2, 0xa5], |
| 283 | }, |
| 284 | ].forEach((testData) => { |
| 285 | [ |
| 286 | { |
| 287 | bufferIncrease: 0, |
| 288 | destinationOffset: 0, |
| 289 | filler: 0, |
| 290 | }, |
| 291 | { |
| 292 | bufferIncrease: 10, |
| 293 | destinationOffset: 4, |
| 294 | filler: 0, |
| 295 | }, |
| 296 | { |
| 297 | bufferIncrease: 0, |
| 298 | destinationOffset: 0, |
| 299 | filler: 0x80, |
| 300 | }, |
| 301 | { |
| 302 | bufferIncrease: 10, |
| 303 | destinationOffset: 4, |
| 304 | filler: 0x80, |
| 305 | }, |
| 306 | { |
| 307 | bufferIncrease: 0, |
| 308 | destinationOffset: 0, |
| 309 | filler: 'random', |
| 310 | }, |
| 311 | { |
| 312 | bufferIncrease: 10, |
| 313 | destinationOffset: 4, |
| 314 | filler: 'random', |
| 315 | }, |
| 316 | ].forEach((destinationData) => { |
| 317 | const bufferLength = |
| 318 | testData.destinationLength + destinationData.bufferIncrease; |
| 319 | const destinationOffset = destinationData.destinationOffset; |
| 320 | const destinationLength = testData.destinationLength; |
| 321 | const destinationFiller = destinationData.filler; |
| 322 | const encoder = new TextEncoder(); |
| 323 | const buffer = new ArrayBuffer(bufferLength); |
| 324 | const view = new Uint8Array( |
| 325 | buffer, |
| 326 | destinationOffset, |
| 327 | destinationLength |
| 328 | ); |
| 329 | const fullView = new Uint8Array(buffer); |
| 330 | const control = Array.from({ length: bufferLength }, () => 0); |
| 331 | let byte = destinationFiller; |
| 332 | for (let i = 0; i < bufferLength; i++) { |
| 333 | if (destinationFiller === 'random') { |
| 334 | byte = Math.floor(Math.random() * 256); |
| 335 | } |
| 336 | control[i] = byte; |
| 337 | fullView[i] = byte; |
| 338 | } |
| 339 | |
| 340 | // It's happening |
| 341 | const result = encoder.encodeInto(testData.input, view); |
| 342 | |
| 343 | // Basics |
| 344 | strictEqual(view.byteLength, destinationLength); |
| 345 | strictEqual(view.length, destinationLength); |
| 346 | |
| 347 | // Remainder |
| 348 | strictEqual(result.read, testData.read); |
| 349 | strictEqual(result.written, testData.written.length); |
| 350 | for (let i = 0; i < bufferLength; i++) { |
| 351 | if ( |
| 352 | i < destinationOffset || |
| 353 | i >= destinationOffset + testData.written.length |
| 354 | ) { |
| 355 | strictEqual(fullView[i], control[i]); |
| 356 | } else { |
| 357 | strictEqual(fullView[i], testData.written[i - destinationOffset]); |
| 358 | } |
| 359 | } |
| 360 | }); |
| 361 | }); |
| 362 | |
| 363 | [ |
| 364 | DataView, |
| 365 | Int8Array, |
| 366 | Int16Array, |
| 367 | Int32Array, |
| 368 | Uint16Array, |
| 369 | Uint32Array, |
| 370 | Uint8ClampedArray, |
| 371 | Float16Array, |
| 372 | Float32Array, |
| 373 | Float64Array, |
| 374 | ].forEach((view) => { |
| 375 | const enc = new TextEncoder(); |
| 376 | throws(() => enc.encodeInto('', new view(new ArrayBuffer(0)))); |
| 377 | }); |
| 378 | |
| 379 | { |
| 380 | const enc = new TextEncoder(); |
| 381 | throws(() => enc.encodeInto('', new ArrayBuffer(0))); |
| 382 | } |
| 383 | } |
| 384 | }, |
| 385 | }; |
| 386 | |
| 387 | export const big5 = { |
| 388 | test() { |
| 389 | // Input is the Big5 encoding for the word 中國人 (meaning Chinese Person) |
| 390 | // Check is the UTF-8 encoding for the same word. |
| 391 | const input = new Uint8Array([0xa4, 0xa4, 0xb0, 0xea, 0xa4, 0x48]); |
| 392 | const check = new Uint8Array([ |
| 393 | 0xe4, 0xb8, 0xad, 0xe5, 0x9c, 0x8b, 0xe4, 0xba, 0xba, |
| 394 | ]); |
| 395 | const enc = new TextEncoder(); |
| 396 | const dec = new TextDecoder('big5', { ignoreBOM: true }); |
| 397 | const result = enc.encode(dec.decode(input)); |
| 398 | strictEqual(result.length, check.length); |
| 399 | for (let n = 0; n < result.length; n++) { |
| 400 | strictEqual(result[n], check[n]); |
| 401 | } |
| 402 | }, |
| 403 | }; |
| 404 | |
| 405 | export const utf16leFastTrack = { |
| 406 | test() { |
| 407 | // Input is the UTF-16le encoding for the word Hello. This should trigger the fast-path |
| 408 | // which should handle the encoding with no problems. Here we only test that the results |
| 409 | // are expected. We cannot verify here that the fast track is actually used. |
| 410 | const input = new Uint8Array([ |
| 411 | 0x68, 0x00, 0x65, 0x00, 0x6c, 0x00, 0x6c, 0x00, 0x6f, 0x00, |
| 412 | ]); |
| 413 | const dec = new TextDecoder('utf-16le'); |
| 414 | strictEqual(dec.decode(input), 'hello'); |
| 415 | }, |
| 416 | }; |
| 417 | |
| 418 | export const allTheDecoders = { |
| 419 | test() { |
| 420 | [ |
| 421 | ['unicode-1-1-utf-8', 'utf-8'], |
| 422 | ['unicode11utf8', 'utf-8'], |
| 423 | ['unicode20utf8', 'utf-8'], |
| 424 | ['utf-8', 'utf-8'], |
| 425 | ['utf8', 'utf-8'], |
| 426 | ['x-unicode20utf8', 'utf-8'], |
| 427 | ['866', 'ibm866'], |
| 428 | ['cp866', 'ibm866'], |
| 429 | ['csibm866', 'ibm866'], |
| 430 | ['ibm866', 'ibm866'], |
| 431 | ['csisolatin2', 'iso-8859-2'], |
| 432 | ['iso-8859-2', 'iso-8859-2'], |
| 433 | ['iso-ir-101', 'iso-8859-2'], |
| 434 | ['iso8859-2', 'iso-8859-2'], |
| 435 | ['iso88592', 'iso-8859-2'], |
| 436 | ['iso_8859-2', 'iso-8859-2'], |
| 437 | ['iso_8859-2:1987', 'iso-8859-2'], |
| 438 | ['l2', 'iso-8859-2'], |
| 439 | ['latin2', 'iso-8859-2'], |
| 440 | ['csisolatin3', 'iso-8859-3'], |
| 441 | ['iso-8859-3', 'iso-8859-3'], |
| 442 | ['iso-ir-109', 'iso-8859-3'], |
| 443 | ['iso8859-3', 'iso-8859-3'], |
| 444 | ['iso88593', 'iso-8859-3'], |
| 445 | ['iso_8859-3', 'iso-8859-3'], |
| 446 | ['iso_8859-3:1988', 'iso-8859-3'], |
| 447 | ['l3', 'iso-8859-3'], |
| 448 | ['latin3', 'iso-8859-3'], |
| 449 | ['csisolatin4', 'iso-8859-4'], |
| 450 | ['iso-8859-4', 'iso-8859-4'], |
| 451 | ['iso-ir-110', 'iso-8859-4'], |
| 452 | ['iso8859-4', 'iso-8859-4'], |
| 453 | ['iso88594', 'iso-8859-4'], |
| 454 | ['iso_8859-4', 'iso-8859-4'], |
| 455 | ['iso_8859-4:1988', 'iso-8859-4'], |
| 456 | ['l4', 'iso-8859-4'], |
| 457 | ['latin4', 'iso-8859-4'], |
| 458 | ['csisolatincyrillic', 'iso-8859-5'], |
| 459 | ['cyrillic', 'iso-8859-5'], |
| 460 | ['iso-8859-5', 'iso-8859-5'], |
| 461 | ['iso-ir-144', 'iso-8859-5'], |
| 462 | ['iso8859-5', 'iso-8859-5'], |
| 463 | ['iso88595', 'iso-8859-5'], |
| 464 | ['iso_8859-5', 'iso-8859-5'], |
| 465 | ['iso_8859-5:1988', 'iso-8859-5'], |
| 466 | ['arabic', 'iso-8859-6'], |
| 467 | ['asmo-708', 'iso-8859-6'], |
| 468 | ['csiso88596e', 'iso-8859-6'], |
| 469 | ['csiso88596i', 'iso-8859-6'], |
| 470 | ['csisolatinarabic', 'iso-8859-6'], |
| 471 | ['ecma-114', 'iso-8859-6'], |
| 472 | ['iso-8859-6', 'iso-8859-6'], |
| 473 | ['iso-8859-6-e', 'iso-8859-6'], |
| 474 | ['iso-8859-6-i', 'iso-8859-6'], |
| 475 | ['iso-ir-127', 'iso-8859-6'], |
| 476 | ['iso8859-6', 'iso-8859-6'], |
| 477 | ['iso88596', 'iso-8859-6'], |
| 478 | ['iso_8859-6', 'iso-8859-6'], |
| 479 | ['iso_8859-6:1987', 'iso-8859-6'], |
| 480 | ['csisolatingreek', 'iso-8859-7'], |
| 481 | ['ecma-118', 'iso-8859-7'], |
| 482 | ['elot_928', 'iso-8859-7'], |
| 483 | ['greek', 'iso-8859-7'], |
| 484 | ['greek8', 'iso-8859-7'], |
| 485 | ['iso-8859-7', 'iso-8859-7'], |
| 486 | ['iso-ir-126', 'iso-8859-7'], |
| 487 | ['iso8859-7', 'iso-8859-7'], |
| 488 | ['iso88597', 'iso-8859-7'], |
| 489 | ['iso_8859-7', 'iso-8859-7'], |
| 490 | ['iso_8859-7:1987', 'iso-8859-7'], |
| 491 | ['sun_eu_greek', 'iso-8859-7'], |
| 492 | ['csiso88598e', 'iso-8859-8'], |
| 493 | ['csisolatinhebrew', 'iso-8859-8'], |
| 494 | ['hebrew', 'iso-8859-8'], |
| 495 | ['iso-8859-8', 'iso-8859-8'], |
| 496 | ['iso-8859-8-e', 'iso-8859-8'], |
| 497 | ['iso-ir-138', 'iso-8859-8'], |
| 498 | ['iso8859-8', 'iso-8859-8'], |
| 499 | ['iso88598', 'iso-8859-8'], |
| 500 | ['iso_8859-8', 'iso-8859-8'], |
| 501 | ['iso_8859-8:1988', 'iso-8859-8'], |
| 502 | ['visual', 'iso-8859-8'], |
| 503 | ['csiso88598i', 'iso-8859-8-i'], |
| 504 | ['iso-8859-8-i', 'iso-8859-8-i'], |
| 505 | ['logical', 'iso-8859-8-i'], |
| 506 | ['csisolatin6', 'iso-8859-10'], |
| 507 | ['iso-8859-10', 'iso-8859-10'], |
| 508 | ['iso-ir-157', 'iso-8859-10'], |
| 509 | ['iso8859-10', 'iso-8859-10'], |
| 510 | ['iso885910', 'iso-8859-10'], |
| 511 | ['l6', 'iso-8859-10'], |
| 512 | ['latin6', 'iso-8859-10'], |
| 513 | ['iso-8859-13', 'iso-8859-13'], |
| 514 | ['iso8859-13', 'iso-8859-13'], |
| 515 | ['iso885913', 'iso-8859-13'], |
| 516 | ['iso-8859-14', 'iso-8859-14'], |
| 517 | ['iso8859-14', 'iso-8859-14'], |
| 518 | ['iso885914', 'iso-8859-14'], |
| 519 | ['csisolatin9', 'iso-8859-15'], |
| 520 | ['iso-8859-15', 'iso-8859-15'], |
| 521 | ['iso8859-15', 'iso-8859-15'], |
| 522 | ['iso885915', 'iso-8859-15'], |
| 523 | ['iso_8859-15', 'iso-8859-15'], |
| 524 | ['l9', 'iso-8859-15'], |
| 525 | ['iso-8859-16', 'iso-8859-16'], |
| 526 | ['cskoi8r', 'koi8-r'], |
| 527 | ['koi', 'koi8-r'], |
| 528 | ['koi8', 'koi8-r'], |
| 529 | ['koi8-r', 'koi8-r'], |
| 530 | ['koi8_r', 'koi8-r'], |
| 531 | ['koi8-ru', 'koi8-u'], |
| 532 | ['koi8-u', 'koi8-u'], |
| 533 | ['csmacintosh', 'macintosh'], |
| 534 | ['mac', 'macintosh'], |
| 535 | ['macintosh', 'macintosh'], |
| 536 | ['x-mac-roman', 'macintosh'], |
| 537 | ['dos-874', 'windows-874'], |
| 538 | ['iso-8859-11', 'windows-874'], |
| 539 | ['iso8859-11', 'windows-874'], |
| 540 | ['iso885911', 'windows-874'], |
| 541 | ['tis-620', 'windows-874'], |
| 542 | ['windows-874', 'windows-874'], |
| 543 | ['cp1250', 'windows-1250'], |
| 544 | ['windows-1250', 'windows-1250'], |
| 545 | ['x-cp1250', 'windows-1250'], |
| 546 | ['cp1251', 'windows-1251'], |
| 547 | ['windows-1251', 'windows-1251'], |
| 548 | ['x-cp1251', 'windows-1251'], |
| 549 | ['ansi_x3.4-1968', 'windows-1252'], |
| 550 | ['ascii', 'windows-1252'], |
| 551 | ['cp1252', 'windows-1252'], |
| 552 | ['cp819', 'windows-1252'], |
| 553 | ['csisolatin1', 'windows-1252'], |
| 554 | ['ibm819', 'windows-1252'], |
| 555 | ['iso-8859-1', 'windows-1252'], |
| 556 | ['iso-ir-100', 'windows-1252'], |
| 557 | ['iso8859-1', 'windows-1252'], |
| 558 | ['iso88591', 'windows-1252'], |
| 559 | ['iso_8859-1', 'windows-1252'], |
| 560 | ['iso_8859-1:1987', 'windows-1252'], |
| 561 | ['l1', 'windows-1252'], |
| 562 | ['latin1', 'windows-1252'], |
| 563 | ['us-ascii', 'windows-1252'], |
| 564 | ['windows-1252', 'windows-1252'], |
| 565 | ['x-cp1252', 'windows-1252'], |
| 566 | ['cp1253', 'windows-1253'], |
| 567 | ['windows-1253', 'windows-1253'], |
| 568 | ['x-cp1253', 'windows-1253'], |
| 569 | ['cp1254', 'windows-1254'], |
| 570 | ['csisolatin5', 'windows-1254'], |
| 571 | ['iso-8859-9', 'windows-1254'], |
| 572 | ['iso-ir-148', 'windows-1254'], |
| 573 | ['iso8859-9', 'windows-1254'], |
| 574 | ['iso88599', 'windows-1254'], |
| 575 | ['iso_8859-9', 'windows-1254'], |
| 576 | ['iso_8859-9:1989', 'windows-1254'], |
| 577 | ['l5', 'windows-1254'], |
| 578 | ['latin5', 'windows-1254'], |
| 579 | ['windows-1254', 'windows-1254'], |
| 580 | ['x-cp1254', 'windows-1254'], |
| 581 | ['cp1255', 'windows-1255'], |
| 582 | ['windows-1255', 'windows-1255'], |
| 583 | ['x-cp1255', 'windows-1255'], |
| 584 | ['cp1256', 'windows-1256'], |
| 585 | ['windows-1256', 'windows-1256'], |
| 586 | ['x-cp1256', 'windows-1256'], |
| 587 | ['cp1257', 'windows-1257'], |
| 588 | ['windows-1257', 'windows-1257'], |
| 589 | ['x-cp1257', 'windows-1257'], |
| 590 | ['cp1258', 'windows-1258'], |
| 591 | ['windows-1258', 'windows-1258'], |
| 592 | ['x-cp1258', 'windows-1258'], |
| 593 | ['x-mac-cyrillic', 'x-mac-cyrillic'], |
| 594 | ['x-mac-ukrainian', 'x-mac-cyrillic'], |
| 595 | ['chinese', 'gbk'], |
| 596 | ['csgb2312', 'gbk'], |
| 597 | ['csiso58gb231280', 'gbk'], |
| 598 | ['gb2312', 'gbk'], |
| 599 | ['gb_2312', 'gbk'], |
| 600 | ['gb_2312-80', 'gbk'], |
| 601 | ['gbk', 'gbk'], |
| 602 | ['iso-ir-58', 'gbk'], |
| 603 | ['x-gbk', 'gbk'], |
| 604 | ['gb18030', 'gb18030'], |
| 605 | ['big5', 'big5'], |
| 606 | ['big5-hkscs', 'big5'], |
| 607 | ['cn-big5', 'big5'], |
| 608 | ['csbig5', 'big5'], |
| 609 | ['x-x-big5', 'big5'], |
| 610 | ['cseucpkdfmtjapanese', 'euc-jp'], |
| 611 | ['euc-jp', 'euc-jp'], |
| 612 | ['x-euc-jp', 'euc-jp'], |
| 613 | ['csiso2022jp', 'iso-2022-jp'], |
| 614 | ['iso-2022-jp', 'iso-2022-jp'], |
| 615 | ['csshiftjis', 'shift_jis'], |
| 616 | ['ms932', 'shift_jis'], |
| 617 | ['ms_kanji', 'shift_jis'], |
| 618 | ['shift-jis', 'shift_jis'], |
| 619 | ['shift_jis', 'shift_jis'], |
| 620 | ['sjis', 'shift_jis'], |
| 621 | ['windows-31j', 'shift_jis'], |
| 622 | ['x-sjis', 'shift_jis'], |
| 623 | ['cseuckr', 'euc-kr'], |
| 624 | ['csksc56011987', 'euc-kr'], |
| 625 | ['euc-kr', 'euc-kr'], |
| 626 | ['iso-ir-149', 'euc-kr'], |
| 627 | ['korean', 'euc-kr'], |
| 628 | ['ks_c_5601-1987', 'euc-kr'], |
| 629 | ['ks_c_5601-1989', 'euc-kr'], |
| 630 | ['ksc5601', 'euc-kr'], |
| 631 | ['ksc_5601', 'euc-kr'], |
| 632 | ['windows-949', 'euc-kr'], |
| 633 | ['csiso2022kr', undefined], |
| 634 | ['hz-gb-2312', undefined], |
| 635 | ['iso-2022-cn', undefined], |
| 636 | ['iso-2022-cn-ext', undefined], |
| 637 | ['iso-2022-kr', undefined], |
| 638 | ['replacement', undefined], |
| 639 | ['unicodefffe', 'utf-16be'], |
| 640 | ['utf-16be', 'utf-16be'], |
| 641 | ['csunicode', 'utf-16le'], |
| 642 | ['iso-10646-ucs-2', 'utf-16le'], |
| 643 | ['ucs-2', 'utf-16le'], |
| 644 | ['unicode', 'utf-16le'], |
| 645 | ['unicodefeff', 'utf-16le'], |
| 646 | ['utf-16', 'utf-16le'], |
| 647 | ['utf-16le', 'utf-16le'], |
| 648 | ['x-user-defined', 'x-user-defined'], |
| 649 | // Test that match is case-insensitive |
| 650 | ['UTF-8', 'utf-8'], |
| 651 | ['UtF-8', 'utf-8'], |
| 652 | ].forEach((pair) => { |
| 653 | const [label, key] = pair; |
| 654 | if (key === undefined) { |
| 655 | throws(() => new TextDecoder(label)); |
| 656 | } else { |
| 657 | { |
| 658 | const dec = new TextDecoder(label); |
| 659 | strictEqual(dec.encoding, key); |
| 660 | } |
| 661 | { |
| 662 | // Whitespace leading and trailing the label will be ignored. |
| 663 | const dec = new TextDecoder(`\t\n\r ${label}\t\n\r`); |
| 664 | strictEqual(dec.encoding, key); |
| 665 | } |
| 666 | } |
| 667 | }); |
| 668 | }, |
| 669 | }; |
| 670 | |
| 671 | // Test that windows-1252 correctly decodes bytes 0x80-0x9F. |
| 672 | // These bytes differ between windows-1252 and ISO-8859-1/Latin-1. |
| 673 | // Per the WHATWG Encoding Standard, labels like "ascii", "latin1", and |
| 674 | // "iso-8859-1" all map to windows-1252, so they must produce the |
| 675 | // windows-1252 code points — not the raw byte values (C1 control chars). |
| 676 | // See: https://encoding.spec.whatwg.org/#names-and-labels |
| 677 | export const windows1252Decode = { |
| 678 | test() { |
| 679 | // The windows-1252 mapping for bytes 0x80-0x9F per the WHATWG Encoding Standard. |
| 680 | // See: https://encoding.spec.whatwg.org/index-windows-1252.txt |
| 681 | // Bytes 0x81, 0x8D, 0x8F, 0x90, 0x9D map to their C1 control character |
| 682 | // code points (identity) rather than being undefined. |
| 683 | const win1252UpperTable = { |
| 684 | 0x80: 0x20ac, // € Euro Sign |
| 685 | 0x81: 0x0081, // <control> |
| 686 | 0x82: 0x201a, // ‚ Single Low-9 Quotation Mark |
| 687 | 0x83: 0x0192, // ƒ Latin Small Letter F With Hook |
| 688 | 0x84: 0x201e, // „ Double Low-9 Quotation Mark |
| 689 | 0x85: 0x2026, // … Horizontal Ellipsis |
| 690 | 0x86: 0x2020, // † Dagger |
| 691 | 0x87: 0x2021, // ‡ Double Dagger |
| 692 | 0x88: 0x02c6, // ˆ Modifier Letter Circumflex Accent |
| 693 | 0x89: 0x2030, // ‰ Per Mille Sign |
| 694 | 0x8a: 0x0160, // Š Latin Capital Letter S With Caron |
| 695 | 0x8b: 0x2039, // ‹ Single Left-Pointing Angle Quotation Mark |
| 696 | 0x8c: 0x0152, // Œ Latin Capital Ligature OE |
| 697 | 0x8d: 0x008d, // <control> |
| 698 | 0x8e: 0x017d, // Ž Latin Capital Letter Z With Caron |
| 699 | 0x8f: 0x008f, // <control> |
| 700 | 0x90: 0x0090, // <control> |
| 701 | 0x91: 0x2018, // ' Left Single Quotation Mark |
| 702 | 0x92: 0x2019, // ' Right Single Quotation Mark |
| 703 | 0x93: 0x201c, // " Left Double Quotation Mark |
| 704 | 0x94: 0x201d, // " Right Double Quotation Mark |
| 705 | 0x95: 0x2022, // • Bullet |
| 706 | 0x96: 0x2013, // – En Dash |
| 707 | 0x97: 0x2014, // — Em Dash |
| 708 | 0x98: 0x02dc, // ˜ Small Tilde |
| 709 | 0x99: 0x2122, // ™ Trade Mark Sign |
| 710 | 0x9a: 0x0161, // š Latin Small Letter S With Caron |
| 711 | 0x9b: 0x203a, // › Single Right-Pointing Angle Quotation Mark |
| 712 | 0x9c: 0x0153, // œ Latin Small Ligature OE |
| 713 | 0x9d: 0x009d, // <control> |
| 714 | 0x9e: 0x017e, // ž Latin Small Letter Z With Caron |
| 715 | 0x9f: 0x0178, // Ÿ Latin Capital Letter Y With Diaeresis |
| 716 | }; |
| 717 | |
| 718 | // Test with all aliases that map to windows-1252 |
| 719 | for (const label of [ |
| 720 | 'windows-1252', |
| 721 | 'ascii', |
| 722 | 'latin1', |
| 723 | 'iso-8859-1', |
| 724 | 'us-ascii', |
| 725 | ]) { |
| 726 | const decoder = new TextDecoder(label); |
| 727 | for (const [byte, expectedCodePoint] of Object.entries( |
| 728 | win1252UpperTable |
| 729 | )) { |
| 730 | const byteVal = Number(byte); |
| 731 | const decoded = decoder.decode(Uint8Array.of(byteVal)); |
| 732 | const actualCodePoint = decoded.codePointAt(0); |
| 733 | strictEqual( |
| 734 | actualCodePoint, |
| 735 | expectedCodePoint, |
| 736 | `${label}: byte 0x${byteVal.toString(16)} should decode to ` + |
| 737 | `U+${expectedCodePoint.toString(16).toUpperCase().padStart(4, '0')}, ` + |
| 738 | `got U+${actualCodePoint.toString(16).toUpperCase().padStart(4, '0')}` |
| 739 | ); |
| 740 | } |
| 741 | } |
| 742 | |
| 743 | // Verify that bytes outside 0x80-0x9F still work correctly |
| 744 | const decoder = new TextDecoder('windows-1252'); |
| 745 | // ASCII range (0x00-0x7F) should be identity |
| 746 | strictEqual(decoder.decode(Uint8Array.of(0x41)).codePointAt(0), 0x41); // 'A' |
| 747 | strictEqual(decoder.decode(Uint8Array.of(0x7f)).codePointAt(0), 0x7f); |
| 748 | // 0xA0-0xFF range is identical between latin-1 and windows-1252 |
| 749 | strictEqual(decoder.decode(Uint8Array.of(0xa2)).codePointAt(0), 0xa2); // ¢ |
| 750 | strictEqual(decoder.decode(Uint8Array.of(0xff)).codePointAt(0), 0xff); // ÿ |
| 751 | }, |
| 752 | }; |
| 753 | |
| 754 | // Per the WHATWG encoding spec (section 10.1.1), GBK's decoder is gb18030's decoder. |
| 755 | // The .encoding property must still return "gbk", but decoding results must match gb18030. |
| 756 | // https://encoding.spec.whatwg.org/#gbk-decoder |
| 757 | export const gbkDecoderIsGb18030Decoder = { |
| 758 | test() { |
| 759 | const gbk = new TextDecoder('gbk'); |
| 760 | const gb18030 = new TextDecoder('gb18030'); |
| 761 | |
| 762 | // .encoding property must still distinguish the two |
| 763 | strictEqual(gbk.encoding, 'gbk'); |
| 764 | strictEqual(gb18030.encoding, 'gb18030'); |
| 765 | |
| 766 | // Decoding results must be identical. These byte pairs exercise boundary |
| 767 | // conditions where gbk and gb18030 would diverge if the ICU converter |
| 768 | // for gbk were used directly instead of delegating to gb18030. |
| 769 | const testBytes = [ |
| 770 | [0, 255], |
| 771 | [128, 255], |
| 772 | [129, 48], |
| 773 | [129, 255], |
| 774 | [254, 48], |
| 775 | [254, 255], |
| 776 | [255, 0], |
| 777 | [255, 255], |
| 778 | ]; |
| 779 | for (const bytes of testBytes) { |
| 780 | const u8 = Uint8Array.from(bytes); |
| 781 | strictEqual( |
| 782 | gbk.decode(u8), |
| 783 | gb18030.decode(u8), |
| 784 | `gbk and gb18030 must decode [${bytes}] identically` |
| 785 | ); |
| 786 | } |
| 787 | }, |
| 788 | }; |
| 789 | |
| 790 | const gbVersionAndRangesTest = (encoding) => { |
| 791 | const loose = new TextDecoder(encoding); |
| 792 | const checkAll = (...list) => list.forEach((x) => check(...x)); |
| 793 | const check = (bytes, str, invalid = false) => { |
| 794 | const fatal = new TextDecoder(encoding, { fatal: true }); |
| 795 | const u8 = Uint8Array.from(bytes); |
| 796 | strictEqual(loose.decode(u8), str); |
| 797 | if (!invalid) strictEqual(fatal.decode(u8), str); |
| 798 | if (invalid) throws(() => fatal.decode(u8)); |
| 799 | }; |
| 800 | |
| 801 | check([0x84, 0x31, 0xa4, 0x36], '\uFFFC'); |
| 802 | check([0x84, 0x31, 0xa4, 0x37], '\uFFFD'); |
| 803 | check([0x84, 0x31, 0xa4, 0x38], '\uFFFE'); |
| 804 | check([0x84, 0x31, 0xa4, 0x39], '\uFFFF'); |
| 805 | check([0x84, 0x31, 0xa5, 0x30], '\uFFFD', true); |
| 806 | check([0x8f, 0x39, 0xfe, 0x39], '\uFFFD', true); |
| 807 | check([0x90, 0x30, 0x81, 0x30], String.fromCodePoint(0x1_00_00)); |
| 808 | check([0x90, 0x30, 0x81, 0x31], String.fromCodePoint(0x1_00_01)); |
| 809 | |
| 810 | check([0xe3, 0x32, 0x9a, 0x35], String.fromCodePoint(0x10_ff_ff)); |
| 811 | check([0xe3, 0x32, 0x9a, 0x36], '\uFFFD', true); |
| 812 | check([0xe3, 0x32, 0x9a, 0x37], '\uFFFD', true); |
| 813 | |
| 814 | check([0xfe, 0x39, 0xfe, 0x39], '\uFFFD', true); |
| 815 | check([0xff, 0x39, 0xfe, 0x39], '\uFFFD9\uFFFD', true); |
| 816 | check([0xfe, 0x40, 0xfe, 0x39], '\uFA0C\uFFFD', true); |
| 817 | check([0xfe, 0x39, 0xff, 0x39], '\uFFFD9\uFFFD9', true); |
| 818 | check([0xfe, 0x39, 0xfe, 0x40], '\uFFFD9\uFA0C', true); |
| 819 | |
| 820 | checkAll( |
| 821 | [[0xa8, 0xbb], '\u0251'], |
| 822 | [[0xa8, 0xbc], '\u1E3F'], |
| 823 | [[0xa8, 0xbd], '\u0144'] |
| 824 | ); |
| 825 | check([0x81, 0x35, 0xf4, 0x36], '\u1E3E'); |
| 826 | check([0x81, 0x35, 0xf4, 0x37], '\uE7C7'); |
| 827 | check([0x81, 0x35, 0xf4, 0x38], '\u1E40'); |
| 828 | |
| 829 | checkAll( |
| 830 | [[0xa6, 0xd9], '\uFE10'], |
| 831 | [[0xa6, 0xed], '\uFE18'], |
| 832 | [[0xa6, 0xf3], '\uFE19'] |
| 833 | ); |
| 834 | checkAll([[0xfe, 0x59], '\u9FB4'], [[0xfe, 0xa0], '\u9FBB']); |
| 835 | }; |
| 836 | |
| 837 | export const gb18030VersionAndRanges = { |
| 838 | test() { |
| 839 | gbVersionAndRangesTest('gb18030'); |
| 840 | }, |
| 841 | }; |
| 842 | |
| 843 | export const gbkVersionAndRanges = { |
| 844 | test() { |
| 845 | gbVersionAndRangesTest('gbk'); |
| 846 | }, |
| 847 | }; |
| 848 | |
| 849 | // Verify that the WHATWG-required mapping corrections also produce the |
| 850 | // correct output when the corrected byte sequences appear inside a larger |
| 851 | // buffer (surrounded by ASCII), not only when they are the entire input. |
| 852 | export const gb18030OverridesEmbedded = { |
| 853 | test() { |
| 854 | const d = new TextDecoder('gb18030'); |
| 855 | |
| 856 | // 0x80 → U+20AC (Euro sign) surrounded by ASCII |
| 857 | strictEqual(d.decode(Uint8Array.of(0x41, 0x80, 0x42)), 'A\u20ACB'); |
| 858 | |
| 859 | // Two-byte mapping corrections surrounded by ASCII |
| 860 | strictEqual(d.decode(Uint8Array.of(0x41, 0xa8, 0xbb, 0x42)), 'A\u0251B'); |
| 861 | strictEqual(d.decode(Uint8Array.of(0x41, 0xa8, 0xbc, 0x42)), 'A\u1E3FB'); |
| 862 | strictEqual(d.decode(Uint8Array.of(0x41, 0xa8, 0xbd, 0x42)), 'A\u0144B'); |
| 863 | |
| 864 | // Vertical form corrections surrounded by ASCII |
| 865 | strictEqual(d.decode(Uint8Array.of(0x41, 0xa6, 0xd9, 0x42)), 'A\uFE10B'); |
| 866 | |
| 867 | // CJK extension corrections surrounded by ASCII |
| 868 | strictEqual(d.decode(Uint8Array.of(0x41, 0xfe, 0x59, 0x42)), 'A\u9FB4B'); |
| 869 | }, |
| 870 | }; |
| 871 | |
| 872 | export const replacementPushbackAsciiCharactersLoose = { |
| 873 | test() { |
| 874 | const vectors = { |
| 875 | big5: [ |
| 876 | [[0x80], '\uFFFD'], |
| 877 | [[0x81, 0x40], '\uFFFD@'], |
| 878 | [[0x83, 0x5c], '\uFFFD\\'], |
| 879 | [[0x87, 0x87, 0x40], '\uFFFD@'], |
| 880 | [[0x81, 0x81], '\uFFFD'], |
| 881 | ], |
| 882 | 'iso-2022-jp': [ |
| 883 | [[0x1b, 0x24], '\uFFFD$'], |
| 884 | [[0x1b, 0x24, 0x40, 0x1b, 0x24], '\uFFFD\uFFFD'], |
| 885 | ], |
| 886 | 'euc-jp': [ |
| 887 | [[0x80], '\uFFFD'], |
| 888 | [[0x8d, 0x8d], '\uFFFD\uFFFD'], |
| 889 | [[0x8e, 0x8e], '\uFFFD'], |
| 890 | ], |
| 891 | }; |
| 892 | |
| 893 | for (const [encoding, list] of Object.entries(vectors)) { |
| 894 | const d = new TextDecoder(encoding); |
| 895 | for (const [bytes, text] of list) { |
| 896 | strictEqual(d.decode(Uint8Array.from(bytes)), text); |
| 897 | } |
| 898 | } |
| 899 | }, |
| 900 | }; |
| 901 | |
| 902 | export const stickyMultibyteStateIso2022JpLoose = { |
| 903 | test() { |
| 904 | const vectors = [ |
| 905 | [[27], '\uFFFD'], |
| 906 | [[27, 0x28], '\uFFFD('], |
| 907 | [[0x1b, 0x28, 0x49], ''], |
| 908 | ]; |
| 909 | |
| 910 | const d = new TextDecoder('iso-2022-jp'); |
| 911 | for (const [bytes, text] of vectors) { |
| 912 | strictEqual(d.decode(Uint8Array.of(0x40)), '@'); |
| 913 | strictEqual(d.decode(Uint8Array.from(bytes)), text); |
| 914 | strictEqual(d.decode(Uint8Array.of(0x40)), '@'); |
| 915 | strictEqual(d.decode(Uint8Array.of(0x2a)), '*'); |
| 916 | strictEqual(d.decode(Uint8Array.of(0x42)), 'B'); |
| 917 | } |
| 918 | }, |
| 919 | }; |
| 920 | |
| 921 | export const fatalStreamGb18030Gbk = { |
| 922 | test() { |
| 923 | for (const encoding of ['gb18030', 'gbk']) { |
| 924 | { |
| 925 | const d = new TextDecoder(encoding, { fatal: true }); |
| 926 | strictEqual(d.decode(Uint8Array.of(0x80), { stream: true }), '\u20AC'); |
| 927 | throws(() => |
| 928 | d.decode(Uint8Array.of(0x81, 0x30, 0x21, 0x21, 0x21), { |
| 929 | stream: true, |
| 930 | }) |
| 931 | ); |
| 932 | strictEqual(d.decode(Uint8Array.of(0x80)), '\u20AC'); |
| 933 | } |
| 934 | |
| 935 | { |
| 936 | const d = new TextDecoder(encoding, { fatal: true }); |
| 937 | strictEqual(d.decode(Uint8Array.of(0x80), { stream: true }), '\u20AC'); |
| 938 | throws(() => |
| 939 | d.decode(Uint8Array.of(0x81, 0x30, 0x81, 0x42, 0x42), { |
| 940 | stream: true, |
| 941 | }) |
| 942 | ); |
| 943 | strictEqual(d.decode(Uint8Array.of(0x80)), '\u20AC'); |
| 944 | } |
| 945 | } |
| 946 | }, |
| 947 | }; |
| 948 | |
| 949 | export const textDecoderStream = { |
| 950 | test() { |
| 951 | const stream = new TextDecoderStream('utf-16', { |
| 952 | fatal: true, |
| 953 | ignoreBOM: true, |
| 954 | }); |
| 955 | strictEqual(stream.encoding, 'utf-16le'); |
| 956 | strictEqual(stream.fatal, true); |
| 957 | strictEqual(stream.ignoreBOM, true); |
| 958 | |
| 959 | const enc = new TextEncoderStream(); |
| 960 | strictEqual(enc.encoding, 'utf-8'); |
| 961 | }, |
| 962 | }; |
| 963 | |
| 964 | // Per WHATWG Big5 decoder step 1, when end-of-queue is reached with a |
| 965 | // pending lead byte, the decoder must return error (U+FFFD in replacement |
| 966 | // mode, throw in fatal mode). This tests the streaming case where a lead |
| 967 | // byte is buffered in one call and then flushed without a trail byte. |
| 968 | export const big5OrphanedLeadOnFlush = { |
| 969 | test() { |
| 970 | // 0xA4 is a valid Big5 lead byte (e.g., first byte of 中 = 0xA4 0xA4). |
| 971 | // Streaming it alone, then flushing, must produce U+FFFD. |
| 972 | { |
| 973 | const dec = new TextDecoder('big5'); |
| 974 | strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), ''); |
| 975 | strictEqual(dec.decode(), '\uFFFD'); |
| 976 | } |
| 977 | |
| 978 | // Fatal mode must throw on the orphaned lead. |
| 979 | { |
| 980 | const dec = new TextDecoder('big5', { fatal: true }); |
| 981 | strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), ''); |
| 982 | throws(() => dec.decode()); |
| 983 | } |
| 984 | |
| 985 | // Orphaned lead followed by an invalid trail byte on flush: the lead |
| 986 | // must produce U+FFFD. 0x20 (space) is not a valid Big5 trail byte |
| 987 | // (valid trails are 0x40-0x7E and 0xA1-0xFE). |
| 988 | { |
| 989 | const dec = new TextDecoder('big5'); |
| 990 | strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), ''); |
| 991 | const result = dec.decode(Uint8Array.of(0x20)); |
| 992 | // The orphaned lead must produce at least one U+FFFD. |
| 993 | ok( |
| 994 | result.includes('\uFFFD'), |
| 995 | `expected U+FFFD in output, got: ${JSON.stringify(result)}` |
| 996 | ); |
| 997 | // The space byte must not be swallowed. |
| 998 | ok( |
| 999 | result.includes(' '), |
| 1000 | `expected space in output, got: ${JSON.stringify(result)}` |
| 1001 | ); |
| 1002 | } |
| 1003 | |
| 1004 | // Streaming a complete pair across two calls must still work. |
| 1005 | { |
| 1006 | const dec = new TextDecoder('big5'); |
| 1007 | strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), ''); |
| 1008 | strictEqual(dec.decode(Uint8Array.of(0xa4)), '中'); |
| 1009 | } |
| 1010 | }, |
| 1011 | }; |
| 1012 | |
| 1013 | // Test x-user-defined encoding per WHATWG spec |
| 1014 | // https://encoding.spec.whatwg.org/#x-user-defined-decoder |
| 1015 | export const xUserDefinedDecode = { |
| 1016 | test() { |
| 1017 | const decoder = new TextDecoder('x-user-defined'); |
| 1018 | strictEqual(decoder.encoding, 'x-user-defined'); |
| 1019 | strictEqual(decoder.fatal, false); |
| 1020 | strictEqual(decoder.ignoreBOM, false); |
| 1021 | |
| 1022 | // Test ASCII bytes (0x00-0x7F) - identity mapping |
| 1023 | strictEqual(decoder.decode(Uint8Array.of(0x41)), 'A'); |
| 1024 | strictEqual(decoder.decode(Uint8Array.of(0x00)), '\u0000'); |
| 1025 | strictEqual(decoder.decode(Uint8Array.of(0x7f)), '\u007F'); |
| 1026 | |
| 1027 | // Test high bytes (0x80-0xFF) - map to Private Use Area U+F780-U+F7FF |
| 1028 | strictEqual(decoder.decode(Uint8Array.of(0x80)), '\uF780'); |
| 1029 | strictEqual(decoder.decode(Uint8Array.of(0x81)), '\uF781'); |
| 1030 | strictEqual(decoder.decode(Uint8Array.of(0xff)), '\uF7FF'); |
| 1031 | |
| 1032 | // Test mixed sequence |
| 1033 | const mixed = new Uint8Array([0x00, 0x7f, 0x80, 0x81, 0xff]); |
| 1034 | strictEqual(decoder.decode(mixed), '\u0000\u007F\uF780\uF781\uF7FF'); |
| 1035 | |
| 1036 | // Test empty input |
| 1037 | strictEqual(decoder.decode(new Uint8Array([])), ''); |
| 1038 | strictEqual(decoder.decode(), ''); |
| 1039 | |
| 1040 | // Test pure ASCII input (fast path) |
| 1041 | strictEqual( |
| 1042 | decoder.decode(new Uint8Array([0x48, 0x65, 0x6c, 0x6c, 0x6f])), |
| 1043 | 'Hello' |
| 1044 | ); |
| 1045 | |
| 1046 | // Test streaming (x-user-defined is single-byte, streaming is trivial) |
| 1047 | const streamDecoder = new TextDecoder('x-user-defined'); |
| 1048 | let result = ''; |
| 1049 | result += streamDecoder.decode(Uint8Array.of(0x41), { stream: true }); |
| 1050 | result += streamDecoder.decode(Uint8Array.of(0x80), { stream: true }); |
| 1051 | result += streamDecoder.decode(Uint8Array.of(0xff), { stream: true }); |
| 1052 | result += streamDecoder.decode(); |
| 1053 | strictEqual(result, 'A\uF780\uF7FF'); |
| 1054 | }, |
| 1055 | }; |
| 1056 | |
| 1057 | // Test x-user-defined with fatal option (all 256 bytes are valid) |
| 1058 | export const xUserDefinedFatal = { |
| 1059 | test() { |
| 1060 | const decoder = new TextDecoder('x-user-defined', { fatal: true }); |
| 1061 | strictEqual(decoder.fatal, true); |
| 1062 | |
| 1063 | // All 256 byte values are valid, fatal mode should never throw |
| 1064 | for (let byte = 0; byte < 256; byte++) { |
| 1065 | const decoded = decoder.decode(Uint8Array.of(byte)); |
| 1066 | if (byte < 0x80) { |
| 1067 | strictEqual(decoded.codePointAt(0), byte); |
| 1068 | } else { |
| 1069 | strictEqual(decoded.codePointAt(0), 0xf700 + byte); |
| 1070 | } |
| 1071 | } |
| 1072 | }, |
| 1073 | }; |
| 1074 | |
| 1075 | // Verify that streaming with zero-length input works for every legacy |
| 1076 | // encoding handled by the Rust LegacyDecoder. An empty chunk in streaming |
| 1077 | // mode must produce an empty string and leave the decoder in a valid state |
| 1078 | // for subsequent calls. |
| 1079 | export const legacyStreamEmptyInput = { |
| 1080 | test() { |
| 1081 | const encodings = [ |
| 1082 | 'big5', |
| 1083 | 'euc-jp', |
| 1084 | 'euc-kr', |
| 1085 | 'gb18030', |
| 1086 | 'gbk', |
| 1087 | 'iso-2022-jp', |
| 1088 | 'shift_jis', |
| 1089 | 'windows-1252', |
| 1090 | 'x-user-defined', |
| 1091 | ]; |
| 1092 | |
| 1093 | const empty = new Uint8Array(0); |
| 1094 | |
| 1095 | for (const label of encodings) { |
| 1096 | for (const fatal of [false, true]) { |
| 1097 | const dec = new TextDecoder(label, { fatal }); |
| 1098 | |
| 1099 | // Empty stream chunk must produce empty string. |
| 1100 | strictEqual( |
| 1101 | dec.decode(empty, { stream: true }), |
| 1102 | '', |
| 1103 | `${label} (fatal=${fatal}): empty stream chunk should be ''` |
| 1104 | ); |
| 1105 | |
| 1106 | // A second empty stream chunk must also be fine. |
| 1107 | strictEqual( |
| 1108 | dec.decode(empty, { stream: true }), |
| 1109 | '', |
| 1110 | `${label} (fatal=${fatal}): second empty stream chunk should be ''` |
| 1111 | ); |
| 1112 | |
| 1113 | // Final flush with no pending bytes must produce empty string. |
| 1114 | strictEqual( |
| 1115 | dec.decode(), |
| 1116 | '', |
| 1117 | `${label} (fatal=${fatal}): flush after empty chunks should be ''` |
| 1118 | ); |
| 1119 | |
| 1120 | // Decoder must still work normally after the empty-stream sequence. |
| 1121 | // Feed a single ASCII byte to verify. |
| 1122 | strictEqual( |
| 1123 | dec.decode(Uint8Array.of(0x41)), |
| 1124 | 'A', |
| 1125 | `${label} (fatal=${fatal}): decode 'A' after empty stream should work` |
| 1126 | ); |
| 1127 | } |
| 1128 | } |
| 1129 | }, |
| 1130 | }; |