Skip to content
File

Blob: src/workerd/api/tests/encoding-test.js

javascript1131 lines
1// Copyright (c) 2023 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4import { deepStrictEqual, strictEqual, throws, ok } from 'node:assert';
5 
6// Test for the Encoding standard Web API implementation.
7// The implementation for these are in api/encoding.{h|c++}
8 
9function decodeStreaming(decoder, input) {
10 // Test truncation behavior while streaming by feeding the decoder a single byte at a time.
11 // Note we don't try-catch here because we don't expect this ever to fail, because we're
12 // streaming text.
13 let x = '';
14 for (let i = 0; i < input.length; ++i) {
15 x += decoder.decode(input.slice(i, i + 1), { stream: true });
16 }
17 x += decoder.decode();
18 return x;
19}
20 
21// From https://developer.mozilla.org/en-US/docs/Web/API/Encoding_API/Encodings
22const windows1252Labels = [
23 'ansi_x3.4-1968',
24 'ascii',
25 'cp1252',
26 'cp819',
27 'csisolatin1',
28 'ibm819',
29 'iso-8859-1',
30 'iso-ir-100',
31 'iso8859-1',
32 'iso88591',
33 'iso_8859-1',
34 'iso_8859-1:1987',
35 'l1',
36 'latin1',
37 'us-ascii',
38 'windows-1252',
39 'x-cp1252',
40];
41 
42const utf8Labels = ['unicode-1-1-utf-8', 'utf-8', 'utf8'];
43 
44export const decodeStreamingTest = {
45 test() {
46 let _results = [];
47 
48 for (const label of windows1252Labels) {
49 ok(
50 new TextDecoder(`${label}`).encoding === 'windows-1252',
51 `TextDecoder constructed with '${label}' label to have 'windows-1252' encoding.`
52 );
53 }
54 for (const label of utf8Labels) {
55 ok(
56 new TextDecoder(`${label}`).encoding === 'utf-8',
57 `TextDecoder constructed with '${label}' label to have 'utf-8' encoding.`
58 );
59 }
60 
61 const decoder = new TextDecoder();
62 const fatalDecoder = new TextDecoder('utf-8', { fatal: true });
63 const fatalIgnoreBomDecoder = new TextDecoder('utf-8', {
64 fatal: true,
65 ignoreBOM: true,
66 });
67 
68 ok(decoder.encoding === 'utf-8', "default encoding property to be 'utf-8'");
69 ok(decoder.fatal === false, 'default fatal property to be false');
70 ok(decoder.ignoreBOM === false, 'default ignoreBOM property to be false');
71 
72 const fooCat = new Uint8Array([102, 111, 111, 32, 240, 159, 152, 186]);
73 
74 ok(decoder.decode().length === 0, 'decoded undefined array length to be 0');
75 ok(
76 decoder.decode(fooCat) === 'foo 😺',
77 'foo-cat from Uint8Buffer to be foo 😺'
78 );
79 ok(
80 decoder.decode(fooCat.buffer) === 'foo 😺',
81 'foo-cat from ArrayBuffer to be foo 😺'
82 );
83 
84 const twoByteCodePoint = new Uint8Array([0xc2, 0xa2]); // cent sign
85 const threeByteCodePoint = new Uint8Array([0xe2, 0x82, 0xac]); // euro sign
86 const fourByteCodePoint = new Uint8Array([240, 159, 152, 186]); // cat emoji
87 
88 [twoByteCodePoint, threeByteCodePoint, fourByteCodePoint].forEach(
89 (input) => {
90 // For each input sequence of code units, try decoding each subsequence of code units except the
91 // full code point itself.
92 for (let i = 1; i < input.length; ++i) {
93 const head = input.slice(0, input.length - i);
94 const tail = input.slice(input.length - i);
95 
96 ok(
97 decoder.decode(head) === '�',
98 'code point fragment (head) to be replaced with replacement character'
99 );
100 ok(
101 decoder.decode(tail) === '�'.repeat(tail.length),
102 'code point fragment (tail) to be replaced with replacement character'
103 );
104 
105 const _errMsg = 'Failed to decode input.';
106 
107 // Exception to be thrown decoding code point fragment (tail) in fatal mode
108 throws(() => fatalDecoder.decode(head));
109 throws(() => fatalDecoder.decode(tail));
110 }
111 }
112 );
113 
114 // Test ASCII
115 const asciiDecoder = new TextDecoder('ascii');
116 ok(
117 asciiDecoder.decode().length === 0,
118 'decoded undefined array length to be 0'
119 );
120 ok(
121 asciiDecoder.decode(new Uint8Array([162, 174, 255])) === '¢®ÿ',
122 'decoded extended ascii correctly'
123 );
124 
125 // Test streaming
126 
127 ok(
128 decodeStreaming(fatalDecoder, twoByteCodePoint) === '¢',
129 '2-byte code point (cent sign) to be decoded correctly'
130 );
131 ok(
132 decodeStreaming(fatalDecoder, threeByteCodePoint) === '€',
133 '3-byte code point (euro sign) to be decoded correctly'
134 );
135 ok(
136 decodeStreaming(fatalDecoder, fourByteCodePoint) === '😺',
137 '4-byte code point (cat emoji) to be decoded correctly'
138 );
139 
140 const bom = new Uint8Array([0xef, 0xbb, 0xbf]);
141 const bomBom = new Uint8Array([0xef, 0xbb, 0xbf, 0xef, 0xbb, 0xbf]);
142 
143 ok(
144 decodeStreaming(fatalDecoder, bom) === '',
145 'BOM to be stripped by TextDecoder without ignoreBOM set'
146 );
147 ok(
148 decodeStreaming(fatalDecoder, bomBom) === '\ufeff',
149 'first BOM to be stripped by TextDecoder without ignoreBOM set'
150 );
151 
152 ok(
153 decodeStreaming(fatalIgnoreBomDecoder, bom) === '\ufeff',
154 'BOM not to be stripped by TextDecoder with ignoreBOM set'
155 );
156 ok(
157 decodeStreaming(fatalIgnoreBomDecoder, bomBom) === '\ufeff\ufeff',
158 'first BOM not to be stripped by TextDecoder with ignoreBOM set'
159 );
160 
161 const shiftJisDecoder = new TextDecoder('shift_jis');
162 let shiftJisResult = '';
163 shiftJisResult += shiftJisDecoder.decode(new Uint8Array([0x82]), {
164 stream: true,
165 });
166 shiftJisResult += shiftJisDecoder.decode(new Uint8Array(), {
167 stream: true,
168 });
169 shiftJisResult += shiftJisDecoder.decode(new Uint8Array([0xa0]), {
170 stream: true,
171 });
172 shiftJisResult += shiftJisDecoder.decode();
173 ok(
174 shiftJisResult === 'あ',
175 'streaming Shift_JIS should tolerate empty chunks without flushing state'
176 );
177 },
178};
179 
180export const textEncoderTest = {
181 test() {
182 const encoder = new TextEncoder();
183 strictEqual(encoder.encoding, 'utf-8');
184 strictEqual(encoder.encode().length, 0);
185 deepStrictEqual(
186 encoder.encode('foo 😺'),
187 new Uint8Array([102, 111, 111, 32, 240, 159, 152, 186])
188 );
189 },
190};
191 
192export const encodeWptTest = {
193 test() {
194 w3cTestEncode();
195 w3cTestEncodeInto();
196 
197 function w3cTestEncode() {
198 const bad = [
199 {
200 input: '\uD800',
201 expected: '\uFFFD',
202 name: 'lone surrogate lead',
203 },
204 {
205 input: '\uDC00',
206 expected: '\uFFFD',
207 name: 'lone surrogate trail',
208 },
209 {
210 input: '\uD800\u0000',
211 expected: '\uFFFD\u0000',
212 name: 'unmatched surrogate lead',
213 },
214 {
215 input: '\uDC00\u0000',
216 expected: '\uFFFD\u0000',
217 name: 'unmatched surrogate trail',
218 },
219 {
220 input: '\uDC00\uD800',
221 expected: '\uFFFD\uFFFD',
222 name: 'swapped surrogate pair',
223 },
224 {
225 input: '\uD834\uDD1E',
226 expected: '\uD834\uDD1E',
227 name: 'properly encoded MUSICAL SYMBOL G CLEF (U+1D11E)',
228 },
229 ];
230 
231 bad.forEach(function (t) {
232 const encoded = new TextEncoder().encode(t.input);
233 const decoded = new TextDecoder().decode(encoded);
234 strictEqual(decoded, t.expected);
235 });
236 
237 strictEqual(new TextEncoder().encode().length, 0);
238 }
239 
240 function w3cTestEncodeInto() {
241 [
242 {
243 input: 'Hi',
244 read: 0,
245 destinationLength: 0,
246 written: [],
247 },
248 {
249 input: 'A',
250 read: 1,
251 destinationLength: 10,
252 written: [0x41],
253 },
254 {
255 input: '\u{1D306}', // "\uD834\uDF06"
256 read: 2,
257 destinationLength: 4,
258 written: [0xf0, 0x9d, 0x8c, 0x86],
259 },
260 {
261 input: '\u{1D306}A',
262 read: 0,
263 destinationLength: 3,
264 written: [],
265 },
266 {
267 input: '\uD834A\uDF06A¥Hi',
268 read: 5,
269 destinationLength: 10,
270 written: [0xef, 0xbf, 0xbd, 0x41, 0xef, 0xbf, 0xbd, 0x41, 0xc2, 0xa5],
271 },
272 {
273 input: 'A\uDF06',
274 read: 2,
275 destinationLength: 4,
276 written: [0x41, 0xef, 0xbf, 0xbd],
277 },
278 {
279 input: '¥¥',
280 read: 2,
281 destinationLength: 4,
282 written: [0xc2, 0xa5, 0xc2, 0xa5],
283 },
284 ].forEach((testData) => {
285 [
286 {
287 bufferIncrease: 0,
288 destinationOffset: 0,
289 filler: 0,
290 },
291 {
292 bufferIncrease: 10,
293 destinationOffset: 4,
294 filler: 0,
295 },
296 {
297 bufferIncrease: 0,
298 destinationOffset: 0,
299 filler: 0x80,
300 },
301 {
302 bufferIncrease: 10,
303 destinationOffset: 4,
304 filler: 0x80,
305 },
306 {
307 bufferIncrease: 0,
308 destinationOffset: 0,
309 filler: 'random',
310 },
311 {
312 bufferIncrease: 10,
313 destinationOffset: 4,
314 filler: 'random',
315 },
316 ].forEach((destinationData) => {
317 const bufferLength =
318 testData.destinationLength + destinationData.bufferIncrease;
319 const destinationOffset = destinationData.destinationOffset;
320 const destinationLength = testData.destinationLength;
321 const destinationFiller = destinationData.filler;
322 const encoder = new TextEncoder();
323 const buffer = new ArrayBuffer(bufferLength);
324 const view = new Uint8Array(
325 buffer,
326 destinationOffset,
327 destinationLength
328 );
329 const fullView = new Uint8Array(buffer);
330 const control = Array.from({ length: bufferLength }, () => 0);
331 let byte = destinationFiller;
332 for (let i = 0; i < bufferLength; i++) {
333 if (destinationFiller === 'random') {
334 byte = Math.floor(Math.random() * 256);
335 }
336 control[i] = byte;
337 fullView[i] = byte;
338 }
339 
340 // It's happening
341 const result = encoder.encodeInto(testData.input, view);
342 
343 // Basics
344 strictEqual(view.byteLength, destinationLength);
345 strictEqual(view.length, destinationLength);
346 
347 // Remainder
348 strictEqual(result.read, testData.read);
349 strictEqual(result.written, testData.written.length);
350 for (let i = 0; i < bufferLength; i++) {
351 if (
352 i < destinationOffset ||
353 i >= destinationOffset + testData.written.length
354 ) {
355 strictEqual(fullView[i], control[i]);
356 } else {
357 strictEqual(fullView[i], testData.written[i - destinationOffset]);
358 }
359 }
360 });
361 });
362 
363 [
364 DataView,
365 Int8Array,
366 Int16Array,
367 Int32Array,
368 Uint16Array,
369 Uint32Array,
370 Uint8ClampedArray,
371 Float16Array,
372 Float32Array,
373 Float64Array,
374 ].forEach((view) => {
375 const enc = new TextEncoder();
376 throws(() => enc.encodeInto('', new view(new ArrayBuffer(0))));
377 });
378 
379 {
380 const enc = new TextEncoder();
381 throws(() => enc.encodeInto('', new ArrayBuffer(0)));
382 }
383 }
384 },
385};
386 
387export const big5 = {
388 test() {
389 // Input is the Big5 encoding for the word 中國人 (meaning Chinese Person)
390 // Check is the UTF-8 encoding for the same word.
391 const input = new Uint8Array([0xa4, 0xa4, 0xb0, 0xea, 0xa4, 0x48]);
392 const check = new Uint8Array([
393 0xe4, 0xb8, 0xad, 0xe5, 0x9c, 0x8b, 0xe4, 0xba, 0xba,
394 ]);
395 const enc = new TextEncoder();
396 const dec = new TextDecoder('big5', { ignoreBOM: true });
397 const result = enc.encode(dec.decode(input));
398 strictEqual(result.length, check.length);
399 for (let n = 0; n < result.length; n++) {
400 strictEqual(result[n], check[n]);
401 }
402 },
403};
404 
405export const utf16leFastTrack = {
406 test() {
407 // Input is the UTF-16le encoding for the word Hello. This should trigger the fast-path
408 // which should handle the encoding with no problems. Here we only test that the results
409 // are expected. We cannot verify here that the fast track is actually used.
410 const input = new Uint8Array([
411 0x68, 0x00, 0x65, 0x00, 0x6c, 0x00, 0x6c, 0x00, 0x6f, 0x00,
412 ]);
413 const dec = new TextDecoder('utf-16le');
414 strictEqual(dec.decode(input), 'hello');
415 },
416};
417 
418export const allTheDecoders = {
419 test() {
420 [
421 ['unicode-1-1-utf-8', 'utf-8'],
422 ['unicode11utf8', 'utf-8'],
423 ['unicode20utf8', 'utf-8'],
424 ['utf-8', 'utf-8'],
425 ['utf8', 'utf-8'],
426 ['x-unicode20utf8', 'utf-8'],
427 ['866', 'ibm866'],
428 ['cp866', 'ibm866'],
429 ['csibm866', 'ibm866'],
430 ['ibm866', 'ibm866'],
431 ['csisolatin2', 'iso-8859-2'],
432 ['iso-8859-2', 'iso-8859-2'],
433 ['iso-ir-101', 'iso-8859-2'],
434 ['iso8859-2', 'iso-8859-2'],
435 ['iso88592', 'iso-8859-2'],
436 ['iso_8859-2', 'iso-8859-2'],
437 ['iso_8859-2:1987', 'iso-8859-2'],
438 ['l2', 'iso-8859-2'],
439 ['latin2', 'iso-8859-2'],
440 ['csisolatin3', 'iso-8859-3'],
441 ['iso-8859-3', 'iso-8859-3'],
442 ['iso-ir-109', 'iso-8859-3'],
443 ['iso8859-3', 'iso-8859-3'],
444 ['iso88593', 'iso-8859-3'],
445 ['iso_8859-3', 'iso-8859-3'],
446 ['iso_8859-3:1988', 'iso-8859-3'],
447 ['l3', 'iso-8859-3'],
448 ['latin3', 'iso-8859-3'],
449 ['csisolatin4', 'iso-8859-4'],
450 ['iso-8859-4', 'iso-8859-4'],
451 ['iso-ir-110', 'iso-8859-4'],
452 ['iso8859-4', 'iso-8859-4'],
453 ['iso88594', 'iso-8859-4'],
454 ['iso_8859-4', 'iso-8859-4'],
455 ['iso_8859-4:1988', 'iso-8859-4'],
456 ['l4', 'iso-8859-4'],
457 ['latin4', 'iso-8859-4'],
458 ['csisolatincyrillic', 'iso-8859-5'],
459 ['cyrillic', 'iso-8859-5'],
460 ['iso-8859-5', 'iso-8859-5'],
461 ['iso-ir-144', 'iso-8859-5'],
462 ['iso8859-5', 'iso-8859-5'],
463 ['iso88595', 'iso-8859-5'],
464 ['iso_8859-5', 'iso-8859-5'],
465 ['iso_8859-5:1988', 'iso-8859-5'],
466 ['arabic', 'iso-8859-6'],
467 ['asmo-708', 'iso-8859-6'],
468 ['csiso88596e', 'iso-8859-6'],
469 ['csiso88596i', 'iso-8859-6'],
470 ['csisolatinarabic', 'iso-8859-6'],
471 ['ecma-114', 'iso-8859-6'],
472 ['iso-8859-6', 'iso-8859-6'],
473 ['iso-8859-6-e', 'iso-8859-6'],
474 ['iso-8859-6-i', 'iso-8859-6'],
475 ['iso-ir-127', 'iso-8859-6'],
476 ['iso8859-6', 'iso-8859-6'],
477 ['iso88596', 'iso-8859-6'],
478 ['iso_8859-6', 'iso-8859-6'],
479 ['iso_8859-6:1987', 'iso-8859-6'],
480 ['csisolatingreek', 'iso-8859-7'],
481 ['ecma-118', 'iso-8859-7'],
482 ['elot_928', 'iso-8859-7'],
483 ['greek', 'iso-8859-7'],
484 ['greek8', 'iso-8859-7'],
485 ['iso-8859-7', 'iso-8859-7'],
486 ['iso-ir-126', 'iso-8859-7'],
487 ['iso8859-7', 'iso-8859-7'],
488 ['iso88597', 'iso-8859-7'],
489 ['iso_8859-7', 'iso-8859-7'],
490 ['iso_8859-7:1987', 'iso-8859-7'],
491 ['sun_eu_greek', 'iso-8859-7'],
492 ['csiso88598e', 'iso-8859-8'],
493 ['csisolatinhebrew', 'iso-8859-8'],
494 ['hebrew', 'iso-8859-8'],
495 ['iso-8859-8', 'iso-8859-8'],
496 ['iso-8859-8-e', 'iso-8859-8'],
497 ['iso-ir-138', 'iso-8859-8'],
498 ['iso8859-8', 'iso-8859-8'],
499 ['iso88598', 'iso-8859-8'],
500 ['iso_8859-8', 'iso-8859-8'],
501 ['iso_8859-8:1988', 'iso-8859-8'],
502 ['visual', 'iso-8859-8'],
503 ['csiso88598i', 'iso-8859-8-i'],
504 ['iso-8859-8-i', 'iso-8859-8-i'],
505 ['logical', 'iso-8859-8-i'],
506 ['csisolatin6', 'iso-8859-10'],
507 ['iso-8859-10', 'iso-8859-10'],
508 ['iso-ir-157', 'iso-8859-10'],
509 ['iso8859-10', 'iso-8859-10'],
510 ['iso885910', 'iso-8859-10'],
511 ['l6', 'iso-8859-10'],
512 ['latin6', 'iso-8859-10'],
513 ['iso-8859-13', 'iso-8859-13'],
514 ['iso8859-13', 'iso-8859-13'],
515 ['iso885913', 'iso-8859-13'],
516 ['iso-8859-14', 'iso-8859-14'],
517 ['iso8859-14', 'iso-8859-14'],
518 ['iso885914', 'iso-8859-14'],
519 ['csisolatin9', 'iso-8859-15'],
520 ['iso-8859-15', 'iso-8859-15'],
521 ['iso8859-15', 'iso-8859-15'],
522 ['iso885915', 'iso-8859-15'],
523 ['iso_8859-15', 'iso-8859-15'],
524 ['l9', 'iso-8859-15'],
525 ['iso-8859-16', 'iso-8859-16'],
526 ['cskoi8r', 'koi8-r'],
527 ['koi', 'koi8-r'],
528 ['koi8', 'koi8-r'],
529 ['koi8-r', 'koi8-r'],
530 ['koi8_r', 'koi8-r'],
531 ['koi8-ru', 'koi8-u'],
532 ['koi8-u', 'koi8-u'],
533 ['csmacintosh', 'macintosh'],
534 ['mac', 'macintosh'],
535 ['macintosh', 'macintosh'],
536 ['x-mac-roman', 'macintosh'],
537 ['dos-874', 'windows-874'],
538 ['iso-8859-11', 'windows-874'],
539 ['iso8859-11', 'windows-874'],
540 ['iso885911', 'windows-874'],
541 ['tis-620', 'windows-874'],
542 ['windows-874', 'windows-874'],
543 ['cp1250', 'windows-1250'],
544 ['windows-1250', 'windows-1250'],
545 ['x-cp1250', 'windows-1250'],
546 ['cp1251', 'windows-1251'],
547 ['windows-1251', 'windows-1251'],
548 ['x-cp1251', 'windows-1251'],
549 ['ansi_x3.4-1968', 'windows-1252'],
550 ['ascii', 'windows-1252'],
551 ['cp1252', 'windows-1252'],
552 ['cp819', 'windows-1252'],
553 ['csisolatin1', 'windows-1252'],
554 ['ibm819', 'windows-1252'],
555 ['iso-8859-1', 'windows-1252'],
556 ['iso-ir-100', 'windows-1252'],
557 ['iso8859-1', 'windows-1252'],
558 ['iso88591', 'windows-1252'],
559 ['iso_8859-1', 'windows-1252'],
560 ['iso_8859-1:1987', 'windows-1252'],
561 ['l1', 'windows-1252'],
562 ['latin1', 'windows-1252'],
563 ['us-ascii', 'windows-1252'],
564 ['windows-1252', 'windows-1252'],
565 ['x-cp1252', 'windows-1252'],
566 ['cp1253', 'windows-1253'],
567 ['windows-1253', 'windows-1253'],
568 ['x-cp1253', 'windows-1253'],
569 ['cp1254', 'windows-1254'],
570 ['csisolatin5', 'windows-1254'],
571 ['iso-8859-9', 'windows-1254'],
572 ['iso-ir-148', 'windows-1254'],
573 ['iso8859-9', 'windows-1254'],
574 ['iso88599', 'windows-1254'],
575 ['iso_8859-9', 'windows-1254'],
576 ['iso_8859-9:1989', 'windows-1254'],
577 ['l5', 'windows-1254'],
578 ['latin5', 'windows-1254'],
579 ['windows-1254', 'windows-1254'],
580 ['x-cp1254', 'windows-1254'],
581 ['cp1255', 'windows-1255'],
582 ['windows-1255', 'windows-1255'],
583 ['x-cp1255', 'windows-1255'],
584 ['cp1256', 'windows-1256'],
585 ['windows-1256', 'windows-1256'],
586 ['x-cp1256', 'windows-1256'],
587 ['cp1257', 'windows-1257'],
588 ['windows-1257', 'windows-1257'],
589 ['x-cp1257', 'windows-1257'],
590 ['cp1258', 'windows-1258'],
591 ['windows-1258', 'windows-1258'],
592 ['x-cp1258', 'windows-1258'],
593 ['x-mac-cyrillic', 'x-mac-cyrillic'],
594 ['x-mac-ukrainian', 'x-mac-cyrillic'],
595 ['chinese', 'gbk'],
596 ['csgb2312', 'gbk'],
597 ['csiso58gb231280', 'gbk'],
598 ['gb2312', 'gbk'],
599 ['gb_2312', 'gbk'],
600 ['gb_2312-80', 'gbk'],
601 ['gbk', 'gbk'],
602 ['iso-ir-58', 'gbk'],
603 ['x-gbk', 'gbk'],
604 ['gb18030', 'gb18030'],
605 ['big5', 'big5'],
606 ['big5-hkscs', 'big5'],
607 ['cn-big5', 'big5'],
608 ['csbig5', 'big5'],
609 ['x-x-big5', 'big5'],
610 ['cseucpkdfmtjapanese', 'euc-jp'],
611 ['euc-jp', 'euc-jp'],
612 ['x-euc-jp', 'euc-jp'],
613 ['csiso2022jp', 'iso-2022-jp'],
614 ['iso-2022-jp', 'iso-2022-jp'],
615 ['csshiftjis', 'shift_jis'],
616 ['ms932', 'shift_jis'],
617 ['ms_kanji', 'shift_jis'],
618 ['shift-jis', 'shift_jis'],
619 ['shift_jis', 'shift_jis'],
620 ['sjis', 'shift_jis'],
621 ['windows-31j', 'shift_jis'],
622 ['x-sjis', 'shift_jis'],
623 ['cseuckr', 'euc-kr'],
624 ['csksc56011987', 'euc-kr'],
625 ['euc-kr', 'euc-kr'],
626 ['iso-ir-149', 'euc-kr'],
627 ['korean', 'euc-kr'],
628 ['ks_c_5601-1987', 'euc-kr'],
629 ['ks_c_5601-1989', 'euc-kr'],
630 ['ksc5601', 'euc-kr'],
631 ['ksc_5601', 'euc-kr'],
632 ['windows-949', 'euc-kr'],
633 ['csiso2022kr', undefined],
634 ['hz-gb-2312', undefined],
635 ['iso-2022-cn', undefined],
636 ['iso-2022-cn-ext', undefined],
637 ['iso-2022-kr', undefined],
638 ['replacement', undefined],
639 ['unicodefffe', 'utf-16be'],
640 ['utf-16be', 'utf-16be'],
641 ['csunicode', 'utf-16le'],
642 ['iso-10646-ucs-2', 'utf-16le'],
643 ['ucs-2', 'utf-16le'],
644 ['unicode', 'utf-16le'],
645 ['unicodefeff', 'utf-16le'],
646 ['utf-16', 'utf-16le'],
647 ['utf-16le', 'utf-16le'],
648 ['x-user-defined', 'x-user-defined'],
649 // Test that match is case-insensitive
650 ['UTF-8', 'utf-8'],
651 ['UtF-8', 'utf-8'],
652 ].forEach((pair) => {
653 const [label, key] = pair;
654 if (key === undefined) {
655 throws(() => new TextDecoder(label));
656 } else {
657 {
658 const dec = new TextDecoder(label);
659 strictEqual(dec.encoding, key);
660 }
661 {
662 // Whitespace leading and trailing the label will be ignored.
663 const dec = new TextDecoder(`\t\n\r ${label}\t\n\r`);
664 strictEqual(dec.encoding, key);
665 }
666 }
667 });
668 },
669};
670 
671// Test that windows-1252 correctly decodes bytes 0x80-0x9F.
672// These bytes differ between windows-1252 and ISO-8859-1/Latin-1.
673// Per the WHATWG Encoding Standard, labels like "ascii", "latin1", and
674// "iso-8859-1" all map to windows-1252, so they must produce the
675// windows-1252 code points — not the raw byte values (C1 control chars).
676// See: https://encoding.spec.whatwg.org/#names-and-labels
677export const windows1252Decode = {
678 test() {
679 // The windows-1252 mapping for bytes 0x80-0x9F per the WHATWG Encoding Standard.
680 // See: https://encoding.spec.whatwg.org/index-windows-1252.txt
681 // Bytes 0x81, 0x8D, 0x8F, 0x90, 0x9D map to their C1 control character
682 // code points (identity) rather than being undefined.
683 const win1252UpperTable = {
684 0x80: 0x20ac, // € Euro Sign
685 0x81: 0x0081, // <control>
686 0x82: 0x201a, // ‚ Single Low-9 Quotation Mark
687 0x83: 0x0192, // ƒ Latin Small Letter F With Hook
688 0x84: 0x201e, // „ Double Low-9 Quotation Mark
689 0x85: 0x2026, // … Horizontal Ellipsis
690 0x86: 0x2020, // † Dagger
691 0x87: 0x2021, // ‡ Double Dagger
692 0x88: 0x02c6, // ˆ Modifier Letter Circumflex Accent
693 0x89: 0x2030, // ‰ Per Mille Sign
694 0x8a: 0x0160, // Š Latin Capital Letter S With Caron
695 0x8b: 0x2039, // ‹ Single Left-Pointing Angle Quotation Mark
696 0x8c: 0x0152, // Œ Latin Capital Ligature OE
697 0x8d: 0x008d, // <control>
698 0x8e: 0x017d, // Ž Latin Capital Letter Z With Caron
699 0x8f: 0x008f, // <control>
700 0x90: 0x0090, // <control>
701 0x91: 0x2018, // ' Left Single Quotation Mark
702 0x92: 0x2019, // ' Right Single Quotation Mark
703 0x93: 0x201c, // " Left Double Quotation Mark
704 0x94: 0x201d, // " Right Double Quotation Mark
705 0x95: 0x2022, // • Bullet
706 0x96: 0x2013, // – En Dash
707 0x97: 0x2014, // — Em Dash
708 0x98: 0x02dc, // ˜ Small Tilde
709 0x99: 0x2122, // ™ Trade Mark Sign
710 0x9a: 0x0161, // š Latin Small Letter S With Caron
711 0x9b: 0x203a, // › Single Right-Pointing Angle Quotation Mark
712 0x9c: 0x0153, // œ Latin Small Ligature OE
713 0x9d: 0x009d, // <control>
714 0x9e: 0x017e, // ž Latin Small Letter Z With Caron
715 0x9f: 0x0178, // Ÿ Latin Capital Letter Y With Diaeresis
716 };
717 
718 // Test with all aliases that map to windows-1252
719 for (const label of [
720 'windows-1252',
721 'ascii',
722 'latin1',
723 'iso-8859-1',
724 'us-ascii',
725 ]) {
726 const decoder = new TextDecoder(label);
727 for (const [byte, expectedCodePoint] of Object.entries(
728 win1252UpperTable
729 )) {
730 const byteVal = Number(byte);
731 const decoded = decoder.decode(Uint8Array.of(byteVal));
732 const actualCodePoint = decoded.codePointAt(0);
733 strictEqual(
734 actualCodePoint,
735 expectedCodePoint,
736 `${label}: byte 0x${byteVal.toString(16)} should decode to ` +
737 `U+${expectedCodePoint.toString(16).toUpperCase().padStart(4, '0')}, ` +
738 `got U+${actualCodePoint.toString(16).toUpperCase().padStart(4, '0')}`
739 );
740 }
741 }
742 
743 // Verify that bytes outside 0x80-0x9F still work correctly
744 const decoder = new TextDecoder('windows-1252');
745 // ASCII range (0x00-0x7F) should be identity
746 strictEqual(decoder.decode(Uint8Array.of(0x41)).codePointAt(0), 0x41); // 'A'
747 strictEqual(decoder.decode(Uint8Array.of(0x7f)).codePointAt(0), 0x7f);
748 // 0xA0-0xFF range is identical between latin-1 and windows-1252
749 strictEqual(decoder.decode(Uint8Array.of(0xa2)).codePointAt(0), 0xa2); // ¢
750 strictEqual(decoder.decode(Uint8Array.of(0xff)).codePointAt(0), 0xff); // ÿ
751 },
752};
753 
754// Per the WHATWG encoding spec (section 10.1.1), GBK's decoder is gb18030's decoder.
755// The .encoding property must still return "gbk", but decoding results must match gb18030.
756// https://encoding.spec.whatwg.org/#gbk-decoder
757export const gbkDecoderIsGb18030Decoder = {
758 test() {
759 const gbk = new TextDecoder('gbk');
760 const gb18030 = new TextDecoder('gb18030');
761 
762 // .encoding property must still distinguish the two
763 strictEqual(gbk.encoding, 'gbk');
764 strictEqual(gb18030.encoding, 'gb18030');
765 
766 // Decoding results must be identical. These byte pairs exercise boundary
767 // conditions where gbk and gb18030 would diverge if the ICU converter
768 // for gbk were used directly instead of delegating to gb18030.
769 const testBytes = [
770 [0, 255],
771 [128, 255],
772 [129, 48],
773 [129, 255],
774 [254, 48],
775 [254, 255],
776 [255, 0],
777 [255, 255],
778 ];
779 for (const bytes of testBytes) {
780 const u8 = Uint8Array.from(bytes);
781 strictEqual(
782 gbk.decode(u8),
783 gb18030.decode(u8),
784 `gbk and gb18030 must decode [${bytes}] identically`
785 );
786 }
787 },
788};
789 
790const gbVersionAndRangesTest = (encoding) => {
791 const loose = new TextDecoder(encoding);
792 const checkAll = (...list) => list.forEach((x) => check(...x));
793 const check = (bytes, str, invalid = false) => {
794 const fatal = new TextDecoder(encoding, { fatal: true });
795 const u8 = Uint8Array.from(bytes);
796 strictEqual(loose.decode(u8), str);
797 if (!invalid) strictEqual(fatal.decode(u8), str);
798 if (invalid) throws(() => fatal.decode(u8));
799 };
800 
801 check([0x84, 0x31, 0xa4, 0x36], '\uFFFC');
802 check([0x84, 0x31, 0xa4, 0x37], '\uFFFD');
803 check([0x84, 0x31, 0xa4, 0x38], '\uFFFE');
804 check([0x84, 0x31, 0xa4, 0x39], '\uFFFF');
805 check([0x84, 0x31, 0xa5, 0x30], '\uFFFD', true);
806 check([0x8f, 0x39, 0xfe, 0x39], '\uFFFD', true);
807 check([0x90, 0x30, 0x81, 0x30], String.fromCodePoint(0x1_00_00));
808 check([0x90, 0x30, 0x81, 0x31], String.fromCodePoint(0x1_00_01));
809 
810 check([0xe3, 0x32, 0x9a, 0x35], String.fromCodePoint(0x10_ff_ff));
811 check([0xe3, 0x32, 0x9a, 0x36], '\uFFFD', true);
812 check([0xe3, 0x32, 0x9a, 0x37], '\uFFFD', true);
813 
814 check([0xfe, 0x39, 0xfe, 0x39], '\uFFFD', true);
815 check([0xff, 0x39, 0xfe, 0x39], '\uFFFD9\uFFFD', true);
816 check([0xfe, 0x40, 0xfe, 0x39], '\uFA0C\uFFFD', true);
817 check([0xfe, 0x39, 0xff, 0x39], '\uFFFD9\uFFFD9', true);
818 check([0xfe, 0x39, 0xfe, 0x40], '\uFFFD9\uFA0C', true);
819 
820 checkAll(
821 [[0xa8, 0xbb], '\u0251'],
822 [[0xa8, 0xbc], '\u1E3F'],
823 [[0xa8, 0xbd], '\u0144']
824 );
825 check([0x81, 0x35, 0xf4, 0x36], '\u1E3E');
826 check([0x81, 0x35, 0xf4, 0x37], '\uE7C7');
827 check([0x81, 0x35, 0xf4, 0x38], '\u1E40');
828 
829 checkAll(
830 [[0xa6, 0xd9], '\uFE10'],
831 [[0xa6, 0xed], '\uFE18'],
832 [[0xa6, 0xf3], '\uFE19']
833 );
834 checkAll([[0xfe, 0x59], '\u9FB4'], [[0xfe, 0xa0], '\u9FBB']);
835};
836 
837export const gb18030VersionAndRanges = {
838 test() {
839 gbVersionAndRangesTest('gb18030');
840 },
841};
842 
843export const gbkVersionAndRanges = {
844 test() {
845 gbVersionAndRangesTest('gbk');
846 },
847};
848 
849// Verify that the WHATWG-required mapping corrections also produce the
850// correct output when the corrected byte sequences appear inside a larger
851// buffer (surrounded by ASCII), not only when they are the entire input.
852export const gb18030OverridesEmbedded = {
853 test() {
854 const d = new TextDecoder('gb18030');
855 
856 // 0x80 → U+20AC (Euro sign) surrounded by ASCII
857 strictEqual(d.decode(Uint8Array.of(0x41, 0x80, 0x42)), 'A\u20ACB');
858 
859 // Two-byte mapping corrections surrounded by ASCII
860 strictEqual(d.decode(Uint8Array.of(0x41, 0xa8, 0xbb, 0x42)), 'A\u0251B');
861 strictEqual(d.decode(Uint8Array.of(0x41, 0xa8, 0xbc, 0x42)), 'A\u1E3FB');
862 strictEqual(d.decode(Uint8Array.of(0x41, 0xa8, 0xbd, 0x42)), 'A\u0144B');
863 
864 // Vertical form corrections surrounded by ASCII
865 strictEqual(d.decode(Uint8Array.of(0x41, 0xa6, 0xd9, 0x42)), 'A\uFE10B');
866 
867 // CJK extension corrections surrounded by ASCII
868 strictEqual(d.decode(Uint8Array.of(0x41, 0xfe, 0x59, 0x42)), 'A\u9FB4B');
869 },
870};
871 
872export const replacementPushbackAsciiCharactersLoose = {
873 test() {
874 const vectors = {
875 big5: [
876 [[0x80], '\uFFFD'],
877 [[0x81, 0x40], '\uFFFD@'],
878 [[0x83, 0x5c], '\uFFFD\\'],
879 [[0x87, 0x87, 0x40], '\uFFFD@'],
880 [[0x81, 0x81], '\uFFFD'],
881 ],
882 'iso-2022-jp': [
883 [[0x1b, 0x24], '\uFFFD$'],
884 [[0x1b, 0x24, 0x40, 0x1b, 0x24], '\uFFFD\uFFFD'],
885 ],
886 'euc-jp': [
887 [[0x80], '\uFFFD'],
888 [[0x8d, 0x8d], '\uFFFD\uFFFD'],
889 [[0x8e, 0x8e], '\uFFFD'],
890 ],
891 };
892 
893 for (const [encoding, list] of Object.entries(vectors)) {
894 const d = new TextDecoder(encoding);
895 for (const [bytes, text] of list) {
896 strictEqual(d.decode(Uint8Array.from(bytes)), text);
897 }
898 }
899 },
900};
901 
902export const stickyMultibyteStateIso2022JpLoose = {
903 test() {
904 const vectors = [
905 [[27], '\uFFFD'],
906 [[27, 0x28], '\uFFFD('],
907 [[0x1b, 0x28, 0x49], ''],
908 ];
909 
910 const d = new TextDecoder('iso-2022-jp');
911 for (const [bytes, text] of vectors) {
912 strictEqual(d.decode(Uint8Array.of(0x40)), '@');
913 strictEqual(d.decode(Uint8Array.from(bytes)), text);
914 strictEqual(d.decode(Uint8Array.of(0x40)), '@');
915 strictEqual(d.decode(Uint8Array.of(0x2a)), '*');
916 strictEqual(d.decode(Uint8Array.of(0x42)), 'B');
917 }
918 },
919};
920 
921export const fatalStreamGb18030Gbk = {
922 test() {
923 for (const encoding of ['gb18030', 'gbk']) {
924 {
925 const d = new TextDecoder(encoding, { fatal: true });
926 strictEqual(d.decode(Uint8Array.of(0x80), { stream: true }), '\u20AC');
927 throws(() =>
928 d.decode(Uint8Array.of(0x81, 0x30, 0x21, 0x21, 0x21), {
929 stream: true,
930 })
931 );
932 strictEqual(d.decode(Uint8Array.of(0x80)), '\u20AC');
933 }
934 
935 {
936 const d = new TextDecoder(encoding, { fatal: true });
937 strictEqual(d.decode(Uint8Array.of(0x80), { stream: true }), '\u20AC');
938 throws(() =>
939 d.decode(Uint8Array.of(0x81, 0x30, 0x81, 0x42, 0x42), {
940 stream: true,
941 })
942 );
943 strictEqual(d.decode(Uint8Array.of(0x80)), '\u20AC');
944 }
945 }
946 },
947};
948 
949export const textDecoderStream = {
950 test() {
951 const stream = new TextDecoderStream('utf-16', {
952 fatal: true,
953 ignoreBOM: true,
954 });
955 strictEqual(stream.encoding, 'utf-16le');
956 strictEqual(stream.fatal, true);
957 strictEqual(stream.ignoreBOM, true);
958 
959 const enc = new TextEncoderStream();
960 strictEqual(enc.encoding, 'utf-8');
961 },
962};
963 
964// Per WHATWG Big5 decoder step 1, when end-of-queue is reached with a
965// pending lead byte, the decoder must return error (U+FFFD in replacement
966// mode, throw in fatal mode). This tests the streaming case where a lead
967// byte is buffered in one call and then flushed without a trail byte.
968export const big5OrphanedLeadOnFlush = {
969 test() {
970 // 0xA4 is a valid Big5 lead byte (e.g., first byte of 中 = 0xA4 0xA4).
971 // Streaming it alone, then flushing, must produce U+FFFD.
972 {
973 const dec = new TextDecoder('big5');
974 strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), '');
975 strictEqual(dec.decode(), '\uFFFD');
976 }
977 
978 // Fatal mode must throw on the orphaned lead.
979 {
980 const dec = new TextDecoder('big5', { fatal: true });
981 strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), '');
982 throws(() => dec.decode());
983 }
984 
985 // Orphaned lead followed by an invalid trail byte on flush: the lead
986 // must produce U+FFFD. 0x20 (space) is not a valid Big5 trail byte
987 // (valid trails are 0x40-0x7E and 0xA1-0xFE).
988 {
989 const dec = new TextDecoder('big5');
990 strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), '');
991 const result = dec.decode(Uint8Array.of(0x20));
992 // The orphaned lead must produce at least one U+FFFD.
993 ok(
994 result.includes('\uFFFD'),
995 `expected U+FFFD in output, got: ${JSON.stringify(result)}`
996 );
997 // The space byte must not be swallowed.
998 ok(
999 result.includes(' '),
1000 `expected space in output, got: ${JSON.stringify(result)}`
1001 );
1002 }
1003 
1004 // Streaming a complete pair across two calls must still work.
1005 {
1006 const dec = new TextDecoder('big5');
1007 strictEqual(dec.decode(Uint8Array.of(0xa4), { stream: true }), '');
1008 strictEqual(dec.decode(Uint8Array.of(0xa4)), '中');
1009 }
1010 },
1011};
1012 
1013// Test x-user-defined encoding per WHATWG spec
1014// https://encoding.spec.whatwg.org/#x-user-defined-decoder
1015export const xUserDefinedDecode = {
1016 test() {
1017 const decoder = new TextDecoder('x-user-defined');
1018 strictEqual(decoder.encoding, 'x-user-defined');
1019 strictEqual(decoder.fatal, false);
1020 strictEqual(decoder.ignoreBOM, false);
1021 
1022 // Test ASCII bytes (0x00-0x7F) - identity mapping
1023 strictEqual(decoder.decode(Uint8Array.of(0x41)), 'A');
1024 strictEqual(decoder.decode(Uint8Array.of(0x00)), '\u0000');
1025 strictEqual(decoder.decode(Uint8Array.of(0x7f)), '\u007F');
1026 
1027 // Test high bytes (0x80-0xFF) - map to Private Use Area U+F780-U+F7FF
1028 strictEqual(decoder.decode(Uint8Array.of(0x80)), '\uF780');
1029 strictEqual(decoder.decode(Uint8Array.of(0x81)), '\uF781');
1030 strictEqual(decoder.decode(Uint8Array.of(0xff)), '\uF7FF');
1031 
1032 // Test mixed sequence
1033 const mixed = new Uint8Array([0x00, 0x7f, 0x80, 0x81, 0xff]);
1034 strictEqual(decoder.decode(mixed), '\u0000\u007F\uF780\uF781\uF7FF');
1035 
1036 // Test empty input
1037 strictEqual(decoder.decode(new Uint8Array([])), '');
1038 strictEqual(decoder.decode(), '');
1039 
1040 // Test pure ASCII input (fast path)
1041 strictEqual(
1042 decoder.decode(new Uint8Array([0x48, 0x65, 0x6c, 0x6c, 0x6f])),
1043 'Hello'
1044 );
1045 
1046 // Test streaming (x-user-defined is single-byte, streaming is trivial)
1047 const streamDecoder = new TextDecoder('x-user-defined');
1048 let result = '';
1049 result += streamDecoder.decode(Uint8Array.of(0x41), { stream: true });
1050 result += streamDecoder.decode(Uint8Array.of(0x80), { stream: true });
1051 result += streamDecoder.decode(Uint8Array.of(0xff), { stream: true });
1052 result += streamDecoder.decode();
1053 strictEqual(result, 'A\uF780\uF7FF');
1054 },
1055};
1056 
1057// Test x-user-defined with fatal option (all 256 bytes are valid)
1058export const xUserDefinedFatal = {
1059 test() {
1060 const decoder = new TextDecoder('x-user-defined', { fatal: true });
1061 strictEqual(decoder.fatal, true);
1062 
1063 // All 256 byte values are valid, fatal mode should never throw
1064 for (let byte = 0; byte < 256; byte++) {
1065 const decoded = decoder.decode(Uint8Array.of(byte));
1066 if (byte < 0x80) {
1067 strictEqual(decoded.codePointAt(0), byte);
1068 } else {
1069 strictEqual(decoded.codePointAt(0), 0xf700 + byte);
1070 }
1071 }
1072 },
1073};
1074 
1075// Verify that streaming with zero-length input works for every legacy
1076// encoding handled by the Rust LegacyDecoder. An empty chunk in streaming
1077// mode must produce an empty string and leave the decoder in a valid state
1078// for subsequent calls.
1079export const legacyStreamEmptyInput = {
1080 test() {
1081 const encodings = [
1082 'big5',
1083 'euc-jp',
1084 'euc-kr',
1085 'gb18030',
1086 'gbk',
1087 'iso-2022-jp',
1088 'shift_jis',
1089 'windows-1252',
1090 'x-user-defined',
1091 ];
1092 
1093 const empty = new Uint8Array(0);
1094 
1095 for (const label of encodings) {
1096 for (const fatal of [false, true]) {
1097 const dec = new TextDecoder(label, { fatal });
1098 
1099 // Empty stream chunk must produce empty string.
1100 strictEqual(
1101 dec.decode(empty, { stream: true }),
1102 '',
1103 `${label} (fatal=${fatal}): empty stream chunk should be ''`
1104 );
1105 
1106 // A second empty stream chunk must also be fine.
1107 strictEqual(
1108 dec.decode(empty, { stream: true }),
1109 '',
1110 `${label} (fatal=${fatal}): second empty stream chunk should be ''`
1111 );
1112 
1113 // Final flush with no pending bytes must produce empty string.
1114 strictEqual(
1115 dec.decode(),
1116 '',
1117 `${label} (fatal=${fatal}): flush after empty chunks should be ''`
1118 );
1119 
1120 // Decoder must still work normally after the empty-stream sequence.
1121 // Feed a single ASCII byte to verify.
1122 strictEqual(
1123 dec.decode(Uint8Array.of(0x41)),
1124 'A',
1125 `${label} (fatal=${fatal}): decode 'A' after empty stream should work`
1126 );
1127 }
1128 }
1129 },
1130};