Skip to content
File

Blob: src/workerd/api/node/tests/string-decoder-test.js

javascript584 lines
1// Copyright (c) 2017-2022 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4//
5// Adapted from Node.js. Copyright Joyent, Inc. and other Node contributors.
6//
7// Permission is hereby granted, free of charge, to any person obtaining a
8// copy of this software and associated documentation files (the
9// "Software"), to deal in the Software without restriction, including
10// without limitation the rights to use, copy, modify, merge, publish,
11// distribute, sublicense, and/or sell copies of the Software, and to permit
12// persons to whom the Software is furnished to do so, subject to the
13// following conditions:
14//
15// The above copyright notice and this permission notice shall be included
16// in all copies or substantial portions of the Software.
17//
18// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
19// OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
20// MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN
21// NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM,
22// DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
23// OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE
24// USE OR OTHER DEALINGS IN THE SOFTWARE.
25 
26import { ok, fail, strictEqual, throws } from 'node:assert';
27 
28import { Buffer } from 'node:buffer';
29 
30import { StringDecoder } from 'node:string_decoder';
31 
32import * as string_decoder from 'node:string_decoder';
33 
34if (string_decoder.StringDecoder !== StringDecoder) {
35 throw new Error('Incorrect default exports');
36}
37 
38function getArrayBufferViews(buf) {
39 const { buffer, byteOffset, byteLength } = buf;
40 
41 const out = [];
42 
43 const arrayBufferViews = [
44 Int8Array,
45 Uint8Array,
46 Uint8ClampedArray,
47 Int16Array,
48 Uint16Array,
49 Int32Array,
50 Uint32Array,
51 Float16Array,
52 Float32Array,
53 Float64Array,
54 BigInt64Array,
55 BigUint64Array,
56 DataView,
57 ];
58 
59 for (const type of arrayBufferViews) {
60 const { BYTES_PER_ELEMENT = 1 } = type;
61 if (byteLength % BYTES_PER_ELEMENT === 0) {
62 out.push(new type(buffer, byteOffset, byteLength / BYTES_PER_ELEMENT));
63 }
64 }
65 return out;
66}
67 
68// Test verifies that StringDecoder will correctly decode the given input
69// buffer with the given encoding to the expected output. It will attempt all
70// possible ways to write() the input buffer, see writeSequences(). The
71// singleSequence allows for easy debugging of a specific sequence which is
72// useful in case of test failures.
73function test_(encoding, input, expected, singleSequence) {
74 let sequences;
75 if (!singleSequence) {
76 sequences = writeSequences(input.length);
77 } else {
78 sequences = [singleSequence];
79 }
80 const hexNumberRE = /.{2}/g;
81 sequences.forEach((sequence) => {
82 const decoder = new StringDecoder(encoding);
83 let output = '';
84 sequence.forEach((write) => {
85 output += decoder.write(input.slice(write[0], write[1]));
86 });
87 output += decoder.end();
88 if (output !== expected) {
89 const message =
90 `Expected "${unicodeEscape(expected)}", ` +
91 `but got "${unicodeEscape(output)}"\n` +
92 `input: ${input.toString('hex').match(hexNumberRE)}\n` +
93 `Write sequence: ${JSON.stringify(sequence)}\n` +
94 `Full Decoder State: ${decoder}`;
95 fail(message);
96 }
97 });
98}
99 
100// writeSequences returns an array of arrays that describes all possible ways a
101// buffer of the given length could be split up and passed to sequential write
102// calls.
103//
104// e.G. writeSequences(3) will return: [
105// [ [ 0, 3 ] ],
106// [ [ 0, 2 ], [ 2, 3 ] ],
107// [ [ 0, 1 ], [ 1, 3 ] ],
108// [ [ 0, 1 ], [ 1, 2 ], [ 2, 3 ] ]
109// ]
110function writeSequences(length, start, sequence) {
111 if (start === undefined) {
112 start = 0;
113 sequence = [];
114 } else if (start === length) {
115 return [sequence];
116 }
117 let sequences = [];
118 for (let end = length; end > start; end--) {
119 const subSequence = sequence.concat([[start, end]]);
120 const subSequences = writeSequences(length, end, subSequence, sequences);
121 sequences = sequences.concat(subSequences);
122 }
123 return sequences;
124}
125 
126// unicodeEscape prints the str contents as unicode escape codes.
127function unicodeEscape(str) {
128 let r = '';
129 for (let i = 0; i < str.length; i++) {
130 r += `\\u${str.charCodeAt(i).toString(16)}`;
131 }
132 return r;
133}
134 
135export const stringDecoder = {
136 test(ctrl, env, ctx) {
137 // Test default encoding
138 let decoder = new StringDecoder();
139 strictEqual(decoder.encoding, 'utf8');
140 
141 // Should work without 'new' keyword
142 const decoder2 = {};
143 StringDecoder.call(decoder2);
144 strictEqual(decoder2.encoding, 'utf8');
145 
146 // UTF-8
147 test_('utf-8', Buffer.from('$', 'utf-8'), '$');
148 test_('utf-8', Buffer.from('¢', 'utf-8'), '¢');
149 test_('utf-8', Buffer.from('€', 'utf-8'), '€');
150 test_('utf-8', Buffer.from('𤭢', 'utf-8'), '𤭢');
151 // A mixed ascii and non-ascii string
152 // Test stolen from deps/v8/test/cctest/test-strings.cc
153 // U+02E4 -> CB A4
154 // U+0064 -> 64
155 // U+12E4 -> E1 8B A4
156 // U+0030 -> 30
157 // U+3045 -> E3 81 85
158 
159 test_(
160 'utf-8',
161 Buffer.from([0xcb, 0xa4, 0x64, 0xe1, 0x8b, 0xa4, 0x30, 0xe3, 0x81, 0x85]),
162 '\u02e4\u0064\u12e4\u0030\u3045'
163 );
164 
165 // Some invalid input, known to have caused trouble with chunking
166 // in https://github.com/nodejs/node/pull/7310#issuecomment-226445923
167 // 00: |00000000 ASCII
168 // 41: |01000001 ASCII
169 // B8: 10|111000 continuation
170 // CC: 110|01100 two-byte head
171 // E2: 1110|0010 three-byte head
172 // F0: 11110|000 four-byte head
173 // F1: 11110|001'another four-byte head
174 // FB: 111110|11 "five-byte head", not UTF-8
175 test_('utf-8', Buffer.from('C9B5A941', 'hex'), '\u0275\ufffdA');
176 test_('utf-8', Buffer.from('E2', 'hex'), '\ufffd');
177 test_('utf-8', Buffer.from('E241', 'hex'), '\ufffdA');
178 test_('utf-8', Buffer.from('CCCCB8', 'hex'), '\ufffd\u0338');
179 test_('utf-8', Buffer.from('F0B841', 'hex'), '\ufffdA');
180 test_('utf-8', Buffer.from('F1CCB8', 'hex'), '\ufffd\u0338');
181 test_('utf-8', Buffer.from('F0FB00', 'hex'), '\ufffd\ufffd\0');
182 test_('utf-8', Buffer.from('CCE2B8B8', 'hex'), '\ufffd\u2e38');
183 test_('utf-8', Buffer.from('E2B8CCB8', 'hex'), '\ufffd\u0338');
184 test_('utf-8', Buffer.from('E2FBCC01', 'hex'), '\ufffd\ufffd\ufffd\u0001');
185 test_('utf-8', Buffer.from('CCB8CDB9', 'hex'), '\u0338\u0379');
186 // CESU-8 of U+1D40D
187 
188 // V8 has changed their invalid UTF-8 handling, see
189 // https://chromium-review.googlesource.com/c/v8/v8/+/671020 for more info.
190 test_(
191 'utf-8',
192 Buffer.from('EDA0B5EDB08D', 'hex'),
193 '\ufffd\ufffd\ufffd\ufffd\ufffd\ufffd'
194 );
195 
196 // UCS-2
197 test_('ucs2', Buffer.from('ababc', 'ucs2'), 'ababc');
198 
199 // UTF-16LE
200 test_('utf16le', Buffer.from('3DD84DDC', 'hex'), '\ud83d\udc4d'); // thumbs up
201 
202 // Additional UTF-8 tests
203 decoder = new StringDecoder('utf8');
204 strictEqual(decoder.write(Buffer.from('E1', 'hex')), '');
205 
206 // A quick test for lastChar, lastNeed & lastTotal which are undocumented.
207 ok(decoder.lastChar.equals(new Uint8Array([0xe1, 0, 0, 0])));
208 strictEqual(decoder.lastNeed, 2);
209 strictEqual(decoder.lastTotal, 3);
210 
211 strictEqual(decoder.end(), '\ufffd');
212 
213 // ArrayBufferView tests
214 const arrayBufferViewStr = 'String for ArrayBufferView tests\n';
215 const inputBuffer = Buffer.from(arrayBufferViewStr.repeat(8), 'utf8');
216 for (const expectView of getArrayBufferViews(inputBuffer)) {
217 strictEqual(decoder.write(expectView), inputBuffer.toString('utf8'));
218 strictEqual(decoder.end(), '');
219 }
220 
221 decoder = new StringDecoder('utf8');
222 strictEqual(decoder.write(Buffer.from('E18B', 'hex')), '');
223 strictEqual(decoder.end(), '\ufffd');
224 
225 decoder = new StringDecoder('utf8');
226 strictEqual(decoder.write(Buffer.from('\ufffd')), '\ufffd');
227 strictEqual(decoder.end(), '');
228 
229 decoder = new StringDecoder('utf8');
230 strictEqual(
231 decoder.write(Buffer.from('\ufffd\ufffd\ufffd')),
232 '\ufffd\ufffd\ufffd'
233 );
234 strictEqual(decoder.end(), '');
235 
236 decoder = new StringDecoder('utf8');
237 strictEqual(decoder.write(Buffer.from('EFBFBDE2', 'hex')), '\ufffd');
238 strictEqual(decoder.end(), '\ufffd');
239 
240 decoder = new StringDecoder('utf8');
241 strictEqual(decoder.write(Buffer.from('F1', 'hex')), '');
242 strictEqual(decoder.write(Buffer.from('41F2', 'hex')), '\ufffdA');
243 strictEqual(decoder.end(), '\ufffd');
244 
245 // Additional utf8Text test
246 decoder = new StringDecoder('utf8');
247 strictEqual(decoder.text(Buffer.from([0x41]), 2), '');
248 
249 // Additional UTF-16LE surrogate pair tests
250 decoder = new StringDecoder('utf16le');
251 strictEqual(decoder.write(Buffer.from('3DD8', 'hex')), '');
252 strictEqual(decoder.write(Buffer.from('4D', 'hex')), '');
253 strictEqual(decoder.write(Buffer.from('DC', 'hex')), '\ud83d\udc4d');
254 strictEqual(decoder.end(), '');
255 
256 decoder = new StringDecoder('utf16le');
257 strictEqual(decoder.write(Buffer.from('3DD8', 'hex')), '');
258 strictEqual(decoder.end(), '\ud83d');
259 
260 decoder = new StringDecoder('utf16le');
261 strictEqual(decoder.write(Buffer.from('3DD8', 'hex')), '');
262 strictEqual(decoder.write(Buffer.from('4D', 'hex')), '');
263 strictEqual(decoder.end(), '\ud83d');
264 
265 decoder = new StringDecoder('utf16le');
266 strictEqual(decoder.write(Buffer.from('3DD84D', 'hex')), '\ud83d');
267 strictEqual(decoder.end(), '');
268 
269 // Regression test for https://github.com/nodejs/node/issues/22358
270 // (unaligned UTF-16 access).
271 decoder = new StringDecoder('utf16le');
272 strictEqual(decoder.write(Buffer.alloc(1)), '');
273 strictEqual(decoder.write(Buffer.alloc(20)), '\0'.repeat(10));
274 strictEqual(decoder.write(Buffer.alloc(48)), '\0'.repeat(24));
275 strictEqual(decoder.end(), '');
276 
277 // Regression tests for https://github.com/nodejs/node/issues/22626
278 // (not enough replacement chars when having seen more than one byte of an
279 // incomplete multibyte characters).
280 decoder = new StringDecoder('utf8');
281 strictEqual(decoder.write(Buffer.from('f69b', 'hex')), '');
282 strictEqual(decoder.write(Buffer.from('d1', 'hex')), '\ufffd\ufffd');
283 strictEqual(decoder.end(), '\ufffd');
284 strictEqual(decoder.write(Buffer.from('f4', 'hex')), '');
285 strictEqual(decoder.write(Buffer.from('bde5', 'hex')), '\ufffd\ufffd');
286 strictEqual(decoder.end(), '\ufffd');
287 
288 throws(() => new StringDecoder(1), {
289 code: 'ERR_UNKNOWN_ENCODING',
290 name: 'TypeError',
291 message: 'Unknown encoding: 1',
292 });
293 
294 throws(() => new StringDecoder('test'), {
295 code: 'ERR_UNKNOWN_ENCODING',
296 name: 'TypeError',
297 message: 'Unknown encoding: test',
298 });
299 
300 throws(() => new StringDecoder('utf8').write(null), {
301 code: 'ERR_INVALID_ARG_TYPE',
302 name: 'TypeError',
303 });
304 
305 throws(
306 () => new StringDecoder('utf8').__proto__.write(Buffer.from('abc')),
307 {
308 code: 'ERR_INVALID_THIS',
309 }
310 );
311 },
312};
313 
314export const stringDecoderEnd = {
315 test(ctrl, env, ctx) {
316 const encodings = ['base64', 'base64url', 'hex', 'utf8', 'utf16le', 'ucs2'];
317 const bufs = ['☃💩', 'asdf'].map((b) => Buffer.from(b));
318 
319 // Also test just arbitrary bytes from 0-15.
320 for (let i = 1; i <= 16; i++) {
321 const bytes = '.'
322 .repeat(i - 1)
323 .split('.')
324 .map((_, j) => j + 0x78);
325 bufs.push(Buffer.from(bytes));
326 }
327 
328 encodings.forEach(testEncoding);
329 
330 testEnd('utf8', Buffer.of(0xe2), Buffer.of(0x61), '\uFFFDa');
331 testEnd('utf8', Buffer.of(0xe2), Buffer.of(0x82), '\uFFFD\uFFFD');
332 testEnd('utf8', Buffer.of(0xe2), Buffer.of(0xe2), '\uFFFD\uFFFD');
333 testEnd('utf8', Buffer.of(0xe2, 0x82), Buffer.of(0x61), '\uFFFDa');
334 testEnd('utf8', Buffer.of(0xe2, 0x82), Buffer.of(0xac), '\uFFFD\uFFFD');
335 testEnd('utf8', Buffer.of(0xe2, 0x82), Buffer.of(0xe2), '\uFFFD\uFFFD');
336 testEnd('utf8', Buffer.of(0xe2, 0x82, 0xac), Buffer.of(0x61), '€a');
337 
338 testEnd('utf16le', Buffer.of(0x3d), Buffer.of(0x61, 0x00), 'a');
339 testEnd('utf16le', Buffer.of(0x3d), Buffer.of(0xd8, 0x4d, 0xdc), '\u4DD8');
340 testEnd('utf16le', Buffer.of(0x3d, 0xd8), Buffer.of(), '\uD83D');
341 testEnd('utf16le', Buffer.of(0x3d, 0xd8), Buffer.of(0x61, 0x00), '\uD83Da');
342 testEnd(
343 'utf16le',
344 Buffer.of(0x3d, 0xd8),
345 Buffer.of(0x4d, 0xdc),
346 '\uD83D\uDC4D'
347 );
348 testEnd('utf16le', Buffer.of(0x3d, 0xd8, 0x4d), Buffer.of(), '\uD83D');
349 testEnd(
350 'utf16le',
351 Buffer.of(0x3d, 0xd8, 0x4d),
352 Buffer.of(0x61, 0x00),
353 '\uD83Da'
354 );
355 testEnd('utf16le', Buffer.of(0x3d, 0xd8, 0x4d), Buffer.of(0xdc), '\uD83D');
356 testEnd(
357 'utf16le',
358 Buffer.of(0x3d, 0xd8, 0x4d, 0xdc),
359 Buffer.of(0x61, 0x00),
360 '👍a'
361 );
362 
363 testEnd('base64', Buffer.of(0x61), Buffer.of(), 'YQ==');
364 testEnd('base64', Buffer.of(0x61), Buffer.of(0x61), 'YQ==YQ==');
365 testEnd('base64', Buffer.of(0x61, 0x61), Buffer.of(), 'YWE=');
366 testEnd('base64', Buffer.of(0x61, 0x61), Buffer.of(0x61), 'YWE=YQ==');
367 testEnd('base64', Buffer.of(0x61, 0x61, 0x61), Buffer.of(), 'YWFh');
368 testEnd('base64', Buffer.of(0x61, 0x61, 0x61), Buffer.of(0x61), 'YWFhYQ==');
369 
370 testEnd('base64url', Buffer.of(0x61), Buffer.of(), 'YQ');
371 testEnd('base64url', Buffer.of(0x61), Buffer.of(0x61), 'YQYQ');
372 testEnd('base64url', Buffer.of(0x61, 0x61), Buffer.of(), 'YWE');
373 testEnd('base64url', Buffer.of(0x61, 0x61), Buffer.of(0x61), 'YWEYQ');
374 testEnd('base64url', Buffer.of(0x61, 0x61, 0x61), Buffer.of(), 'YWFh');
375 testEnd(
376 'base64url',
377 Buffer.of(0x61, 0x61, 0x61),
378 Buffer.of(0x61),
379 'YWFhYQ'
380 );
381 
382 function testEncoding(encoding) {
383 bufs.forEach((buf) => {
384 testBuf(encoding, buf);
385 });
386 }
387 
388 function testBuf(encoding, buf) {
389 // Write one byte at a time.
390 let s = new StringDecoder(encoding);
391 let res1 = '';
392 for (let i = 0; i < buf.length; i++) {
393 res1 += s.write(buf.slice(i, i + 1));
394 }
395 res1 += s.end();
396 
397 // Write the whole buffer at once.
398 let res2 = '';
399 s = new StringDecoder(encoding);
400 res2 += s.write(buf);
401 res2 += s.end();
402 
403 // .toString() on the buffer
404 const res3 = buf.toString(encoding);
405 
406 // One byte at a time should match toString
407 strictEqual(res1, res3);
408 // All bytes at once should match toString
409 strictEqual(res2, res3);
410 }
411 
412 function testEnd(encoding, incomplete, next, expected) {
413 let res = '';
414 const s = new StringDecoder(encoding);
415 res += s.write(incomplete);
416 res += s.end();
417 res += s.write(next);
418 res += s.end();
419 
420 strictEqual(res, expected);
421 }
422 },
423};
424 
425export const stringDecoderFuzz = {
426 test(ctrl, env, ctx) {
427 function rand(max) {
428 return Math.floor(Math.random() * max);
429 }
430 
431 function randBuf(maxLen) {
432 const buf = Buffer.allocUnsafe(rand(maxLen));
433 for (let i = 0; i < buf.length; i++) buf[i] = rand(256);
434 return buf;
435 }
436 
437 const encodings = [
438 'utf16le',
439 'utf8',
440 'ascii',
441 'hex',
442 'base64',
443 'latin1',
444 'base64url',
445 ];
446 
447 function runSingleFuzzTest() {
448 const enc = encodings[rand(encodings.length)];
449 const sd = new StringDecoder(enc);
450 const bufs = [];
451 const strings = [];
452 
453 const N = rand(10);
454 for (let i = 0; i < N; ++i) {
455 const buf = randBuf(50);
456 bufs.push(buf);
457 strings.push(sd.write(buf));
458 }
459 strings.push(sd.end());
460 
461 strictEqual(
462 strings.join(''),
463 Buffer.concat(bufs).toString(enc),
464 `Mismatch:\n${strings}\n` +
465 bufs.map((buf) => buf.toString('hex')) +
466 `\nfor encoding ${enc}`
467 );
468 }
469 
470 const start = Date.now();
471 while (Date.now() - start < 100) runSingleFuzzTest();
472 },
473};
474 
475export const stringDecoderHacking = {
476 test(ctrl, env, ctx) {
477 throws(
478 () => {
479 const sd = new StringDecoder();
480 const sym = Object.getOwnPropertySymbols(sd)[0];
481 sd[sym] = 'not a buffer';
482 sd.write(Buffer.from("this shouldn't crash"));
483 },
484 {
485 name: 'TypeError',
486 }
487 );
488 
489 throws(
490 () => {
491 const sd = new StringDecoder();
492 const sym = Object.getOwnPropertySymbols(sd)[0];
493 sd[sym] = Buffer.alloc(1);
494 sd.write(Buffer.from("this shouldn't crash"));
495 },
496 {
497 message: 'Invalid StringDecoder',
498 }
499 );
500 
501 throws(
502 () => {
503 const sd = new StringDecoder();
504 const sym = Object.getOwnPropertySymbols(sd)[0];
505 sd[sym] = Buffer.alloc(9);
506 sd.write(Buffer.from("this shouldn't crash"));
507 },
508 {
509 message: 'Invalid StringDecoder',
510 }
511 );
512 
513 throws(
514 () => {
515 const sd = new StringDecoder();
516 const sym = Object.getOwnPropertySymbols(sd)[0];
517 sd[sym][5] = 100;
518 sd.write(Buffer.from("this shouldn't crash"));
519 },
520 {
521 message: 'Buffered bytes cannot exceed 4',
522 }
523 );
524 
525 throws(
526 () => {
527 const sd = new StringDecoder();
528 const sym = Object.getOwnPropertySymbols(sd)[0];
529 sd[sym][4] = 100;
530 sd.write(Buffer.from("this shouldn't crash"));
531 },
532 {
533 message: 'Missing bytes cannot exceed 4',
534 }
535 );
536 
537 throws(
538 () => {
539 const sd = new StringDecoder();
540 const sym = Object.getOwnPropertySymbols(sd)[0];
541 sd[sym][6] = 100;
542 sd.write(Buffer.from("this shouldn't crash"));
543 },
544 {
545 message: 'Invalid StringDecoder state',
546 }
547 );
548 
549 throws(
550 () => {
551 const sd = new StringDecoder();
552 const sym = Object.getOwnPropertySymbols(sd)[0];
553 sd[sym][4] = 3;
554 sd[sym][5] = 2;
555 sd.write(Buffer.from("this shouldn't crash"));
556 },
557 {
558 message: 'Invalid StringDecoder state',
559 }
560 );
561 
562 {
563 // fuzz a bit with random values
564 const messages = [
565 'Invalid StringDecoder state',
566 'Missing bytes cannot exceed 4',
567 'Buffered bytes cannot exceed 4',
568 ];
569 for (let n = 0; n < 255; n++) {
570 try {
571 const sd = new StringDecoder();
572 const sym = Object.getOwnPropertySymbols(sd)[0];
573 crypto.getRandomValues(sd[sym]);
574 sd.write(Buffer.from("this shouldn't crash"));
575 } catch (err) {
576 if (!messages.includes(err.message)) {
577 throw err;
578 }
579 }
580 }
581 }
582 },
583};