Skip to content
File

Blob: src/node/punycode.ts

typescript477 lines
1// Copyright (c) 2017-2022 Cloudflare, Inc.
2// Licensed under the Apache 2.0 license found in the LICENSE file or at:
3// https://opensource.org/licenses/Apache-2.0
4// Copyright Mathias Bynens <https://mathiasbynens.be/>
5//
6// Permission is hereby granted, free of charge, to any person obtaining
7// a copy of this software and associated documentation files (the
8// "Software"), to deal in the Software without restriction, including
9// without limitation the rights to use, copy, modify, merge, publish,
10// distribute, sublicense, and/or sell copies of the Software, and to
11// permit persons to whom the Software is furnished to do so, subject to
12// the following conditions:
13//
14// The above copyright notice and this permission notice shall be
15// included in all copies or substantial portions of the Software.
16//
17// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
18// EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
19// MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
20// NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE
21// LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
22// OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
23// WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
24 
25/** Highest positive signed 32-bit float value */
26const maxInt = 2147483647; // aka. 0x7FFFFFFF or 2^31-1
27 
28/** Bootstring parameters */
29const base = 36;
30const tMin = 1;
31const tMax = 26;
32const skew = 38;
33const damp = 700;
34const initialBias = 72;
35const initialN = 128; // 0x80
36const delimiter = '-'; // '\x2D'
37 
38/** Regular expressions */
39const regexPunycode = /^xn--/;
40const regexNonASCII = /[^\0-\x7F]/; // Note: U+007F DEL is excluded too.
41const regexSeparators = /[\x2E\u3002\uFF0E\uFF61]/g; // RFC 3490 separators
42 
43/** Error messages */
44const errors = {
45 overflow: 'Overflow: input needs wider integers to process',
46 'not-basic': 'Illegal input >= 0x80 (not a basic code point)',
47 'invalid-input': 'Invalid input',
48};
49 
50/** Convenience shortcuts */
51const baseMinusTMin = base - tMin;
52const floor = Math.floor;
53const stringFromCharCode = String.fromCharCode;
54 
55/*--------------------------------------------------------------------------*/
56 
57/**
58 * A generic error utility function.
59 * @private
60 * @param {String} type The error type.
61 * @returns {Error} Throws a `RangeError` with the applicable error message.
62 */
63function error(type: keyof typeof errors): void {
64 throw new RangeError(errors[type]);
65}
66 
67/**
68 * A generic `Array#map` utility function.
69 * @private
70 * @param {Array} array The array to iterate over.
71 * @param {Function} callback The function that gets called for every array
72 * item.
73 * @returns {Array} A new array of values returned by the callback function.
74 */
75function map<T>(array: T[], callback: (item: T) => T): T[] {
76 const result = [];
77 let length = array.length;
78 while (length--) {
79 result[length] = callback(array[length] as T);
80 }
81 return result;
82}
83 
84/**
85 * A simple `Array#map`-like wrapper to work with domain name strings or email
86 * addresses.
87 * @private
88 * @param {String} domain The domain name or email address.
89 * @param {Function} callback The function that gets called for every
90 * character.
91 * @returns {String} A new string of characters returned by the callback
92 * function.
93 */
94function mapDomain(domain: string, callback: (char: string) => string): string {
95 const parts = domain.split('@');
96 let result = '';
97 if (parts.length > 1) {
98 // In email addresses, only the domain name should be punycoded. Leave
99 // the local part (i.e. everything up to `@`) intact.
100 result = parts[0] + '@'; // eslint-disable-line @typescript-eslint/restrict-plus-operands
101 domain = parts[1] as string;
102 }
103 // Avoid `split(regex)` for IE8 compatibility. See #17.
104 domain = domain.replace(regexSeparators, '\x2E');
105 const labels = domain.split('.');
106 const encoded = map(labels, callback).join('.');
107 return result + encoded;
108}
109 
110/**
111 * Creates an array containing the numeric code points of each Unicode
112 * character in the string. While JavaScript uses UCS-2 internally,
113 * this function will convert a pair of surrogate halves (each of which
114 * UCS-2 exposes as separate characters) into a single code point,
115 * matching UTF-16.
116 * @see `punycode.ucs2.encode`
117 * @see <https://mathiasbynens.be/notes/javascript-encoding>
118 * @memberOf punycode.ucs2
119 * @name decode
120 * @param {String} string The Unicode input string (UCS-2).
121 * @returns {Array} The new array of code points.
122 */
123function ucs2decode(string: string): number[] {
124 const output = [];
125 let counter = 0;
126 const length = string.length;
127 while (counter < length) {
128 const value = string.charCodeAt(counter++);
129 if (value >= 0xd800 && value <= 0xdbff && counter < length) {
130 // It's a high surrogate, and there is a next character.
131 const extra = string.charCodeAt(counter++);
132 if ((extra & 0xfc00) == 0xdc00) {
133 // Low surrogate.
134 output.push(((value & 0x3ff) << 10) + (extra & 0x3ff) + 0x10000);
135 } else {
136 // It's an unmatched surrogate; only append this code unit, in case the
137 // next code unit is the high surrogate of a surrogate pair.
138 output.push(value);
139 counter--;
140 }
141 } else {
142 output.push(value);
143 }
144 }
145 return output;
146}
147 
148/**
149 * Creates a string based on an array of numeric code points.
150 * @see `punycode.ucs2.decode`
151 * @memberOf punycode.ucs2
152 * @name encode
153 * @param {Array} codePoints The array of numeric code points.
154 * @returns {String} The new Unicode string (UCS-2).
155 */
156const ucs2encode = (codePoints: number[]): string =>
157 String.fromCodePoint(...codePoints);
158 
159/**
160 * Converts a basic code point into a digit/integer.
161 * @see `digitToBasic()`
162 * @private
163 * @param {Number} codePoint The basic numeric code point value.
164 * @returns {Number} The numeric value of a basic code point (for use in
165 * representing integers) in the range `0` to `base - 1`, or `base` if
166 * the code point does not represent a value.
167 */
168const basicToDigit = function (codePoint: number): number {
169 if (codePoint >= 0x30 && codePoint < 0x3a) {
170 return 26 + (codePoint - 0x30);
171 }
172 if (codePoint >= 0x41 && codePoint < 0x5b) {
173 return codePoint - 0x41;
174 }
175 if (codePoint >= 0x61 && codePoint < 0x7b) {
176 return codePoint - 0x61;
177 }
178 return base;
179};
180 
181/**
182 * Converts a digit/integer into a basic code point.
183 * @see `basicToDigit()`
184 * @private
185 * @param {Number} digit The numeric value of a basic code point.
186 * @returns {Number} The basic code point whose value (when used for
187 * representing integers) is `digit`, which needs to be in the range
188 * `0` to `base - 1`. If `flag` is non-zero, the uppercase form is
189 * used; else, the lowercase form is used. The behavior is undefined
190 * if `flag` is non-zero and `digit` has no uppercase form.
191 */
192const digitToBasic = function (digit: number, flag: number): number {
193 // 0..25 map to ASCII a..z or A..Z
194 // 26..35 map to ASCII 0..9
195 return digit + 22 + 75 * Number(digit < 26) - (Number(flag != 0) << 5);
196};
197 
198/**
199 * Bias adaptation function as per section 3.4 of RFC 3492.
200 * https://tools.ietf.org/html/rfc3492#section-3.4
201 * @private
202 */
203const adapt = function (
204 delta: number,
205 numPoints: number,
206 firstTime: boolean
207): number {
208 let k = 0;
209 delta = firstTime ? floor(delta / damp) : delta >> 1;
210 delta += floor(delta / numPoints);
211 for (
212 ;
213 /* no initialization */ delta > (baseMinusTMin * tMax) >> 1;
214 k += base
215 ) {
216 delta = floor(delta / baseMinusTMin);
217 }
218 return floor(k + ((baseMinusTMin + 1) * delta) / (delta + skew));
219};
220 
221/**
222 * Converts a Punycode string of ASCII-only symbols to a string of Unicode
223 * symbols.
224 * @memberOf punycode
225 * @param {String} input The Punycode string of ASCII-only symbols.
226 * @returns {String} The resulting string of Unicode symbols.
227 */
228export const decode = function (input: string): string {
229 // Don't use UCS-2.
230 const output = [];
231 const inputLength = input.length;
232 let i = 0;
233 let n = initialN;
234 let bias = initialBias;
235 
236 // Handle the basic code points: let `basic` be the number of input code
237 // points before the last delimiter, or `0` if there is none, then copy
238 // the first basic code points to the output.
239 
240 let basic = input.lastIndexOf(delimiter);
241 if (basic < 0) {
242 basic = 0;
243 }
244 
245 for (let j = 0; j < basic; ++j) {
246 // if it's not a basic code point
247 if (input.charCodeAt(j) >= 0x80) {
248 error('not-basic');
249 }
250 output.push(input.charCodeAt(j));
251 }
252 
253 // Main decoding loop: start just after the last delimiter if any basic code
254 // points were copied; start at the beginning otherwise.
255 
256 for (
257 let index = basic > 0 ? basic + 1 : 0;
258 index < inputLength /* no final expression */;
259 ) {
260 // `index` is the index of the next character to be consumed.
261 // Decode a generalized variable-length integer into `delta`,
262 // which gets added to `i`. The overflow checking is easier
263 // if we increase `i` as we go, then subtract off its starting
264 // value at the end to obtain `delta`.
265 const oldi = i;
266 for (let w = 1, k = base /* no condition */; ; k += base) {
267 if (index >= inputLength) {
268 error('invalid-input');
269 }
270 
271 const digit = basicToDigit(input.charCodeAt(index++));
272 
273 if (digit >= base) {
274 error('invalid-input');
275 }
276 if (digit > floor((maxInt - i) / w)) {
277 error('overflow');
278 }
279 
280 i += digit * w;
281 const t = k <= bias ? tMin : k >= bias + tMax ? tMax : k - bias;
282 
283 if (digit < t) {
284 break;
285 }
286 
287 const baseMinusT = base - t;
288 if (w > floor(maxInt / baseMinusT)) {
289 error('overflow');
290 }
291 
292 w *= baseMinusT;
293 }
294 
295 const out = output.length + 1;
296 bias = adapt(i - oldi, out, oldi == 0);
297 
298 // `i` was supposed to wrap around from `out` to `0`,
299 // incrementing `n` each time, so we'll fix that now:
300 if (floor(i / out) > maxInt - n) {
301 error('overflow');
302 }
303 
304 n += floor(i / out);
305 i %= out;
306 
307 // Insert `n` at position `i` of the output.
308 output.splice(i++, 0, n);
309 }
310 
311 return String.fromCodePoint(...output);
312};
313 
314/**
315 * Converts a string of Unicode symbols (e.g. a domain name label) to a
316 * Punycode string of ASCII-only symbols.
317 * @memberOf punycode
318 * @param {String} input The string of Unicode symbols.
319 * @returns {String} The resulting Punycode string of ASCII-only symbols.
320 */
321export const encode = function (input: string | number[]): string {
322 const output = [];
323 
324 // Convert the input in UCS-2 to an array of Unicode code points.
325 input = ucs2decode(input as string);
326 
327 // Cache the length.
328 const inputLength = input.length;
329 
330 // Initialize the state.
331 let n = initialN;
332 let delta = 0;
333 let bias = initialBias;
334 
335 // Handle the basic code points.
336 for (const currentValue of input) {
337 if (currentValue < 0x80) {
338 output.push(stringFromCharCode(currentValue));
339 }
340 }
341 
342 const basicLength = output.length;
343 let handledCPCount = basicLength;
344 
345 // `handledCPCount` is the number of code points that have been handled;
346 // `basicLength` is the number of basic code points.
347 
348 // Finish the basic string with a delimiter unless it's empty.
349 if (basicLength) {
350 output.push(delimiter);
351 }
352 
353 // Main encoding loop:
354 while (handledCPCount < inputLength) {
355 // All non-basic code points < n have been handled already. Find the next
356 // larger one:
357 let m = maxInt;
358 for (const currentValue of input) {
359 if (currentValue >= n && currentValue < m) {
360 m = currentValue;
361 }
362 }
363 
364 // Increase `delta` enough to advance the decoder's <n,i> state to <m,0>,
365 // but guard against overflow.
366 const handledCPCountPlusOne = handledCPCount + 1;
367 if (m - n > floor((maxInt - delta) / handledCPCountPlusOne)) {
368 error('overflow');
369 }
370 
371 delta += (m - n) * handledCPCountPlusOne;
372 n = m;
373 
374 for (const currentValue of input) {
375 if (currentValue < n && ++delta > maxInt) {
376 error('overflow');
377 }
378 if (currentValue === n) {
379 // Represent delta as a generalized variable-length integer.
380 let q = delta;
381 for (let k = base /* no condition */; ; k += base) {
382 const t = k <= bias ? tMin : k >= bias + tMax ? tMax : k - bias;
383 if (q < t) {
384 break;
385 }
386 const qMinusT = q - t;
387 const baseMinusT = base - t;
388 output.push(
389 stringFromCharCode(digitToBasic(t + (qMinusT % baseMinusT), 0))
390 );
391 q = floor(qMinusT / baseMinusT);
392 }
393 
394 output.push(stringFromCharCode(digitToBasic(q, 0)));
395 bias = adapt(
396 delta,
397 handledCPCountPlusOne,
398 handledCPCount === basicLength
399 );
400 delta = 0;
401 ++handledCPCount;
402 }
403 }
404 
405 ++delta;
406 ++n;
407 }
408 return output.join('');
409};
410 
411/**
412 * Converts a Punycode string representing a domain name or an email address
413 * to Unicode. Only the Punycoded parts of the input will be converted, i.e.
414 * it doesn't matter if you call it on a string that has already been
415 * converted to Unicode.
416 * @memberOf punycode
417 * @param {String} input The Punycoded domain name or email address to
418 * convert to Unicode.
419 * @returns {String} The Unicode representation of the given Punycode
420 * string.
421 */
422export const toUnicode = function (input: string): string {
423 return mapDomain(input, function (string) {
424 return regexPunycode.test(string)
425 ? decode(string.slice(4).toLowerCase())
426 : string;
427 });
428};
429 
430/**
431 * Converts a Unicode string representing a domain name or an email address to
432 * Punycode. Only the non-ASCII parts of the domain name will be converted,
433 * i.e. it doesn't matter if you call it with a domain that's already in
434 * ASCII.
435 * @memberOf punycode
436 * @param {String} input The domain name or email address to convert, as a
437 * Unicode string.
438 * @returns {String} The Punycode representation of the given domain name or
439 * email address.
440 */
441export const toASCII = function (input: string): string {
442 return mapDomain(input, function (string) {
443 return regexNonASCII.test(string) ? 'xn--' + encode(string) : string;
444 });
445};
446 
447/*--------------------------------------------------------------------------*/
448 
449/** Define the public API */
450export const version = '2.1.0';
451 
452export const ucs2 = {
453 decode: ucs2decode,
454 encode: ucs2encode,
455};
456 
457export default {
458 /**
459 * A string representing the current Punycode.js version number.
460 * @memberOf punycode
461 * @type String
462 */
463 version,
464 /**
465 * An object of methods to convert from JavaScript's internal character
466 * representation (UCS-2) to Unicode code points, and back.
467 * @see <https://mathiasbynens.be/notes/javascript-encoding>
468 * @memberOf punycode
469 * @type Object
470 */
471 ucs2,
472 decode,
473 encode,
474 toASCII,
475 toUnicode,
476};