File
Blob: src/node/punycode.ts
| 1 | // Copyright (c) 2017-2022 Cloudflare, Inc. |
| 2 | // Licensed under the Apache 2.0 license found in the LICENSE file or at: |
| 3 | // https://opensource.org/licenses/Apache-2.0 |
| 4 | // Copyright Mathias Bynens <https://mathiasbynens.be/> |
| 5 | // |
| 6 | // Permission is hereby granted, free of charge, to any person obtaining |
| 7 | // a copy of this software and associated documentation files (the |
| 8 | // "Software"), to deal in the Software without restriction, including |
| 9 | // without limitation the rights to use, copy, modify, merge, publish, |
| 10 | // distribute, sublicense, and/or sell copies of the Software, and to |
| 11 | // permit persons to whom the Software is furnished to do so, subject to |
| 12 | // the following conditions: |
| 13 | // |
| 14 | // The above copyright notice and this permission notice shall be |
| 15 | // included in all copies or substantial portions of the Software. |
| 16 | // |
| 17 | // THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, |
| 18 | // EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF |
| 19 | // MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND |
| 20 | // NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE |
| 21 | // LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION |
| 22 | // OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION |
| 23 | // WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. |
| 24 | |
| 25 | /** Highest positive signed 32-bit float value */ |
| 26 | const maxInt = 2147483647; // aka. 0x7FFFFFFF or 2^31-1 |
| 27 | |
| 28 | /** Bootstring parameters */ |
| 29 | const base = 36; |
| 30 | const tMin = 1; |
| 31 | const tMax = 26; |
| 32 | const skew = 38; |
| 33 | const damp = 700; |
| 34 | const initialBias = 72; |
| 35 | const initialN = 128; // 0x80 |
| 36 | const delimiter = '-'; // '\x2D' |
| 37 | |
| 38 | /** Regular expressions */ |
| 39 | const regexPunycode = /^xn--/; |
| 40 | const regexNonASCII = /[^\0-\x7F]/; // Note: U+007F DEL is excluded too. |
| 41 | const regexSeparators = /[\x2E\u3002\uFF0E\uFF61]/g; // RFC 3490 separators |
| 42 | |
| 43 | /** Error messages */ |
| 44 | const errors = { |
| 45 | overflow: 'Overflow: input needs wider integers to process', |
| 46 | 'not-basic': 'Illegal input >= 0x80 (not a basic code point)', |
| 47 | 'invalid-input': 'Invalid input', |
| 48 | }; |
| 49 | |
| 50 | /** Convenience shortcuts */ |
| 51 | const baseMinusTMin = base - tMin; |
| 52 | const floor = Math.floor; |
| 53 | const stringFromCharCode = String.fromCharCode; |
| 54 | |
| 55 | /*--------------------------------------------------------------------------*/ |
| 56 | |
| 57 | /** |
| 58 | * A generic error utility function. |
| 59 | * @private |
| 60 | * @param {String} type The error type. |
| 61 | * @returns {Error} Throws a `RangeError` with the applicable error message. |
| 62 | */ |
| 63 | function error(type: keyof typeof errors): void { |
| 64 | throw new RangeError(errors[type]); |
| 65 | } |
| 66 | |
| 67 | /** |
| 68 | * A generic `Array#map` utility function. |
| 69 | * @private |
| 70 | * @param {Array} array The array to iterate over. |
| 71 | * @param {Function} callback The function that gets called for every array |
| 72 | * item. |
| 73 | * @returns {Array} A new array of values returned by the callback function. |
| 74 | */ |
| 75 | function map<T>(array: T[], callback: (item: T) => T): T[] { |
| 76 | const result = []; |
| 77 | let length = array.length; |
| 78 | while (length--) { |
| 79 | result[length] = callback(array[length] as T); |
| 80 | } |
| 81 | return result; |
| 82 | } |
| 83 | |
| 84 | /** |
| 85 | * A simple `Array#map`-like wrapper to work with domain name strings or email |
| 86 | * addresses. |
| 87 | * @private |
| 88 | * @param {String} domain The domain name or email address. |
| 89 | * @param {Function} callback The function that gets called for every |
| 90 | * character. |
| 91 | * @returns {String} A new string of characters returned by the callback |
| 92 | * function. |
| 93 | */ |
| 94 | function mapDomain(domain: string, callback: (char: string) => string): string { |
| 95 | const parts = domain.split('@'); |
| 96 | let result = ''; |
| 97 | if (parts.length > 1) { |
| 98 | // In email addresses, only the domain name should be punycoded. Leave |
| 99 | // the local part (i.e. everything up to `@`) intact. |
| 100 | result = parts[0] + '@'; // eslint-disable-line @typescript-eslint/restrict-plus-operands |
| 101 | domain = parts[1] as string; |
| 102 | } |
| 103 | // Avoid `split(regex)` for IE8 compatibility. See #17. |
| 104 | domain = domain.replace(regexSeparators, '\x2E'); |
| 105 | const labels = domain.split('.'); |
| 106 | const encoded = map(labels, callback).join('.'); |
| 107 | return result + encoded; |
| 108 | } |
| 109 | |
| 110 | /** |
| 111 | * Creates an array containing the numeric code points of each Unicode |
| 112 | * character in the string. While JavaScript uses UCS-2 internally, |
| 113 | * this function will convert a pair of surrogate halves (each of which |
| 114 | * UCS-2 exposes as separate characters) into a single code point, |
| 115 | * matching UTF-16. |
| 116 | * @see `punycode.ucs2.encode` |
| 117 | * @see <https://mathiasbynens.be/notes/javascript-encoding> |
| 118 | * @memberOf punycode.ucs2 |
| 119 | * @name decode |
| 120 | * @param {String} string The Unicode input string (UCS-2). |
| 121 | * @returns {Array} The new array of code points. |
| 122 | */ |
| 123 | function ucs2decode(string: string): number[] { |
| 124 | const output = []; |
| 125 | let counter = 0; |
| 126 | const length = string.length; |
| 127 | while (counter < length) { |
| 128 | const value = string.charCodeAt(counter++); |
| 129 | if (value >= 0xd800 && value <= 0xdbff && counter < length) { |
| 130 | // It's a high surrogate, and there is a next character. |
| 131 | const extra = string.charCodeAt(counter++); |
| 132 | if ((extra & 0xfc00) == 0xdc00) { |
| 133 | // Low surrogate. |
| 134 | output.push(((value & 0x3ff) << 10) + (extra & 0x3ff) + 0x10000); |
| 135 | } else { |
| 136 | // It's an unmatched surrogate; only append this code unit, in case the |
| 137 | // next code unit is the high surrogate of a surrogate pair. |
| 138 | output.push(value); |
| 139 | counter--; |
| 140 | } |
| 141 | } else { |
| 142 | output.push(value); |
| 143 | } |
| 144 | } |
| 145 | return output; |
| 146 | } |
| 147 | |
| 148 | /** |
| 149 | * Creates a string based on an array of numeric code points. |
| 150 | * @see `punycode.ucs2.decode` |
| 151 | * @memberOf punycode.ucs2 |
| 152 | * @name encode |
| 153 | * @param {Array} codePoints The array of numeric code points. |
| 154 | * @returns {String} The new Unicode string (UCS-2). |
| 155 | */ |
| 156 | const ucs2encode = (codePoints: number[]): string => |
| 157 | String.fromCodePoint(...codePoints); |
| 158 | |
| 159 | /** |
| 160 | * Converts a basic code point into a digit/integer. |
| 161 | * @see `digitToBasic()` |
| 162 | * @private |
| 163 | * @param {Number} codePoint The basic numeric code point value. |
| 164 | * @returns {Number} The numeric value of a basic code point (for use in |
| 165 | * representing integers) in the range `0` to `base - 1`, or `base` if |
| 166 | * the code point does not represent a value. |
| 167 | */ |
| 168 | const basicToDigit = function (codePoint: number): number { |
| 169 | if (codePoint >= 0x30 && codePoint < 0x3a) { |
| 170 | return 26 + (codePoint - 0x30); |
| 171 | } |
| 172 | if (codePoint >= 0x41 && codePoint < 0x5b) { |
| 173 | return codePoint - 0x41; |
| 174 | } |
| 175 | if (codePoint >= 0x61 && codePoint < 0x7b) { |
| 176 | return codePoint - 0x61; |
| 177 | } |
| 178 | return base; |
| 179 | }; |
| 180 | |
| 181 | /** |
| 182 | * Converts a digit/integer into a basic code point. |
| 183 | * @see `basicToDigit()` |
| 184 | * @private |
| 185 | * @param {Number} digit The numeric value of a basic code point. |
| 186 | * @returns {Number} The basic code point whose value (when used for |
| 187 | * representing integers) is `digit`, which needs to be in the range |
| 188 | * `0` to `base - 1`. If `flag` is non-zero, the uppercase form is |
| 189 | * used; else, the lowercase form is used. The behavior is undefined |
| 190 | * if `flag` is non-zero and `digit` has no uppercase form. |
| 191 | */ |
| 192 | const digitToBasic = function (digit: number, flag: number): number { |
| 193 | // 0..25 map to ASCII a..z or A..Z |
| 194 | // 26..35 map to ASCII 0..9 |
| 195 | return digit + 22 + 75 * Number(digit < 26) - (Number(flag != 0) << 5); |
| 196 | }; |
| 197 | |
| 198 | /** |
| 199 | * Bias adaptation function as per section 3.4 of RFC 3492. |
| 200 | * https://tools.ietf.org/html/rfc3492#section-3.4 |
| 201 | * @private |
| 202 | */ |
| 203 | const adapt = function ( |
| 204 | delta: number, |
| 205 | numPoints: number, |
| 206 | firstTime: boolean |
| 207 | ): number { |
| 208 | let k = 0; |
| 209 | delta = firstTime ? floor(delta / damp) : delta >> 1; |
| 210 | delta += floor(delta / numPoints); |
| 211 | for ( |
| 212 | ; |
| 213 | /* no initialization */ delta > (baseMinusTMin * tMax) >> 1; |
| 214 | k += base |
| 215 | ) { |
| 216 | delta = floor(delta / baseMinusTMin); |
| 217 | } |
| 218 | return floor(k + ((baseMinusTMin + 1) * delta) / (delta + skew)); |
| 219 | }; |
| 220 | |
| 221 | /** |
| 222 | * Converts a Punycode string of ASCII-only symbols to a string of Unicode |
| 223 | * symbols. |
| 224 | * @memberOf punycode |
| 225 | * @param {String} input The Punycode string of ASCII-only symbols. |
| 226 | * @returns {String} The resulting string of Unicode symbols. |
| 227 | */ |
| 228 | export const decode = function (input: string): string { |
| 229 | // Don't use UCS-2. |
| 230 | const output = []; |
| 231 | const inputLength = input.length; |
| 232 | let i = 0; |
| 233 | let n = initialN; |
| 234 | let bias = initialBias; |
| 235 | |
| 236 | // Handle the basic code points: let `basic` be the number of input code |
| 237 | // points before the last delimiter, or `0` if there is none, then copy |
| 238 | // the first basic code points to the output. |
| 239 | |
| 240 | let basic = input.lastIndexOf(delimiter); |
| 241 | if (basic < 0) { |
| 242 | basic = 0; |
| 243 | } |
| 244 | |
| 245 | for (let j = 0; j < basic; ++j) { |
| 246 | // if it's not a basic code point |
| 247 | if (input.charCodeAt(j) >= 0x80) { |
| 248 | error('not-basic'); |
| 249 | } |
| 250 | output.push(input.charCodeAt(j)); |
| 251 | } |
| 252 | |
| 253 | // Main decoding loop: start just after the last delimiter if any basic code |
| 254 | // points were copied; start at the beginning otherwise. |
| 255 | |
| 256 | for ( |
| 257 | let index = basic > 0 ? basic + 1 : 0; |
| 258 | index < inputLength /* no final expression */; |
| 259 | ) { |
| 260 | // `index` is the index of the next character to be consumed. |
| 261 | // Decode a generalized variable-length integer into `delta`, |
| 262 | // which gets added to `i`. The overflow checking is easier |
| 263 | // if we increase `i` as we go, then subtract off its starting |
| 264 | // value at the end to obtain `delta`. |
| 265 | const oldi = i; |
| 266 | for (let w = 1, k = base /* no condition */; ; k += base) { |
| 267 | if (index >= inputLength) { |
| 268 | error('invalid-input'); |
| 269 | } |
| 270 | |
| 271 | const digit = basicToDigit(input.charCodeAt(index++)); |
| 272 | |
| 273 | if (digit >= base) { |
| 274 | error('invalid-input'); |
| 275 | } |
| 276 | if (digit > floor((maxInt - i) / w)) { |
| 277 | error('overflow'); |
| 278 | } |
| 279 | |
| 280 | i += digit * w; |
| 281 | const t = k <= bias ? tMin : k >= bias + tMax ? tMax : k - bias; |
| 282 | |
| 283 | if (digit < t) { |
| 284 | break; |
| 285 | } |
| 286 | |
| 287 | const baseMinusT = base - t; |
| 288 | if (w > floor(maxInt / baseMinusT)) { |
| 289 | error('overflow'); |
| 290 | } |
| 291 | |
| 292 | w *= baseMinusT; |
| 293 | } |
| 294 | |
| 295 | const out = output.length + 1; |
| 296 | bias = adapt(i - oldi, out, oldi == 0); |
| 297 | |
| 298 | // `i` was supposed to wrap around from `out` to `0`, |
| 299 | // incrementing `n` each time, so we'll fix that now: |
| 300 | if (floor(i / out) > maxInt - n) { |
| 301 | error('overflow'); |
| 302 | } |
| 303 | |
| 304 | n += floor(i / out); |
| 305 | i %= out; |
| 306 | |
| 307 | // Insert `n` at position `i` of the output. |
| 308 | output.splice(i++, 0, n); |
| 309 | } |
| 310 | |
| 311 | return String.fromCodePoint(...output); |
| 312 | }; |
| 313 | |
| 314 | /** |
| 315 | * Converts a string of Unicode symbols (e.g. a domain name label) to a |
| 316 | * Punycode string of ASCII-only symbols. |
| 317 | * @memberOf punycode |
| 318 | * @param {String} input The string of Unicode symbols. |
| 319 | * @returns {String} The resulting Punycode string of ASCII-only symbols. |
| 320 | */ |
| 321 | export const encode = function (input: string | number[]): string { |
| 322 | const output = []; |
| 323 | |
| 324 | // Convert the input in UCS-2 to an array of Unicode code points. |
| 325 | input = ucs2decode(input as string); |
| 326 | |
| 327 | // Cache the length. |
| 328 | const inputLength = input.length; |
| 329 | |
| 330 | // Initialize the state. |
| 331 | let n = initialN; |
| 332 | let delta = 0; |
| 333 | let bias = initialBias; |
| 334 | |
| 335 | // Handle the basic code points. |
| 336 | for (const currentValue of input) { |
| 337 | if (currentValue < 0x80) { |
| 338 | output.push(stringFromCharCode(currentValue)); |
| 339 | } |
| 340 | } |
| 341 | |
| 342 | const basicLength = output.length; |
| 343 | let handledCPCount = basicLength; |
| 344 | |
| 345 | // `handledCPCount` is the number of code points that have been handled; |
| 346 | // `basicLength` is the number of basic code points. |
| 347 | |
| 348 | // Finish the basic string with a delimiter unless it's empty. |
| 349 | if (basicLength) { |
| 350 | output.push(delimiter); |
| 351 | } |
| 352 | |
| 353 | // Main encoding loop: |
| 354 | while (handledCPCount < inputLength) { |
| 355 | // All non-basic code points < n have been handled already. Find the next |
| 356 | // larger one: |
| 357 | let m = maxInt; |
| 358 | for (const currentValue of input) { |
| 359 | if (currentValue >= n && currentValue < m) { |
| 360 | m = currentValue; |
| 361 | } |
| 362 | } |
| 363 | |
| 364 | // Increase `delta` enough to advance the decoder's <n,i> state to <m,0>, |
| 365 | // but guard against overflow. |
| 366 | const handledCPCountPlusOne = handledCPCount + 1; |
| 367 | if (m - n > floor((maxInt - delta) / handledCPCountPlusOne)) { |
| 368 | error('overflow'); |
| 369 | } |
| 370 | |
| 371 | delta += (m - n) * handledCPCountPlusOne; |
| 372 | n = m; |
| 373 | |
| 374 | for (const currentValue of input) { |
| 375 | if (currentValue < n && ++delta > maxInt) { |
| 376 | error('overflow'); |
| 377 | } |
| 378 | if (currentValue === n) { |
| 379 | // Represent delta as a generalized variable-length integer. |
| 380 | let q = delta; |
| 381 | for (let k = base /* no condition */; ; k += base) { |
| 382 | const t = k <= bias ? tMin : k >= bias + tMax ? tMax : k - bias; |
| 383 | if (q < t) { |
| 384 | break; |
| 385 | } |
| 386 | const qMinusT = q - t; |
| 387 | const baseMinusT = base - t; |
| 388 | output.push( |
| 389 | stringFromCharCode(digitToBasic(t + (qMinusT % baseMinusT), 0)) |
| 390 | ); |
| 391 | q = floor(qMinusT / baseMinusT); |
| 392 | } |
| 393 | |
| 394 | output.push(stringFromCharCode(digitToBasic(q, 0))); |
| 395 | bias = adapt( |
| 396 | delta, |
| 397 | handledCPCountPlusOne, |
| 398 | handledCPCount === basicLength |
| 399 | ); |
| 400 | delta = 0; |
| 401 | ++handledCPCount; |
| 402 | } |
| 403 | } |
| 404 | |
| 405 | ++delta; |
| 406 | ++n; |
| 407 | } |
| 408 | return output.join(''); |
| 409 | }; |
| 410 | |
| 411 | /** |
| 412 | * Converts a Punycode string representing a domain name or an email address |
| 413 | * to Unicode. Only the Punycoded parts of the input will be converted, i.e. |
| 414 | * it doesn't matter if you call it on a string that has already been |
| 415 | * converted to Unicode. |
| 416 | * @memberOf punycode |
| 417 | * @param {String} input The Punycoded domain name or email address to |
| 418 | * convert to Unicode. |
| 419 | * @returns {String} The Unicode representation of the given Punycode |
| 420 | * string. |
| 421 | */ |
| 422 | export const toUnicode = function (input: string): string { |
| 423 | return mapDomain(input, function (string) { |
| 424 | return regexPunycode.test(string) |
| 425 | ? decode(string.slice(4).toLowerCase()) |
| 426 | : string; |
| 427 | }); |
| 428 | }; |
| 429 | |
| 430 | /** |
| 431 | * Converts a Unicode string representing a domain name or an email address to |
| 432 | * Punycode. Only the non-ASCII parts of the domain name will be converted, |
| 433 | * i.e. it doesn't matter if you call it with a domain that's already in |
| 434 | * ASCII. |
| 435 | * @memberOf punycode |
| 436 | * @param {String} input The domain name or email address to convert, as a |
| 437 | * Unicode string. |
| 438 | * @returns {String} The Punycode representation of the given domain name or |
| 439 | * email address. |
| 440 | */ |
| 441 | export const toASCII = function (input: string): string { |
| 442 | return mapDomain(input, function (string) { |
| 443 | return regexNonASCII.test(string) ? 'xn--' + encode(string) : string; |
| 444 | }); |
| 445 | }; |
| 446 | |
| 447 | /*--------------------------------------------------------------------------*/ |
| 448 | |
| 449 | /** Define the public API */ |
| 450 | export const version = '2.1.0'; |
| 451 | |
| 452 | export const ucs2 = { |
| 453 | decode: ucs2decode, |
| 454 | encode: ucs2encode, |
| 455 | }; |
| 456 | |
| 457 | export default { |
| 458 | /** |
| 459 | * A string representing the current Punycode.js version number. |
| 460 | * @memberOf punycode |
| 461 | * @type String |
| 462 | */ |
| 463 | version, |
| 464 | /** |
| 465 | * An object of methods to convert from JavaScript's internal character |
| 466 | * representation (UCS-2) to Unicode code points, and back. |
| 467 | * @see <https://mathiasbynens.be/notes/javascript-encoding> |
| 468 | * @memberOf punycode |
| 469 | * @type Object |
| 470 | */ |
| 471 | ucs2, |
| 472 | decode, |
| 473 | encode, |
| 474 | toASCII, |
| 475 | toUnicode, |
| 476 | }; |