utf8.js 2.5 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465
  1. import { assertU8, fromUint8, E_STRING, E_STRICT_UNICODE } from './fallback/_utils.js'
  2. import { nativeDecoder, nativeEncoder } from './fallback/platform.js'
  3. import * as js from './fallback/utf8.auto.js'
  4. // ignoreBOM: true means that BOM will be left as-is, i.e. will be present in the output
  5. // We don't want to strip anything unexpectedly
  6. const decoderLoose = nativeDecoder
  7. const decoderFatal = nativeDecoder
  8. ? new TextDecoder('utf-8', { ignoreBOM: true, fatal: true })
  9. : null
  10. const { isWellFormed } = String.prototype
  11. function deLoose(str, loose, res) {
  12. if (loose || str.length === res.length) return res // length is equal only for ascii, which is automatically fine
  13. if (isWellFormed) {
  14. // We have a fast native method
  15. if (isWellFormed.call(str)) return res
  16. throw new TypeError(E_STRICT_UNICODE)
  17. }
  18. // Recheck if the string was encoded correctly
  19. let start = 0
  20. const last = res.length - 3
  21. // Search for EFBFBD (3-byte sequence)
  22. while (start <= last) {
  23. const pos = res.indexOf(0xef, start)
  24. if (pos === -1 || pos > last) break
  25. start = pos + 1
  26. if (res[pos + 1] === 0xbf && res[pos + 2] === 0xbd) {
  27. // Found a replacement char in output, need to recheck if we encoded the input correctly
  28. if (js.decodeFast && !nativeDecoder && str.length < 1e7) {
  29. // This is ~2x faster than decode in Hermes
  30. try {
  31. if (encodeURI(str) !== null) return res // guard against optimizing out
  32. } catch {}
  33. } else if (str === decode(res)) return res
  34. throw new TypeError(E_STRICT_UNICODE)
  35. }
  36. }
  37. return res
  38. }
  39. function encode(str, loose = false) {
  40. if (typeof str !== 'string') throw new TypeError(E_STRING)
  41. if (str.length === 0) return new Uint8Array() // faster than Uint8Array.of
  42. if (nativeEncoder || !js.encode) return deLoose(str, loose, nativeEncoder.encode(str))
  43. // No reason to use unescape + encodeURIComponent: it's slower than JS on normal engines, and modern Hermes already has TextEncoder
  44. return js.encode(str, loose)
  45. }
  46. function decode(arr, loose = false) {
  47. assertU8(arr)
  48. if (arr.byteLength === 0) return ''
  49. if (nativeDecoder || !js.decodeFast) {
  50. return loose ? decoderLoose.decode(arr) : decoderFatal.decode(arr) // Node.js and browsers
  51. }
  52. return js.decodeFast(arr, loose)
  53. }
  54. export const utf8fromString = (str, format = 'uint8') => fromUint8(encode(str, false), format)
  55. export const utf8fromStringLoose = (str, format = 'uint8') => fromUint8(encode(str, true), format)
  56. export const utf8toString = (arr) => decode(arr, false)
  57. export const utf8toStringLoose = (arr) => decode(arr, true)