Pudu programming language
Menu
Package

@chrismichaelps / pudu-lang-environment

Environment files for Pudu: load .env variables safely with layered files, discovery, expansion, typed reads, and secrets kept out of every message

0.1.0Apache-2.01

InstallClose

Encoding.pudu

Pudu151 lines5.2 KB

GitHub ↗
1/** @Domain.Encoding.Module — bytes to text in a declared encoding */2module PuduLangEnvironment.Domain.Encoding34import Std.Bytes as Bytes5import Std.Option as Option6import Std.Result as Result78/** @Domain.Encoding.Encoding — how an environment file's bytes spell text */9export type Encoding10  = Utf811  | Ascii12  | Latin113  | Utf16Le14  | Utf16Be15  | Utf32Le16  | Utf32Be1718/// The first octet value ASCII does not define.19const ASCII_LIMIT: Int = 0x802021/// The first unit of a UTF-16 surrogate pair.22const HIGH_SURROGATE_FIRST: Int = 0xD8002324/// The last first unit of a UTF-16 surrogate pair.25const HIGH_SURROGATE_LAST: Int = 0xDBFF2627/// The first second unit of a UTF-16 surrogate pair.28const LOW_SURROGATE_FIRST: Int = 0xDC002930/// The last second unit of a UTF-16 surrogate pair.31const LOW_SURROGATE_LAST: Int = 0xDFFF3233/// The first code point a surrogate pair spells.34const SUPPLEMENTARY_FIRST: Int = 0x100003536/// How many code points one high surrogate spans.37const SURROGATE_SPAN: Int = 0x4003839/// The text the bytes spell. A UTF-8, UTF-32, or UTF-16 byte-order mark selects its encoding and40/// is dropped; otherwise `declared` applies. Bytes that do not spell text answer `None`.41export fn decode(data: &Bytes, declared: Encoding) -> Option[Str] {42  let octets = data.toArray().map(|octet: UInt8| Option.unwrapOr(convertInteger[Int](octet), 0))43  if startsWith(&octets, &[0xEF, 0xBB, 0xBF]) { return Result.ok(Bytes.toText(&data.drop(3))) }44  if startsWith(&octets, &[0xFF, 0xFE, 0x00, 0x00]) { return utf32(&octets.slice(4, octets.length()), true) }45  if startsWith(&octets, &[0x00, 0x00, 0xFE, 0xFF]) { return utf32(&octets.slice(4, octets.length()), false) }46  if startsWith(&octets, &[0xFF, 0xFE]) { return utf16(&octets.slice(2, octets.length()), true) }47  if startsWith(&octets, &[0xFE, 0xFF]) { return utf16(&octets.slice(2, octets.length()), false) }48  match declared {49    case Utf8 => Result.ok(Bytes.toText(data))50    case Ascii => ascii(&octets)51    case Latin1 => Some(latin1(&octets))52    case Utf16Le => utf16(&octets, true)53    case Utf16Be => utf16(&octets, false)54    case Utf32Le => utf32(&octets, true)55    case Utf32Be => utf32(&octets, false)56  }57}5859/// The encoding's conventional name.60export fn name(encoding: &Encoding) -> Str {61  match encoding {62    case Utf8 => "UTF-8"63    case Ascii => "ASCII"64    case Latin1 => "Latin-1"65    case Utf16Le => "UTF-16LE"66    case Utf16Be => "UTF-16BE"67    case Utf32Le => "UTF-32LE"68    case Utf32Be => "UTF-32BE"69  }70}7172/// Whether the octets begin with the mark.73fn startsWith(octets: &Array[Int], mark: &Array[Int]) -> Bool {74  octets.length() >= mark.length() && octets.slice(0, mark.length()) == *mark75}7677/// Each octet as the character with its code, when every octet is below the first code ASCII78/// leaves undefined.79fn ascii(octets: &Array[Int]) -> Option[Str] {80  for octet in octets {81    if octet >= ASCII_LIMIT { return None }82  }83  Some(latin1(octets))84}8586/// Each octet as the character with its code.87fn latin1(octets: &Array[Int]) -> Str {88  var pieces: Array[Str] = []89  for octet in octets {90    if let Some(symbol) = charFromCode(octet) { pieces = pieces.push(symbol.toText()) }91  }92  pieces.join("")93}9495/// The text of UTF-16 code units in little- or big-endian order, joining surrogate pairs; an odd96/// length or an unpaired surrogate answers `None`, a lone low surrogate because it is no character.97fn utf16(octets: &Array[Int], littleEndian: Bool) -> Option[Str] {98  if octets.length() % 2 == 1 { return None }99  let units = octets.length() / 2100  var pieces: Array[Str] = []101  var at = 0102  while at < units {103    let unit = unitAt(octets, at, littleEndian)104    var code = unit105    if unit >= HIGH_SURROGATE_FIRST && unit <= HIGH_SURROGATE_LAST {106      if at + 1 >= units { return None }107      let low = unitAt(octets, at + 1, littleEndian)108      if low < LOW_SURROGATE_FIRST || low > LOW_SURROGATE_LAST { return None }109      code = SUPPLEMENTARY_FIRST + (unit - HIGH_SURROGATE_FIRST) * SURROGATE_SPAN + (low - LOW_SURROGATE_FIRST)110      at = at + 1111    }112    let symbol = charFromCode(code) ?113    pieces = pieces.push(symbol.toText())114    at = at + 1115  }116  Some(pieces.join(""))117}118119/// The text of UTF-32 code units in little- or big-endian order; a length that is not a multiple120/// of four, or a unit that is no character, answers `None`.121fn utf32(octets: &Array[Int], littleEndian: Bool) -> Option[Str] {122  if octets.length() % 4 != 0 { return None }123  var pieces: Array[Str] = []124  var at = 0125  while at < octets.length() {126    let symbol = charFromCode(codeAt(octets, at, littleEndian)) ?127    pieces = pieces.push(symbol.toText())128    at = at + 4129  }130  Some(pieces.join(""))131}132133/// The UTF-32 code at an octet offset, from its four octets in the stated order.134fn codeAt(octets: &Array[Int], offset: Int, littleEndian: Bool) -> Int {135  var code = 0136  var step = 0137  while step < 4 {138    let octet = if littleEndian { octets[offset + 3 - step] } else { octets[offset + step] }139    code = code * 256 + octet140    step = step + 1141  }142  code143}144145/// The code unit at a position, from its two octets in the stated order.146fn unitAt(octets: &Array[Int], position: Int, littleEndian: Bool) -> Int {147  let first = octets[position * 2]148  let second = octets[position * 2 + 1]149  if littleEndian { second * 256 + first } else { first * 256 + second }150}151