Pudu programming language
Menu
Package

@chrismichaelps / pudu-lang-docgen

Documentation publishing for Pudu: articles, API references, navigation, search, and static websites

0.1.0Apache-2.01

InstallClose

Lexer.pudu

Pudu145 lines5.2 KB

GitHub ↗
1/** @Docgen.Api.Lexer — Pudu source reduced to tokens and documentation comments */2module PuduLangDocgen.Api.Lexer34import Std.Char as Char56/** @Docgen.Api.Token — one token with its kind, text, and one-based line */7export type Token = { kind: TokenKind, text: Str, line: Int }89/** @Docgen.Api.TokenKind — the lexical classes declarations are read from */10export type TokenKind = Word | Symbol | Literal | LineDoc | BlockDoc1112/// Two-character operators kept as one symbol.13const PAIRS: Array[Str] = ["->", "=>", "==", "!=", "<=", ">=", "&&", "||", "..", "::"]1415/// Tokens of a source file; ordinary comments and white space are dropped.16export fn tokens(source: Str) -> Array[Token] {17  let cs = source.chars()18  var result: Array[Token] = []19  var index = 020  var line = 121  while index < cs.length() {22    let character = cs[index]23    if character == '\n' {24      line = line + 125      index = index + 126    } else if Char.isWhitespace(character) {27      index = index + 128    } else if startsWith(&cs, index, "///") && !startsWith(&cs, index, "////") {29      var end = index + 330      while end < cs.length() && cs[end] != '\n' { end = end + 1 }31      let body = slice(&cs, index + 3, end)32      result = result.push(Token{kind: LineDoc, text: if body.startsWith(" ") { body.drop(1) } else { body }, line: line})33      index = end34    } else if startsWith(&cs, index, "//") {35      while index < cs.length() && cs[index] != '\n' { index = index + 1 }36    } else if startsWith(&cs, index, "/*") {37      let documented = startsWith(&cs, index, "/**") && !startsWith(&cs, index, "/**/")38      var end = index + 239      var depth = 140      let first = line41      while end < cs.length() && depth > 0 {42        if startsWith(&cs, end, "/*") {43          depth = depth + 144          end = end + 245        } else if startsWith(&cs, end, "*/") {46          depth = depth - 147          end = end + 248        } else {49          if cs[end] == '\n' { line = line + 1 }50          end = end + 151        }52      }53      if documented { result = result.push(Token{kind: BlockDoc, text: blockText(slice(&cs, index + 3, if depth == 0 { end - 2 } else { end })), line: first}) }54      index = end55    } else if character == '"' {56      let end = stringEnd(&cs, index)57      let text = slice(&cs, index, end)58      result = result.push(Token{kind: Literal, text: text, line: line})59      for held in text.chars() {60        if held == '\n' { line = line + 1 }61      }62      index = end63    } else if character == '\'' && charLiteral(&cs, index) > index {64      let end = charLiteral(&cs, index)65      result = result.push(Token{kind: Literal, text: slice(&cs, index, end), line: line})66      index = end67    } else if Char.isAlphanumeric(character) || character == '_' {68      var end = index + 169      while end < cs.length() && (Char.isAlphanumeric(cs[end]) || cs[end] == '_') { end = end + 1 }70      let text = slice(&cs, index, end)71      result = result.push(Token{kind: if Char.isDigit(character) { Literal } else { Word }, text: text, line: line})72      index = end73    } else {74      let pair = if index + 1 < cs.length() { slice(&cs, index, index + 2) } else { "" }75      if PAIRS.contains(pair) {76        result = result.push(Token{kind: Symbol, text: pair, line: line})77        index = index + 278      } else {79        result = result.push(Token{kind: Symbol, text: character.toText(), line: line})80        index = index + 181      }82    }83  }84  result85}8687/// Whether text continues with a prefix at an index.88fn startsWith(cs: &Array[Char], index: Int, prefix: Str) -> Bool {89  let wanted = prefix.chars()90  if index + wanted.length() > cs.length() { return false }91  for offset in 0..wanted.length() {92    if cs[index + offset] != wanted[offset] { return false }93  }94  true95}9697/// Text of the characters in a half-open range.98fn slice(cs: &Array[Char], from: Int, to: Int) -> Str {99  var pieces: Array[Str] = []100  for index in from..to { pieces = pieces.push(cs[index].toText()) }101  pieces.join("")102}103104/// The index after a string literal, following interpolations and the strings inside them.105fn stringEnd(cs: &Array[Char], index: Int) -> Int {106  var end = index + 1107  var depth = 0108  while end < cs.length() {109    let character = cs[end]110    if character == '\\' {111      end = end + 2112      continue113    }114    if depth == 0 && character == '"' { return end + 1 }115    if character == '{' { depth = depth + 1 }116    if character == '}' && depth > 0 { depth = depth - 1 }117    if character == '"' {118      end = stringEnd(cs, end)119      continue120    }121    end = end + 1122  }123  end124}125126/// The index after a character literal at an index, or the index itself when there is none.127fn charLiteral(cs: &Array[Char], index: Int) -> Int {128  if index + 2 < cs.length() && cs[index + 1] != '\\' && cs[index + 2] == '\'' { return index + 3 }129  if index + 1 < cs.length() && cs[index + 1] == '\\' {130    var end = index + 2131    while end < cs.length() && end - index < 12 && cs[end] != '\'' { end = end + 1 }132    if end < cs.length() && cs[end] == '\'' { return end + 1 }133  }134  index135}136137/// The text of a block documentation comment without its leading asterisks.138fn blockText(inner: Str) -> Str {139  let lines = inner.split("\n").map(fn(line: Str) -> Str {140      let trimmed = line.trim()141      if trimmed.startsWith("* ") { trimmed.drop(2) } else if trimmed == "*" { "" } else { trimmed }142    })143  lines.join("\n").trim()144}145