# lex.ludic — source text -> a token slice. Tokens carry their kind, their # text (identifiers, strings, operators), an integer value (numbers, char # literals) and a line for diagnostics. const TK_ID: int = 0 const TK_INT: int = 1 const TK_STR: int = 2 const TK_OP: int = 3 const TK_NL: int = 4 const TK_EOF: int = 5 const TK_FLOAT: int = 6 const TK_INTERP: int = 7 # `text {expr} text` — raw content, split by the parser property Tok { kind: int = 0, text: pointer = null, ival: int = 0, line: int = 0 } var toks: []Tok function tok_push(kind: int, text: pointer, ival: int, line: int) -> void { let t = new Tok t.kind = kind; t.text = text; t.ival = ival; t.line = line push(toks, t) } # does src match the 2-char operator op at position i? function two_at(src: pointer, i: int, a: int, b: int) -> bool { return src[i] == a and src[i + 1] == b } function is_op1(c: int) -> bool { # + - * / % < > = ( ) { } [ ] , : . ! @ if c == '+' or c == '-' or c == '*' or c == '/' or c == '%' { return true } if c == '<' or c == '>' or c == '=' { return true } if c == '(' or c == ')' or c == '{' or c == '}' { return true } if c == '[' or c == ']' or c == ',' or c == ':' or c == '.' { return true } if c == '!' or c == '@' { return true } if c == '&' or c == '|' or c == '^' or c == '~' { return true } # & | ^ ~ return false } # The two bytes that delimit escapes, spelled numerically on purpose: this file # is what teaches the compiler to read `'\''` and `'\\'`, and the seed that # bootstraps it must lex it without already knowing those spellings. const CH_SQUOTE: int = 39 # ' const CH_BACKSLASH: int = 92 # \ # the byte an escape sequence `\e` stands for, shared by "strings" and 'chars': # \n \r \t \0 are the named ones; anything else (\\ \' \" \`) is itself. function unescape(e: int) -> int { if e == 'n' { return 10 } if e == 'r' { return 13 } if e == 't' { return 9 } if e == '0' { return 0 } return e } # a character the language has no use for is an error, never silently dropped: # `x = 5 $ 3` must not compile as `x = 5 3`. function lex_error(line: int, c: int) -> void { let ch = bytes(2); ch[0] = c; ch[1] = 0 lex_fail(line, `unexpected character '{ch}' (byte {c})`) } # the closing backtick of the template literal that opens at s[i]: a hole's `{...}` is code, so a # string, a char or another template inside it is skipped whole - `a {f(`b {c}`)} d` is one literal function tmpl_end(s: pointer, i0: int, n: int) -> int { var i = i0 + 1 while i < n and s[i] != '`' { if s[i] == CH_BACKSLASH { i += 2; continue } if s[i] == '{' and i + 1 < n and s[i + 1] == '{' { i += 2; continue } if s[i] == '{' { i = hole_end(s, i + 1, n) } i += 1 } return i } # the `}` that closes a hole whose code starts at s[i]; n when there is none function hole_end(s: pointer, i0: int, n: int) -> int { var i = i0 var depth = 1 while i < n { let d = s[i] if d == '"' or d == CH_SQUOTE { i += 1 while i < n and s[i] != d { if s[i] == CH_BACKSLASH { i += 1 }; i += 1 } } else if d == '`' { i = tmpl_end(s, i, n) } else if d == '{' { depth += 1 } else if d == '}' { depth -= 1 if depth == 0 { return i } } i += 1 } return n } # report a lexical error in the standard `file:line: error: msg` shape and stop function lex_fail(line: int, msg: pointer) -> void { let m = `{g_parse_file}:{line}: error: {msg}\n` file_write(file_stderr(), m, len(m)) exit(1) } function lex(src: pointer) -> void { lex_at(src, 1) } # lex `src` with its first line numbered `first_line` (an interpolation hole is # re-lexed on its own, and keeps the line of the string it sits in) function lex_at(src: pointer, first_line: int) -> void { toks = new []Tok var i = 0 var line = first_line let n = len(src) while i < n { let c = src[i] if c == '\n' { tok_push(TK_NL, null, 0, line); line += 1; i += 1; continue } if c == ' ' or c == '\t' or c == '\r' { i += 1; continue } if c == '#' { # '#' comment to end of line while i < n and src[i] != '\n' { i += 1 } continue } if c == '"' { # "string" i += 1 let start = i let out = bytes(n) var j = 0 while i < n and src[i] != '"' { if src[i] == CH_BACKSLASH { # backslash escape out[j] = unescape(src[i + 1]); j += 1; i += 2 } else { out[j] = src[i]; j += 1; i += 1 } } i += 1 out[j] = 0 tok_push(TK_STR, out, 0, line) continue } if c == '`' { # `interpolated string` — captured raw let e = tmpl_end(src, i, n) let out = bytes(n) var j = 0 i += 1 while i < e { out[j] = src[i] j += 1 i += 1 } i = e + 1 out[j] = 0 tok_push(TK_INTERP, out, 0, line) continue } if c == CH_SQUOTE { # 'c' char literal -> int i += 1 var v = 0 if src[i] == CH_BACKSLASH { v = unescape(src[i + 1]); i += 2 } else { v = src[i]; i += 1 } if src[i] != CH_SQUOTE { lex_fail(line, "unterminated character literal (expected closing ')") } i += 1 tok_push(TK_INT, null, v, line) continue } if char_is_digit(c) { if c == '0' and src[i + 1] == 'x' { # 0x hex var v = 0 var lv: long = 0 var digits = 0 i += 2 while i < n { let h = src[i] var d = 0 if char_is_digit(h) { d = h - 48 } else { if h >= 'a' and h <= 'f' { d = h - 87 } else { if h >= 'A' and h <= 'F' { d = h - 55 } else { break } } } v = v * 16 + d lv = lv * long(16) + long(d) if digits > 0 or d > 0 { digits += 1 } i += 1 } # up to 8 hex digits is a 32-bit pattern (0xFFFFFFFF is -1, as it always was); more is a long var big: pointer = null if digits > 8 { big = string(lv) } tok_push(TK_INT, big, v, line) continue } var v = 0 var lv: long = 0 let nstart = i while i < n and char_is_digit(src[i]) { v = v * 10 + (src[i] - 48) lv = lv * long(10) + long(src[i] - 48) i += 1 } # a fractional part makes it a Q16.16 fixed literal (its text is kept: in a # float context the literal is exactly that decimal instead) if i < n and src[i] == '.' and char_is_digit(src[i + 1]) { i += 1 # Accumulate only the first 4 fractional digits: `fnum << 16` must stay # in i32 (5+ digits overflow), and Q16.16 resolves ~4-5 decimals anyway. # Extra digits are still consumed so they don't become a stray token. var fnum = 0; var fden = 1 while i < n and char_is_digit(src[i]) { if fden < 10000 { fnum = fnum * 10 + (src[i] - 48); fden *= 10 } i += 1 } let bits = (v << 16) + ((fnum << 16) + (fden >> 1)) / fden tok_push(TK_FLOAT, src[nstart..i], bits, line) continue } # a decimal past 2^31 - 1 does not fit an int: it is a long, and keeps its value as text var big: pointer = null if lv > long(2147483647) { big = string(lv) } tok_push(TK_INT, big, v, line) continue } if char_is_alpha(c) { let start = i while i < n and char_is_alnum(src[i]) { i += 1 } tok_push(TK_ID, src[start..i], 0, line) continue } if c == ';' { tok_push(TK_NL, null, 0, line); i += 1; continue } # ';' # two-character operators if two_at(src, i, 45, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # -> if two_at(src, i, 61, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # => if two_at(src, i, 61, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # == if two_at(src, i, 33, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # != if two_at(src, i, 60, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # <= if two_at(src, i, 62, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # >= if two_at(src, i, 43, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # += if two_at(src, i, 45, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # -= if two_at(src, i, 42, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # *= if two_at(src, i, 47, 61) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # /= if two_at(src, i, 46, 46) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # .. if two_at(src, i, 60, 60) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # << if two_at(src, i, 62, 62) { tok_push(TK_OP, src[i..i + 2], 0, line); i += 2; continue } # >> if is_op1(c) { tok_push(TK_OP, src[i..i + 1], 0, line); i += 1; continue } lex_error(line, c) } tok_push(TK_EOF, null, 0, line) }