init
This commit is contained in:
commit
8f4f858d86
30 changed files with 9765 additions and 0 deletions
167
internal/lang/lexer.go
Normal file
167
internal/lang/lexer.go
Normal file
|
|
@ -0,0 +1,167 @@
|
|||
package lang
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"unicode"
|
||||
)
|
||||
|
||||
type lexer struct {
|
||||
src []rune
|
||||
pos int
|
||||
}
|
||||
|
||||
func lex(input string) ([]token, error) {
|
||||
l := &lexer{src: []rune(input)}
|
||||
var tokens []token
|
||||
for {
|
||||
tok, err := l.next()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
tokens = append(tokens, tok)
|
||||
if tok.kind == tokenEOF {
|
||||
return tokens, nil
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (l *lexer) next() (token, error) {
|
||||
l.skipWhitespace()
|
||||
start := l.pos
|
||||
if l.pos >= len(l.src) {
|
||||
return token{kind: tokenEOF, pos: start}, nil
|
||||
}
|
||||
|
||||
ch := l.src[l.pos]
|
||||
switch {
|
||||
case isIdentStart(ch):
|
||||
l.pos++
|
||||
for l.pos < len(l.src) && isIdentPart(l.src[l.pos]) {
|
||||
l.pos++
|
||||
}
|
||||
lexeme := string(l.src[start:l.pos])
|
||||
if kind, ok := keywords[lexeme]; ok {
|
||||
return token{kind: kind, lexeme: lexeme, pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenIdent, lexeme: lexeme, pos: start}, nil
|
||||
case unicode.IsDigit(ch):
|
||||
l.pos++
|
||||
for l.pos < len(l.src) && unicode.IsDigit(l.src[l.pos]) {
|
||||
l.pos++
|
||||
}
|
||||
return token{kind: tokenInt, lexeme: string(l.src[start:l.pos]), pos: start}, nil
|
||||
case ch == '"':
|
||||
l.pos++
|
||||
for l.pos < len(l.src) && l.src[l.pos] != '"' {
|
||||
if l.src[l.pos] == '\\' {
|
||||
l.pos++
|
||||
if l.pos >= len(l.src) {
|
||||
return token{}, fmt.Errorf("unterminated string at %d", start)
|
||||
}
|
||||
}
|
||||
l.pos++
|
||||
}
|
||||
if l.pos >= len(l.src) {
|
||||
return token{}, fmt.Errorf("unterminated string at %d", start)
|
||||
}
|
||||
l.pos++
|
||||
return token{kind: tokenString, lexeme: string(l.src[start:l.pos]), pos: start}, nil
|
||||
default:
|
||||
l.pos++
|
||||
switch ch {
|
||||
case '(':
|
||||
return token{kind: tokenLParen, lexeme: "(", pos: start}, nil
|
||||
case ')':
|
||||
return token{kind: tokenRParen, lexeme: ")", pos: start}, nil
|
||||
case '{':
|
||||
return token{kind: tokenLBrace, lexeme: "{", pos: start}, nil
|
||||
case '}':
|
||||
return token{kind: tokenRBrace, lexeme: "}", pos: start}, nil
|
||||
case ',':
|
||||
return token{kind: tokenComma, lexeme: ",", pos: start}, nil
|
||||
case '.':
|
||||
return token{kind: tokenDot, lexeme: ".", pos: start}, nil
|
||||
case ':':
|
||||
return token{kind: tokenColon, lexeme: ":", pos: start}, nil
|
||||
case ';':
|
||||
return token{kind: tokenSemicolon, lexeme: ";", pos: start}, nil
|
||||
case '+':
|
||||
if l.match('=') {
|
||||
return token{kind: tokenPlusAssign, lexeme: "+=", pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenPlus, lexeme: "+", pos: start}, nil
|
||||
case '-':
|
||||
if l.match('>') {
|
||||
return token{kind: tokenArrow, lexeme: "->", pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenMinus, lexeme: "-", pos: start}, nil
|
||||
case '*':
|
||||
return token{kind: tokenStar, lexeme: "*", pos: start}, nil
|
||||
case '/':
|
||||
if l.match('/') {
|
||||
for l.pos < len(l.src) && l.src[l.pos] != '\n' {
|
||||
l.pos++
|
||||
}
|
||||
return l.next()
|
||||
}
|
||||
return token{kind: tokenSlash, lexeme: "/", pos: start}, nil
|
||||
case '%':
|
||||
return token{kind: tokenPercent, lexeme: "%", pos: start}, nil
|
||||
case '!':
|
||||
if l.match('=') {
|
||||
return token{kind: tokenNeq, lexeme: "!=", pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenBang, lexeme: "!", pos: start}, nil
|
||||
case '=':
|
||||
if l.match('=') {
|
||||
return token{kind: tokenEq, lexeme: "==", pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenAssign, lexeme: "=", pos: start}, nil
|
||||
case '<':
|
||||
if l.match('=') {
|
||||
return token{kind: tokenLte, lexeme: "<=", pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenLt, lexeme: "<", pos: start}, nil
|
||||
case '>':
|
||||
if l.match('=') {
|
||||
return token{kind: tokenGte, lexeme: ">=", pos: start}, nil
|
||||
}
|
||||
return token{kind: tokenGt, lexeme: ">", pos: start}, nil
|
||||
case '&':
|
||||
if l.match('&') {
|
||||
return token{kind: tokenAnd, lexeme: "&&", pos: start}, nil
|
||||
}
|
||||
case '|':
|
||||
if l.match('|') {
|
||||
return token{kind: tokenOr, lexeme: "||", pos: start}, nil
|
||||
}
|
||||
}
|
||||
return token{}, fmt.Errorf("unexpected character %q at %d", ch, start)
|
||||
}
|
||||
}
|
||||
|
||||
func (l *lexer) match(expected rune) bool {
|
||||
if l.pos >= len(l.src) || l.src[l.pos] != expected {
|
||||
return false
|
||||
}
|
||||
l.pos++
|
||||
return true
|
||||
}
|
||||
|
||||
func (l *lexer) skipWhitespace() {
|
||||
for l.pos < len(l.src) {
|
||||
if unicode.IsSpace(l.src[l.pos]) {
|
||||
l.pos++
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
func isIdentStart(ch rune) bool {
|
||||
return unicode.IsLetter(ch) || ch == '_'
|
||||
}
|
||||
|
||||
func isIdentPart(ch rune) bool {
|
||||
return isIdentStart(ch) || unicode.IsDigit(ch)
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue