From fec3c39de2e93ae625ec440121336e3055a392c5 Mon Sep 17 00:00:00 2001 From: Leo dev Date: Fri, 9 Aug 2024 21:05:03 +0200 Subject: [PATCH] go lexer --- ast/ast.go | 2 ++ go.mod | 3 +++ lexer.go | 54 ++++++++++++++++++++++++++++++++++++++++++++++++++ main.go | 18 +++++++++++++++++ token/token.go | 26 ++++++++++++++++++++++++ utils.go | 2 ++ 6 files changed, 105 insertions(+) create mode 100644 ast/ast.go create mode 100644 go.mod create mode 100644 lexer.go create mode 100644 main.go create mode 100644 token/token.go create mode 100644 utils.go diff --git a/ast/ast.go b/ast/ast.go new file mode 100644 index 0000000..9d89e55 --- /dev/null +++ b/ast/ast.go @@ -0,0 +1,2 @@ +package ast + diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..6eb8106 --- /dev/null +++ b/go.mod @@ -0,0 +1,3 @@ +module github.com/wyst-lang/wyst + +go 1.21.5 diff --git a/lexer.go b/lexer.go new file mode 100644 index 0000000..dba8230 --- /dev/null +++ b/lexer.go @@ -0,0 +1,54 @@ +package main + +import ( + "fmt" + "strings" + + "github.com/wyst-lang/wyst/token" +) + +type State struct { + Line int + Column int +} + +type Token struct { + Rule token.Rule + Value string + Position State +} + +func (t Token) String() string { + return fmt.Sprintf("Token(\n rule=%s,\n value=%s\n)", t.Rule, t.Value) +} + +func Tokenize(code string, rules []token.Rule) ([]Token, error) { + var tokens = []Token{} + var state = State{1, 0} + for code != "" { + matched := false + for i := 0; i < len(rules); i++ { + if rules[i].Pattern.MatchString(code) { + match := rules[i].Pattern.FindString(code) + matched = true + code = code[len(match):] + tokens = append(tokens, Token{Rule: rules[i], Value: match, Position: state}) + state.Column += len(match) + break + } else if strings.HasPrefix(code, "\n") { + state.Line += 1 + state.Column = 0 + code = code[1:] + matched = true + } else if strings.HasPrefix(code, "\t") || strings.HasPrefix(code, " ") { + state.Column += 1 + code = code[1:] + matched = true + } + } + if !matched { + return tokens, fmt.Errorf("%d:%d", state.Line, state.Column) + } + } + return tokens, nil +} diff --git a/main.go b/main.go new file mode 100644 index 0000000..c94e1af --- /dev/null +++ b/main.go @@ -0,0 +1,18 @@ +package main + +import ( + "fmt" + + "github.com/wyst-lang/wyst/token" +) + +func main() { + tokens, err := Tokenize("test 123", token.RULES_TOP) + if err != nil { + fmt.Printf("Syntax error at %s \n", err) + return + } + for i := 0; i < len(tokens); i++ { + fmt.Printf("%s\n", tokens[i]) + } +} diff --git a/token/token.go b/token/token.go new file mode 100644 index 0000000..1d7aba7 --- /dev/null +++ b/token/token.go @@ -0,0 +1,26 @@ +package token + +import ( + "fmt" + "regexp" +) + +type Rule struct { + Pattern regexp.Regexp + Name string +} + +func (t Rule) String() string { + return fmt.Sprintf("TokenRule(%s)", t.Name) +} + +func NewToken(name string, pattern regexp.Regexp) Rule { + return Rule{Name: name, Pattern: pattern} +} + +var ( + NUMBER = NewToken("NUMBER", *regexp.MustCompile("^[0-9]*")) + IDENTIFIER = NewToken("IDENTIFIER", *regexp.MustCompile("^[A-Za-z_][A-Za-z_0-9]*")) +) + +var RULES_TOP = []Rule{IDENTIFIER, NUMBER} diff --git a/utils.go b/utils.go new file mode 100644 index 0000000..c9ecbf5 --- /dev/null +++ b/utils.go @@ -0,0 +1,2 @@ +package main +