From b79b00e2d99d4015a3014f14088cc465dc12deb4 Mon Sep 17 00:00:00 2001 From: Leo dev Date: Sun, 11 Aug 2024 15:01:18 +0200 Subject: [PATCH] antlr parser --- .gitignore | 29 +++++++++++++++++ README.md | 6 +++- Wyst.g4 | 11 +++++++ go.mod | 11 ++++++- go.sum | 8 +++++ main.go | 13 ++------ parser/ast.go | 65 -------------------------------------- parser/parser.go | 71 ------------------------------------------ tokenizer/token.go | 29 ----------------- tokenizer/tokenizer.go | 54 -------------------------------- utils.go | 39 +++++++++++++++++++++++ 11 files changed, 104 insertions(+), 232 deletions(-) create mode 100644 .gitignore create mode 100644 Wyst.g4 create mode 100644 go.sum delete mode 100644 parser/ast.go delete mode 100644 parser/parser.go delete mode 100644 tokenizer/token.go delete mode 100644 tokenizer/tokenizer.go diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..c4fba4e --- /dev/null +++ b/.gitignore @@ -0,0 +1,29 @@ +# Created by https://www.toptal.com/developers/gitignore/api/go +# Edit at https://www.toptal.com/developers/gitignore?templates=go + +### Go ### +# If you prefer the allow list template instead of the deny list, see community template: +# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore +# +# Binaries for programs and plugins +*.exe +*.exe~ +*.dll +*.so +*.dylib + +# Test binary, built with `go test -c` +*.test + +# Output of the go coverage tool, specifically when used with LiteIDE +*.out + +# Dependency directories (remove the comment below to include it) +# vendor/ + +# Go workspace file +go.work + +# End of https://www.toptal.com/developers/gitignore/api/go + +parser/ \ No newline at end of file diff --git a/README.md b/README.md index 772cd4c..10493ee 100644 --- a/README.md +++ b/README.md @@ -14,4 +14,8 @@ int main() { printf("Hello, World!"); return 0; } -``` \ No newline at end of file +``` + +# For developers + +If the grammar has been updated run: `antlr -Dlanguage=Go -o parser Wyst.g4` \ No newline at end of file diff --git a/Wyst.g4 b/Wyst.g4 new file mode 100644 index 0000000..8d2d2a9 --- /dev/null +++ b/Wyst.g4 @@ -0,0 +1,11 @@ +grammar Wyst; + +WS : [ \t\r\n]+ -> skip; + +expr: expr ('*'|'/') expr + | expr ('+'|'-') expr + | INT + | '(' expr ')' + ; + +INT: [0-9]+; \ No newline at end of file diff --git a/go.mod b/go.mod index 6eb8106..af7a2e8 100644 --- a/go.mod +++ b/go.mod @@ -1,3 +1,12 @@ module github.com/wyst-lang/wyst -go 1.21.5 +go 1.22 + +toolchain go1.22.6 + +require github.com/antlr4-go/antlr/v4 v4.13.1 // direct + +require ( + // github.com/antlr/antlr4/runtime/Go/antlr v1.4.10 // indirect + golang.org/x/exp v0.0.0-20240506185415-9bf2ced13842 // indirect +) diff --git a/go.sum b/go.sum new file mode 100644 index 0000000..d57d439 --- /dev/null +++ b/go.sum @@ -0,0 +1,8 @@ +github.com/antlr/antlr4/runtime/Go/antlr v1.4.10 h1:yL7+Jz0jTC6yykIK/Wh74gnTJnrGr5AyrNMXuA0gves= +github.com/antlr/antlr4/runtime/Go/antlr v1.4.10/go.mod h1:F7bn7fEU90QkQ3tnmaTx3LTKLEDqnwWODIYppRQ5hnY= +github.com/antlr4-go/antlr v0.0.0-20230518091524-98b52378c522 h1:o+W7GDFUwWtVkN28CW/nhh/aCmHn6OJddUs3+8vMMjs= +github.com/antlr4-go/antlr v0.0.0-20230518091524-98b52378c522/go.mod h1:srLVvW4JLxy+tCG9Nn2l8al77mUIMCwAOLQocfLDU2w= +github.com/antlr4-go/antlr/v4 v4.13.1 h1:SqQKkuVZ+zWkMMNkjy5FZe5mr5WURWnlpmOuzYWrPrQ= +github.com/antlr4-go/antlr/v4 v4.13.1/go.mod h1:GKmUxMtwp6ZgGwZSva4eWPC5mS6vUAmOABFgjdkM7Nw= +golang.org/x/exp v0.0.0-20240506185415-9bf2ced13842 h1:vr/HnozRka3pE4EsMEg1lgkXJkTFJCVUX+S/ZT6wYzM= +golang.org/x/exp v0.0.0-20240506185415-9bf2ced13842/go.mod h1:XtvwrStGgqGPLc4cjQfWqZHG1YFdYs6swckp8vpsjnc= diff --git a/main.go b/main.go index 2d3a6db..f8dc3fd 100644 --- a/main.go +++ b/main.go @@ -1,18 +1,9 @@ package main import ( - "fmt" - - "github.com/wyst-lang/wyst/parser" ) func main() { - code := "int x;" - ast, err := parser.ParseString(code) - if err != nil { - fmt.Printf("SyntaxErr: %s\n", err) - } - for i := 0; i < len(ast); i++ { - fmt.Printf("%s\n", ast[i]) - } + tree, parser := Parse("") + iterateTree(tree, parser, 0) } diff --git a/parser/ast.go b/parser/ast.go deleted file mode 100644 index 659f456..0000000 --- a/parser/ast.go +++ /dev/null @@ -1,65 +0,0 @@ -package parser - -import ( - "fmt" -) - -type RuleKind int - -const ( - RK_Required RuleKind = iota - RK_Optional - RK_Repeat - RK_Port -) - -type Rule struct { - Name string - Kind RuleKind - Inner []Rule -} - -type AstNode struct { - Rule Rule - Value string - Inner []AstNode -} - -func (t AstNode) String() string { - return fmt.Sprintf("AstNode(\n value=%s\n rule=%v,\n inner=%v,\n)", t.Value, t.Rule, t.Inner) -} -func (t Rule) String() string { - return t.Name -} - -func NewRule(name string, inner []Rule) Rule { - return Rule{name, RK_Required, inner} -} - -func NewPort(name string, inner []Rule) Rule { - return Rule{name, RK_Port, inner} -} - -func Optional(rule Rule) Rule { - rule.Kind = RK_Optional - return rule -} - -func Repeat(rule Rule) Rule { - rule.Kind = RK_Repeat - return rule -} - -var ( - IDENTIFIER = NewPort("IDENTIFIER", []Rule{}) - NUMBER = NewPort("NUMBER", []Rule{}) - SEMICOLON = NewPort("SEMICOLON", []Rule{}) - EXPR = NewRule("EXPR", []Rule{IDENTIFIER}) - CODE_BLOCK = NewRule("CODE_BLOCK", []Rule{EXPR, SEMICOLON}) - VAR_DEF = NewRule("VAR_DEF", []Rule{IDENTIFIER, IDENTIFIER}) - FUNC_DEF = NewRule("FUNC_DEF", []Rule{IDENTIFIER, IDENTIFIER, CODE_BLOCK}) -) - -var ( - TOP_RULE = []Rule{FUNC_DEF} -) diff --git a/parser/parser.go b/parser/parser.go deleted file mode 100644 index 25407b0..0000000 --- a/parser/parser.go +++ /dev/null @@ -1,71 +0,0 @@ -package parser - -import ( - "fmt" - - "github.com/wyst-lang/wyst/tokenizer" -) - -func ParseString(code string) ([]AstNode, error) { - tokens, state, err := tokenizer.Tokenize(code) - ast, err1 := Parse(tokens, TOP_RULE) - if err != nil { - return ast, fmt.Errorf("LexingError at %d:%d: %s", state.Line, state.Column, err) - } else if err1 != nil { - return ast, fmt.Errorf("ParserError") - } - return ast, nil -} - -func Parse(tokens []tokenizer.Token, rule_set []Rule) ([]AstNode, error) { - var ast = []AstNode{} - for i := 0; i < len(tokens); i++ { - for r := 0; r < len(rule_set); r++ { - matching := false - var match_nodes = []AstNode{} - if rule_set[r].Kind == RK_Port { - if tokens[i].Rule.Name == rule_set[r].Name { - matching = true - match_nodes = append(match_nodes, AstNode{Rule: rule_set[r], Value: tokens[i].Value}) - } - } else { - for v := 0; v < len(rule_set[r].Inner); v++ { - if rule_set[r].Inner[v].Kind == RK_Required { - if len(tokens)-i > v && tokens[i].Rule.Name == rule_set[r].Inner[v].Name { - matching = true - match_nodes = append(match_nodes, AstNode{Rule: rule_set[r].Inner[v], Value: tokens[i].Value}) - } else { - matching = false - break - } - } else if rule_set[r].Inner[v].Kind == RK_Repeat { - for l := 0; len(tokens)-i > l && tokens[i+l].Rule.Name == rule_set[r].Inner[v].Name; l++ { - match_nodes = append(match_nodes, AstNode{Rule: rule_set[r].Inner[v], Value: tokens[i+l].Value}) - matching = true - } - } else if rule_set[r].Inner[v].Kind == RK_Optional { - match_nodes = append(match_nodes, AstNode{Rule: rule_set[r].Inner[v], Value: tokens[i].Value}) - matching = true - } - } - } - if matching { - ast = append(ast, AstNode{Rule: rule_set[r], Inner: match_nodes, Value: merge_nodes(match_nodes)}) - i += len(match_nodes) - 1 - break - } - } - } - return ast, nil -} - -func merge_nodes(match_nodes []AstNode) string { - str := "" - for i := 0; i < len(match_nodes); i++ { - str += match_nodes[i].Value - if len(match_nodes) > i { - str += " " - } - } - return str -} diff --git a/tokenizer/token.go b/tokenizer/token.go deleted file mode 100644 index 3caf7c3..0000000 --- a/tokenizer/token.go +++ /dev/null @@ -1,29 +0,0 @@ -package tokenizer - -import ( - "fmt" - "regexp" -) - -type Rule struct { - Pattern regexp.Regexp - Name string -} - -func (t Rule) String() string { - return fmt.Sprintf("TokenRule(%s)", t.Name) -} - -var RULES = []Rule{} - -func NewRule(name string, pattern regexp.Regexp) Rule { - var rule = Rule{Name: name, Pattern: pattern} - RULES = append(RULES, rule) - return rule -} - -var ( - NUMBER = NewRule("NUMBER", *regexp.MustCompile(`^\d+`)) - IDENTIFIER = NewRule("IDENTIFIER", *regexp.MustCompile("^[A-Za-z_][A-Za-z_0-9]*")) - SEMICOLON = NewRule("SEMICOLON", *regexp.MustCompile("^;")) -) diff --git a/tokenizer/tokenizer.go b/tokenizer/tokenizer.go deleted file mode 100644 index 3075aff..0000000 --- a/tokenizer/tokenizer.go +++ /dev/null @@ -1,54 +0,0 @@ -package tokenizer - -import ( - "fmt" - "strings" -) - -type State struct { - Line int - Column int -} - -type Token struct { - Rule Rule - Value string - Position State -} - -func (t Token) String() string { - return fmt.Sprintf("Token(\n rule=%s,\n value=%s\n)", t.Rule, t.Value) -} - -func Tokenize(code string) ([]Token, State, error) { - var tokens = []Token{} - var state = State{1, 0} - var err error = nil - for code != "" { - matched := false - for i := 0; i < len(RULES); i++ { - if RULES[i].Pattern.MatchString(code) { - match := RULES[i].Pattern.FindString(code) - matched = true - code = code[len(match):] - tokens = append(tokens, Token{Rule: RULES[i], Value: match, Position: state}) - state.Column += len(match) - break - } else if strings.HasPrefix(code, "\n") { - state.Line += 1 - state.Column = 0 - code = code[1:] - matched = true - } else if strings.HasPrefix(code, "\t") || strings.HasPrefix(code, " ") { - state.Column += 1 - code = code[1:] - matched = true - } - } - if !matched { - code = code[1:] - err = fmt.Errorf("invalid character or token") - } - } - return tokens, state, err -} diff --git a/utils.go b/utils.go index c9ecbf5..83afb7f 100644 --- a/utils.go +++ b/utils.go @@ -1,2 +1,41 @@ package main +import ( + "fmt" + + "github.com/antlr4-go/antlr/v4" + "github.com/wyst-lang/wyst/parser" +) + +func Parse(code string) (antlr.ParseTree, *parser.WystParser) { + chars := antlr.NewInputStream(code) + lexer := parser.NewWystLexer(chars) + stream := antlr.NewCommonTokenStream(lexer, 0) + + p := parser.NewWystParser(stream) + // p.AddErrorListener(antlr.NewDiagnosticErrorListener(true).WithContext(p)) + + tree := p.Expr() + return tree, p +} + +func iterateTree(node antlr.Tree, wparser *parser.WystParser, depth int) { + if ruleContext, ok := node.(antlr.RuleContext); ok { + ruleIndex := ruleContext.GetRuleIndex() + ruleName := wparser.RuleNames[ruleIndex] + fmt.Printf("%sRule: %s\n", indent(depth), ruleName) + } + + if parseTree, ok := node.(antlr.ParseTree); ok { + fmt.Printf("%sNode: %s\n", indent(depth), parseTree.GetText()) + } + + for i := 0; i < node.GetChildCount(); i++ { + child := node.GetChild(i) + iterateTree(child, wparser, depth+1) + } +} + +func indent(depth int) string { + return fmt.Sprintf("%s", string(make([]byte, depth*2))) +}