antlr parser

This commit is contained in:
2024-08-11 15:01:18 +02:00
parent 154e9597ae
commit b79b00e2d9
11 changed files with 104 additions and 232 deletions
+29
View File
@@ -0,0 +1,29 @@
# Created by https://www.toptal.com/developers/gitignore/api/go
# Edit at https://www.toptal.com/developers/gitignore?templates=go
### Go ###
# If you prefer the allow list template instead of the deny list, see community template:
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
#
# Binaries for programs and plugins
*.exe
*.exe~
*.dll
*.so
*.dylib
# Test binary, built with `go test -c`
*.test
# Output of the go coverage tool, specifically when used with LiteIDE
*.out
# Dependency directories (remove the comment below to include it)
# vendor/
# Go workspace file
go.work
# End of https://www.toptal.com/developers/gitignore/api/go
parser/
+5 -1
View File
@@ -14,4 +14,8 @@ int main() {
printf("Hello, World!"); printf("Hello, World!");
return 0; return 0;
} }
``` ```
# For developers
If the grammar has been updated run: `antlr -Dlanguage=Go -o parser Wyst.g4`
+11
View File
@@ -0,0 +1,11 @@
grammar Wyst;
WS : [ \t\r\n]+ -> skip;
expr: expr ('*'|'/') expr
| expr ('+'|'-') expr
| INT
| '(' expr ')'
;
INT: [0-9]+;
+10 -1
View File
@@ -1,3 +1,12 @@
module github.com/wyst-lang/wyst module github.com/wyst-lang/wyst
go 1.21.5 go 1.22
toolchain go1.22.6
require github.com/antlr4-go/antlr/v4 v4.13.1 // direct
require (
// github.com/antlr/antlr4/runtime/Go/antlr v1.4.10 // indirect
golang.org/x/exp v0.0.0-20240506185415-9bf2ced13842 // indirect
)
+8
View File
@@ -0,0 +1,8 @@
github.com/antlr/antlr4/runtime/Go/antlr v1.4.10 h1:yL7+Jz0jTC6yykIK/Wh74gnTJnrGr5AyrNMXuA0gves=
github.com/antlr/antlr4/runtime/Go/antlr v1.4.10/go.mod h1:F7bn7fEU90QkQ3tnmaTx3LTKLEDqnwWODIYppRQ5hnY=
github.com/antlr4-go/antlr v0.0.0-20230518091524-98b52378c522 h1:o+W7GDFUwWtVkN28CW/nhh/aCmHn6OJddUs3+8vMMjs=
github.com/antlr4-go/antlr v0.0.0-20230518091524-98b52378c522/go.mod h1:srLVvW4JLxy+tCG9Nn2l8al77mUIMCwAOLQocfLDU2w=
github.com/antlr4-go/antlr/v4 v4.13.1 h1:SqQKkuVZ+zWkMMNkjy5FZe5mr5WURWnlpmOuzYWrPrQ=
github.com/antlr4-go/antlr/v4 v4.13.1/go.mod h1:GKmUxMtwp6ZgGwZSva4eWPC5mS6vUAmOABFgjdkM7Nw=
golang.org/x/exp v0.0.0-20240506185415-9bf2ced13842 h1:vr/HnozRka3pE4EsMEg1lgkXJkTFJCVUX+S/ZT6wYzM=
golang.org/x/exp v0.0.0-20240506185415-9bf2ced13842/go.mod h1:XtvwrStGgqGPLc4cjQfWqZHG1YFdYs6swckp8vpsjnc=
+2 -11
View File
@@ -1,18 +1,9 @@
package main package main
import ( import (
"fmt"
"github.com/wyst-lang/wyst/parser"
) )
func main() { func main() {
code := "int x;" tree, parser := Parse("")
ast, err := parser.ParseString(code) iterateTree(tree, parser, 0)
if err != nil {
fmt.Printf("SyntaxErr: %s\n", err)
}
for i := 0; i < len(ast); i++ {
fmt.Printf("%s\n", ast[i])
}
} }
-65
View File
@@ -1,65 +0,0 @@
package parser
import (
"fmt"
)
type RuleKind int
const (
RK_Required RuleKind = iota
RK_Optional
RK_Repeat
RK_Port
)
type Rule struct {
Name string
Kind RuleKind
Inner []Rule
}
type AstNode struct {
Rule Rule
Value string
Inner []AstNode
}
func (t AstNode) String() string {
return fmt.Sprintf("AstNode(\n value=%s\n rule=%v,\n inner=%v,\n)", t.Value, t.Rule, t.Inner)
}
func (t Rule) String() string {
return t.Name
}
func NewRule(name string, inner []Rule) Rule {
return Rule{name, RK_Required, inner}
}
func NewPort(name string, inner []Rule) Rule {
return Rule{name, RK_Port, inner}
}
func Optional(rule Rule) Rule {
rule.Kind = RK_Optional
return rule
}
func Repeat(rule Rule) Rule {
rule.Kind = RK_Repeat
return rule
}
var (
IDENTIFIER = NewPort("IDENTIFIER", []Rule{})
NUMBER = NewPort("NUMBER", []Rule{})
SEMICOLON = NewPort("SEMICOLON", []Rule{})
EXPR = NewRule("EXPR", []Rule{IDENTIFIER})
CODE_BLOCK = NewRule("CODE_BLOCK", []Rule{EXPR, SEMICOLON})
VAR_DEF = NewRule("VAR_DEF", []Rule{IDENTIFIER, IDENTIFIER})
FUNC_DEF = NewRule("FUNC_DEF", []Rule{IDENTIFIER, IDENTIFIER, CODE_BLOCK})
)
var (
TOP_RULE = []Rule{FUNC_DEF}
)
-71
View File
@@ -1,71 +0,0 @@
package parser
import (
"fmt"
"github.com/wyst-lang/wyst/tokenizer"
)
func ParseString(code string) ([]AstNode, error) {
tokens, state, err := tokenizer.Tokenize(code)
ast, err1 := Parse(tokens, TOP_RULE)
if err != nil {
return ast, fmt.Errorf("LexingError at %d:%d: %s", state.Line, state.Column, err)
} else if err1 != nil {
return ast, fmt.Errorf("ParserError")
}
return ast, nil
}
func Parse(tokens []tokenizer.Token, rule_set []Rule) ([]AstNode, error) {
var ast = []AstNode{}
for i := 0; i < len(tokens); i++ {
for r := 0; r < len(rule_set); r++ {
matching := false
var match_nodes = []AstNode{}
if rule_set[r].Kind == RK_Port {
if tokens[i].Rule.Name == rule_set[r].Name {
matching = true
match_nodes = append(match_nodes, AstNode{Rule: rule_set[r], Value: tokens[i].Value})
}
} else {
for v := 0; v < len(rule_set[r].Inner); v++ {
if rule_set[r].Inner[v].Kind == RK_Required {
if len(tokens)-i > v && tokens[i].Rule.Name == rule_set[r].Inner[v].Name {
matching = true
match_nodes = append(match_nodes, AstNode{Rule: rule_set[r].Inner[v], Value: tokens[i].Value})
} else {
matching = false
break
}
} else if rule_set[r].Inner[v].Kind == RK_Repeat {
for l := 0; len(tokens)-i > l && tokens[i+l].Rule.Name == rule_set[r].Inner[v].Name; l++ {
match_nodes = append(match_nodes, AstNode{Rule: rule_set[r].Inner[v], Value: tokens[i+l].Value})
matching = true
}
} else if rule_set[r].Inner[v].Kind == RK_Optional {
match_nodes = append(match_nodes, AstNode{Rule: rule_set[r].Inner[v], Value: tokens[i].Value})
matching = true
}
}
}
if matching {
ast = append(ast, AstNode{Rule: rule_set[r], Inner: match_nodes, Value: merge_nodes(match_nodes)})
i += len(match_nodes) - 1
break
}
}
}
return ast, nil
}
func merge_nodes(match_nodes []AstNode) string {
str := ""
for i := 0; i < len(match_nodes); i++ {
str += match_nodes[i].Value
if len(match_nodes) > i {
str += " "
}
}
return str
}
-29
View File
@@ -1,29 +0,0 @@
package tokenizer
import (
"fmt"
"regexp"
)
type Rule struct {
Pattern regexp.Regexp
Name string
}
func (t Rule) String() string {
return fmt.Sprintf("TokenRule(%s)", t.Name)
}
var RULES = []Rule{}
func NewRule(name string, pattern regexp.Regexp) Rule {
var rule = Rule{Name: name, Pattern: pattern}
RULES = append(RULES, rule)
return rule
}
var (
NUMBER = NewRule("NUMBER", *regexp.MustCompile(`^\d+`))
IDENTIFIER = NewRule("IDENTIFIER", *regexp.MustCompile("^[A-Za-z_][A-Za-z_0-9]*"))
SEMICOLON = NewRule("SEMICOLON", *regexp.MustCompile("^;"))
)
-54
View File
@@ -1,54 +0,0 @@
package tokenizer
import (
"fmt"
"strings"
)
type State struct {
Line int
Column int
}
type Token struct {
Rule Rule
Value string
Position State
}
func (t Token) String() string {
return fmt.Sprintf("Token(\n rule=%s,\n value=%s\n)", t.Rule, t.Value)
}
func Tokenize(code string) ([]Token, State, error) {
var tokens = []Token{}
var state = State{1, 0}
var err error = nil
for code != "" {
matched := false
for i := 0; i < len(RULES); i++ {
if RULES[i].Pattern.MatchString(code) {
match := RULES[i].Pattern.FindString(code)
matched = true
code = code[len(match):]
tokens = append(tokens, Token{Rule: RULES[i], Value: match, Position: state})
state.Column += len(match)
break
} else if strings.HasPrefix(code, "\n") {
state.Line += 1
state.Column = 0
code = code[1:]
matched = true
} else if strings.HasPrefix(code, "\t") || strings.HasPrefix(code, " ") {
state.Column += 1
code = code[1:]
matched = true
}
}
if !matched {
code = code[1:]
err = fmt.Errorf("invalid character or token")
}
}
return tokens, state, err
}
+39
View File
@@ -1,2 +1,41 @@
package main package main
import (
"fmt"
"github.com/antlr4-go/antlr/v4"
"github.com/wyst-lang/wyst/parser"
)
func Parse(code string) (antlr.ParseTree, *parser.WystParser) {
chars := antlr.NewInputStream(code)
lexer := parser.NewWystLexer(chars)
stream := antlr.NewCommonTokenStream(lexer, 0)
p := parser.NewWystParser(stream)
// p.AddErrorListener(antlr.NewDiagnosticErrorListener(true).WithContext(p))
tree := p.Expr()
return tree, p
}
func iterateTree(node antlr.Tree, wparser *parser.WystParser, depth int) {
if ruleContext, ok := node.(antlr.RuleContext); ok {
ruleIndex := ruleContext.GetRuleIndex()
ruleName := wparser.RuleNames[ruleIndex]
fmt.Printf("%sRule: %s\n", indent(depth), ruleName)
}
if parseTree, ok := node.(antlr.ParseTree); ok {
fmt.Printf("%sNode: %s\n", indent(depth), parseTree.GetText())
}
for i := 0; i < node.GetChildCount(); i++ {
child := node.GetChild(i)
iterateTree(child, wparser, depth+1)
}
}
func indent(depth int) string {
return fmt.Sprintf("%s", string(make([]byte, depth*2)))
}