parser beta

This commit is contained in:
2024-04-02 21:49:30 +02:00
parent 17b4ee7ef0
commit 55fe45717c
3 changed files with 110 additions and 10 deletions
+19 -6
View File
@@ -5,6 +5,7 @@ use std::fmt;
// Define token types // Define token types
#[derive(Debug, PartialEq, Clone, Copy)] #[derive(Debug, PartialEq, Clone, Copy)]
pub enum TokenType { pub enum TokenType {
Keyword,
Newline, Newline,
Whitespace, Whitespace,
Number, Number,
@@ -16,14 +17,15 @@ pub enum TokenType {
// EOF, // EOF,
} }
#[derive(Clone)]
pub struct Token { pub struct Token {
pub token_type: TokenType, pub token_type: TokenType,
pub token_value: String pub token_value: Vec<String>
} }
impl<'a> fmt::Display for Token { impl<'a> fmt::Debug for Token {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value) write!(f, "(TokenType: {:?}, TokenValue: {})", self.token_type, self.token_value.join(", "))
} }
} }
@@ -32,7 +34,7 @@ pub struct Node {
token_regex: Lazy<Regex> token_regex: Lazy<Regex>
} }
const SYNTAX: [Node; 8] = [ const SYNTAX: [Node; 9] = [
Node { Node {
token_type: TokenType::Newline, token_type: TokenType::Newline,
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()) token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
@@ -45,6 +47,10 @@ const SYNTAX: [Node; 8] = [
token_type: TokenType::Number, token_type: TokenType::Number,
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap()) token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
}, },
Node {
token_type: TokenType::Keyword,
token_regex: Lazy::new(|| Regex::new(r"^mut|try|catch|return|fn").unwrap())
},
Node { Node {
token_type: TokenType::Identifier, token_type: TokenType::Identifier,
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap()) token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
@@ -67,7 +73,7 @@ const SYNTAX: [Node; 8] = [
} }
]; ];
pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> { pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
let mut tokens: Vec<Token> = Vec::new(); let mut tokens: Vec<Token> = Vec::new();
while !code.is_empty() { while !code.is_empty() {
let mut is_match = false; let mut is_match = false;
@@ -75,10 +81,17 @@ pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
if let Some(caps) = s.token_regex.captures(code) { if let Some(caps) = s.token_regex.captures(code) {
is_match = true; is_match = true;
code = code.strip_prefix(&caps[0]).unwrap_or(code); code = code.strip_prefix(&caps[0]).unwrap_or(code);
let mut vcaps: Vec<String> = Vec::new();
let capsl = caps.len();
let mut x = 0;
while x < capsl {
vcaps.push(caps[x].to_string());
x+=1;
}
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace { if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
tokens.push(Token { tokens.push(Token {
token_type: s.token_type, token_type: s.token_type,
token_value: caps[0].to_string(), token_value: vcaps,
}); });
} }
} else { } else {
+12 -4
View File
@@ -1,11 +1,19 @@
mod lexer; mod lexer;
use lexer::lex; use lexer::lex;
mod parser;
use parser::parse;
fn main() { fn main() {
let input = "int test() {}"; let input = "int main() {";
// let input = "test->xy"; // let input = "test->xy";
let tokens = lex(input, true); let mut tokens = lex(input, false);
for token in tokens { let ast = *parse(&mut tokens);
println!("{:?} {}", token.token_type, token.token_value); for t in tokens {
println!("{:?}", t);
}
let mut _x = 1;
for a in ast {
println!("{:?}: {:?}", a.type_, a.values);
_x+=1;
} }
} }
+79
View File
@@ -0,0 +1,79 @@
// use std::collections::HashMap;
use crate::lexer::{Token, TokenType};
use std::fmt;
#[derive(Debug, Clone)]
pub enum AstTypes {
FunctionDecleration
}
pub struct Ast {
pub type_: AstTypes,
pub values: Vec<Token>
}
impl<'a> fmt::Debug for Ast {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "type={:?} values={:?}", self.type_, self.values)
}
}
#[derive(Debug, PartialEq, Clone, Copy)]
pub enum AstDef {
Normal(TokenType),
Repeated(TokenType),
Optional(TokenType)
}
pub struct PNode {
ast_type: AstTypes,
ast_match: Vec<AstDef>
}
pub fn parse(tokens: &mut Vec<Token>) -> Box<Vec<Ast>> {
let ast_def: Vec<PNode> = vec![
PNode {
ast_type: AstTypes::FunctionDecleration,
ast_match: vec![AstDef::Normal(TokenType::Identifier), AstDef::Normal(TokenType::Identifier), AstDef::Normal(TokenType::Round), AstDef::Normal(TokenType::Curly)]
},
PNode {
ast_type: AstTypes::FunctionDecleration,
ast_match: vec![AstDef::Normal(TokenType::Keyword)]
},
];
let tokens_len = tokens.len();
let mut ast: Vec<Ast> = Vec::new();
let mut current = 0;
let mut i = 0;
while tokens.len()>0 {
for ast_node in &ast_def {
if tokens_len < ast_node.ast_match.len() {
continue;
}
let mut ast_types: Vec<AstTypes> = vec![];
let mut is_match = false;
let mut matched: Vec<Token> = Vec::new();
for ast_match in &ast_node.ast_match {
match ast_match {
AstDef::Normal(token_type) => {
if &tokens[i].token_type == token_type {
matched.push(tokens[i].clone());
tokens.drain(0..1);
is_match = true;
} else {
is_match = false;
}
}
_ => {}
}
}
if is_match {
i = 0;
ast.push(Ast {values: matched, type_: ast_node.ast_type.clone()});
matched = Vec::new();
break;
}
}
}
Box::new(ast)
}