From 6faedc7c0dc218d16100940074685da8d890d504 Mon Sep 17 00:00:00 2001 From: Leo Date: Sun, 7 Apr 2024 20:49:21 +0200 Subject: [PATCH] parser change --- Cargo.toml | 2 +- src/lexer.rs | 29 +++++---- src/main.rs | 22 ++----- src/parser.rs | 162 +++++++++++++++++++------------------------------- 4 files changed, 83 insertions(+), 132 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 2edb360..f5ca315 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,5 @@ [package] -name = "wisp" +name = "wyst" version = "0.1.0" edition = "2021" diff --git a/src/lexer.rs b/src/lexer.rs index dd6d0ba..ef7f8d5 100644 --- a/src/lexer.rs +++ b/src/lexer.rs @@ -17,6 +17,7 @@ pub enum TokenType { Identifier, Ptr, Operator, + SecondOperator, Round, Curly, Square, @@ -27,14 +28,20 @@ pub enum TokenType { #[derive(Clone)] pub struct Token { pub token_type: TokenType, - pub token_values: Vec, + pub value: String, pub line: usize, pub column: usize } -impl<'a> fmt::Debug for Token { +impl fmt::Debug for Token { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "(TokenType: {:?}, TokenValue: {}, Line: {}, Column {})", self.token_type, self.token_values.join(", "), self.line, self.column) + write!(f, "(TokenType: {:?}, TokenValue: {}, Line: {}, Column {})", self.token_type, self.value, self.line, self.column) + } +} + +impl fmt::Display for Token { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + write!(f, "(TokenType: {:?}, TokenValue: {}, Line: {}, Column {})", self.token_type, self.value, self.line, self.column) } } @@ -43,7 +50,7 @@ pub struct Node { token_regex: Lazy } -const SYNTAX: [Node; 11] = [ +const SYNTAX: [Node; 12] = [ Node { token_type: TokenType::Newline, token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()), @@ -88,6 +95,10 @@ const SYNTAX: [Node; 11] = [ token_type: TokenType::Angle, token_regex: Lazy::new(|| Regex::new(r"<(?:[^<>]|(?R))*>").unwrap()) }, + Node { + token_type: TokenType::SecondOperator, + token_regex: Lazy::new(|| Regex::new(r"^,|;").unwrap()) + }, ]; pub fn lex(mut code: &str, use_whitespace: bool) -> Vec { @@ -100,18 +111,10 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec { if let Some(caps) = s.token_regex.captures(code) { is_match = true; code = code.strip_prefix(&caps[0]).unwrap_or(code); - let mut vcaps: Vec = Vec::new(); - let capsl = caps.len(); - let mut x = 0; - while x < capsl { - vcaps.push(caps[x].to_string()); - x+=1; - } - if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace { tokens.push(Token { token_type: s.token_type, - token_values: vcaps, + value: caps[0].to_string(), line: state.line, column: state.column }); diff --git a/src/main.rs b/src/main.rs index 0a99b0a..e311b1d 100644 --- a/src/main.rs +++ b/src/main.rs @@ -1,24 +1,14 @@ mod lexer; use lexer::lex; mod parser; -use parser::parse; +use parser::Parser; fn main() { - let input = "void main() {} <>"; + let input = "int main() {}"; println!("\n\n\n\n"); - // let input = "test->xy"; - let mut tokens = lex(input, false); - let ast = *parse(&mut tokens); - // for t in tokens { - // println!("{:?}", t); - // } - let mut _x = 1; - for a in ast { - println!("{:?}: [", a.type_); - for v in a.values { - println!(" {:?},", v); - } - println!("]"); - _x+=1; + let tokens = lex(input, false); + let mut ast = Parser::new(tokens.clone()); + while ast.tokens.len() > ast.index as usize { + println!("{}\n\n", ast.next()); } } diff --git a/src/parser.rs b/src/parser.rs index de6afbf..81b40eb 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -1,127 +1,85 @@ -// use std::collections::HashMap; use crate::lexer::{Token, TokenType}; use std::fmt; -use crate::parser::AstDef::{Else, Normal}; -use crate::parser::AstTypes::{FunctionDecleration, Other}; -#[derive(Debug, Clone)] -pub enum AstTypes { +#[derive(Clone, Debug)] +pub enum AstType { FunctionDecleration, VariableDecleration, Other } pub struct Ast { - pub type_: AstTypes, - pub values: Vec + pub tokens: Vec, + pub ast_type: AstType } -impl<'a> fmt::Debug for Ast { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(f, "type={:?} values={:?}", self.type_, self.values) +impl fmt::Display for Ast { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + write!(f, "{:?}: [\n", self.ast_type)?; + for (i, token) in self.tokens.iter().enumerate() { + if i < self.tokens.len() - 1 { + write!(f, " {},\n", token)?; + } else { + write!(f, " {}\n", token)?; + } + } + write!(f, "]") } } -#[derive(Debug, PartialEq, Clone)] -pub enum AstDef { - Normal(TokenType), - NormalValue(TokenType, Vec), - Repeated(Vec), - Optional(TokenType), - Else + + +pub struct Parser { + pub tokens: Vec, + pub index: u32 } -pub struct PNode { - ast_type: AstTypes, - ast_match: Vec -} -pub fn parse(tokens: &mut Vec) -> Box> { - let ast_def: Vec = vec![ - PNode { - ast_type: FunctionDecleration, - ast_match: vec![Normal(TokenType::Identifier), Normal(TokenType::Identifier), Normal(TokenType::Round), Normal(TokenType::Curly)] - }, - PNode { - ast_type: AstTypes::VariableDecleration, - ast_match: vec![Normal(TokenType::Identifier), Normal(TokenType::Identifier)] - }, - PNode { - ast_type: Other, - ast_match: vec![Else] - } - ]; - let tokens_len = tokens.len(); - let mut ast: Vec = Vec::new(); - // let mut i = 0; - while tokens.len()>0 { - for ast_node in &ast_def { - if tokens_len < ast_node.ast_match.len() { - continue; - } - let mut is_match = false; - let mut matched: Vec = Vec::new(); - for ast_match in &ast_node.ast_match { - match ast_match { - AstDef::Normal(token_type) => { - if &tokens[0].token_type == token_type { - matched.push(tokens[0].clone()); - tokens.drain(0..1); - is_match = true; +impl Parser { + pub fn new(tokens: Vec) -> Parser { + Parser {tokens: tokens, index: 0} + } + pub fn next(&mut self) -> Ast { + let mut ast_res: Ast = Ast {tokens: vec![], ast_type: AstType::Other}; + let mut index = self.index as usize; + if index == self.tokens.len() {panic!("Reached the end of tokens")} + let token = &self.tokens[index]; + match token.token_type { + TokenType::Identifier => { + if self.tokens[index+1].token_type==TokenType::Identifier && self.tokens[index+2].token_type==TokenType::Round && self.tokens[index+3].token_type==TokenType::Curly { + ast_res.tokens.push(self.tokens[index].clone()); + ast_res.tokens.push(self.tokens[index+1].clone()); + ast_res.tokens.push(self.tokens[index+2].clone()); + ast_res.tokens.push(self.tokens[index+3].clone()); + ast_res.ast_type = AstType::FunctionDecleration; + self.index += 3; + } else { + loop { + if index+1 >= self.tokens.len() {break;} + index = self.index as usize; // Update the index + let ntk = &self.tokens[index]; // Stands for next token + if ntk.value=="," { + ast_res.tokens.push(ntk.clone()); + self.index += 1; + } else if ntk.token_type==TokenType::Identifier { + ast_res.tokens.push(ntk.clone()); + ast_res.ast_type = AstType::VariableDecleration; + self.index += 1; + } else if ntk.token_type==TokenType::Angle { + ast_res.tokens.push(ntk.clone()); + self.index += 1; } else { - is_match = false; + self.index -= 1; break; - } + } } - AstDef::NormalValue(token_type, token_value) => { - if &tokens[0].token_type == token_type && &tokens[0].token_values == token_value { - matched.push(tokens[0].clone()); - tokens.drain(0..1); - is_match = true; - } else { - is_match = false; - break; - } - } - AstDef::Repeated(nodes) => { - loop { - let mut node_match = false; - for &node_ in nodes { - if node_ == tokens[0].token_type { - matched.push(tokens[0].clone()); - node_match = true; - tokens.drain(0..1); - } - } - if node_match { - is_match = true; - } else { - break; - } - } - } - AstDef::Optional(token_type) => { - if &tokens[0].token_type == token_type { - matched.push(tokens[0].clone()); - tokens.drain(0..1); - is_match = true; - break; - } - } - AstDef::Else => { - matched.push(tokens[0].clone()); - tokens.drain(0..1); - is_match = true; - } - _ => {} } } - if is_match { - ast.push(Ast {values: matched, type_: ast_node.ast_type.clone()}); - matched = Vec::new(); - break; + _ => { + ast_res.tokens.push(token.clone()); } } + self.index += 1; + ast_res } - Box::new(ast) -} +} \ No newline at end of file