From e9cf9273900dda8494a7398bc7b9fc8337bb7936 Mon Sep 17 00:00:00 2001 From: Leo Date: Wed, 10 Apr 2024 20:58:55 +0200 Subject: [PATCH] another bug fix --- src/lexer.rs | 42 ++++++++++++++--------------- src/parser.rs | 67 +++++++++++++++++++++++++++++++++++++++++++++-- src/transpiler.rs | 2 +- 3 files changed, 85 insertions(+), 26 deletions(-) diff --git a/src/lexer.rs b/src/lexer.rs index a844a22..b756fe3 100644 --- a/src/lexer.rs +++ b/src/lexer.rs @@ -3,8 +3,8 @@ use once_cell::sync::Lazy; use std::fmt; pub struct LexerState { - line: usize, - column: usize + pub line: usize, + pub column: usize } // Define token types @@ -51,7 +51,15 @@ pub struct Node { token_regex: Lazy } -const SYNTAX: [Node; 13] = [ +const SYNTAX: [Node; 12] = [ + Node { + token_type: TokenType::Semicolon, + token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap()) + }, + Node { + token_type: TokenType::SecondOperator, + token_regex: Lazy::new(|| Regex::new(r"^,").unwrap()) + }, Node { token_type: TokenType::Newline, token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()), @@ -60,17 +68,17 @@ const SYNTAX: [Node; 13] = [ token_type: TokenType::Whitespace, token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap()) }, - Node { - token_type: TokenType::Number, - token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap()) - }, Node { token_type: TokenType::Keyword, token_regex: Lazy::new(|| Regex::new(r"^mut|try|catch|return|fn\b").unwrap()) }, Node { token_type: TokenType::Identifier, - token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*(:?<[._a-zA-Z][a-zA-Z0-9_<>]*>)?").unwrap()) + token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*(:?\s*<[._a-zA-Z][a-zA-Z0-9_<>]*>)?").unwrap()) + }, + Node { + token_type: TokenType::Number, + token_regex: Lazy::new(|| Regex::new(r"^(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap()) }, Node { token_type: TokenType::Ptr, @@ -82,27 +90,15 @@ const SYNTAX: [Node; 13] = [ }, Node { token_type: TokenType::Round, - token_regex: Lazy::new(|| Regex::new(r"^\((?:[^()]|(?R))*\)").unwrap()) + token_regex: Lazy::new(|| Regex::new(r"^[\(|\)]").unwrap()) }, Node { token_type: TokenType::Curly, - token_regex: Lazy::new(|| Regex::new(r"^\{(?:[^{}]|(?R))*}").unwrap()) + token_regex: Lazy::new(|| Regex::new(r"^[\{|\}]").unwrap()) }, Node { token_type: TokenType::Square, - token_regex: Lazy::new(|| Regex::new(r"^\[(?:[^\[\]]|(?R))*]").unwrap()) - }, - Node { - token_type: TokenType::Angle, - token_regex: Lazy::new(|| Regex::new(r"^<(?:[^<>]|(?R))*>").unwrap()) - }, - Node { - token_type: TokenType::Semicolon, - token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap()) - }, - Node { - token_type: TokenType::SecondOperator, - token_regex: Lazy::new(|| Regex::new(r"^,").unwrap()) + token_regex: Lazy::new(|| Regex::new(r"^[\[|\]]").unwrap()) }, ]; diff --git a/src/parser.rs b/src/parser.rs index ccc5b9e..95ba003 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -1,4 +1,4 @@ -use crate::lexer::{Token, TokenType}; +use crate::lexer::{LexerState, Token, TokenType}; use std::fmt; #[derive(Clone, Debug, PartialEq, Eq)] @@ -48,10 +48,73 @@ pub struct Parser { pub index: u32 } +const BRACKETS: [TokenType; 3] = [TokenType::Round, TokenType::Curly, TokenType::Square]; + +pub fn pre_parse (tokens: Vec) -> Vec { + println!("{:?}", tokens); + let mut result: Vec = Vec::new(); + let mut bracket: Vec = Vec::new(); + let mut inner_string = String::new(); + let mut bracket_state: LexerState = LexerState {line: 0, column: 0}; + for token in &tokens { + let is_bracket = BRACKETS.contains(&token.token_type); + match token.value.as_str() { + "(" => { + bracket.push(token.token_type); + inner_string += token.value.as_str(); + } + ")" => { + if bracket[bracket.len()-1] == token.token_type { + bracket.pop(); + inner_string += token.value.as_str(); + bracket_state.line = token.line; + bracket_state.column = token.column; + } + } + "{" => { + bracket.push(token.token_type); + inner_string += token.value.as_str(); + } + "}" => { + if bracket[bracket.len()-1] == token.token_type { + bracket.pop(); + inner_string += token.value.as_str(); + bracket_state.line = token.line; + bracket_state.column = token.column; + } + } + "[" => { + bracket.push(token.token_type); + inner_string += token.value.as_str(); + } + "]" => { + if bracket[bracket.len()-1] == token.token_type { + bracket.pop(); + inner_string += token.value.as_str(); + bracket_state.line = token.line; + bracket_state.column = token.column; + } + } + _ => { + if bracket.len() > 0 { + inner_string += token.value.as_str() + } else if token.token_type != TokenType::Whitespace { + result.push(token.clone()) + } + } + } + if is_bracket && bracket.len() == 0 { + result.push(Token {token_type: token.token_type, value: inner_string.to_string(), line: bracket_state.line, column: bracket_state.column}); + inner_string = String::new(); + } + } + + result +} impl Parser { pub fn new(tokens: Vec) -> Parser { - Parser {tokens: tokens, index: 0} + Parser {tokens: pre_parse(tokens), index: 0} } pub fn next(&mut self) -> Ast { let mut ast_res: Ast = Ast {tokens: vec![], ast_type: AstType::Other}; diff --git a/src/transpiler.rs b/src/transpiler.rs index f2b70ee..7ccf96b 100644 --- a/src/transpiler.rs +++ b/src/transpiler.rs @@ -21,7 +21,7 @@ pub fn transpile(input: String, indent: u32) -> String { input = auto_strip(input); } let mut result = String::new(); - let tokens = lex(input.as_str(), false); + let tokens = lex(input.as_str(), true); println!("\n\n\n\n"); let mut full_ast = Parser::new(tokens.clone()); while full_ast.tokens.len() > full_ast.index as usize {