From ac7df0a9fe5cd44db8831041f0e64d7709a93119 Mon Sep 17 00:00:00 2001 From: Leo Dev Date: Fri, 12 Apr 2024 20:23:51 +0200 Subject: [PATCH] optimized brackets --- code.ws | 5 -- src/lexer.rs | 118 +++++++++++++++++++++++++++++++++++++++------- src/parser.rs | 70 ++------------------------- src/transpiler.rs | 2 +- 4 files changed, 104 insertions(+), 91 deletions(-) delete mode 100644 code.ws diff --git a/code.ws b/code.ws deleted file mode 100644 index 85c17bf..0000000 --- a/code.ws +++ /dev/null @@ -1,5 +0,0 @@ -#include - -void main() { - String x = "Hello, wOrld"; -} \ No newline at end of file diff --git a/src/lexer.rs b/src/lexer.rs index a78ac6a..6b0d658 100644 --- a/src/lexer.rs +++ b/src/lexer.rs @@ -1,6 +1,6 @@ use regex::Regex; use once_cell::sync::Lazy; -use std::fmt; +use std::{fmt, os::linux::raw::stat}; pub struct LexerState { pub line: usize, @@ -52,7 +52,7 @@ pub struct Node { token_regex: Lazy } -const SYNTAX: [Node; 15] = [ +const SYNTAX: [Node; 14] = [ Node { token_type: TokenType::Semicolon, token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap()) @@ -65,10 +65,6 @@ const SYNTAX: [Node; 15] = [ token_type: TokenType::String, token_regex: Lazy::new(|| Regex::new("^\"").unwrap()) }, - Node { - token_type: TokenType::Newline, - token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()), - }, Node { token_type: TokenType::Whitespace, token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap()) @@ -128,21 +124,21 @@ fn get_second_char(value: &str) -> String { pub fn lex(mut code: &str, use_whitespace: bool) -> Vec { let mut state = LexerState { line: 1, column: 1 }; let mut tokens: Vec = Vec::new(); - let mut string_vec: Vec = Vec::new(); + let mut bracket_vec: Vec = Vec::new(); let mut inner_string = String::new(); // let mut bracket_state: LexerState = LexerState {line: 0, column: 0}; while !code.is_empty() { let mut is_match = false; - let svl = string_vec.len(); + let svl = bracket_vec.len(); match code.chars().next().unwrap() { '\"' => { is_match = true; code = code.strip_prefix("\"").unwrap_or(code); inner_string += "\""; if svl == 0 { - string_vec.push('\"'); - } else if string_vec[svl-1] == '\"' { - string_vec.pop(); + bracket_vec.push('\"'); + } else if bracket_vec[svl-1] == '\"' { + bracket_vec.pop(); if svl == 1 { tokens.push(Token { token_type:TokenType::String, @@ -159,9 +155,9 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec { code = code.strip_prefix("\'").unwrap_or(code); inner_string += "\'"; if svl == 0 { - string_vec.push('\''); - } else if string_vec[svl-1] == '\'' { - string_vec.pop(); + bracket_vec.push('\''); + } else if bracket_vec[svl-1] == '\'' { + bracket_vec.pop(); if svl == 1 { tokens.push(Token { token_type:TokenType::String, @@ -173,6 +169,96 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec { } } } + '{' => { + is_match = true; + code = code.strip_prefix("{").unwrap_or(code); + inner_string += "{"; + if svl == 0 { + bracket_vec.push('{'); + } + } + '}' => { + is_match = true; + code = code.strip_prefix("}").unwrap_or(code); + println!("{code} 585"); + inner_string += "}"; + if bracket_vec[svl-1] == '{' { + bracket_vec.pop(); + tokens.push(Token { + token_type: TokenType::Curly, + value: inner_string.clone(), + line: state.line, + column: state.column + }); + inner_string = String::new(); + state.column += inner_string.len(); + } + } + '(' => { + is_match = true; + code = code.strip_prefix("(").unwrap_or(code); + inner_string += "("; + if svl == 0 { + bracket_vec.push('('); + } + } + ')' => { + is_match = true; + code = code.strip_prefix(")").unwrap_or(code); + inner_string += ")"; + if bracket_vec[svl-1] == '(' { + bracket_vec.pop(); + tokens.push(Token { + token_type:TokenType::Round, + value: inner_string.clone(), + line: state.line, + column: state.column + }); + inner_string = String::new(); + state.column += inner_string.len(); + } + } + '[' => { + is_match = true; + code = code.strip_prefix("[").unwrap_or(code); + inner_string += "["; + + if svl == 0 { + bracket_vec.push('['); + } + } + ']' => { + is_match = true; + code = code.strip_prefix("]").unwrap_or(code); + inner_string += "]"; + if bracket_vec[svl-1] == '[' { + bracket_vec.pop(); + tokens.push(Token { + token_type:TokenType::Square, + value: inner_string.clone(), + line: state.line, + column: state.column + }); + inner_string = String::new(); + state.column += inner_string.len(); + } + } + '\n' => { + is_match = true; + code = code.strip_prefix("\n").unwrap_or(code); + state.line += 1; + state.column = 1; + if svl > 0 { + inner_string+="\n"; + } else { + tokens.push(Token { + token_type: TokenType::Newline, + value: "\n".to_string(), + line: state.line, + column: state.column + }); + } + } _ => { if svl > 0 { let cfc = get_first_char(&code); @@ -199,10 +285,6 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec { }); } match s.token_type { - TokenType::Newline => { - state.line += caps[0].len(); - state.column = 1; - }, _ => { state.column += caps[0].len(); } diff --git a/src/parser.rs b/src/parser.rs index 45660f4..5b9b937 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -1,4 +1,4 @@ -use crate::lexer::{LexerState, Token, TokenType}; +use crate::lexer::{Token, TokenType}; use std::fmt; use regex::Regex; use once_cell::sync::Lazy; @@ -54,74 +54,10 @@ pub struct Parser { pub include_regex_local: Lazy } -const BRACKETS: [TokenType; 3] = [TokenType::Round, TokenType::Curly, TokenType::Square]; - -pub fn pre_parse (tokens: Vec) -> Vec { - println!("{:?}", tokens); - let mut result: Vec = Vec::new(); - let mut bracket: Vec = Vec::new(); - let mut inner_string = String::new(); - let mut bracket_state: LexerState = LexerState {line: 0, column: 0}; - for token in &tokens { - let is_bracket = BRACKETS.contains(&token.token_type); - match token.value.as_str() { - "(" => { - bracket.push(token.token_type); - inner_string += token.value.as_str(); - } - ")" => { - if bracket[bracket.len()-1] == token.token_type { - bracket.pop(); - inner_string += token.value.as_str(); - bracket_state.line = token.line; - bracket_state.column = token.column; - } - } - "{" => { - bracket.push(token.token_type); - inner_string += token.value.as_str(); - } - "}" => { - if bracket[bracket.len()-1] == token.token_type { - bracket.pop(); - inner_string += token.value.as_str(); - bracket_state.line = token.line; - bracket_state.column = token.column; - } - } - "[" => { - bracket.push(token.token_type); - inner_string += token.value.as_str(); - } - "]" => { - if bracket[bracket.len()-1] == token.token_type { - bracket.pop(); - inner_string += token.value.as_str(); - bracket_state.line = token.line; - bracket_state.column = token.column; - } - } - _ => { - if bracket.len() > 0 { - inner_string += token.value.as_str() - } else if token.token_type != TokenType::Whitespace { - result.push(token.clone()) - } - } - } - if is_bracket && bracket.len() == 0 { - result.push(Token {token_type: token.token_type, value: inner_string.to_string(), line: bracket_state.line, column: bracket_state.column}); - inner_string = String::new(); - } - } - - result -} - impl Parser { pub fn new(tokens: Vec) -> Parser { Parser { - tokens: pre_parse(tokens), + tokens: tokens, index: 0, include_regex: Lazy::new(|| Regex::new(r"^(#include *)<(.*?)>").unwrap()), include_regex_local: Lazy::new(|| Regex::new(r#"^(#include *)"(.*?)""#).unwrap()) @@ -145,7 +81,7 @@ impl Parser { } else { ast_res.ast_type = AstType::FunctionDeceleration; } - self.index += 3; + self.index += 2; } else { loop { if index+1 >= self.tokens.len() {break;} diff --git a/src/transpiler.rs b/src/transpiler.rs index 7ccf96b..f2b70ee 100644 --- a/src/transpiler.rs +++ b/src/transpiler.rs @@ -21,7 +21,7 @@ pub fn transpile(input: String, indent: u32) -> String { input = auto_strip(input); } let mut result = String::new(); - let tokens = lex(input.as_str(), true); + let tokens = lex(input.as_str(), false); println!("\n\n\n\n"); let mut full_ast = Parser::new(tokens.clone()); while full_ast.tokens.len() > full_ast.index as usize {