From cf588ab7e713fa8ce20a0d88c6dd74625c763d99 Mon Sep 17 00:00:00 2001 From: Leo dev Date: Thu, 25 Apr 2024 02:37:07 +0200 Subject: [PATCH] A LOT of bug fixes --- src/lexer.rs | 241 ++++++++++++++++++++-------------------------- src/main.rs | 2 +- src/parser.rs | 38 +++----- src/transpiler.rs | 65 ++++--------- 4 files changed, 140 insertions(+), 206 deletions(-) diff --git a/src/lexer.rs b/src/lexer.rs index 2f8b660..c1e109d 100644 --- a/src/lexer.rs +++ b/src/lexer.rs @@ -115,155 +115,132 @@ fn get_first_char(value: &str) -> String { value.chars().next().unwrap().to_string() } -fn get_second_char(value: &str) -> String { - let mut sv2 = value.chars(); - sv2.next(); - sv2.next().unwrap().to_string() -} - pub fn lex(mut code: &str, use_whitespace: bool) -> Result, (LexerState, Vec)> { let mut state = LexerState { line: 1, column: 1 }; let mut tokens: Vec = Vec::new(); - let mut bracket_vec: Vec = Vec::new(); - let mut inner_string = String::new(); - // let mut bracket_state: LexerState = LexerState {line: 0, column: 0}; + let mut brstr: String = String::new(); + let mut brvct: Vec = Vec::new(); + let mut inside_str: u8 = 0; while !code.is_empty() { let mut is_match = false; - let svl = bracket_vec.len(); - match code.chars().next().unwrap() { - '\"' => { - is_match = true; - code = code.strip_prefix("\"").unwrap_or(code); - inner_string += "\""; - if svl == 0 { - bracket_vec.push('\"'); - } else if bracket_vec[svl-1] == '\"' { - bracket_vec.pop(); - if svl == 1 { + let fch = get_first_char(code); + match fch.as_str() { + "\"" => { + brstr += "\""; + code = code.strip_prefix(fch.as_str()).expect(""); + if inside_str == 1 { + inside_str = 0; + if brvct.len() == 0 { tokens.push(Token { - token_type:TokenType::String, - value: inner_string.clone(), - line: state.line, - column: state.column + token_type: TokenType::String, + value: brstr.clone(), + column: state.line, + line: state.column }); - inner_string = String::new(); + brstr = String::new(); } + } else if inside_str == 0 { + inside_str = 1; } } - '\'' => { - is_match = true; - code = code.strip_prefix("\'").unwrap_or(code); - inner_string += "\'"; - if svl == 0 { - bracket_vec.push('\''); - } else if bracket_vec[svl-1] == '\'' { - bracket_vec.pop(); - if svl == 1 { + "'" => { + brstr += "'"; + code = code.strip_prefix(fch.as_str()).expect(""); + if inside_str == 2 { + inside_str = 0; + if brvct.len() == 0 { tokens.push(Token { - token_type:TokenType::String, - value: inner_string.clone(), - line: state.line, - column: state.column + token_type: TokenType::String, + value: brstr.clone(), + column: state.line, + line: state.column }); - inner_string = String::new(); + brstr = String::new(); } + } else if inside_str == 0 { + inside_str = 2; } } - '{' => { - is_match = true; - code = code.strip_prefix("{").unwrap_or(code); - if svl == 0 { - bracket_vec.push('{'); + "\\" => { + brstr += "\\"; + code = code.strip_prefix(fch.as_str()).expect(""); + if code.len() > 0 { + let sch = get_first_char(code); + code = code.strip_prefix(sch.as_str()).expect(""); + brstr += &sch; } } - '}' => { - is_match = true; - code = code.strip_prefix("}").unwrap_or(code); - if bracket_vec[svl-1] == '{' { - bracket_vec.pop(); + "{" => { + code = code.strip_prefix(fch.as_str()).expect(""); + if brvct.len() > 0 { + brstr += "{"; + } + brvct.push(0); + } + "}" => { + code = code.strip_prefix(fch.as_str()).expect(""); + brvct.pop(); + if brvct.len() == 0 { tokens.push(Token { token_type: TokenType::Curly, - value: inner_string.clone(), - line: state.line, - column: state.column + value: brstr.clone(), + column: state.line, + line: state.column }); - inner_string = String::new(); - state.column += inner_string.len(); - } - } - '(' => { - is_match = true; - code = code.strip_prefix("(").unwrap_or(code); - if svl == 0 { - bracket_vec.push('('); - } - } - ')' => { - is_match = true; - code = code.strip_prefix(")").unwrap_or(code); - if bracket_vec[svl-1] == '(' { - bracket_vec.pop(); - tokens.push(Token { - token_type:TokenType::Round, - value: inner_string.clone(), - line: state.line, - column: state.column - }); - inner_string = String::new(); - state.column += inner_string.len(); - } - } - '[' => { - is_match = true; - code = code.strip_prefix("[").unwrap_or(code); - - if svl == 0 { - bracket_vec.push('['); - } - } - ']' => { - is_match = true; - code = code.strip_prefix("]").unwrap_or(code); - if bracket_vec[svl-1] == '[' { - bracket_vec.pop(); - tokens.push(Token { - token_type:TokenType::Square, - value: inner_string.clone(), - line: state.line, - column: state.column - }); - inner_string = String::new(); - state.column += inner_string.len(); - } - } - '\n' => { - is_match = true; - code = code.strip_prefix("\n").unwrap_or(code); - state.line += 1; - state.column = 1; - if svl > 0 { - inner_string+="\n"; } else { - tokens.push(Token { - token_type: TokenType::Newline, - value: "\n".to_string(), - line: state.line, - column: state.column - }); + brstr += "}"; } } + "(" => { + code = code.strip_prefix(fch.as_str()).expect(""); + if brvct.len() > 0 { + brstr += "("; + } + brvct.push(1); + } + ")" => { + code = code.strip_prefix(fch.as_str()).expect(""); + brvct.pop(); + if brvct.len() == 0 { + tokens.push(Token { + token_type: TokenType::Round, + value: brstr.clone(), + column: state.line, + line: state.column + }); + } else { + brstr += ")"; + } + } + "[" => { + code = code.strip_prefix(fch.as_str()).expect(""); + if brvct.len() > 0 { + brstr += "["; + } + brvct.push(2); + } + "]" => { + code = code.strip_prefix(fch.as_str()).expect(""); + brvct.pop(); + if brvct.len() == 0 { + tokens.push(Token { + token_type: TokenType::Square, + value: brstr.clone(), + column: state.line, + line: state.column + }); + } else { + brstr += "]"; + } + } + "\n" => { + code = code.strip_prefix(fch.as_str()).expect(""); + state.line += 1; + } _ => { - if svl > 0 { - let cfc = get_first_char(&code); - let cfc1 = get_second_char(&code); - inner_string += cfc.as_str(); - code = code.strip_prefix(&cfc).unwrap_or(code); - is_match = true; - if cfc == "\\" && (cfc1 == "\"" || cfc1 == "'") { - inner_string += cfc1.as_str(); - code = code.strip_prefix(&cfc1).unwrap_or(code); - is_match = true; - } + if inside_str > 0 || brvct.len() > 0 { + code = code.strip_prefix(fch.as_str()).expect(""); + brstr += fch.as_str(); } else { for s in &SYNTAX { if let Some(caps) = s.token_regex.captures(code) { @@ -277,24 +254,18 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Result, (LexerSta column: state.column }); } - match s.token_type { - _ => { - state.column += caps[0].len(); - } - } + state.column += caps[0].len(); } else { continue; }; break; } + if !is_match { + return Err((state, tokens)); + } } - } } - if !is_match { - return Err((state, tokens)); - // break; - } } Ok(tokens) } diff --git a/src/main.rs b/src/main.rs index 9a7eb99..e6f310f 100644 --- a/src/main.rs +++ b/src/main.rs @@ -5,7 +5,7 @@ use std::fs; fn main() { - let code_file = "code.ws"; + let code_file = "code.wst"; let contents = fs::read_to_string(code_file).expect("cannot read for some reason"); let result = transpiler::transpile(contents, 0); println!("{result}") diff --git a/src/parser.rs b/src/parser.rs index 5b9b937..75b7341 100644 --- a/src/parser.rs +++ b/src/parser.rs @@ -11,6 +11,7 @@ pub enum AstType { MutVariableDeceleration, Include, IncludeLocal, + CodeBlock, Other } @@ -65,13 +66,12 @@ impl Parser { } pub fn next(&mut self) -> Ast { let mut ast_res: Ast = Ast {tokens: vec![], ast_type: AstType::Other}; - let mut index = self.index as usize; + let index = self.index as usize; if index == self.tokens.len() {panic!("Reached the end of tokens")} let token = &self.tokens[index]; match token.token_type { TokenType::Identifier => { ast_res.tokens.push(self.tokens[index].clone()); - self.index += 1; if self.tokens.len()-index > 3 && self.tokens[index+1].token_type==TokenType::Identifier && self.tokens[index+2].token_type==TokenType::Round && self.tokens[index+3].token_type==TokenType::Curly { ast_res.tokens.push(self.tokens[index+1].clone()); ast_res.tokens.push(self.tokens[index+2].clone()); @@ -81,29 +81,19 @@ impl Parser { } else { ast_res.ast_type = AstType::FunctionDeceleration; } + self.index += 3; + } else if self.tokens.len()-index > 1 && token.value == "cb" && self.tokens[index+1].token_type == TokenType::Curly { + ast_res.tokens.push(self.tokens[index+1].clone()); + ast_res.ast_type = AstType::CodeBlock; + self.index += 1; + } else if self.tokens[index+1].token_type==TokenType::Identifier { + ast_res.tokens.push(self.tokens[index+1].clone()); + ast_res.ast_type = AstType::VariableDeceleration; + self.index += 1; + } else if self.tokens[index+1].value=="mut" && self.tokens[index+2].token_type==TokenType::Identifier { + ast_res.tokens.push(self.tokens[index+2].clone()); + ast_res.ast_type = AstType::MutVariableDeceleration; self.index += 2; - } else { - loop { - if index+1 >= self.tokens.len() {break;} - index = self.index as usize; // Update the index - let ntk = &self.tokens[index]; // Stands for next token - if ntk.value=="," { - ast_res.tokens.push(ntk.clone()); - self.index += 1; - } else if ntk.token_type==TokenType::Identifier { - ast_res.tokens.push(ntk.clone()); - if ast_res.ast_type != AstType::VariableDeceleration && ast_res.ast_type != AstType::MutVariableDeceleration { - ast_res.ast_type = AstType::VariableDeceleration; - } - self.index += 1; - } else if ntk.value=="mut" { - ast_res.ast_type = AstType::MutVariableDeceleration; - self.index += 1; - } else { - self.index -= 1; - break; - } - } } } TokenType::Include => { diff --git a/src/transpiler.rs b/src/transpiler.rs index 35edb42..a65a79d 100644 --- a/src/transpiler.rs +++ b/src/transpiler.rs @@ -3,69 +3,42 @@ use crate::parser::{Parser, AstType}; pub fn transpile(input: String, indent: u32) -> String { let mut result = String::new(); + if indent > 0 { + result += " ".repeat((indent as usize)*2).as_str(); + } let lexer_out = lex(input.as_str(), false); match lexer_out { Ok(tokens) => { - println!("\n\n\n\n"); let mut full_ast = Parser::new(tokens.clone()); while full_ast.tokens.len() > full_ast.index as usize { let ast = full_ast.next(); println!("{ast}"); if ast.ast_type == AstType::FunctionDeceleration { - result += format!("fn {}{} -> {} {}", ast.tokens[1].value, ast.tokens[2].value, ast.tokens[0].value, transpile(ast.tokens[3].value.clone(), indent+1)).as_str(); + result += format!("fn {}({}) -> {} {}", ast.tokens[1].value, ast.tokens[2].value, ast.tokens[0].value, transpile(ast.tokens[3].value.clone(), indent+1)).as_str(); } else if ast.ast_type == AstType::VoidFunctionDeceleration { - result += format!("fn {}{} {}", ast.tokens[1].value, ast.tokens[2].value, transpile(ast.tokens[3].value.clone(), indent+1)).as_str(); + result += format!("fn {}({}) {}", ast.tokens[1].value, ast.tokens[2].value, transpile(ast.tokens[3].value.clone(), indent+1)).as_str(); } else if ast.ast_type == AstType::VariableDeceleration { - let _o = String::new(); - let mut oy = ast.tokens.clone(); - oy.remove(0); - if oy.len() > 1 { - result += "("; - for t in &oy { - result+=t.value.as_str() - } - result += "): ("; - for t in oy { - if t.value == "," { - result+=", " - } else { - result+=ast.tokens[0].value.as_str() - } - } - result += ")"; - } else { - result += format!("let {}: {}", ast.tokens[1].value, ast.tokens[0].value).as_str(); - } + result += format!("let {}: {}", ast.tokens[1].value, ast.tokens[0].value).as_str(); } else if ast.ast_type == AstType::MutVariableDeceleration { - let _o = String::new(); - let mut oy = ast.tokens.clone(); - oy.remove(0); - if oy.len() > 1 { - result += "("; - for t in &oy { - result+=t.value.as_str() - } - result += "): ("; - for t in oy { - if t.value == "," { - result+=", " - } else { - result+=ast.tokens[0].value.as_str() - } - } - result += ")"; - } else { - result += format!("let mut {}: {}", ast.tokens[1].value, ast.tokens[0].value).as_str(); - } + result += format!("let mut {}: {}", ast.tokens[1].value, ast.tokens[0].value).as_str(); } else if ast.tokens.len() == 1 && ast.tokens[0].token_type == TokenType::Round { result += format!("({})", ast.tokens[0].value).as_str(); - } else { + } else if ast.tokens.len() == 1 && ast.tokens[0].token_type == TokenType::Curly { if ast.tokens[0].token_type==TokenType::Newline { result += (ast.tokens[0].value.as_str().to_owned() + (" ".repeat((indent as usize)*2).as_str())).as_str() } else { result += ast.tokens[0].value.as_str() } + } else { + if ast.tokens[0].token_type==TokenType::Newline { + result += (ast.tokens[0].value.as_str().to_owned() + (" ".repeat((indent as usize)*2).as_str())).as_str() + } else if ast.tokens[0].token_type==TokenType::Semicolon { + result += ";\n"; + result += " ".repeat((indent as usize)*2).as_str(); + } else { + result += ast.tokens[0].value.as_str() + } } } @@ -73,13 +46,13 @@ pub fn transpile(input: String, indent: u32) -> String { if indent > 0 { result += "\n"; result += " ".repeat((indent as usize-1)*2).as_str(); - return "{".to_owned()+result.as_str()+"}" + return "{\n".to_owned()+result.as_str()+"}" } else { return result } }, Err((state, _tokens)) => { - println!("Invalid character at code.ws:{}:{}", state.line, state.column); + println!("Invalid syntax at code.ws:{}:{}", state.line, state.column); return "".to_string(); } }