even more bug fixes
This commit is contained in:
+113
-26
@@ -22,7 +22,8 @@ pub enum TokenType {
|
|||||||
Round,
|
Round,
|
||||||
Curly,
|
Curly,
|
||||||
Square,
|
Square,
|
||||||
Angle,
|
Include,
|
||||||
|
String,
|
||||||
// EOF,
|
// EOF,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -51,7 +52,7 @@ pub struct Node {
|
|||||||
token_regex: Lazy<Regex>
|
token_regex: Lazy<Regex>
|
||||||
}
|
}
|
||||||
|
|
||||||
const SYNTAX: [Node; 12] = [
|
const SYNTAX: [Node; 15] = [
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Semicolon,
|
token_type: TokenType::Semicolon,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap())
|
||||||
@@ -60,6 +61,10 @@ const SYNTAX: [Node; 12] = [
|
|||||||
token_type: TokenType::SecondOperator,
|
token_type: TokenType::SecondOperator,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^,").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^,").unwrap())
|
||||||
},
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::String,
|
||||||
|
token_regex: Lazy::new(|| Regex::new("^\"").unwrap())
|
||||||
|
},
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Newline,
|
token_type: TokenType::Newline,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()),
|
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()),
|
||||||
@@ -86,7 +91,7 @@ const SYNTAX: [Node; 12] = [
|
|||||||
},
|
},
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Operator,
|
token_type: TokenType::Operator,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^[\-|\+|\*|\=|\!]").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^[|\-|\+|\*|\=|\!]").unwrap())
|
||||||
},
|
},
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Round,
|
token_type: TokenType::Round,
|
||||||
@@ -100,41 +105,123 @@ const SYNTAX: [Node; 12] = [
|
|||||||
token_type: TokenType::Square,
|
token_type: TokenType::Square,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^[\[|\]]").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^[\[|\]]").unwrap())
|
||||||
},
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Include,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^#include *<(.*?)>").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Include,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r#"^#include *"(.*?)""#).unwrap())
|
||||||
|
}
|
||||||
];
|
];
|
||||||
|
|
||||||
|
fn remove_first_char(value: &str) -> String {
|
||||||
|
value.chars().skip(1).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_first_char(value: &str) -> String {
|
||||||
|
value.chars().next().unwrap().to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_second_char(value: &str) -> String {
|
||||||
|
let mut sv2 = value.chars();
|
||||||
|
sv2.next();
|
||||||
|
sv2.next().unwrap().to_string()
|
||||||
|
}
|
||||||
|
|
||||||
pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
||||||
let mut state = LexerState { line: 1, column: 1 };
|
let mut state = LexerState { line: 1, column: 1 };
|
||||||
let mut tokens: Vec<Token> = Vec::new();
|
let mut tokens: Vec<Token> = Vec::new();
|
||||||
|
let mut string_vec: Vec<char> = Vec::new();
|
||||||
|
let mut inner_string = String::new();
|
||||||
|
// let mut bracket_state: LexerState = LexerState {line: 0, column: 0};
|
||||||
while !code.is_empty() {
|
while !code.is_empty() {
|
||||||
let mut is_match = false;
|
let mut is_match = false;
|
||||||
for s in &SYNTAX {
|
let svl = string_vec.len();
|
||||||
if let Some(caps) = s.token_regex.captures(code) {
|
match code.chars().next().unwrap() {
|
||||||
|
'\"' => {
|
||||||
is_match = true;
|
is_match = true;
|
||||||
code = code.strip_prefix(&caps[0]).unwrap_or(code);
|
code = code.strip_prefix("\"").unwrap_or(code);
|
||||||
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
|
inner_string += "\"";
|
||||||
tokens.push(Token {
|
if svl == 0 {
|
||||||
token_type: s.token_type,
|
string_vec.push('\"');
|
||||||
value: caps[0].to_string(),
|
} else if string_vec[svl-1] == '\"' {
|
||||||
line: state.line,
|
string_vec.pop();
|
||||||
column: state.column
|
if svl == 1 {
|
||||||
});
|
tokens.push(Token {
|
||||||
|
token_type:TokenType::String,
|
||||||
|
value: inner_string.clone(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
inner_string = String::new();
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
match s.token_type {
|
'\'' => {
|
||||||
TokenType::Newline => {
|
is_match = true;
|
||||||
state.line += caps[0].len();
|
code = code.strip_prefix("\'").unwrap_or(code);
|
||||||
state.column = 1;
|
inner_string += "\'";
|
||||||
},
|
if svl == 0 {
|
||||||
_ => { state.column += caps[0].len(); }
|
string_vec.push('\'');
|
||||||
|
} else if string_vec[svl-1] == '\'' {
|
||||||
|
string_vec.pop();
|
||||||
|
if svl == 1 {
|
||||||
|
tokens.push(Token {
|
||||||
|
token_type:TokenType::String,
|
||||||
|
value: inner_string.clone(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
inner_string = String::new();
|
||||||
|
}
|
||||||
}
|
}
|
||||||
} else {
|
}
|
||||||
continue;
|
_ => {
|
||||||
};
|
if svl > 0 {
|
||||||
break;
|
let cfc = get_first_char(&code);
|
||||||
|
let cfc1 = get_second_char(&code);
|
||||||
|
inner_string += cfc.as_str();
|
||||||
|
code = code.strip_prefix(&cfc).unwrap_or(code);
|
||||||
|
is_match = true;
|
||||||
|
if cfc == "\\" && (cfc1 == "\"" || cfc1 == "'") {
|
||||||
|
inner_string += cfc1.as_str();
|
||||||
|
code = code.strip_prefix(&cfc1).unwrap_or(code);
|
||||||
|
is_match = true;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
for s in &SYNTAX {
|
||||||
|
if let Some(caps) = s.token_regex.captures(code) {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix(&caps[0]).unwrap_or(code);
|
||||||
|
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
|
||||||
|
tokens.push(Token {
|
||||||
|
token_type: s.token_type,
|
||||||
|
value: caps[0].to_string(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
}
|
||||||
|
match s.token_type {
|
||||||
|
TokenType::Newline => {
|
||||||
|
state.line += caps[0].len();
|
||||||
|
state.column = 1;
|
||||||
|
},
|
||||||
|
_ => {
|
||||||
|
state.column += caps[0].len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
}
|
}
|
||||||
if !is_match {
|
if !is_match {
|
||||||
println!("Error: syntax -> {code}");
|
println!("Error: syntax ->{code}");
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+22
-2
@@ -1,5 +1,7 @@
|
|||||||
use crate::lexer::{LexerState, Token, TokenType};
|
use crate::lexer::{LexerState, Token, TokenType};
|
||||||
use std::fmt;
|
use std::fmt;
|
||||||
|
use regex::Regex;
|
||||||
|
use once_cell::sync::Lazy;
|
||||||
|
|
||||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||||
pub enum AstType {
|
pub enum AstType {
|
||||||
@@ -7,6 +9,7 @@ pub enum AstType {
|
|||||||
VoidFunctionDeceleration,
|
VoidFunctionDeceleration,
|
||||||
VariableDeceleration,
|
VariableDeceleration,
|
||||||
MutVariableDeceleration,
|
MutVariableDeceleration,
|
||||||
|
Include,
|
||||||
Other
|
Other
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -45,12 +48,15 @@ impl fmt::Debug for Ast {
|
|||||||
|
|
||||||
pub struct Parser {
|
pub struct Parser {
|
||||||
pub tokens: Vec<Token>,
|
pub tokens: Vec<Token>,
|
||||||
pub index: u32
|
pub index: u32,
|
||||||
|
pub include_regex: Lazy<Regex>,
|
||||||
|
pub include_regex_local: Lazy<Regex>
|
||||||
}
|
}
|
||||||
|
|
||||||
const BRACKETS: [TokenType; 3] = [TokenType::Round, TokenType::Curly, TokenType::Square];
|
const BRACKETS: [TokenType; 3] = [TokenType::Round, TokenType::Curly, TokenType::Square];
|
||||||
|
|
||||||
pub fn pre_parse (tokens: Vec<Token>) -> Vec<Token> {
|
pub fn pre_parse (tokens: Vec<Token>) -> Vec<Token> {
|
||||||
|
println!("{:?}", tokens);
|
||||||
let mut result: Vec<Token> = Vec::new();
|
let mut result: Vec<Token> = Vec::new();
|
||||||
let mut bracket: Vec<TokenType> = Vec::new();
|
let mut bracket: Vec<TokenType> = Vec::new();
|
||||||
let mut inner_string = String::new();
|
let mut inner_string = String::new();
|
||||||
@@ -113,7 +119,12 @@ pub fn pre_parse (tokens: Vec<Token>) -> Vec<Token> {
|
|||||||
|
|
||||||
impl Parser {
|
impl Parser {
|
||||||
pub fn new(tokens: Vec<Token>) -> Parser {
|
pub fn new(tokens: Vec<Token>) -> Parser {
|
||||||
Parser {tokens: pre_parse(tokens), index: 0}
|
Parser {
|
||||||
|
tokens: pre_parse(tokens),
|
||||||
|
index: 0,
|
||||||
|
include_regex: Lazy::new(|| Regex::new(r"^(#include *)<(.*?)>").unwrap()),
|
||||||
|
include_regex_local: Lazy::new(|| Regex::new(r#"^(#include *)"(.*?)""#).unwrap())
|
||||||
|
}
|
||||||
}
|
}
|
||||||
pub fn next(&mut self) -> Ast {
|
pub fn next(&mut self) -> Ast {
|
||||||
let mut ast_res: Ast = Ast {tokens: vec![], ast_type: AstType::Other};
|
let mut ast_res: Ast = Ast {tokens: vec![], ast_type: AstType::Other};
|
||||||
@@ -158,6 +169,15 @@ impl Parser {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
TokenType::Include => {
|
||||||
|
if let Some(caps) = self.include_regex.captures(&token.value) {
|
||||||
|
println!("bruh: {}", &caps[1]);
|
||||||
|
} else if let Some(caps) = self.include_regex_local.captures(&token.value) {
|
||||||
|
// ast_res.tokens.push(Token {token_type: TokenType::, value: &caps[1], line: 0, column: 0});
|
||||||
|
} else {
|
||||||
|
ast_res.tokens.push(token.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
_ => {
|
_ => {
|
||||||
ast_res.tokens.push(token.clone());
|
ast_res.tokens.push(token.clone());
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user