optimized brackets
This commit is contained in:
+100
-18
@@ -1,6 +1,6 @@
|
|||||||
use regex::Regex;
|
use regex::Regex;
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
use std::fmt;
|
use std::{fmt, os::linux::raw::stat};
|
||||||
|
|
||||||
pub struct LexerState {
|
pub struct LexerState {
|
||||||
pub line: usize,
|
pub line: usize,
|
||||||
@@ -52,7 +52,7 @@ pub struct Node {
|
|||||||
token_regex: Lazy<Regex>
|
token_regex: Lazy<Regex>
|
||||||
}
|
}
|
||||||
|
|
||||||
const SYNTAX: [Node; 15] = [
|
const SYNTAX: [Node; 14] = [
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Semicolon,
|
token_type: TokenType::Semicolon,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^\;").unwrap())
|
||||||
@@ -65,10 +65,6 @@ const SYNTAX: [Node; 15] = [
|
|||||||
token_type: TokenType::String,
|
token_type: TokenType::String,
|
||||||
token_regex: Lazy::new(|| Regex::new("^\"").unwrap())
|
token_regex: Lazy::new(|| Regex::new("^\"").unwrap())
|
||||||
},
|
},
|
||||||
Node {
|
|
||||||
token_type: TokenType::Newline,
|
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()),
|
|
||||||
},
|
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Whitespace,
|
token_type: TokenType::Whitespace,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap())
|
||||||
@@ -128,21 +124,21 @@ fn get_second_char(value: &str) -> String {
|
|||||||
pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
||||||
let mut state = LexerState { line: 1, column: 1 };
|
let mut state = LexerState { line: 1, column: 1 };
|
||||||
let mut tokens: Vec<Token> = Vec::new();
|
let mut tokens: Vec<Token> = Vec::new();
|
||||||
let mut string_vec: Vec<char> = Vec::new();
|
let mut bracket_vec: Vec<char> = Vec::new();
|
||||||
let mut inner_string = String::new();
|
let mut inner_string = String::new();
|
||||||
// let mut bracket_state: LexerState = LexerState {line: 0, column: 0};
|
// let mut bracket_state: LexerState = LexerState {line: 0, column: 0};
|
||||||
while !code.is_empty() {
|
while !code.is_empty() {
|
||||||
let mut is_match = false;
|
let mut is_match = false;
|
||||||
let svl = string_vec.len();
|
let svl = bracket_vec.len();
|
||||||
match code.chars().next().unwrap() {
|
match code.chars().next().unwrap() {
|
||||||
'\"' => {
|
'\"' => {
|
||||||
is_match = true;
|
is_match = true;
|
||||||
code = code.strip_prefix("\"").unwrap_or(code);
|
code = code.strip_prefix("\"").unwrap_or(code);
|
||||||
inner_string += "\"";
|
inner_string += "\"";
|
||||||
if svl == 0 {
|
if svl == 0 {
|
||||||
string_vec.push('\"');
|
bracket_vec.push('\"');
|
||||||
} else if string_vec[svl-1] == '\"' {
|
} else if bracket_vec[svl-1] == '\"' {
|
||||||
string_vec.pop();
|
bracket_vec.pop();
|
||||||
if svl == 1 {
|
if svl == 1 {
|
||||||
tokens.push(Token {
|
tokens.push(Token {
|
||||||
token_type:TokenType::String,
|
token_type:TokenType::String,
|
||||||
@@ -159,9 +155,9 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
|||||||
code = code.strip_prefix("\'").unwrap_or(code);
|
code = code.strip_prefix("\'").unwrap_or(code);
|
||||||
inner_string += "\'";
|
inner_string += "\'";
|
||||||
if svl == 0 {
|
if svl == 0 {
|
||||||
string_vec.push('\'');
|
bracket_vec.push('\'');
|
||||||
} else if string_vec[svl-1] == '\'' {
|
} else if bracket_vec[svl-1] == '\'' {
|
||||||
string_vec.pop();
|
bracket_vec.pop();
|
||||||
if svl == 1 {
|
if svl == 1 {
|
||||||
tokens.push(Token {
|
tokens.push(Token {
|
||||||
token_type:TokenType::String,
|
token_type:TokenType::String,
|
||||||
@@ -173,6 +169,96 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
'{' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix("{").unwrap_or(code);
|
||||||
|
inner_string += "{";
|
||||||
|
if svl == 0 {
|
||||||
|
bracket_vec.push('{');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'}' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix("}").unwrap_or(code);
|
||||||
|
println!("{code} 585");
|
||||||
|
inner_string += "}";
|
||||||
|
if bracket_vec[svl-1] == '{' {
|
||||||
|
bracket_vec.pop();
|
||||||
|
tokens.push(Token {
|
||||||
|
token_type: TokenType::Curly,
|
||||||
|
value: inner_string.clone(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
inner_string = String::new();
|
||||||
|
state.column += inner_string.len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'(' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix("(").unwrap_or(code);
|
||||||
|
inner_string += "(";
|
||||||
|
if svl == 0 {
|
||||||
|
bracket_vec.push('(');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
')' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix(")").unwrap_or(code);
|
||||||
|
inner_string += ")";
|
||||||
|
if bracket_vec[svl-1] == '(' {
|
||||||
|
bracket_vec.pop();
|
||||||
|
tokens.push(Token {
|
||||||
|
token_type:TokenType::Round,
|
||||||
|
value: inner_string.clone(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
inner_string = String::new();
|
||||||
|
state.column += inner_string.len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'[' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix("[").unwrap_or(code);
|
||||||
|
inner_string += "[";
|
||||||
|
|
||||||
|
if svl == 0 {
|
||||||
|
bracket_vec.push('[');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
']' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix("]").unwrap_or(code);
|
||||||
|
inner_string += "]";
|
||||||
|
if bracket_vec[svl-1] == '[' {
|
||||||
|
bracket_vec.pop();
|
||||||
|
tokens.push(Token {
|
||||||
|
token_type:TokenType::Square,
|
||||||
|
value: inner_string.clone(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
inner_string = String::new();
|
||||||
|
state.column += inner_string.len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'\n' => {
|
||||||
|
is_match = true;
|
||||||
|
code = code.strip_prefix("\n").unwrap_or(code);
|
||||||
|
state.line += 1;
|
||||||
|
state.column = 1;
|
||||||
|
if svl > 0 {
|
||||||
|
inner_string+="\n";
|
||||||
|
} else {
|
||||||
|
tokens.push(Token {
|
||||||
|
token_type: TokenType::Newline,
|
||||||
|
value: "\n".to_string(),
|
||||||
|
line: state.line,
|
||||||
|
column: state.column
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
_ => {
|
_ => {
|
||||||
if svl > 0 {
|
if svl > 0 {
|
||||||
let cfc = get_first_char(&code);
|
let cfc = get_first_char(&code);
|
||||||
@@ -199,10 +285,6 @@ pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
match s.token_type {
|
match s.token_type {
|
||||||
TokenType::Newline => {
|
|
||||||
state.line += caps[0].len();
|
|
||||||
state.column = 1;
|
|
||||||
},
|
|
||||||
_ => {
|
_ => {
|
||||||
state.column += caps[0].len();
|
state.column += caps[0].len();
|
||||||
}
|
}
|
||||||
|
|||||||
+3
-67
@@ -1,4 +1,4 @@
|
|||||||
use crate::lexer::{LexerState, Token, TokenType};
|
use crate::lexer::{Token, TokenType};
|
||||||
use std::fmt;
|
use std::fmt;
|
||||||
use regex::Regex;
|
use regex::Regex;
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
@@ -54,74 +54,10 @@ pub struct Parser {
|
|||||||
pub include_regex_local: Lazy<Regex>
|
pub include_regex_local: Lazy<Regex>
|
||||||
}
|
}
|
||||||
|
|
||||||
const BRACKETS: [TokenType; 3] = [TokenType::Round, TokenType::Curly, TokenType::Square];
|
|
||||||
|
|
||||||
pub fn pre_parse (tokens: Vec<Token>) -> Vec<Token> {
|
|
||||||
println!("{:?}", tokens);
|
|
||||||
let mut result: Vec<Token> = Vec::new();
|
|
||||||
let mut bracket: Vec<TokenType> = Vec::new();
|
|
||||||
let mut inner_string = String::new();
|
|
||||||
let mut bracket_state: LexerState = LexerState {line: 0, column: 0};
|
|
||||||
for token in &tokens {
|
|
||||||
let is_bracket = BRACKETS.contains(&token.token_type);
|
|
||||||
match token.value.as_str() {
|
|
||||||
"(" => {
|
|
||||||
bracket.push(token.token_type);
|
|
||||||
inner_string += token.value.as_str();
|
|
||||||
}
|
|
||||||
")" => {
|
|
||||||
if bracket[bracket.len()-1] == token.token_type {
|
|
||||||
bracket.pop();
|
|
||||||
inner_string += token.value.as_str();
|
|
||||||
bracket_state.line = token.line;
|
|
||||||
bracket_state.column = token.column;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
"{" => {
|
|
||||||
bracket.push(token.token_type);
|
|
||||||
inner_string += token.value.as_str();
|
|
||||||
}
|
|
||||||
"}" => {
|
|
||||||
if bracket[bracket.len()-1] == token.token_type {
|
|
||||||
bracket.pop();
|
|
||||||
inner_string += token.value.as_str();
|
|
||||||
bracket_state.line = token.line;
|
|
||||||
bracket_state.column = token.column;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
"[" => {
|
|
||||||
bracket.push(token.token_type);
|
|
||||||
inner_string += token.value.as_str();
|
|
||||||
}
|
|
||||||
"]" => {
|
|
||||||
if bracket[bracket.len()-1] == token.token_type {
|
|
||||||
bracket.pop();
|
|
||||||
inner_string += token.value.as_str();
|
|
||||||
bracket_state.line = token.line;
|
|
||||||
bracket_state.column = token.column;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
_ => {
|
|
||||||
if bracket.len() > 0 {
|
|
||||||
inner_string += token.value.as_str()
|
|
||||||
} else if token.token_type != TokenType::Whitespace {
|
|
||||||
result.push(token.clone())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if is_bracket && bracket.len() == 0 {
|
|
||||||
result.push(Token {token_type: token.token_type, value: inner_string.to_string(), line: bracket_state.line, column: bracket_state.column});
|
|
||||||
inner_string = String::new();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
result
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Parser {
|
impl Parser {
|
||||||
pub fn new(tokens: Vec<Token>) -> Parser {
|
pub fn new(tokens: Vec<Token>) -> Parser {
|
||||||
Parser {
|
Parser {
|
||||||
tokens: pre_parse(tokens),
|
tokens: tokens,
|
||||||
index: 0,
|
index: 0,
|
||||||
include_regex: Lazy::new(|| Regex::new(r"^(#include *)<(.*?)>").unwrap()),
|
include_regex: Lazy::new(|| Regex::new(r"^(#include *)<(.*?)>").unwrap()),
|
||||||
include_regex_local: Lazy::new(|| Regex::new(r#"^(#include *)"(.*?)""#).unwrap())
|
include_regex_local: Lazy::new(|| Regex::new(r#"^(#include *)"(.*?)""#).unwrap())
|
||||||
@@ -145,7 +81,7 @@ impl Parser {
|
|||||||
} else {
|
} else {
|
||||||
ast_res.ast_type = AstType::FunctionDeceleration;
|
ast_res.ast_type = AstType::FunctionDeceleration;
|
||||||
}
|
}
|
||||||
self.index += 3;
|
self.index += 2;
|
||||||
} else {
|
} else {
|
||||||
loop {
|
loop {
|
||||||
if index+1 >= self.tokens.len() {break;}
|
if index+1 >= self.tokens.len() {break;}
|
||||||
|
|||||||
+1
-1
@@ -21,7 +21,7 @@ pub fn transpile(input: String, indent: u32) -> String {
|
|||||||
input = auto_strip(input);
|
input = auto_strip(input);
|
||||||
}
|
}
|
||||||
let mut result = String::new();
|
let mut result = String::new();
|
||||||
let tokens = lex(input.as_str(), true);
|
let tokens = lex(input.as_str(), false);
|
||||||
println!("\n\n\n\n");
|
println!("\n\n\n\n");
|
||||||
let mut full_ast = Parser::new(tokens.clone());
|
let mut full_ast = Parser::new(tokens.clone());
|
||||||
while full_ast.tokens.len() > full_ast.index as usize {
|
while full_ast.tokens.len() > full_ast.index as usize {
|
||||||
|
|||||||
Reference in New Issue
Block a user