the lexer now works

This commit is contained in:
2024-04-01 12:06:00 +02:00
parent e909e00144
commit 17b4ee7ef0
2 changed files with 85 additions and 126 deletions
+78 -122
View File
@@ -1,139 +1,95 @@
use regex::Regex; use regex::Regex;
use once_cell::sync::Lazy; use once_cell::sync::Lazy;
use std::fmt;
// Define token types // Define token types
#[derive(Debug, PartialEq)] #[derive(Debug, PartialEq, Clone, Copy)]
pub enum Token { pub enum TokenType {
Newline,
Whitespace,
Number, Number,
Dot, Identifier,
Plus, Ptr,
Minus, Operator,
Multiply, Round,
Divide, Curly,
LParen, // EOF,
RParen,
EOF,
} }
pub struct Token {
pub struct Node<'a> { pub token_type: TokenType,
token_type: Token, pub token_value: String
token_regex: Lazy<Regex>,
token_value: &'a str
} }
const SYNTAX: [Node; 1] = [ impl<'a> fmt::Display for Token {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value)
}
}
pub struct Node {
token_type: TokenType,
token_regex: Lazy<Regex>
}
const SYNTAX: [Node; 8] = [
Node { Node {
token_type: Token::Dot, token_type: TokenType::Newline,
token_regex: Lazy::new(|| Regex::new(r"^test").unwrap()), token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
token_value: "" },
Node {
token_type: TokenType::Whitespace,
token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap())
},
Node {
token_type: TokenType::Number,
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
},
Node {
token_type: TokenType::Identifier,
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
},
Node {
token_type: TokenType::Ptr,
token_regex: Lazy::new(|| Regex::new(r"^->").unwrap())
},
Node {
token_type: TokenType::Operator,
token_regex: Lazy::new(|| Regex::new(r"^[\-|\+|\*]").unwrap())
},
Node {
token_type: TokenType::Round,
token_regex: Lazy::new(|| Regex::new(r"\((?:[^()]|(?R))*\)").unwrap())
},
Node {
token_type: TokenType::Curly,
token_regex: Lazy::new(|| Regex::new(r"^[\{|\}]").unwrap())
} }
]; ];
pub fn lex(code: &str) -> Vec<Node> { pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
let mut tokens: Vec<Node> = Vec::new(); let mut tokens: Vec<Token> = Vec::new();
while (code!="") { while !code.is_empty() {
for s in SYNTAX { let mut is_match = false;
if (s.token_regex.is_match(code)) { for s in &SYNTAX {
println!("It works!"); if let Some(caps) = s.token_regex.captures(code) {
tokens.push(Node { is_match = true;
token_type: s.token_type, code = code.strip_prefix(&caps[0]).unwrap_or(code);
token_regex: s.token_regex, if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
token_value: s.token_regex.captures(wisp) tokens.push(Token {
}); token_type: s.token_type,
} token_value: caps[0].to_string(),
});
}
} else {
continue;
};
break;
}
if !is_match {
println!("Error: syntax -> {code}");
break;
} }
} {
} }
tokens tokens
} }
// Define Lexer struct
// pub struct Lexer<'a> {
// input: &'a str,
// position: usize,
// }
// impl<'a> Lexer<'a> {
// // Constructor
// pub fn new(input: &'a str) -> Self {
// // Lexer { input, position: 0 }
// }
// // Advance position in input
// fn advance(&mut self) {
// self.position += 1;
// }
// // Get current character without advancing position
// fn current_char(&self) -> Option<char> {
// self.input.chars().nth(self.position)
// }
// // Lexical analysis
// pub fn next_token(&mut self) -> Token {
// // Skip whitespace
// while let Some(c) = self.current_char() {
// if c.is_whitespace() {
// self.advance();
// } else {
// break;
// }
// }
// // Check for end of input
// if let None = self.current_char() {
// return Token::EOF;
// }
// // Match characters to tokens
// match self.current_char().unwrap() {
// '+' => {
// self.advance();
// Token::Plus
// }
// '-' => {
// self.advance();
// Token::Minus
// }
// '*' => {
// self.advance();
// Token::Multiply
// }
// '/' => {
// self.advance();
// Token::Divide
// }
// '(' => {
// self.advance();
// Token::LParen
// }
// ')' => {
// self.advance();
// Token::RParen
// }
// '.' => {
// self.advance();
// Token::Dot
// }
// // Match numbers
// digit if digit.is_digit(10) => {
// let mut num_str = String::new();
// while let Some(digit) = self.current_char() {
// if digit.is_digit(10) {
// num_str.push(digit);
// self.advance();
// } else {
// break;
// }
// }
// Token::Number(num_str.parse().unwrap())
// }
// // Unknown character
// _ => {
// panic!("Invalid character: {}", self.current_char().unwrap());
// }
// }
// }
// }
+6 -3
View File
@@ -2,7 +2,10 @@ mod lexer;
use lexer::lex; use lexer::lex;
fn main() { fn main() {
// let input = "-3.232313 + 4 * (10 - 2)"; let input = "int test() {}";
let input = "test"; // let input = "test->xy";
let mut tokens = lex(input); let tokens = lex(input, true);
for token in tokens {
println!("{:?} {}", token.token_type, token.token_value);
}
} }