the lexer now works

This commit is contained in:
2024-04-01 12:06:00 +02:00
parent e909e00144
commit 17b4ee7ef0
2 changed files with 85 additions and 126 deletions
+79 -123
View File
@@ -1,139 +1,95 @@
use regex::Regex;
use once_cell::sync::Lazy;
use std::fmt;
// Define token types
#[derive(Debug, PartialEq)]
pub enum Token {
#[derive(Debug, PartialEq, Clone, Copy)]
pub enum TokenType {
Newline,
Whitespace,
Number,
Dot,
Plus,
Minus,
Multiply,
Divide,
LParen,
RParen,
EOF,
Identifier,
Ptr,
Operator,
Round,
Curly,
// EOF,
}
pub struct Node<'a> {
token_type: Token,
token_regex: Lazy<Regex>,
token_value: &'a str
pub struct Token {
pub token_type: TokenType,
pub token_value: String
}
const SYNTAX: [Node; 1] = [
impl<'a> fmt::Display for Token {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value)
}
}
pub struct Node {
token_type: TokenType,
token_regex: Lazy<Regex>
}
const SYNTAX: [Node; 8] = [
Node {
token_type: Token::Dot,
token_regex: Lazy::new(|| Regex::new(r"^test").unwrap()),
token_value: ""
token_type: TokenType::Newline,
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
},
Node {
token_type: TokenType::Whitespace,
token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap())
},
Node {
token_type: TokenType::Number,
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
},
Node {
token_type: TokenType::Identifier,
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
},
Node {
token_type: TokenType::Ptr,
token_regex: Lazy::new(|| Regex::new(r"^->").unwrap())
},
Node {
token_type: TokenType::Operator,
token_regex: Lazy::new(|| Regex::new(r"^[\-|\+|\*]").unwrap())
},
Node {
token_type: TokenType::Round,
token_regex: Lazy::new(|| Regex::new(r"\((?:[^()]|(?R))*\)").unwrap())
},
Node {
token_type: TokenType::Curly,
token_regex: Lazy::new(|| Regex::new(r"^[\{|\}]").unwrap())
}
];
pub fn lex(code: &str) -> Vec<Node> {
let mut tokens: Vec<Node> = Vec::new();
while (code!="") {
for s in SYNTAX {
if (s.token_regex.is_match(code)) {
println!("It works!");
tokens.push(Node {
token_type: s.token_type,
token_regex: s.token_regex,
token_value: s.token_regex.captures(wisp)
});
}
pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
let mut tokens: Vec<Token> = Vec::new();
while !code.is_empty() {
let mut is_match = false;
for s in &SYNTAX {
if let Some(caps) = s.token_regex.captures(code) {
is_match = true;
code = code.strip_prefix(&caps[0]).unwrap_or(code);
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
tokens.push(Token {
token_type: s.token_type,
token_value: caps[0].to_string(),
});
}
} else {
continue;
};
break;
}
if !is_match {
println!("Error: syntax -> {code}");
break;
}
} {
}
tokens
}
// Define Lexer struct
// pub struct Lexer<'a> {
// input: &'a str,
// position: usize,
// }
// impl<'a> Lexer<'a> {
// // Constructor
// pub fn new(input: &'a str) -> Self {
// // Lexer { input, position: 0 }
// }
// // Advance position in input
// fn advance(&mut self) {
// self.position += 1;
// }
// // Get current character without advancing position
// fn current_char(&self) -> Option<char> {
// self.input.chars().nth(self.position)
// }
// // Lexical analysis
// pub fn next_token(&mut self) -> Token {
// // Skip whitespace
// while let Some(c) = self.current_char() {
// if c.is_whitespace() {
// self.advance();
// } else {
// break;
// }
// }
// // Check for end of input
// if let None = self.current_char() {
// return Token::EOF;
// }
// // Match characters to tokens
// match self.current_char().unwrap() {
// '+' => {
// self.advance();
// Token::Plus
// }
// '-' => {
// self.advance();
// Token::Minus
// }
// '*' => {
// self.advance();
// Token::Multiply
// }
// '/' => {
// self.advance();
// Token::Divide
// }
// '(' => {
// self.advance();
// Token::LParen
// }
// ')' => {
// self.advance();
// Token::RParen
// }
// '.' => {
// self.advance();
// Token::Dot
// }
// // Match numbers
// digit if digit.is_digit(10) => {
// let mut num_str = String::new();
// while let Some(digit) = self.current_char() {
// if digit.is_digit(10) {
// num_str.push(digit);
// self.advance();
// } else {
// break;
// }
// }
// Token::Number(num_str.parse().unwrap())
// }
// // Unknown character
// _ => {
// panic!("Invalid character: {}", self.current_char().unwrap());
// }
// }
// }
// }
}