diff --git a/src/lexer.rs b/src/lexer.rs index e7b55b2..ed07b6c 100644 --- a/src/lexer.rs +++ b/src/lexer.rs @@ -1,139 +1,95 @@ use regex::Regex; use once_cell::sync::Lazy; +use std::fmt; // Define token types -#[derive(Debug, PartialEq)] -pub enum Token { +#[derive(Debug, PartialEq, Clone, Copy)] +pub enum TokenType { + Newline, + Whitespace, Number, - Dot, - Plus, - Minus, - Multiply, - Divide, - LParen, - RParen, - EOF, + Identifier, + Ptr, + Operator, + Round, + Curly, + // EOF, } - -pub struct Node<'a> { - token_type: Token, - token_regex: Lazy, - token_value: &'a str +pub struct Token { + pub token_type: TokenType, + pub token_value: String } -const SYNTAX: [Node; 1] = [ +impl<'a> fmt::Display for Token { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value) + } +} + +pub struct Node { + token_type: TokenType, + token_regex: Lazy +} + +const SYNTAX: [Node; 8] = [ Node { - token_type: Token::Dot, - token_regex: Lazy::new(|| Regex::new(r"^test").unwrap()), - token_value: "" + token_type: TokenType::Newline, + token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap()) + }, + Node { + token_type: TokenType::Whitespace, + token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap()) + }, + Node { + token_type: TokenType::Number, + token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap()) + }, + Node { + token_type: TokenType::Identifier, + token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap()) + }, + Node { + token_type: TokenType::Ptr, + token_regex: Lazy::new(|| Regex::new(r"^->").unwrap()) + }, + Node { + token_type: TokenType::Operator, + token_regex: Lazy::new(|| Regex::new(r"^[\-|\+|\*]").unwrap()) + }, + Node { + token_type: TokenType::Round, + token_regex: Lazy::new(|| Regex::new(r"\((?:[^()]|(?R))*\)").unwrap()) + }, + Node { + token_type: TokenType::Curly, + token_regex: Lazy::new(|| Regex::new(r"^[\{|\}]").unwrap()) } ]; -pub fn lex(code: &str) -> Vec { - let mut tokens: Vec = Vec::new(); - while (code!="") { - for s in SYNTAX { - if (s.token_regex.is_match(code)) { - println!("It works!"); - tokens.push(Node { - token_type: s.token_type, - token_regex: s.token_regex, - token_value: s.token_regex.captures(wisp) - }); - } +pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec { + let mut tokens: Vec = Vec::new(); + while !code.is_empty() { + let mut is_match = false; + for s in &SYNTAX { + if let Some(caps) = s.token_regex.captures(code) { + is_match = true; + code = code.strip_prefix(&caps[0]).unwrap_or(code); + if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace { + tokens.push(Token { + token_type: s.token_type, + token_value: caps[0].to_string(), + }); + } + } else { + continue; + }; + break; + } + if !is_match { + println!("Error: syntax -> {code}"); + break; } - } { - } tokens -} - -// Define Lexer struct -// pub struct Lexer<'a> { -// input: &'a str, -// position: usize, -// } - -// impl<'a> Lexer<'a> { -// // Constructor -// pub fn new(input: &'a str) -> Self { -// // Lexer { input, position: 0 } -// } - -// // Advance position in input -// fn advance(&mut self) { -// self.position += 1; -// } - -// // Get current character without advancing position -// fn current_char(&self) -> Option { -// self.input.chars().nth(self.position) -// } - -// // Lexical analysis -// pub fn next_token(&mut self) -> Token { -// // Skip whitespace -// while let Some(c) = self.current_char() { -// if c.is_whitespace() { -// self.advance(); -// } else { -// break; -// } -// } - -// // Check for end of input -// if let None = self.current_char() { -// return Token::EOF; -// } - -// // Match characters to tokens -// match self.current_char().unwrap() { -// '+' => { -// self.advance(); -// Token::Plus -// } -// '-' => { -// self.advance(); -// Token::Minus -// } -// '*' => { -// self.advance(); -// Token::Multiply -// } -// '/' => { -// self.advance(); -// Token::Divide -// } -// '(' => { -// self.advance(); -// Token::LParen -// } -// ')' => { -// self.advance(); -// Token::RParen -// } -// '.' => { -// self.advance(); -// Token::Dot -// } -// // Match numbers -// digit if digit.is_digit(10) => { -// let mut num_str = String::new(); -// while let Some(digit) = self.current_char() { -// if digit.is_digit(10) { -// num_str.push(digit); -// self.advance(); -// } else { -// break; -// } -// } -// Token::Number(num_str.parse().unwrap()) -// } -// // Unknown character -// _ => { -// panic!("Invalid character: {}", self.current_char().unwrap()); -// } -// } -// } -// } +} \ No newline at end of file diff --git a/src/main.rs b/src/main.rs index 40ba416..32439bb 100644 --- a/src/main.rs +++ b/src/main.rs @@ -2,7 +2,10 @@ mod lexer; use lexer::lex; fn main() { - // let input = "-3.232313 + 4 * (10 - 2)"; - let input = "test"; - let mut tokens = lex(input); + let input = "int test() {}"; + // let input = "test->xy"; + let tokens = lex(input, true); + for token in tokens { + println!("{:?} {}", token.token_type, token.token_value); + } }