the lexer now works
This commit is contained in:
+78
-122
@@ -1,139 +1,95 @@
|
|||||||
use regex::Regex;
|
use regex::Regex;
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
|
use std::fmt;
|
||||||
|
|
||||||
// Define token types
|
// Define token types
|
||||||
#[derive(Debug, PartialEq)]
|
#[derive(Debug, PartialEq, Clone, Copy)]
|
||||||
pub enum Token {
|
pub enum TokenType {
|
||||||
|
Newline,
|
||||||
|
Whitespace,
|
||||||
Number,
|
Number,
|
||||||
Dot,
|
Identifier,
|
||||||
Plus,
|
Ptr,
|
||||||
Minus,
|
Operator,
|
||||||
Multiply,
|
Round,
|
||||||
Divide,
|
Curly,
|
||||||
LParen,
|
// EOF,
|
||||||
RParen,
|
|
||||||
EOF,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub struct Token {
|
||||||
pub struct Node<'a> {
|
pub token_type: TokenType,
|
||||||
token_type: Token,
|
pub token_value: String
|
||||||
token_regex: Lazy<Regex>,
|
|
||||||
token_value: &'a str
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const SYNTAX: [Node; 1] = [
|
impl<'a> fmt::Display for Token {
|
||||||
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
|
write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct Node {
|
||||||
|
token_type: TokenType,
|
||||||
|
token_regex: Lazy<Regex>
|
||||||
|
}
|
||||||
|
|
||||||
|
const SYNTAX: [Node; 8] = [
|
||||||
Node {
|
Node {
|
||||||
token_type: Token::Dot,
|
token_type: TokenType::Newline,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^test").unwrap()),
|
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
|
||||||
token_value: ""
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Whitespace,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^\s+").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Number,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Identifier,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Ptr,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^->").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Operator,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^[\-|\+|\*]").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Round,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"\((?:[^()]|(?R))*\)").unwrap())
|
||||||
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Curly,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^[\{|\}]").unwrap())
|
||||||
}
|
}
|
||||||
];
|
];
|
||||||
|
|
||||||
pub fn lex(code: &str) -> Vec<Node> {
|
pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
|
||||||
let mut tokens: Vec<Node> = Vec::new();
|
let mut tokens: Vec<Token> = Vec::new();
|
||||||
while (code!="") {
|
while !code.is_empty() {
|
||||||
for s in SYNTAX {
|
let mut is_match = false;
|
||||||
if (s.token_regex.is_match(code)) {
|
for s in &SYNTAX {
|
||||||
println!("It works!");
|
if let Some(caps) = s.token_regex.captures(code) {
|
||||||
tokens.push(Node {
|
is_match = true;
|
||||||
token_type: s.token_type,
|
code = code.strip_prefix(&caps[0]).unwrap_or(code);
|
||||||
token_regex: s.token_regex,
|
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
|
||||||
token_value: s.token_regex.captures(wisp)
|
tokens.push(Token {
|
||||||
});
|
token_type: s.token_type,
|
||||||
}
|
token_value: caps[0].to_string(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if !is_match {
|
||||||
|
println!("Error: syntax -> {code}");
|
||||||
|
break;
|
||||||
}
|
}
|
||||||
} {
|
|
||||||
|
|
||||||
}
|
}
|
||||||
tokens
|
tokens
|
||||||
}
|
}
|
||||||
|
|
||||||
// Define Lexer struct
|
|
||||||
// pub struct Lexer<'a> {
|
|
||||||
// input: &'a str,
|
|
||||||
// position: usize,
|
|
||||||
// }
|
|
||||||
|
|
||||||
// impl<'a> Lexer<'a> {
|
|
||||||
// // Constructor
|
|
||||||
// pub fn new(input: &'a str) -> Self {
|
|
||||||
// // Lexer { input, position: 0 }
|
|
||||||
// }
|
|
||||||
|
|
||||||
// // Advance position in input
|
|
||||||
// fn advance(&mut self) {
|
|
||||||
// self.position += 1;
|
|
||||||
// }
|
|
||||||
|
|
||||||
// // Get current character without advancing position
|
|
||||||
// fn current_char(&self) -> Option<char> {
|
|
||||||
// self.input.chars().nth(self.position)
|
|
||||||
// }
|
|
||||||
|
|
||||||
// // Lexical analysis
|
|
||||||
// pub fn next_token(&mut self) -> Token {
|
|
||||||
// // Skip whitespace
|
|
||||||
// while let Some(c) = self.current_char() {
|
|
||||||
// if c.is_whitespace() {
|
|
||||||
// self.advance();
|
|
||||||
// } else {
|
|
||||||
// break;
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
|
|
||||||
// // Check for end of input
|
|
||||||
// if let None = self.current_char() {
|
|
||||||
// return Token::EOF;
|
|
||||||
// }
|
|
||||||
|
|
||||||
// // Match characters to tokens
|
|
||||||
// match self.current_char().unwrap() {
|
|
||||||
// '+' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::Plus
|
|
||||||
// }
|
|
||||||
// '-' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::Minus
|
|
||||||
// }
|
|
||||||
// '*' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::Multiply
|
|
||||||
// }
|
|
||||||
// '/' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::Divide
|
|
||||||
// }
|
|
||||||
// '(' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::LParen
|
|
||||||
// }
|
|
||||||
// ')' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::RParen
|
|
||||||
// }
|
|
||||||
// '.' => {
|
|
||||||
// self.advance();
|
|
||||||
// Token::Dot
|
|
||||||
// }
|
|
||||||
// // Match numbers
|
|
||||||
// digit if digit.is_digit(10) => {
|
|
||||||
// let mut num_str = String::new();
|
|
||||||
// while let Some(digit) = self.current_char() {
|
|
||||||
// if digit.is_digit(10) {
|
|
||||||
// num_str.push(digit);
|
|
||||||
// self.advance();
|
|
||||||
// } else {
|
|
||||||
// break;
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
// Token::Number(num_str.parse().unwrap())
|
|
||||||
// }
|
|
||||||
// // Unknown character
|
|
||||||
// _ => {
|
|
||||||
// panic!("Invalid character: {}", self.current_char().unwrap());
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
|
|||||||
+6
-3
@@ -2,7 +2,10 @@ mod lexer;
|
|||||||
use lexer::lex;
|
use lexer::lex;
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
// let input = "-3.232313 + 4 * (10 - 2)";
|
let input = "int test() {}";
|
||||||
let input = "test";
|
// let input = "test->xy";
|
||||||
let mut tokens = lex(input);
|
let tokens = lex(input, true);
|
||||||
|
for token in tokens {
|
||||||
|
println!("{:?} {}", token.token_type, token.token_value);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user