Lexer
A basic lexer for the wisp programming language.
This commit is contained in:
Generated
+7
@@ -0,0 +1,7 @@
|
|||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 3
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "wisp"
|
||||||
|
version = "0.1.0"
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
[package]
|
||||||
|
name = "wisp"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2021"
|
||||||
|
|
||||||
|
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
+14
@@ -0,0 +1,14 @@
|
|||||||
|
# Lexer
|
||||||
|
The Lexer struct is responsible for tokenizing input strings. It reads characters from the input string and converts them into tokens, which represent the smallest units of meaning in the programming language.
|
||||||
|
|
||||||
|
# Constructor: new
|
||||||
|
The new method is a constructor for the Lexer struct. It takes an input string as its argument and returns a new Lexer instance initialized with the input string and an initial position of 0.
|
||||||
|
|
||||||
|
# Method: advance
|
||||||
|
The advance method moves the lexer's position to the next character in the input string. It increments the position field by 1, allowing the lexer to progress through the input string.
|
||||||
|
|
||||||
|
# Method: current_char
|
||||||
|
The current_char method retrieves the current character from the input string without advancing the lexer's position. It returns the character at the current position as an Option<char>. If the end of the input string is reached, it returns None.
|
||||||
|
|
||||||
|
# Method: next_token
|
||||||
|
The next_token method tokenizes the input string and returns the next token. It uses the advance and current_char methods to iterate through the input string, skipping whitespace characters and returning tokens based on the current character. It matches characters to known token types such as numbers, operators, and parentheses, and constructs tokens accordingly. If an invalid character is encountered, it panics with an error message.
|
||||||
@@ -0,0 +1,97 @@
|
|||||||
|
// Define token types
|
||||||
|
#[derive(Debug, PartialEq)]
|
||||||
|
pub enum Token {
|
||||||
|
Number(i32),
|
||||||
|
Plus,
|
||||||
|
Minus,
|
||||||
|
Multiply,
|
||||||
|
Divide,
|
||||||
|
LParen,
|
||||||
|
RParen,
|
||||||
|
EOF,
|
||||||
|
}
|
||||||
|
|
||||||
|
// Define Lexer struct
|
||||||
|
pub struct Lexer<'a> {
|
||||||
|
input: &'a str,
|
||||||
|
position: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> Lexer<'a> {
|
||||||
|
// Constructor
|
||||||
|
pub fn new(input: &'a str) -> Self {
|
||||||
|
Lexer { input, position: 0 }
|
||||||
|
}
|
||||||
|
|
||||||
|
// Advance position in input
|
||||||
|
fn advance(&mut self) {
|
||||||
|
self.position += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get current character without advancing position
|
||||||
|
fn current_char(&self) -> Option<char> {
|
||||||
|
self.input.chars().nth(self.position)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lexical analysis
|
||||||
|
pub fn next_token(&mut self) -> Token {
|
||||||
|
// Skip whitespace
|
||||||
|
while let Some(c) = self.current_char() {
|
||||||
|
if c.is_whitespace() {
|
||||||
|
self.advance();
|
||||||
|
} else {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check for end of input
|
||||||
|
if let None = self.current_char() {
|
||||||
|
return Token::EOF;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match characters to tokens
|
||||||
|
match self.current_char().unwrap() {
|
||||||
|
'+' => {
|
||||||
|
self.advance();
|
||||||
|
Token::Plus
|
||||||
|
}
|
||||||
|
'-' => {
|
||||||
|
self.advance();
|
||||||
|
Token::Minus
|
||||||
|
}
|
||||||
|
'*' => {
|
||||||
|
self.advance();
|
||||||
|
Token::Multiply
|
||||||
|
}
|
||||||
|
'/' => {
|
||||||
|
self.advance();
|
||||||
|
Token::Divide
|
||||||
|
}
|
||||||
|
'(' => {
|
||||||
|
self.advance();
|
||||||
|
Token::LParen
|
||||||
|
}
|
||||||
|
')' => {
|
||||||
|
self.advance();
|
||||||
|
Token::RParen
|
||||||
|
}
|
||||||
|
// Match numbers
|
||||||
|
digit if digit.is_digit(10) => {
|
||||||
|
let mut num_str = String::new();
|
||||||
|
while let Some(digit) = self.current_char() {
|
||||||
|
if digit.is_digit(10) {
|
||||||
|
num_str.push(digit);
|
||||||
|
self.advance();
|
||||||
|
} else {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Token::Number(num_str.parse().unwrap())
|
||||||
|
}
|
||||||
|
// Unknown character
|
||||||
|
_ => {
|
||||||
|
panic!("Invalid character: {}", self.current_char().unwrap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
pub mod lexer;
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
mod lexer; // Import the lexer module
|
||||||
|
|
||||||
|
use lexer::lexer::Lexer; // Import the Lexer struct
|
||||||
|
use lexer::lexer::Token; // Import the Token enum
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let input = "3 + 4 * (10 - 2)";
|
||||||
|
let mut lexer = Lexer::new(input);
|
||||||
|
|
||||||
|
loop {
|
||||||
|
let token = lexer.next_token();
|
||||||
|
println!("{:?}", token);
|
||||||
|
if token == Token::EOF {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
Signature: 8a477f597d28d172789f06886806bc55
|
||||||
|
# This file is a cache directory tag created by cargo.
|
||||||
|
# For information about cache directory tags see https://bford.info/cachedir/
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
C:\Users\charlie\Desktop\projects\wisp\wisp\target\debug\deps\libwisp-028a7a1f0b66b9bd.rmeta: src\main.rs src\lexer\mod.rs src\lexer\lexer.rs
|
||||||
|
|
||||||
|
C:\Users\charlie\Desktop\projects\wisp\wisp\target\debug\deps\wisp-028a7a1f0b66b9bd.d: src\main.rs src\lexer\mod.rs src\lexer\lexer.rs
|
||||||
|
|
||||||
|
src\main.rs:
|
||||||
|
src\lexer\mod.rs:
|
||||||
|
src\lexer\lexer.rs:
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
C:\Users\charlie\Desktop\projects\wisp\wisp\target\debug\deps\wisp.exe: src\main.rs src\lexer\mod.rs src\lexer\lexer.rs
|
||||||
|
|
||||||
|
C:\Users\charlie\Desktop\projects\wisp\wisp\target\debug\deps\wisp.d: src\main.rs src\lexer\mod.rs src\lexer\lexer.rs
|
||||||
|
|
||||||
|
src\main.rs:
|
||||||
|
src\lexer\mod.rs:
|
||||||
|
src\lexer\lexer.rs:
|
||||||
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
@@ -0,0 +1 @@
|
|||||||
|
C:\Users\charlie\Desktop\projects\wisp\wisp\target\debug\wisp.exe: C:\Users\charlie\Desktop\projects\wisp\wisp\src\lexer\lexer.rs C:\Users\charlie\Desktop\projects\wisp\wisp\src\lexer\mod.rs C:\Users\charlie\Desktop\projects\wisp\wisp\src\main.rs
|
||||||
Binary file not shown.
Binary file not shown.
Reference in New Issue
Block a user