parser beta
This commit is contained in:
+19
-6
@@ -5,6 +5,7 @@ use std::fmt;
|
||||
// Define token types
|
||||
#[derive(Debug, PartialEq, Clone, Copy)]
|
||||
pub enum TokenType {
|
||||
Keyword,
|
||||
Newline,
|
||||
Whitespace,
|
||||
Number,
|
||||
@@ -16,14 +17,15 @@ pub enum TokenType {
|
||||
// EOF,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct Token {
|
||||
pub token_type: TokenType,
|
||||
pub token_value: String
|
||||
pub token_value: Vec<String>
|
||||
}
|
||||
|
||||
impl<'a> fmt::Display for Token {
|
||||
impl<'a> fmt::Debug for Token {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value)
|
||||
write!(f, "(TokenType: {:?}, TokenValue: {})", self.token_type, self.token_value.join(", "))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -32,7 +34,7 @@ pub struct Node {
|
||||
token_regex: Lazy<Regex>
|
||||
}
|
||||
|
||||
const SYNTAX: [Node; 8] = [
|
||||
const SYNTAX: [Node; 9] = [
|
||||
Node {
|
||||
token_type: TokenType::Newline,
|
||||
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
|
||||
@@ -45,6 +47,10 @@ const SYNTAX: [Node; 8] = [
|
||||
token_type: TokenType::Number,
|
||||
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
|
||||
},
|
||||
Node {
|
||||
token_type: TokenType::Keyword,
|
||||
token_regex: Lazy::new(|| Regex::new(r"^mut|try|catch|return|fn").unwrap())
|
||||
},
|
||||
Node {
|
||||
token_type: TokenType::Identifier,
|
||||
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
|
||||
@@ -67,7 +73,7 @@ const SYNTAX: [Node; 8] = [
|
||||
}
|
||||
];
|
||||
|
||||
pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
|
||||
pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
||||
let mut tokens: Vec<Token> = Vec::new();
|
||||
while !code.is_empty() {
|
||||
let mut is_match = false;
|
||||
@@ -75,10 +81,17 @@ pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
|
||||
if let Some(caps) = s.token_regex.captures(code) {
|
||||
is_match = true;
|
||||
code = code.strip_prefix(&caps[0]).unwrap_or(code);
|
||||
let mut vcaps: Vec<String> = Vec::new();
|
||||
let capsl = caps.len();
|
||||
let mut x = 0;
|
||||
while x < capsl {
|
||||
vcaps.push(caps[x].to_string());
|
||||
x+=1;
|
||||
}
|
||||
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
|
||||
tokens.push(Token {
|
||||
token_type: s.token_type,
|
||||
token_value: caps[0].to_string(),
|
||||
token_value: vcaps,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
|
||||
+12
-4
@@ -1,11 +1,19 @@
|
||||
mod lexer;
|
||||
use lexer::lex;
|
||||
mod parser;
|
||||
use parser::parse;
|
||||
|
||||
fn main() {
|
||||
let input = "int test() {}";
|
||||
let input = "int main() {";
|
||||
// let input = "test->xy";
|
||||
let tokens = lex(input, true);
|
||||
for token in tokens {
|
||||
println!("{:?} {}", token.token_type, token.token_value);
|
||||
let mut tokens = lex(input, false);
|
||||
let ast = *parse(&mut tokens);
|
||||
for t in tokens {
|
||||
println!("{:?}", t);
|
||||
}
|
||||
let mut _x = 1;
|
||||
for a in ast {
|
||||
println!("{:?}: {:?}", a.type_, a.values);
|
||||
_x+=1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
// use std::collections::HashMap;
|
||||
use crate::lexer::{Token, TokenType};
|
||||
use std::fmt;
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum AstTypes {
|
||||
FunctionDecleration
|
||||
}
|
||||
|
||||
pub struct Ast {
|
||||
pub type_: AstTypes,
|
||||
pub values: Vec<Token>
|
||||
}
|
||||
|
||||
impl<'a> fmt::Debug for Ast {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
write!(f, "type={:?} values={:?}", self.type_, self.values)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq, Clone, Copy)]
|
||||
pub enum AstDef {
|
||||
Normal(TokenType),
|
||||
Repeated(TokenType),
|
||||
Optional(TokenType)
|
||||
}
|
||||
|
||||
pub struct PNode {
|
||||
ast_type: AstTypes,
|
||||
ast_match: Vec<AstDef>
|
||||
}
|
||||
|
||||
pub fn parse(tokens: &mut Vec<Token>) -> Box<Vec<Ast>> {
|
||||
let ast_def: Vec<PNode> = vec![
|
||||
PNode {
|
||||
ast_type: AstTypes::FunctionDecleration,
|
||||
ast_match: vec![AstDef::Normal(TokenType::Identifier), AstDef::Normal(TokenType::Identifier), AstDef::Normal(TokenType::Round), AstDef::Normal(TokenType::Curly)]
|
||||
},
|
||||
PNode {
|
||||
ast_type: AstTypes::FunctionDecleration,
|
||||
ast_match: vec![AstDef::Normal(TokenType::Keyword)]
|
||||
},
|
||||
];
|
||||
let tokens_len = tokens.len();
|
||||
let mut ast: Vec<Ast> = Vec::new();
|
||||
let mut current = 0;
|
||||
let mut i = 0;
|
||||
while tokens.len()>0 {
|
||||
for ast_node in &ast_def {
|
||||
if tokens_len < ast_node.ast_match.len() {
|
||||
continue;
|
||||
}
|
||||
let mut ast_types: Vec<AstTypes> = vec![];
|
||||
let mut is_match = false;
|
||||
let mut matched: Vec<Token> = Vec::new();
|
||||
for ast_match in &ast_node.ast_match {
|
||||
match ast_match {
|
||||
AstDef::Normal(token_type) => {
|
||||
if &tokens[i].token_type == token_type {
|
||||
matched.push(tokens[i].clone());
|
||||
tokens.drain(0..1);
|
||||
is_match = true;
|
||||
} else {
|
||||
is_match = false;
|
||||
}
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
if is_match {
|
||||
i = 0;
|
||||
ast.push(Ast {values: matched, type_: ast_node.ast_type.clone()});
|
||||
matched = Vec::new();
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
Box::new(ast)
|
||||
}
|
||||
Reference in New Issue
Block a user