parser beta
This commit is contained in:
+19
-6
@@ -5,6 +5,7 @@ use std::fmt;
|
|||||||
// Define token types
|
// Define token types
|
||||||
#[derive(Debug, PartialEq, Clone, Copy)]
|
#[derive(Debug, PartialEq, Clone, Copy)]
|
||||||
pub enum TokenType {
|
pub enum TokenType {
|
||||||
|
Keyword,
|
||||||
Newline,
|
Newline,
|
||||||
Whitespace,
|
Whitespace,
|
||||||
Number,
|
Number,
|
||||||
@@ -16,14 +17,15 @@ pub enum TokenType {
|
|||||||
// EOF,
|
// EOF,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
pub struct Token {
|
pub struct Token {
|
||||||
pub token_type: TokenType,
|
pub token_type: TokenType,
|
||||||
pub token_value: String
|
pub token_value: Vec<String>
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<'a> fmt::Display for Token {
|
impl<'a> fmt::Debug for Token {
|
||||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
write!(f, "TokenType: {:?}, TokenValue: {}", self.token_type, self.token_value)
|
write!(f, "(TokenType: {:?}, TokenValue: {})", self.token_type, self.token_value.join(", "))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -32,7 +34,7 @@ pub struct Node {
|
|||||||
token_regex: Lazy<Regex>
|
token_regex: Lazy<Regex>
|
||||||
}
|
}
|
||||||
|
|
||||||
const SYNTAX: [Node; 8] = [
|
const SYNTAX: [Node; 9] = [
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Newline,
|
token_type: TokenType::Newline,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^\n+").unwrap())
|
||||||
@@ -45,6 +47,10 @@ const SYNTAX: [Node; 8] = [
|
|||||||
token_type: TokenType::Number,
|
token_type: TokenType::Number,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^\b(:?.)?(:?0[x|X])?\d+(:?.\d+)?\b").unwrap())
|
||||||
},
|
},
|
||||||
|
Node {
|
||||||
|
token_type: TokenType::Keyword,
|
||||||
|
token_regex: Lazy::new(|| Regex::new(r"^mut|try|catch|return|fn").unwrap())
|
||||||
|
},
|
||||||
Node {
|
Node {
|
||||||
token_type: TokenType::Identifier,
|
token_type: TokenType::Identifier,
|
||||||
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
|
token_regex: Lazy::new(|| Regex::new(r"^[._a-zA-Z][a-zA-Z0-9_]*").unwrap())
|
||||||
@@ -67,7 +73,7 @@ const SYNTAX: [Node; 8] = [
|
|||||||
}
|
}
|
||||||
];
|
];
|
||||||
|
|
||||||
pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
|
pub fn lex(mut code: &str, use_whitespace: bool) -> Vec<Token> {
|
||||||
let mut tokens: Vec<Token> = Vec::new();
|
let mut tokens: Vec<Token> = Vec::new();
|
||||||
while !code.is_empty() {
|
while !code.is_empty() {
|
||||||
let mut is_match = false;
|
let mut is_match = false;
|
||||||
@@ -75,10 +81,17 @@ pub fn lex(mut code: &str, use_whitespace: bool ) -> Vec<Token> {
|
|||||||
if let Some(caps) = s.token_regex.captures(code) {
|
if let Some(caps) = s.token_regex.captures(code) {
|
||||||
is_match = true;
|
is_match = true;
|
||||||
code = code.strip_prefix(&caps[0]).unwrap_or(code);
|
code = code.strip_prefix(&caps[0]).unwrap_or(code);
|
||||||
|
let mut vcaps: Vec<String> = Vec::new();
|
||||||
|
let capsl = caps.len();
|
||||||
|
let mut x = 0;
|
||||||
|
while x < capsl {
|
||||||
|
vcaps.push(caps[x].to_string());
|
||||||
|
x+=1;
|
||||||
|
}
|
||||||
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
|
if (!use_whitespace && s.token_type!=TokenType::Whitespace) || use_whitespace {
|
||||||
tokens.push(Token {
|
tokens.push(Token {
|
||||||
token_type: s.token_type,
|
token_type: s.token_type,
|
||||||
token_value: caps[0].to_string(),
|
token_value: vcaps,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
+12
-4
@@ -1,11 +1,19 @@
|
|||||||
mod lexer;
|
mod lexer;
|
||||||
use lexer::lex;
|
use lexer::lex;
|
||||||
|
mod parser;
|
||||||
|
use parser::parse;
|
||||||
|
|
||||||
fn main() {
|
fn main() {
|
||||||
let input = "int test() {}";
|
let input = "int main() {";
|
||||||
// let input = "test->xy";
|
// let input = "test->xy";
|
||||||
let tokens = lex(input, true);
|
let mut tokens = lex(input, false);
|
||||||
for token in tokens {
|
let ast = *parse(&mut tokens);
|
||||||
println!("{:?} {}", token.token_type, token.token_value);
|
for t in tokens {
|
||||||
|
println!("{:?}", t);
|
||||||
|
}
|
||||||
|
let mut _x = 1;
|
||||||
|
for a in ast {
|
||||||
|
println!("{:?}: {:?}", a.type_, a.values);
|
||||||
|
_x+=1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,79 @@
|
|||||||
|
// use std::collections::HashMap;
|
||||||
|
use crate::lexer::{Token, TokenType};
|
||||||
|
use std::fmt;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub enum AstTypes {
|
||||||
|
FunctionDecleration
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct Ast {
|
||||||
|
pub type_: AstTypes,
|
||||||
|
pub values: Vec<Token>
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> fmt::Debug for Ast {
|
||||||
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
|
write!(f, "type={:?} values={:?}", self.type_, self.values)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, PartialEq, Clone, Copy)]
|
||||||
|
pub enum AstDef {
|
||||||
|
Normal(TokenType),
|
||||||
|
Repeated(TokenType),
|
||||||
|
Optional(TokenType)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct PNode {
|
||||||
|
ast_type: AstTypes,
|
||||||
|
ast_match: Vec<AstDef>
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn parse(tokens: &mut Vec<Token>) -> Box<Vec<Ast>> {
|
||||||
|
let ast_def: Vec<PNode> = vec![
|
||||||
|
PNode {
|
||||||
|
ast_type: AstTypes::FunctionDecleration,
|
||||||
|
ast_match: vec![AstDef::Normal(TokenType::Identifier), AstDef::Normal(TokenType::Identifier), AstDef::Normal(TokenType::Round), AstDef::Normal(TokenType::Curly)]
|
||||||
|
},
|
||||||
|
PNode {
|
||||||
|
ast_type: AstTypes::FunctionDecleration,
|
||||||
|
ast_match: vec![AstDef::Normal(TokenType::Keyword)]
|
||||||
|
},
|
||||||
|
];
|
||||||
|
let tokens_len = tokens.len();
|
||||||
|
let mut ast: Vec<Ast> = Vec::new();
|
||||||
|
let mut current = 0;
|
||||||
|
let mut i = 0;
|
||||||
|
while tokens.len()>0 {
|
||||||
|
for ast_node in &ast_def {
|
||||||
|
if tokens_len < ast_node.ast_match.len() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut ast_types: Vec<AstTypes> = vec![];
|
||||||
|
let mut is_match = false;
|
||||||
|
let mut matched: Vec<Token> = Vec::new();
|
||||||
|
for ast_match in &ast_node.ast_match {
|
||||||
|
match ast_match {
|
||||||
|
AstDef::Normal(token_type) => {
|
||||||
|
if &tokens[i].token_type == token_type {
|
||||||
|
matched.push(tokens[i].clone());
|
||||||
|
tokens.drain(0..1);
|
||||||
|
is_match = true;
|
||||||
|
} else {
|
||||||
|
is_match = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if is_match {
|
||||||
|
i = 0;
|
||||||
|
ast.push(Ast {values: matched, type_: ast_node.ast_type.clone()});
|
||||||
|
matched = Vec::new();
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Box::new(ast)
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user