time to test
This commit is contained in:
@@ -2,12 +2,18 @@
|
||||
|
||||
namespace spider::asm_ebnf {
|
||||
|
||||
// Token Factory
|
||||
|
||||
static TokenFactory tf;
|
||||
|
||||
// Char Functions
|
||||
|
||||
bool isUTF8Alpha(u32 ch) {
|
||||
return false;
|
||||
return (u32('a') <= ch && ch <= u32('z')) || (u32('A') <= ch && ch <= u32('Z'));
|
||||
}
|
||||
|
||||
bool isWhithespaceCharNotCrLf(u32 ch) {
|
||||
return false;
|
||||
return ch == u32(' ');
|
||||
}
|
||||
|
||||
bool isUTF8CharNotCrLf(u32 ch) {
|
||||
@@ -22,37 +28,117 @@ namespace spider::asm_ebnf {
|
||||
return ch != u32('"');
|
||||
}
|
||||
|
||||
LitToken numbers[] = {
|
||||
"0",
|
||||
"1","2","3",
|
||||
"4","5","6",
|
||||
"7","8","9",
|
||||
};
|
||||
|
||||
LitToken hex_digits[][2] = {
|
||||
{"A", "a"},
|
||||
{"B", "b"},
|
||||
{"C", "c"},
|
||||
{"D", "d"},
|
||||
{"E", "e"},
|
||||
{"F", "f"},
|
||||
};
|
||||
// (* Characters & Basic Predicates *)
|
||||
static const Token* letter = tf.fn(isUTF8Alpha);
|
||||
static const Token* digit = tf.choice("0123456789");
|
||||
static const Token* alpha_num_char = tf.choice({ letter, digit });
|
||||
|
||||
LitToken new_line[] = { "\r\n", "\r", "\n" };
|
||||
static const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
|
||||
static const Token* octal_digit = tf.choice("01234567");
|
||||
static const Token* binary_digit = tf.choice("01");
|
||||
|
||||
LitToken symbols[] = {
|
||||
"\\", "\'", "\"",
|
||||
"_" , ";" , "(" ,
|
||||
"#",
|
||||
"$" , "." , "+" , "-", ",", ")", "@", ":",
|
||||
};
|
||||
static const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
|
||||
static const Token* ws_optional = tf.rep(ws_char);
|
||||
static const Token* whitespace = tf.seq({ ws_char, tf.rep(ws_char) });
|
||||
static const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
|
||||
static const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
|
||||
|
||||
LitToken lit_letter[] = {
|
||||
"x", "c", "b"
|
||||
};
|
||||
static const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
|
||||
static const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
|
||||
static const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
|
||||
|
||||
LitToken type_letter[] = {
|
||||
"B", "S", "I", "L", "F", "D"
|
||||
};
|
||||
static const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
|
||||
static const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
|
||||
|
||||
// (* Literals *)
|
||||
static const Token* identifier = tf.seq({
|
||||
tf.choice({ letter, tf["_"] }),
|
||||
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
|
||||
});
|
||||
|
||||
static const Token* comment = tf.seq({ tf[";"], tf.rep(utf8_char) });
|
||||
|
||||
static const Token* sign = tf.choice("+-");
|
||||
static const Token* exponent_marker = tf.choice("eE");
|
||||
static const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
|
||||
|
||||
static const Token* decimal_lit = tf.seq({
|
||||
tf.opt(sign),
|
||||
digit,
|
||||
tf.rep(digit),
|
||||
tf.opt(tf.choice("BSIL"))
|
||||
});
|
||||
|
||||
static const Token* float_lit = tf.seq({
|
||||
tf.opt(sign),
|
||||
tf.choice({
|
||||
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
|
||||
tf.seq({ digit, tf.rep(digit), exponent })
|
||||
}),
|
||||
tf.opt(tf.choice("FD"))
|
||||
});
|
||||
|
||||
static const Token* hex_lit = tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) });
|
||||
static const Token* octal_lit = tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) });
|
||||
static const Token* binary_lit = tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) });
|
||||
|
||||
static const Token* literal = tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit });
|
||||
static const Token* literal_cast = tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] });
|
||||
static const Token* literal_decl = tf.choice({ literal, literal_cast });
|
||||
|
||||
// (* Operands *)
|
||||
static const Token* register_tok = tf.seq({ tf["R"], alpha_num_char });
|
||||
|
||||
static const Token* addrm_ind = tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] });
|
||||
static const Token* addrm_ptr = tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] });
|
||||
|
||||
static const Token* addrm_idx = tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
});
|
||||
|
||||
static const Token* addrm_sca = tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, register_tok, ws_optional,
|
||||
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
});
|
||||
|
||||
static const Token* addrm_dis = tf.seq({
|
||||
tf["["], ws_optional, register_tok, ws_optional,
|
||||
tf["+"], ws_optional, register_tok, ws_optional,
|
||||
tf["*"], ws_optional, literal_decl, ws_optional,
|
||||
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
|
||||
});
|
||||
|
||||
static const Token* addr_modes = tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind });
|
||||
static const Token* operand = tf.choice({ register_tok, identifier, literal_decl, addr_modes });
|
||||
|
||||
// (* Generalized Instructions *)
|
||||
|
||||
static const Token* opcode = tf.seq({ letter, tf.rep(alpha_num_char) });
|
||||
static const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
|
||||
static const Token* instruction = tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) });
|
||||
|
||||
// (* Added Preprocessor, Annotation *)
|
||||
|
||||
static const Token* annotation_named = tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl });
|
||||
static const Token* annotation_arg = tf.choice({ annotation_named, literal_decl });
|
||||
static const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
|
||||
static const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
|
||||
static const Token* annotation = tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) });
|
||||
|
||||
static const Token* preprocessor_val = tf.choice({ identifier, string_lit });
|
||||
static const Token* preprocessor = tf.seq({ tf["#"], identifier, whitespace, preprocessor_val });
|
||||
|
||||
// (* Line Structure & Program *)
|
||||
|
||||
static const Token* label = tf.seq({ identifier, tf[":"] });
|
||||
static const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
static const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
|
||||
static const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
|
||||
static const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
|
||||
static const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
|
||||
static const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
|
||||
|
||||
}
|
||||
|
||||
@@ -6,4 +6,6 @@ namespace spider::asm_ebnf {
|
||||
|
||||
extern LitToken letter;
|
||||
|
||||
void createTokens();
|
||||
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user