text reader passes tests

This commit is contained in:
2026-08-04 14:22:57 -06:00
parent fa511520c6
commit 897e155b0e
6 changed files with 406 additions and 162 deletions
+162 -89
View File
@@ -4,7 +4,7 @@ namespace spider::asm_ebnf {
// Token Factory
TokenFactory tf;
//TokenFactory tf;
// Char Functions
@@ -28,117 +28,190 @@ namespace spider::asm_ebnf {
return ch != u32('"');
}
// (* Characters & Basic Predicates *)
const Token* letter = tf.fn(isUTF8Alpha);
const Token* digit = tf.choice("0123456789");
const Token* alpha_num_char = tf.choice({ letter, digit });
const Token* letter;
const Token* digit;
const Token* alpha_num_char;
const Token* hex_digit = tf.choice("0123456789ABCDEFabcdef");
const Token* octal_digit = tf.choice("01234567");
const Token* binary_digit = tf.choice("01");
const Token* hex_digit;
const Token* octal_digit;
const Token* binary_digit;
const Token* ws_char = tf.fn(isWhithespaceCharNotCrLf);
const Token* ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
const Token* whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
const Token* newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
const Token* utf8_char = tf.fn(isUTF8CharNotCrLf);
const Token* ws_char;
const Token* ws_optional;
const Token* whitespace;
const Token* newline;
const Token* utf8_char;
const Token* char_escape = tf.seq({ tf["\\"], utf8_char });
const Token* char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
const Token* char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
const Token* char_escape;
const Token* char_content;
const Token* char_lit;
const Token* string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
const Token* string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
const Token* string_char;
const Token* string_lit;
// (* Literals *)
const Token* identifier = tf.tag(tf.seq({
tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
}), "identifier", true);
const Token* identifier;
const Token* comment;
const Token* comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
const Token* sign;
const Token* exponent_marker;
const Token* exponent;
const Token* sign = tf.choice("+-");
const Token* exponent_marker = tf.choice("eE");
const Token* exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
const Token* decimal_lit;
const Token* float_lit;
const Token* decimal_lit = tf.tag(tf.seq({
tf.opt(sign),
digit,
tf.rep(digit),
tf.opt(tf.choice("BSIL"))
}), "decimal_lit", true);
const Token* hex_lit;
const Token* octal_lit;
const Token* binary_lit;
const Token* float_lit = tf.tag(tf.seq({
tf.opt(sign),
tf.choice({
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ digit, tf.rep(digit), exponent })
}),
tf.opt(tf.choice("FD"))
}), "float_lit", true);
const Token* literal;
const Token* literal_cast;
const Token* literal_decl;
const Token* hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
const Token* octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
const Token* binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
const Token* register_tok;
const Token* literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
const Token* literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
const Token* literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
const Token* addrm_ind;
const Token* addrm_ptr;
const Token* addrm_idx;
const Token* addrm_sca;
const Token* addrm_dis;
// (* Operands *)
const Token* register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
const Token* addr_modes;
const Token* operand;
const Token* addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
const Token* addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
const Token* opcode;
const Token* operand_list;
const Token* instruction;
const Token* addrm_idx = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_idx", true);
const Token* annotation_named;
const Token* annotation_arg;
const Token* annotation_args;
const Token* annotation_pars;
const Token* annotation;
const Token* addrm_sca = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_sca", true);
const Token* preprocessor_val;
const Token* preprocessor;
const Token* addrm_dis = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_dis", true);
const Token* label;
const Token* line_label;
const Token* line_annotation;
const Token* line_content;
const Token* line;
const Token* line_last;
const Token* program;
const Token* addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
const Token* operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
void initTokens(TokenFactory& tf) {
// (* Characters & Basic Predicates *)
letter = tf.fn(isUTF8Alpha);
digit = tf.choice("0123456789");
alpha_num_char = tf.choice({ letter, digit });
// (* Generalized Instructions *)
hex_digit = tf.choice("0123456789ABCDEFabcdef");
octal_digit = tf.choice("01234567");
binary_digit = tf.choice("01");
const Token* opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
const Token* operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
const Token* instruction = tf.tag(tf.seq({opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
ws_char = tf.fn(isWhithespaceCharNotCrLf);
ws_optional = tf.tag(tf.rep(ws_char), "whitespace", true);
whitespace = tf.tag(tf.seq({ ws_char, tf.rep(ws_char) }), "whitespace", true);
newline = tf.choice({ tf["\r\n"], tf["\r"], tf["\n"] });
utf8_char = tf.fn(isUTF8CharNotCrLf);
// (* Added Preprocessor, Annotation *)
char_escape = tf.seq({ tf["\\"], utf8_char });
char_content = tf.choice({ char_escape, tf.fn(isUTF8CharLitCont) });
char_lit = tf.seq({ tf["'"], char_content, tf["'"] });
const Token* annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
const Token* annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
const Token* annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
const Token* annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
const Token* annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
string_char = tf.choice({ char_escape, tf.fn(isUTF8StringLitCont) });
string_lit = tf.seq({ tf["\""], tf.rep(string_char), tf["\""] });
const Token* preprocessor_val = tf.choice({ identifier, literal_decl });
const Token* preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
// (* Literals *)
identifier = tf.tag(tf.seq({
tf.choice({ letter, tf["_"] }),
tf.rep(tf.choice({ alpha_num_char, tf["_"] }))
}), "identifier", true);
// (* Line Structure & Program *)
comment = tf.tag(tf.seq({ tf[";"], tf.rep(utf8_char) }), "comment", true);
const Token* label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
const Token* line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
const Token* line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
const Token* line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
const Token* line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
const Token* line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
const Token* program = tf.seq({ tf.rep(line), tf.opt(line_last) });
sign = tf.choice("+-");
exponent_marker = tf.choice("eE");
exponent = tf.seq({ exponent_marker, tf.opt(sign), digit, tf.rep(digit) });
decimal_lit = tf.tag(tf.seq({
tf.opt(sign),
digit,
tf.rep(digit),
tf.opt(tf.choice("BSIL"))
}), "decimal_lit", true);
float_lit = tf.tag(tf.seq({
tf.opt(sign),
tf.choice({
tf.seq({ digit, tf.rep(digit), tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ tf["."], digit, tf.rep(digit), tf.opt(exponent) }),
tf.seq({ digit, tf.rep(digit), exponent })
}),
tf.opt(tf.choice("FD"))
}), "float_lit", true);
hex_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0x"], hex_digit, tf.rep(hex_digit) }), "hex_lit", true);
octal_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0c"], octal_digit, tf.rep(octal_digit) }), "octal_lit", true);
binary_lit = tf.tag(tf.seq({ tf.opt(sign), tf["0b"], binary_digit, tf.rep(binary_digit) }), "binary_lit", true);
literal = tf.tag(tf.choice({ float_lit, decimal_lit, hex_lit, octal_lit, binary_lit, string_lit, char_lit }), "literal");
literal_cast = tf.tag(tf.seq({ tf.choice("BSILFD"), ws_optional, tf["("], ws_optional, literal, ws_optional, tf[")"] }), "literal_cast");
literal_decl = tf.tag(tf.choice({ literal, literal_cast }), "literal_decl");
// (* Operands *)
register_tok = tf.tag(tf.seq({ tf["R"], alpha_num_char }), "register", true);
addrm_ind = tf.tag(tf.seq({ tf["["], ws_optional, literal_decl, ws_optional, tf["]"] }), "addrm_ind", true);
addrm_ptr = tf.tag(tf.seq({ tf["["], ws_optional, register_tok, ws_optional, tf["]"] }), "addrm_ptr", true);
addrm_idx = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_idx", true);
addrm_sca = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_sca", true);
addrm_dis = tf.tag(tf.seq({
tf["["], ws_optional, register_tok, ws_optional,
tf["+"], ws_optional, register_tok, ws_optional,
tf["*"], ws_optional, literal_decl, ws_optional,
tf["+"], ws_optional, literal_decl, ws_optional, tf["]"]
}), "addrm_dis", true);
addr_modes = tf.tag(tf.choice({ addrm_dis, addrm_sca, addrm_idx, addrm_ptr, addrm_ind }), "addrm");
operand = tf.tag(tf.choice({ register_tok, identifier, literal_decl, addr_modes }), "operand");
// (* Generalized Instructions *)
opcode = tf.tag(tf.seq({ letter, tf.rep(alpha_num_char) }), "opcode", true);
operand_list = tf.seq({ operand, tf.rep(tf.seq({ tf[","], ws_optional, operand })) });
instruction = tf.tag(tf.seq({ opcode, tf.opt(tf.seq({ whitespace, operand_list })) }), "instruction");
// (* Added Preprocessor, Annotation *)
annotation_named = tf.tag(tf.seq({ identifier, ws_optional, tf["="], ws_optional, literal_decl }), "annotation_arg_named");
annotation_arg = tf.tag(tf.choice({ annotation_named, literal_decl }), "annotation_arg");
annotation_args = tf.seq({ annotation_arg, tf.rep(tf.seq({ ws_optional, tf[","], ws_optional, annotation_arg })) });
annotation_pars = tf.seq({ tf["("], ws_optional, annotation_args, ws_optional, tf[")"] });
annotation = tf.tag(tf.seq({ tf["@"], identifier, tf.opt(annotation_pars) }), "annotation");
preprocessor_val = tf.choice({ identifier, literal_decl });
preprocessor = tf.tag(tf.seq({ tf["#"], identifier, whitespace, preprocessor_val }), "preprocessor");
// (* Line Structure & Program *)
label = tf.tag(tf.seq({ identifier, tf[":"] }), "label");
line_label = tf.seq({ label, tf.opt(tf.seq({ whitespace, instruction })) });
line_annotation = tf.seq({ annotation, tf.opt(tf.seq({ whitespace, instruction })) });
line_content = tf.choice({ preprocessor, line_annotation, line_label, instruction });
line = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment), newline });
line_last = tf.seq({ ws_optional, tf.opt(line_content), ws_optional, tf.opt(comment) });
program = tf.seq({ tf.rep(line), tf.opt(line_last) });
}
}