diff options
| author | hachem <im@hachem.wtf> | 2026-08-30 22:42:20 +0200 |
|---|---|---|
| committer | hachem <im@hachem.wtf> | 2026-08-30 22:42:20 +0200 |
| commit | 81799f3e41ad9cab305930b7743a752003ff16ec (patch) | |
| tree | 20fd67914ac143ab97a00392abcf5817ba11db22 | |
| parent | 294777178f9203283364f3e672569becb4dd1e7c (diff) | |
feat: sized memory stores bia ^
| -rw-r--r-- | examples/fibonacci.hdasm | 2 | ||||
| -rwxr-xr-x | scripts/test_examples.sh | 10 | ||||
| -rw-r--r-- | src/codegen/nasm.c | 86 | ||||
| -rw-r--r-- | src/lexer/lexer.c | 8 | ||||
| -rw-r--r-- | src/lexer/lexer.h | 4 | ||||
| -rw-r--r-- | src/parser/ast.h | 10 | ||||
| -rw-r--r-- | src/parser/parser.c | 14 | ||||
| -rw-r--r-- | tests/codegen_test.c | 26 | ||||
| -rw-r--r-- | tests/lexer_test.c | 10 | ||||
| -rw-r--r-- | tests/parser_test.c | 21 |
10 files changed, 188 insertions, 3 deletions
diff --git a/examples/fibonacci.hdasm b/examples/fibonacci.hdasm index b044d84..58f71fe 100644 --- a/examples/fibonacci.hdasm +++ b/examples/fibonacci.hdasm @@ -18,7 +18,7 @@ convert: rax /= rbx rdx += '0' - ^rsi = rdx + ^byte rsi = rdx rsi -= 1 if rax != 0 diff --git a/scripts/test_examples.sh b/scripts/test_examples.sh index 2fb4e5f..d5b5491 100755 --- a/scripts/test_examples.sh +++ b/scripts/test_examples.sh @@ -70,6 +70,16 @@ check arithmetic examples/arithmetic.hdass 15 "" check loop_sum examples/loop_sum.hdass 15 "" check branch examples/branch.hdass 8 "" check call examples/call.hdass 21 "" +check fibonacci examples/fibonacci.hdasm 0 "0 +1 +1 +2 +3 +5 +8 +13 +21 +34" echo echo "$pass passed, $fail failed" diff --git a/src/codegen/nasm.c b/src/codegen/nasm.c index 1062d7a..323adfc 100644 --- a/src/codegen/nasm.c +++ b/src/codegen/nasm.c @@ -193,6 +193,71 @@ static void emit_expr_into(struct Emitter* emitter, const char* dst, struct Expr fprintf(emitter->out, "\n"); } +static const char* store_size_keyword(enum StoreSize size) +{ + switch (size) + { + case STORE_SIZE_BYTE: return "byte "; + case STORE_SIZE_WORD: return "word "; + case STORE_SIZE_DWORD: return "dword "; + case STORE_SIZE_QWORD: return "qword "; + default: return ""; + } +} + +// maps a full 64-bit register to its byte/word/dword sub-register for a sized +// store, so `^byte rsi = rdx` writes `dl` rather than the whole register. +// returns NULL when the token is not a full register, or no resizing applies. +static const char* sized_register(struct Token reg, enum StoreSize size) +{ + if (size == STORE_SIZE_NONE || size == STORE_SIZE_QWORD) + return NULL; + + static const struct RegisterSizes + { + const char* quad; + const char* dword; + const char* word; + const char* byte; + } registers[] = + { + { "rax", "eax", "ax", "al" }, + { "rbx", "ebx", "bx", "bl" }, + { "rcx", "ecx", "cx", "cl" }, + { "rdx", "edx", "dx", "dl" }, + { "rsi", "esi", "si", "sil" }, + { "rdi", "edi", "di", "dil" }, + { "rbp", "ebp", "bp", "bpl" }, + { "rsp", "esp", "sp", "spl" }, + { "r8", "r8d", "r8w", "r8b" }, + { "r9", "r9d", "r9w", "r9b" }, + { "r10", "r10d", "r10w", "r10b" }, + { "r11", "r11d", "r11w", "r11b" }, + { "r12", "r12d", "r12w", "r12b" }, + { "r13", "r13d", "r13w", "r13b" }, + { "r14", "r14d", "r14w", "r14b" }, + { "r15", "r15d", "r15w", "r15b" }, + }; + + for (size_t i = 0; i < sizeof(registers) / sizeof(registers[0]); i += 1) + { + const struct RegisterSizes* entry = ®isters[i]; + size_t length = strlen(entry->quad); + if (reg.length != length || memcmp(reg.start, entry->quad, length) != 0) + continue; + + switch (size) + { + case STORE_SIZE_DWORD: return entry->dword; + case STORE_SIZE_WORD: return entry->word; + case STORE_SIZE_BYTE: return entry->byte; + default: return NULL; + } + } + + return NULL; +} + static void emit_assign(struct Emitter* emitter, struct AssignStatement* assign) { if (assign->op.type == TOKEN_SLASH_EQUAL) @@ -227,11 +292,28 @@ static void emit_assign(struct Emitter* emitter, struct AssignStatement* assign) } if (assign->target_deref) - fprintf(emitter->out, "\t%s [%.*s], ", mnemonic, (int)target.length, target.start); + { + fprintf(emitter->out, "\t%s %s[%.*s], ", mnemonic, + store_size_keyword(assign->store_size), (int)target.length, target.start); + + const char* sized = NULL; + if (assign->value->kind == EXPR_PRIMARY) + { + struct Token value = resolve_token(emitter, assign->value->primary.token); + sized = sized_register(value, assign->store_size); + if (sized != NULL) + fprintf(emitter->out, "%s", sized); + } + + if (sized == NULL) + emit_operand(emitter, assign->value); + } else + { fprintf(emitter->out, "\t%s %.*s, ", mnemonic, (int)target.length, target.start); + emit_operand(emitter, assign->value); + } - emit_operand(emitter, assign->value); fprintf(emitter->out, "\n"); } diff --git a/src/lexer/lexer.c b/src/lexer/lexer.c index 97a91b1..da45741 100644 --- a/src/lexer/lexer.c +++ b/src/lexer/lexer.c @@ -31,6 +31,10 @@ static enum TokenType identifier_type(const char* start, size_t length) { "if", 2, TOKEN_IF }, { "goto", 4, TOKEN_GOTO }, { "syscall", 7, TOKEN_SYSCALL }, + { "byte", 4, TOKEN_BYTE }, + { "word", 4, TOKEN_WORD }, + { "dword", 5, TOKEN_DWORD }, + { "qword", 5, TOKEN_QWORD }, }; for (size_t i = 0; i < sizeof(keywords) / sizeof(keywords[0]); i += 1) @@ -224,6 +228,10 @@ const char* token_type_name(enum TokenType type) case TOKEN_IF: return "if"; case TOKEN_GOTO: return "goto"; case TOKEN_SYSCALL: return "syscall"; + case TOKEN_BYTE: return "byte"; + case TOKEN_WORD: return "word"; + case TOKEN_DWORD: return "dword"; + case TOKEN_QWORD: return "qword"; case TOKEN_EQUAL: return "equal"; case TOKEN_PLUS: return "plus"; case TOKEN_MINUS: return "minus"; diff --git a/src/lexer/lexer.h b/src/lexer/lexer.h index 56c1feb..441fe10 100644 --- a/src/lexer/lexer.h +++ b/src/lexer/lexer.h @@ -18,6 +18,10 @@ enum TokenType TOKEN_IF, TOKEN_GOTO, TOKEN_SYSCALL, + TOKEN_BYTE, + TOKEN_WORD, + TOKEN_DWORD, + TOKEN_QWORD, TOKEN_EQUAL, TOKEN_PLUS, diff --git a/src/parser/ast.h b/src/parser/ast.h index 8281085..6071c4a 100644 --- a/src/parser/ast.h +++ b/src/parser/ast.h @@ -59,6 +59,15 @@ struct Expr }; }; +enum StoreSize +{ + STORE_SIZE_NONE, + STORE_SIZE_BYTE, + STORE_SIZE_WORD, + STORE_SIZE_DWORD, + STORE_SIZE_QWORD, +}; + enum StatementKind { STATEMENT_ASSIGN, @@ -73,6 +82,7 @@ enum StatementKind struct AssignStatement { bool target_deref; + enum StoreSize store_size; struct Token target; struct Token op; struct Expr* value; diff --git a/src/parser/parser.c b/src/parser/parser.c index d2a7285..ca14023 100644 --- a/src/parser/parser.c +++ b/src/parser/parser.c @@ -344,6 +344,19 @@ static bool parse_statement(struct Parser* parser, struct Statement* out) bool deref = match_token(parser, TOKEN_CARET); + enum StoreSize store_size = STORE_SIZE_NONE; + if (deref) + { + if (match_token(parser, TOKEN_BYTE)) + store_size = STORE_SIZE_BYTE; + else if (match_token(parser, TOKEN_WORD)) + store_size = STORE_SIZE_WORD; + else if (match_token(parser, TOKEN_DWORD)) + store_size = STORE_SIZE_DWORD; + else if (match_token(parser, TOKEN_QWORD)) + store_size = STORE_SIZE_QWORD; + } + if (!consume(parser, TOKEN_IDENTIFIER, "expected a statement")) return false; struct Token name = parser->previous; @@ -373,6 +386,7 @@ static bool parse_statement(struct Parser* parser, struct Statement* out) out->kind = STATEMENT_ASSIGN; out->assign.target_deref = deref; + out->assign.store_size = store_size; out->assign.target = name; out->assign.op = op; out->assign.value = value; diff --git a/tests/codegen_test.c b/tests/codegen_test.c index a72638b..64a422d 100644 --- a/tests/codegen_test.c +++ b/tests/codegen_test.c @@ -203,6 +203,31 @@ static void test_generate_address_expr(struct TestContext* context) free_program(&program); } +static void test_generate_sized_store(struct TestContext* context) +{ + struct Lexer lexer = create_lexer( + "proc main\n{\n^byte rsi = rdx\n^dword rsi = rax\n^byte rsi = 5\n^rsi = rbx\n}\n"); + struct Program program; + check(context, parse_program(&lexer, &program)); + + FILE* out = tmpfile(); + generate_nasm(&program, out); + fflush(out); + rewind(out); + + char buffer[1024]; + size_t read = fread(buffer, 1, sizeof(buffer) - 1, out); + buffer[read] = '\0'; + fclose(out); + + check(context, strstr(buffer, "mov byte [rsi], dl") != NULL); + check(context, strstr(buffer, "mov dword [rsi], eax") != NULL); + check(context, strstr(buffer, "mov byte [rsi], 5") != NULL); + check(context, strstr(buffer, "mov [rsi], rbx") != NULL); + + free_program(&program); +} + void run_codegen_tests(struct TestContext* context) { test_generate_consts_and_data(context); @@ -213,4 +238,5 @@ void run_codegen_tests(struct TestContext* context) test_generate_divide(context); test_generate_stack_frame(context); test_generate_address_expr(context); + test_generate_sized_store(context); } diff --git a/tests/lexer_test.c b/tests/lexer_test.c index 11b15b9..f677793 100644 --- a/tests/lexer_test.c +++ b/tests/lexer_test.c @@ -45,6 +45,15 @@ static void test_keywords(struct TestContext* context) check(context, scan_token(&lexer).type == TOKEN_IDENTIFIER); } +static void test_size_keywords(struct TestContext* context) +{ + struct Lexer lexer = create_lexer("byte word dword qword"); + check(context, scan_token(&lexer).type == TOKEN_BYTE); + check(context, scan_token(&lexer).type == TOKEN_WORD); + check(context, scan_token(&lexer).type == TOKEN_DWORD); + check(context, scan_token(&lexer).type == TOKEN_QWORD); +} + static void test_line_counting(struct TestContext* context) { struct Lexer lexer = create_lexer("a\nb\nc"); @@ -70,6 +79,7 @@ void run_lexer_tests(struct TestContext* context) test_operators(context); test_literals(context); test_keywords(context); + test_size_keywords(context); test_line_counting(context); test_comments(context); } diff --git a/tests/parser_test.c b/tests/parser_test.c index bbbbaa2..61f8c37 100644 --- a/tests/parser_test.c +++ b/tests/parser_test.c @@ -188,6 +188,26 @@ static void test_parse_stack(struct TestContext* context) free_program(&program); } +static void test_parse_sized_deref(struct TestContext* context) +{ + struct Lexer lexer = create_lexer("proc main\n{\n^byte rsi = rdx\n^rdi = rax\n}\n"); + struct Program program; + + check(context, parse_program(&lexer, &program)); + check(context, program.procs[0].body_count == 2); + + struct AssignStatement sized = program.procs[0].body[0].assign; + check(context, sized.target_deref); + check(context, sized.store_size == STORE_SIZE_BYTE); + check(context, text_is(sized.target, "rsi")); + + struct AssignStatement plain = program.procs[0].body[1].assign; + check(context, plain.target_deref); + check(context, plain.store_size == STORE_SIZE_NONE); + + free_program(&program); +} + static void test_parse_errors(struct TestContext* context) { struct Program program; @@ -216,5 +236,6 @@ void run_parser_tests(struct TestContext* context) test_parse_if(context); test_parse_call(context); test_parse_stack(context); + test_parse_sized_deref(context); test_parse_errors(context); } |
