aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorhachem <im@hachem.wtf>2026-08-30 22:42:20 +0200
committerhachem <im@hachem.wtf>2026-08-30 22:42:20 +0200
commit81799f3e41ad9cab305930b7743a752003ff16ec (patch)
tree20fd67914ac143ab97a00392abcf5817ba11db22
parent294777178f9203283364f3e672569becb4dd1e7c (diff)
feat: sized memory stores bia ^
-rw-r--r--examples/fibonacci.hdasm2
-rwxr-xr-xscripts/test_examples.sh10
-rw-r--r--src/codegen/nasm.c86
-rw-r--r--src/lexer/lexer.c8
-rw-r--r--src/lexer/lexer.h4
-rw-r--r--src/parser/ast.h10
-rw-r--r--src/parser/parser.c14
-rw-r--r--tests/codegen_test.c26
-rw-r--r--tests/lexer_test.c10
-rw-r--r--tests/parser_test.c21
10 files changed, 188 insertions, 3 deletions
diff --git a/examples/fibonacci.hdasm b/examples/fibonacci.hdasm
index b044d84..58f71fe 100644
--- a/examples/fibonacci.hdasm
+++ b/examples/fibonacci.hdasm
@@ -18,7 +18,7 @@ convert:
rax /= rbx
rdx += '0'
- ^rsi = rdx
+ ^byte rsi = rdx
rsi -= 1
if rax != 0
diff --git a/scripts/test_examples.sh b/scripts/test_examples.sh
index 2fb4e5f..d5b5491 100755
--- a/scripts/test_examples.sh
+++ b/scripts/test_examples.sh
@@ -70,6 +70,16 @@ check arithmetic examples/arithmetic.hdass 15 ""
check loop_sum examples/loop_sum.hdass 15 ""
check branch examples/branch.hdass 8 ""
check call examples/call.hdass 21 ""
+check fibonacci examples/fibonacci.hdasm 0 "0
+1
+1
+2
+3
+5
+8
+13
+21
+34"
echo
echo "$pass passed, $fail failed"
diff --git a/src/codegen/nasm.c b/src/codegen/nasm.c
index 1062d7a..323adfc 100644
--- a/src/codegen/nasm.c
+++ b/src/codegen/nasm.c
@@ -193,6 +193,71 @@ static void emit_expr_into(struct Emitter* emitter, const char* dst, struct Expr
fprintf(emitter->out, "\n");
}
+static const char* store_size_keyword(enum StoreSize size)
+{
+ switch (size)
+ {
+ case STORE_SIZE_BYTE: return "byte ";
+ case STORE_SIZE_WORD: return "word ";
+ case STORE_SIZE_DWORD: return "dword ";
+ case STORE_SIZE_QWORD: return "qword ";
+ default: return "";
+ }
+}
+
+// maps a full 64-bit register to its byte/word/dword sub-register for a sized
+// store, so `^byte rsi = rdx` writes `dl` rather than the whole register.
+// returns NULL when the token is not a full register, or no resizing applies.
+static const char* sized_register(struct Token reg, enum StoreSize size)
+{
+ if (size == STORE_SIZE_NONE || size == STORE_SIZE_QWORD)
+ return NULL;
+
+ static const struct RegisterSizes
+ {
+ const char* quad;
+ const char* dword;
+ const char* word;
+ const char* byte;
+ } registers[] =
+ {
+ { "rax", "eax", "ax", "al" },
+ { "rbx", "ebx", "bx", "bl" },
+ { "rcx", "ecx", "cx", "cl" },
+ { "rdx", "edx", "dx", "dl" },
+ { "rsi", "esi", "si", "sil" },
+ { "rdi", "edi", "di", "dil" },
+ { "rbp", "ebp", "bp", "bpl" },
+ { "rsp", "esp", "sp", "spl" },
+ { "r8", "r8d", "r8w", "r8b" },
+ { "r9", "r9d", "r9w", "r9b" },
+ { "r10", "r10d", "r10w", "r10b" },
+ { "r11", "r11d", "r11w", "r11b" },
+ { "r12", "r12d", "r12w", "r12b" },
+ { "r13", "r13d", "r13w", "r13b" },
+ { "r14", "r14d", "r14w", "r14b" },
+ { "r15", "r15d", "r15w", "r15b" },
+ };
+
+ for (size_t i = 0; i < sizeof(registers) / sizeof(registers[0]); i += 1)
+ {
+ const struct RegisterSizes* entry = &registers[i];
+ size_t length = strlen(entry->quad);
+ if (reg.length != length || memcmp(reg.start, entry->quad, length) != 0)
+ continue;
+
+ switch (size)
+ {
+ case STORE_SIZE_DWORD: return entry->dword;
+ case STORE_SIZE_WORD: return entry->word;
+ case STORE_SIZE_BYTE: return entry->byte;
+ default: return NULL;
+ }
+ }
+
+ return NULL;
+}
+
static void emit_assign(struct Emitter* emitter, struct AssignStatement* assign)
{
if (assign->op.type == TOKEN_SLASH_EQUAL)
@@ -227,11 +292,28 @@ static void emit_assign(struct Emitter* emitter, struct AssignStatement* assign)
}
if (assign->target_deref)
- fprintf(emitter->out, "\t%s [%.*s], ", mnemonic, (int)target.length, target.start);
+ {
+ fprintf(emitter->out, "\t%s %s[%.*s], ", mnemonic,
+ store_size_keyword(assign->store_size), (int)target.length, target.start);
+
+ const char* sized = NULL;
+ if (assign->value->kind == EXPR_PRIMARY)
+ {
+ struct Token value = resolve_token(emitter, assign->value->primary.token);
+ sized = sized_register(value, assign->store_size);
+ if (sized != NULL)
+ fprintf(emitter->out, "%s", sized);
+ }
+
+ if (sized == NULL)
+ emit_operand(emitter, assign->value);
+ }
else
+ {
fprintf(emitter->out, "\t%s %.*s, ", mnemonic, (int)target.length, target.start);
+ emit_operand(emitter, assign->value);
+ }
- emit_operand(emitter, assign->value);
fprintf(emitter->out, "\n");
}
diff --git a/src/lexer/lexer.c b/src/lexer/lexer.c
index 97a91b1..da45741 100644
--- a/src/lexer/lexer.c
+++ b/src/lexer/lexer.c
@@ -31,6 +31,10 @@ static enum TokenType identifier_type(const char* start, size_t length)
{ "if", 2, TOKEN_IF },
{ "goto", 4, TOKEN_GOTO },
{ "syscall", 7, TOKEN_SYSCALL },
+ { "byte", 4, TOKEN_BYTE },
+ { "word", 4, TOKEN_WORD },
+ { "dword", 5, TOKEN_DWORD },
+ { "qword", 5, TOKEN_QWORD },
};
for (size_t i = 0; i < sizeof(keywords) / sizeof(keywords[0]); i += 1)
@@ -224,6 +228,10 @@ const char* token_type_name(enum TokenType type)
case TOKEN_IF: return "if";
case TOKEN_GOTO: return "goto";
case TOKEN_SYSCALL: return "syscall";
+ case TOKEN_BYTE: return "byte";
+ case TOKEN_WORD: return "word";
+ case TOKEN_DWORD: return "dword";
+ case TOKEN_QWORD: return "qword";
case TOKEN_EQUAL: return "equal";
case TOKEN_PLUS: return "plus";
case TOKEN_MINUS: return "minus";
diff --git a/src/lexer/lexer.h b/src/lexer/lexer.h
index 56c1feb..441fe10 100644
--- a/src/lexer/lexer.h
+++ b/src/lexer/lexer.h
@@ -18,6 +18,10 @@ enum TokenType
TOKEN_IF,
TOKEN_GOTO,
TOKEN_SYSCALL,
+ TOKEN_BYTE,
+ TOKEN_WORD,
+ TOKEN_DWORD,
+ TOKEN_QWORD,
TOKEN_EQUAL,
TOKEN_PLUS,
diff --git a/src/parser/ast.h b/src/parser/ast.h
index 8281085..6071c4a 100644
--- a/src/parser/ast.h
+++ b/src/parser/ast.h
@@ -59,6 +59,15 @@ struct Expr
};
};
+enum StoreSize
+{
+ STORE_SIZE_NONE,
+ STORE_SIZE_BYTE,
+ STORE_SIZE_WORD,
+ STORE_SIZE_DWORD,
+ STORE_SIZE_QWORD,
+};
+
enum StatementKind
{
STATEMENT_ASSIGN,
@@ -73,6 +82,7 @@ enum StatementKind
struct AssignStatement
{
bool target_deref;
+ enum StoreSize store_size;
struct Token target;
struct Token op;
struct Expr* value;
diff --git a/src/parser/parser.c b/src/parser/parser.c
index d2a7285..ca14023 100644
--- a/src/parser/parser.c
+++ b/src/parser/parser.c
@@ -344,6 +344,19 @@ static bool parse_statement(struct Parser* parser, struct Statement* out)
bool deref = match_token(parser, TOKEN_CARET);
+ enum StoreSize store_size = STORE_SIZE_NONE;
+ if (deref)
+ {
+ if (match_token(parser, TOKEN_BYTE))
+ store_size = STORE_SIZE_BYTE;
+ else if (match_token(parser, TOKEN_WORD))
+ store_size = STORE_SIZE_WORD;
+ else if (match_token(parser, TOKEN_DWORD))
+ store_size = STORE_SIZE_DWORD;
+ else if (match_token(parser, TOKEN_QWORD))
+ store_size = STORE_SIZE_QWORD;
+ }
+
if (!consume(parser, TOKEN_IDENTIFIER, "expected a statement"))
return false;
struct Token name = parser->previous;
@@ -373,6 +386,7 @@ static bool parse_statement(struct Parser* parser, struct Statement* out)
out->kind = STATEMENT_ASSIGN;
out->assign.target_deref = deref;
+ out->assign.store_size = store_size;
out->assign.target = name;
out->assign.op = op;
out->assign.value = value;
diff --git a/tests/codegen_test.c b/tests/codegen_test.c
index a72638b..64a422d 100644
--- a/tests/codegen_test.c
+++ b/tests/codegen_test.c
@@ -203,6 +203,31 @@ static void test_generate_address_expr(struct TestContext* context)
free_program(&program);
}
+static void test_generate_sized_store(struct TestContext* context)
+{
+ struct Lexer lexer = create_lexer(
+ "proc main\n{\n^byte rsi = rdx\n^dword rsi = rax\n^byte rsi = 5\n^rsi = rbx\n}\n");
+ struct Program program;
+ check(context, parse_program(&lexer, &program));
+
+ FILE* out = tmpfile();
+ generate_nasm(&program, out);
+ fflush(out);
+ rewind(out);
+
+ char buffer[1024];
+ size_t read = fread(buffer, 1, sizeof(buffer) - 1, out);
+ buffer[read] = '\0';
+ fclose(out);
+
+ check(context, strstr(buffer, "mov byte [rsi], dl") != NULL);
+ check(context, strstr(buffer, "mov dword [rsi], eax") != NULL);
+ check(context, strstr(buffer, "mov byte [rsi], 5") != NULL);
+ check(context, strstr(buffer, "mov [rsi], rbx") != NULL);
+
+ free_program(&program);
+}
+
void run_codegen_tests(struct TestContext* context)
{
test_generate_consts_and_data(context);
@@ -213,4 +238,5 @@ void run_codegen_tests(struct TestContext* context)
test_generate_divide(context);
test_generate_stack_frame(context);
test_generate_address_expr(context);
+ test_generate_sized_store(context);
}
diff --git a/tests/lexer_test.c b/tests/lexer_test.c
index 11b15b9..f677793 100644
--- a/tests/lexer_test.c
+++ b/tests/lexer_test.c
@@ -45,6 +45,15 @@ static void test_keywords(struct TestContext* context)
check(context, scan_token(&lexer).type == TOKEN_IDENTIFIER);
}
+static void test_size_keywords(struct TestContext* context)
+{
+ struct Lexer lexer = create_lexer("byte word dword qword");
+ check(context, scan_token(&lexer).type == TOKEN_BYTE);
+ check(context, scan_token(&lexer).type == TOKEN_WORD);
+ check(context, scan_token(&lexer).type == TOKEN_DWORD);
+ check(context, scan_token(&lexer).type == TOKEN_QWORD);
+}
+
static void test_line_counting(struct TestContext* context)
{
struct Lexer lexer = create_lexer("a\nb\nc");
@@ -70,6 +79,7 @@ void run_lexer_tests(struct TestContext* context)
test_operators(context);
test_literals(context);
test_keywords(context);
+ test_size_keywords(context);
test_line_counting(context);
test_comments(context);
}
diff --git a/tests/parser_test.c b/tests/parser_test.c
index bbbbaa2..61f8c37 100644
--- a/tests/parser_test.c
+++ b/tests/parser_test.c
@@ -188,6 +188,26 @@ static void test_parse_stack(struct TestContext* context)
free_program(&program);
}
+static void test_parse_sized_deref(struct TestContext* context)
+{
+ struct Lexer lexer = create_lexer("proc main\n{\n^byte rsi = rdx\n^rdi = rax\n}\n");
+ struct Program program;
+
+ check(context, parse_program(&lexer, &program));
+ check(context, program.procs[0].body_count == 2);
+
+ struct AssignStatement sized = program.procs[0].body[0].assign;
+ check(context, sized.target_deref);
+ check(context, sized.store_size == STORE_SIZE_BYTE);
+ check(context, text_is(sized.target, "rsi"));
+
+ struct AssignStatement plain = program.procs[0].body[1].assign;
+ check(context, plain.target_deref);
+ check(context, plain.store_size == STORE_SIZE_NONE);
+
+ free_program(&program);
+}
+
static void test_parse_errors(struct TestContext* context)
{
struct Program program;
@@ -216,5 +236,6 @@ void run_parser_tests(struct TestContext* context)
test_parse_if(context);
test_parse_call(context);
test_parse_stack(context);
+ test_parse_sized_deref(context);
test_parse_errors(context);
}