diff options
| author | hachem <im@hachem.wtf> | 2026-09-10 04:42:33 +0200 |
|---|---|---|
| committer | hachem <im@hachem.wtf> | 2026-09-10 04:42:33 +0200 |
| commit | bf157cf09b8b0888c54a165b887309a4e801f608 (patch) | |
| tree | b72735972c2edb46c9324a74aeda9ff46c64682a | |
| parent | bf3bd19f79936fa519a5285701763da58935592a (diff) | |
feat: add aarch64 as target
| -rw-r--r-- | Dockerfile | 2 | ||||
| -rw-r--r-- | docs/language.md | 12 | ||||
| -rw-r--r-- | examples/arm64/exit.hdass | 14 | ||||
| -rw-r--r-- | examples/arm64/sum.hdass | 19 | ||||
| -rw-r--r-- | src/args.c | 19 | ||||
| -rw-r--r-- | src/args.h | 11 | ||||
| -rw-r--r-- | src/codegen.c | 439 | ||||
| -rw-r--r-- | src/codegen.h | 1 | ||||
| -rw-r--r-- | src/main.c | 6 | ||||
| -rw-r--r-- | tests/codegen_test.c | 44 |
10 files changed, 552 insertions, 15 deletions
@@ -5,6 +5,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ nasm \ fasm \ binutils \ + binutils-aarch64-linux-gnu \ + qemu-user \ meson \ ninja-build \ && rm -rf /var/lib/apt/lists/* diff --git a/docs/language.md b/docs/language.md index 79d7866..a4743b3 100644 --- a/docs/language.md +++ b/docs/language.md @@ -1,6 +1,16 @@ # hdass language reference -hdass emits NASM or FASM for x86-64 (`-t nasm` by default, `-t fasm`); masm is planned. The instruction bodies are the same Intel syntax for both — only the framing (headers, sections, constants, data) differs. The compiler output itself isn't tied to an OS, but the examples and toolchain here target Linux (Linux syscall numbers, ELF64, `ld`). Pipeline: `lex → parse → analyze → emit`. +hdass has two independent axes: the **architecture** (which instructions and registers) and the **assembler syntax** (how they are written). A target is a pairing: + +| `-t` | architecture | assembler | +| --- | --- | --- | +| `nasm` (default) | x86-64 | NASM | +| `fasm` | x86-64 | fasm | +| `arm64` | AArch64 | GNU as | + +For x86-64 the two syntaxes emit the same Intel instruction bodies and differ only in framing (headers, sections, constants, data). `arm64` is a separate instruction selector — different registers, three-operand arithmetic, `ldr`/`str`, `cmp`+`b.cond`, `svc #0` — and is early: assignments, arithmetic (`+ - * /`), control flow, calls, `syscall`, and the raw instruction statement work; floats, stack buffers, and division-remainder do not yet. masm is planned. Pipeline: `lex → parse → analyze → emit`. + +The portable way to write for more than one architecture is the [`logical_registers`](#extensions) extension: `r1..r14` are the general-purpose registers, mapped per target (x86-64 `r1 = rax`; AArch64 `r1 = x0`, i.e. `rN → x(N-1)`). Architecture-native register names (`rax`, `x0`) and the raw instruction statement are, by definition, locked to one architecture. Syscall ABIs also differ per architecture — Linux exit is `60` in `rax` on x86-64 but `93` in `x8` (logical `r9`) on AArch64 — so programs still carry arch-specific ABI constants even when the language is portable. The output isn't tied to an OS, but the examples and toolchain here target Linux (ELF, `ld` / `qemu-aarch64`). ## A first program diff --git a/examples/arm64/exit.hdass b/examples/arm64/exit.hdass new file mode 100644 index 0000000..a86e5e2 --- /dev/null +++ b/examples/arm64/exit.hdass @@ -0,0 +1,14 @@ +[entry: main] +[enable: logical_registers] + +// Exits with status 42 on AArch64 Linux. Logical registers map r1 -> x0 and +// r9 -> x8, so the exit code goes in x0 and the syscall number in x8; `syscall` +// lowers to `svc #0`. +const SYS_EXIT = 93 + +proc main +{ + r1 = 42 // x0 = exit code + r9 = SYS_EXIT // x8 = syscall number + syscall +} diff --git a/examples/arm64/sum.hdass b/examples/arm64/sum.hdass new file mode 100644 index 0000000..85820c9 --- /dev/null +++ b/examples/arm64/sum.hdass @@ -0,0 +1,19 @@ +[entry: main] +[enable: logical_registers] + +// Sums 5 + 4 + 3 + 2 + 1 = 15 with a named countdown loop, returns it as the +// exit status. Shows three-operand arithmetic (r1 = r1 + r2 -> add x0, x0, x1). +const SYS_EXIT = 93 + +proc main +{ + r1 = 0 + r2 = 5 + while .countdown r2 > 0 + { + r1 = r1 + r2 + r2 = r2 - 1 + } + r9 = SYS_EXIT + syscall +} @@ -6,21 +6,26 @@ #define HDASS_VERSION "0.1.0" -static bool match_assembler(const char* name, enum Assembler* out) +static bool match_target(const char* name, enum Target* out) { if (strcmp(name, "nasm") == 0) { - *out = ASSEMBLER_NASM; + *out = TARGET_NASM; return true; } if (strcmp(name, "fasm") == 0) { - *out = ASSEMBLER_FASM; + *out = TARGET_FASM; return true; } if (strcmp(name, "masm") == 0) { - *out = ASSEMBLER_MASM; + *out = TARGET_MASM; + return true; + } + if (strcmp(name, "arm64") == 0) + { + *out = TARGET_ARM64; return true; } return false; @@ -37,7 +42,7 @@ void print_usage(const char* program) printf("usage: %s <input.hdass> [options]\n\n", program); printf("options:\n"); printf(" -o, --output <file> write output to <file> (default: stdout)\n"); - printf(" -t, --target <name> target assembler: nasm, fasm, masm (default: nasm)\n"); + printf(" -t, --target <name> target: nasm, fasm, masm, arm64 (default: nasm)\n"); printf(" -h, --help print this help and exit\n"); printf(" -v, --version print version and exit\n"); } @@ -48,7 +53,7 @@ enum ParseResult parse_args(int argc, char** argv, struct Args* args) args->input_path = NULL; args->output_path = NULL; - args->target = ASSEMBLER_NASM; + args->target = TARGET_NASM; for (int i = 1; i < argc; i += 1) { @@ -86,7 +91,7 @@ enum ParseResult parse_args(int argc, char** argv, struct Args* args) fprintf(stderr, "error: '%s' requires an argument\n", arg); return PARSE_ERROR; } - if (!match_assembler(argv[i], &args->target)) + if (!match_target(argv[i], &args->target)) { fprintf(stderr, "error: unknown target '%s'\n", argv[i]); return PARSE_ERROR; @@ -1,10 +1,11 @@ #pragma once -enum Assembler +enum Target { - ASSEMBLER_NASM, - ASSEMBLER_FASM, - ASSEMBLER_MASM, + TARGET_NASM, + TARGET_FASM, + TARGET_MASM, + TARGET_ARM64, }; enum ParseResult @@ -18,7 +19,7 @@ struct Args { const char* input_path; const char* output_path; - enum Assembler target; + enum Target target; }; enum ParseResult parse_args(int argc, char** argv, struct Args* args); diff --git a/src/codegen.c b/src/codegen.c index bb38f17..271c5ae 100644 --- a/src/codegen.c +++ b/src/codegen.c @@ -1328,6 +1328,372 @@ static void emit_proc_x86(struct Program* program, struct FloatTable* floats, st } } +// --------------------------------------------------------------------------- +// AArch64 target +// +// A separate instruction selector: the register model (logical rN -> xN-1, or +// native x0.., w0.., sp, lr), 3-operand arithmetic, ldr/str memory, cmp + b.cond +// control flow and svc #0 syscalls are all its own. Shares only the arch-neutral +// helpers above (fold_const, resolve_token, the AST). +// --------------------------------------------------------------------------- + +static bool is_a64_register(struct Token token) +{ + // logical rN (mapped to xN-1) + if (token.length >= 2 && token.start[0] == 'r' && token.start[1] >= '0' && token.start[1] <= '9') + { + for (size_t i = 1; i < token.length; i += 1) + if (token.start[i] < '0' || token.start[i] > '9') + return false; + return true; + } + + // native names x0..x30 / w0..w30 + if ((token.start[0] == 'x' || token.start[0] == 'w') && token.length >= 2 + && token.start[1] >= '0' && token.start[1] <= '9') + return true; + + return token_matches(token, "sp") || token_matches(token, "lr") + || token_matches(token, "fp") || token_matches(token, "xzr") + || token_matches(token, "wzr"); +} + +static void emit_a64_reg(struct Emitter* emitter, struct Token token) +{ + struct Token r = resolve_token(emitter, token); + if (r.length >= 2 && r.start[0] == 'r' && r.start[1] >= '0' && r.start[1] <= '9') + { + uint32_t index = 0; + for (size_t i = 1; i < r.length; i += 1) + index = index * 10 + (uint32_t)(r.start[i] - '0'); + fprintf(emitter->out, "x%u", index - 1); + return; + } + fprintf(emitter->out, "%.*s", (int)r.length, r.start); +} + +// an operand in register or immediate position: a register maps through, and +// anything that folds to a constant becomes an #immediate +static void emit_a64_operand(struct Emitter* emitter, struct Expr* expr) +{ + if (expr->kind == EXPR_PRIMARY && is_a64_register(resolve_token(emitter, expr->primary.token))) + { + emit_a64_reg(emitter, expr->primary.token); + return; + } + + uint64_t value; + if (fold_const(emitter->program, expr, &value)) + { + fprintf(emitter->out, "#%lld", (long long)value); + return; + } + + if (expr->kind == EXPR_PRIMARY) + fprintf(emitter->out, "#%.*s", (int)expr->primary.token.length, expr->primary.token.start); + else + fprintf(emitter->out, "; TODO: unsupported operand"); +} + +static const char* a64_binop(enum TokenType op) +{ + switch (op) + { + case TOKEN_PLUS: case TOKEN_PLUS_EQUAL: return "add"; + case TOKEN_MINUS: case TOKEN_MINUS_EQUAL: return "sub"; + case TOKEN_STAR: case TOKEN_STAR_EQUAL: return "mul"; + case TOKEN_SLASH: case TOKEN_SLASH_EQUAL: return "sdiv"; + default: return NULL; + } +} + +// branch taken when the comparison is false (to skip the guarded body) +static const char* a64_jump_if_false(enum TokenType comparison) +{ + switch (comparison) + { + case TOKEN_EQUAL_EQUAL: return "ne"; + case TOKEN_BANG_EQUAL: return "eq"; + case TOKEN_LESS: return "ge"; + case TOKEN_LESS_EQUAL: return "gt"; + case TOKEN_GREATER: return "le"; + case TOKEN_GREATER_EQUAL: return "lt"; + default: return NULL; + } +} + +static void emit_a64_statement(struct Emitter* emitter, struct Statement* statement); + +static void emit_a64_block(struct Emitter* emitter, struct Statement* body, size_t count) +{ + for (size_t i = 0; i < count; i += 1) + emit_a64_statement(emitter, &body[i]); +} + +static void emit_a64_assign(struct Emitter* emitter, struct AssignStatement* assign) +{ + FILE* out = emitter->out; + + // store through a pointer: ^[size] p = value + if (assign->target_deref) + { + const char* store = assign->store_size == STORE_SIZE_BYTE ? "strb" + : assign->store_size == STORE_SIZE_WORD ? "strh" : "str"; + fprintf(out, "\t%s ", store); + emit_a64_operand(emitter, assign->value); + fprintf(out, ", ["); + emit_a64_reg(emitter, assign->target); + fprintf(out, "]\n"); + return; + } + + struct Expr* value = assign->value; + + // load through a pointer: dst = ^[size] p + if (assign->op.type == TOKEN_EQUAL && value->kind == EXPR_DEREF + && value->deref.address->kind == EXPR_PRIMARY) + { + const char* load = value->deref.size == STORE_SIZE_BYTE ? "ldrb" + : value->deref.size == STORE_SIZE_WORD ? "ldrh" : "ldr"; + fprintf(out, "\t%s ", load); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", ["); + emit_a64_reg(emitter, value->deref.address->primary.token); + fprintf(out, "]\n"); + return; + } + + // three-operand arithmetic: dst = a op b + if (assign->op.type == TOKEN_EQUAL && value->kind == EXPR_BINARY) + { + const char* mnemonic = a64_binop(value->binary.op.type); + if (mnemonic == NULL) + { + fprintf(out, "\t; TODO: unsupported expression\n"); + return; + } + fprintf(out, "\t%s ", mnemonic); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", "); + emit_a64_operand(emitter, value->binary.left); + fprintf(out, ", "); + emit_a64_operand(emitter, value->binary.right); + fprintf(out, "\n"); + return; + } + + // compound assignment: dst op= value -> op dst, dst, value + if (assign->op.type != TOKEN_EQUAL) + { + const char* mnemonic = a64_binop(assign->op.type); + if (mnemonic == NULL) + { + fprintf(out, "\t; TODO: unsupported assignment\n"); + return; + } + fprintf(out, "\t%s ", mnemonic); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", "); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", "); + emit_a64_operand(emitter, value); + fprintf(out, "\n"); + return; + } + + // plain move: dst = <register | immediate | symbol/address> + if (value->kind == EXPR_PRIMARY && is_a64_register(resolve_token(emitter, value->primary.token))) + { + fprintf(out, "\tmov "); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", "); + emit_a64_reg(emitter, value->primary.token); + fprintf(out, "\n"); + return; + } + + uint64_t folded; + if (fold_const(emitter->program, value, &folded)) + { + fprintf(out, "\tmov "); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", #%lld\n", (long long)folded); + return; + } + + // a data label or other symbol: load its address/value through the pool + if (value->kind == EXPR_PRIMARY) + { + fprintf(out, "\tldr "); + emit_a64_reg(emitter, assign->target); + fprintf(out, ", =%.*s\n", (int)value->primary.token.length, value->primary.token.start); + return; + } + + fprintf(out, "\t; TODO: unsupported assignment\n"); +} + +static bool emit_a64_branch_test(struct Emitter* emitter, struct Expr* left, + struct Token comparison, struct Expr* right, const char* target) +{ + const char* cond = a64_jump_if_false(comparison.type); + if (cond == NULL || left->kind != EXPR_PRIMARY) + { + fprintf(emitter->out, "\t; TODO: unsupported condition\n"); + return false; + } + + fprintf(emitter->out, "\tcmp "); + emit_a64_operand(emitter, left); + fprintf(emitter->out, ", "); + emit_a64_operand(emitter, right); + fprintf(emitter->out, "\n\tb.%s %s\n", cond, target); + return true; +} + +static void emit_a64_if(struct Emitter* emitter, struct IfStatement* branch) +{ + bool has_else = branch->else_count > 0; + uint32_t id = emitter->label_id; + emitter->label_id += 1; + + char target[32]; + snprintf(target, sizeof(target), ".if_%s_%u", has_else ? "else" : "end", id); + + if (!emit_a64_branch_test(emitter, branch->left, branch->comparison, branch->right, target)) + return; + + emit_a64_block(emitter, branch->body, branch->body_count); + + if (has_else) + { + fprintf(emitter->out, "\tb .if_end_%u\n", id); + fprintf(emitter->out, ".if_else_%u:\n", id); + emit_a64_block(emitter, branch->else_body, branch->else_count); + } + + fprintf(emitter->out, ".if_end_%u:\n", id); +} + +static void emit_a64_while(struct Emitter* emitter, struct WhileStatement* loop) +{ + char top[64]; + char end[64]; + if (loop->named) + { + snprintf(top, sizeof(top), ".%.*s", (int)loop->name.length, loop->name.start); + snprintf(end, sizeof(end), ".%.*s_end", (int)loop->name.length, loop->name.start); + } + else + { + uint32_t id = emitter->label_id; + emitter->label_id += 1; + snprintf(top, sizeof(top), ".while_%u", id); + snprintf(end, sizeof(end), ".while_end_%u", id); + } + + fprintf(emitter->out, "%s:\n", top); + if (!emit_a64_branch_test(emitter, loop->left, loop->comparison, loop->right, end)) + return; + emit_a64_block(emitter, loop->body, loop->body_count); + fprintf(emitter->out, "\tb %s\n", top); + fprintf(emitter->out, "%s:\n", end); +} + +static void emit_a64_call(struct Emitter* emitter, struct CallStatement* call) +{ + const struct ProcDecl* callee = NULL; + for (size_t i = 0; i < emitter->program->proc_count; i += 1) + if (tokens_equal(emitter->program->procs[i].name, call->name)) + callee = &emitter->program->procs[i]; + + if (callee != NULL) + for (size_t i = 0; i < call->arg_count && i < callee->param_count; i += 1) + { + fprintf(emitter->out, "\tmov "); + emit_a64_reg(emitter, callee->params[i].reg); + fprintf(emitter->out, ", "); + emit_a64_operand(emitter, call->args[i]); + fprintf(emitter->out, "\n"); + } + + fprintf(emitter->out, "\tbl %.*s\n", (int)call->name.length, call->name.start); +} + +static void emit_a64_instruction(struct Emitter* emitter, struct InstructionStatement* insn) +{ + fprintf(emitter->out, "\t%.*s", (int)insn->mnemonic.length, insn->mnemonic.start); + for (size_t i = 0; i < insn->operand_count; i += 1) + { + struct Expr* operand = insn->operands[i]; + fprintf(emitter->out, "%s", i == 0 ? " " : ", "); + if (operand->kind == EXPR_DEREF && operand->deref.address->kind == EXPR_PRIMARY) + { + fprintf(emitter->out, "["); + emit_a64_reg(emitter, operand->deref.address->primary.token); + fprintf(emitter->out, "]"); + } + else + { + emit_a64_operand(emitter, operand); + } + } + fprintf(emitter->out, "\n"); +} + +static void emit_a64_statement(struct Emitter* emitter, struct Statement* statement) +{ + FILE* out = emitter->out; + switch (statement->kind) + { + case STATEMENT_ASSIGN: + emit_a64_assign(emitter, &statement->assign); + break; + case STATEMENT_LABEL: + fprintf(out, "%.*s:\n", (int)statement->label.name.length, statement->label.name.start); + break; + case STATEMENT_GOTO: + fprintf(out, "\tb %.*s\n", (int)statement->jump.label.length, statement->jump.label.start); + break; + case STATEMENT_SYSCALL: + fprintf(out, "\tsvc #0\n"); + break; + case STATEMENT_IF: + emit_a64_if(emitter, &statement->branch); + break; + case STATEMENT_WHILE: + emit_a64_while(emitter, &statement->loop); + break; + case STATEMENT_CALL: + emit_a64_call(emitter, &statement->call); + break; + case STATEMENT_STACK: + fprintf(out, "\t; TODO: stack buffers not yet supported on aarch64\n"); + break; + case STATEMENT_INSTRUCTION: + emit_a64_instruction(emitter, &statement->instruction); + break; + } +} + +static void emit_proc_aarch64(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, bool is_entry, FILE* out) +{ + struct Emitter emitter; + emitter.program = program; + emitter.proc = proc; + emitter.floats = floats; + emitter.out = out; + emitter.label_id = 0; + + fprintf(out, "%.*s:\n", (int)proc->name.length, proc->name.start); + + for (size_t i = 0; i < proc->body_count; i += 1) + emit_a64_statement(&emitter, &proc->body[i]); + + if (!is_entry) + fprintf(out, "\tret\n"); +} + // Instruction selection lives behind the Arch seam: turning a procedure's // statements into a target's instructions (register model, mnemonics, stack // frames) is all an architecture decides. The Backend below is the orthogonal @@ -1342,6 +1708,10 @@ static const struct Arch x86_arch = { emit_proc_x86, }; +static const struct Arch aarch64_arch = { + emit_proc_aarch64, +}; + struct Backend { void (*prologue)(const struct Program* program, FILE* out); @@ -1536,6 +1906,70 @@ static const struct Backend fasm_backend = { fasm_boot_signature, }; +// GNU as (the assembler for the ARM targets): different directives from the +// Intel-syntax assemblers, but the same framing shape. +static void gas_prologue(const struct Program* program, FILE* out) +{ + (void)program; + fprintf(out, ".arch armv8-a\n"); +} + +static void gas_constant(struct ConstDecl decl, FILE* out) +{ + fprintf(out, ".equ %.*s, ", (int)decl.name.length, decl.name.start); + emit_const_expr(decl.value, out); + fprintf(out, "\n"); +} + +static void gas_data_section(FILE* out) +{ + fprintf(out, ".data\n"); +} + +static void gas_string_data(struct DataDecl decl, FILE* out) +{ + // GNU as .ascii interprets the same C escapes NASM's backtick strings do, + // so the inner content passes through unchanged (no trailing NUL, matching) + fprintf(out, "%.*s: .ascii \"%.*s\"\n", + (int)decl.name.length, decl.name.start, + (int)(decl.value.length - 2), decl.value.start + 1); + fprintf(out, ".equ %.*s.len, . - %.*s\n", + (int)decl.name.length, decl.name.start, + (int)decl.name.length, decl.name.start); +} + +static void gas_float_slot(size_t index, struct Token literal, FILE* out) +{ + fprintf(out, "__float%zu: .double %.*s\n", index, (int)literal.length, literal.start); +} + +static void gas_text_section(FILE* out) +{ + fprintf(out, ".text\n"); +} + +static void gas_global(struct Token name, FILE* out) +{ + fprintf(out, ".global %.*s\n", (int)name.length, name.start); +} + +static void gas_boot_signature(FILE* out) +{ + // boot sectors are an x86/BIOS concept; not meaningful for the ARM targets + (void)out; +} + +static const struct Backend gas_backend = { + gas_prologue, + gas_constant, + gas_data_section, + gas_string_data, + gas_float_slot, + gas_text_section, + gas_global, + gas_boot_signature, +}; + // the entry procedure drops its trailing `ret`. It is the [entry: NAME] proc if // given; otherwise a flat binary starts at its first proc. static bool proc_is_entry(struct Program* program, size_t index) @@ -1624,3 +2058,8 @@ void generate_fasm(struct Program* program, FILE* out) { generate(program, out, &x86_arch, &fasm_backend); } + +void generate_aarch64(struct Program* program, FILE* out) +{ + generate(program, out, &aarch64_arch, &gas_backend); +} diff --git a/src/codegen.h b/src/codegen.h index 38f37ee..dc183ea 100644 --- a/src/codegen.h +++ b/src/codegen.h @@ -6,3 +6,4 @@ void generate_nasm(struct Program* program, FILE* out); void generate_fasm(struct Program* program, FILE* out); +void generate_aarch64(struct Program* program, FILE* out); @@ -18,7 +18,7 @@ int main(int argc, char** argv) if (result == PARSE_ERROR) return 1; - if (args.target == ASSEMBLER_MASM) + if (args.target == TARGET_MASM) { report_error_message("the masm target is not implemented yet"); return 1; @@ -62,8 +62,10 @@ int main(int argc, char** argv) } } - if (args.target == ASSEMBLER_FASM) + if (args.target == TARGET_FASM) generate_fasm(&program, out); + else if (args.target == TARGET_ARM64) + generate_aarch64(&program, out); else generate_nasm(&program, out); diff --git a/tests/codegen_test.c b/tests/codegen_test.c index 24cac6d..2b5fdb7 100644 --- a/tests/codegen_test.c +++ b/tests/codegen_test.c @@ -41,6 +41,24 @@ static void generate_fasm_to_buffer(struct Program* program, char* buffer, size_ fclose(out); } +static void generate_aarch64_to_buffer(struct Program* program, char* buffer, size_t size) +{ + FILE* out = tmpfile(); + if (out == NULL) + { + buffer[0] = '\0'; + return; + } + + generate_aarch64(program, out); + fflush(out); + rewind(out); + + size_t read = fread(buffer, 1, size - 1, out); + buffer[read] = '\0'; + fclose(out); +} + static void test_generate_consts_and_data(struct TestContext* context) { struct Lexer lexer = create_lexer("const N = 5\ndata msg = \"hi\"\n"); @@ -86,6 +104,31 @@ static void test_generate_fasm(struct TestContext* context) free_program(&program); } +static void test_generate_aarch64(struct TestContext* context) +{ + // logical registers map r1 -> x0, r2 -> x1; three-operand arithmetic and svc + struct Lexer lexer = create_lexer( + "[enable: logical_registers]\nproc main\n{\nr1 = 5\nr1 = r1 + r2\n" + "while r2 > 0\nr2 = r2 - 1\nsyscall\n}\n"); + struct Program program; + check(context, parse_program(&lexer, &program)); + + char buffer[1024]; + generate_aarch64_to_buffer(&program, buffer, sizeof(buffer)); + + check(context, strstr(buffer, ".text") != NULL); + check(context, strstr(buffer, "mov x0, #5") != NULL); // r1 = 5 + check(context, strstr(buffer, "add x0, x0, x1") != NULL); // r1 = r1 + r2 + check(context, strstr(buffer, "cmp x1, #0") != NULL); // while r2 > 0 + check(context, strstr(buffer, "sub x1, x1, #1") != NULL); // r2 = r2 - 1 + check(context, strstr(buffer, "svc #0") != NULL); // syscall + // no x86 leaked in + check(context, strstr(buffer, "rax") == NULL); + check(context, strstr(buffer, "; TODO") == NULL); + + free_program(&program); +} + static void test_generate_text(struct TestContext* context) { struct Lexer lexer = create_lexer( @@ -589,6 +632,7 @@ void run_codegen_tests(struct TestContext* context) test_generate_no_entry(context); test_generate_consts_and_data(context); test_generate_fasm(context); + test_generate_aarch64(context); test_generate_text(context); test_generate_if(context); test_generate_if_else(context); |
