aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--Dockerfile2
-rw-r--r--docs/language.md12
-rw-r--r--examples/arm64/exit.hdass14
-rw-r--r--examples/arm64/sum.hdass19
-rw-r--r--src/args.c19
-rw-r--r--src/args.h11
-rw-r--r--src/codegen.c439
-rw-r--r--src/codegen.h1
-rw-r--r--src/main.c6
-rw-r--r--tests/codegen_test.c44
10 files changed, 552 insertions, 15 deletions
diff --git a/Dockerfile b/Dockerfile
index eb8ae2a..fcb2b43 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -5,6 +5,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
nasm \
fasm \
binutils \
+ binutils-aarch64-linux-gnu \
+ qemu-user \
meson \
ninja-build \
&& rm -rf /var/lib/apt/lists/*
diff --git a/docs/language.md b/docs/language.md
index 79d7866..a4743b3 100644
--- a/docs/language.md
+++ b/docs/language.md
@@ -1,6 +1,16 @@
# hdass language reference
-hdass emits NASM or FASM for x86-64 (`-t nasm` by default, `-t fasm`); masm is planned. The instruction bodies are the same Intel syntax for both — only the framing (headers, sections, constants, data) differs. The compiler output itself isn't tied to an OS, but the examples and toolchain here target Linux (Linux syscall numbers, ELF64, `ld`). Pipeline: `lex → parse → analyze → emit`.
+hdass has two independent axes: the **architecture** (which instructions and registers) and the **assembler syntax** (how they are written). A target is a pairing:
+
+| `-t` | architecture | assembler |
+| --- | --- | --- |
+| `nasm` (default) | x86-64 | NASM |
+| `fasm` | x86-64 | fasm |
+| `arm64` | AArch64 | GNU as |
+
+For x86-64 the two syntaxes emit the same Intel instruction bodies and differ only in framing (headers, sections, constants, data). `arm64` is a separate instruction selector — different registers, three-operand arithmetic, `ldr`/`str`, `cmp`+`b.cond`, `svc #0` — and is early: assignments, arithmetic (`+ - * /`), control flow, calls, `syscall`, and the raw instruction statement work; floats, stack buffers, and division-remainder do not yet. masm is planned. Pipeline: `lex → parse → analyze → emit`.
+
+The portable way to write for more than one architecture is the [`logical_registers`](#extensions) extension: `r1..r14` are the general-purpose registers, mapped per target (x86-64 `r1 = rax`; AArch64 `r1 = x0`, i.e. `rN → x(N-1)`). Architecture-native register names (`rax`, `x0`) and the raw instruction statement are, by definition, locked to one architecture. Syscall ABIs also differ per architecture — Linux exit is `60` in `rax` on x86-64 but `93` in `x8` (logical `r9`) on AArch64 — so programs still carry arch-specific ABI constants even when the language is portable. The output isn't tied to an OS, but the examples and toolchain here target Linux (ELF, `ld` / `qemu-aarch64`).
## A first program
diff --git a/examples/arm64/exit.hdass b/examples/arm64/exit.hdass
new file mode 100644
index 0000000..a86e5e2
--- /dev/null
+++ b/examples/arm64/exit.hdass
@@ -0,0 +1,14 @@
+[entry: main]
+[enable: logical_registers]
+
+// Exits with status 42 on AArch64 Linux. Logical registers map r1 -> x0 and
+// r9 -> x8, so the exit code goes in x0 and the syscall number in x8; `syscall`
+// lowers to `svc #0`.
+const SYS_EXIT = 93
+
+proc main
+{
+ r1 = 42 // x0 = exit code
+ r9 = SYS_EXIT // x8 = syscall number
+ syscall
+}
diff --git a/examples/arm64/sum.hdass b/examples/arm64/sum.hdass
new file mode 100644
index 0000000..85820c9
--- /dev/null
+++ b/examples/arm64/sum.hdass
@@ -0,0 +1,19 @@
+[entry: main]
+[enable: logical_registers]
+
+// Sums 5 + 4 + 3 + 2 + 1 = 15 with a named countdown loop, returns it as the
+// exit status. Shows three-operand arithmetic (r1 = r1 + r2 -> add x0, x0, x1).
+const SYS_EXIT = 93
+
+proc main
+{
+ r1 = 0
+ r2 = 5
+ while .countdown r2 > 0
+ {
+ r1 = r1 + r2
+ r2 = r2 - 1
+ }
+ r9 = SYS_EXIT
+ syscall
+}
diff --git a/src/args.c b/src/args.c
index b31def5..beb4c5d 100644
--- a/src/args.c
+++ b/src/args.c
@@ -6,21 +6,26 @@
#define HDASS_VERSION "0.1.0"
-static bool match_assembler(const char* name, enum Assembler* out)
+static bool match_target(const char* name, enum Target* out)
{
if (strcmp(name, "nasm") == 0)
{
- *out = ASSEMBLER_NASM;
+ *out = TARGET_NASM;
return true;
}
if (strcmp(name, "fasm") == 0)
{
- *out = ASSEMBLER_FASM;
+ *out = TARGET_FASM;
return true;
}
if (strcmp(name, "masm") == 0)
{
- *out = ASSEMBLER_MASM;
+ *out = TARGET_MASM;
+ return true;
+ }
+ if (strcmp(name, "arm64") == 0)
+ {
+ *out = TARGET_ARM64;
return true;
}
return false;
@@ -37,7 +42,7 @@ void print_usage(const char* program)
printf("usage: %s <input.hdass> [options]\n\n", program);
printf("options:\n");
printf(" -o, --output <file> write output to <file> (default: stdout)\n");
- printf(" -t, --target <name> target assembler: nasm, fasm, masm (default: nasm)\n");
+ printf(" -t, --target <name> target: nasm, fasm, masm, arm64 (default: nasm)\n");
printf(" -h, --help print this help and exit\n");
printf(" -v, --version print version and exit\n");
}
@@ -48,7 +53,7 @@ enum ParseResult parse_args(int argc, char** argv, struct Args* args)
args->input_path = NULL;
args->output_path = NULL;
- args->target = ASSEMBLER_NASM;
+ args->target = TARGET_NASM;
for (int i = 1; i < argc; i += 1)
{
@@ -86,7 +91,7 @@ enum ParseResult parse_args(int argc, char** argv, struct Args* args)
fprintf(stderr, "error: '%s' requires an argument\n", arg);
return PARSE_ERROR;
}
- if (!match_assembler(argv[i], &args->target))
+ if (!match_target(argv[i], &args->target))
{
fprintf(stderr, "error: unknown target '%s'\n", argv[i]);
return PARSE_ERROR;
diff --git a/src/args.h b/src/args.h
index f80b736..9f01ac4 100644
--- a/src/args.h
+++ b/src/args.h
@@ -1,10 +1,11 @@
#pragma once
-enum Assembler
+enum Target
{
- ASSEMBLER_NASM,
- ASSEMBLER_FASM,
- ASSEMBLER_MASM,
+ TARGET_NASM,
+ TARGET_FASM,
+ TARGET_MASM,
+ TARGET_ARM64,
};
enum ParseResult
@@ -18,7 +19,7 @@ struct Args
{
const char* input_path;
const char* output_path;
- enum Assembler target;
+ enum Target target;
};
enum ParseResult parse_args(int argc, char** argv, struct Args* args);
diff --git a/src/codegen.c b/src/codegen.c
index bb38f17..271c5ae 100644
--- a/src/codegen.c
+++ b/src/codegen.c
@@ -1328,6 +1328,372 @@ static void emit_proc_x86(struct Program* program, struct FloatTable* floats, st
}
}
+// ---------------------------------------------------------------------------
+// AArch64 target
+//
+// A separate instruction selector: the register model (logical rN -> xN-1, or
+// native x0.., w0.., sp, lr), 3-operand arithmetic, ldr/str memory, cmp + b.cond
+// control flow and svc #0 syscalls are all its own. Shares only the arch-neutral
+// helpers above (fold_const, resolve_token, the AST).
+// ---------------------------------------------------------------------------
+
+static bool is_a64_register(struct Token token)
+{
+ // logical rN (mapped to xN-1)
+ if (token.length >= 2 && token.start[0] == 'r' && token.start[1] >= '0' && token.start[1] <= '9')
+ {
+ for (size_t i = 1; i < token.length; i += 1)
+ if (token.start[i] < '0' || token.start[i] > '9')
+ return false;
+ return true;
+ }
+
+ // native names x0..x30 / w0..w30
+ if ((token.start[0] == 'x' || token.start[0] == 'w') && token.length >= 2
+ && token.start[1] >= '0' && token.start[1] <= '9')
+ return true;
+
+ return token_matches(token, "sp") || token_matches(token, "lr")
+ || token_matches(token, "fp") || token_matches(token, "xzr")
+ || token_matches(token, "wzr");
+}
+
+static void emit_a64_reg(struct Emitter* emitter, struct Token token)
+{
+ struct Token r = resolve_token(emitter, token);
+ if (r.length >= 2 && r.start[0] == 'r' && r.start[1] >= '0' && r.start[1] <= '9')
+ {
+ uint32_t index = 0;
+ for (size_t i = 1; i < r.length; i += 1)
+ index = index * 10 + (uint32_t)(r.start[i] - '0');
+ fprintf(emitter->out, "x%u", index - 1);
+ return;
+ }
+ fprintf(emitter->out, "%.*s", (int)r.length, r.start);
+}
+
+// an operand in register or immediate position: a register maps through, and
+// anything that folds to a constant becomes an #immediate
+static void emit_a64_operand(struct Emitter* emitter, struct Expr* expr)
+{
+ if (expr->kind == EXPR_PRIMARY && is_a64_register(resolve_token(emitter, expr->primary.token)))
+ {
+ emit_a64_reg(emitter, expr->primary.token);
+ return;
+ }
+
+ uint64_t value;
+ if (fold_const(emitter->program, expr, &value))
+ {
+ fprintf(emitter->out, "#%lld", (long long)value);
+ return;
+ }
+
+ if (expr->kind == EXPR_PRIMARY)
+ fprintf(emitter->out, "#%.*s", (int)expr->primary.token.length, expr->primary.token.start);
+ else
+ fprintf(emitter->out, "; TODO: unsupported operand");
+}
+
+static const char* a64_binop(enum TokenType op)
+{
+ switch (op)
+ {
+ case TOKEN_PLUS: case TOKEN_PLUS_EQUAL: return "add";
+ case TOKEN_MINUS: case TOKEN_MINUS_EQUAL: return "sub";
+ case TOKEN_STAR: case TOKEN_STAR_EQUAL: return "mul";
+ case TOKEN_SLASH: case TOKEN_SLASH_EQUAL: return "sdiv";
+ default: return NULL;
+ }
+}
+
+// branch taken when the comparison is false (to skip the guarded body)
+static const char* a64_jump_if_false(enum TokenType comparison)
+{
+ switch (comparison)
+ {
+ case TOKEN_EQUAL_EQUAL: return "ne";
+ case TOKEN_BANG_EQUAL: return "eq";
+ case TOKEN_LESS: return "ge";
+ case TOKEN_LESS_EQUAL: return "gt";
+ case TOKEN_GREATER: return "le";
+ case TOKEN_GREATER_EQUAL: return "lt";
+ default: return NULL;
+ }
+}
+
+static void emit_a64_statement(struct Emitter* emitter, struct Statement* statement);
+
+static void emit_a64_block(struct Emitter* emitter, struct Statement* body, size_t count)
+{
+ for (size_t i = 0; i < count; i += 1)
+ emit_a64_statement(emitter, &body[i]);
+}
+
+static void emit_a64_assign(struct Emitter* emitter, struct AssignStatement* assign)
+{
+ FILE* out = emitter->out;
+
+ // store through a pointer: ^[size] p = value
+ if (assign->target_deref)
+ {
+ const char* store = assign->store_size == STORE_SIZE_BYTE ? "strb"
+ : assign->store_size == STORE_SIZE_WORD ? "strh" : "str";
+ fprintf(out, "\t%s ", store);
+ emit_a64_operand(emitter, assign->value);
+ fprintf(out, ", [");
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, "]\n");
+ return;
+ }
+
+ struct Expr* value = assign->value;
+
+ // load through a pointer: dst = ^[size] p
+ if (assign->op.type == TOKEN_EQUAL && value->kind == EXPR_DEREF
+ && value->deref.address->kind == EXPR_PRIMARY)
+ {
+ const char* load = value->deref.size == STORE_SIZE_BYTE ? "ldrb"
+ : value->deref.size == STORE_SIZE_WORD ? "ldrh" : "ldr";
+ fprintf(out, "\t%s ", load);
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", [");
+ emit_a64_reg(emitter, value->deref.address->primary.token);
+ fprintf(out, "]\n");
+ return;
+ }
+
+ // three-operand arithmetic: dst = a op b
+ if (assign->op.type == TOKEN_EQUAL && value->kind == EXPR_BINARY)
+ {
+ const char* mnemonic = a64_binop(value->binary.op.type);
+ if (mnemonic == NULL)
+ {
+ fprintf(out, "\t; TODO: unsupported expression\n");
+ return;
+ }
+ fprintf(out, "\t%s ", mnemonic);
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", ");
+ emit_a64_operand(emitter, value->binary.left);
+ fprintf(out, ", ");
+ emit_a64_operand(emitter, value->binary.right);
+ fprintf(out, "\n");
+ return;
+ }
+
+ // compound assignment: dst op= value -> op dst, dst, value
+ if (assign->op.type != TOKEN_EQUAL)
+ {
+ const char* mnemonic = a64_binop(assign->op.type);
+ if (mnemonic == NULL)
+ {
+ fprintf(out, "\t; TODO: unsupported assignment\n");
+ return;
+ }
+ fprintf(out, "\t%s ", mnemonic);
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", ");
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", ");
+ emit_a64_operand(emitter, value);
+ fprintf(out, "\n");
+ return;
+ }
+
+ // plain move: dst = <register | immediate | symbol/address>
+ if (value->kind == EXPR_PRIMARY && is_a64_register(resolve_token(emitter, value->primary.token)))
+ {
+ fprintf(out, "\tmov ");
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", ");
+ emit_a64_reg(emitter, value->primary.token);
+ fprintf(out, "\n");
+ return;
+ }
+
+ uint64_t folded;
+ if (fold_const(emitter->program, value, &folded))
+ {
+ fprintf(out, "\tmov ");
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", #%lld\n", (long long)folded);
+ return;
+ }
+
+ // a data label or other symbol: load its address/value through the pool
+ if (value->kind == EXPR_PRIMARY)
+ {
+ fprintf(out, "\tldr ");
+ emit_a64_reg(emitter, assign->target);
+ fprintf(out, ", =%.*s\n", (int)value->primary.token.length, value->primary.token.start);
+ return;
+ }
+
+ fprintf(out, "\t; TODO: unsupported assignment\n");
+}
+
+static bool emit_a64_branch_test(struct Emitter* emitter, struct Expr* left,
+ struct Token comparison, struct Expr* right, const char* target)
+{
+ const char* cond = a64_jump_if_false(comparison.type);
+ if (cond == NULL || left->kind != EXPR_PRIMARY)
+ {
+ fprintf(emitter->out, "\t; TODO: unsupported condition\n");
+ return false;
+ }
+
+ fprintf(emitter->out, "\tcmp ");
+ emit_a64_operand(emitter, left);
+ fprintf(emitter->out, ", ");
+ emit_a64_operand(emitter, right);
+ fprintf(emitter->out, "\n\tb.%s %s\n", cond, target);
+ return true;
+}
+
+static void emit_a64_if(struct Emitter* emitter, struct IfStatement* branch)
+{
+ bool has_else = branch->else_count > 0;
+ uint32_t id = emitter->label_id;
+ emitter->label_id += 1;
+
+ char target[32];
+ snprintf(target, sizeof(target), ".if_%s_%u", has_else ? "else" : "end", id);
+
+ if (!emit_a64_branch_test(emitter, branch->left, branch->comparison, branch->right, target))
+ return;
+
+ emit_a64_block(emitter, branch->body, branch->body_count);
+
+ if (has_else)
+ {
+ fprintf(emitter->out, "\tb .if_end_%u\n", id);
+ fprintf(emitter->out, ".if_else_%u:\n", id);
+ emit_a64_block(emitter, branch->else_body, branch->else_count);
+ }
+
+ fprintf(emitter->out, ".if_end_%u:\n", id);
+}
+
+static void emit_a64_while(struct Emitter* emitter, struct WhileStatement* loop)
+{
+ char top[64];
+ char end[64];
+ if (loop->named)
+ {
+ snprintf(top, sizeof(top), ".%.*s", (int)loop->name.length, loop->name.start);
+ snprintf(end, sizeof(end), ".%.*s_end", (int)loop->name.length, loop->name.start);
+ }
+ else
+ {
+ uint32_t id = emitter->label_id;
+ emitter->label_id += 1;
+ snprintf(top, sizeof(top), ".while_%u", id);
+ snprintf(end, sizeof(end), ".while_end_%u", id);
+ }
+
+ fprintf(emitter->out, "%s:\n", top);
+ if (!emit_a64_branch_test(emitter, loop->left, loop->comparison, loop->right, end))
+ return;
+ emit_a64_block(emitter, loop->body, loop->body_count);
+ fprintf(emitter->out, "\tb %s\n", top);
+ fprintf(emitter->out, "%s:\n", end);
+}
+
+static void emit_a64_call(struct Emitter* emitter, struct CallStatement* call)
+{
+ const struct ProcDecl* callee = NULL;
+ for (size_t i = 0; i < emitter->program->proc_count; i += 1)
+ if (tokens_equal(emitter->program->procs[i].name, call->name))
+ callee = &emitter->program->procs[i];
+
+ if (callee != NULL)
+ for (size_t i = 0; i < call->arg_count && i < callee->param_count; i += 1)
+ {
+ fprintf(emitter->out, "\tmov ");
+ emit_a64_reg(emitter, callee->params[i].reg);
+ fprintf(emitter->out, ", ");
+ emit_a64_operand(emitter, call->args[i]);
+ fprintf(emitter->out, "\n");
+ }
+
+ fprintf(emitter->out, "\tbl %.*s\n", (int)call->name.length, call->name.start);
+}
+
+static void emit_a64_instruction(struct Emitter* emitter, struct InstructionStatement* insn)
+{
+ fprintf(emitter->out, "\t%.*s", (int)insn->mnemonic.length, insn->mnemonic.start);
+ for (size_t i = 0; i < insn->operand_count; i += 1)
+ {
+ struct Expr* operand = insn->operands[i];
+ fprintf(emitter->out, "%s", i == 0 ? " " : ", ");
+ if (operand->kind == EXPR_DEREF && operand->deref.address->kind == EXPR_PRIMARY)
+ {
+ fprintf(emitter->out, "[");
+ emit_a64_reg(emitter, operand->deref.address->primary.token);
+ fprintf(emitter->out, "]");
+ }
+ else
+ {
+ emit_a64_operand(emitter, operand);
+ }
+ }
+ fprintf(emitter->out, "\n");
+}
+
+static void emit_a64_statement(struct Emitter* emitter, struct Statement* statement)
+{
+ FILE* out = emitter->out;
+ switch (statement->kind)
+ {
+ case STATEMENT_ASSIGN:
+ emit_a64_assign(emitter, &statement->assign);
+ break;
+ case STATEMENT_LABEL:
+ fprintf(out, "%.*s:\n", (int)statement->label.name.length, statement->label.name.start);
+ break;
+ case STATEMENT_GOTO:
+ fprintf(out, "\tb %.*s\n", (int)statement->jump.label.length, statement->jump.label.start);
+ break;
+ case STATEMENT_SYSCALL:
+ fprintf(out, "\tsvc #0\n");
+ break;
+ case STATEMENT_IF:
+ emit_a64_if(emitter, &statement->branch);
+ break;
+ case STATEMENT_WHILE:
+ emit_a64_while(emitter, &statement->loop);
+ break;
+ case STATEMENT_CALL:
+ emit_a64_call(emitter, &statement->call);
+ break;
+ case STATEMENT_STACK:
+ fprintf(out, "\t; TODO: stack buffers not yet supported on aarch64\n");
+ break;
+ case STATEMENT_INSTRUCTION:
+ emit_a64_instruction(emitter, &statement->instruction);
+ break;
+ }
+}
+
+static void emit_proc_aarch64(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, bool is_entry, FILE* out)
+{
+ struct Emitter emitter;
+ emitter.program = program;
+ emitter.proc = proc;
+ emitter.floats = floats;
+ emitter.out = out;
+ emitter.label_id = 0;
+
+ fprintf(out, "%.*s:\n", (int)proc->name.length, proc->name.start);
+
+ for (size_t i = 0; i < proc->body_count; i += 1)
+ emit_a64_statement(&emitter, &proc->body[i]);
+
+ if (!is_entry)
+ fprintf(out, "\tret\n");
+}
+
// Instruction selection lives behind the Arch seam: turning a procedure's
// statements into a target's instructions (register model, mnemonics, stack
// frames) is all an architecture decides. The Backend below is the orthogonal
@@ -1342,6 +1708,10 @@ static const struct Arch x86_arch = {
emit_proc_x86,
};
+static const struct Arch aarch64_arch = {
+ emit_proc_aarch64,
+};
+
struct Backend
{
void (*prologue)(const struct Program* program, FILE* out);
@@ -1536,6 +1906,70 @@ static const struct Backend fasm_backend = {
fasm_boot_signature,
};
+// GNU as (the assembler for the ARM targets): different directives from the
+// Intel-syntax assemblers, but the same framing shape.
+static void gas_prologue(const struct Program* program, FILE* out)
+{
+ (void)program;
+ fprintf(out, ".arch armv8-a\n");
+}
+
+static void gas_constant(struct ConstDecl decl, FILE* out)
+{
+ fprintf(out, ".equ %.*s, ", (int)decl.name.length, decl.name.start);
+ emit_const_expr(decl.value, out);
+ fprintf(out, "\n");
+}
+
+static void gas_data_section(FILE* out)
+{
+ fprintf(out, ".data\n");
+}
+
+static void gas_string_data(struct DataDecl decl, FILE* out)
+{
+ // GNU as .ascii interprets the same C escapes NASM's backtick strings do,
+ // so the inner content passes through unchanged (no trailing NUL, matching)
+ fprintf(out, "%.*s: .ascii \"%.*s\"\n",
+ (int)decl.name.length, decl.name.start,
+ (int)(decl.value.length - 2), decl.value.start + 1);
+ fprintf(out, ".equ %.*s.len, . - %.*s\n",
+ (int)decl.name.length, decl.name.start,
+ (int)decl.name.length, decl.name.start);
+}
+
+static void gas_float_slot(size_t index, struct Token literal, FILE* out)
+{
+ fprintf(out, "__float%zu: .double %.*s\n", index, (int)literal.length, literal.start);
+}
+
+static void gas_text_section(FILE* out)
+{
+ fprintf(out, ".text\n");
+}
+
+static void gas_global(struct Token name, FILE* out)
+{
+ fprintf(out, ".global %.*s\n", (int)name.length, name.start);
+}
+
+static void gas_boot_signature(FILE* out)
+{
+ // boot sectors are an x86/BIOS concept; not meaningful for the ARM targets
+ (void)out;
+}
+
+static const struct Backend gas_backend = {
+ gas_prologue,
+ gas_constant,
+ gas_data_section,
+ gas_string_data,
+ gas_float_slot,
+ gas_text_section,
+ gas_global,
+ gas_boot_signature,
+};
+
// the entry procedure drops its trailing `ret`. It is the [entry: NAME] proc if
// given; otherwise a flat binary starts at its first proc.
static bool proc_is_entry(struct Program* program, size_t index)
@@ -1624,3 +2058,8 @@ void generate_fasm(struct Program* program, FILE* out)
{
generate(program, out, &x86_arch, &fasm_backend);
}
+
+void generate_aarch64(struct Program* program, FILE* out)
+{
+ generate(program, out, &aarch64_arch, &gas_backend);
+}
diff --git a/src/codegen.h b/src/codegen.h
index 38f37ee..dc183ea 100644
--- a/src/codegen.h
+++ b/src/codegen.h
@@ -6,3 +6,4 @@
void generate_nasm(struct Program* program, FILE* out);
void generate_fasm(struct Program* program, FILE* out);
+void generate_aarch64(struct Program* program, FILE* out);
diff --git a/src/main.c b/src/main.c
index f15af79..7825d08 100644
--- a/src/main.c
+++ b/src/main.c
@@ -18,7 +18,7 @@ int main(int argc, char** argv)
if (result == PARSE_ERROR)
return 1;
- if (args.target == ASSEMBLER_MASM)
+ if (args.target == TARGET_MASM)
{
report_error_message("the masm target is not implemented yet");
return 1;
@@ -62,8 +62,10 @@ int main(int argc, char** argv)
}
}
- if (args.target == ASSEMBLER_FASM)
+ if (args.target == TARGET_FASM)
generate_fasm(&program, out);
+ else if (args.target == TARGET_ARM64)
+ generate_aarch64(&program, out);
else
generate_nasm(&program, out);
diff --git a/tests/codegen_test.c b/tests/codegen_test.c
index 24cac6d..2b5fdb7 100644
--- a/tests/codegen_test.c
+++ b/tests/codegen_test.c
@@ -41,6 +41,24 @@ static void generate_fasm_to_buffer(struct Program* program, char* buffer, size_
fclose(out);
}
+static void generate_aarch64_to_buffer(struct Program* program, char* buffer, size_t size)
+{
+ FILE* out = tmpfile();
+ if (out == NULL)
+ {
+ buffer[0] = '\0';
+ return;
+ }
+
+ generate_aarch64(program, out);
+ fflush(out);
+ rewind(out);
+
+ size_t read = fread(buffer, 1, size - 1, out);
+ buffer[read] = '\0';
+ fclose(out);
+}
+
static void test_generate_consts_and_data(struct TestContext* context)
{
struct Lexer lexer = create_lexer("const N = 5\ndata msg = \"hi\"\n");
@@ -86,6 +104,31 @@ static void test_generate_fasm(struct TestContext* context)
free_program(&program);
}
+static void test_generate_aarch64(struct TestContext* context)
+{
+ // logical registers map r1 -> x0, r2 -> x1; three-operand arithmetic and svc
+ struct Lexer lexer = create_lexer(
+ "[enable: logical_registers]\nproc main\n{\nr1 = 5\nr1 = r1 + r2\n"
+ "while r2 > 0\nr2 = r2 - 1\nsyscall\n}\n");
+ struct Program program;
+ check(context, parse_program(&lexer, &program));
+
+ char buffer[1024];
+ generate_aarch64_to_buffer(&program, buffer, sizeof(buffer));
+
+ check(context, strstr(buffer, ".text") != NULL);
+ check(context, strstr(buffer, "mov x0, #5") != NULL); // r1 = 5
+ check(context, strstr(buffer, "add x0, x0, x1") != NULL); // r1 = r1 + r2
+ check(context, strstr(buffer, "cmp x1, #0") != NULL); // while r2 > 0
+ check(context, strstr(buffer, "sub x1, x1, #1") != NULL); // r2 = r2 - 1
+ check(context, strstr(buffer, "svc #0") != NULL); // syscall
+ // no x86 leaked in
+ check(context, strstr(buffer, "rax") == NULL);
+ check(context, strstr(buffer, "; TODO") == NULL);
+
+ free_program(&program);
+}
+
static void test_generate_text(struct TestContext* context)
{
struct Lexer lexer = create_lexer(
@@ -589,6 +632,7 @@ void run_codegen_tests(struct TestContext* context)
test_generate_no_entry(context);
test_generate_consts_and_data(context);
test_generate_fasm(context);
+ test_generate_aarch64(context);
test_generate_text(context);
test_generate_if(context);
test_generate_if_else(context);