diff options
| author | hachem <im@hachem.wtf> | 2026-09-09 05:14:44 +0200 |
|---|---|---|
| committer | hachem <im@hachem.wtf> | 2026-09-09 05:14:44 +0200 |
| commit | 8b76d35b0a045e0a3277be422513b938ef33dcce (patch) | |
| tree | b876d92ba3afb810e38f3f83792fc3fb849dd457 | |
| parent | b9eff51ee157b22d8fa46aeecac7b95d5ed345d9 (diff) | |
feat: fasm backend
| -rw-r--r-- | Dockerfile | 1 | ||||
| -rw-r--r-- | README.md | 11 | ||||
| -rw-r--r-- | docs/language.md | 21 | ||||
| -rw-r--r-- | examples/fibonacci.hdass | 16 | ||||
| -rw-r--r-- | examples/loop_sum.hdass | 10 | ||||
| -rw-r--r-- | examples/records.hdass | 6 | ||||
| -rw-r--r-- | meson.build | 2 | ||||
| -rwxr-xr-x | scripts/dump_asm.sh | 34 | ||||
| -rwxr-xr-x | scripts/run_suite.sh | 4 | ||||
| -rwxr-xr-x | scripts/test_examples.sh | 19 | ||||
| -rw-r--r-- | src/ast.h | 2 | ||||
| -rw-r--r-- | src/codegen.c (renamed from src/nasm.c) | 261 | ||||
| -rw-r--r-- | src/codegen.h (renamed from src/nasm.h) | 1 | ||||
| -rw-r--r-- | src/main.c | 11 | ||||
| -rw-r--r-- | src/parser.c | 10 | ||||
| -rw-r--r-- | tests/codegen_test.c | 68 | ||||
| -rw-r--r-- | tests/parser_test.c | 18 |
17 files changed, 412 insertions, 83 deletions
@@ -3,6 +3,7 @@ FROM --platform=linux/amd64 debian:trixie-slim RUN apt-get update && apt-get install -y --no-install-recommends \ build-essential \ nasm \ + fasm \ binutils \ meson \ ninja-build \ @@ -3,7 +3,7 @@ hachem's dumb assembly super-set, pronounced "HD Ass", or "headass"...depends on The entire point of this project is to provide a middle ground between C and assembly. If we think about why we still write assembly today, it's usually because we need direct control over what the CPU is doing. We want control over the exact instructions being executed, the memory, the stack and everything else that higher-level programming languages normally abstract away. The thing is, not every program written in assembly actually needs that control. Sometimes you want to write something close to the machine without having to manuall deal with every tiny detail yourself. You still want registers, explicit control over memory and a good understanding of what your program is doing, but you don't necessarily need to manually express everything as individual assembly instructions. -This is where this piece of shit comes in. It's not quite high-level enough to be a C-like language, but it's also not low-level enough to be as annoying to write as raw assembly. The goal is to sit somewhere in between, keeping the parts of assembly that make it useful while making the parts that don't need to be painful a little nicer to work with. hdass transpiles into multiple flavours of assembly, such as NASM and MASM, rather than directly producing machine code. The idea is to provide a single language for writing low-level programs while allowing the backend to translate code into the assembler syntax you want to target. You're still ultimately producing assembly, and you're never particularly far away from the code that gets assembled. The goal isn't to hide the machine from you or turn assembly into C. There are already plenty of high-level languages that do that. hdass just fills the gap between the two, where you might want some more convenience whilst writing assembly without taking away the reason you wanted to work close to the machine in the first place. +This is where this piece of shit comes in. It's not quite high-level enough to be a C-like language, but it's also not low-level enough to be as annoying to write as raw assembly. The goal is to sit somewhere in between, keeping the parts of assembly that make it useful while making the parts that don't need to be painful a little nicer to work with. hdass transpiles into multiple flavours of assembly, currently NASM and FASM (MASM is planned), rather than directly producing machine code. The idea is to provide a single language for writing low-level programs while allowing the backend to translate code into the assembler syntax you want to target. You're still ultimately producing assembly, and you're never particularly far away from the code that gets assembled. The goal isn't to hide the machine from you or turn assembly into C. There are already plenty of high-level languages that do that. hdass just fills the gap between the two, where you might want some more convenience whilst writing assembly without taking away the reason you wanted to work close to the machine in the first place. ## Examples Here's a simple "Hello, World!" world program written using hdass' syntax: @@ -53,7 +53,7 @@ meson test -C build ``` If [cppcheck](https://cppcheck.sourceforge.io/) is installed, `ninja -C build cppcheck` runs static analysis over the sources. -hdass emits x86-64 assembly, so to actually assemble and run its output you need an x86-64 Linux toolchain. The bundled Docker environment provides a consistent one on any host, including Apple Silicon, where the amd64 image runs under emulation. The image is a Debian base with `nasm`, `ld` (binutils), a C toolchain and Meson/Ninja preinstalled. +hdass emits x86-64 assembly, so to actually assemble and run its output you need an x86-64 Linux toolchain. The bundled Docker environment provides a consistent one on any host, including Apple Silicon, where the amd64 image runs under emulation. The image is a Debian base with `nasm`, `fasm`, `ld` (binutils), a C toolchain and Meson/Ninja preinstalled. Start the container (this builds the image the first time): ```bash @@ -74,6 +74,13 @@ ld -e main hello.o -o hello ./hello ``` +hdass targets NASM by default; pass `-t fasm` to emit FASM instead, which assembles in a single step: +```bash +./build-linux/hdass -t fasm examples/hello_world.hdass -o hello.asm +fasm hello.asm hello.o +ld -e main hello.o -o hello +``` + To transpile, assemble, link and run every program in [examples/](examples/) and check its output, use the end-to-end test script (also from inside the container): ```bash meson setup build-linux && meson compile -C build-linux && ./scripts/test_examples.sh diff --git a/docs/language.md b/docs/language.md index 42b178e..79d7866 100644 --- a/docs/language.md +++ b/docs/language.md @@ -1,6 +1,6 @@ # hdass language reference -hdass emits NASM for x86-64; fasm and masm are planned. The compiler output itself isn't tied to an OS, but the examples and toolchain here target Linux (Linux syscall numbers, `nasm -f elf64`, `ld`). Pipeline: `lex → parse → analyze → emit`. +hdass emits NASM or FASM for x86-64 (`-t nasm` by default, `-t fasm`); masm is planned. The instruction bodies are the same Intel syntax for both — only the framing (headers, sections, constants, data) differs. The compiler output itself isn't tied to an OS, but the examples and toolchain here target Linux (Linux syscall numbers, ELF64, `ld`). Pipeline: `lex → parse → analyze → emit`. ## A first program @@ -154,6 +154,13 @@ while rcx > 0 } ``` +An optional `.name` right after `while` names the loop's generated labels, so they read as `.name` (top) and `.name_end` (exit) instead of the anonymous `.while_N` — handy for finding a loop in the emitted assembly. Give nested loops distinct names. + +```hdass +while .countdown rcx > 0 // emits `.countdown:` … `jmp .countdown` … `.countdown_end:` + rcx -= 1 +``` + ## Dereference (`^`) `^reg` is the memory at the address in `reg` — NASM's `[reg]`. On the left of `=` it stores there. The store width comes from the value operand, so a sized sub-register picks the size: @@ -262,14 +269,22 @@ See [examples/mandelbrot.hdass](../examples/mandelbrot.hdass) for a float progra ## Building a program ```bash -hdass program.hdass -o program.asm +hdass program.hdass -o program.asm # nasm (default) nasm -f elf64 program.asm -o program.o ld -e main program.o -o program ./program ``` +Or target fasm with `-t fasm`, which assembles in one step: + +```bash +hdass -t fasm program.hdass -o program.asm +fasm program.asm program.o +ld -e main program.o -o program +``` + The [README](../README.md) has a Docker setup with these tools. ## Some stinkies -Clobbering is your responsibility: `syscall` trashes `rcx` and `r11`, while a callee can trash any registers it touches, so nothing is saved automatically. `examples/fibonacci.hdass`, for example, keeps its counter in `r15` for this reason. Register widths must also match, meaning something like `rax = r1.8` would become `mov rax, al`, which will not assemble. Division clobbers extra registers: `/` `%` and their `=` forms use `idiv` through `rax:rdx`, so both are overwritten regardless of the destination. The divisor can be anything — a register, a constant, or an immediate — but an immediate or an `rax`/`rdx` divisor is first copied into `r11`, so those also clobber `r11`. Finally, the entry procedure has no `ret`; it should end with an exit syscall. +Clobbering is your responsibility: `syscall` trashes `rcx` and `r11`, while a callee can trash any registers it touches, so nothing is saved automatically. `examples/fibonacci.hdass`, for example, keeps its counter in `r15` for this reason. Register widths must also match, meaning something like `rax = r1.8` would become `mov rax, al`, which will not assemble. Division clobbers extra registers: `/` `%` and their `=` forms use `idiv` through `rax:rdx`, so both are overwritten regardless of the destination. The divisor can be anything — a register, a constant, or an immediate — but an immediate or an `rax`/`rdx` divisor is first copied into `r11`, so those also clobber `r11`. Labels and procedures become plain assembler symbols, so avoid names the target assembler reserves: `loop`, for instance, is an instruction mnemonic that fasm rejects as a label (nasm allows it). Finally, the entry procedure has no `ret`; it should end with an exit syscall. diff --git a/examples/fibonacci.hdass b/examples/fibonacci.hdass index 9436580..47224fb 100644 --- a/examples/fibonacci.hdass +++ b/examples/fibonacci.hdass @@ -45,16 +45,16 @@ proc main r13 = 1 r15 = 10 // syscall clobbers rcx/r11, so keep the counter in r15 -loop: - print_number(r12) + while .sequence r15 > 0 + { + print_number(r12) - rax = r12 - r12 = r13 - r13 += rax + rax = r12 + r12 = r13 + r13 += rax - r15 -= 1 - if r15 != 0 - goto loop + r15 -= 1 + } rax = SYS_EXIT rdi = 0 diff --git a/examples/loop_sum.hdass b/examples/loop_sum.hdass index 2427b17..130f055 100644 --- a/examples/loop_sum.hdass +++ b/examples/loop_sum.hdass @@ -8,11 +8,11 @@ proc main rbx = 0 // running total rcx = 5 // counter -loop: - rbx += rcx - rcx -= 1 - if rcx != 0 - goto loop + while .countdown rcx > 0 + { + rbx += rcx + rcx -= 1 + } rdi = rbx rax = SYS_EXIT diff --git a/examples/records.hdass b/examples/records.hdass index f8553d6..8e87a4c 100644 --- a/examples/records.hdass +++ b/examples/records.hdass @@ -22,15 +22,17 @@ proc main stack pair[Pair.size] rsi = pair + rsi += Pair.a rbx = 40 - ^rsi = rbx // pair.a (offset 0) + ^rsi = rbx // pair.a rsi = pair rsi += Pair.b rcx = Status.Fail // 2 - ^rsi = rcx // pair.b (offset 8) + ^rsi = rcx // pair.b rsi = pair + rsi += Pair.a rdi = ^rsi // load a = 40 rsi = pair rsi += Pair.b diff --git a/meson.build b/meson.build index b445e94..75226c3 100644 --- a/meson.build +++ b/meson.build @@ -36,7 +36,7 @@ core = static_library( 'src/diag.c', 'src/file.c', 'src/lexer.c', - 'src/nasm.c', + 'src/codegen.c', 'src/parser.c', 'src/sema.c', ), diff --git a/scripts/dump_asm.sh b/scripts/dump_asm.sh new file mode 100755 index 0000000..cb8a710 --- /dev/null +++ b/scripts/dump_asm.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +set -euo pipefail + +root="$(cd "$(dirname "$0")/.." && pwd)" +cd "$root" + +assemblers="${1:-${ASSEMBLER:-nasm fasm}}" + +docker compose up -d >/dev/null + +docker compose exec -T -e ASMS="$assemblers" hdass bash -c ' + set -e + cd /hdass + [ -d build-linux ] || meson setup build-linux >/dev/null + meson compile -C build-linux >/dev/null + +cat > /tmp/sidebyside.awk <<"AWK" +NR == FNR { left[FNR] = $0; if (length($0) > width) width = length($0); ln = FNR; next } +{ right[FNR] = $0; if (FNR > rn) rn = FNR } +END { + total = ln > rn ? ln : rn + for (i = 1; i <= total; i += 1) + printf "%-*s | %s\n", width, (i in left ? left[i] : ""), (i in right ? right[i] : "") +} +AWK + + for asm in $ASMS; do + for src in examples/*.hdass; do + printf "\n===== %s (%s) =====\n" "$(basename "$src")" "$asm" + ./build-linux/hdass -t "$asm" "$src" -o /tmp/gen.asm + awk -f /tmp/sidebyside.awk <(expand -t 4 "$src") <(expand -t 4 /tmp/gen.asm) + done + done +' </dev/null diff --git a/scripts/run_suite.sh b/scripts/run_suite.sh index 38598ad..56854e5 100755 --- a/scripts/run_suite.sh +++ b/scripts/run_suite.sh @@ -27,4 +27,6 @@ else exit 1 fi -exec bash ./scripts/test_examples.sh +ASSEMBLER=nasm bash ./scripts/test_examples.sh || exit 1 +printf '\n' +ASSEMBLER=fasm bash ./scripts/test_examples.sh diff --git a/scripts/test_examples.sh b/scripts/test_examples.sh index f8d1462..ab4f184 100755 --- a/scripts/test_examples.sh +++ b/scripts/test_examples.sh @@ -17,13 +17,24 @@ if [ ! -x "$hdass" ]; then exit 1 fi -for tool in nasm ld; do +assembler="${ASSEMBLER:-nasm}" + +for tool in "$assembler" ld; do if ! command -v "$tool" >/dev/null 2>&1; then echo "${red}error:${reset} '$tool' not found; run this inside the Docker environment" >&2 exit 1 fi done +assemble() +{ + if [ "$assembler" = "fasm" ]; then + fasm "$1" "$2" >/dev/null + else + nasm -f elf64 "$1" -o "$2" + fi +} + work="$(mktemp -d)" trap 'rm -rf "$work"' EXIT @@ -35,9 +46,9 @@ check() local name="$1" desc="$2" source="$3" expected_exit="$4" expected_stdout="$5" local asm="$work/$name.asm" obj="$work/$name.o" bin="$work/$name" stage="" - if ! "$hdass" "$source" -o "$asm" 2>"$work/err"; then + if ! "$hdass" -t "$assembler" "$source" -o "$asm" 2>"$work/err"; then stage="transpile" - elif ! nasm -f elf64 "$asm" -o "$obj" 2>"$work/err"; then + elif ! assemble "$asm" "$obj" 2>"$work/err"; then stage="assemble" elif ! ld -e main "$obj" -o "$bin" 2>"$work/err"; then stage="link" @@ -71,7 +82,7 @@ check() pass=$((pass + 1)) } -printf '%s━━ example programs ━━%s\n' "$bold" "$reset" +printf '%s━━ example programs (%s) ━━%s\n' "$bold" "$assembler" "$reset" check hello_world "writes a greeting to stdout" examples/hello_world.hdass 0 "Hello, World!" check greet "writes a fixed string" examples/greet.hdass 0 "hdass works!" @@ -152,6 +152,8 @@ struct IfStatement struct WhileStatement { + bool named; + struct Token name; struct Expr* left; struct Token comparison; struct Expr* right; diff --git a/src/nasm.c b/src/codegen.c index 2a98e11..2eaa8bd 100644 --- a/src/nasm.c +++ b/src/codegen.c @@ -5,7 +5,7 @@ #include <string.h> #include <stdbool.h> -#include "nasm.h" +#include "codegen.h" static void emit_const_expr(struct Expr* expr, FILE* out) { @@ -28,35 +28,6 @@ static void emit_const_expr(struct Expr* expr, FILE* out) } } -static void emit_consts(struct Program* program, FILE* out) -{ - for (size_t i = 0; i < program->const_count; i += 1) - { - struct ConstDecl decl = program->consts[i]; - fprintf(out, "%%define %.*s (", (int)decl.name.length, decl.name.start); - emit_const_expr(decl.value, out); - fprintf(out, ")\n"); - } -} - -static void emit_data(struct Program* program, FILE* out) -{ - fprintf(out, "section .data\n"); - - for (size_t i = 0; i < program->data_count; i += 1) - { - struct DataDecl decl = program->data_decls[i]; - - // the value lexeme keeps its surrounding double quotes; NASM backtick - // strings interpret the same escapes, so re-wrap the inner content - fprintf(out, "%.*s: db `%.*s`\n", - (int)decl.name.length, decl.name.start, - (int)(decl.value.length - 2), decl.value.start + 1); - fprintf(out, ".len equ $ - %.*s\n", - (int)decl.name.length, decl.name.start); - } -} - static const char* assign_mnemonic(enum TokenType op) { switch (op) @@ -1129,21 +1100,31 @@ static void emit_if(struct Emitter* emitter, struct IfStatement* branch) static void emit_while(struct Emitter* emitter, struct WhileStatement* loop) { - uint32_t id = emitter->label_id; - emitter->label_id += 1; + char top[64]; + char end[64]; - char target[32]; - snprintf(target, sizeof(target), ".while_end_%u", id); + if (loop->named) + { + snprintf(top, sizeof(top), ".%.*s", (int)loop->name.length, loop->name.start); + snprintf(end, sizeof(end), ".%.*s_end", (int)loop->name.length, loop->name.start); + } + else + { + uint32_t id = emitter->label_id; + emitter->label_id += 1; + snprintf(top, sizeof(top), ".while_%u", id); + snprintf(end, sizeof(end), ".while_end_%u", id); + } - fprintf(emitter->out, ".while_%u:\n", id); + fprintf(emitter->out, "%s:\n", top); - if (!emit_branch_test(emitter, loop->left, loop->comparison, loop->right, target)) + if (!emit_branch_test(emitter, loop->left, loop->comparison, loop->right, end)) return; emit_block(emitter, loop->body, loop->body_count); - fprintf(emitter->out, "\tjmp .while_%u\n", id); - fprintf(emitter->out, ".while_end_%u:\n", id); + fprintf(emitter->out, "\tjmp %s\n", top); + fprintf(emitter->out, "%s:\n", end); } static void emit_statement(struct Emitter* emitter, struct Statement* statement) @@ -1279,13 +1260,6 @@ static struct FloatTable collect_floats(struct Program* program) return floats; } -static void emit_float_data(const struct FloatTable* floats, FILE* out) -{ - for (size_t i = 0; i < floats->count; i += 1) - fprintf(out, "__float%zu: dq %.*s\n", i, - (int)floats->items[i].length, floats->items[i].start); -} - static void emit_proc(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, FILE* out) { struct Emitter emitter; @@ -1321,25 +1295,198 @@ static void emit_proc(struct Program* program, struct FloatTable* floats, struct } } -void generate_nasm(struct Program* program, FILE* out) +// The instruction bodies above are plain Intel syntax, identical for every +// target assembler. Only the framing around them — the file header, constants, +// section directives, data definitions and the exported entry symbol — differs, +// so each backend supplies just those. +struct Backend +{ + void (*prologue)(const struct Program* program, FILE* out); + void (*constant)(struct ConstDecl decl, FILE* out); + void (*data_section)(FILE* out); + void (*string_data)(struct DataDecl decl, FILE* out); + void (*float_slot)(size_t index, struct Token literal, FILE* out); + void (*text_section)(FILE* out); + void (*global)(struct Token name, FILE* out); +}; + +static void nasm_prologue(const struct Program* program, FILE* out) +{ + fprintf(out, "bits %u\n", program->config.bits); +} + +static void nasm_constant(struct ConstDecl decl, FILE* out) +{ + fprintf(out, "%%define %.*s (", (int)decl.name.length, decl.name.start); + emit_const_expr(decl.value, out); + fprintf(out, ")\n"); +} + +static void nasm_data_section(FILE* out) +{ + fprintf(out, "section .data\n"); +} + +static void nasm_string_data(struct DataDecl decl, FILE* out) +{ + // the value lexeme keeps its quotes; NASM backtick strings interpret the + // same escapes, so re-wrap the inner content + fprintf(out, "%.*s: db `%.*s`\n", + (int)decl.name.length, decl.name.start, + (int)(decl.value.length - 2), decl.value.start + 1); + fprintf(out, ".len equ $ - %.*s\n", (int)decl.name.length, decl.name.start); +} + +static void nasm_float_slot(size_t index, struct Token literal, FILE* out) +{ + fprintf(out, "__float%zu: dq %.*s\n", index, (int)literal.length, literal.start); +} + +static void nasm_text_section(FILE* out) +{ + fprintf(out, "section .text\n"); +} + +static void nasm_global(struct Token name, FILE* out) +{ + fprintf(out, "global %.*s\n", (int)name.length, name.start); +} + +static const struct Backend nasm_backend = { + nasm_prologue, + nasm_constant, + nasm_data_section, + nasm_string_data, + nasm_float_slot, + nasm_text_section, + nasm_global, +}; + +static void fasm_prologue(const struct Program* program, FILE* out) +{ + fprintf(out, "format ELF%s\n", program->config.bits == 64 ? "64" : ""); +} + +static void fasm_constant(struct ConstDecl decl, FILE* out) +{ + fprintf(out, "%.*s = ", (int)decl.name.length, decl.name.start); + emit_const_expr(decl.value, out); + fprintf(out, "\n"); +} + +static void fasm_data_section(FILE* out) +{ + fprintf(out, "section '.data' writeable\n"); +} + +// fasm string literals are taken verbatim, so the escapes NASM would interpret +// are expanded here into the byte values fasm expects (db "run", 10, "run"). +static void fasm_string_data(struct DataDecl decl, FILE* out) +{ + fprintf(out, "%.*s db ", (int)decl.name.length, decl.name.start); + + const char* text = decl.value.start + 1; + size_t length = decl.value.length - 2; + bool in_quotes = false; + bool first = true; + + for (size_t i = 0; i < length; i += 1) + { + unsigned char byte = (unsigned char)text[i]; + if (byte == '\\' && i + 1 < length) + { + i += 1; + switch (text[i]) + { + case 'n': byte = '\n'; break; + case 't': byte = '\t'; break; + case 'r': byte = '\r'; break; + case '0': byte = '\0'; break; + case 'a': byte = '\a'; break; + case 'b': byte = '\b'; break; + case 'f': byte = '\f'; break; + case 'v': byte = '\v'; break; + case 'e': byte = 27; break; + default: byte = (unsigned char)text[i]; break; + } + + if (in_quotes) + { + fprintf(out, "\""); + in_quotes = false; + } + fprintf(out, "%s%u", first ? "" : ", ", byte); + first = false; + continue; + } + + if (!in_quotes) + { + fprintf(out, "%s\"", first ? "" : ", "); + in_quotes = true; + first = false; + } + fprintf(out, "%c", byte); + } + + if (in_quotes) + fprintf(out, "\""); + if (first) + fprintf(out, "\"\""); + fprintf(out, "\n"); + + fprintf(out, ".len = $ - %.*s\n", (int)decl.name.length, decl.name.start); +} + +static void fasm_float_slot(size_t index, struct Token literal, FILE* out) +{ + fprintf(out, "__float%zu dq %.*s\n", index, (int)literal.length, literal.start); +} + +static void fasm_text_section(FILE* out) +{ + fprintf(out, "section '.text' executable\n"); +} + +static void fasm_global(struct Token name, FILE* out) +{ + fprintf(out, "public %.*s\n", (int)name.length, name.start); +} + +static const struct Backend fasm_backend = { + fasm_prologue, + fasm_constant, + fasm_data_section, + fasm_string_data, + fasm_float_slot, + fasm_text_section, + fasm_global, +}; + +static void generate(struct Program* program, FILE* out, const struct Backend* backend) { struct FloatTable floats = collect_floats(program); - fprintf(out, "bits %u\n\n", program->config.bits); + backend->prologue(program, out); + fprintf(out, "\n"); if (program->const_count > 0) { - emit_consts(program, out); + for (size_t i = 0; i < program->const_count; i += 1) + backend->constant(program->consts[i], out); fprintf(out, "\n"); } - emit_data(program, out); - emit_float_data(&floats, out); + backend->data_section(out); + for (size_t i = 0; i < program->data_count; i += 1) + backend->string_data(program->data_decls[i], out); + for (size_t i = 0; i < floats.count; i += 1) + backend->float_slot(i, floats.items[i], out); fprintf(out, "\n"); - fprintf(out, "section .text\n"); + backend->text_section(out); if (program->config.has_entry) - fprintf(out, "global %.*s\n", (int)program->config.entry.length, program->config.entry.start); + backend->global(program->config.entry, out); for (size_t i = 0; i < program->proc_count; i += 1) { @@ -1349,3 +1496,13 @@ void generate_nasm(struct Program* program, FILE* out) free(floats.items); } + +void generate_nasm(struct Program* program, FILE* out) +{ + generate(program, out, &nasm_backend); +} + +void generate_fasm(struct Program* program, FILE* out) +{ + generate(program, out, &fasm_backend); +} diff --git a/src/nasm.h b/src/codegen.h index 1a93b31..38f37ee 100644 --- a/src/nasm.h +++ b/src/codegen.h @@ -5,3 +5,4 @@ #include "ast.h" void generate_nasm(struct Program* program, FILE* out); +void generate_fasm(struct Program* program, FILE* out); @@ -5,8 +5,8 @@ #include "diag.h" #include "sema.h" #include "lexer.h" -#include "nasm.h" #include "parser.h" +#include "codegen.h" int main(int argc, char** argv) { @@ -18,9 +18,9 @@ int main(int argc, char** argv) if (result == PARSE_ERROR) return 1; - if (args.target != ASSEMBLER_NASM) + if (args.target == ASSEMBLER_MASM) { - report_error_message("only the nasm target is supported"); + report_error_message("the masm target is not implemented yet"); return 1; } @@ -62,7 +62,10 @@ int main(int argc, char** argv) } } - generate_nasm(&program, out); + if (args.target == ASSEMBLER_FASM) + generate_fasm(&program, out); + else + generate_nasm(&program, out); if (out != stdout) fclose(out); diff --git a/src/parser.c b/src/parser.c index ddaee72..ae7c1b3 100644 --- a/src/parser.c +++ b/src/parser.c @@ -552,6 +552,14 @@ static bool parse_if(struct Parser* parser, struct Statement* out) static bool parse_while(struct Parser* parser, struct Statement* out) { + // an optional .name makes the loop's asm labels readable (.name / .name_end) + struct Token name = { 0 }; + bool named = match_token(parser, TOKEN_DOT); + if (named && !consume(parser, TOKEN_IDENTIFIER, "expected a loop name after '.'")) + return false; + if (named) + name = parser->previous; + struct Expr* left; struct Token comparison; struct Expr* right; @@ -568,6 +576,8 @@ static bool parse_while(struct Parser* parser, struct Statement* out) } out->kind = STATEMENT_WHILE; + out->loop.named = named; + out->loop.name = name; out->loop.left = left; out->loop.comparison = comparison; out->loop.right = right; diff --git a/tests/codegen_test.c b/tests/codegen_test.c index 2ccba98..24cac6d 100644 --- a/tests/codegen_test.c +++ b/tests/codegen_test.c @@ -1,9 +1,9 @@ #include <stdio.h> #include <string.h> -#include "nasm.h" #include "tests.h" #include "parser.h" +#include "codegen.h" static void generate_to_buffer(struct Program* program, char* buffer, size_t size) { @@ -23,6 +23,24 @@ static void generate_to_buffer(struct Program* program, char* buffer, size_t siz fclose(out); } +static void generate_fasm_to_buffer(struct Program* program, char* buffer, size_t size) +{ + FILE* out = tmpfile(); + if (out == NULL) + { + buffer[0] = '\0'; + return; + } + + generate_fasm(program, out); + fflush(out); + rewind(out); + + size_t read = fread(buffer, 1, size - 1, out); + buffer[read] = '\0'; + fclose(out); +} + static void test_generate_consts_and_data(struct TestContext* context) { struct Lexer lexer = create_lexer("const N = 5\ndata msg = \"hi\"\n"); @@ -41,6 +59,33 @@ static void test_generate_consts_and_data(struct TestContext* context) free_program(&program); } +static void test_generate_fasm(struct TestContext* context) +{ + struct Lexer lexer = create_lexer( + "[entry: main]\nconst N = 5\ndata msg = \"hi\\n\"\nproc main\n{\nrax = N\nsyscall\n}\n"); + struct Program program; + check(context, parse_program(&lexer, &program)); + + char buffer[1024]; + generate_fasm_to_buffer(&program, buffer, sizeof(buffer)); + + // fasm framing differs from nasm + check(context, strstr(buffer, "format ELF64") != NULL); + check(context, strstr(buffer, "N = 5") != NULL); + check(context, strstr(buffer, "section '.data' writeable") != NULL); + check(context, strstr(buffer, "msg db \"hi\", 10") != NULL); // escape expanded to a byte + check(context, strstr(buffer, ".len = $ - msg") != NULL); + check(context, strstr(buffer, "section '.text' executable") != NULL); + check(context, strstr(buffer, "public main") != NULL); + // instruction bodies are identical to nasm + check(context, strstr(buffer, "mov rax, N") != NULL); + // nasm-only forms must be absent + check(context, strstr(buffer, "%define") == NULL); + check(context, strstr(buffer, "bits 64") == NULL); + + free_program(&program); +} + static void test_generate_text(struct TestContext* context) { struct Lexer lexer = create_lexer( @@ -121,6 +166,25 @@ static void test_generate_while(struct TestContext* context) free_program(&program); } +static void test_generate_while_named(struct TestContext* context) +{ + struct Lexer lexer = create_lexer( + "proc main\n{\nwhile .drain rcx > 0\n{\nrcx -= 1\n}\n}\n"); + struct Program program; + check(context, parse_program(&lexer, &program)); + + char buffer[1024]; + generate_to_buffer(&program, buffer, sizeof(buffer)); + + check(context, strstr(buffer, ".drain:") != NULL); + check(context, strstr(buffer, "jle .drain_end") != NULL); + check(context, strstr(buffer, "jmp .drain") != NULL); + check(context, strstr(buffer, ".drain_end:") != NULL); + check(context, strstr(buffer, ".while_0") == NULL); + + free_program(&program); +} + static void test_generate_negative(struct TestContext* context) { struct Lexer lexer = create_lexer( @@ -524,10 +588,12 @@ void run_codegen_tests(struct TestContext* context) test_generate_entry_and_bits(context); test_generate_no_entry(context); test_generate_consts_and_data(context); + test_generate_fasm(context); test_generate_text(context); test_generate_if(context); test_generate_if_else(context); test_generate_while(context); + test_generate_while_named(context); test_generate_negative(context); test_generate_call(context); test_generate_param_substitution(context); diff --git a/tests/parser_test.c b/tests/parser_test.c index 88e32a1..a88a319 100644 --- a/tests/parser_test.c +++ b/tests/parser_test.c @@ -241,6 +241,23 @@ static void test_parse_while(struct TestContext* context) check(context, primary_is(loop.loop.right, "0")); check(context, loop.loop.body_count == 2); check(context, text_is(loop.loop.body[1].assign.target, "rcx")); + check(context, !loop.loop.named); + + free_program(&program); +} + +static void test_parse_while_named(struct TestContext* context) +{ + struct Lexer lexer = create_lexer("proc main\n{\nwhile .drain rcx > 0\n{\nrcx -= 1\n}\n}\n"); + struct Program program; + + check(context, parse_program(&lexer, &program)); + + struct Statement loop = program.procs[0].body[0]; + check(context, loop.kind == STATEMENT_WHILE); + check(context, loop.loop.named); + check(context, text_is(loop.loop.name, "drain")); + check(context, primary_is(loop.loop.left, "rcx")); free_program(&program); } @@ -417,6 +434,7 @@ void run_parser_tests(struct TestContext* context) test_parse_if_block(context); test_parse_else_if(context); test_parse_while(context); + test_parse_while_named(context); test_parse_call(context); test_parse_stack(context); test_parse_sized_deref(context); |
