diff options
| author | hachem <im@hachem.wtf> | 2026-09-09 05:14:44 +0200 |
|---|---|---|
| committer | hachem <im@hachem.wtf> | 2026-09-09 05:14:44 +0200 |
| commit | 8b76d35b0a045e0a3277be422513b938ef33dcce (patch) | |
| tree | b876d92ba3afb810e38f3f83792fc3fb849dd457 /src | |
| parent | b9eff51ee157b22d8fa46aeecac7b95d5ed345d9 (diff) | |
feat: fasm backend
Diffstat (limited to 'src')
| -rw-r--r-- | src/ast.h | 2 | ||||
| -rw-r--r-- | src/codegen.c (renamed from src/nasm.c) | 261 | ||||
| -rw-r--r-- | src/codegen.h (renamed from src/nasm.h) | 1 | ||||
| -rw-r--r-- | src/main.c | 11 | ||||
| -rw-r--r-- | src/parser.c | 10 |
5 files changed, 229 insertions, 56 deletions
@@ -152,6 +152,8 @@ struct IfStatement struct WhileStatement { + bool named; + struct Token name; struct Expr* left; struct Token comparison; struct Expr* right; diff --git a/src/nasm.c b/src/codegen.c index 2a98e11..2eaa8bd 100644 --- a/src/nasm.c +++ b/src/codegen.c @@ -5,7 +5,7 @@ #include <string.h> #include <stdbool.h> -#include "nasm.h" +#include "codegen.h" static void emit_const_expr(struct Expr* expr, FILE* out) { @@ -28,35 +28,6 @@ static void emit_const_expr(struct Expr* expr, FILE* out) } } -static void emit_consts(struct Program* program, FILE* out) -{ - for (size_t i = 0; i < program->const_count; i += 1) - { - struct ConstDecl decl = program->consts[i]; - fprintf(out, "%%define %.*s (", (int)decl.name.length, decl.name.start); - emit_const_expr(decl.value, out); - fprintf(out, ")\n"); - } -} - -static void emit_data(struct Program* program, FILE* out) -{ - fprintf(out, "section .data\n"); - - for (size_t i = 0; i < program->data_count; i += 1) - { - struct DataDecl decl = program->data_decls[i]; - - // the value lexeme keeps its surrounding double quotes; NASM backtick - // strings interpret the same escapes, so re-wrap the inner content - fprintf(out, "%.*s: db `%.*s`\n", - (int)decl.name.length, decl.name.start, - (int)(decl.value.length - 2), decl.value.start + 1); - fprintf(out, ".len equ $ - %.*s\n", - (int)decl.name.length, decl.name.start); - } -} - static const char* assign_mnemonic(enum TokenType op) { switch (op) @@ -1129,21 +1100,31 @@ static void emit_if(struct Emitter* emitter, struct IfStatement* branch) static void emit_while(struct Emitter* emitter, struct WhileStatement* loop) { - uint32_t id = emitter->label_id; - emitter->label_id += 1; + char top[64]; + char end[64]; - char target[32]; - snprintf(target, sizeof(target), ".while_end_%u", id); + if (loop->named) + { + snprintf(top, sizeof(top), ".%.*s", (int)loop->name.length, loop->name.start); + snprintf(end, sizeof(end), ".%.*s_end", (int)loop->name.length, loop->name.start); + } + else + { + uint32_t id = emitter->label_id; + emitter->label_id += 1; + snprintf(top, sizeof(top), ".while_%u", id); + snprintf(end, sizeof(end), ".while_end_%u", id); + } - fprintf(emitter->out, ".while_%u:\n", id); + fprintf(emitter->out, "%s:\n", top); - if (!emit_branch_test(emitter, loop->left, loop->comparison, loop->right, target)) + if (!emit_branch_test(emitter, loop->left, loop->comparison, loop->right, end)) return; emit_block(emitter, loop->body, loop->body_count); - fprintf(emitter->out, "\tjmp .while_%u\n", id); - fprintf(emitter->out, ".while_end_%u:\n", id); + fprintf(emitter->out, "\tjmp %s\n", top); + fprintf(emitter->out, "%s:\n", end); } static void emit_statement(struct Emitter* emitter, struct Statement* statement) @@ -1279,13 +1260,6 @@ static struct FloatTable collect_floats(struct Program* program) return floats; } -static void emit_float_data(const struct FloatTable* floats, FILE* out) -{ - for (size_t i = 0; i < floats->count; i += 1) - fprintf(out, "__float%zu: dq %.*s\n", i, - (int)floats->items[i].length, floats->items[i].start); -} - static void emit_proc(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, FILE* out) { struct Emitter emitter; @@ -1321,25 +1295,198 @@ static void emit_proc(struct Program* program, struct FloatTable* floats, struct } } -void generate_nasm(struct Program* program, FILE* out) +// The instruction bodies above are plain Intel syntax, identical for every +// target assembler. Only the framing around them — the file header, constants, +// section directives, data definitions and the exported entry symbol — differs, +// so each backend supplies just those. +struct Backend +{ + void (*prologue)(const struct Program* program, FILE* out); + void (*constant)(struct ConstDecl decl, FILE* out); + void (*data_section)(FILE* out); + void (*string_data)(struct DataDecl decl, FILE* out); + void (*float_slot)(size_t index, struct Token literal, FILE* out); + void (*text_section)(FILE* out); + void (*global)(struct Token name, FILE* out); +}; + +static void nasm_prologue(const struct Program* program, FILE* out) +{ + fprintf(out, "bits %u\n", program->config.bits); +} + +static void nasm_constant(struct ConstDecl decl, FILE* out) +{ + fprintf(out, "%%define %.*s (", (int)decl.name.length, decl.name.start); + emit_const_expr(decl.value, out); + fprintf(out, ")\n"); +} + +static void nasm_data_section(FILE* out) +{ + fprintf(out, "section .data\n"); +} + +static void nasm_string_data(struct DataDecl decl, FILE* out) +{ + // the value lexeme keeps its quotes; NASM backtick strings interpret the + // same escapes, so re-wrap the inner content + fprintf(out, "%.*s: db `%.*s`\n", + (int)decl.name.length, decl.name.start, + (int)(decl.value.length - 2), decl.value.start + 1); + fprintf(out, ".len equ $ - %.*s\n", (int)decl.name.length, decl.name.start); +} + +static void nasm_float_slot(size_t index, struct Token literal, FILE* out) +{ + fprintf(out, "__float%zu: dq %.*s\n", index, (int)literal.length, literal.start); +} + +static void nasm_text_section(FILE* out) +{ + fprintf(out, "section .text\n"); +} + +static void nasm_global(struct Token name, FILE* out) +{ + fprintf(out, "global %.*s\n", (int)name.length, name.start); +} + +static const struct Backend nasm_backend = { + nasm_prologue, + nasm_constant, + nasm_data_section, + nasm_string_data, + nasm_float_slot, + nasm_text_section, + nasm_global, +}; + +static void fasm_prologue(const struct Program* program, FILE* out) +{ + fprintf(out, "format ELF%s\n", program->config.bits == 64 ? "64" : ""); +} + +static void fasm_constant(struct ConstDecl decl, FILE* out) +{ + fprintf(out, "%.*s = ", (int)decl.name.length, decl.name.start); + emit_const_expr(decl.value, out); + fprintf(out, "\n"); +} + +static void fasm_data_section(FILE* out) +{ + fprintf(out, "section '.data' writeable\n"); +} + +// fasm string literals are taken verbatim, so the escapes NASM would interpret +// are expanded here into the byte values fasm expects (db "run", 10, "run"). +static void fasm_string_data(struct DataDecl decl, FILE* out) +{ + fprintf(out, "%.*s db ", (int)decl.name.length, decl.name.start); + + const char* text = decl.value.start + 1; + size_t length = decl.value.length - 2; + bool in_quotes = false; + bool first = true; + + for (size_t i = 0; i < length; i += 1) + { + unsigned char byte = (unsigned char)text[i]; + if (byte == '\\' && i + 1 < length) + { + i += 1; + switch (text[i]) + { + case 'n': byte = '\n'; break; + case 't': byte = '\t'; break; + case 'r': byte = '\r'; break; + case '0': byte = '\0'; break; + case 'a': byte = '\a'; break; + case 'b': byte = '\b'; break; + case 'f': byte = '\f'; break; + case 'v': byte = '\v'; break; + case 'e': byte = 27; break; + default: byte = (unsigned char)text[i]; break; + } + + if (in_quotes) + { + fprintf(out, "\""); + in_quotes = false; + } + fprintf(out, "%s%u", first ? "" : ", ", byte); + first = false; + continue; + } + + if (!in_quotes) + { + fprintf(out, "%s\"", first ? "" : ", "); + in_quotes = true; + first = false; + } + fprintf(out, "%c", byte); + } + + if (in_quotes) + fprintf(out, "\""); + if (first) + fprintf(out, "\"\""); + fprintf(out, "\n"); + + fprintf(out, ".len = $ - %.*s\n", (int)decl.name.length, decl.name.start); +} + +static void fasm_float_slot(size_t index, struct Token literal, FILE* out) +{ + fprintf(out, "__float%zu dq %.*s\n", index, (int)literal.length, literal.start); +} + +static void fasm_text_section(FILE* out) +{ + fprintf(out, "section '.text' executable\n"); +} + +static void fasm_global(struct Token name, FILE* out) +{ + fprintf(out, "public %.*s\n", (int)name.length, name.start); +} + +static const struct Backend fasm_backend = { + fasm_prologue, + fasm_constant, + fasm_data_section, + fasm_string_data, + fasm_float_slot, + fasm_text_section, + fasm_global, +}; + +static void generate(struct Program* program, FILE* out, const struct Backend* backend) { struct FloatTable floats = collect_floats(program); - fprintf(out, "bits %u\n\n", program->config.bits); + backend->prologue(program, out); + fprintf(out, "\n"); if (program->const_count > 0) { - emit_consts(program, out); + for (size_t i = 0; i < program->const_count; i += 1) + backend->constant(program->consts[i], out); fprintf(out, "\n"); } - emit_data(program, out); - emit_float_data(&floats, out); + backend->data_section(out); + for (size_t i = 0; i < program->data_count; i += 1) + backend->string_data(program->data_decls[i], out); + for (size_t i = 0; i < floats.count; i += 1) + backend->float_slot(i, floats.items[i], out); fprintf(out, "\n"); - fprintf(out, "section .text\n"); + backend->text_section(out); if (program->config.has_entry) - fprintf(out, "global %.*s\n", (int)program->config.entry.length, program->config.entry.start); + backend->global(program->config.entry, out); for (size_t i = 0; i < program->proc_count; i += 1) { @@ -1349,3 +1496,13 @@ void generate_nasm(struct Program* program, FILE* out) free(floats.items); } + +void generate_nasm(struct Program* program, FILE* out) +{ + generate(program, out, &nasm_backend); +} + +void generate_fasm(struct Program* program, FILE* out) +{ + generate(program, out, &fasm_backend); +} diff --git a/src/nasm.h b/src/codegen.h index 1a93b31..38f37ee 100644 --- a/src/nasm.h +++ b/src/codegen.h @@ -5,3 +5,4 @@ #include "ast.h" void generate_nasm(struct Program* program, FILE* out); +void generate_fasm(struct Program* program, FILE* out); @@ -5,8 +5,8 @@ #include "diag.h" #include "sema.h" #include "lexer.h" -#include "nasm.h" #include "parser.h" +#include "codegen.h" int main(int argc, char** argv) { @@ -18,9 +18,9 @@ int main(int argc, char** argv) if (result == PARSE_ERROR) return 1; - if (args.target != ASSEMBLER_NASM) + if (args.target == ASSEMBLER_MASM) { - report_error_message("only the nasm target is supported"); + report_error_message("the masm target is not implemented yet"); return 1; } @@ -62,7 +62,10 @@ int main(int argc, char** argv) } } - generate_nasm(&program, out); + if (args.target == ASSEMBLER_FASM) + generate_fasm(&program, out); + else + generate_nasm(&program, out); if (out != stdout) fclose(out); diff --git a/src/parser.c b/src/parser.c index ddaee72..ae7c1b3 100644 --- a/src/parser.c +++ b/src/parser.c @@ -552,6 +552,14 @@ static bool parse_if(struct Parser* parser, struct Statement* out) static bool parse_while(struct Parser* parser, struct Statement* out) { + // an optional .name makes the loop's asm labels readable (.name / .name_end) + struct Token name = { 0 }; + bool named = match_token(parser, TOKEN_DOT); + if (named && !consume(parser, TOKEN_IDENTIFIER, "expected a loop name after '.'")) + return false; + if (named) + name = parser->previous; + struct Expr* left; struct Token comparison; struct Expr* right; @@ -568,6 +576,8 @@ static bool parse_while(struct Parser* parser, struct Statement* out) } out->kind = STATEMENT_WHILE; + out->loop.named = named; + out->loop.name = name; out->loop.left = left; out->loop.comparison = comparison; out->loop.right = right; |
