aboutsummaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
Diffstat (limited to 'src')
-rw-r--r--src/ast.h2
-rw-r--r--src/codegen.c (renamed from src/nasm.c)261
-rw-r--r--src/codegen.h (renamed from src/nasm.h)1
-rw-r--r--src/main.c11
-rw-r--r--src/parser.c10
5 files changed, 229 insertions, 56 deletions
diff --git a/src/ast.h b/src/ast.h
index f5eea17..d764c13 100644
--- a/src/ast.h
+++ b/src/ast.h
@@ -152,6 +152,8 @@ struct IfStatement
struct WhileStatement
{
+ bool named;
+ struct Token name;
struct Expr* left;
struct Token comparison;
struct Expr* right;
diff --git a/src/nasm.c b/src/codegen.c
index 2a98e11..2eaa8bd 100644
--- a/src/nasm.c
+++ b/src/codegen.c
@@ -5,7 +5,7 @@
#include <string.h>
#include <stdbool.h>
-#include "nasm.h"
+#include "codegen.h"
static void emit_const_expr(struct Expr* expr, FILE* out)
{
@@ -28,35 +28,6 @@ static void emit_const_expr(struct Expr* expr, FILE* out)
}
}
-static void emit_consts(struct Program* program, FILE* out)
-{
- for (size_t i = 0; i < program->const_count; i += 1)
- {
- struct ConstDecl decl = program->consts[i];
- fprintf(out, "%%define %.*s (", (int)decl.name.length, decl.name.start);
- emit_const_expr(decl.value, out);
- fprintf(out, ")\n");
- }
-}
-
-static void emit_data(struct Program* program, FILE* out)
-{
- fprintf(out, "section .data\n");
-
- for (size_t i = 0; i < program->data_count; i += 1)
- {
- struct DataDecl decl = program->data_decls[i];
-
- // the value lexeme keeps its surrounding double quotes; NASM backtick
- // strings interpret the same escapes, so re-wrap the inner content
- fprintf(out, "%.*s: db `%.*s`\n",
- (int)decl.name.length, decl.name.start,
- (int)(decl.value.length - 2), decl.value.start + 1);
- fprintf(out, ".len equ $ - %.*s\n",
- (int)decl.name.length, decl.name.start);
- }
-}
-
static const char* assign_mnemonic(enum TokenType op)
{
switch (op)
@@ -1129,21 +1100,31 @@ static void emit_if(struct Emitter* emitter, struct IfStatement* branch)
static void emit_while(struct Emitter* emitter, struct WhileStatement* loop)
{
- uint32_t id = emitter->label_id;
- emitter->label_id += 1;
+ char top[64];
+ char end[64];
- char target[32];
- snprintf(target, sizeof(target), ".while_end_%u", id);
+ if (loop->named)
+ {
+ snprintf(top, sizeof(top), ".%.*s", (int)loop->name.length, loop->name.start);
+ snprintf(end, sizeof(end), ".%.*s_end", (int)loop->name.length, loop->name.start);
+ }
+ else
+ {
+ uint32_t id = emitter->label_id;
+ emitter->label_id += 1;
+ snprintf(top, sizeof(top), ".while_%u", id);
+ snprintf(end, sizeof(end), ".while_end_%u", id);
+ }
- fprintf(emitter->out, ".while_%u:\n", id);
+ fprintf(emitter->out, "%s:\n", top);
- if (!emit_branch_test(emitter, loop->left, loop->comparison, loop->right, target))
+ if (!emit_branch_test(emitter, loop->left, loop->comparison, loop->right, end))
return;
emit_block(emitter, loop->body, loop->body_count);
- fprintf(emitter->out, "\tjmp .while_%u\n", id);
- fprintf(emitter->out, ".while_end_%u:\n", id);
+ fprintf(emitter->out, "\tjmp %s\n", top);
+ fprintf(emitter->out, "%s:\n", end);
}
static void emit_statement(struct Emitter* emitter, struct Statement* statement)
@@ -1279,13 +1260,6 @@ static struct FloatTable collect_floats(struct Program* program)
return floats;
}
-static void emit_float_data(const struct FloatTable* floats, FILE* out)
-{
- for (size_t i = 0; i < floats->count; i += 1)
- fprintf(out, "__float%zu: dq %.*s\n", i,
- (int)floats->items[i].length, floats->items[i].start);
-}
-
static void emit_proc(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, FILE* out)
{
struct Emitter emitter;
@@ -1321,25 +1295,198 @@ static void emit_proc(struct Program* program, struct FloatTable* floats, struct
}
}
-void generate_nasm(struct Program* program, FILE* out)
+// The instruction bodies above are plain Intel syntax, identical for every
+// target assembler. Only the framing around them — the file header, constants,
+// section directives, data definitions and the exported entry symbol — differs,
+// so each backend supplies just those.
+struct Backend
+{
+ void (*prologue)(const struct Program* program, FILE* out);
+ void (*constant)(struct ConstDecl decl, FILE* out);
+ void (*data_section)(FILE* out);
+ void (*string_data)(struct DataDecl decl, FILE* out);
+ void (*float_slot)(size_t index, struct Token literal, FILE* out);
+ void (*text_section)(FILE* out);
+ void (*global)(struct Token name, FILE* out);
+};
+
+static void nasm_prologue(const struct Program* program, FILE* out)
+{
+ fprintf(out, "bits %u\n", program->config.bits);
+}
+
+static void nasm_constant(struct ConstDecl decl, FILE* out)
+{
+ fprintf(out, "%%define %.*s (", (int)decl.name.length, decl.name.start);
+ emit_const_expr(decl.value, out);
+ fprintf(out, ")\n");
+}
+
+static void nasm_data_section(FILE* out)
+{
+ fprintf(out, "section .data\n");
+}
+
+static void nasm_string_data(struct DataDecl decl, FILE* out)
+{
+ // the value lexeme keeps its quotes; NASM backtick strings interpret the
+ // same escapes, so re-wrap the inner content
+ fprintf(out, "%.*s: db `%.*s`\n",
+ (int)decl.name.length, decl.name.start,
+ (int)(decl.value.length - 2), decl.value.start + 1);
+ fprintf(out, ".len equ $ - %.*s\n", (int)decl.name.length, decl.name.start);
+}
+
+static void nasm_float_slot(size_t index, struct Token literal, FILE* out)
+{
+ fprintf(out, "__float%zu: dq %.*s\n", index, (int)literal.length, literal.start);
+}
+
+static void nasm_text_section(FILE* out)
+{
+ fprintf(out, "section .text\n");
+}
+
+static void nasm_global(struct Token name, FILE* out)
+{
+ fprintf(out, "global %.*s\n", (int)name.length, name.start);
+}
+
+static const struct Backend nasm_backend = {
+ nasm_prologue,
+ nasm_constant,
+ nasm_data_section,
+ nasm_string_data,
+ nasm_float_slot,
+ nasm_text_section,
+ nasm_global,
+};
+
+static void fasm_prologue(const struct Program* program, FILE* out)
+{
+ fprintf(out, "format ELF%s\n", program->config.bits == 64 ? "64" : "");
+}
+
+static void fasm_constant(struct ConstDecl decl, FILE* out)
+{
+ fprintf(out, "%.*s = ", (int)decl.name.length, decl.name.start);
+ emit_const_expr(decl.value, out);
+ fprintf(out, "\n");
+}
+
+static void fasm_data_section(FILE* out)
+{
+ fprintf(out, "section '.data' writeable\n");
+}
+
+// fasm string literals are taken verbatim, so the escapes NASM would interpret
+// are expanded here into the byte values fasm expects (db "run", 10, "run").
+static void fasm_string_data(struct DataDecl decl, FILE* out)
+{
+ fprintf(out, "%.*s db ", (int)decl.name.length, decl.name.start);
+
+ const char* text = decl.value.start + 1;
+ size_t length = decl.value.length - 2;
+ bool in_quotes = false;
+ bool first = true;
+
+ for (size_t i = 0; i < length; i += 1)
+ {
+ unsigned char byte = (unsigned char)text[i];
+ if (byte == '\\' && i + 1 < length)
+ {
+ i += 1;
+ switch (text[i])
+ {
+ case 'n': byte = '\n'; break;
+ case 't': byte = '\t'; break;
+ case 'r': byte = '\r'; break;
+ case '0': byte = '\0'; break;
+ case 'a': byte = '\a'; break;
+ case 'b': byte = '\b'; break;
+ case 'f': byte = '\f'; break;
+ case 'v': byte = '\v'; break;
+ case 'e': byte = 27; break;
+ default: byte = (unsigned char)text[i]; break;
+ }
+
+ if (in_quotes)
+ {
+ fprintf(out, "\"");
+ in_quotes = false;
+ }
+ fprintf(out, "%s%u", first ? "" : ", ", byte);
+ first = false;
+ continue;
+ }
+
+ if (!in_quotes)
+ {
+ fprintf(out, "%s\"", first ? "" : ", ");
+ in_quotes = true;
+ first = false;
+ }
+ fprintf(out, "%c", byte);
+ }
+
+ if (in_quotes)
+ fprintf(out, "\"");
+ if (first)
+ fprintf(out, "\"\"");
+ fprintf(out, "\n");
+
+ fprintf(out, ".len = $ - %.*s\n", (int)decl.name.length, decl.name.start);
+}
+
+static void fasm_float_slot(size_t index, struct Token literal, FILE* out)
+{
+ fprintf(out, "__float%zu dq %.*s\n", index, (int)literal.length, literal.start);
+}
+
+static void fasm_text_section(FILE* out)
+{
+ fprintf(out, "section '.text' executable\n");
+}
+
+static void fasm_global(struct Token name, FILE* out)
+{
+ fprintf(out, "public %.*s\n", (int)name.length, name.start);
+}
+
+static const struct Backend fasm_backend = {
+ fasm_prologue,
+ fasm_constant,
+ fasm_data_section,
+ fasm_string_data,
+ fasm_float_slot,
+ fasm_text_section,
+ fasm_global,
+};
+
+static void generate(struct Program* program, FILE* out, const struct Backend* backend)
{
struct FloatTable floats = collect_floats(program);
- fprintf(out, "bits %u\n\n", program->config.bits);
+ backend->prologue(program, out);
+ fprintf(out, "\n");
if (program->const_count > 0)
{
- emit_consts(program, out);
+ for (size_t i = 0; i < program->const_count; i += 1)
+ backend->constant(program->consts[i], out);
fprintf(out, "\n");
}
- emit_data(program, out);
- emit_float_data(&floats, out);
+ backend->data_section(out);
+ for (size_t i = 0; i < program->data_count; i += 1)
+ backend->string_data(program->data_decls[i], out);
+ for (size_t i = 0; i < floats.count; i += 1)
+ backend->float_slot(i, floats.items[i], out);
fprintf(out, "\n");
- fprintf(out, "section .text\n");
+ backend->text_section(out);
if (program->config.has_entry)
- fprintf(out, "global %.*s\n", (int)program->config.entry.length, program->config.entry.start);
+ backend->global(program->config.entry, out);
for (size_t i = 0; i < program->proc_count; i += 1)
{
@@ -1349,3 +1496,13 @@ void generate_nasm(struct Program* program, FILE* out)
free(floats.items);
}
+
+void generate_nasm(struct Program* program, FILE* out)
+{
+ generate(program, out, &nasm_backend);
+}
+
+void generate_fasm(struct Program* program, FILE* out)
+{
+ generate(program, out, &fasm_backend);
+}
diff --git a/src/nasm.h b/src/codegen.h
index 1a93b31..38f37ee 100644
--- a/src/nasm.h
+++ b/src/codegen.h
@@ -5,3 +5,4 @@
#include "ast.h"
void generate_nasm(struct Program* program, FILE* out);
+void generate_fasm(struct Program* program, FILE* out);
diff --git a/src/main.c b/src/main.c
index 4153a6e..f15af79 100644
--- a/src/main.c
+++ b/src/main.c
@@ -5,8 +5,8 @@
#include "diag.h"
#include "sema.h"
#include "lexer.h"
-#include "nasm.h"
#include "parser.h"
+#include "codegen.h"
int main(int argc, char** argv)
{
@@ -18,9 +18,9 @@ int main(int argc, char** argv)
if (result == PARSE_ERROR)
return 1;
- if (args.target != ASSEMBLER_NASM)
+ if (args.target == ASSEMBLER_MASM)
{
- report_error_message("only the nasm target is supported");
+ report_error_message("the masm target is not implemented yet");
return 1;
}
@@ -62,7 +62,10 @@ int main(int argc, char** argv)
}
}
- generate_nasm(&program, out);
+ if (args.target == ASSEMBLER_FASM)
+ generate_fasm(&program, out);
+ else
+ generate_nasm(&program, out);
if (out != stdout)
fclose(out);
diff --git a/src/parser.c b/src/parser.c
index ddaee72..ae7c1b3 100644
--- a/src/parser.c
+++ b/src/parser.c
@@ -552,6 +552,14 @@ static bool parse_if(struct Parser* parser, struct Statement* out)
static bool parse_while(struct Parser* parser, struct Statement* out)
{
+ // an optional .name makes the loop's asm labels readable (.name / .name_end)
+ struct Token name = { 0 };
+ bool named = match_token(parser, TOKEN_DOT);
+ if (named && !consume(parser, TOKEN_IDENTIFIER, "expected a loop name after '.'"))
+ return false;
+ if (named)
+ name = parser->previous;
+
struct Expr* left;
struct Token comparison;
struct Expr* right;
@@ -568,6 +576,8 @@ static bool parse_while(struct Parser* parser, struct Statement* out)
}
out->kind = STATEMENT_WHILE;
+ out->loop.named = named;
+ out->loop.name = name;
out->loop.left = left;
out->loop.comparison = comparison;
out->loop.right = right;