diff options
Diffstat (limited to 'src')
| -rw-r--r-- | src/ast.c | 8 | ||||
| -rw-r--r-- | src/ast.h | 23 | ||||
| -rw-r--r-- | src/codegen.c | 170 | ||||
| -rw-r--r-- | src/parser.c | 101 | ||||
| -rw-r--r-- | src/sema.c | 7 |
5 files changed, 280 insertions, 29 deletions
@@ -58,6 +58,11 @@ void free_statement(struct Statement* statement) free_expr(statement->call.args[i]); free(statement->call.args); break; + case STATEMENT_INSTRUCTION: + for (size_t i = 0; i < statement->instruction.operand_count; i += 1) + free_expr(statement->instruction.operands[i]); + free(statement->instruction.operands); + break; case STATEMENT_STACK: free_expr(statement->stack.size); break; @@ -81,6 +86,9 @@ struct Program create_program(void) program.config.bits = 64; program.config.has_entry = false; program.config.logical_registers = false; + program.config.format = OUTPUT_ELF; + program.config.has_org = false; + program.config.boot = false; program.consts = NULL; program.const_count = 0; program.const_capacity = 0; @@ -118,6 +118,7 @@ enum StatementKind STATEMENT_WHILE, STATEMENT_CALL, STATEMENT_STACK, + STATEMENT_INSTRUCTION, }; struct AssignStatement @@ -175,6 +176,17 @@ struct StackStatement struct Expr* size; }; +// a bare instruction: a mnemonic and its operands emitted verbatim, the escape +// hatch for anything outside the assignment / control-flow model (int, hlt, +// lgdt, in/out, ...) +struct InstructionStatement +{ + struct Token mnemonic; + struct Expr** operands; + size_t operand_count; + size_t operand_capacity; +}; + struct Statement { enum StatementKind kind; @@ -187,6 +199,7 @@ struct Statement struct WhileStatement loop; struct CallStatement call; struct StackStatement stack; + struct InstructionStatement instruction; }; }; @@ -202,12 +215,22 @@ struct ProcDecl size_t body_capacity; }; +enum OutputFormat +{ + OUTPUT_ELF, + OUTPUT_BIN, +}; + struct Config { uint32_t bits; bool has_entry; struct Token entry; bool logical_registers; + enum OutputFormat format; + bool has_org; + struct Token org; + bool boot; }; struct Program diff --git a/src/codegen.c b/src/codegen.c index 2eaa8bd..bb38f17 100644 --- a/src/codegen.c +++ b/src/codegen.c @@ -1127,6 +1127,41 @@ static void emit_while(struct Emitter* emitter, struct WhileStatement* loop) fprintf(emitter->out, "%s:\n", end); } +// renders a raw instruction's operand: registers, immediates, constants and +// members reuse emit_operand; memory (^x -> [x], with an optional size) and +// address math are handled here so operands like `^byte si` and `gdt + 2` work +static void emit_instruction_operand(struct Emitter* emitter, struct Expr* operand) +{ + switch (operand->kind) + { + case EXPR_DEREF: + fprintf(emitter->out, "%s[", store_size_keyword(operand->deref.size)); + emit_instruction_operand(emitter, operand->deref.address); + fprintf(emitter->out, "]"); + break; + case EXPR_BINARY: + emit_instruction_operand(emitter, operand->binary.left); + fprintf(emitter->out, " %.*s ", + (int)operand->binary.op.length, operand->binary.op.start); + emit_instruction_operand(emitter, operand->binary.right); + break; + default: + emit_operand(emitter, operand); + break; + } +} + +static void emit_instruction(struct Emitter* emitter, struct InstructionStatement* insn) +{ + fprintf(emitter->out, "\t%.*s", (int)insn->mnemonic.length, insn->mnemonic.start); + for (size_t i = 0; i < insn->operand_count; i += 1) + { + fprintf(emitter->out, "%s", i == 0 ? " " : ", "); + emit_instruction_operand(emitter, insn->operands[i]); + } + fprintf(emitter->out, "\n"); +} + static void emit_statement(struct Emitter* emitter, struct Statement* statement) { FILE* out = emitter->out; @@ -1155,6 +1190,9 @@ static void emit_statement(struct Emitter* emitter, struct Statement* statement) break; case STATEMENT_STACK: break; + case STATEMENT_INSTRUCTION: + emit_instruction(emitter, &statement->instruction); + break; default: fprintf(out, "\t; TODO: unsupported statement\n"); break; @@ -1260,7 +1298,7 @@ static struct FloatTable collect_floats(struct Program* program) return floats; } -static void emit_proc(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, FILE* out) +static void emit_proc_x86(struct Program* program, struct FloatTable* floats, struct ProcDecl* proc, bool is_entry, FILE* out) { struct Emitter emitter; emitter.program = program; @@ -1269,11 +1307,6 @@ static void emit_proc(struct Program* program, struct FloatTable* floats, struct emitter.out = out; emitter.label_id = 0; - struct Config config = program->config; - bool is_entry = config.has_entry - && proc->name.length == config.entry.length - && memcmp(proc->name.start, config.entry.start, proc->name.length) == 0; - fprintf(out, "%.*s:\n", (int)proc->name.length, proc->name.start); uint64_t stack_size = proc_stack_size(program, proc); @@ -1295,10 +1328,20 @@ static void emit_proc(struct Program* program, struct FloatTable* floats, struct } } -// The instruction bodies above are plain Intel syntax, identical for every -// target assembler. Only the framing around them — the file header, constants, -// section directives, data definitions and the exported entry symbol — differs, -// so each backend supplies just those. +// Instruction selection lives behind the Arch seam: turning a procedure's +// statements into a target's instructions (register model, mnemonics, stack +// frames) is all an architecture decides. The Backend below is the orthogonal +// axis — the assembler *syntax* (framing, data, labels) for a given arch. +struct Arch +{ + void (*emit_proc)(struct Program* program, struct FloatTable* floats, + struct ProcDecl* proc, bool is_entry, FILE* out); +}; + +static const struct Arch x86_arch = { + emit_proc_x86, +}; + struct Backend { void (*prologue)(const struct Program* program, FILE* out); @@ -1308,11 +1351,15 @@ struct Backend void (*float_slot)(size_t index, struct Token literal, FILE* out); void (*text_section)(FILE* out); void (*global)(struct Token name, FILE* out); + void (*boot_signature)(FILE* out); }; static void nasm_prologue(const struct Program* program, FILE* out) { - fprintf(out, "bits %u\n", program->config.bits); + struct Config config = program->config; + fprintf(out, "bits %u\n", config.bits); + if (config.has_org) + fprintf(out, "org %.*s\n", (int)config.org.length, config.org.start); } static void nasm_constant(struct ConstDecl decl, FILE* out) @@ -1352,6 +1399,13 @@ static void nasm_global(struct Token name, FILE* out) fprintf(out, "global %.*s\n", (int)name.length, name.start); } +// pad to 510 bytes and append the 0x55AA boot signature (little-endian dw) +static void nasm_boot_signature(FILE* out) +{ + fprintf(out, "times 510-($-$$) db 0\n"); + fprintf(out, "dw 0xAA55\n"); +} + static const struct Backend nasm_backend = { nasm_prologue, nasm_constant, @@ -1360,11 +1414,23 @@ static const struct Backend nasm_backend = { nasm_float_slot, nasm_text_section, nasm_global, + nasm_boot_signature, }; static void fasm_prologue(const struct Program* program, FILE* out) { - fprintf(out, "format ELF%s\n", program->config.bits == 64 ? "64" : ""); + struct Config config = program->config; + if (config.format == OUTPUT_BIN) + { + fprintf(out, "format binary\n"); + if (config.has_org) + fprintf(out, "org %.*s\n", (int)config.org.length, config.org.start); + fprintf(out, "use%u\n", config.bits); + } + else + { + fprintf(out, "format ELF%s\n", config.bits == 64 ? "64" : ""); + } } static void fasm_constant(struct ConstDecl decl, FILE* out) @@ -1453,6 +1519,12 @@ static void fasm_global(struct Token name, FILE* out) fprintf(out, "public %.*s\n", (int)name.length, name.start); } +static void fasm_boot_signature(FILE* out) +{ + fprintf(out, "db (510 - ($ - $$)) dup (0)\n"); + fprintf(out, "dw 0xAA55\n"); +} + static const struct Backend fasm_backend = { fasm_prologue, fasm_constant, @@ -1461,9 +1533,33 @@ static const struct Backend fasm_backend = { fasm_float_slot, fasm_text_section, fasm_global, + fasm_boot_signature, }; -static void generate(struct Program* program, FILE* out, const struct Backend* backend) +// the entry procedure drops its trailing `ret`. It is the [entry: NAME] proc if +// given; otherwise a flat binary starts at its first proc. +static bool proc_is_entry(struct Program* program, size_t index) +{ + struct Config config = program->config; + struct ProcDecl* proc = &program->procs[index]; + + if (config.has_entry) + return proc->name.length == config.entry.length + && memcmp(proc->name.start, config.entry.start, proc->name.length) == 0; + + return config.format == OUTPUT_BIN && index == 0; +} + +static void emit_data_block(struct Program* program, struct FloatTable* floats, + const struct Backend* backend, FILE* out) +{ + for (size_t i = 0; i < program->data_count; i += 1) + backend->string_data(program->data_decls[i], out); + for (size_t i = 0; i < floats->count; i += 1) + backend->float_slot(i, floats->items[i], out); +} + +static void generate(struct Program* program, FILE* out, const struct Arch* arch, const struct Backend* backend) { struct FloatTable floats = collect_floats(program); @@ -1477,21 +1573,43 @@ static void generate(struct Program* program, FILE* out, const struct Backend* b fprintf(out, "\n"); } - backend->data_section(out); - for (size_t i = 0; i < program->data_count; i += 1) - backend->string_data(program->data_decls[i], out); - for (size_t i = 0; i < floats.count; i += 1) - backend->float_slot(i, floats.items[i], out); - fprintf(out, "\n"); + if (program->config.format == OUTPUT_BIN) + { + // a flat binary executes from its origin, so code comes first, then data + for (size_t i = 0; i < program->proc_count; i += 1) + { + if (i > 0) + fprintf(out, "\n"); + arch->emit_proc(program, &floats, &program->procs[i], proc_is_entry(program, i), out); + } - backend->text_section(out); - if (program->config.has_entry) - backend->global(program->config.entry, out); + if (program->data_count > 0 || floats.count > 0) + { + fprintf(out, "\n"); + emit_data_block(program, &floats, backend, out); + } - for (size_t i = 0; i < program->proc_count; i += 1) + if (program->config.boot) + { + fprintf(out, "\n"); + backend->boot_signature(out); + } + } + else { + backend->data_section(out); + emit_data_block(program, &floats, backend, out); fprintf(out, "\n"); - emit_proc(program, &floats, &program->procs[i], out); + + backend->text_section(out); + if (program->config.has_entry) + backend->global(program->config.entry, out); + + for (size_t i = 0; i < program->proc_count; i += 1) + { + fprintf(out, "\n"); + arch->emit_proc(program, &floats, &program->procs[i], proc_is_entry(program, i), out); + } } free(floats.items); @@ -1499,10 +1617,10 @@ static void generate(struct Program* program, FILE* out, const struct Backend* b void generate_nasm(struct Program* program, FILE* out) { - generate(program, out, &nasm_backend); + generate(program, out, &x86_arch, &nasm_backend); } void generate_fasm(struct Program* program, FILE* out) { - generate(program, out, &fasm_backend); + generate(program, out, &x86_arch, &fasm_backend); } diff --git a/src/parser.c b/src/parser.c index ae7c1b3..5379114 100644 --- a/src/parser.c +++ b/src/parser.c @@ -586,6 +586,59 @@ static bool parse_while(struct Parser* parser, struct Statement* out) return true; } +static bool token_starts_operand(enum TokenType type) +{ + return type == TOKEN_IDENTIFIER || type == TOKEN_INTEGER || type == TOKEN_FLOAT + || type == TOKEN_CHAR || type == TOKEN_CARET || type == TOKEN_MINUS; +} + +static bool parse_instruction(struct Parser* parser, struct Token mnemonic, struct Statement* out) +{ + struct Expr** operands = NULL; + size_t count = 0; + size_t capacity = 0; + + // assembly is line-oriented: operands share the mnemonic's line, and a bare + // mnemonic like `hlt` is just followed by the next statement + if (parser->current.line == mnemonic.line && token_starts_operand(parser->current.type)) + { + do + { + struct Expr* operand = parse_expression(parser); + if (operand == NULL) + goto error; + + if (count == capacity) + { + capacity = capacity < 4 ? 4 : capacity * 2; + struct Expr** grown = realloc(operands, capacity * sizeof(struct Expr*)); + if (grown == NULL) + { + free_expr(operand); + goto error; + } + operands = grown; + } + operands[count] = operand; + count += 1; + } + while (match_token(parser, TOKEN_COMMA)); + } + + out->kind = STATEMENT_INSTRUCTION; + out->instruction.mnemonic = mnemonic; + out->instruction.operands = operands; + out->instruction.operand_count = count; + out->instruction.operand_capacity = capacity; + return true; + +error: + for (size_t i = 0; i < count; i += 1) + free_expr(operands[i]); + free(operands); + return false; +} + static bool parse_statement(struct Parser* parser, struct Statement* out) { if (match_token(parser, TOKEN_IF)) @@ -652,6 +705,9 @@ static bool parse_statement(struct Parser* parser, struct Statement* out) return true; } + if (!deref && !is_assign_op(parser->current.type)) + return parse_instruction(parser, name, out); + if (!is_assign_op(parser->current.type)) { error_at(parser, parser->current, "expected an assignment operator"); @@ -729,6 +785,18 @@ static bool parse_directive(struct Parser* parser, struct Program* program) return false; struct Token key = parser->previous; + // bare directive with no value, e.g. [boot] + if (match_token(parser, TOKEN_RIGHT_BRACKET)) + { + if (token_text_is(key, "boot")) + { + program->config.boot = true; + return true; + } + error_at(parser, key, "unknown directive"); + return false; + } + if (!consume(parser, TOKEN_COLON, "expected ':' after directive name")) return false; @@ -745,12 +813,39 @@ static bool parse_directive(struct Parser* parser, struct Program* program) if (token_text_is(key, "bits")) { - if (value.type != TOKEN_INTEGER || (!token_text_is(value, "64") && !token_text_is(value, "32"))) + if (value.type != TOKEN_INTEGER + || (!token_text_is(value, "64") && !token_text_is(value, "32") && !token_text_is(value, "16"))) + { + error_at(parser, value, "bits must be 16, 32 or 64"); + return false; + } + program->config.bits = token_text_is(value, "64") ? 64 : token_text_is(value, "32") ? 32 : 16; + return true; + } + + if (token_text_is(key, "format")) + { + if (value.type == TOKEN_IDENTIFIER && (token_text_is(value, "elf") || token_text_is(value, "elf64"))) + program->config.format = OUTPUT_ELF; + else if (value.type == TOKEN_IDENTIFIER && (token_text_is(value, "bin") || token_text_is(value, "binary"))) + program->config.format = OUTPUT_BIN; + else + { + error_at(parser, value, "format must be elf or bin"); + return false; + } + return true; + } + + if (token_text_is(key, "org")) + { + if (value.type != TOKEN_INTEGER) { - error_at(parser, value, "bits must be 32 or 64"); + error_at(parser, value, "org must be an integer address"); return false; } - program->config.bits = token_text_is(value, "64") ? 64 : 32; + program->config.has_org = true; + program->config.org = value; return true; } @@ -199,6 +199,8 @@ static bool is_arch_register(struct Token token) "rip", "xmm0", "xmm1", "xmm2", "xmm3", "xmm4", "xmm5", "xmm6", "xmm7", "xmm8", "xmm9", "xmm10", "xmm11", "xmm12", "xmm13", "xmm14", "xmm15", + "cs", "ds", "es", "fs", "gs", "ss", + "cr0", "cr2", "cr3", "cr4", "cr8", }; for (size_t i = 0; i < sizeof(names) / sizeof(names[0]); i += 1) @@ -497,6 +499,11 @@ static void check_statement(struct RefCheck* check, struct Statement* statement) case STATEMENT_LABEL: case STATEMENT_SYSCALL: break; + case STATEMENT_INSTRUCTION: + // a raw instruction is an escape hatch; its operands may name asm + // symbols or registers hdass does not model, so leave them to the + // assembler rather than flagging them as undefined + break; } } |
