From 8eac30c13cfa4f47a8a0aa0772b96f0813fe33a5 Mon Sep 17 00:00:00 2001 From: hachem Date: Sun, 30 Aug 2026 23:14:01 +0200 Subject: feat: add sematic analysis --- src/main.c | 8 ++++++ src/sema/sema.c | 73 +++++++++++++++++++++++++++++++++++++++++++++++++++++++ src/sema/sema.h | 7 ++++++ tests/main.c | 1 + tests/sema_test.c | 48 ++++++++++++++++++++++++++++++++++++ tests/tests.h | 1 + 6 files changed, 138 insertions(+) create mode 100644 src/sema/sema.c create mode 100644 src/sema/sema.h create mode 100644 tests/sema_test.c diff --git a/src/main.c b/src/main.c index a908a73..90b3c82 100644 --- a/src/main.c +++ b/src/main.c @@ -2,6 +2,7 @@ #include "io/file.h" #include "cli/args.h" +#include "sema/sema.h" #include "lexer/lexer.h" #include "codegen/nasm.h" #include "parser/parser.h" @@ -36,6 +37,13 @@ int main(int argc, char** argv) return 1; } + if (!analyze_program(&program)) + { + free_program(&program); + free_file(&source); + return 1; + } + FILE* out = stdout; if (args.output_path != NULL) { diff --git a/src/sema/sema.c b/src/sema/sema.c new file mode 100644 index 0000000..99ec4f1 --- /dev/null +++ b/src/sema/sema.c @@ -0,0 +1,73 @@ +#include +#include +#include + +#include "sema/sema.h" + +static bool names_equal(struct Token a, struct Token b) +{ + return a.length == b.length && memcmp(a.start, b.start, a.length) == 0; +} + +static bool check_duplicate_names(struct Program* program) +{ + size_t count = program->const_count + program->data_count + program->proc_count; + if (count == 0) + return true; + + struct Token* names = malloc(count * sizeof(struct Token)); + size_t n = 0; + for (size_t i = 0; i < program->const_count; i += 1) + { + names[n] = program->consts[i].name; + n += 1; + } + for (size_t i = 0; i < program->data_count; i += 1) + { + names[n] = program->data_decls[i].name; + n += 1; + } + for (size_t i = 0; i < program->proc_count; i += 1) + { + names[n] = program->procs[i].name; + n += 1; + } + + bool ok = true; + for (size_t i = 0; i < count; i += 1) + for (size_t j = 0; j < i; j += 1) + if (names_equal(names[i], names[j])) + { + fprintf(stderr, "error: line %u: '%.*s' is already defined\n", + names[i].line, (int)names[i].length, names[i].start); + ok = false; + } + + free(names); + return ok; +} + +static bool check_entry_point(struct Program* program) +{ + for (size_t i = 0; i < program->proc_count; i += 1) + { + struct Token name = program->procs[i].name; + if (name.length == 4 && memcmp(name.start, "main", 4) == 0) + return true; + } + + fprintf(stderr, "error: no 'main' procedure defined\n"); + return false; +} + +bool analyze_program(struct Program* program) +{ + bool ok = true; + + if (!check_duplicate_names(program)) + ok = false; + if (!check_entry_point(program)) + ok = false; + + return ok; +} diff --git a/src/sema/sema.h b/src/sema/sema.h new file mode 100644 index 0000000..e015f41 --- /dev/null +++ b/src/sema/sema.h @@ -0,0 +1,7 @@ +#pragma once + +#include + +#include "parser/ast.h" + +bool analyze_program(struct Program* program); diff --git a/tests/main.c b/tests/main.c index f0caf52..dfd3ed9 100644 --- a/tests/main.c +++ b/tests/main.c @@ -8,6 +8,7 @@ int main(void) run_lexer_tests(&context); run_parser_tests(&context); + run_sema_tests(&context); run_codegen_tests(&context); printf("\n%d checks, %d failure(s)\n", context.checks, context.failures); diff --git a/tests/sema_test.c b/tests/sema_test.c new file mode 100644 index 0000000..a1946d9 --- /dev/null +++ b/tests/sema_test.c @@ -0,0 +1,48 @@ +#include + +#include "parser/parser.h" +#include "sema/sema.h" +#include "tests.h" + +static bool analyze_source(const char* source) +{ + struct Lexer lexer = create_lexer(source); + struct Program program; + if (!parse_program(&lexer, &program)) + { + free_program(&program); + return false; + } + + bool ok = analyze_program(&program); + free_program(&program); + return ok; +} + +static void test_valid_program(struct TestContext* context) +{ + check(context, analyze_source("const N = 1\ndata msg = \"hi\"\nproc main\n{\nsyscall\n}\n")); +} + +static void test_missing_main(struct TestContext* context) +{ + check(context, !analyze_source("proc helper\n{\nsyscall\n}\n")); +} + +static void test_duplicate_const(struct TestContext* context) +{ + check(context, !analyze_source("const X = 1\nconst X = 2\nproc main\n{\nsyscall\n}\n")); +} + +static void test_duplicate_across_kinds(struct TestContext* context) +{ + check(context, !analyze_source("data foo = \"a\"\nproc foo\n{\nsyscall\n}\nproc main\n{\nsyscall\n}\n")); +} + +void run_sema_tests(struct TestContext* context) +{ + test_valid_program(context); + test_missing_main(context); + test_duplicate_const(context); + test_duplicate_across_kinds(context); +} diff --git a/tests/tests.h b/tests/tests.h index 32ee4a6..2caf9b3 100644 --- a/tests/tests.h +++ b/tests/tests.h @@ -4,4 +4,5 @@ void run_lexer_tests(struct TestContext* context); void run_parser_tests(struct TestContext* context); +void run_sema_tests(struct TestContext* context); void run_codegen_tests(struct TestContext* context); -- cgit v1.3