aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorhachem <im@hachem.wtf>2026-08-30 23:14:01 +0200
committerhachem <im@hachem.wtf>2026-08-30 23:14:01 +0200
commit8eac30c13cfa4f47a8a0aa0772b96f0813fe33a5 (patch)
tree8ddf6be2ffd623e3c8bd931ba1f43699f10a447a
parent81799f3e41ad9cab305930b7743a752003ff16ec (diff)
feat: add sematic analysis
-rw-r--r--src/main.c8
-rw-r--r--src/sema/sema.c73
-rw-r--r--src/sema/sema.h7
-rw-r--r--tests/main.c1
-rw-r--r--tests/sema_test.c48
-rw-r--r--tests/tests.h1
6 files changed, 138 insertions, 0 deletions
diff --git a/src/main.c b/src/main.c
index a908a73..90b3c82 100644
--- a/src/main.c
+++ b/src/main.c
@@ -2,6 +2,7 @@
#include "io/file.h"
#include "cli/args.h"
+#include "sema/sema.h"
#include "lexer/lexer.h"
#include "codegen/nasm.h"
#include "parser/parser.h"
@@ -36,6 +37,13 @@ int main(int argc, char** argv)
return 1;
}
+ if (!analyze_program(&program))
+ {
+ free_program(&program);
+ free_file(&source);
+ return 1;
+ }
+
FILE* out = stdout;
if (args.output_path != NULL)
{
diff --git a/src/sema/sema.c b/src/sema/sema.c
new file mode 100644
index 0000000..99ec4f1
--- /dev/null
+++ b/src/sema/sema.c
@@ -0,0 +1,73 @@
+#include <stdio.h>
+#include <stdlib.h>
+#include <string.h>
+
+#include "sema/sema.h"
+
+static bool names_equal(struct Token a, struct Token b)
+{
+ return a.length == b.length && memcmp(a.start, b.start, a.length) == 0;
+}
+
+static bool check_duplicate_names(struct Program* program)
+{
+ size_t count = program->const_count + program->data_count + program->proc_count;
+ if (count == 0)
+ return true;
+
+ struct Token* names = malloc(count * sizeof(struct Token));
+ size_t n = 0;
+ for (size_t i = 0; i < program->const_count; i += 1)
+ {
+ names[n] = program->consts[i].name;
+ n += 1;
+ }
+ for (size_t i = 0; i < program->data_count; i += 1)
+ {
+ names[n] = program->data_decls[i].name;
+ n += 1;
+ }
+ for (size_t i = 0; i < program->proc_count; i += 1)
+ {
+ names[n] = program->procs[i].name;
+ n += 1;
+ }
+
+ bool ok = true;
+ for (size_t i = 0; i < count; i += 1)
+ for (size_t j = 0; j < i; j += 1)
+ if (names_equal(names[i], names[j]))
+ {
+ fprintf(stderr, "error: line %u: '%.*s' is already defined\n",
+ names[i].line, (int)names[i].length, names[i].start);
+ ok = false;
+ }
+
+ free(names);
+ return ok;
+}
+
+static bool check_entry_point(struct Program* program)
+{
+ for (size_t i = 0; i < program->proc_count; i += 1)
+ {
+ struct Token name = program->procs[i].name;
+ if (name.length == 4 && memcmp(name.start, "main", 4) == 0)
+ return true;
+ }
+
+ fprintf(stderr, "error: no 'main' procedure defined\n");
+ return false;
+}
+
+bool analyze_program(struct Program* program)
+{
+ bool ok = true;
+
+ if (!check_duplicate_names(program))
+ ok = false;
+ if (!check_entry_point(program))
+ ok = false;
+
+ return ok;
+}
diff --git a/src/sema/sema.h b/src/sema/sema.h
new file mode 100644
index 0000000..e015f41
--- /dev/null
+++ b/src/sema/sema.h
@@ -0,0 +1,7 @@
+#pragma once
+
+#include <stdbool.h>
+
+#include "parser/ast.h"
+
+bool analyze_program(struct Program* program);
diff --git a/tests/main.c b/tests/main.c
index f0caf52..dfd3ed9 100644
--- a/tests/main.c
+++ b/tests/main.c
@@ -8,6 +8,7 @@ int main(void)
run_lexer_tests(&context);
run_parser_tests(&context);
+ run_sema_tests(&context);
run_codegen_tests(&context);
printf("\n%d checks, %d failure(s)\n", context.checks, context.failures);
diff --git a/tests/sema_test.c b/tests/sema_test.c
new file mode 100644
index 0000000..a1946d9
--- /dev/null
+++ b/tests/sema_test.c
@@ -0,0 +1,48 @@
+#include <stdbool.h>
+
+#include "parser/parser.h"
+#include "sema/sema.h"
+#include "tests.h"
+
+static bool analyze_source(const char* source)
+{
+ struct Lexer lexer = create_lexer(source);
+ struct Program program;
+ if (!parse_program(&lexer, &program))
+ {
+ free_program(&program);
+ return false;
+ }
+
+ bool ok = analyze_program(&program);
+ free_program(&program);
+ return ok;
+}
+
+static void test_valid_program(struct TestContext* context)
+{
+ check(context, analyze_source("const N = 1\ndata msg = \"hi\"\nproc main\n{\nsyscall\n}\n"));
+}
+
+static void test_missing_main(struct TestContext* context)
+{
+ check(context, !analyze_source("proc helper\n{\nsyscall\n}\n"));
+}
+
+static void test_duplicate_const(struct TestContext* context)
+{
+ check(context, !analyze_source("const X = 1\nconst X = 2\nproc main\n{\nsyscall\n}\n"));
+}
+
+static void test_duplicate_across_kinds(struct TestContext* context)
+{
+ check(context, !analyze_source("data foo = \"a\"\nproc foo\n{\nsyscall\n}\nproc main\n{\nsyscall\n}\n"));
+}
+
+void run_sema_tests(struct TestContext* context)
+{
+ test_valid_program(context);
+ test_missing_main(context);
+ test_duplicate_const(context);
+ test_duplicate_across_kinds(context);
+}
diff --git a/tests/tests.h b/tests/tests.h
index 32ee4a6..2caf9b3 100644
--- a/tests/tests.h
+++ b/tests/tests.h
@@ -4,4 +4,5 @@
void run_lexer_tests(struct TestContext* context);
void run_parser_tests(struct TestContext* context);
+void run_sema_tests(struct TestContext* context);
void run_codegen_tests(struct TestContext* context);