summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--.gitignore2
-rw-r--r--README.md141
-rw-r--r--TODO5
-rwxr-xr-xbuild.sh50
-rw-r--r--example.m6
-rw-r--r--include/lex.h57
-rw-r--r--include/lib/vector.h21
-rw-r--r--include/platform/osal.h23
-rw-r--r--lex/lex.c212
-rw-r--r--lib/vector.c55
-rw-r--r--platform/elf/main.c72
-rw-r--r--platform/elf/osal.c57
12 files changed, 701 insertions, 0 deletions
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..8a67ec0
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,2 @@
+build/
+metalc
diff --git a/README.md b/README.md
new file mode 100644
index 0000000..eb31eb3
--- /dev/null
+++ b/README.md
@@ -0,0 +1,141 @@
+## Description
+------------------
+This compiler is meant to be a full os dev-
+elopment toolchain. Replaces the function-
+ality of GNU/make, GNU/bash, GNU/gcc, GNU/ld
+and combines its functionality for an easier
+OS-development experience.
+
+"What you see is what you get"
+
+This compiler doubles as AOT shell on my own
+operating system.
+
+If the C language and this compiler has the
+keyword, the functionality of that keyword
+is self-explanatory. If not, or it differs,
+a description is provided below.
+
+There is no floating point support.
+
+Syntax of this language is deterministic,
+yet similar to C, removing function call
+and parantheses ambiguity through the
+"fn" keyword.
+
+For example;
+
+fn g64 test() {
+ <contents>
+ return;
+}
+
+This way, function pointer declarations also
+become deterministic.
+
+E.g.,
+
+fn g64 *ptr_test = &test;
+
+This compiler does not accept ambigous
+parsing.
+
+## Error Levels
+------------------
+- Warning
+Just fix the warnings, they are specific
+for your purposes.
+
+- Error
+Code does not produce functional outcome.
+
+- No Error & No Warning
+Code produces functional and architecturally
+correct outcome, however, said function may
+not be the intended one.
+
+This compiler does not guarantee "correct
+code" in the sense that it will 100% result
+in intended functionality.
+
+## Error Types
+------------------
+- Syntax Error
+- Token Error
+
+## Keywords
+------------------
+- if
+- else
+- elif
+- while
+- for
+- continue
+- break
+- asm
+
+## Data Types
+------------------
+- g<8/16/32/64>
+- u<8/16/32/64>
+- s<8/16/32/64>
+
+## Operators
+------------------
+- Arithmetic; + - * / %
+- Comparison; == != > < => <=
+- Bit; & | ^ << >> ~
+
+## Link-time processing units
+------------------
+Starts with "$" and instructs the layout of
+the produced binary.
+
+- $region "attr" ""extended attr""
+
+## Directives
+------------------
+Starts with "#" and either manipulates the
+outer environment that the compiler lives in,
+or manipulates code depending on the environment.
+
+Functionality depends on platform.
+
+All Directives:
+
+- #run "<command>"
+Runs the following command in the environment
+when compiling.
+
+- #inject FILE/BYTE "<file_path/bytes>" "<addr/symbol>"
+Injects the specified file into the compiled
+binary at the specified location.
+
+## g<8/16/32/64>
+------------------
+This keyword, when used, does not force any
+bit composition to the value like signed or
+unsigned or zeroed.
+
+## u<8/16/32/64>
+------------------
+This keyword is similar to g, however, it
+forces unsigned composition to the loaded
+value.
+
+## s<8/16/32/64>
+------------------
+Forces signed composition.
+
+## region "name" "attr" ""extended attr""
+------------------
+Puts the following code, unless otherwise
+specified, to the specified data region.
+
+This keyword has different functionality
+depending on the used platform.
+
+## asm
+------------------
+This keyword declares an assembly block
+
diff --git a/TODO b/TODO
new file mode 100644
index 0000000..87f77e2
--- /dev/null
+++ b/TODO
@@ -0,0 +1,5 @@
+TODO
+- Abstract lexer error messages.
+- Cleanup lexer, reduce LoCs.
+- Tokenizer should not malloc() for each token, use arena alloc and reserve instead.
+- Add the same error-logic to hex and binary values in lex_number() in lex/lex.c
diff --git a/build.sh b/build.sh
new file mode 100755
index 0000000..5996c05
--- /dev/null
+++ b/build.sh
@@ -0,0 +1,50 @@
+#!/bin/bash
+
+PLATFORM=elf
+
+SRC_DIRECTORIES=(
+ "./arch/"
+ "./lex/"
+ "./parser/"
+ "./lib/"
+ "./platform/${PLATFORM}/"
+)
+
+TARGET_NAME="metalc"
+
+BUILD_DIR="./build/"
+
+CFLAGS=(-I./include/ -g -O1) # -fsanitize=address)
+LDFLAGS=()
+
+function build {
+ mkdir -p "$BUILD_DIR"
+
+ set -e
+
+ local files=()
+
+ echo "[BUILD] Building metalc..."
+ for i in ${SRC_DIRECTORIES[@]}; do
+ local arr=$(find "$i" -name "*.c")
+ for j in ${arr[@]}; do
+ files+=("$j")
+ done
+ done
+
+ gcc "${files[@]}" "${CFLAGS[@]}" -o "$TARGET_NAME"
+ echo "Built ${TARGET_NAME}."
+}
+
+function clean {
+ if [[ -e "$TARGET_NAME" && -f "$TARGET_NAME" ]]; then
+ rm -rf "$TARGET_NAME"
+ echo "[CLEAN] deleted ${TARGET_NAME}"
+ fi
+}
+
+case "$1" in
+ "build") build ;;
+ "clean") clean ;;
+ *) "invalid command" ;;
+esac
diff --git a/example.m b/example.m
new file mode 100644
index 0000000..b54ad5a
--- /dev/null
+++ b/example.m
@@ -0,0 +1,6 @@
+fn g64 bentry() {
+ g64 test;
+ g32 test = (g32)0xB8000;
+ g64 test = 15;
+ return test;
+}
diff --git a/include/lex.h b/include/lex.h
new file mode 100644
index 0000000..a712204
--- /dev/null
+++ b/include/lex.h
@@ -0,0 +1,57 @@
+#ifndef LEX_H
+#define LEX_H
+
+#include <stddef.h>
+
+// Only to be used with compile-time known tokens.
+#define TOKEN(tt, me) (struct token){.type=tt,.lexeme=me,.length=sizeof(me) - 1}
+
+#define TOKEN_INVALID 1
+#define TOKEN_EOP 2
+#define TOKEN_NUMBER 3
+#define TOKEN_HEX 4
+#define TOKEN_BINARY 5
+#define TOKEN_SEMICOLON 6
+#define TOKEN_NEWLINE 7
+#define TOKEN_FUNCTION 8
+#define TOKEN_G8 9
+#define TOKEN_G16 10
+#define TOKEN_G32 11
+#define TOKEN_G64 12
+#define TOKEN_IDENTIFIER 13
+#define TOKEN_PACK 14
+#define TOKEN_BRACES_OPEN 15
+#define TOKEN_BRACES_CLOSE 16
+#define TOKEN_PARAN_OPEN 17
+#define TOKEN_PARAN_CLOSE 18
+#define TOKEN_RETURN 19
+#define TOKEN_EQUALSIGN 20
+#define TOKEN_PLUS 21
+#define TOKEN_MINUS 22
+#define TOKEN_STAR 23
+#define TOKEN_SLASH 24
+
+struct token {
+ int type;
+// size_t line;
+// size_t pos;
+ size_t length;
+ char *lexeme;
+};
+
+struct lex {
+ size_t pos;
+ size_t size;
+ char* buffer;
+ size_t line;
+};
+
+char lex_advance(struct lex *l);
+char lex_seek(struct lex *l, size_t o);
+
+struct token lex_number(struct lex *l);
+struct token lex_directive(struct lex *l);
+struct token lex_link_unit(struct lex *l);
+struct token lex_next(struct lex *l);
+
+#endif
diff --git a/include/lib/vector.h b/include/lib/vector.h
new file mode 100644
index 0000000..33e5b97
--- /dev/null
+++ b/include/lib/vector.h
@@ -0,0 +1,21 @@
+#ifndef VECTOR_H
+#define VECTOR_H
+
+#include <stddef.h>
+
+/**
+ * A simple pointer-only C++-style vector.
+ */
+
+struct vector {
+ size_t size;
+ size_t capacity;
+ void** buffer;
+};
+
+int vector_init(struct vector *v);
+int vector_resize(struct vector *v, size_t nc);
+int vector_emplace_back(struct vector *v, void *p);
+int vector_deinit(struct vector *v);
+
+#endif
diff --git a/include/platform/osal.h b/include/platform/osal.h
new file mode 100644
index 0000000..8eee16b
--- /dev/null
+++ b/include/platform/osal.h
@@ -0,0 +1,23 @@
+#ifndef OSAL_H
+#define OSAL_H
+
+#include <stddef.h>
+#include <stdint.h>
+
+void* mem_alloc(size_t s);
+void mem_free(void *p);
+
+#define FILE_READ (uint8_t) 0b00000001
+#define FILE_WRITE (uint8_t) 0b00000010
+
+struct file {
+ const char *path;
+ const char *buffer;
+ const size_t size;
+};
+
+struct file open_file(const char *path, int flags);
+
+void print(const char *fmt, ...);
+
+#endif
diff --git a/lex/lex.c b/lex/lex.c
new file mode 100644
index 0000000..7ff470b
--- /dev/null
+++ b/lex/lex.c
@@ -0,0 +1,212 @@
+#include "lex.h"
+
+#include <assert.h>
+#include <string.h>
+
+#include <platform/osal.h>
+
+#include <stdio.h>
+
+static struct token keywords[] = {
+ TOKEN(TOKEN_FUNCTION, "fn"),
+ TOKEN(TOKEN_G8, "g8"),
+ TOKEN(TOKEN_G16, "g16"),
+ TOKEN(TOKEN_G32, "g32"),
+ TOKEN(TOKEN_G64, "g64"),
+ TOKEN(TOKEN_PACK, "pack"),
+ TOKEN(TOKEN_RETURN, "return")
+};
+
+inline int lex_is_digit(char c)
+{
+ return c >= '0' && c <= '9';
+}
+
+inline int lex_is_char(char c)
+{
+ return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z');
+}
+
+inline int lex_is_whitespace(char c)
+{
+ return (c == ' ' || c == '\r' || c == '\t');
+}
+
+void lex_skipwhitespace(struct lex *l)
+{
+ int len = 0;
+ while (lex_seek(l, len) &&
+ lex_is_whitespace(lex_seek(l, len))) {
+ len++;
+ }
+
+ l->pos += len; // maybe add safety? (edit lex_advance to add offset)
+}
+
+char lex_advance(struct lex *l)
+{
+ assert(l != NULL);
+ if (!l->buffer[l->pos+1])
+ return l->buffer[l->pos];
+ l->pos++;
+ return l->buffer[l->pos];
+}
+
+char lex_seek(struct lex *l, size_t o)
+{
+ assert(l != NULL);
+ if (l->buffer[l->pos + o])
+ return l->buffer[l->pos + o];
+ return '\0';
+}
+
+struct token lex_token_create(int tt, char* lexeme, size_t s)
+{
+ return (struct token) {.type = tt, .length = s, .lexeme = lexeme};
+}
+
+struct token lex_number(struct lex *l)
+{
+ assert(l != NULL);
+
+ int state = 0;
+
+ if (lex_seek(l, 0) == '0') {
+
+ if (lex_seek(l, 1) && lex_seek(l, 1) == 'x') {
+ state = 1;
+ l->pos += 2;
+ }
+
+ if (lex_seek(l, 1) && lex_seek(l, 1) == 'b') {
+ state = 2;
+ l->pos += 2;
+ }
+ }
+
+ int start = l->pos;
+ int len = 0;
+
+ int tt;
+
+ switch (state) {
+ case 0:
+ tt = TOKEN_NUMBER;
+ while (lex_seek(l, len) && lex_is_digit(lex_seek(l, len)))
+ len++;
+ break;
+ case 1:
+ tt = TOKEN_HEX;
+ while (lex_seek(l, len) && (lex_is_digit(lex_seek(l, len)) || lex_is_char(lex_seek(l, len))))
+ len++;
+ break;
+ case 2:
+ tt = TOKEN_BINARY;
+ while (lex_seek(l, len) && (lex_seek(l, len) == '0' || lex_seek(l, len) == '1')) {
+ len++;
+ }
+ break;
+ default:
+ return (struct token) {.type = TOKEN_INVALID, .lexeme = "number"};
+ }
+
+ l->pos += len;
+
+ return lex_token_create(tt, &l->buffer[start], len);
+}
+
+struct token lex_directive(struct lex *l)
+{
+// return TOKEN(TOKEN_TODO, "directive");
+}
+
+struct token lex_link_unit(struct lex *l)
+{
+// return TOKEN(TOKEN_TODO, "link_unit");
+}
+
+struct token lex_keyword(struct lex *l) //APPROPRIATE INC
+{
+ size_t len = 0;
+ size_t start = l->pos;
+
+ while (lex_seek(l, len)) {
+ if (lex_is_char(lex_seek(l, len)) ||
+ lex_is_digit(lex_seek(l, len)) ||
+ lex_seek(l, len) == '_') {
+ len++;
+ } else {
+ break;
+ }
+ }
+
+ l->pos += len;
+
+ int tt = TOKEN_IDENTIFIER;
+
+ for (size_t i = 0;
+ i < (sizeof(keywords) / sizeof(struct token)); // maybe convert to macro if used somewhere else
+ i++) {
+
+ if (len != keywords[i].length) { // len check for security purposes.
+ continue;
+ }
+
+ if (strncmp(&l->buffer[start], keywords[i].lexeme, keywords[i].length) == 0) {
+ tt = keywords[i].type;
+ break;
+ }
+ }
+
+ return (struct token) {.type = tt, .length=len, .lexeme = &l->buffer[start]};
+
+}
+
+struct token lex_next(struct lex *l)
+{
+ assert(l != NULL);
+
+ if (l->buffer[l->pos] == '\0' || l->pos >= l->size) {
+ return (struct token) {.type = TOKEN_EOP, .lexeme = ""};
+ }
+
+ lex_skipwhitespace(l);
+
+ if (lex_is_digit(lex_seek(l, 0))) {
+ return lex_number(l);
+ }
+
+ switch (lex_seek(l, 0)) {
+ case '\n':
+ l->pos++;
+ l->line++;
+ return TOKEN(TOKEN_NEWLINE, "\\n");
+ case '#':
+ return lex_directive(l);
+ case '$':
+ return lex_link_unit(l);
+ case ';':
+ l->pos++;
+ return TOKEN(TOKEN_SEMICOLON, ";");
+ case '{':
+ l->pos++;
+ return TOKEN(TOKEN_BRACES_OPEN, "{");
+ case '}':
+ l->pos++;
+ return TOKEN(TOKEN_BRACES_CLOSE, "}");
+ case '(':
+ l->pos++;
+ return TOKEN(TOKEN_PARAN_OPEN, "(");
+ case ')':
+ l->pos++;
+ return TOKEN(TOKEN_PARAN_CLOSE, ")");
+ case '=':
+ l->pos++;
+ return TOKEN(TOKEN_EQUALSIGN, "=");
+ case 'a'...'z':
+ case 'A'...'Z':
+ return lex_keyword(l);
+ default:
+ return TOKEN(TOKEN_INVALID, "INVALID");
+ }
+}
diff --git a/lib/vector.c b/lib/vector.c
new file mode 100644
index 0000000..d338db2
--- /dev/null
+++ b/lib/vector.c
@@ -0,0 +1,55 @@
+#include <lib/vector.h>
+
+#include <platform/osal.h>
+
+#include <assert.h>
+#include <string.h>
+
+int vector_init(struct vector *v)
+{
+ assert(v != NULL);
+ size_t initial = 50;
+ v->size = 0;
+ v->capacity = initial;
+ v->buffer = mem_alloc(initial * sizeof(void*)); // add safety
+ if (v->buffer == NULL)
+ return -1;
+ return 0;
+}
+
+int vector_reserve(struct vector *v, size_t nc)
+{
+ assert(v != NULL);
+ void *nb = mem_alloc((nc) * sizeof(void*));
+
+ if (nb == NULL)
+ return -1;
+
+ memcpy(nb, v->buffer, v->size * sizeof(void*));
+ v->capacity = nc;
+
+ if (v->buffer == NULL)
+ return -1;
+
+ mem_free(v->buffer);
+ v->buffer = nb;
+ return 0;
+}
+
+int vector_emplace_back(struct vector *v, void *p)
+{
+ assert(v != NULL);
+
+ if (p == NULL)
+ return -1;
+
+ if (v->size >= v->capacity)
+ if (vector_reserve(v, v->capacity * 2) == -1)
+ return -1;
+
+
+ v->buffer[v->size] = p;
+ v->size++;
+
+ return 0;
+}
diff --git a/platform/elf/main.c b/platform/elf/main.c
new file mode 100644
index 0000000..b2ebb34
--- /dev/null
+++ b/platform/elf/main.c
@@ -0,0 +1,72 @@
+#include <stdio.h>
+
+#include <lex.h>
+#include <lib/vector.h>
+
+#include <platform/osal.h>
+
+#include <stdlib.h>
+
+int main(int argc, char **argv)
+{
+ printf("Platform set to 'linux-elf'.\n");
+
+ if (argc == 1) {
+ printf("You must specify a file to compile!\n");
+ exit(1);
+ }
+
+ struct file fl = open_file(argv[1], FILE_READ);
+
+ struct vector token_storage = {0};
+ if (vector_init(&token_storage)) {
+ printf("error: vector_init()");
+ exit(1);
+ }
+
+ struct lex lex = {0};
+ lex.line = 0;
+ lex.buffer = fl.buffer;
+ lex.size = fl.size;
+ lex.pos = 0;
+
+ struct token *tkn = NULL;
+
+ size_t size = 0;
+
+ do {
+ tkn = mem_alloc(sizeof(struct token)); // Change mem alloc with arena alloc
+
+ if (tkn == NULL) {
+ printf("error: token alloc");
+ exit(1);
+ }
+
+ *tkn = lex_next(&lex);
+
+ // printf("Tokenized: %.*s, with length: %d.\n", (int)tkn->length, tkn->lexeme, (int)tkn->type);
+
+ if (vector_emplace_back(&token_storage, tkn) == -1) {
+ printf("error: vector_emplace_back()");
+ exit(1);
+ }
+
+ } while (tkn != NULL &&
+ tkn->type != TOKEN_INVALID &&
+ tkn->type != TOKEN_EOP);
+
+ for (int i = 0; i < token_storage.size; i++) {
+ if (token_storage.buffer[i] == NULL)
+ break;
+
+ printf("Tokenized '%.*s' with type: %d\n", //, on line: %d at position: %d\n",
+ (int)((struct token*) token_storage.buffer[i])->length,
+ ((struct token*)token_storage.buffer[i])->lexeme,
+ (int)((struct token*)token_storage.buffer[i])->type
+// ((struct token*)token_storage.buffer[i])->line,
+// ((struct token*)token_storage.buffer[i])->pos
+ );
+ }
+
+ return 0;
+}
diff --git a/platform/elf/osal.c b/platform/elf/osal.c
new file mode 100644
index 0000000..414cd80
--- /dev/null
+++ b/platform/elf/osal.c
@@ -0,0 +1,57 @@
+#include <platform/osal.h>
+
+#include <stdlib.h>
+#include <stdio.h>
+
+#include <stdarg.h>
+
+#include <assert.h>
+
+void *mem_alloc(size_t s)
+{
+ return malloc(s);
+}
+
+void mem_free(void *p)
+{
+ free(p);
+}
+
+struct file open_file(const char *path, int flags)
+{
+ assert(path != NULL);
+
+ FILE *file;
+
+ if (flags & FILE_READ) {
+ file = fopen(path, "rb");
+ } else {
+ printf("Read option not specified for file read!");
+ exit(1);
+ }
+
+ fseek(file, 0, SEEK_END);
+ long fsize = ftell(file);
+ fseek(file, 0, SEEK_SET);
+
+ char *content = malloc(fsize + 1);
+ fread(content, fsize, 1, file);
+ fclose(file);
+
+ content[fsize] = 0;
+
+ return (struct file) {.path=path, .buffer=content, .size=fsize};
+}
+
+void print(const char *fmt, ...)
+{
+ va_list args;
+ va_start(args, fmt);
+ vprintf(fmt, args);
+ va_end(args);
+}
+
+void exitc(int c)
+{
+ exit(c);
+}