diff options
| -rw-r--r-- | .gitignore | 2 | ||||
| -rw-r--r-- | README.md | 141 | ||||
| -rw-r--r-- | TODO | 5 | ||||
| -rwxr-xr-x | build.sh | 50 | ||||
| -rw-r--r-- | example.m | 6 | ||||
| -rw-r--r-- | include/lex.h | 57 | ||||
| -rw-r--r-- | include/lib/vector.h | 21 | ||||
| -rw-r--r-- | include/platform/osal.h | 23 | ||||
| -rw-r--r-- | lex/lex.c | 212 | ||||
| -rw-r--r-- | lib/vector.c | 55 | ||||
| -rw-r--r-- | platform/elf/main.c | 72 | ||||
| -rw-r--r-- | platform/elf/osal.c | 57 |
12 files changed, 701 insertions, 0 deletions
diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..8a67ec0 --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +build/ +metalc diff --git a/README.md b/README.md new file mode 100644 index 0000000..eb31eb3 --- /dev/null +++ b/README.md @@ -0,0 +1,141 @@ +## Description +------------------ +This compiler is meant to be a full os dev- +elopment toolchain. Replaces the function- +ality of GNU/make, GNU/bash, GNU/gcc, GNU/ld +and combines its functionality for an easier +OS-development experience. + +"What you see is what you get" + +This compiler doubles as AOT shell on my own +operating system. + +If the C language and this compiler has the +keyword, the functionality of that keyword +is self-explanatory. If not, or it differs, +a description is provided below. + +There is no floating point support. + +Syntax of this language is deterministic, +yet similar to C, removing function call +and parantheses ambiguity through the +"fn" keyword. + +For example; + +fn g64 test() { + <contents> + return; +} + +This way, function pointer declarations also +become deterministic. + +E.g., + +fn g64 *ptr_test = &test; + +This compiler does not accept ambigous +parsing. + +## Error Levels +------------------ +- Warning +Just fix the warnings, they are specific +for your purposes. + +- Error +Code does not produce functional outcome. + +- No Error & No Warning +Code produces functional and architecturally +correct outcome, however, said function may +not be the intended one. + +This compiler does not guarantee "correct +code" in the sense that it will 100% result +in intended functionality. + +## Error Types +------------------ +- Syntax Error +- Token Error + +## Keywords +------------------ +- if +- else +- elif +- while +- for +- continue +- break +- asm + +## Data Types +------------------ +- g<8/16/32/64> +- u<8/16/32/64> +- s<8/16/32/64> + +## Operators +------------------ +- Arithmetic; + - * / % +- Comparison; == != > < => <= +- Bit; & | ^ << >> ~ + +## Link-time processing units +------------------ +Starts with "$" and instructs the layout of +the produced binary. + +- $region "attr" ""extended attr"" + +## Directives +------------------ +Starts with "#" and either manipulates the +outer environment that the compiler lives in, +or manipulates code depending on the environment. + +Functionality depends on platform. + +All Directives: + +- #run "<command>" +Runs the following command in the environment +when compiling. + +- #inject FILE/BYTE "<file_path/bytes>" "<addr/symbol>" +Injects the specified file into the compiled +binary at the specified location. + +## g<8/16/32/64> +------------------ +This keyword, when used, does not force any +bit composition to the value like signed or +unsigned or zeroed. + +## u<8/16/32/64> +------------------ +This keyword is similar to g, however, it +forces unsigned composition to the loaded +value. + +## s<8/16/32/64> +------------------ +Forces signed composition. + +## region "name" "attr" ""extended attr"" +------------------ +Puts the following code, unless otherwise +specified, to the specified data region. + +This keyword has different functionality +depending on the used platform. + +## asm +------------------ +This keyword declares an assembly block + @@ -0,0 +1,5 @@ +TODO +- Abstract lexer error messages. +- Cleanup lexer, reduce LoCs. +- Tokenizer should not malloc() for each token, use arena alloc and reserve instead. +- Add the same error-logic to hex and binary values in lex_number() in lex/lex.c diff --git a/build.sh b/build.sh new file mode 100755 index 0000000..5996c05 --- /dev/null +++ b/build.sh @@ -0,0 +1,50 @@ +#!/bin/bash + +PLATFORM=elf + +SRC_DIRECTORIES=( + "./arch/" + "./lex/" + "./parser/" + "./lib/" + "./platform/${PLATFORM}/" +) + +TARGET_NAME="metalc" + +BUILD_DIR="./build/" + +CFLAGS=(-I./include/ -g -O1) # -fsanitize=address) +LDFLAGS=() + +function build { + mkdir -p "$BUILD_DIR" + + set -e + + local files=() + + echo "[BUILD] Building metalc..." + for i in ${SRC_DIRECTORIES[@]}; do + local arr=$(find "$i" -name "*.c") + for j in ${arr[@]}; do + files+=("$j") + done + done + + gcc "${files[@]}" "${CFLAGS[@]}" -o "$TARGET_NAME" + echo "Built ${TARGET_NAME}." +} + +function clean { + if [[ -e "$TARGET_NAME" && -f "$TARGET_NAME" ]]; then + rm -rf "$TARGET_NAME" + echo "[CLEAN] deleted ${TARGET_NAME}" + fi +} + +case "$1" in + "build") build ;; + "clean") clean ;; + *) "invalid command" ;; +esac diff --git a/example.m b/example.m new file mode 100644 index 0000000..b54ad5a --- /dev/null +++ b/example.m @@ -0,0 +1,6 @@ +fn g64 bentry() { + g64 test; + g32 test = (g32)0xB8000; + g64 test = 15; + return test; +} diff --git a/include/lex.h b/include/lex.h new file mode 100644 index 0000000..a712204 --- /dev/null +++ b/include/lex.h @@ -0,0 +1,57 @@ +#ifndef LEX_H +#define LEX_H + +#include <stddef.h> + +// Only to be used with compile-time known tokens. +#define TOKEN(tt, me) (struct token){.type=tt,.lexeme=me,.length=sizeof(me) - 1} + +#define TOKEN_INVALID 1 +#define TOKEN_EOP 2 +#define TOKEN_NUMBER 3 +#define TOKEN_HEX 4 +#define TOKEN_BINARY 5 +#define TOKEN_SEMICOLON 6 +#define TOKEN_NEWLINE 7 +#define TOKEN_FUNCTION 8 +#define TOKEN_G8 9 +#define TOKEN_G16 10 +#define TOKEN_G32 11 +#define TOKEN_G64 12 +#define TOKEN_IDENTIFIER 13 +#define TOKEN_PACK 14 +#define TOKEN_BRACES_OPEN 15 +#define TOKEN_BRACES_CLOSE 16 +#define TOKEN_PARAN_OPEN 17 +#define TOKEN_PARAN_CLOSE 18 +#define TOKEN_RETURN 19 +#define TOKEN_EQUALSIGN 20 +#define TOKEN_PLUS 21 +#define TOKEN_MINUS 22 +#define TOKEN_STAR 23 +#define TOKEN_SLASH 24 + +struct token { + int type; +// size_t line; +// size_t pos; + size_t length; + char *lexeme; +}; + +struct lex { + size_t pos; + size_t size; + char* buffer; + size_t line; +}; + +char lex_advance(struct lex *l); +char lex_seek(struct lex *l, size_t o); + +struct token lex_number(struct lex *l); +struct token lex_directive(struct lex *l); +struct token lex_link_unit(struct lex *l); +struct token lex_next(struct lex *l); + +#endif diff --git a/include/lib/vector.h b/include/lib/vector.h new file mode 100644 index 0000000..33e5b97 --- /dev/null +++ b/include/lib/vector.h @@ -0,0 +1,21 @@ +#ifndef VECTOR_H +#define VECTOR_H + +#include <stddef.h> + +/** + * A simple pointer-only C++-style vector. + */ + +struct vector { + size_t size; + size_t capacity; + void** buffer; +}; + +int vector_init(struct vector *v); +int vector_resize(struct vector *v, size_t nc); +int vector_emplace_back(struct vector *v, void *p); +int vector_deinit(struct vector *v); + +#endif diff --git a/include/platform/osal.h b/include/platform/osal.h new file mode 100644 index 0000000..8eee16b --- /dev/null +++ b/include/platform/osal.h @@ -0,0 +1,23 @@ +#ifndef OSAL_H +#define OSAL_H + +#include <stddef.h> +#include <stdint.h> + +void* mem_alloc(size_t s); +void mem_free(void *p); + +#define FILE_READ (uint8_t) 0b00000001 +#define FILE_WRITE (uint8_t) 0b00000010 + +struct file { + const char *path; + const char *buffer; + const size_t size; +}; + +struct file open_file(const char *path, int flags); + +void print(const char *fmt, ...); + +#endif diff --git a/lex/lex.c b/lex/lex.c new file mode 100644 index 0000000..7ff470b --- /dev/null +++ b/lex/lex.c @@ -0,0 +1,212 @@ +#include "lex.h" + +#include <assert.h> +#include <string.h> + +#include <platform/osal.h> + +#include <stdio.h> + +static struct token keywords[] = { + TOKEN(TOKEN_FUNCTION, "fn"), + TOKEN(TOKEN_G8, "g8"), + TOKEN(TOKEN_G16, "g16"), + TOKEN(TOKEN_G32, "g32"), + TOKEN(TOKEN_G64, "g64"), + TOKEN(TOKEN_PACK, "pack"), + TOKEN(TOKEN_RETURN, "return") +}; + +inline int lex_is_digit(char c) +{ + return c >= '0' && c <= '9'; +} + +inline int lex_is_char(char c) +{ + return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z'); +} + +inline int lex_is_whitespace(char c) +{ + return (c == ' ' || c == '\r' || c == '\t'); +} + +void lex_skipwhitespace(struct lex *l) +{ + int len = 0; + while (lex_seek(l, len) && + lex_is_whitespace(lex_seek(l, len))) { + len++; + } + + l->pos += len; // maybe add safety? (edit lex_advance to add offset) +} + +char lex_advance(struct lex *l) +{ + assert(l != NULL); + if (!l->buffer[l->pos+1]) + return l->buffer[l->pos]; + l->pos++; + return l->buffer[l->pos]; +} + +char lex_seek(struct lex *l, size_t o) +{ + assert(l != NULL); + if (l->buffer[l->pos + o]) + return l->buffer[l->pos + o]; + return '\0'; +} + +struct token lex_token_create(int tt, char* lexeme, size_t s) +{ + return (struct token) {.type = tt, .length = s, .lexeme = lexeme}; +} + +struct token lex_number(struct lex *l) +{ + assert(l != NULL); + + int state = 0; + + if (lex_seek(l, 0) == '0') { + + if (lex_seek(l, 1) && lex_seek(l, 1) == 'x') { + state = 1; + l->pos += 2; + } + + if (lex_seek(l, 1) && lex_seek(l, 1) == 'b') { + state = 2; + l->pos += 2; + } + } + + int start = l->pos; + int len = 0; + + int tt; + + switch (state) { + case 0: + tt = TOKEN_NUMBER; + while (lex_seek(l, len) && lex_is_digit(lex_seek(l, len))) + len++; + break; + case 1: + tt = TOKEN_HEX; + while (lex_seek(l, len) && (lex_is_digit(lex_seek(l, len)) || lex_is_char(lex_seek(l, len)))) + len++; + break; + case 2: + tt = TOKEN_BINARY; + while (lex_seek(l, len) && (lex_seek(l, len) == '0' || lex_seek(l, len) == '1')) { + len++; + } + break; + default: + return (struct token) {.type = TOKEN_INVALID, .lexeme = "number"}; + } + + l->pos += len; + + return lex_token_create(tt, &l->buffer[start], len); +} + +struct token lex_directive(struct lex *l) +{ +// return TOKEN(TOKEN_TODO, "directive"); +} + +struct token lex_link_unit(struct lex *l) +{ +// return TOKEN(TOKEN_TODO, "link_unit"); +} + +struct token lex_keyword(struct lex *l) //APPROPRIATE INC +{ + size_t len = 0; + size_t start = l->pos; + + while (lex_seek(l, len)) { + if (lex_is_char(lex_seek(l, len)) || + lex_is_digit(lex_seek(l, len)) || + lex_seek(l, len) == '_') { + len++; + } else { + break; + } + } + + l->pos += len; + + int tt = TOKEN_IDENTIFIER; + + for (size_t i = 0; + i < (sizeof(keywords) / sizeof(struct token)); // maybe convert to macro if used somewhere else + i++) { + + if (len != keywords[i].length) { // len check for security purposes. + continue; + } + + if (strncmp(&l->buffer[start], keywords[i].lexeme, keywords[i].length) == 0) { + tt = keywords[i].type; + break; + } + } + + return (struct token) {.type = tt, .length=len, .lexeme = &l->buffer[start]}; + +} + +struct token lex_next(struct lex *l) +{ + assert(l != NULL); + + if (l->buffer[l->pos] == '\0' || l->pos >= l->size) { + return (struct token) {.type = TOKEN_EOP, .lexeme = ""}; + } + + lex_skipwhitespace(l); + + if (lex_is_digit(lex_seek(l, 0))) { + return lex_number(l); + } + + switch (lex_seek(l, 0)) { + case '\n': + l->pos++; + l->line++; + return TOKEN(TOKEN_NEWLINE, "\\n"); + case '#': + return lex_directive(l); + case '$': + return lex_link_unit(l); + case ';': + l->pos++; + return TOKEN(TOKEN_SEMICOLON, ";"); + case '{': + l->pos++; + return TOKEN(TOKEN_BRACES_OPEN, "{"); + case '}': + l->pos++; + return TOKEN(TOKEN_BRACES_CLOSE, "}"); + case '(': + l->pos++; + return TOKEN(TOKEN_PARAN_OPEN, "("); + case ')': + l->pos++; + return TOKEN(TOKEN_PARAN_CLOSE, ")"); + case '=': + l->pos++; + return TOKEN(TOKEN_EQUALSIGN, "="); + case 'a'...'z': + case 'A'...'Z': + return lex_keyword(l); + default: + return TOKEN(TOKEN_INVALID, "INVALID"); + } +} diff --git a/lib/vector.c b/lib/vector.c new file mode 100644 index 0000000..d338db2 --- /dev/null +++ b/lib/vector.c @@ -0,0 +1,55 @@ +#include <lib/vector.h> + +#include <platform/osal.h> + +#include <assert.h> +#include <string.h> + +int vector_init(struct vector *v) +{ + assert(v != NULL); + size_t initial = 50; + v->size = 0; + v->capacity = initial; + v->buffer = mem_alloc(initial * sizeof(void*)); // add safety + if (v->buffer == NULL) + return -1; + return 0; +} + +int vector_reserve(struct vector *v, size_t nc) +{ + assert(v != NULL); + void *nb = mem_alloc((nc) * sizeof(void*)); + + if (nb == NULL) + return -1; + + memcpy(nb, v->buffer, v->size * sizeof(void*)); + v->capacity = nc; + + if (v->buffer == NULL) + return -1; + + mem_free(v->buffer); + v->buffer = nb; + return 0; +} + +int vector_emplace_back(struct vector *v, void *p) +{ + assert(v != NULL); + + if (p == NULL) + return -1; + + if (v->size >= v->capacity) + if (vector_reserve(v, v->capacity * 2) == -1) + return -1; + + + v->buffer[v->size] = p; + v->size++; + + return 0; +} diff --git a/platform/elf/main.c b/platform/elf/main.c new file mode 100644 index 0000000..b2ebb34 --- /dev/null +++ b/platform/elf/main.c @@ -0,0 +1,72 @@ +#include <stdio.h> + +#include <lex.h> +#include <lib/vector.h> + +#include <platform/osal.h> + +#include <stdlib.h> + +int main(int argc, char **argv) +{ + printf("Platform set to 'linux-elf'.\n"); + + if (argc == 1) { + printf("You must specify a file to compile!\n"); + exit(1); + } + + struct file fl = open_file(argv[1], FILE_READ); + + struct vector token_storage = {0}; + if (vector_init(&token_storage)) { + printf("error: vector_init()"); + exit(1); + } + + struct lex lex = {0}; + lex.line = 0; + lex.buffer = fl.buffer; + lex.size = fl.size; + lex.pos = 0; + + struct token *tkn = NULL; + + size_t size = 0; + + do { + tkn = mem_alloc(sizeof(struct token)); // Change mem alloc with arena alloc + + if (tkn == NULL) { + printf("error: token alloc"); + exit(1); + } + + *tkn = lex_next(&lex); + + // printf("Tokenized: %.*s, with length: %d.\n", (int)tkn->length, tkn->lexeme, (int)tkn->type); + + if (vector_emplace_back(&token_storage, tkn) == -1) { + printf("error: vector_emplace_back()"); + exit(1); + } + + } while (tkn != NULL && + tkn->type != TOKEN_INVALID && + tkn->type != TOKEN_EOP); + + for (int i = 0; i < token_storage.size; i++) { + if (token_storage.buffer[i] == NULL) + break; + + printf("Tokenized '%.*s' with type: %d\n", //, on line: %d at position: %d\n", + (int)((struct token*) token_storage.buffer[i])->length, + ((struct token*)token_storage.buffer[i])->lexeme, + (int)((struct token*)token_storage.buffer[i])->type +// ((struct token*)token_storage.buffer[i])->line, +// ((struct token*)token_storage.buffer[i])->pos + ); + } + + return 0; +} diff --git a/platform/elf/osal.c b/platform/elf/osal.c new file mode 100644 index 0000000..414cd80 --- /dev/null +++ b/platform/elf/osal.c @@ -0,0 +1,57 @@ +#include <platform/osal.h> + +#include <stdlib.h> +#include <stdio.h> + +#include <stdarg.h> + +#include <assert.h> + +void *mem_alloc(size_t s) +{ + return malloc(s); +} + +void mem_free(void *p) +{ + free(p); +} + +struct file open_file(const char *path, int flags) +{ + assert(path != NULL); + + FILE *file; + + if (flags & FILE_READ) { + file = fopen(path, "rb"); + } else { + printf("Read option not specified for file read!"); + exit(1); + } + + fseek(file, 0, SEEK_END); + long fsize = ftell(file); + fseek(file, 0, SEEK_SET); + + char *content = malloc(fsize + 1); + fread(content, fsize, 1, file); + fclose(file); + + content[fsize] = 0; + + return (struct file) {.path=path, .buffer=content, .size=fsize}; +} + +void print(const char *fmt, ...) +{ + va_list args; + va_start(args, fmt); + vprintf(fmt, args); + va_end(args); +} + +void exitc(int c) +{ + exit(c); +} |
