summaryrefslogtreecommitdiff
path: root/src/build/cmeta.c
diff options
context:
space:
mode:
Diffstat (limited to 'src/build/cmeta.c')
-rw-r--r--src/build/cmeta.c464
1 files changed, 273 insertions, 191 deletions
diff --git a/src/build/cmeta.c b/src/build/cmeta.c
index 434be76..88ac327 100644
--- a/src/build/cmeta.c
+++ b/src/build/cmeta.c
@@ -17,56 +17,12 @@
#include <stdio.h>
#include <stdlib.h>
+#include "../chunklets/clex.h"
#include "../intdefs.h"
#include "../langext.h"
#include "../os.h"
#include "cmeta.h"
-// lazy inlined 3rd party stuff {{{
-// too lazy to write a C tokenizer at the moment, or indeed probably ever, so
-// let's just yoink some code from a hacked-up copy of chibicc, a nice minimal C
-// compiler with code that's pretty easy to work with. it does leak memory by
-// design, but build stuff is all one-shot so that's fine.
-#include "../3p/chibicc/chibicc.h"
-#include "../3p/chibicc/unicode.c"
-// type sentinels from type.c (don't bring in the rest of type.c because it
-// circularly depends on other stuff and we really only want tokenize here)
-Type *ty_void = &(Type){TY_VOID, 1, 1};
-Type *ty_bool = &(Type){TY_BOOL, 1, 1};
-Type *ty_char = &(Type){TY_CHAR, 1, 1};
-Type *ty_short = &(Type){TY_SHORT, 2, 2};
-Type *ty_int = &(Type){TY_INT, 4, 4};
-Type *ty_long = &(Type){TY_LONG, 8, 8};
-Type *ty_uchar = &(Type){TY_CHAR, 1, 1, true};
-Type *ty_ushort = &(Type){TY_SHORT, 2, 2, true};
-Type *ty_uint = &(Type){TY_INT, 4, 4, true};
-Type *ty_ulong = &(Type){TY_LONG, 8, 8, true};
-Type *ty_float = &(Type){TY_FLOAT, 4, 4};
-Type *ty_double = &(Type){TY_DOUBLE, 8, 8};
-Type *ty_ldouble = &(Type){TY_LDOUBLE, 16, 16};
-// inline just a couple more things, super lazy, but whatever
-static Type *new_type(TypeKind kind, int size, int align) {
- Type *ty = calloc(1, sizeof(Type));
- ty->kind = kind;
- ty->size = size;
- ty->align = align;
- return ty;
-}
-Type *array_of(Type *base, int len) {
- Type *ty = new_type(TY_ARRAY, base->size * len, base->align);
- ty->base = base;
- ty->array_len = len;
- return ty;
-}
-#include "../3p/chibicc/hashmap.c"
-#include "../3p/chibicc/strings.c"
-#include "../3p/chibicc/tokenize.c"
-// }}}
-
-#ifdef _WIN32
-#include "../3p/openbsd/asprintf.c" // missing from libc; plonked here for now
-#endif
-
static cold noreturn die(int status, const char *s) {
fprintf(stderr, "cmeta: fatal: %s\n", s);
exit(status);
@@ -76,188 +32,314 @@ struct cmeta cmeta_loadfile(const os_char *path) {
int f = os_open_read(path);
if_cold (f == -1) die(100, "couldn't open file");
vlong len = os_fsize(f);
- if_cold (len > 1u << 30 - 1) die(2, "input file is far too large");
+ if_cold (len == 0) die(2, "empty source file");
+ // limit this a little lower so that clex_memreq doesn't overflow if (for
+ // some reason!?) the host compiler target is 32-bit. no point worrying as
+ // we should never have a 256MiB source file anyway!
+ if_cold (len > 1u << 28 - 1) die(2, "input file is far too large");
struct cmeta ret;
- ret.sbase = malloc(len + 1);
- ret.sbase[len] = '\0'; // chibicc needs a null terminator
- if_cold (!ret.sbase) die(100, "couldn't allocate memory");
- if_cold (os_read(f, ret.sbase, len) != len) die(100, "couldn't read file");
- int maxitems = len / 4; // shortest word is "END"
+ usize lexmemreq = clex_memreq(len, os_strlen(path));
+ // smallest possible item is END{} (5 chars), but it's nice to be able to go
+ // >> 2, so pretend it's 4 chars. for each item we store 1 32-bit ints and
+ // an 8-bit int, so the memory requirement ends up being clex_memreq() +
+ // len + len >> 2. in total, including the file buffer, that should be 7.25x
+ // the file size. which is pretty reasonable for normal files.
+ usize memreq = lexmemreq + (len << 1) + (len >> 2);
+ void *mem = malloc(memreq);
ret.nitems = 0;
- // eventual overall memory requirement: file size * 6. seems fine to me.
- // current memory requirement: file size * 10, + all the chibicc linked list
- // crap. not as good but we'll continue tolerating it... probably for years!
- //ret.itemoffs = malloc(maxitems * sizeof(*ret.itemoffs));
- //if (!ret.itemoffs) die(100, "couldn't allocate memory");
- ret.itemtoks = malloc(maxitems * sizeof(*ret.itemtoks));
- if_cold (!ret.itemtoks) die(100, "couldn't allocate memory");
- ret.itemtypes = malloc(maxitems * sizeof(*ret.itemtypes));
- if_cold (!ret.itemtypes) die(100, "couldn't allocate memory");
+ ret.itemtoks = mem;
+ // put the string and item bytes at the end so the cmeta stuff is aligned.
+ ret.sbase = (char *)mem + memreq - len;
+ ret.items = (struct cmeta_item *)mem + memreq - (len << 1);
+ if_cold (!mem) die(100, "couldn't allocate memory");
+ if_cold (os_read(f, ret.sbase, len) != len) die(100, "couldn't read file");
os_close(f);
#ifdef _WIN32
- char *realname = malloc(wcslen(path) + 1);
- if_cold (!realname) die(100, "couldn't allocate memory");
+ char *asciiname = malloc(wcslen(path) + 1);
+ if_cold (!asciiname) die(100, "couldn't allocate memory");
// XXX: being lazy about Unicode right now; a general purpose tool should
// implement WTF8 or something. SST itself doesn't have any unicode paths
// though, so we don't really care as much. this code still sucks though.
- *realname = *path;
- for (const ushort *p = path + 1; p[-1]; ++p) realname[p - path] = *p;
+ *asciiname = *path;
+ for (const ushort *p = path + 1; p[-1]; ++p) asciiname[p - path] = *p;
#else
- const char *realname = f;
+ const char *asciiname = f;
#endif
- struct Token *t = tokenize_buf(realname, ret.sbase);
- // everything is THING() or THING {} so we need at least 3 tokens ahead - if
- // we have fewer tokens left in the file we can bail
- if (t && t->next) while (t->next->next) {
- if (!t->at_bol) {
- t = t->next;
- continue;
+ ret.lexer = clex(ret.sbase, len, (u32 *)mem + (len >> 2), asciiname);
+ if (ret.lexer.err) die(2, ret.lexer.err);
+ // everything is THING() or THING {}, and file also ends in an EOL, so once
+ // there's less than 4 tokens left in the file, we can bail.
+ for (u32 i = 0, end = ret.lexer.ntoks - 4; i < end; ++i) {
+ if_hot (ret.lexer.toks[i] != CLEX_TOK_IDENT) continue;
+ // technically we don't have to validate *every* token, but doing so
+ // gives less confusing syntax errors.
+ struct clex_ident_validate_ret val = clex_ident_validate(
+ &ret.lexer, ret.sbase, i);
+ if (val.err) {
+ char buf[CLEX_IDENT_ERRSTR_MEMREQ(PATH_MAX)];
+ clex_ident_errstr(buf, ret.sbase, val.err, val.err_off, asciiname);
+ die(2, buf);
}
+ // everything we match has to be followed by either ( or {
+ if (ret.lexer.toks[i + 1] != CLEX_TOK_OP1) { ++i; continue; }
+ if (val.len > 24) continue; // longer than the longest string below
+ char name[24];
+ clex_ident(&ret.lexer, ret.sbase, i, name);
int type;
- if ((equal(t, "DEF_CVAR") || equal(t, "DEF_CVAR_MIN") ||
- equal(t, "DEF_CVAR_MAX") || equal(t, "DEF_CVAR_MINMAX") ||
- equal(t, "DEF_CVAR_UNREG") || equal(t, "DEF_CVAR_MIN_UNREG") ||
- equal(t, "DEF_CVAR_MAX_UNREG") ||
- equal(t, "DEF_CVAR_MINMAX_UNREG") ||
- equal(t, "DEF_FEAT_CVAR") || equal(t, "DEF_FEAT_CVAR_MIN") ||
- equal(t, "DEF_FEAT_CVAR_MAX") ||
- equal(t, "DEF_FEAT_CVAR_MINMAX")) && equal(t->next, "(")) {
- type = CMETA_ITEM_DEF_CVAR;
- }
- else if ((equal(t, "DEF_CCMD") || equal(t, "DEF_CCMD_HERE") ||
- equal(t, "DEF_CCMD_UNREG") || equal(t, "DEF_CCMD_HERE_UNREG") ||
- equal(t, "DEF_CCMD_PLUSMINUS") ||
- equal(t, "DEF_CCMD_PLUSMINUS_UNREG") ||
- equal(t, "DEF_FEAT_CCMD") || equal(t, "DEF_FEAT_CCMD_HERE") ||
- equal(t, "DEF_FEAT_CCMD_PLUSMINUS")) && equal(t->next, "(")) {
- type = CMETA_ITEM_DEF_CCMD;
- }
- else if ((equal(t, "DEF_EVENT") || equal(t, "DEF_PREDICATE")) &&
- equal(t->next, "(")) {
- type = CMETA_ITEM_DEF_EVENT;
- }
- else if (equal(t, "HANDLE_EVENT") && equal(t->next, "(")) {
- type = CMETA_ITEM_HANDLE_EVENT;
- }
- else if (equal(t, "FEATURE") && equal(t->next, "(")) {
- type = CMETA_ITEM_FEATURE;
- }
- else if ((equal(t, "REQUIRE") || equal(t, "REQUIRE_GAMEDATA") ||
- equal(t, "REQUIRE_GLOBAL") || equal(t, "REQUEST")) &&
- equal(t->next, "(")) {
- type = CMETA_ITEM_REQUIRE;
- }
- else if (equal(t, "GAMESPECIFIC") && equal(t->next, "(")) {
- type = CMETA_ITEM_GAMESPECIFIC;
- }
- else if (equal(t, "PREINIT") && equal(t->next, "{")) {
- type = CMETA_ITEM_PREINIT;
+ int flags = 0;
+ char nextop = '(';
+ // this is kind of dumb code. oh well, good enough probably.
+ switch (val.len) {
+ case 3:
+ if (!memcmp(name, "END", 3)) {
+ type = CMETA_ITEM_END;
+ nextop = '{';
+ break;
+ }
+ continue;
+ case 4:
+ if (!memcmp(name, "INIT", 4)) {
+ type = CMETA_ITEM_INIT;
+ nextop = '{';
+ break;
+ }
+ continue;
+ case 7:
+ if (!memcmp(name, "FEATURE", 7)) {
+ type = CMETA_ITEM_FEATURE;
+ break;
+ }
+ if (!memcmp(name, "PREINIT", 7)) {
+ type = CMETA_ITEM_PREINIT;
+ nextop = '{';
+ break;
+ }
+ if (!memcmp(name, "REQUIRE", 7)) {
+ type = CMETA_ITEM_REQUIRE;
+ break;
+ }
+ if (!memcmp(name, "REQUEST", 7)) {
+ type = CMETA_ITEM_REQUIRE;
+ flags = CMETA_REQUIRE_OPTIONAL;
+ break;
+ }
+ continue;
+ case 8:
+ if (!memcmp(name, "DEF_CCMD", 8)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ break;
+ }
+ if (!memcmp(name, "DEF_CVAR", 8)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ break;
+ }
+ continue;
+ case 9:
+ if (!memcmp(name, "DEF_EVENT", 9)) {
+ type = CMETA_ITEM_DEF_EVENT;
+ break;
+ }
+ continue;
+ case 12:
+ if (!memcmp(name, "DEF_CVAR_MAX", 12) ||
+ !memcmp(name, "DEF_CVAR_MIN", 12)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ break;
+ }
+ if (!memcmp(name, "GAMESPECIFIC", 12)) {
+ type = CMETA_ITEM_GAMESPECIFIC;
+ break;
+ }
+ if (!memcmp(name, "HANDLE_EVENT", 12)) {
+ type = CMETA_ITEM_HANDLE_EVENT;
+ break;
+ }
+ continue;
+ case 13:
+ if (!memcmp(name, "DEF_CCMD_HERE", 13)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ break;
+ }
+ if (!memcmp(name, "DEF_FEAT_CCMD", 13)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ if (!memcmp(name, "DEF_FEAT_CVAR", 13)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ if (!memcmp(name, "DEF_PREDICATE", 13)) {
+ type = CMETA_ITEM_DEF_EVENT;
+ flags = CMETA_EVENT_ISPREDICATE;
+ break;
+ }
+ continue;
+ case 14:
+ if (!memcmp(name, "DEF_CCMD_UNREG", 14)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_UNREG;
+ break;
+ }
+ if (!memcmp(name, "DEF_CVAR_UNREG", 14)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_UNREG;
+ break;
+ }
+ if (!memcmp(name, "REQUIRE_GLOBAL", 14)) {
+ type = CMETA_ITEM_REQUIRE;
+ flags = CMETA_REQUIRE_GLOBAL;
+ break;
+ }
+ continue;
+ case 15:
+ if (!memcmp(name, "DEF_CVAR_MINMAX", 15)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ break;
+ }
+ continue;
+ case 16:
+ if (!memcmp(name, "REQUIRE_GAMEDATA", 16)) {
+ type = CMETA_ITEM_REQUIRE;
+ flags = CMETA_REQUIRE_GAMEDATA;
+ break;
+ }
+ continue;
+ case 17:
+ if (!memcmp(name, "DEF_FEAT_CVAR_MAX", 17) ||
+ !memcmp(name, "DEF_FEAT_CVAR_MIN", 17)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ continue;
+ case 18:
+ if (!memcmp(name, "DEF_CCMD_PLUSMINUS", 18)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_PLUSMINUS;
+ break;
+ }
+ if (!memcmp(name, "DEF_CVAR_MAX_UNREG", 18) ||
+ !memcmp(name, "DEF_CVAR_MIN_UNREG", 18)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_UNREG;
+ break;
+ }
+ if (!memcmp(name, "DEF_FEAT_CCMD_HERE", 18)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_FEAT;
+ break;
+ }
+ continue;
+ case 19:
+ if (!memcmp(name, "DEF_CCMD_HERE_UNREG", 19)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_UNREG;
+ break;
+ }
+ continue;
+ case 20:
+ if (!memcmp(name, "DEF_FEAT_CVAR_MINMAX", 20)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ continue;
+ case 21:
+ if (!memcmp(name, "DEF_CVAR_MINMAX_UNREG", 21)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_UNREG;
+ break;
+ }
+ continue;
+ case 23:
+ if (!memcmp(name, "DEF_FEAT_CCMD_PLUSMINUS", 23)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_FEAT | CMETA_CCMD_PLUSMINUS;
+ break;
+ }
+ continue;
+ case 24:
+ if (!memcmp(name, "DEF_CCMD_PLUSMINUS_UNREG", 24)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_UNREG | CMETA_CCMD_PLUSMINUS;
+ break;
+ }
+ default:
+ continue;
}
- else if (equal(t, "INIT") && equal(t->next, "{")) {
- type = CMETA_ITEM_INIT;
- }
- else if (equal(t, "END") && equal(t->next, "{")) {
- type = CMETA_ITEM_END;
- }
- else {
- t = t->next;
+ if (ret.sbase[ret.lexer.tokoffs[i + 1]] != nextop) {
+ // bump i a little further as we've already looked at at least 2
+ // tokens. this is technically kind of inefficient; in most cases we
+ // can skip more stuff, but we're always scanning for something
+ // specific, so who cares actually, this is good enough.
+ ++i;
continue;
}
- ret.itemtoks[ret.nitems] = t;
- ret.itemtypes[ret.nitems] = type;
+ ret.itemtoks[ret.nitems] = i;
+ ret.items[ret.nitems] = (struct cmeta_item){type, flags};
++ret.nitems;
- // this is kind of inefficient; in most cases we can skip more stuff,
- // but then also, we're always scanning for something specific, so who
- // cares actually, this will do for now.
- t = t->next->next;
+ ++i;
}
return ret;
}
-int cmeta_flags_cvar(const struct cmeta *cm, u32 i) {
- struct Token *t = cm->itemtoks[i];
- switch_exhaust (t->len) {
- // It JUST so happens all of the possible tokens here have a unique
- // length. I swear this wasn't planned. But it IS convenient!
- case 8: case 12: case 15: return 0;
- case 14: case 18: case 21: return CMETA_CVAR_UNREG;
- case 13: case 17: case 20: return CMETA_CVAR_FEAT;
- }
-}
-
-int cmeta_flags_ccmd(const struct cmeta *cm, u32 i) {
- struct Token *t = cm->itemtoks[i];
- switch_exhaust (t->len) {
- case 13: if (t->loc[4] == 'F') return CMETA_CCMD_FEAT;
- case 8: return 0;
- case 18: if (t->loc[4] == 'F') return CMETA_CCMD_FEAT;
- return CMETA_CCMD_PLUSMINUS;
- case 14: case 19: return CMETA_CCMD_UNREG;
- case 23: return CMETA_CCMD_FEAT | CMETA_CCMD_PLUSMINUS;
- case 24: return CMETA_CCMD_UNREG | CMETA_CCMD_PLUSMINUS;
- }
-}
-
-int cmeta_flags_event(const struct cmeta *cm, u32 i) {
- // assuming CMETA_EVENT_ISPREDICATE remains 1, the ternary should
- // optimise out
- return cm->itemtoks[i]->len == 13 ? CMETA_EVENT_ISPREDICATE : 0;
-}
-
-int cmeta_flags_require(const struct cmeta *cm, u32 i) {
- struct Token *t = cm->itemtoks[i];
- // NOTE: this is somewhat more flexible to enable REQUEST_GAMEDATA or
- // something in future, although that's kind of useless currently
- int optflag = t->loc[4] == 'E'; // REQU[E]ST
- switch_exhaust (t->len) {
- case 7: return optflag;
- case 16: return optflag | CMETA_REQUIRE_GAMEDATA;
- case 14: return optflag | CMETA_REQUIRE_GLOBAL;
- };
-}
-
-int cmeta_nparams(const struct cmeta *cm, u32 i) {
+int cmeta_nparams(const struct cmeta *cm, u32 item) {
int argc = 1, nest = 0;
- struct Token *t = cm->itemtoks[i]->next->next;
- if (equal(t, ")")) return 0; // XXX: stupid special case, surely improvable?
- for (; t; t = t->next) {
- if (equal(t, "(")) { ++nest; continue; }
- if (!nest && equal(t, ",")) ++argc;
- else if (equal(t, ")") && !nest--) break;
+ int i = cm->itemtoks[item] + 2; // get past the first (
+ // handle immediate ) - XXX: stupid special case, surely improvable?
+ if (clex_isrparen(&cm->lexer, cm->sbase, i)) return 0;
+ for (; i < cm->lexer.ntoks; ++i) {
+ if (clex_islparen(&cm->lexer, cm->sbase, i)) { ++nest; continue; }
+ if (!nest && clex_iscomma(&cm->lexer, cm->sbase, i)) ++argc;
+ else if (clex_isrparen(&cm->lexer, cm->sbase, i) && !nest--) break;
}
if (nest != -1) return 0; // XXX: any need to do anything better here?
return argc;
}
struct cmeta_param_iter cmeta_param_iter_init(const struct cmeta *cm, u32 i) {
- return (struct cmeta_param_iter){cm->itemtoks[i]->next->next};
+ return (struct cmeta_param_iter){cm->itemtoks[i] + 2};
}
-struct cmeta_slice cmeta_param_iter(struct cmeta_param_iter *it) {
+struct cmeta_slice cmeta_param_iter(const struct cmeta *cm,
+ struct cmeta_param_iter *it) {
int nest = 0;
- const char *start = it->cur->loc;
- for (struct Token *last = 0; it->cur;
- last = it->cur, it->cur = it->cur->next) {
- if (equal(it->cur, "(")) { ++nest; continue; }
- if (!nest && equal(it->cur, ",")) {
- if (!last) { // , immediately after (, for some reason. treat as ""
- return (struct cmeta_slice){start, 0};
- }
- it->cur = it->cur->next;
+ const char *start = cm->sbase + cm->lexer.tokoffs[it->i];
+ for (; it->i < cm->lexer.ntoks; ++it->i) {
+ if (clex_islparen(&cm->lexer, cm->sbase, it->i)) {
+ ++nest;
+ continue;
}
- else if (equal(it->cur, ")") && !nest--) {
- if (!last) break;
+ if (!nest && clex_iscomma(&cm->lexer, cm->sbase, it->i)) {
+ const char *end = cm->sbase + cm->lexer.tokoffs[it->i];
+ // XXX: to avoid picking up random whitespace we should get the
+ // previous token and ask clex for its extent, but I've not yet
+ // implemented full parsing of all variable-length tokens (i.e.
+ // numbers and string/char literals). not worrying about it too much
+ // at the moment since it's not really a problem anyway in practice
+ ++it->i; // skip past comma for next time
+ return (struct cmeta_slice){start, end - start};
}
- else {
- continue;
+ else if (clex_isrparen(&cm->lexer, cm->sbase, it->i) && !nest--) {
+ // annoying case: 0 args is different from ", )" (empty arg)
+ if (clex_islparen(&cm->lexer, cm->sbase, it->i - 1)) break;
+ const char *end = cm->sbase + cm->lexer.tokoffs[it->i];
+ it->i = -1u; // force next call to return {0, 0}
+ return (struct cmeta_slice){start, end - start};
}
- return (struct cmeta_slice){start, last->loc - start + last->len};
}
return (struct cmeta_slice){0, 0};
}
-u32 cmeta_line(const struct cmeta *cm, u32 i) {
- return cm->itemtoks[i]->line_no;
+cold u32 cmeta_line(const struct cmeta *cm, u32 i) {
+ u32 line = 1;
+ for (u32 off = 0, end = cm->lexer.tokoffs[cm->itemtoks[i]];
+ off != end; ++off) {
+ line += cm->sbase[off] == '\n';
+ }
+ return line;
}
// vi: sw=4 ts=4 noet tw=80 cc=80 fdm=marker