summaryrefslogtreecommitdiff
path: root/src/build/cmeta.c
diff options
context:
space:
mode:
authorGravatar Michael Smith <mikesmiffy128@gmail.com> 2025-12-13 18:26:51 +0000
committerGravatar Michael Smith <mikesmiffy128@gmail.com> 2026-02-16 19:11:24 +0000
commit40f9d989df2c1ff2f567656ccbdbfc0d97e34a77 (patch)
treec3a0e69b2823adfe1d5ccf0f48d77b288e55740a /src/build/cmeta.c
parentf386f2a1fcae885967278710fe997c56e2183476 (diff)
downloadsst-40f9d989df2c1ff2f567656ccbdbfc0d97e34a77.tar.gz
sst-40f9d989df2c1ff2f567656ccbdbfc0d97e34a77.zip
Switch cmeta from chibicc to a new homegrown lexer
This lexer is being written as a chunklet, not quite quite complete yet as it lacks some functionality to make it generally useful for things, but good enough for the gluegen use case now. So it's in the repo now and we can go ahead and use it for this instead of having this hacked-to-pieces third party thing that's nowhere near as efficient. In the future the goal is to have a decently usable library for any sort of C metaprogramming needs, not just within this project. This design does the data-oriented thing of storing as little about each token as possible (just 5 bytes), and re-lexing specific pieces only when necessary. In some cases, full validation of correct syntax is only possible through this secondary step. Since the use case is metaprogramming and code-generation rather than the reimplementation of Clang, this slight sloppiness in validation doesn't seem too bad. It's been a while since I did a proper performance measurement of this code but in some crude tests I did in the past the primary tokenisation step was running at well over 500MB/s, which is fast enough for me. It's possible that this version is a tiny bit slower due to the added complexity of making UCNs work. UCNs, incidentally, are one of the dumbest features of C by far. However, unlike trigraphs - which this lexer does not handle - UCNs are still in the language, so it's kind of sort of necessary to support them. It's probably still possible to come up with a faster design with some SIMD trickery, but the main loop is currently branchless and the lookup tables are too big for the SIMD lookup things, so that seems kind of hard. A separate SIMD path for whitespace or comment runs seems dubious as it would introduce branch mispredictions everywhere. I also made previous attempts to unroll the main loop and every attempt just made it slower, so I guess code size is a significant factor. Optimising the secondary tokenisation of identifiers (and later numerals and string/character literals once those are handled) is still on the cards, but since that happens less often, I don't know how much difference it'll make. At any rate, in a multithreaded context this thing would already come pretty close to SSD speeds, if open-read-close syscall overhead doesn't get in the way first. I imagine it's fast enough for anyone who hasn't *already* written something faster.
Diffstat (limited to 'src/build/cmeta.c')
-rw-r--r--src/build/cmeta.c464
1 files changed, 273 insertions, 191 deletions
diff --git a/src/build/cmeta.c b/src/build/cmeta.c
index 434be76..88ac327 100644
--- a/src/build/cmeta.c
+++ b/src/build/cmeta.c
@@ -17,56 +17,12 @@
#include <stdio.h>
#include <stdlib.h>
+#include "../chunklets/clex.h"
#include "../intdefs.h"
#include "../langext.h"
#include "../os.h"
#include "cmeta.h"
-// lazy inlined 3rd party stuff {{{
-// too lazy to write a C tokenizer at the moment, or indeed probably ever, so
-// let's just yoink some code from a hacked-up copy of chibicc, a nice minimal C
-// compiler with code that's pretty easy to work with. it does leak memory by
-// design, but build stuff is all one-shot so that's fine.
-#include "../3p/chibicc/chibicc.h"
-#include "../3p/chibicc/unicode.c"
-// type sentinels from type.c (don't bring in the rest of type.c because it
-// circularly depends on other stuff and we really only want tokenize here)
-Type *ty_void = &(Type){TY_VOID, 1, 1};
-Type *ty_bool = &(Type){TY_BOOL, 1, 1};
-Type *ty_char = &(Type){TY_CHAR, 1, 1};
-Type *ty_short = &(Type){TY_SHORT, 2, 2};
-Type *ty_int = &(Type){TY_INT, 4, 4};
-Type *ty_long = &(Type){TY_LONG, 8, 8};
-Type *ty_uchar = &(Type){TY_CHAR, 1, 1, true};
-Type *ty_ushort = &(Type){TY_SHORT, 2, 2, true};
-Type *ty_uint = &(Type){TY_INT, 4, 4, true};
-Type *ty_ulong = &(Type){TY_LONG, 8, 8, true};
-Type *ty_float = &(Type){TY_FLOAT, 4, 4};
-Type *ty_double = &(Type){TY_DOUBLE, 8, 8};
-Type *ty_ldouble = &(Type){TY_LDOUBLE, 16, 16};
-// inline just a couple more things, super lazy, but whatever
-static Type *new_type(TypeKind kind, int size, int align) {
- Type *ty = calloc(1, sizeof(Type));
- ty->kind = kind;
- ty->size = size;
- ty->align = align;
- return ty;
-}
-Type *array_of(Type *base, int len) {
- Type *ty = new_type(TY_ARRAY, base->size * len, base->align);
- ty->base = base;
- ty->array_len = len;
- return ty;
-}
-#include "../3p/chibicc/hashmap.c"
-#include "../3p/chibicc/strings.c"
-#include "../3p/chibicc/tokenize.c"
-// }}}
-
-#ifdef _WIN32
-#include "../3p/openbsd/asprintf.c" // missing from libc; plonked here for now
-#endif
-
static cold noreturn die(int status, const char *s) {
fprintf(stderr, "cmeta: fatal: %s\n", s);
exit(status);
@@ -76,188 +32,314 @@ struct cmeta cmeta_loadfile(const os_char *path) {
int f = os_open_read(path);
if_cold (f == -1) die(100, "couldn't open file");
vlong len = os_fsize(f);
- if_cold (len > 1u << 30 - 1) die(2, "input file is far too large");
+ if_cold (len == 0) die(2, "empty source file");
+ // limit this a little lower so that clex_memreq doesn't overflow if (for
+ // some reason!?) the host compiler target is 32-bit. no point worrying as
+ // we should never have a 256MiB source file anyway!
+ if_cold (len > 1u << 28 - 1) die(2, "input file is far too large");
struct cmeta ret;
- ret.sbase = malloc(len + 1);
- ret.sbase[len] = '\0'; // chibicc needs a null terminator
- if_cold (!ret.sbase) die(100, "couldn't allocate memory");
- if_cold (os_read(f, ret.sbase, len) != len) die(100, "couldn't read file");
- int maxitems = len / 4; // shortest word is "END"
+ usize lexmemreq = clex_memreq(len, os_strlen(path));
+ // smallest possible item is END{} (5 chars), but it's nice to be able to go
+ // >> 2, so pretend it's 4 chars. for each item we store 1 32-bit ints and
+ // an 8-bit int, so the memory requirement ends up being clex_memreq() +
+ // len + len >> 2. in total, including the file buffer, that should be 7.25x
+ // the file size. which is pretty reasonable for normal files.
+ usize memreq = lexmemreq + (len << 1) + (len >> 2);
+ void *mem = malloc(memreq);
ret.nitems = 0;
- // eventual overall memory requirement: file size * 6. seems fine to me.
- // current memory requirement: file size * 10, + all the chibicc linked list
- // crap. not as good but we'll continue tolerating it... probably for years!
- //ret.itemoffs = malloc(maxitems * sizeof(*ret.itemoffs));
- //if (!ret.itemoffs) die(100, "couldn't allocate memory");
- ret.itemtoks = malloc(maxitems * sizeof(*ret.itemtoks));
- if_cold (!ret.itemtoks) die(100, "couldn't allocate memory");
- ret.itemtypes = malloc(maxitems * sizeof(*ret.itemtypes));
- if_cold (!ret.itemtypes) die(100, "couldn't allocate memory");
+ ret.itemtoks = mem;
+ // put the string and item bytes at the end so the cmeta stuff is aligned.
+ ret.sbase = (char *)mem + memreq - len;
+ ret.items = (struct cmeta_item *)mem + memreq - (len << 1);
+ if_cold (!mem) die(100, "couldn't allocate memory");
+ if_cold (os_read(f, ret.sbase, len) != len) die(100, "couldn't read file");
os_close(f);
#ifdef _WIN32
- char *realname = malloc(wcslen(path) + 1);
- if_cold (!realname) die(100, "couldn't allocate memory");
+ char *asciiname = malloc(wcslen(path) + 1);
+ if_cold (!asciiname) die(100, "couldn't allocate memory");
// XXX: being lazy about Unicode right now; a general purpose tool should
// implement WTF8 or something. SST itself doesn't have any unicode paths
// though, so we don't really care as much. this code still sucks though.
- *realname = *path;
- for (const ushort *p = path + 1; p[-1]; ++p) realname[p - path] = *p;
+ *asciiname = *path;
+ for (const ushort *p = path + 1; p[-1]; ++p) asciiname[p - path] = *p;
#else
- const char *realname = f;
+ const char *asciiname = f;
#endif
- struct Token *t = tokenize_buf(realname, ret.sbase);
- // everything is THING() or THING {} so we need at least 3 tokens ahead - if
- // we have fewer tokens left in the file we can bail
- if (t && t->next) while (t->next->next) {
- if (!t->at_bol) {
- t = t->next;
- continue;
+ ret.lexer = clex(ret.sbase, len, (u32 *)mem + (len >> 2), asciiname);
+ if (ret.lexer.err) die(2, ret.lexer.err);
+ // everything is THING() or THING {}, and file also ends in an EOL, so once
+ // there's less than 4 tokens left in the file, we can bail.
+ for (u32 i = 0, end = ret.lexer.ntoks - 4; i < end; ++i) {
+ if_hot (ret.lexer.toks[i] != CLEX_TOK_IDENT) continue;
+ // technically we don't have to validate *every* token, but doing so
+ // gives less confusing syntax errors.
+ struct clex_ident_validate_ret val = clex_ident_validate(
+ &ret.lexer, ret.sbase, i);
+ if (val.err) {
+ char buf[CLEX_IDENT_ERRSTR_MEMREQ(PATH_MAX)];
+ clex_ident_errstr(buf, ret.sbase, val.err, val.err_off, asciiname);
+ die(2, buf);
}
+ // everything we match has to be followed by either ( or {
+ if (ret.lexer.toks[i + 1] != CLEX_TOK_OP1) { ++i; continue; }
+ if (val.len > 24) continue; // longer than the longest string below
+ char name[24];
+ clex_ident(&ret.lexer, ret.sbase, i, name);
int type;
- if ((equal(t, "DEF_CVAR") || equal(t, "DEF_CVAR_MIN") ||
- equal(t, "DEF_CVAR_MAX") || equal(t, "DEF_CVAR_MINMAX") ||
- equal(t, "DEF_CVAR_UNREG") || equal(t, "DEF_CVAR_MIN_UNREG") ||
- equal(t, "DEF_CVAR_MAX_UNREG") ||
- equal(t, "DEF_CVAR_MINMAX_UNREG") ||
- equal(t, "DEF_FEAT_CVAR") || equal(t, "DEF_FEAT_CVAR_MIN") ||
- equal(t, "DEF_FEAT_CVAR_MAX") ||
- equal(t, "DEF_FEAT_CVAR_MINMAX")) && equal(t->next, "(")) {
- type = CMETA_ITEM_DEF_CVAR;
- }
- else if ((equal(t, "DEF_CCMD") || equal(t, "DEF_CCMD_HERE") ||
- equal(t, "DEF_CCMD_UNREG") || equal(t, "DEF_CCMD_HERE_UNREG") ||
- equal(t, "DEF_CCMD_PLUSMINUS") ||
- equal(t, "DEF_CCMD_PLUSMINUS_UNREG") ||
- equal(t, "DEF_FEAT_CCMD") || equal(t, "DEF_FEAT_CCMD_HERE") ||
- equal(t, "DEF_FEAT_CCMD_PLUSMINUS")) && equal(t->next, "(")) {
- type = CMETA_ITEM_DEF_CCMD;
- }
- else if ((equal(t, "DEF_EVENT") || equal(t, "DEF_PREDICATE")) &&
- equal(t->next, "(")) {
- type = CMETA_ITEM_DEF_EVENT;
- }
- else if (equal(t, "HANDLE_EVENT") && equal(t->next, "(")) {
- type = CMETA_ITEM_HANDLE_EVENT;
- }
- else if (equal(t, "FEATURE") && equal(t->next, "(")) {
- type = CMETA_ITEM_FEATURE;
- }
- else if ((equal(t, "REQUIRE") || equal(t, "REQUIRE_GAMEDATA") ||
- equal(t, "REQUIRE_GLOBAL") || equal(t, "REQUEST")) &&
- equal(t->next, "(")) {
- type = CMETA_ITEM_REQUIRE;
- }
- else if (equal(t, "GAMESPECIFIC") && equal(t->next, "(")) {
- type = CMETA_ITEM_GAMESPECIFIC;
- }
- else if (equal(t, "PREINIT") && equal(t->next, "{")) {
- type = CMETA_ITEM_PREINIT;
+ int flags = 0;
+ char nextop = '(';
+ // this is kind of dumb code. oh well, good enough probably.
+ switch (val.len) {
+ case 3:
+ if (!memcmp(name, "END", 3)) {
+ type = CMETA_ITEM_END;
+ nextop = '{';
+ break;
+ }
+ continue;
+ case 4:
+ if (!memcmp(name, "INIT", 4)) {
+ type = CMETA_ITEM_INIT;
+ nextop = '{';
+ break;
+ }
+ continue;
+ case 7:
+ if (!memcmp(name, "FEATURE", 7)) {
+ type = CMETA_ITEM_FEATURE;
+ break;
+ }
+ if (!memcmp(name, "PREINIT", 7)) {
+ type = CMETA_ITEM_PREINIT;
+ nextop = '{';
+ break;
+ }
+ if (!memcmp(name, "REQUIRE", 7)) {
+ type = CMETA_ITEM_REQUIRE;
+ break;
+ }
+ if (!memcmp(name, "REQUEST", 7)) {
+ type = CMETA_ITEM_REQUIRE;
+ flags = CMETA_REQUIRE_OPTIONAL;
+ break;
+ }
+ continue;
+ case 8:
+ if (!memcmp(name, "DEF_CCMD", 8)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ break;
+ }
+ if (!memcmp(name, "DEF_CVAR", 8)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ break;
+ }
+ continue;
+ case 9:
+ if (!memcmp(name, "DEF_EVENT", 9)) {
+ type = CMETA_ITEM_DEF_EVENT;
+ break;
+ }
+ continue;
+ case 12:
+ if (!memcmp(name, "DEF_CVAR_MAX", 12) ||
+ !memcmp(name, "DEF_CVAR_MIN", 12)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ break;
+ }
+ if (!memcmp(name, "GAMESPECIFIC", 12)) {
+ type = CMETA_ITEM_GAMESPECIFIC;
+ break;
+ }
+ if (!memcmp(name, "HANDLE_EVENT", 12)) {
+ type = CMETA_ITEM_HANDLE_EVENT;
+ break;
+ }
+ continue;
+ case 13:
+ if (!memcmp(name, "DEF_CCMD_HERE", 13)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ break;
+ }
+ if (!memcmp(name, "DEF_FEAT_CCMD", 13)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ if (!memcmp(name, "DEF_FEAT_CVAR", 13)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ if (!memcmp(name, "DEF_PREDICATE", 13)) {
+ type = CMETA_ITEM_DEF_EVENT;
+ flags = CMETA_EVENT_ISPREDICATE;
+ break;
+ }
+ continue;
+ case 14:
+ if (!memcmp(name, "DEF_CCMD_UNREG", 14)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_UNREG;
+ break;
+ }
+ if (!memcmp(name, "DEF_CVAR_UNREG", 14)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_UNREG;
+ break;
+ }
+ if (!memcmp(name, "REQUIRE_GLOBAL", 14)) {
+ type = CMETA_ITEM_REQUIRE;
+ flags = CMETA_REQUIRE_GLOBAL;
+ break;
+ }
+ continue;
+ case 15:
+ if (!memcmp(name, "DEF_CVAR_MINMAX", 15)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ break;
+ }
+ continue;
+ case 16:
+ if (!memcmp(name, "REQUIRE_GAMEDATA", 16)) {
+ type = CMETA_ITEM_REQUIRE;
+ flags = CMETA_REQUIRE_GAMEDATA;
+ break;
+ }
+ continue;
+ case 17:
+ if (!memcmp(name, "DEF_FEAT_CVAR_MAX", 17) ||
+ !memcmp(name, "DEF_FEAT_CVAR_MIN", 17)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ continue;
+ case 18:
+ if (!memcmp(name, "DEF_CCMD_PLUSMINUS", 18)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_PLUSMINUS;
+ break;
+ }
+ if (!memcmp(name, "DEF_CVAR_MAX_UNREG", 18) ||
+ !memcmp(name, "DEF_CVAR_MIN_UNREG", 18)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_UNREG;
+ break;
+ }
+ if (!memcmp(name, "DEF_FEAT_CCMD_HERE", 18)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_FEAT;
+ break;
+ }
+ continue;
+ case 19:
+ if (!memcmp(name, "DEF_CCMD_HERE_UNREG", 19)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_UNREG;
+ break;
+ }
+ continue;
+ case 20:
+ if (!memcmp(name, "DEF_FEAT_CVAR_MINMAX", 20)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_FEAT;
+ break;
+ }
+ continue;
+ case 21:
+ if (!memcmp(name, "DEF_CVAR_MINMAX_UNREG", 21)) {
+ type = CMETA_ITEM_DEF_CVAR;
+ flags = CMETA_CVAR_UNREG;
+ break;
+ }
+ continue;
+ case 23:
+ if (!memcmp(name, "DEF_FEAT_CCMD_PLUSMINUS", 23)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_FEAT | CMETA_CCMD_PLUSMINUS;
+ break;
+ }
+ continue;
+ case 24:
+ if (!memcmp(name, "DEF_CCMD_PLUSMINUS_UNREG", 24)) {
+ type = CMETA_ITEM_DEF_CCMD;
+ flags = CMETA_CCMD_UNREG | CMETA_CCMD_PLUSMINUS;
+ break;
+ }
+ default:
+ continue;
}
- else if (equal(t, "INIT") && equal(t->next, "{")) {
- type = CMETA_ITEM_INIT;
- }
- else if (equal(t, "END") && equal(t->next, "{")) {
- type = CMETA_ITEM_END;
- }
- else {
- t = t->next;
+ if (ret.sbase[ret.lexer.tokoffs[i + 1]] != nextop) {
+ // bump i a little further as we've already looked at at least 2
+ // tokens. this is technically kind of inefficient; in most cases we
+ // can skip more stuff, but we're always scanning for something
+ // specific, so who cares actually, this is good enough.
+ ++i;
continue;
}
- ret.itemtoks[ret.nitems] = t;
- ret.itemtypes[ret.nitems] = type;
+ ret.itemtoks[ret.nitems] = i;
+ ret.items[ret.nitems] = (struct cmeta_item){type, flags};
++ret.nitems;
- // this is kind of inefficient; in most cases we can skip more stuff,
- // but then also, we're always scanning for something specific, so who
- // cares actually, this will do for now.
- t = t->next->next;
+ ++i;
}
return ret;
}
-int cmeta_flags_cvar(const struct cmeta *cm, u32 i) {
- struct Token *t = cm->itemtoks[i];
- switch_exhaust (t->len) {
- // It JUST so happens all of the possible tokens here have a unique
- // length. I swear this wasn't planned. But it IS convenient!
- case 8: case 12: case 15: return 0;
- case 14: case 18: case 21: return CMETA_CVAR_UNREG;
- case 13: case 17: case 20: return CMETA_CVAR_FEAT;
- }
-}
-
-int cmeta_flags_ccmd(const struct cmeta *cm, u32 i) {
- struct Token *t = cm->itemtoks[i];
- switch_exhaust (t->len) {
- case 13: if (t->loc[4] == 'F') return CMETA_CCMD_FEAT;
- case 8: return 0;
- case 18: if (t->loc[4] == 'F') return CMETA_CCMD_FEAT;
- return CMETA_CCMD_PLUSMINUS;
- case 14: case 19: return CMETA_CCMD_UNREG;
- case 23: return CMETA_CCMD_FEAT | CMETA_CCMD_PLUSMINUS;
- case 24: return CMETA_CCMD_UNREG | CMETA_CCMD_PLUSMINUS;
- }
-}
-
-int cmeta_flags_event(const struct cmeta *cm, u32 i) {
- // assuming CMETA_EVENT_ISPREDICATE remains 1, the ternary should
- // optimise out
- return cm->itemtoks[i]->len == 13 ? CMETA_EVENT_ISPREDICATE : 0;
-}
-
-int cmeta_flags_require(const struct cmeta *cm, u32 i) {
- struct Token *t = cm->itemtoks[i];
- // NOTE: this is somewhat more flexible to enable REQUEST_GAMEDATA or
- // something in future, although that's kind of useless currently
- int optflag = t->loc[4] == 'E'; // REQU[E]ST
- switch_exhaust (t->len) {
- case 7: return optflag;
- case 16: return optflag | CMETA_REQUIRE_GAMEDATA;
- case 14: return optflag | CMETA_REQUIRE_GLOBAL;
- };
-}
-
-int cmeta_nparams(const struct cmeta *cm, u32 i) {
+int cmeta_nparams(const struct cmeta *cm, u32 item) {
int argc = 1, nest = 0;
- struct Token *t = cm->itemtoks[i]->next->next;
- if (equal(t, ")")) return 0; // XXX: stupid special case, surely improvable?
- for (; t; t = t->next) {
- if (equal(t, "(")) { ++nest; continue; }
- if (!nest && equal(t, ",")) ++argc;
- else if (equal(t, ")") && !nest--) break;
+ int i = cm->itemtoks[item] + 2; // get past the first (
+ // handle immediate ) - XXX: stupid special case, surely improvable?
+ if (clex_isrparen(&cm->lexer, cm->sbase, i)) return 0;
+ for (; i < cm->lexer.ntoks; ++i) {
+ if (clex_islparen(&cm->lexer, cm->sbase, i)) { ++nest; continue; }
+ if (!nest && clex_iscomma(&cm->lexer, cm->sbase, i)) ++argc;
+ else if (clex_isrparen(&cm->lexer, cm->sbase, i) && !nest--) break;
}
if (nest != -1) return 0; // XXX: any need to do anything better here?
return argc;
}
struct cmeta_param_iter cmeta_param_iter_init(const struct cmeta *cm, u32 i) {
- return (struct cmeta_param_iter){cm->itemtoks[i]->next->next};
+ return (struct cmeta_param_iter){cm->itemtoks[i] + 2};
}
-struct cmeta_slice cmeta_param_iter(struct cmeta_param_iter *it) {
+struct cmeta_slice cmeta_param_iter(const struct cmeta *cm,
+ struct cmeta_param_iter *it) {
int nest = 0;
- const char *start = it->cur->loc;
- for (struct Token *last = 0; it->cur;
- last = it->cur, it->cur = it->cur->next) {
- if (equal(it->cur, "(")) { ++nest; continue; }
- if (!nest && equal(it->cur, ",")) {
- if (!last) { // , immediately after (, for some reason. treat as ""
- return (struct cmeta_slice){start, 0};
- }
- it->cur = it->cur->next;
+ const char *start = cm->sbase + cm->lexer.tokoffs[it->i];
+ for (; it->i < cm->lexer.ntoks; ++it->i) {
+ if (clex_islparen(&cm->lexer, cm->sbase, it->i)) {
+ ++nest;
+ continue;
}
- else if (equal(it->cur, ")") && !nest--) {
- if (!last) break;
+ if (!nest && clex_iscomma(&cm->lexer, cm->sbase, it->i)) {
+ const char *end = cm->sbase + cm->lexer.tokoffs[it->i];
+ // XXX: to avoid picking up random whitespace we should get the
+ // previous token and ask clex for its extent, but I've not yet
+ // implemented full parsing of all variable-length tokens (i.e.
+ // numbers and string/char literals). not worrying about it too much
+ // at the moment since it's not really a problem anyway in practice
+ ++it->i; // skip past comma for next time
+ return (struct cmeta_slice){start, end - start};
}
- else {
- continue;
+ else if (clex_isrparen(&cm->lexer, cm->sbase, it->i) && !nest--) {
+ // annoying case: 0 args is different from ", )" (empty arg)
+ if (clex_islparen(&cm->lexer, cm->sbase, it->i - 1)) break;
+ const char *end = cm->sbase + cm->lexer.tokoffs[it->i];
+ it->i = -1u; // force next call to return {0, 0}
+ return (struct cmeta_slice){start, end - start};
}
- return (struct cmeta_slice){start, last->loc - start + last->len};
}
return (struct cmeta_slice){0, 0};
}
-u32 cmeta_line(const struct cmeta *cm, u32 i) {
- return cm->itemtoks[i]->line_no;
+cold u32 cmeta_line(const struct cmeta *cm, u32 i) {
+ u32 line = 1;
+ for (u32 off = 0, end = cm->lexer.tokoffs[cm->itemtoks[i]];
+ off != end; ++off) {
+ line += cm->sbase[off] == '\n';
+ }
+ return line;
}
// vi: sw=4 ts=4 noet tw=80 cc=80 fdm=marker