summaryrefslogtreecommitdiff
path: root/src/chunklets/clex.h
diff options
context:
space:
mode:
Diffstat (limited to 'src/chunklets/clex.h')
-rw-r--r--src/chunklets/clex.h731
1 files changed, 731 insertions, 0 deletions
diff --git a/src/chunklets/clex.h b/src/chunklets/clex.h
new file mode 100644
index 0000000..5d3a636
--- /dev/null
+++ b/src/chunklets/clex.h
@@ -0,0 +1,731 @@
+/*
+ * Copyright © Michael Smith <mikesmiffy128@gmail.com>
+ *
+ * Permission to use, copy, modify, and/or distribute this software for any
+ * purpose with or without fee is hereby granted, provided that the above
+ * copyright notice and this permission notice appear in all copies.
+ *
+ * THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
+ * REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
+ * AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
+ * INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
+ * LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
+ * OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
+ * PERFORMANCE OF THIS SOFTWARE.
+ */
+
+/* WARNING: this library is incomplete and still subject to change! */
+
+#ifndef INC_CHUNKLETS_CLEX_H
+#define INC_CHUNKLETS_CLEX_H
+
+#ifdef __cplusplus
+#define _clex_bool bool
+#define _clex_restrict __restrict // XXX: is this portable enough?
+#else
+#define _clex_bool _Bool
+#define _clex_restrict restrict
+#endif
+
+#ifdef _WIN32
+#ifdef __cplusplus
+typedef wchar_t clex_os_char; // ugh
+#else
+typedef unsigned short clex_os_char;
+#endif
+#else
+typedef char clex_os_char;
+#endif
+
+#ifdef _WIN64
+typedef long long clex_size;
+#else
+typedef long clex_size;
+#endif
+
+#if defined(__GNUC__) || defined(__clang__)
+#define _clex_unreachable __builtin_unreachable()
+#elif defined(_MSC_VER)
+#define _clex_unreachable __assume(0)
+#else
+#ifdef __cplusplus
+[[noreturn]] static inline void _clex_invoke_ub(void) {}
+#else
+static inline _Noreturn void _clex_invoke_ub(void) {}
+#endif
+#define _clex_unreachable (_clex_invoke_ub())
+#endif
+
+/*
+ * Determines the amount of memory required to call clex() on a given file.
+ * That memory must be provided as a contiguous block, with at least 4 byte
+ * alignment.
+ *
+ * filesz is the number of bytes in the file, and namelen is the length of the
+ * name. The reason namelen is required is because the memory is reused to
+ * produce an error string when lexing fails.
+ *
+ * filesz is assumed to be non-zero, since empty C source files are not allowed,
+ * and common memory allocators do not accept zero sizes either. To gracefully
+ * handle an empty file if desired, check for a zero size and avoid attempting
+ * tokenisation altogether.
+ *
+ * If a null filename will be passed to clex(), which skips error message
+ * formatting, namelen can be set to 0.
+ *
+ * NOTE: This assumes that the file is reasonably-sized. On 64-bit architectures,
+ * the file size must be not exceed 4GiB *minus 2 bytes*, due to an internal
+ * implementation detail. On 32-bit architectures, the file has to be even
+ * smaller in order to fit the token data in memory; it is recommended to stay
+ * below ~256MiB.
+ */
+static inline clex_size clex_memreq(unsigned int filesz, unsigned int namelen) {
+ clex_size sz = ((clex_size)filesz + 2) * 5 + 5;
+ if (sz < namelen + 64) return namelen + 64; // make space for error messages
+ return sz;
+}
+
+/* The main structure returned by a call to clex() - see below. */
+struct clex {
+ /*
+ * Check this first. It is a null pointer if lexing succeeded, or points to
+ * a null-terminated error string on failure. See also clex() for a
+ * description of how the error string is created.
+ */
+ const char *err;
+ union {
+ /*
+ * If err is not null, contains the length of the string, excluding the
+ * null terminator.
+ */
+ int errlen;
+ /*
+ * If lexing was successful (err is null), contains the number of tokens
+ * lexed. Use this to loop over toks and tokoffs.
+ */
+ unsigned int ntoks;
+ };
+ //char pad[4];
+ /*
+ * If lexing was successful (err is null), points to an array of token
+ * offsets within the lexed file. On failure, the value is undefined.
+ */
+ unsigned int *tokoffs;
+ /*
+ * If lexing was successful (err is null), points to an array of basic token
+ * types corresponding to each position in tokoffs. On failure, the value is
+ * undefined.
+ */
+ unsigned char *toks;
+};
+
+/*
+ * Lexes a C file into a list of tokens. See above comments on struct clex for
+ * return type information.
+ *
+ * buf and sz point to the contents of the file, which has to be fully loaded
+ * into memory prior to lexing.
+ *
+ * outmem is an opaque block of memory of at sufficient size calculated by
+ * clex_memreq(), and at least 4-byte alignment. It can be allocated by any
+ * means desired.
+ *
+ * filename is used purely for error reporting, though typically it would be the
+ * name of the file being lexed, of course. Normally the error string output
+ * will include the filename, row and column, and a brief description of the
+ * error. If this information is not needed, filename can be a null pointer
+ * instead, which disables this functionality.
+ */
+struct clex clex(const char *_clex_restrict buf, unsigned int sz,
+ void *_clex_restrict outmem, const char *_clex_restrict filename);
+
+/*
+ * These are the basic token types recognised by the lexer in its main pass.
+ * All the operators are grouped in such a way that they can be distinguished by
+ * looking at a single character.
+ *
+ * Identifiers and literals might not necessarily be totally valid; this is
+ * checked in more detail when parsing out specific values.
+ */
+enum clex_tok_type {
+ CLEX_TOK_IDENT, /* An identifier or keyword. */
+ CLEX_TOK_NUM, /* A numeric literal, which may or may not be totally valid */
+ CLEX_TOK_OP1, /* One of: + - * / = < > ! & | ^ ~ . , ( ) [ ] { } : ; ? # */
+ CLEX_TOK_OP2, /* One of: ++ -- -> && || ## :: */
+ CLEX_TOK_SHIFT, /* One of: >> << */
+ CLEX_TOK_OPEQ, /* One of: += -= *= /= &= |= ^= <= >= != == */
+ CLEX_TOK_SHIFTEQ, /* One of: >>= <<= */
+ CLEX_TOK_CHAR, /* A character literal. */
+ CLEX_TOK_STR, /* A string literal. */
+ CLEX_TOK_LNCOMM, /* A C++/C99-style one-line comment. */
+ CLEX_TOK_BLKCOMM, /* A classic C-style multi-line comment. */
+ CLEX_TOK_EOL /* A run of end-of-line characters, for parsing #directives. */
+};
+
+/* These are all the C operators, hopefully self-explanatory. */
+enum clex_op {
+ // -- op1 --
+ CLEX_OP_PLUS, /* + */
+ CLEX_OP_MINUS, /* - */
+ CLEX_OP_MULT, /* * (could also be a deref) */ // XXX: bad name maybe?
+ CLEX_OP_DIV, /* / */
+ CLEX_OP_ASSIGN, /* = */
+ CLEX_OP_LT, /* < */
+ CLEX_OP_GT, /* > */
+ CLEX_OP_NOT, /* ! */
+ CLEX_OP_BITAND, /* & (could also be an address-of) */ // XXX: bad name?
+ CLEX_OP_BITOR, /* | */
+ CLEX_OP_XOR, /* ^ */
+ CLEX_OP_BITNOT, /* ~ */
+ CLEX_OP_DOT, /* . */
+ CLEX_OP_COMMA, /* , */
+ CLEX_OP_LPAREN, /* ( */
+ CLEX_OP_RPAREN, /* ) */
+ CLEX_OP_LSQ, /* [ */
+ CLEX_OP_RSQ, /* ] */
+ CLEX_OP_LCURL, /* { */
+ CLEX_OP_RCURL, /* } */
+ CLEX_OP_COLON, /* : */
+ CLEX_OP_SEMICOL, /* ; */
+ CLEX_OP_QUESTION, /* ? */
+ CLEX_OP_HASH, /* # */
+ // -- op2 --
+ CLEX_OP_INC, /* ++ */
+ CLEX_OP_DEC, /* -- */
+ CLEX_OP_ARROW, /* -> */
+ CLEX_OP_AND, /* && */
+ CLEX_OP_OR, /* || */
+ CLEX_OP_PASTE, /* ## */
+ CLEX_OP_DCOLON, /* :: C23 attribute vendors/namespaces */
+ // -- shift --
+ CLEX_OP_LSH, /* << */
+ CLEX_OP_RSH, /* >> */
+ // -- opeq --
+ CLEX_OP_PLUSEQ, /* += */
+ CLEX_OP_MINUSEQ, /* -= */
+ CLEX_OP_MULTEQ, /* *= */
+ CLEX_OP_DIVEQ, /* /= */
+ CLEX_OP_ANDEQ, /* &= */
+ CLEX_OP_OREQ, /* |= */
+ CLEX_OP_XOREQ, /* ^= */
+ CLEX_OP_LTEQ, /* <= */
+ CLEX_OP_GTEQ, /* >= */
+ CLEX_OP_NOTEQ, /* != */
+ CLEX_OP_EQ, /* == */
+ // -- shifteq --
+ CLEX_OP_LSHEQ, /* <<= */
+ CLEX_OP_RSHEQ /* >>= */
+};
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OP1 at offset off
+ * by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_OP1, the behaviour is undefined.
+ */
+static inline enum clex_op clex_op1(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '+': return CLEX_OP_PLUS;
+ case '-': return CLEX_OP_MINUS;
+ case '*': return CLEX_OP_MULT;
+ case '/': return CLEX_OP_DIV;
+ case '=': return CLEX_OP_ASSIGN;
+ case '<': return CLEX_OP_LT;
+ case '>': return CLEX_OP_GT;
+ case '!': return CLEX_OP_NOT;
+ case '&': return CLEX_OP_BITAND;
+ case '|': return CLEX_OP_BITOR;
+ case '^': return CLEX_OP_XOR;
+ case '~': return CLEX_OP_BITNOT;
+ case '.': return CLEX_OP_DOT;
+ case ',': return CLEX_OP_COMMA;
+ case '(': return CLEX_OP_LPAREN;
+ case ')': return CLEX_OP_RPAREN;
+ case '[': return CLEX_OP_LSQ;
+ case ']': return CLEX_OP_RSQ;
+ case '{': return CLEX_OP_LCURL;
+ case '}': return CLEX_OP_RCURL;
+ case ':': return CLEX_OP_COLON;
+ case ';': return CLEX_OP_SEMICOL;
+ case '?': return CLEX_OP_QUESTION;
+ case '#': return CLEX_OP_HASH;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OP2 at offset off
+ * by inspecting the end of the token.
+ *
+ * If the token is not of type CLEX_TOK_OP2, the behaviour is undefined.
+ */
+static inline enum clex_op clex_op2(const char *buf, unsigned int off) {
+ for (;;) {
+ switch (buf[off + 1]) {
+ case '+': return CLEX_OP_INC;
+ case '-': return CLEX_OP_DEC;
+ case '>': return CLEX_OP_ARROW;
+ case '&': return CLEX_OP_AND;
+ case '|': return CLEX_OP_OR;
+ case '#': return CLEX_OP_PASTE;
+ case ':': return CLEX_OP_DCOLON;
+ // initial lex has validated backslash-newlines, so we can just
+ // ignore any of these characters here
+ case '\\': case ' ': case '\t': case '\r': case '\n': continue;
+ }
+ _clex_unreachable;
+ }
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_SHIFT at offset
+ * off by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_SHIFT, the behaviour is undefined.
+ */
+static inline enum clex_op clex_shift(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '>': return CLEX_OP_RSH;
+ case '<': return CLEX_OP_LSH;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OPEQ at offset
+ * off by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_OPEQ, the behaviour is undefined.
+ */
+static inline enum clex_op clex_opeq(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '+': return CLEX_OP_PLUSEQ;
+ case '-': return CLEX_OP_MINUSEQ;
+ case '*': return CLEX_OP_MULTEQ;
+ case '/': return CLEX_OP_DIVEQ;
+ case '&': return CLEX_OP_ANDEQ;
+ case '|': return CLEX_OP_OREQ;
+ case '^': return CLEX_OP_XOREQ;
+ case '<': return CLEX_OP_LTEQ;
+ case '>': return CLEX_OP_GTEQ;
+ case '!': return CLEX_OP_NOTEQ;
+ case '=': return CLEX_OP_EQ;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_SHIFTEQ at offset
+ * off by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_SHIFTEQ, the behaviour is undefined.
+ */
+static inline enum clex_op clex_shifteq(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '<': return CLEX_OP_RSHEQ;
+ case '>': return CLEX_OP_LSHEQ;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines whether the lexed token at a given index is an operator (not
+ * counting the likes of sizeof, which will have to be handled as a special case
+ * of an identifier/keyword for syntax purposes).
+ */
+static inline _clex_bool clex_isop(const struct clex *c, unsigned int idx) {
+ switch (c->toks[idx]) {
+ case CLEX_TOK_IDENT: return 0;
+ case CLEX_TOK_NUM: return 0;
+ case CLEX_TOK_OP1: return 1;
+ case CLEX_TOK_OP2: return 1;
+ case CLEX_TOK_SHIFT: return 1;
+ case CLEX_TOK_OPEQ: return 1;
+ case CLEX_TOK_SHIFTEQ: return 1;
+ case CLEX_TOK_CHAR: return 0;
+ case CLEX_TOK_STR: return 0;
+ case CLEX_TOK_LNCOMM: return 0;
+ case CLEX_TOK_BLKCOMM: return 0;
+ case CLEX_TOK_EOL: return 0;
+ }
+ _clex_unreachable;
+}
+
+/* Returns true if a token is either a unary or binary plus operator. */
+static inline _clex_bool clex_isplus(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '+';
+}
+
+/* Returns true if a token is either a unary or binary minus operator. */
+static inline _clex_bool clex_isminus(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '-';
+}
+
+/* Returns true if a token is either a multiply or deference operator. */
+static inline _clex_bool clex_ismult(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '*';
+}
+
+/* Returns true a token is a division operator. */
+static inline _clex_bool clex_isdiv(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '/';
+}
+
+/* Returns true if a token is an assignment operator. */
+static inline _clex_bool clex_isassign(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '=';
+}
+
+/* Returns true if a token is a less-than operator. */
+static inline _clex_bool clex_islt(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a greater-than operator. */
+static inline _clex_bool clex_isgt(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '>';
+}
+
+/* Returns true if a token is a negation operator. */
+static inline _clex_bool clex_isnot(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '!';
+}
+
+/* Returns true if a token is either a bitwise and operator or an indirection. */
+static inline _clex_bool clex_isbitand(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '&';
+}
+
+/* Returns true if a token is a bitwise or operator. */
+static inline _clex_bool clex_isbitor(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '|';
+}
+
+/* Returns true if a token is a bitwise exclusive-or operator. */
+static inline _clex_bool clex_isxor(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '^';
+}
+
+/* Returns true if a token is a bitwise negation operator. */
+static inline _clex_bool clex_isbitnot(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '~';
+}
+
+/* Returns true if a token is a struct member dot. */
+static inline _clex_bool clex_isdot(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '.';
+}
+
+/* Returns true if a token is a comma. */
+static inline _clex_bool clex_iscomma(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ',';
+}
+
+/* Returns true if a token is an opening/left parenthesis. */
+static inline _clex_bool clex_islparen(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '(';
+}
+
+/* Returns true if a token is a closing/right parenthesis. */
+static inline _clex_bool clex_isrparen(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ')';
+}
+
+/* Returns true if a token is an opening/left square bracket. */
+static inline _clex_bool clex_islsq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '[';
+}
+
+/* Returns true if a token is a closing/right square bracket. */
+static inline _clex_bool clex_isrsq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ']';
+}
+
+/* Returns true if a token is an opening/left curly brace. */
+static inline _clex_bool clex_islcurl(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '{';
+}
+
+/* Returns true if a token is a closing/right curly brace. */
+static inline _clex_bool clex_isrcurl(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '}';
+}
+
+/* Returns true if a token is a colon. */
+static inline _clex_bool clex_iscolon(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ':';
+}
+
+/* Returns true if a token is a semicolon. */
+static inline _clex_bool clex_issemicol(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ';';
+}
+
+/* Returns true if a token is a question mark. */
+static inline _clex_bool clex_isquestion(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '?';
+}
+
+/* Returns true if a token is a preprocessor hash. */
+static inline _clex_bool clex_ishash(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '#';
+}
+
+/* Returns true if a token is a unary increment operator. */
+static inline _clex_bool clex_isinc(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '+';
+}
+
+/* Returns true if a token is a unary decrement operator. */
+static inline _clex_bool clex_isdec(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '-';
+}
+
+/* Returns true if a token is an arrow for indirect struct member access. */
+static inline _clex_bool clex_isarrow(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '>';
+}
+
+/* Returns true if a token is a conditional and operator. */
+static inline _clex_bool clex_isand(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '&';
+}
+
+/* Returns true if a token is a conditional or operator. */
+static inline _clex_bool clex_isor(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '|';
+}
+
+/* Returns true if a token is a token-pasting operator. */
+static inline _clex_bool clex_ispaste(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '#';
+}
+
+/* Returns true if a token is a double colon for C23 attribute namespacing. */
+static inline _clex_bool clex_isdcol(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == ':';
+}
+
+/* Returns true if a token is a left shift operator. */
+static inline _clex_bool clex_islsh(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFT && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a right shift operator. */
+static inline _clex_bool clex_isrsh(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFT && buf[c->tokoffs[idx]] == '>';
+}
+
+/* Returns true if a token is a binary increment (plus-assignment) operator. */
+static inline _clex_bool clex_ispluseq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '+';
+}
+
+/* Returns true if a token is a binary decrement (plus-assignment) operator. */
+static inline _clex_bool clex_isminuseq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '-';
+}
+
+/* Returns true if a token is a multiply-assignment operator. */
+static inline _clex_bool clex_ismulteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '*';
+}
+
+/* Returns true if a token is a divide-assignment operator. */
+static inline _clex_bool clex_isdiveq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '/';
+}
+
+/* Returns true if a token is a bitwise-and-assignment operator. */
+static inline _clex_bool clex_isandeq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '&';
+}
+
+/* Returns true if a token is a bitwise-or-assignment operator. */
+static inline _clex_bool clex_isoreq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '|';
+}
+
+/* Returns true if a token is a bitwise-xor-assignment operator. */
+static inline _clex_bool clex_isxoreq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '^';
+}
+
+/* Returns true if a token is a less-than-or-equal comparison operator. */
+static inline _clex_bool clex_islteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a greater-than-or-equal comparison operator. */
+static inline _clex_bool clex_isgteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '>';
+}
+
+/* Returns true if a token is an inequality comparison operator. */
+static inline _clex_bool clex_isnoteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '!';
+}
+
+/* Returns true if a token is an equality comparison operator. */
+static inline _clex_bool clex_iseq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '=';
+}
+
+/* Returns true if a token is a left-shift-assignment operator. */
+static inline _clex_bool clex_islshifteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFTEQ && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a right-shift-assignment operator. */
+static inline _clex_bool clex_isrshifteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFTEQ && buf[c->tokoffs[idx]] == '>';
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OP1,
+ * CLEX_TOK_OP2, CLEX_TOK_SHIFT, CLEX_TOK_OPEQ, or CLEX_TOK_SHIFTEQ.
+ *
+ * If the token is not of one of the above types, the behaviour is undefined.
+ * Call clex_isop() first to determine whether a token is of a valid type.
+ */
+static inline enum clex_op clex_op(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ unsigned int off = c->tokoffs[idx];
+ switch (c->toks[idx]) {
+ case CLEX_TOK_OP1: return clex_op1(buf, off);
+ case CLEX_TOK_OP2: return clex_op2(buf, off);
+ case CLEX_TOK_SHIFT: return clex_shift(buf, off);
+ case CLEX_TOK_OPEQ: return clex_opeq(buf, off);
+ case CLEX_TOK_SHIFTEQ: return clex_shifteq(buf, off);
+ }
+ _clex_unreachable;
+}
+
+enum clex_ident_validate_err {
+ // NOTE: do not reorder these values! clex_ident_errstr() relies on them
+ CLEX_IDENT_OK,
+ CLEX_IDENT_BADUCNHEX4, /* invalid \uXXXX UCN sequence */
+ CLEX_IDENT_BADCODEPOINT, /* invalid Unicode codepoint in UCN */
+ CLEX_IDENT_BADUCNHEX6, /* invalid \UXXXXXX UCN sequence */
+ CLEX_IDENT_UNEXPBACKSLASH /* unexpected backslash in identifier */
+};
+
+/*
+ * Double-checks that a TOK_IDENT token is syntactically valid (as the initial
+ * lexer pass does not account for bad UCN sequences or misplaced backslashes).
+ *
+ * Determines the extent of the token, i.e. how many bytes it occupies in the
+ * source, along with the length of the evaluated name, i.e. how many bytes are
+ * required to store the evaluated name as UTF-8 once line-continuations and
+ * UCNs are taken into account.
+ *
+ * This function may only be called on tokens of type TOK_IDENT; anything else
+ * produces undefined behaviour.
+ */
+struct clex_ident_validate_ret {
+ enum clex_ident_validate_err err; /* zero on success, nonzero on error */
+ // wonky union interweaving here because C++ doesn't do anonymous structs :(
+ union {
+ int ext; /* if !err: extent (source length) */
+ unsigned int err_off; /* if err: position of error in file */
+ };
+ int len; /* if !err: length of evaluated name (as utf-8). else undefined! */
+} clex_ident_validate(const struct clex *c, const char *_clex_restrict buf,
+ unsigned int idx);
+
+/*
+ * Evaluates the name of an identifier, taking into account line continuations
+ * and UCNs. The identifier has to have first been successfully validated using
+ * clex_ident_validate(). The buffer out must be large enough to hold the result
+ * which is indicated by the len member of the clex_ident_validate_ret struct.
+ */
+void clex_ident(const struct clex *c, const char *_clex_restrict buf,
+ unsigned int idx, char *_clex_restrict out);
+
+// NOTE: this is precalculated based on the implementation of clex_ident_errstr
+// and will need changed if the function changes!
+/*
+ * Determines the buffer size required to format an error message using
+ * clex_ident_errstr(), given the length of the filename.
+ *
+ * clex_ident_errstr() formats a string in the form "filename:line:col: message"
+ * so this macro can determine the worst-case memory requirement in advance.
+ *
+ * If a constant value is desired, e.g. for a stack buffer, simply use the known
+ * longest possible file path, for instance PATH_MAX on Unix-likes or MAX_PATH
+ * on Windows. The reason this is a macro is to allow it to produce an integer
+ * constant expression, so that a stack buffer can be used without creating a
+ * VLA.
+ */
+#define CLEX_IDENT_ERRSTR_MEMREQ(namelen) ((namelen) + 24 + 33)
+
+/*
+ * Formats an error message in the form "filename:line:col: message", given an
+ * error result from clex_ident_validate(). The resulting string is written to
+ * the buffer pointed to by out.
+ *
+ * out must point to a buffer of at least CLEX_IDENT_ERRSTR_MEMREQ(
+ * strlen(filename)) bytes. The resulting string is null-terminated for
+ * convenience, and the length of the string excluding the null terminator is
+ * returned.
+ *
+ * err has to be an error value from the clex_ident_validate_err enum; calling
+ * this function with a success result will produce undefined behaviour.
+ */
+int clex_ident_errstr(char *_clex_restrict out, const char *_clex_restrict buf,
+ enum clex_ident_validate_err err, unsigned int err_off,
+ const char *_clex_restrict filename);
+
+#undef _clex_unreachable
+#undef _clex_restrict
+#undef _clex_bool
+
+#endif
+
+// vi: sw=4 ts=4 noet tw=80 cc=80