summaryrefslogtreecommitdiff
path: root/src/chunklets
diff options
context:
space:
mode:
Diffstat (limited to 'src/chunklets')
-rw-r--r--src/chunklets/clex.c1664
-rw-r--r--src/chunklets/clex.h731
2 files changed, 2395 insertions, 0 deletions
diff --git a/src/chunklets/clex.c b/src/chunklets/clex.c
new file mode 100644
index 0000000..d20986c
--- /dev/null
+++ b/src/chunklets/clex.c
@@ -0,0 +1,1664 @@
+/*
+ * Copyright © Michael Smith <mikesmiffy128@gmail.com>
+ *
+ * Permission to use, copy, modify, and/or distribute this software for any
+ * purpose with or without fee is hereby granted, provided that the above
+ * copyright notice and this permission notice appear in all copies.
+ *
+ * THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
+ * REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
+ * AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
+ * INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
+ * LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
+ * OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
+ * PERFORMANCE OF THIS SOFTWARE.
+ */
+
+/* WARNING: this library is incomplete and still subject to change! */
+
+// NYI:
+// - numeric literal parsing
+// - string/char literal parsing
+// - more performance measurements
+// - if needed: more optimisation (target: >500MB/s give or take)
+// - security/robustness tests (e.g. fuzzing)
+// - UTF-8 aware column numbers (assuming this is what most tools do/expect?)
+
+#ifdef __cplusplus
+#error This file should not be compiled as C++. It relies on numerous C-only \
+features.
+#endif
+
+#if defined(_MSC_VER) && !defined(__clang__)
+#if !defined(_MSVC_TRADITIONAL)
+#error This version of MSVC is too old: upgrade to 2019 or newer and use \
+`-std:c17`, or better yet use Clang.
+#elif _MSVC_TRADITIONAL
+#error This file must be compiled using `-std:c17` when using MSVC.
+#endif
+#endif
+
+#ifndef _WIN32
+_Static_assert(
+ (unsigned char)-1 == 255 &&
+ sizeof(short) == 2 &&
+ sizeof(int) == 4 &&
+ sizeof(long long) == 8 &&
+ sizeof(void *) == 4 || sizeof(void *) == 8 &&
+ sizeof(long) == sizeof(void *),
+ "this code is only designed for relatively sane environments, plus Windows"
+);
+#endif
+
+#ifndef __clang__ // see `#define copy` further down
+#include <string.h>
+#endif
+
+#include "clex.h"
+
+#if defined(__GNUC__) || defined(__clang__)
+#define if_cold(x) if (__builtin_expect(!!(x), 0))
+#else
+#define if_cold(x) if (x)
+#endif
+
+#if defined(__GNUC__) || defined(__clang__)
+#define cold __attribute__((cold, noinline))
+#elif defined(_MSC_VER)
+#define cold __declspec(noinline)
+#else
+#define cold
+#endif
+
+// duping this from clex.h because it's undef'd there to pollute a little less
+#if defined(__GNUC__) || defined(__clang__)
+#define unreachable __builtin_unreachable()
+#elif defined(_MSC_VER)
+#define unreachable __assume(0)
+#else
+static inline _Noreturn void _clex_invoke_ub(void) {}
+#define unreachable (_clex_invoke_ub())
+#endif
+
+// note: a couple of these have been moved from seemingly more neat/logical
+// positions in order to make statetoks below more compact. classes and state
+// transitions use designated initialisers so this stuff can be moved around
+// without ruining the state transitions. the visual grouping here shows how
+// statetoks[] will be indexed, using halved state values.
+#define STATES(S, s) \
+ S(BOL) \
+ s(OP1) s(WS) \
+ s(OP2) s(WS_BS) \
+ S(LITPFX) \
+ S(uPFX) \
+ S(WORD) \
+ S(NUM) \
+ S(MIGHTEQ) \
+ S(MINUS) \
+ S(PLUS) \
+ S(AND) \
+ S(OR) \
+ S(HASH) \
+ S(COLON) \
+ S(FS) \
+ S(LT) \
+ S(GT) \
+ S(DOT) \
+ S(SHIFT) \
+ S(LNCOMM) \
+ s(BLKCOMM) s(BLKSTAR) \
+ s(OPEQ) s(BLKSTAR_BS) \
+ s(SHIFTEQ) s(STRLIT_BS_WS) /* < STUPID case, needed for gcc/clang compat! */ \
+ s(STRLIT) s(STRLIT_BS) \
+ s(CHRLIT) s(CHRLIT_BS) \
+ s(ERROR) /* special case for bad syntax */ \
+ s(CHRLIT_BS_WS) /* also stupid case */
+
+#define ENUMSTATE(name) STATE_##name,
+#define ENUMSTATEBS(name) STATE_##name, STATE_##name##_BS,
+#define ENUMTRANS(name) STATE_TRANS_##name = STATE_##name << 1,
+#define ENUMTRANSBS(name) \
+ ENUMTRANS(name) STATE_TRANS_##name##_BS = STATE_##name##_BS << 1,
+enum {
+ STATES(ENUMSTATEBS, ENUMSTATE)
+ NSTATES,
+ STATES(ENUMTRANSBS, ENUMTRANS)
+};
+#undef SHIFTENUMBS
+#undef SHIFTENUM
+#undef ENUMSTATEBS
+#undef ENUMSTATE
+
+#define CLASSES(X) \
+ X(OTHER) /* includes (most) alpha, underscores, anything unicode */ \
+ X(LU) /* L and U (lit prefixes) */ \
+ X(u) /* u prefix (could become u8) */ \
+ X(DIGIT) /* excluding 8 */ \
+ X(8) \
+ /* ^^ those must be before symbols */ \
+ X(OP1) /* 1 char ops/symbols */ \
+ X(MIGHTEQ) /* chars that could be followed by = */ \
+ X(MINUS) \
+ X(PLUS) \
+ X(EQ) \
+ X(LT) \
+ X(GT) \
+ X(AND) \
+ X(OR) \
+ X(DOT) \
+ X(STAR) \
+ X(HASH) \
+ X(COLON) \
+ X(FS) \
+ X(BS) \
+ X(SQ) \
+ X(DQ) \
+ X(SP) \
+ X(NL)
+
+// premultiply class values for transition table.
+#define CLASSBASEVALUE(name) CLASSBASEVAL_##name,
+enum { CLASSES(CLASSBASEVALUE) NCLASSES };
+#undef CLASSBASEVALUE
+#define CLASSREALVALUE(name) CLASS_##name = CLASSBASEVAL_##name * NSTATES,
+enum { CLASSES(CLASSREALVALUE) };
+#undef CLASSREALVALUE
+
+#define CLS(cls, c) [c] = CLASS_##cls,
+#define CLS2(cls, c1, c2) CLS(cls, c1) CLS(cls, c2)
+#define CLS3(cls, c1, ...) CLS(cls, c1) CLS2(cls, __VA_ARGS__)
+#define CLS4(cls, c1, ...) CLS(cls, c1) CLS3(cls, __VA_ARGS__)
+#define CLS5(cls, c1, ...) CLS(cls, c1) CLS4(cls, __VA_ARGS__)
+#define CLS6(cls, c1, ...) CLS(cls, c1) CLS5(cls, __VA_ARGS__)
+#define CLS7(cls, c1, ...) CLS(cls, c1) CLS6(cls, __VA_ARGS__)
+#define CLS8(cls, c1, ...) CLS(cls, c1) CLS7(cls, __VA_ARGS__)
+#define CLS9(cls, c1, ...) CLS(cls, c1) CLS8(cls, __VA_ARGS__)
+#define CLS10(cls, c1, ...) CLS(cls, c1) CLS9(cls, __VA_ARGS__)
+
+static const short classes[256] = {
+ /*CLSX(OTHER, everything, else)*/
+ CLS2(LU, 'L', 'U')
+ CLS(u, 'u')
+ CLS9(DIGIT, '0', '1', '2', '3', '4', '5', '6', '7', '9')
+ CLS(8, '8')
+ CLS10(OP1, '(', ')', '[', ']', '{', '}', ',', '?', ';', '~')
+ CLS3(MIGHTEQ, '!', '^', '%')
+ CLS(PLUS, '+')
+ CLS(MINUS, '-')
+ CLS(EQ, '=')
+ CLS(LT, '<')
+ CLS(GT, '>')
+ CLS(AND, '&')
+ CLS(OR, '|')
+ CLS(DOT, '.')
+ CLS(STAR, '*')
+ CLS(HASH, '#')
+ CLS(COLON, ':')
+ CLS(FS, '/')
+ CLS(BS, '\\')
+ CLS(SQ, '\'')
+ CLS(DQ, '"')
+ CLS3(SP, ' ', '\t', '\r')
+ CLS(NL, '\n')
+};
+
+#define FWD 1 // advance token index (i.e. create new token)
+#define FIXPOS 128 // in \u cases, token starts before the u. fixup file offset
+
+// abstracting transition table definition to allow fiddling with the layout.
+#define TRANS(from, class, to) \
+ [CLASS_##class + STATE_##from] = STATE_TRANS_##to,
+#define ST(s, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, s11, s12, \
+ s13, s14, s15, s16, s17, s18, s19, s20, s21, s22, s23, s24) \
+ TRANS(s, OTHER, s1) \
+ TRANS(s, LU, s2) \
+ TRANS(s, u, s3) \
+ TRANS(s, DIGIT, s4) \
+ TRANS(s, 8, s5) \
+ TRANS(s, OP1, s6) \
+ TRANS(s, MIGHTEQ, s7) \
+ TRANS(s, PLUS, s8) \
+ TRANS(s, MINUS, s9) \
+ TRANS(s, EQ, s10) \
+ TRANS(s, LT, s11) \
+ TRANS(s, GT, s12) \
+ TRANS(s, AND, s13) \
+ TRANS(s, OR, s14) \
+ TRANS(s, DOT, s15) \
+ TRANS(s, STAR, s16) \
+ TRANS(s, HASH, s17) \
+ TRANS(s, COLON, s18) \
+ TRANS(s, FS, s19) \
+ TRANS(s, BS, s20) \
+ TRANS(s, SQ, s21) \
+ TRANS(s, DQ, s22) \
+ TRANS(s, SP, s23) \
+ TRANS(s, NL, s24)
+
+#define ST_BS(s) \
+ ST(s##_BS, \
+ /* OTHER => */ ERROR | FWD, \
+ /* LU => */ WORD | FWD | FIXPOS, /* *could* be UCN (unlikely) */ \
+ /* u => */ WORD | FWD | FIXPOS, /* same */ \
+ /* DIGIT => */ ERROR | FWD, \
+ /* 8 => */ ERROR | FWD, \
+ /* OP1 => */ ERROR | FWD, \
+ /* MIGHTEQ => */ ERROR | FWD, \
+ /* PLUS => */ ERROR | FWD, \
+ /* MINUS => */ ERROR | FWD, \
+ /* EQ => */ ERROR | FWD, \
+ /* LT => */ ERROR | FWD, \
+ /* GT => */ ERROR | FWD, \
+ /* AND => */ ERROR | FWD, \
+ /* OR => */ ERROR | FWD, \
+ /* DOT => */ ERROR | FWD, \
+ /* STAR => */ ERROR | FWD, \
+ /* HASH => */ ERROR | FWD, \
+ /* COLON => */ ERROR | FWD, \
+ /* FS => */ ERROR | FWD, \
+ /* BS => */ ERROR | FWD, \
+ /* SQ => */ ERROR | FWD, \
+ /* DQ => */ ERROR | FWD, \
+ /* SP => */ s##_BS, /* nonstandard crap done by GCC and Clang */ \
+ /* NL => */ s \
+ )
+
+#define ST_TERM(s) /* "terminal" state (i.e. not mid-token) */ \
+ ST(s, \
+ /* OTHER => */ WORD | FWD, \
+ /* LU => */ LITPFX | FWD, \
+ /* u => */ uPFX | FWD, \
+ /* DIGIT => */ NUM | FWD, \
+ /* 8 => */ NUM | FWD, \
+ /* OP1 => */ OP1 | FWD, \
+ /* MIGHTEQ => */ MIGHTEQ | FWD, \
+ /* PLUS => */ PLUS | FWD, \
+ /* MINUS => */ MINUS | FWD, \
+ /* EQ => */ MIGHTEQ | FWD, \
+ /* LT => */ LT | FWD, \
+ /* GT => */ GT | FWD, \
+ /* AND => */ AND | FWD, \
+ /* OR => */ OR | FWD, \
+ /* DOT => */ DOT | FWD, \
+ /* STAR => */ MIGHTEQ | FWD, \
+ /* HASH => */ HASH | FWD, \
+ /* COLON => */ COLON | FWD, \
+ /* FS => */ FS | FWD, \
+ /* BS => */ WS_BS, \
+ /* SQ => */ CHRLIT | FWD, \
+ /* DQ => */ STRLIT | FWD, \
+ /* SP => */ s, \
+ /* NL => */ BOL | FWD \
+ )
+
+// VERY long table {{{
+static const unsigned char trans[NSTATES * NCLASSES] = {
+ ST(BOL, // note: same as ST_TERM but without FWD so we don't repeat NLs
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ BOL_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ BOL,
+ /* NL => */ BOL
+ )
+ ST(BOL_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD | FWD | FIXPOS, // maybe UCN
+ /* u => */ WORD | FWD | FIXPOS, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS | FWD | FIXPOS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ BOL_BS,
+ /* NL => */ BOL
+ )
+ ST_TERM(WS)
+ ST(WS_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD | FWD | FIXPOS, // maybe UCN
+ /* u => */ WORD | FWD | FIXPOS, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS | FWD | FIXPOS, // maybe UCN split over liens
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ WS_BS,
+ /* NL => */ WS
+ )
+ ST(uPFX,
+ /* OTHER => */ WORD,
+ /* LU => */ WORD,
+ /* u => */ WORD,
+ /* DIGIT => */ WORD,
+ /* 8 => */ LITPFX,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ uPFX_BS,
+ /* SQ => */ CHRLIT,
+ /* DQ => */ STRLIT,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(uPFX_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD, // maybe UCN
+ /* u => */ WORD, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ uPFX_BS,
+ /* NL => */ uPFX
+ )
+ ST(LITPFX,
+ /* OTHER => */ WORD,
+ /* LU => */ WORD,
+ /* u => */ WORD,
+ /* DIGIT => */ WORD,
+ /* 8 => */ WORD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ LITPFX_BS,
+ /* SQ => */ CHRLIT,
+ /* DQ => */ STRLIT,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(LITPFX_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD, // maybe UCN
+ /* u => */ WORD, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ LITPFX_BS,
+ /* NL => */ LITPFX
+ )
+ ST(WORD,
+ /* OTHER => */ WORD,
+ /* LU => */ WORD,
+ /* u => */ WORD,
+ /* DIGIT => */ WORD,
+ /* 8 => */ WORD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ WORD_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(WORD_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD, // maybe UCN
+ /* u => */ WORD, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ WORD_BS,
+ /* NL => */ WORD
+ )
+ ST(NUM,
+ /* OTHER => */ NUM,
+ /* LU => */ NUM,
+ /* u => */ NUM,
+ /* DIGIT => */ NUM,
+ /* 8 => */ NUM,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ NUM,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ NUM_BS,
+ /* SQ => */ NUM, // C23
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(NUM_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ ERROR | FWD, // UCN mid-number makes no sense
+ /* u => */ ERROR | FWD,
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ ERROR | FWD,
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ NUM_BS,
+ /* NL => */ NUM
+ )
+ ST(MIGHTEQ,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ MIGHTEQ_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(MIGHTEQ)
+ ST(MINUS,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ OP2,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ OP2,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ MINUS_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(MINUS)
+ ST(PLUS,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ DOT | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ OP2,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ MIGHTEQ | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ PLUS_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(PLUS)
+ ST(AND,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ OP2,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ AND_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(AND)
+ ST(OR,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OP2,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ OR_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(OR)
+ ST(HASH,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ OP2,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ HASH_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(HASH)
+ ST(COLON,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ OP2,
+ /* FS => */ FS | FWD,
+ /* BS => */ COLON_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(COLON)
+ ST(FS,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ BLKCOMM,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ LNCOMM,
+ /* BS => */ FS_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(FS)
+ ST(LT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ SHIFT,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ LT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(LT)
+ ST(GT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ SHIFT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ GT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(GT)
+ ST(DOT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM,
+ /* 8 => */ NUM,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ DOT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(DOT)
+ ST(SHIFT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ SHIFTEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ SHIFT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(SHIFT)
+ ST(LNCOMM,
+ /* OTHER => */ LNCOMM,
+ /* LU => */ LNCOMM,
+ /* u => */ LNCOMM,
+ /* DIGIT => */ LNCOMM,
+ /* 8 => */ LNCOMM,
+ /* OP1 => */ LNCOMM,
+ /* MIGHTEQ => */ LNCOMM,
+ /* PLUS => */ LNCOMM,
+ /* MINUS => */ LNCOMM,
+ /* EQ => */ LNCOMM,
+ /* LT => */ LNCOMM,
+ /* GT => */ LNCOMM,
+ /* AND => */ LNCOMM,
+ /* OR => */ LNCOMM,
+ /* DOT => */ LNCOMM,
+ /* STAR => */ LNCOMM,
+ /* HASH => */ LNCOMM,
+ /* COLON => */ LNCOMM,
+ /* FS => */ LNCOMM,
+ /* BS => */ LNCOMM_BS,
+ /* SQ => */ LNCOMM,
+ /* DQ => */ LNCOMM,
+ /* SP => */ LNCOMM,
+ /* NL => */ BOL | FWD
+ )
+ ST(LNCOMM_BS,
+ /* OTHER => */ LNCOMM,
+ /* LU => */ LNCOMM,
+ /* u => */ LNCOMM,
+ /* DIGIT => */ LNCOMM,
+ /* 8 => */ LNCOMM,
+ /* OP1 => */ LNCOMM,
+ /* MIGHTEQ => */ LNCOMM,
+ /* PLUS => */ LNCOMM,
+ /* MINUS => */ LNCOMM,
+ /* EQ => */ LNCOMM,
+ /* LT => */ LNCOMM,
+ /* GT => */ LNCOMM,
+ /* AND => */ LNCOMM,
+ /* OR => */ LNCOMM,
+ /* DOT => */ LNCOMM,
+ /* STAR => */ LNCOMM,
+ /* HASH => */ LNCOMM,
+ /* COLON => */ LNCOMM,
+ /* FS => */ LNCOMM,
+ /* BS => */ LNCOMM_BS,
+ /* SQ => */ LNCOMM,
+ /* DQ => */ LNCOMM,
+ /* SP => */ LNCOMM_BS,
+ /* NL => */ LNCOMM
+ )
+ ST(BLKCOMM,
+ /* OTHER => */ BLKCOMM,
+ /* LU => */ BLKCOMM,
+ /* u => */ BLKCOMM,
+ /* DIGIT => */ BLKCOMM,
+ /* 8 => */ BLKCOMM,
+ /* OP1 => */ BLKCOMM,
+ /* MIGHTEQ => */ BLKCOMM,
+ /* PLUS => */ BLKCOMM,
+ /* MINUS => */ BLKCOMM,
+ /* EQ => */ BLKCOMM,
+ /* LT => */ BLKCOMM,
+ /* GT => */ BLKCOMM,
+ /* AND => */ BLKCOMM,
+ /* OR => */ BLKCOMM,
+ /* DOT => */ BLKCOMM,
+ /* STAR => */ BLKSTAR,
+ /* HASH => */ BLKCOMM,
+ /* COLON => */ BLKCOMM,
+ /* FS => */ BLKCOMM,
+ /* BS => */ BLKCOMM,
+ /* SQ => */ BLKCOMM,
+ /* DQ => */ BLKCOMM,
+ /* SP => */ BLKCOMM,
+ /* NL => */ BLKCOMM
+ )
+ ST(BLKSTAR,
+ /* OTHER => */ BLKCOMM,
+ /* LU => */ BLKCOMM,
+ /* u => */ BLKCOMM,
+ /* DIGIT => */ BLKCOMM,
+ /* 8 => */ BLKCOMM,
+ /* OP1 => */ BLKCOMM,
+ /* MIGHTEQ => */ BLKCOMM,
+ /* PLUS => */ BLKCOMM,
+ /* MINUS => */ BLKCOMM,
+ /* EQ => */ BLKCOMM,
+ /* LT => */ BLKCOMM,
+ /* GT => */ BLKCOMM,
+ /* AND => */ BLKCOMM,
+ /* OR => */ BLKCOMM,
+ /* DOT => */ BLKCOMM,
+ /* STAR => */ BLKSTAR,
+ /* HASH => */ BLKCOMM,
+ /* COLON => */ BLKCOMM,
+ /* FS => */ WS,
+ /* BS => */ BLKSTAR_BS,
+ /* SQ => */ BLKCOMM,
+ /* DQ => */ BLKCOMM,
+ /* SP => */ BLKCOMM,
+ /* NL => */ BLKCOMM
+ )
+ ST(BLKSTAR_BS,
+ /* OTHER => */ BLKCOMM,
+ /* LU => */ BLKCOMM,
+ /* u => */ BLKCOMM,
+ /* DIGIT => */ BLKCOMM,
+ /* 8 => */ BLKCOMM,
+ /* OP1 => */ BLKCOMM,
+ /* MIGHTEQ => */ BLKCOMM,
+ /* PLUS => */ BLKCOMM,
+ /* MINUS => */ BLKCOMM,
+ /* EQ => */ BLKCOMM,
+ /* LT => */ BLKCOMM,
+ /* GT => */ BLKCOMM,
+ /* AND => */ BLKCOMM,
+ /* OR => */ BLKCOMM,
+ /* DOT => */ BLKCOMM,
+ /* STAR => */ BLKSTAR,
+ /* HASH => */ BLKCOMM,
+ /* COLON => */ BLKCOMM,
+ /* FS => */ WS,
+ /* BS => */ BLKCOMM,
+ /* SQ => */ BLKCOMM,
+ /* DQ => */ BLKCOMM,
+ /* SP => */ BLKSTAR_BS,
+ /* NL => */ BLKSTAR
+ )
+ ST_TERM(OP1)
+ ST_TERM(OP2)
+ ST_TERM(OPEQ)
+ ST_TERM(SHIFTEQ)
+ ST(STRLIT,
+ /* OTHER => */ STRLIT,
+ /* LU => */ STRLIT,
+ /* u => */ STRLIT,
+ /* DIGIT => */ STRLIT,
+ /* 8 => */ STRLIT,
+ /* OP1 => */ STRLIT,
+ /* MIGHTEQ => */ STRLIT,
+ /* PLUS => */ STRLIT,
+ /* MINUS => */ STRLIT,
+ /* EQ => */ STRLIT,
+ /* LT => */ STRLIT,
+ /* GT => */ STRLIT,
+ /* AND => */ STRLIT,
+ /* OR => */ STRLIT,
+ /* DOT => */ STRLIT,
+ /* STAR => */ STRLIT,
+ /* HASH => */ STRLIT,
+ /* COLON => */ STRLIT,
+ /* FS => */ STRLIT,
+ /* BS => */ STRLIT_BS,
+ /* SQ => */ STRLIT,
+ /* DQ => */ WS,
+ /* SP => */ STRLIT,
+ /* NL => */ ERROR | FWD
+ )
+ ST(STRLIT_BS,
+ /* OTHER => */ STRLIT,
+ /* LU => */ STRLIT,
+ /* u => */ STRLIT,
+ /* DIGIT => */ STRLIT,
+ /* 8 => */ STRLIT,
+ /* OP1 => */ STRLIT,
+ /* MIGHTEQ => */ STRLIT,
+ /* PLUS => */ STRLIT,
+ /* MINUS => */ STRLIT,
+ /* EQ => */ STRLIT,
+ /* LT => */ STRLIT,
+ /* GT => */ STRLIT,
+ /* AND => */ STRLIT,
+ /* OR => */ STRLIT,
+ /* DOT => */ STRLIT,
+ /* STAR => */ STRLIT,
+ /* HASH => */ STRLIT,
+ /* COLON => */ STRLIT,
+ /* FS => */ STRLIT,
+ /* BS => */ STRLIT,
+ /* SQ => */ STRLIT,
+ /* DQ => */ STRLIT,
+ /* SP => */ STRLIT_BS_WS,
+ /* NL => */ STRLIT
+ )
+ ST(STRLIT_BS_WS, // ugh this really does suck
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ ERROR | FWD,
+ /* u => */ ERROR | FWD,
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ ERROR | FWD,
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ STRLIT_BS_WS,
+ /* NL => */ STRLIT
+ )
+ ST(CHRLIT,
+ /* OTHER => */ CHRLIT,
+ /* LU => */ CHRLIT,
+ /* u => */ CHRLIT,
+ /* DIGIT => */ CHRLIT,
+ /* 8 => */ CHRLIT,
+ /* OP1 => */ CHRLIT,
+ /* MIGHTEQ => */ CHRLIT,
+ /* PLUS => */ CHRLIT,
+ /* MINUS => */ CHRLIT,
+ /* EQ => */ CHRLIT,
+ /* LT => */ CHRLIT,
+ /* GT => */ CHRLIT,
+ /* AND => */ CHRLIT,
+ /* OR => */ CHRLIT,
+ /* DOT => */ CHRLIT,
+ /* STAR => */ CHRLIT,
+ /* HASH => */ CHRLIT,
+ /* COLON => */ CHRLIT,
+ /* FS => */ CHRLIT,
+ /* BS => */ CHRLIT_BS,
+ /* SQ => */ WS,
+ /* DQ => */ CHRLIT,
+ /* SP => */ CHRLIT,
+ /* NL => */ ERROR | FWD
+ )
+ ST(CHRLIT_BS,
+ /* OTHER => */ CHRLIT,
+ /* LU => */ CHRLIT,
+ /* u => */ CHRLIT,
+ /* DIGIT => */ CHRLIT,
+ /* 8 => */ CHRLIT,
+ /* OP1 => */ CHRLIT,
+ /* MIGHTEQ => */ CHRLIT,
+ /* PLUS => */ CHRLIT,
+ /* MINUS => */ CHRLIT,
+ /* EQ => */ CHRLIT,
+ /* LT => */ CHRLIT,
+ /* GT => */ CHRLIT,
+ /* AND => */ CHRLIT,
+ /* OR => */ CHRLIT,
+ /* DOT => */ CHRLIT,
+ /* STAR => */ CHRLIT,
+ /* HASH => */ CHRLIT,
+ /* COLON => */ CHRLIT,
+ /* FS => */ CHRLIT,
+ /* BS => */ CHRLIT,
+ /* SQ => */ CHRLIT,
+ /* DQ => */ CHRLIT,
+ /* SP => */ CHRLIT_BS_WS,
+ /* NL => */ CHRLIT
+ )
+ ST(CHRLIT_BS_WS, // ugh this really does also suck
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ ERROR | FWD,
+ /* u => */ ERROR | FWD,
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ ERROR | FWD,
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ CHRLIT_BS_WS,
+ /* NL => */ CHRLIT
+ )
+ ST(ERROR, // dead end, caught outside the branchless main loop
+ /* OTHER => */ ERROR,
+ /* LU => */ ERROR,
+ /* u => */ ERROR,
+ /* DIGIT => */ ERROR,
+ /* 8 => */ ERROR,
+ /* OP1 => */ ERROR,
+ /* MIGHTEQ => */ ERROR,
+ /* PLUS => */ ERROR,
+ /* MINUS => */ ERROR,
+ /* EQ => */ ERROR,
+ /* LT => */ ERROR,
+ /* GT => */ ERROR,
+ /* AND => */ ERROR,
+ /* OR => */ ERROR,
+ /* DOT => */ ERROR,
+ /* STAR => */ ERROR,
+ /* HASH => */ ERROR,
+ /* COLON => */ ERROR,
+ /* FS => */ ERROR,
+ /* BS => */ ERROR,
+ /* SQ => */ ERROR,
+ /* DQ => */ ERROR,
+ /* SP => */ ERROR,
+ /* NL => */ ERROR
+ )
+};
+// }}}
+
+static const unsigned char statetoks[] = {
+ // comments here cover corresponding tokens. stuff in parentheses doesn't
+ // matter (could have any value)
+ CLEX_TOK_EOL, // BOL, (BOL_BS)
+ CLEX_TOK_OP1, // OP1, (WS)
+ CLEX_TOK_OP2, // OP2, (WS_BS)
+ CLEX_TOK_IDENT, // LITPFX, (LITPFX_BS)
+ CLEX_TOK_IDENT, // uPFX, (uPFX_BS)
+ CLEX_TOK_IDENT, // WORD, (WORD_BS)
+ CLEX_TOK_NUM, // NUM, (NUM_BS)
+ CLEX_TOK_OP1, // MIGHTEQ, (MIGHTEQ_BS)
+ CLEX_TOK_OP1, // MINUS, (MINUS_BS)
+ CLEX_TOK_OP1, // PLUS, (PLUS_BS)
+ CLEX_TOK_OP1, // AND, (AND_BS)
+ CLEX_TOK_OP1, // OR, (OR_BS)
+ CLEX_TOK_OP1, // HASH, (HASH_BS)
+ CLEX_TOK_OP1, // COLON, (COLON_BS)
+ CLEX_TOK_OP1, // FS, (FS_BS)
+ CLEX_TOK_OP1, // LT, (LT_BS)
+ CLEX_TOK_OP1, // GT, (GT_BS)
+ CLEX_TOK_OP1, // DOT, (DOT_BS)
+ CLEX_TOK_SHIFT, // SHIFT, (SHIFT_BS)
+ CLEX_TOK_LNCOMM, // LNCOMM, (LNCOMM_BS)
+ CLEX_TOK_BLKCOMM, // BLKCOMM, (BLKSTAR)
+ CLEX_TOK_OPEQ, // OPEQ, (BLKSTAR_BS)
+ CLEX_TOK_SHIFTEQ, // SHIFTEQ, (STRLIT_BS_WS)
+ CLEX_TOK_STR, // STRLIT, (STRLIT_BS)
+ CLEX_TOK_CHAR, // CHRLIT, (CHRLIT_BS)
+ 200 // (ERROR), (CHRLIT_BS_WS) // completely arbitrary value!
+};
+
+// Uncomment and LSP-inspect this for an idea of how much space is used:
+//enum { TABLESPACE = sizeof(classes) + sizeof(trans) + sizeof(statetoks) };
+// => 1786 bytes.
+
+// somewhat inefficient number formatter - size matters more than speed here.
+// also assumes there's extra space in the buffer, because there always is.
+static cold int err_fmtnum(char *out, unsigned int n) {
+ int i = 0, j = 10;
+ do {
+ // explicitly optimised division by 10 to make sure we NEVER have a
+ // divide instruction because integer division is evil.
+ unsigned int div10 = n * 3435973837ull >> 35;
+ // apparently doing * 10 here emits a more efficient lea-based shift-add
+ // thingy than attempting to shift-add by hand, at least with clang.
+ unsigned int remainder = n - div10 * 10;
+ out[--j] = '0' + remainder;
+ n = div10;
+ } while (n);
+ do out[i++] = out[j++]; while (j < 10);
+ return i;
+}
+
+static cold struct linecol {
+ unsigned int ln, col;
+} getlinecol(const char *buf, unsigned int off) {
+ // we don't store line and col for every token as it wastes a lot of memory.
+ // instead we can simply re-scan and count the newlines.
+ // FIXME: grapheme widths for columns? horrendous, but technically correct!
+ // if not that, then probably at least count utf8 codepoints.
+ unsigned int ln = 1, col = 0;
+ for (unsigned int i = 0;; ++i, ++col) {
+ if (buf[i] == '\n') { ++ln; col = 0; }
+ if (i == off) break;
+ }
+ return (struct linecol) {ln, col};
+}
+
+static cold char *err_putprefix(char *restrict out, const char *f,
+ const char *restrict buf, unsigned int off) {
+ struct linecol lc = getlinecol(buf, off);
+ // "filename:line:col: "
+ while (*f) *out++ = *f++;
+ *out++ = ':';
+ out += err_fmtnum(out, lc.ln);
+ *out++ = ':';
+ out += err_fmtnum(out, lc.col);
+ *out++ = ':';
+ *out++ = ' ';
+ return out;
+}
+
+// if we have clang we can avoid even doing a library call here. otherwise we
+// still only depend on memcpy which is pretty reasonable. we could also in
+// theory do some manual word-wise copying of the strings below, but eww.
+#ifdef __clang__
+#define copy __builtin_memcpy_inline
+#else
+#define copy memcpy
+#endif
+
+static cold void err(struct clex *c, const char *p, unsigned int off, int state,
+ int prevtok, const char *f) {
+ if (state == STATE_ERROR) {
+ // all transitions to ERROR have FWD, so we get a dummy token at the
+ // point where the error was detected. if this token points at a
+ // newline, then we have an unterminated literal. if it does *not* point
+ // at a newline, we have a non-newline after a backslash. in the latter
+ // case, backtrack to the backslash to report it as unexpected. in the
+ // former case, move back one character to point to the end of the line.
+ if (p[off--] != '\n') while (p[off] != '\\') --off;
+ }
+ // format the message as f:ln:col: msg. pretty verbose due to not using
+ // stdio or any other convenient formatting library, but also pretty simple
+ // IMPORTANT: clex_memreq() must be recalculated after changing any of this!
+ char *msg = (char *)c->tokoffs, *msgp = msg;
+ c->err = msg;
+ msgp = err_putprefix(msgp, f, p, off);
+ static const char strblock[78] =
+ "unterminated " // + 0, len 13
+ "string" // +13, len 6
+ "character" // +19, len 9
+ " literal" // +28, len 8
+ "block comment" // +36, len 13
+ "expected" // +49, len 8
+ " line after" // +57, len 11
+ " backslash"; // +68, len 10
+ // minor code size trick: write a little too much first, then replace parts
+ copy(msgp, strblock, 19); // "unterminated string"
+ if (state == STATE_ERROR) {
+ // note: ERROR can only follow another token-producing state. prevtok
+ // could otherwise contain garbage from one of the dummy slots (see
+ // dotok() below) but in this scenario that will never be the case.
+ if (prevtok == CLEX_TOK_CHAR) {
+ // s/string /character literal/
+ copy(msgp + 13, strblock + 19, 9 + 8);
+ msgp += 13 + 9 + 8;
+ }
+ else if (prevtok == CLEX_TOK_STR) {
+ // append "literal" -> "
+ copy(msgp + 19, strblock + 29, 8);
+ msgp += 13 + 6 + 8;
+ }
+ else /* prevtok == 200 (backslash case above) */ {
+ // s/terminated/expected/ -> unexpected
+ copy(msgp + 2, strblock + 49, 8);
+ // append " backslash" -> "unexpected backslash"
+ copy(msgp + 10, strblock + 68, 10);
+ msgp += 10 + 10;
+ }
+ }
+ else if (state == STATE_BLKCOMM) {
+ // s/string/block comment/ -> "unterminated block comment"
+ copy(msgp + 13, strblock + 36, 13);
+ msgp += 13 + 13;
+ }
+ else /* state == STATE_BS */ {
+ // s/unterminated string/expected line after backslash/
+ copy(msgp, strblock + 49, 8 + 11 + 10);
+ msgp += 8 + 11 + 10;
+ }
+ *msgp = '\0';
+ c->errlen = msgp - msg;
+}
+
+static inline void dotok(struct clex *c, const unsigned char *restrict p,
+ unsigned int sz, const char *restrict f) {
+ unsigned char *toks = c->toks;
+ unsigned int *tokoffs = c->tokoffs;
+ int state = STATE_BOL;
+ unsigned int off = 0;
+ unsigned int ntoks = 1; // skip first slot, see below
+ for (; off != sz; ++off) {
+ unsigned char state_trans = trans[classes[p[off]] + state];
+ unsigned int lower7 = state_trans & 127;
+ unsigned int off_adj = off - (state_trans >> 7); // see FIXPOS
+ unsigned int fwdbit = state_trans & 1; // see FWD
+ // bit 2 (low state bit): set -> 1; unset -> -1.
+ // => only even-indexed states (which are terminal) update the type
+ // note: this means we have to waste the bottom two slots. oh well!
+ unsigned int idxmask = (state_trans | -3u) + 2;
+ // set the starting offset for the *next* token, that way if we're not
+ // starting a new one we don't clobber the existing position.
+ tokoffs[ntoks + 1] = off_adj;
+ // if FWD bit is set then we have a new token; at this point we increase
+ // ntoks meaning the first token will be at position 2. hence the waste.
+ ntoks += fwdbit; // advance if we have a new token
+ state = lower7 >> 1; // middle 6 bits are our new state
+ toks[ntoks & idxmask] = statetoks[lower7 >> 2];
+ }
+ // hide the wasted slots from the caller
+ c->ntoks = ntoks - 2; c->toks += 2; c->tokoffs += 2;
+ if_cold (state != STATE_BOL) {
+ if (f) {
+ err(c, (char *)p, tokoffs[ntoks], state, toks[ntoks - 1], f);
+ }
+ else {
+ c->err = "syntax error";
+ c->errlen = 12;
+ }
+ }
+}
+
+struct clex clex(const char *restrict buf, unsigned int sz,
+ void *restrict outmem, const char *restrict filename) {
+ struct clex c = {
+ .tokoffs = outmem,
+ .toks = (unsigned char *)(c.tokoffs + sz)
+ };
+ if_cold (sz != 0 && buf[sz - 1] != '\n') {
+ c.err = "input is not a valid text file (must end with newline)";
+ return c;
+ }
+ dotok(&c, (unsigned char *)buf, sz, filename);
+ return c;
+}
+
+static inline bool ishex(char c) {
+ char lower = c | 32;
+ return c >= '0' & c <= '9' | lower >= 'a' & lower <= 'f';
+}
+static inline int hexval(char c) {
+ unsigned char u = c;
+ return (u & 15) + (u >> 6) * 9; // assumes ishex(c)
+}
+
+static inline int utf8len_ucs2(int codepoint) { // assumes codepoint <= 0xFFFF
+ if (codepoint > 0x7FF) return 3;
+ // would be weird to use \u for ascii, so do this branch last
+ if (codepoint <= 0x7F) return 1;
+ return 2;
+}
+static inline int utf8len(int codepoint) { // assumes codepoint <= 0x10FFFF
+ if (codepoint <= 0xFFFF) { // most likely use for \U over \u
+ if (codepoint <= 0x7FF) {
+ if (codepoint <= 0x7F) return 1; // should still be least likely
+ return 2;
+ }
+ return 3;
+ }
+ return 4;
+}
+
+static inline bool isbadcodepoint_ucs2(int codepoint) { // assumes <= 0xFFFF
+ return codepoint >= 0xD800 && codepoint <= 0xDFFF || // surrogates
+ codepoint >= 0xFDD0 && codepoint <= 0xFDEF || // non-characters
+ (codepoint & 0x0FFE) == 0x0FFE; // more non-characters
+}
+static inline bool isbadcodepoint(int codepoint) {
+ return isbadcodepoint_ucs2(codepoint) ||
+ codepoint >= 0x110000 && codepoint <= 0x1FFFFF || // more non-chars
+ codepoint > 0x10FFFF; // max valid codepoint
+}
+
+static struct bsiter_ret { char c; const char *next; } bsiter(const char *p) {
+ char c;
+ // skip line continuations, but don't ignore backslashes mid-line. may need
+ // some backtracking/re-scanning - oh well. this is the slow path.
+ while ((c = *p) == '\\') {
+ const char *q = p;
+ while (classes[(unsigned char)*++q] == CLASS_SP);
+ if (*q != '\n') break; // backslash is significant, return it
+ p = q + 1; // skip newline and keep scanning forward
+ }
+ return (struct bsiter_ret){c, p + 1};
+}
+
+struct clex_ident_validate_ret clex_ident_validate(const struct clex *c,
+ const char *restrict buf, unsigned int idx) {
+ const char *p = buf + c->tokoffs[idx], *start = p;
+ // N.B. file always ends in an EOL, so no need for bounds check on this loop
+ struct clex_ident_validate_ret ret;
+ for (int namelen = 0;;) {
+ struct bsiter_ret iter = bsiter(p); char c = iter.c; p = iter.next;
+ // first 5 entries in CLASSES() are alphanumeric; if it's a symbol or
+ // space, we're past this token...
+ if (classes[(unsigned char)c] > 4 * NSTATES) {
+ if (c == '\\') { // ... unless it's a UCN, or bad syntax
+ const char *pbs = p;
+ iter = bsiter(p); c = iter.c; p = iter.next;
+ if (c == 'u') {
+ char hex[4];
+ iter = bsiter(p); hex[0] = iter.c;
+ const char *p0 = iter.next;
+ for (int i = 0;;) {
+ if_cold (!ishex(hex[i])) {
+ ret.err = CLEX_IDENT_BADUCNHEX4;
+ ret.err_off = p - buf;
+ return ret;
+ }
+ p = iter.next;
+ if (++i == 4) break;
+ iter = bsiter(p); hex[i] = iter.c;
+ }
+ int codepoint = hexval(hex[0]) << 12 | hexval(hex[1]) << 8 |
+ hexval(hex[2]) << 4 | hexval(hex[3]);
+ if_cold (isbadcodepoint_ucs2(codepoint)) {
+ ret.err = CLEX_IDENT_BADCODEPOINT;
+ ret.err_off = p0 - buf;
+ return ret;
+ }
+ namelen += utf8len_ucs2(codepoint);
+ }
+ else if (c == 'U') {
+ char hex[6];
+ iter = bsiter(p); hex[0] = iter.c;
+ const char *p0 = iter.next;
+ for (int i = 0;;) {
+ if_cold (!ishex(hex[i])) {
+ ret.err = CLEX_IDENT_BADUCNHEX6;
+ ret.err_off = p - buf;
+ return ret;
+ }
+ p = iter.next;
+ if (++i == 6) break;
+ iter = bsiter(p); hex[i] = iter.c;
+ }
+ int codepoint = hexval(hex[0]) << 20 | hexval(hex[1]) << 16 |
+ hexval(hex[2]) << 12 | hexval(hex[3] << 8) |
+ hexval(hex[4]) << 4 | hexval(hex[5]);
+ if_cold (isbadcodepoint(codepoint)) {
+ ret.err = CLEX_IDENT_BADCODEPOINT;
+ ret.err_off = p0 - buf;
+ return ret;
+ }
+ namelen += utf8len(codepoint);
+ }
+ else { // we skipped line conts already; must be bad syntax here
+ ret.err = CLEX_IDENT_UNEXPBACKSLASH;
+ ret.err_off = pbs - buf;
+ return ret;
+ }
+ }
+ else {
+ ret.err = 0; ret.ext = p - start; ret.len = namelen;
+ return ret;
+ }
+ }
+ else {
+ ++namelen;
+ }
+ }
+}
+
+cold int clex_ident_errstr(char *restrict out, const char *restrict buf,
+ enum clex_ident_validate_err err, unsigned int err_off,
+ const char *restrict filename) {
+ char *p = out;
+ static const char strblock[81] =
+ "illegal character in 3-digit UCN" // (becomes 4 or 6, see below)
+ "Unicode codepoint"
+ "identifier"
+ "unexpected end of line";
+ if (buf[err_off] == '\n') {
+ p = err_putprefix(p, filename, buf, err_off - 1);
+ copy(p, strblock + 59, 22);
+ p[22] = '\0';
+ return p + 23 - out;
+ }
+ switch (err) {
+ case CLEX_IDENT_BADUCNHEX4:
+ case CLEX_IDENT_BADUCNHEX6:
+ copy(p, strblock, 32);
+ p[21] += err; // 3 -> 4 or 6 (N.B. relies on enum value!!!)
+ p += 32;
+ break;
+ case CLEX_IDENT_BADCODEPOINT:
+ copy(p, strblock, 8); // "illegal "
+ copy(p + 8, strblock + 32, 17); // "Unicode codepoint"
+ p += 25;
+ break;
+ case CLEX_IDENT_UNEXPBACKSLASH:
+ copy(p, strblock, 18); // "illegal character in "
+ copy(p + 18, strblock + 49, 10); // "identifier"
+ p += 28;
+ break;
+ default:
+ unreachable;
+ }
+ *p = '\0';
+ return p - out;
+}
+
+static int utf8put_ucs2(char *out, int codepoint) { // assumes valid UCS2
+ if (codepoint > 0x7FF) {
+ out[0] = 0xE0 | codepoint >> 12;
+ out[1] = 0x80 | codepoint >> 6 & 0x3F;
+ out[2] = 0x80 | codepoint & 0x3F;
+ return 3;
+ }
+ if (codepoint <= 0x7F) {
+ out[0] = codepoint;
+ return 1;
+ }
+ out[0] = 0xC0 | codepoint >> 6;
+ out[1] = 0x80 | codepoint & 0x3F;
+ return 2;
+}
+static int utf8put(char *out, int codepoint) { // assumes valid Unicode
+ if (codepoint <= 0xFFFF) return utf8put_ucs2(out, codepoint);
+ out[0] = 0xF0 | codepoint >> 18;
+ out[1] = 0x80 | codepoint >> 12 & 0x3F;
+ out[2] = 0x80 | codepoint >> 6 & 0x3F;
+ out[3] = 0x80 | codepoint & 0x3F;
+ return 4;
+}
+
+void clex_ident(const struct clex *c, const char *restrict buf,
+ unsigned int idx, char *restrict out) {
+ const char *p = buf + c->tokoffs[idx];
+ struct bsiter_ret iter = bsiter(p); char ch = iter.c; p = iter.next;
+ for (;;) {
+ if (classes[(unsigned char)ch] > 4 * NSTATES) {
+ if (ch == '\\') {
+ iter = bsiter(p); ch = iter.c; p = iter.next;
+ if (ch == 'u') {
+ iter = bsiter(p); char c0 = iter.c; p = iter.next;
+ iter = bsiter(p); char c1 = iter.c; p = iter.next;
+ iter = bsiter(p); char c2 = iter.c; p = iter.next;
+ iter = bsiter(p); char c3 = iter.c; p = iter.next;
+ int codepoint = hexval(c0) << 12 | hexval(c1) << 8 |
+ hexval(c2) << 4 | hexval(c3);
+ out += utf8put_ucs2(out, codepoint);
+ }
+ else /* ch == 'U' */ {
+ iter = bsiter(p); char c0 = iter.c; p = iter.next;
+ iter = bsiter(p); char c1 = iter.c; p = iter.next;
+ iter = bsiter(p); char c2 = iter.c; p = iter.next;
+ iter = bsiter(p); char c3 = iter.c; p = iter.next;
+ iter = bsiter(p); char c4 = iter.c; p = iter.next;
+ iter = bsiter(p); char c5 = iter.c; p = iter.next;
+ int codepoint = hexval(c0) << 20 | hexval(c1) << 16 |
+ hexval(c2) << 12 | (hexval(c3) << 8) |
+ hexval(c4) << 4 | hexval(c5);
+ out += utf8put(out, codepoint);
+ }
+ }
+ else {
+ return;
+ }
+ }
+ else {
+ *out++ = ch;
+ iter = bsiter(p); ch = iter.c; p = iter.next;
+ }
+ }
+}
+
+// vi: sw=4 ts=4 noet tw=80 cc=80 fdm=marker
diff --git a/src/chunklets/clex.h b/src/chunklets/clex.h
new file mode 100644
index 0000000..5d3a636
--- /dev/null
+++ b/src/chunklets/clex.h
@@ -0,0 +1,731 @@
+/*
+ * Copyright © Michael Smith <mikesmiffy128@gmail.com>
+ *
+ * Permission to use, copy, modify, and/or distribute this software for any
+ * purpose with or without fee is hereby granted, provided that the above
+ * copyright notice and this permission notice appear in all copies.
+ *
+ * THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
+ * REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
+ * AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
+ * INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
+ * LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
+ * OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
+ * PERFORMANCE OF THIS SOFTWARE.
+ */
+
+/* WARNING: this library is incomplete and still subject to change! */
+
+#ifndef INC_CHUNKLETS_CLEX_H
+#define INC_CHUNKLETS_CLEX_H
+
+#ifdef __cplusplus
+#define _clex_bool bool
+#define _clex_restrict __restrict // XXX: is this portable enough?
+#else
+#define _clex_bool _Bool
+#define _clex_restrict restrict
+#endif
+
+#ifdef _WIN32
+#ifdef __cplusplus
+typedef wchar_t clex_os_char; // ugh
+#else
+typedef unsigned short clex_os_char;
+#endif
+#else
+typedef char clex_os_char;
+#endif
+
+#ifdef _WIN64
+typedef long long clex_size;
+#else
+typedef long clex_size;
+#endif
+
+#if defined(__GNUC__) || defined(__clang__)
+#define _clex_unreachable __builtin_unreachable()
+#elif defined(_MSC_VER)
+#define _clex_unreachable __assume(0)
+#else
+#ifdef __cplusplus
+[[noreturn]] static inline void _clex_invoke_ub(void) {}
+#else
+static inline _Noreturn void _clex_invoke_ub(void) {}
+#endif
+#define _clex_unreachable (_clex_invoke_ub())
+#endif
+
+/*
+ * Determines the amount of memory required to call clex() on a given file.
+ * That memory must be provided as a contiguous block, with at least 4 byte
+ * alignment.
+ *
+ * filesz is the number of bytes in the file, and namelen is the length of the
+ * name. The reason namelen is required is because the memory is reused to
+ * produce an error string when lexing fails.
+ *
+ * filesz is assumed to be non-zero, since empty C source files are not allowed,
+ * and common memory allocators do not accept zero sizes either. To gracefully
+ * handle an empty file if desired, check for a zero size and avoid attempting
+ * tokenisation altogether.
+ *
+ * If a null filename will be passed to clex(), which skips error message
+ * formatting, namelen can be set to 0.
+ *
+ * NOTE: This assumes that the file is reasonably-sized. On 64-bit architectures,
+ * the file size must be not exceed 4GiB *minus 2 bytes*, due to an internal
+ * implementation detail. On 32-bit architectures, the file has to be even
+ * smaller in order to fit the token data in memory; it is recommended to stay
+ * below ~256MiB.
+ */
+static inline clex_size clex_memreq(unsigned int filesz, unsigned int namelen) {
+ clex_size sz = ((clex_size)filesz + 2) * 5 + 5;
+ if (sz < namelen + 64) return namelen + 64; // make space for error messages
+ return sz;
+}
+
+/* The main structure returned by a call to clex() - see below. */
+struct clex {
+ /*
+ * Check this first. It is a null pointer if lexing succeeded, or points to
+ * a null-terminated error string on failure. See also clex() for a
+ * description of how the error string is created.
+ */
+ const char *err;
+ union {
+ /*
+ * If err is not null, contains the length of the string, excluding the
+ * null terminator.
+ */
+ int errlen;
+ /*
+ * If lexing was successful (err is null), contains the number of tokens
+ * lexed. Use this to loop over toks and tokoffs.
+ */
+ unsigned int ntoks;
+ };
+ //char pad[4];
+ /*
+ * If lexing was successful (err is null), points to an array of token
+ * offsets within the lexed file. On failure, the value is undefined.
+ */
+ unsigned int *tokoffs;
+ /*
+ * If lexing was successful (err is null), points to an array of basic token
+ * types corresponding to each position in tokoffs. On failure, the value is
+ * undefined.
+ */
+ unsigned char *toks;
+};
+
+/*
+ * Lexes a C file into a list of tokens. See above comments on struct clex for
+ * return type information.
+ *
+ * buf and sz point to the contents of the file, which has to be fully loaded
+ * into memory prior to lexing.
+ *
+ * outmem is an opaque block of memory of at sufficient size calculated by
+ * clex_memreq(), and at least 4-byte alignment. It can be allocated by any
+ * means desired.
+ *
+ * filename is used purely for error reporting, though typically it would be the
+ * name of the file being lexed, of course. Normally the error string output
+ * will include the filename, row and column, and a brief description of the
+ * error. If this information is not needed, filename can be a null pointer
+ * instead, which disables this functionality.
+ */
+struct clex clex(const char *_clex_restrict buf, unsigned int sz,
+ void *_clex_restrict outmem, const char *_clex_restrict filename);
+
+/*
+ * These are the basic token types recognised by the lexer in its main pass.
+ * All the operators are grouped in such a way that they can be distinguished by
+ * looking at a single character.
+ *
+ * Identifiers and literals might not necessarily be totally valid; this is
+ * checked in more detail when parsing out specific values.
+ */
+enum clex_tok_type {
+ CLEX_TOK_IDENT, /* An identifier or keyword. */
+ CLEX_TOK_NUM, /* A numeric literal, which may or may not be totally valid */
+ CLEX_TOK_OP1, /* One of: + - * / = < > ! & | ^ ~ . , ( ) [ ] { } : ; ? # */
+ CLEX_TOK_OP2, /* One of: ++ -- -> && || ## :: */
+ CLEX_TOK_SHIFT, /* One of: >> << */
+ CLEX_TOK_OPEQ, /* One of: += -= *= /= &= |= ^= <= >= != == */
+ CLEX_TOK_SHIFTEQ, /* One of: >>= <<= */
+ CLEX_TOK_CHAR, /* A character literal. */
+ CLEX_TOK_STR, /* A string literal. */
+ CLEX_TOK_LNCOMM, /* A C++/C99-style one-line comment. */
+ CLEX_TOK_BLKCOMM, /* A classic C-style multi-line comment. */
+ CLEX_TOK_EOL /* A run of end-of-line characters, for parsing #directives. */
+};
+
+/* These are all the C operators, hopefully self-explanatory. */
+enum clex_op {
+ // -- op1 --
+ CLEX_OP_PLUS, /* + */
+ CLEX_OP_MINUS, /* - */
+ CLEX_OP_MULT, /* * (could also be a deref) */ // XXX: bad name maybe?
+ CLEX_OP_DIV, /* / */
+ CLEX_OP_ASSIGN, /* = */
+ CLEX_OP_LT, /* < */
+ CLEX_OP_GT, /* > */
+ CLEX_OP_NOT, /* ! */
+ CLEX_OP_BITAND, /* & (could also be an address-of) */ // XXX: bad name?
+ CLEX_OP_BITOR, /* | */
+ CLEX_OP_XOR, /* ^ */
+ CLEX_OP_BITNOT, /* ~ */
+ CLEX_OP_DOT, /* . */
+ CLEX_OP_COMMA, /* , */
+ CLEX_OP_LPAREN, /* ( */
+ CLEX_OP_RPAREN, /* ) */
+ CLEX_OP_LSQ, /* [ */
+ CLEX_OP_RSQ, /* ] */
+ CLEX_OP_LCURL, /* { */
+ CLEX_OP_RCURL, /* } */
+ CLEX_OP_COLON, /* : */
+ CLEX_OP_SEMICOL, /* ; */
+ CLEX_OP_QUESTION, /* ? */
+ CLEX_OP_HASH, /* # */
+ // -- op2 --
+ CLEX_OP_INC, /* ++ */
+ CLEX_OP_DEC, /* -- */
+ CLEX_OP_ARROW, /* -> */
+ CLEX_OP_AND, /* && */
+ CLEX_OP_OR, /* || */
+ CLEX_OP_PASTE, /* ## */
+ CLEX_OP_DCOLON, /* :: C23 attribute vendors/namespaces */
+ // -- shift --
+ CLEX_OP_LSH, /* << */
+ CLEX_OP_RSH, /* >> */
+ // -- opeq --
+ CLEX_OP_PLUSEQ, /* += */
+ CLEX_OP_MINUSEQ, /* -= */
+ CLEX_OP_MULTEQ, /* *= */
+ CLEX_OP_DIVEQ, /* /= */
+ CLEX_OP_ANDEQ, /* &= */
+ CLEX_OP_OREQ, /* |= */
+ CLEX_OP_XOREQ, /* ^= */
+ CLEX_OP_LTEQ, /* <= */
+ CLEX_OP_GTEQ, /* >= */
+ CLEX_OP_NOTEQ, /* != */
+ CLEX_OP_EQ, /* == */
+ // -- shifteq --
+ CLEX_OP_LSHEQ, /* <<= */
+ CLEX_OP_RSHEQ /* >>= */
+};
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OP1 at offset off
+ * by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_OP1, the behaviour is undefined.
+ */
+static inline enum clex_op clex_op1(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '+': return CLEX_OP_PLUS;
+ case '-': return CLEX_OP_MINUS;
+ case '*': return CLEX_OP_MULT;
+ case '/': return CLEX_OP_DIV;
+ case '=': return CLEX_OP_ASSIGN;
+ case '<': return CLEX_OP_LT;
+ case '>': return CLEX_OP_GT;
+ case '!': return CLEX_OP_NOT;
+ case '&': return CLEX_OP_BITAND;
+ case '|': return CLEX_OP_BITOR;
+ case '^': return CLEX_OP_XOR;
+ case '~': return CLEX_OP_BITNOT;
+ case '.': return CLEX_OP_DOT;
+ case ',': return CLEX_OP_COMMA;
+ case '(': return CLEX_OP_LPAREN;
+ case ')': return CLEX_OP_RPAREN;
+ case '[': return CLEX_OP_LSQ;
+ case ']': return CLEX_OP_RSQ;
+ case '{': return CLEX_OP_LCURL;
+ case '}': return CLEX_OP_RCURL;
+ case ':': return CLEX_OP_COLON;
+ case ';': return CLEX_OP_SEMICOL;
+ case '?': return CLEX_OP_QUESTION;
+ case '#': return CLEX_OP_HASH;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OP2 at offset off
+ * by inspecting the end of the token.
+ *
+ * If the token is not of type CLEX_TOK_OP2, the behaviour is undefined.
+ */
+static inline enum clex_op clex_op2(const char *buf, unsigned int off) {
+ for (;;) {
+ switch (buf[off + 1]) {
+ case '+': return CLEX_OP_INC;
+ case '-': return CLEX_OP_DEC;
+ case '>': return CLEX_OP_ARROW;
+ case '&': return CLEX_OP_AND;
+ case '|': return CLEX_OP_OR;
+ case '#': return CLEX_OP_PASTE;
+ case ':': return CLEX_OP_DCOLON;
+ // initial lex has validated backslash-newlines, so we can just
+ // ignore any of these characters here
+ case '\\': case ' ': case '\t': case '\r': case '\n': continue;
+ }
+ _clex_unreachable;
+ }
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_SHIFT at offset
+ * off by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_SHIFT, the behaviour is undefined.
+ */
+static inline enum clex_op clex_shift(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '>': return CLEX_OP_RSH;
+ case '<': return CLEX_OP_LSH;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OPEQ at offset
+ * off by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_OPEQ, the behaviour is undefined.
+ */
+static inline enum clex_op clex_opeq(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '+': return CLEX_OP_PLUSEQ;
+ case '-': return CLEX_OP_MINUSEQ;
+ case '*': return CLEX_OP_MULTEQ;
+ case '/': return CLEX_OP_DIVEQ;
+ case '&': return CLEX_OP_ANDEQ;
+ case '|': return CLEX_OP_OREQ;
+ case '^': return CLEX_OP_XOREQ;
+ case '<': return CLEX_OP_LTEQ;
+ case '>': return CLEX_OP_GTEQ;
+ case '!': return CLEX_OP_NOTEQ;
+ case '=': return CLEX_OP_EQ;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_SHIFTEQ at offset
+ * off by inspecting the first byte of the token.
+ *
+ * If the token is not of type CLEX_TOK_SHIFTEQ, the behaviour is undefined.
+ */
+static inline enum clex_op clex_shifteq(const char *buf, unsigned int off) {
+ switch (buf[off]) {
+ case '<': return CLEX_OP_RSHEQ;
+ case '>': return CLEX_OP_LSHEQ;
+ }
+ _clex_unreachable;
+}
+
+/*
+ * Determines whether the lexed token at a given index is an operator (not
+ * counting the likes of sizeof, which will have to be handled as a special case
+ * of an identifier/keyword for syntax purposes).
+ */
+static inline _clex_bool clex_isop(const struct clex *c, unsigned int idx) {
+ switch (c->toks[idx]) {
+ case CLEX_TOK_IDENT: return 0;
+ case CLEX_TOK_NUM: return 0;
+ case CLEX_TOK_OP1: return 1;
+ case CLEX_TOK_OP2: return 1;
+ case CLEX_TOK_SHIFT: return 1;
+ case CLEX_TOK_OPEQ: return 1;
+ case CLEX_TOK_SHIFTEQ: return 1;
+ case CLEX_TOK_CHAR: return 0;
+ case CLEX_TOK_STR: return 0;
+ case CLEX_TOK_LNCOMM: return 0;
+ case CLEX_TOK_BLKCOMM: return 0;
+ case CLEX_TOK_EOL: return 0;
+ }
+ _clex_unreachable;
+}
+
+/* Returns true if a token is either a unary or binary plus operator. */
+static inline _clex_bool clex_isplus(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '+';
+}
+
+/* Returns true if a token is either a unary or binary minus operator. */
+static inline _clex_bool clex_isminus(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '-';
+}
+
+/* Returns true if a token is either a multiply or deference operator. */
+static inline _clex_bool clex_ismult(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '*';
+}
+
+/* Returns true a token is a division operator. */
+static inline _clex_bool clex_isdiv(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '/';
+}
+
+/* Returns true if a token is an assignment operator. */
+static inline _clex_bool clex_isassign(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '=';
+}
+
+/* Returns true if a token is a less-than operator. */
+static inline _clex_bool clex_islt(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a greater-than operator. */
+static inline _clex_bool clex_isgt(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '>';
+}
+
+/* Returns true if a token is a negation operator. */
+static inline _clex_bool clex_isnot(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '!';
+}
+
+/* Returns true if a token is either a bitwise and operator or an indirection. */
+static inline _clex_bool clex_isbitand(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '&';
+}
+
+/* Returns true if a token is a bitwise or operator. */
+static inline _clex_bool clex_isbitor(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '|';
+}
+
+/* Returns true if a token is a bitwise exclusive-or operator. */
+static inline _clex_bool clex_isxor(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '^';
+}
+
+/* Returns true if a token is a bitwise negation operator. */
+static inline _clex_bool clex_isbitnot(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '~';
+}
+
+/* Returns true if a token is a struct member dot. */
+static inline _clex_bool clex_isdot(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '.';
+}
+
+/* Returns true if a token is a comma. */
+static inline _clex_bool clex_iscomma(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ',';
+}
+
+/* Returns true if a token is an opening/left parenthesis. */
+static inline _clex_bool clex_islparen(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '(';
+}
+
+/* Returns true if a token is a closing/right parenthesis. */
+static inline _clex_bool clex_isrparen(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ')';
+}
+
+/* Returns true if a token is an opening/left square bracket. */
+static inline _clex_bool clex_islsq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '[';
+}
+
+/* Returns true if a token is a closing/right square bracket. */
+static inline _clex_bool clex_isrsq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ']';
+}
+
+/* Returns true if a token is an opening/left curly brace. */
+static inline _clex_bool clex_islcurl(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '{';
+}
+
+/* Returns true if a token is a closing/right curly brace. */
+static inline _clex_bool clex_isrcurl(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '}';
+}
+
+/* Returns true if a token is a colon. */
+static inline _clex_bool clex_iscolon(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ':';
+}
+
+/* Returns true if a token is a semicolon. */
+static inline _clex_bool clex_issemicol(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == ';';
+}
+
+/* Returns true if a token is a question mark. */
+static inline _clex_bool clex_isquestion(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '?';
+}
+
+/* Returns true if a token is a preprocessor hash. */
+static inline _clex_bool clex_ishash(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP1 && buf[c->tokoffs[idx]] == '#';
+}
+
+/* Returns true if a token is a unary increment operator. */
+static inline _clex_bool clex_isinc(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '+';
+}
+
+/* Returns true if a token is a unary decrement operator. */
+static inline _clex_bool clex_isdec(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '-';
+}
+
+/* Returns true if a token is an arrow for indirect struct member access. */
+static inline _clex_bool clex_isarrow(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '>';
+}
+
+/* Returns true if a token is a conditional and operator. */
+static inline _clex_bool clex_isand(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '&';
+}
+
+/* Returns true if a token is a conditional or operator. */
+static inline _clex_bool clex_isor(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '|';
+}
+
+/* Returns true if a token is a token-pasting operator. */
+static inline _clex_bool clex_ispaste(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == '#';
+}
+
+/* Returns true if a token is a double colon for C23 attribute namespacing. */
+static inline _clex_bool clex_isdcol(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OP2 && buf[c->tokoffs[idx] + 1] == ':';
+}
+
+/* Returns true if a token is a left shift operator. */
+static inline _clex_bool clex_islsh(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFT && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a right shift operator. */
+static inline _clex_bool clex_isrsh(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFT && buf[c->tokoffs[idx]] == '>';
+}
+
+/* Returns true if a token is a binary increment (plus-assignment) operator. */
+static inline _clex_bool clex_ispluseq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '+';
+}
+
+/* Returns true if a token is a binary decrement (plus-assignment) operator. */
+static inline _clex_bool clex_isminuseq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '-';
+}
+
+/* Returns true if a token is a multiply-assignment operator. */
+static inline _clex_bool clex_ismulteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '*';
+}
+
+/* Returns true if a token is a divide-assignment operator. */
+static inline _clex_bool clex_isdiveq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '/';
+}
+
+/* Returns true if a token is a bitwise-and-assignment operator. */
+static inline _clex_bool clex_isandeq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '&';
+}
+
+/* Returns true if a token is a bitwise-or-assignment operator. */
+static inline _clex_bool clex_isoreq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '|';
+}
+
+/* Returns true if a token is a bitwise-xor-assignment operator. */
+static inline _clex_bool clex_isxoreq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '^';
+}
+
+/* Returns true if a token is a less-than-or-equal comparison operator. */
+static inline _clex_bool clex_islteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a greater-than-or-equal comparison operator. */
+static inline _clex_bool clex_isgteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '>';
+}
+
+/* Returns true if a token is an inequality comparison operator. */
+static inline _clex_bool clex_isnoteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '!';
+}
+
+/* Returns true if a token is an equality comparison operator. */
+static inline _clex_bool clex_iseq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_OPEQ && buf[c->tokoffs[idx]] == '=';
+}
+
+/* Returns true if a token is a left-shift-assignment operator. */
+static inline _clex_bool clex_islshifteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFTEQ && buf[c->tokoffs[idx]] == '<';
+}
+
+/* Returns true if a token is a right-shift-assignment operator. */
+static inline _clex_bool clex_isrshifteq(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ return c->toks[idx] == CLEX_TOK_SHIFTEQ && buf[c->tokoffs[idx]] == '>';
+}
+
+/*
+ * Determines the exact operator from a token of type CLEX_TOK_OP1,
+ * CLEX_TOK_OP2, CLEX_TOK_SHIFT, CLEX_TOK_OPEQ, or CLEX_TOK_SHIFTEQ.
+ *
+ * If the token is not of one of the above types, the behaviour is undefined.
+ * Call clex_isop() first to determine whether a token is of a valid type.
+ */
+static inline enum clex_op clex_op(const struct clex *c,
+ const char *_clex_restrict buf, unsigned int idx) {
+ unsigned int off = c->tokoffs[idx];
+ switch (c->toks[idx]) {
+ case CLEX_TOK_OP1: return clex_op1(buf, off);
+ case CLEX_TOK_OP2: return clex_op2(buf, off);
+ case CLEX_TOK_SHIFT: return clex_shift(buf, off);
+ case CLEX_TOK_OPEQ: return clex_opeq(buf, off);
+ case CLEX_TOK_SHIFTEQ: return clex_shifteq(buf, off);
+ }
+ _clex_unreachable;
+}
+
+enum clex_ident_validate_err {
+ // NOTE: do not reorder these values! clex_ident_errstr() relies on them
+ CLEX_IDENT_OK,
+ CLEX_IDENT_BADUCNHEX4, /* invalid \uXXXX UCN sequence */
+ CLEX_IDENT_BADCODEPOINT, /* invalid Unicode codepoint in UCN */
+ CLEX_IDENT_BADUCNHEX6, /* invalid \UXXXXXX UCN sequence */
+ CLEX_IDENT_UNEXPBACKSLASH /* unexpected backslash in identifier */
+};
+
+/*
+ * Double-checks that a TOK_IDENT token is syntactically valid (as the initial
+ * lexer pass does not account for bad UCN sequences or misplaced backslashes).
+ *
+ * Determines the extent of the token, i.e. how many bytes it occupies in the
+ * source, along with the length of the evaluated name, i.e. how many bytes are
+ * required to store the evaluated name as UTF-8 once line-continuations and
+ * UCNs are taken into account.
+ *
+ * This function may only be called on tokens of type TOK_IDENT; anything else
+ * produces undefined behaviour.
+ */
+struct clex_ident_validate_ret {
+ enum clex_ident_validate_err err; /* zero on success, nonzero on error */
+ // wonky union interweaving here because C++ doesn't do anonymous structs :(
+ union {
+ int ext; /* if !err: extent (source length) */
+ unsigned int err_off; /* if err: position of error in file */
+ };
+ int len; /* if !err: length of evaluated name (as utf-8). else undefined! */
+} clex_ident_validate(const struct clex *c, const char *_clex_restrict buf,
+ unsigned int idx);
+
+/*
+ * Evaluates the name of an identifier, taking into account line continuations
+ * and UCNs. The identifier has to have first been successfully validated using
+ * clex_ident_validate(). The buffer out must be large enough to hold the result
+ * which is indicated by the len member of the clex_ident_validate_ret struct.
+ */
+void clex_ident(const struct clex *c, const char *_clex_restrict buf,
+ unsigned int idx, char *_clex_restrict out);
+
+// NOTE: this is precalculated based on the implementation of clex_ident_errstr
+// and will need changed if the function changes!
+/*
+ * Determines the buffer size required to format an error message using
+ * clex_ident_errstr(), given the length of the filename.
+ *
+ * clex_ident_errstr() formats a string in the form "filename:line:col: message"
+ * so this macro can determine the worst-case memory requirement in advance.
+ *
+ * If a constant value is desired, e.g. for a stack buffer, simply use the known
+ * longest possible file path, for instance PATH_MAX on Unix-likes or MAX_PATH
+ * on Windows. The reason this is a macro is to allow it to produce an integer
+ * constant expression, so that a stack buffer can be used without creating a
+ * VLA.
+ */
+#define CLEX_IDENT_ERRSTR_MEMREQ(namelen) ((namelen) + 24 + 33)
+
+/*
+ * Formats an error message in the form "filename:line:col: message", given an
+ * error result from clex_ident_validate(). The resulting string is written to
+ * the buffer pointed to by out.
+ *
+ * out must point to a buffer of at least CLEX_IDENT_ERRSTR_MEMREQ(
+ * strlen(filename)) bytes. The resulting string is null-terminated for
+ * convenience, and the length of the string excluding the null terminator is
+ * returned.
+ *
+ * err has to be an error value from the clex_ident_validate_err enum; calling
+ * this function with a success result will produce undefined behaviour.
+ */
+int clex_ident_errstr(char *_clex_restrict out, const char *_clex_restrict buf,
+ enum clex_ident_validate_err err, unsigned int err_off,
+ const char *_clex_restrict filename);
+
+#undef _clex_unreachable
+#undef _clex_restrict
+#undef _clex_bool
+
+#endif
+
+// vi: sw=4 ts=4 noet tw=80 cc=80