summaryrefslogtreecommitdiff
path: root/src/chunklets/clex.c
diff options
context:
space:
mode:
Diffstat (limited to 'src/chunklets/clex.c')
-rw-r--r--src/chunklets/clex.c1664
1 files changed, 1664 insertions, 0 deletions
diff --git a/src/chunklets/clex.c b/src/chunklets/clex.c
new file mode 100644
index 0000000..d20986c
--- /dev/null
+++ b/src/chunklets/clex.c
@@ -0,0 +1,1664 @@
+/*
+ * Copyright © Michael Smith <mikesmiffy128@gmail.com>
+ *
+ * Permission to use, copy, modify, and/or distribute this software for any
+ * purpose with or without fee is hereby granted, provided that the above
+ * copyright notice and this permission notice appear in all copies.
+ *
+ * THE SOFTWARE IS PROVIDED “AS IS” AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
+ * REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
+ * AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
+ * INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
+ * LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
+ * OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
+ * PERFORMANCE OF THIS SOFTWARE.
+ */
+
+/* WARNING: this library is incomplete and still subject to change! */
+
+// NYI:
+// - numeric literal parsing
+// - string/char literal parsing
+// - more performance measurements
+// - if needed: more optimisation (target: >500MB/s give or take)
+// - security/robustness tests (e.g. fuzzing)
+// - UTF-8 aware column numbers (assuming this is what most tools do/expect?)
+
+#ifdef __cplusplus
+#error This file should not be compiled as C++. It relies on numerous C-only \
+features.
+#endif
+
+#if defined(_MSC_VER) && !defined(__clang__)
+#if !defined(_MSVC_TRADITIONAL)
+#error This version of MSVC is too old: upgrade to 2019 or newer and use \
+`-std:c17`, or better yet use Clang.
+#elif _MSVC_TRADITIONAL
+#error This file must be compiled using `-std:c17` when using MSVC.
+#endif
+#endif
+
+#ifndef _WIN32
+_Static_assert(
+ (unsigned char)-1 == 255 &&
+ sizeof(short) == 2 &&
+ sizeof(int) == 4 &&
+ sizeof(long long) == 8 &&
+ sizeof(void *) == 4 || sizeof(void *) == 8 &&
+ sizeof(long) == sizeof(void *),
+ "this code is only designed for relatively sane environments, plus Windows"
+);
+#endif
+
+#ifndef __clang__ // see `#define copy` further down
+#include <string.h>
+#endif
+
+#include "clex.h"
+
+#if defined(__GNUC__) || defined(__clang__)
+#define if_cold(x) if (__builtin_expect(!!(x), 0))
+#else
+#define if_cold(x) if (x)
+#endif
+
+#if defined(__GNUC__) || defined(__clang__)
+#define cold __attribute__((cold, noinline))
+#elif defined(_MSC_VER)
+#define cold __declspec(noinline)
+#else
+#define cold
+#endif
+
+// duping this from clex.h because it's undef'd there to pollute a little less
+#if defined(__GNUC__) || defined(__clang__)
+#define unreachable __builtin_unreachable()
+#elif defined(_MSC_VER)
+#define unreachable __assume(0)
+#else
+static inline _Noreturn void _clex_invoke_ub(void) {}
+#define unreachable (_clex_invoke_ub())
+#endif
+
+// note: a couple of these have been moved from seemingly more neat/logical
+// positions in order to make statetoks below more compact. classes and state
+// transitions use designated initialisers so this stuff can be moved around
+// without ruining the state transitions. the visual grouping here shows how
+// statetoks[] will be indexed, using halved state values.
+#define STATES(S, s) \
+ S(BOL) \
+ s(OP1) s(WS) \
+ s(OP2) s(WS_BS) \
+ S(LITPFX) \
+ S(uPFX) \
+ S(WORD) \
+ S(NUM) \
+ S(MIGHTEQ) \
+ S(MINUS) \
+ S(PLUS) \
+ S(AND) \
+ S(OR) \
+ S(HASH) \
+ S(COLON) \
+ S(FS) \
+ S(LT) \
+ S(GT) \
+ S(DOT) \
+ S(SHIFT) \
+ S(LNCOMM) \
+ s(BLKCOMM) s(BLKSTAR) \
+ s(OPEQ) s(BLKSTAR_BS) \
+ s(SHIFTEQ) s(STRLIT_BS_WS) /* < STUPID case, needed for gcc/clang compat! */ \
+ s(STRLIT) s(STRLIT_BS) \
+ s(CHRLIT) s(CHRLIT_BS) \
+ s(ERROR) /* special case for bad syntax */ \
+ s(CHRLIT_BS_WS) /* also stupid case */
+
+#define ENUMSTATE(name) STATE_##name,
+#define ENUMSTATEBS(name) STATE_##name, STATE_##name##_BS,
+#define ENUMTRANS(name) STATE_TRANS_##name = STATE_##name << 1,
+#define ENUMTRANSBS(name) \
+ ENUMTRANS(name) STATE_TRANS_##name##_BS = STATE_##name##_BS << 1,
+enum {
+ STATES(ENUMSTATEBS, ENUMSTATE)
+ NSTATES,
+ STATES(ENUMTRANSBS, ENUMTRANS)
+};
+#undef SHIFTENUMBS
+#undef SHIFTENUM
+#undef ENUMSTATEBS
+#undef ENUMSTATE
+
+#define CLASSES(X) \
+ X(OTHER) /* includes (most) alpha, underscores, anything unicode */ \
+ X(LU) /* L and U (lit prefixes) */ \
+ X(u) /* u prefix (could become u8) */ \
+ X(DIGIT) /* excluding 8 */ \
+ X(8) \
+ /* ^^ those must be before symbols */ \
+ X(OP1) /* 1 char ops/symbols */ \
+ X(MIGHTEQ) /* chars that could be followed by = */ \
+ X(MINUS) \
+ X(PLUS) \
+ X(EQ) \
+ X(LT) \
+ X(GT) \
+ X(AND) \
+ X(OR) \
+ X(DOT) \
+ X(STAR) \
+ X(HASH) \
+ X(COLON) \
+ X(FS) \
+ X(BS) \
+ X(SQ) \
+ X(DQ) \
+ X(SP) \
+ X(NL)
+
+// premultiply class values for transition table.
+#define CLASSBASEVALUE(name) CLASSBASEVAL_##name,
+enum { CLASSES(CLASSBASEVALUE) NCLASSES };
+#undef CLASSBASEVALUE
+#define CLASSREALVALUE(name) CLASS_##name = CLASSBASEVAL_##name * NSTATES,
+enum { CLASSES(CLASSREALVALUE) };
+#undef CLASSREALVALUE
+
+#define CLS(cls, c) [c] = CLASS_##cls,
+#define CLS2(cls, c1, c2) CLS(cls, c1) CLS(cls, c2)
+#define CLS3(cls, c1, ...) CLS(cls, c1) CLS2(cls, __VA_ARGS__)
+#define CLS4(cls, c1, ...) CLS(cls, c1) CLS3(cls, __VA_ARGS__)
+#define CLS5(cls, c1, ...) CLS(cls, c1) CLS4(cls, __VA_ARGS__)
+#define CLS6(cls, c1, ...) CLS(cls, c1) CLS5(cls, __VA_ARGS__)
+#define CLS7(cls, c1, ...) CLS(cls, c1) CLS6(cls, __VA_ARGS__)
+#define CLS8(cls, c1, ...) CLS(cls, c1) CLS7(cls, __VA_ARGS__)
+#define CLS9(cls, c1, ...) CLS(cls, c1) CLS8(cls, __VA_ARGS__)
+#define CLS10(cls, c1, ...) CLS(cls, c1) CLS9(cls, __VA_ARGS__)
+
+static const short classes[256] = {
+ /*CLSX(OTHER, everything, else)*/
+ CLS2(LU, 'L', 'U')
+ CLS(u, 'u')
+ CLS9(DIGIT, '0', '1', '2', '3', '4', '5', '6', '7', '9')
+ CLS(8, '8')
+ CLS10(OP1, '(', ')', '[', ']', '{', '}', ',', '?', ';', '~')
+ CLS3(MIGHTEQ, '!', '^', '%')
+ CLS(PLUS, '+')
+ CLS(MINUS, '-')
+ CLS(EQ, '=')
+ CLS(LT, '<')
+ CLS(GT, '>')
+ CLS(AND, '&')
+ CLS(OR, '|')
+ CLS(DOT, '.')
+ CLS(STAR, '*')
+ CLS(HASH, '#')
+ CLS(COLON, ':')
+ CLS(FS, '/')
+ CLS(BS, '\\')
+ CLS(SQ, '\'')
+ CLS(DQ, '"')
+ CLS3(SP, ' ', '\t', '\r')
+ CLS(NL, '\n')
+};
+
+#define FWD 1 // advance token index (i.e. create new token)
+#define FIXPOS 128 // in \u cases, token starts before the u. fixup file offset
+
+// abstracting transition table definition to allow fiddling with the layout.
+#define TRANS(from, class, to) \
+ [CLASS_##class + STATE_##from] = STATE_TRANS_##to,
+#define ST(s, s1, s2, s3, s4, s5, s6, s7, s8, s9, s10, s11, s12, \
+ s13, s14, s15, s16, s17, s18, s19, s20, s21, s22, s23, s24) \
+ TRANS(s, OTHER, s1) \
+ TRANS(s, LU, s2) \
+ TRANS(s, u, s3) \
+ TRANS(s, DIGIT, s4) \
+ TRANS(s, 8, s5) \
+ TRANS(s, OP1, s6) \
+ TRANS(s, MIGHTEQ, s7) \
+ TRANS(s, PLUS, s8) \
+ TRANS(s, MINUS, s9) \
+ TRANS(s, EQ, s10) \
+ TRANS(s, LT, s11) \
+ TRANS(s, GT, s12) \
+ TRANS(s, AND, s13) \
+ TRANS(s, OR, s14) \
+ TRANS(s, DOT, s15) \
+ TRANS(s, STAR, s16) \
+ TRANS(s, HASH, s17) \
+ TRANS(s, COLON, s18) \
+ TRANS(s, FS, s19) \
+ TRANS(s, BS, s20) \
+ TRANS(s, SQ, s21) \
+ TRANS(s, DQ, s22) \
+ TRANS(s, SP, s23) \
+ TRANS(s, NL, s24)
+
+#define ST_BS(s) \
+ ST(s##_BS, \
+ /* OTHER => */ ERROR | FWD, \
+ /* LU => */ WORD | FWD | FIXPOS, /* *could* be UCN (unlikely) */ \
+ /* u => */ WORD | FWD | FIXPOS, /* same */ \
+ /* DIGIT => */ ERROR | FWD, \
+ /* 8 => */ ERROR | FWD, \
+ /* OP1 => */ ERROR | FWD, \
+ /* MIGHTEQ => */ ERROR | FWD, \
+ /* PLUS => */ ERROR | FWD, \
+ /* MINUS => */ ERROR | FWD, \
+ /* EQ => */ ERROR | FWD, \
+ /* LT => */ ERROR | FWD, \
+ /* GT => */ ERROR | FWD, \
+ /* AND => */ ERROR | FWD, \
+ /* OR => */ ERROR | FWD, \
+ /* DOT => */ ERROR | FWD, \
+ /* STAR => */ ERROR | FWD, \
+ /* HASH => */ ERROR | FWD, \
+ /* COLON => */ ERROR | FWD, \
+ /* FS => */ ERROR | FWD, \
+ /* BS => */ ERROR | FWD, \
+ /* SQ => */ ERROR | FWD, \
+ /* DQ => */ ERROR | FWD, \
+ /* SP => */ s##_BS, /* nonstandard crap done by GCC and Clang */ \
+ /* NL => */ s \
+ )
+
+#define ST_TERM(s) /* "terminal" state (i.e. not mid-token) */ \
+ ST(s, \
+ /* OTHER => */ WORD | FWD, \
+ /* LU => */ LITPFX | FWD, \
+ /* u => */ uPFX | FWD, \
+ /* DIGIT => */ NUM | FWD, \
+ /* 8 => */ NUM | FWD, \
+ /* OP1 => */ OP1 | FWD, \
+ /* MIGHTEQ => */ MIGHTEQ | FWD, \
+ /* PLUS => */ PLUS | FWD, \
+ /* MINUS => */ MINUS | FWD, \
+ /* EQ => */ MIGHTEQ | FWD, \
+ /* LT => */ LT | FWD, \
+ /* GT => */ GT | FWD, \
+ /* AND => */ AND | FWD, \
+ /* OR => */ OR | FWD, \
+ /* DOT => */ DOT | FWD, \
+ /* STAR => */ MIGHTEQ | FWD, \
+ /* HASH => */ HASH | FWD, \
+ /* COLON => */ COLON | FWD, \
+ /* FS => */ FS | FWD, \
+ /* BS => */ WS_BS, \
+ /* SQ => */ CHRLIT | FWD, \
+ /* DQ => */ STRLIT | FWD, \
+ /* SP => */ s, \
+ /* NL => */ BOL | FWD \
+ )
+
+// VERY long table {{{
+static const unsigned char trans[NSTATES * NCLASSES] = {
+ ST(BOL, // note: same as ST_TERM but without FWD so we don't repeat NLs
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ BOL_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ BOL,
+ /* NL => */ BOL
+ )
+ ST(BOL_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD | FWD | FIXPOS, // maybe UCN
+ /* u => */ WORD | FWD | FIXPOS, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS | FWD | FIXPOS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ BOL_BS,
+ /* NL => */ BOL
+ )
+ ST_TERM(WS)
+ ST(WS_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD | FWD | FIXPOS, // maybe UCN
+ /* u => */ WORD | FWD | FIXPOS, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS | FWD | FIXPOS, // maybe UCN split over liens
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ WS_BS,
+ /* NL => */ WS
+ )
+ ST(uPFX,
+ /* OTHER => */ WORD,
+ /* LU => */ WORD,
+ /* u => */ WORD,
+ /* DIGIT => */ WORD,
+ /* 8 => */ LITPFX,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ uPFX_BS,
+ /* SQ => */ CHRLIT,
+ /* DQ => */ STRLIT,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(uPFX_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD, // maybe UCN
+ /* u => */ WORD, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ uPFX_BS,
+ /* NL => */ uPFX
+ )
+ ST(LITPFX,
+ /* OTHER => */ WORD,
+ /* LU => */ WORD,
+ /* u => */ WORD,
+ /* DIGIT => */ WORD,
+ /* 8 => */ WORD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ LITPFX_BS,
+ /* SQ => */ CHRLIT,
+ /* DQ => */ STRLIT,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(LITPFX_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD, // maybe UCN
+ /* u => */ WORD, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ LITPFX_BS,
+ /* NL => */ LITPFX
+ )
+ ST(WORD,
+ /* OTHER => */ WORD,
+ /* LU => */ WORD,
+ /* u => */ WORD,
+ /* DIGIT => */ WORD,
+ /* 8 => */ WORD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ WORD_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(WORD_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ WORD, // maybe UCN
+ /* u => */ WORD, // "
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ WORD_BS, // maybe UCN split over lines
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ WORD_BS,
+ /* NL => */ WORD
+ )
+ ST(NUM,
+ /* OTHER => */ NUM,
+ /* LU => */ NUM,
+ /* u => */ NUM,
+ /* DIGIT => */ NUM,
+ /* 8 => */ NUM,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ NUM,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ NUM_BS,
+ /* SQ => */ NUM, // C23
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST(NUM_BS,
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ ERROR | FWD, // UCN mid-number makes no sense
+ /* u => */ ERROR | FWD,
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ ERROR | FWD,
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ NUM_BS,
+ /* NL => */ NUM
+ )
+ ST(MIGHTEQ,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ MIGHTEQ_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(MIGHTEQ)
+ ST(MINUS,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ OP2,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ OP2,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ MINUS_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(MINUS)
+ ST(PLUS,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ DOT | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ OP2,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ MIGHTEQ | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ PLUS_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(PLUS)
+ ST(AND,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ OP2,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ AND_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(AND)
+ ST(OR,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OP2,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ OR_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(OR)
+ ST(HASH,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ OP2,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ HASH_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(HASH)
+ ST(COLON,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ OP2,
+ /* FS => */ FS | FWD,
+ /* BS => */ COLON_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(COLON)
+ ST(FS,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ BLKCOMM,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ LNCOMM,
+ /* BS => */ FS_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(FS)
+ ST(LT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ SHIFT,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ LT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(LT)
+ ST(GT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ OPEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ SHIFT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ GT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(GT)
+ ST(DOT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM,
+ /* 8 => */ NUM,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ MIGHTEQ | FWD,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ DOT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(DOT)
+ ST(SHIFT,
+ /* OTHER => */ WORD | FWD,
+ /* LU => */ LITPFX | FWD,
+ /* u => */ uPFX | FWD,
+ /* DIGIT => */ NUM | FWD,
+ /* 8 => */ NUM | FWD,
+ /* OP1 => */ OP1 | FWD,
+ /* MIGHTEQ => */ MIGHTEQ | FWD,
+ /* PLUS => */ PLUS | FWD,
+ /* MINUS => */ MINUS | FWD,
+ /* EQ => */ SHIFTEQ,
+ /* LT => */ LT | FWD,
+ /* GT => */ GT | FWD,
+ /* AND => */ AND | FWD,
+ /* OR => */ OR | FWD,
+ /* DOT => */ DOT | FWD,
+ /* STAR => */ MIGHTEQ | FWD,
+ /* HASH => */ HASH | FWD,
+ /* COLON => */ COLON | FWD,
+ /* FS => */ FS | FWD,
+ /* BS => */ SHIFT_BS,
+ /* SQ => */ CHRLIT | FWD,
+ /* DQ => */ STRLIT | FWD,
+ /* SP => */ WS,
+ /* NL => */ BOL | FWD
+ )
+ ST_BS(SHIFT)
+ ST(LNCOMM,
+ /* OTHER => */ LNCOMM,
+ /* LU => */ LNCOMM,
+ /* u => */ LNCOMM,
+ /* DIGIT => */ LNCOMM,
+ /* 8 => */ LNCOMM,
+ /* OP1 => */ LNCOMM,
+ /* MIGHTEQ => */ LNCOMM,
+ /* PLUS => */ LNCOMM,
+ /* MINUS => */ LNCOMM,
+ /* EQ => */ LNCOMM,
+ /* LT => */ LNCOMM,
+ /* GT => */ LNCOMM,
+ /* AND => */ LNCOMM,
+ /* OR => */ LNCOMM,
+ /* DOT => */ LNCOMM,
+ /* STAR => */ LNCOMM,
+ /* HASH => */ LNCOMM,
+ /* COLON => */ LNCOMM,
+ /* FS => */ LNCOMM,
+ /* BS => */ LNCOMM_BS,
+ /* SQ => */ LNCOMM,
+ /* DQ => */ LNCOMM,
+ /* SP => */ LNCOMM,
+ /* NL => */ BOL | FWD
+ )
+ ST(LNCOMM_BS,
+ /* OTHER => */ LNCOMM,
+ /* LU => */ LNCOMM,
+ /* u => */ LNCOMM,
+ /* DIGIT => */ LNCOMM,
+ /* 8 => */ LNCOMM,
+ /* OP1 => */ LNCOMM,
+ /* MIGHTEQ => */ LNCOMM,
+ /* PLUS => */ LNCOMM,
+ /* MINUS => */ LNCOMM,
+ /* EQ => */ LNCOMM,
+ /* LT => */ LNCOMM,
+ /* GT => */ LNCOMM,
+ /* AND => */ LNCOMM,
+ /* OR => */ LNCOMM,
+ /* DOT => */ LNCOMM,
+ /* STAR => */ LNCOMM,
+ /* HASH => */ LNCOMM,
+ /* COLON => */ LNCOMM,
+ /* FS => */ LNCOMM,
+ /* BS => */ LNCOMM_BS,
+ /* SQ => */ LNCOMM,
+ /* DQ => */ LNCOMM,
+ /* SP => */ LNCOMM_BS,
+ /* NL => */ LNCOMM
+ )
+ ST(BLKCOMM,
+ /* OTHER => */ BLKCOMM,
+ /* LU => */ BLKCOMM,
+ /* u => */ BLKCOMM,
+ /* DIGIT => */ BLKCOMM,
+ /* 8 => */ BLKCOMM,
+ /* OP1 => */ BLKCOMM,
+ /* MIGHTEQ => */ BLKCOMM,
+ /* PLUS => */ BLKCOMM,
+ /* MINUS => */ BLKCOMM,
+ /* EQ => */ BLKCOMM,
+ /* LT => */ BLKCOMM,
+ /* GT => */ BLKCOMM,
+ /* AND => */ BLKCOMM,
+ /* OR => */ BLKCOMM,
+ /* DOT => */ BLKCOMM,
+ /* STAR => */ BLKSTAR,
+ /* HASH => */ BLKCOMM,
+ /* COLON => */ BLKCOMM,
+ /* FS => */ BLKCOMM,
+ /* BS => */ BLKCOMM,
+ /* SQ => */ BLKCOMM,
+ /* DQ => */ BLKCOMM,
+ /* SP => */ BLKCOMM,
+ /* NL => */ BLKCOMM
+ )
+ ST(BLKSTAR,
+ /* OTHER => */ BLKCOMM,
+ /* LU => */ BLKCOMM,
+ /* u => */ BLKCOMM,
+ /* DIGIT => */ BLKCOMM,
+ /* 8 => */ BLKCOMM,
+ /* OP1 => */ BLKCOMM,
+ /* MIGHTEQ => */ BLKCOMM,
+ /* PLUS => */ BLKCOMM,
+ /* MINUS => */ BLKCOMM,
+ /* EQ => */ BLKCOMM,
+ /* LT => */ BLKCOMM,
+ /* GT => */ BLKCOMM,
+ /* AND => */ BLKCOMM,
+ /* OR => */ BLKCOMM,
+ /* DOT => */ BLKCOMM,
+ /* STAR => */ BLKSTAR,
+ /* HASH => */ BLKCOMM,
+ /* COLON => */ BLKCOMM,
+ /* FS => */ WS,
+ /* BS => */ BLKSTAR_BS,
+ /* SQ => */ BLKCOMM,
+ /* DQ => */ BLKCOMM,
+ /* SP => */ BLKCOMM,
+ /* NL => */ BLKCOMM
+ )
+ ST(BLKSTAR_BS,
+ /* OTHER => */ BLKCOMM,
+ /* LU => */ BLKCOMM,
+ /* u => */ BLKCOMM,
+ /* DIGIT => */ BLKCOMM,
+ /* 8 => */ BLKCOMM,
+ /* OP1 => */ BLKCOMM,
+ /* MIGHTEQ => */ BLKCOMM,
+ /* PLUS => */ BLKCOMM,
+ /* MINUS => */ BLKCOMM,
+ /* EQ => */ BLKCOMM,
+ /* LT => */ BLKCOMM,
+ /* GT => */ BLKCOMM,
+ /* AND => */ BLKCOMM,
+ /* OR => */ BLKCOMM,
+ /* DOT => */ BLKCOMM,
+ /* STAR => */ BLKSTAR,
+ /* HASH => */ BLKCOMM,
+ /* COLON => */ BLKCOMM,
+ /* FS => */ WS,
+ /* BS => */ BLKCOMM,
+ /* SQ => */ BLKCOMM,
+ /* DQ => */ BLKCOMM,
+ /* SP => */ BLKSTAR_BS,
+ /* NL => */ BLKSTAR
+ )
+ ST_TERM(OP1)
+ ST_TERM(OP2)
+ ST_TERM(OPEQ)
+ ST_TERM(SHIFTEQ)
+ ST(STRLIT,
+ /* OTHER => */ STRLIT,
+ /* LU => */ STRLIT,
+ /* u => */ STRLIT,
+ /* DIGIT => */ STRLIT,
+ /* 8 => */ STRLIT,
+ /* OP1 => */ STRLIT,
+ /* MIGHTEQ => */ STRLIT,
+ /* PLUS => */ STRLIT,
+ /* MINUS => */ STRLIT,
+ /* EQ => */ STRLIT,
+ /* LT => */ STRLIT,
+ /* GT => */ STRLIT,
+ /* AND => */ STRLIT,
+ /* OR => */ STRLIT,
+ /* DOT => */ STRLIT,
+ /* STAR => */ STRLIT,
+ /* HASH => */ STRLIT,
+ /* COLON => */ STRLIT,
+ /* FS => */ STRLIT,
+ /* BS => */ STRLIT_BS,
+ /* SQ => */ STRLIT,
+ /* DQ => */ WS,
+ /* SP => */ STRLIT,
+ /* NL => */ ERROR | FWD
+ )
+ ST(STRLIT_BS,
+ /* OTHER => */ STRLIT,
+ /* LU => */ STRLIT,
+ /* u => */ STRLIT,
+ /* DIGIT => */ STRLIT,
+ /* 8 => */ STRLIT,
+ /* OP1 => */ STRLIT,
+ /* MIGHTEQ => */ STRLIT,
+ /* PLUS => */ STRLIT,
+ /* MINUS => */ STRLIT,
+ /* EQ => */ STRLIT,
+ /* LT => */ STRLIT,
+ /* GT => */ STRLIT,
+ /* AND => */ STRLIT,
+ /* OR => */ STRLIT,
+ /* DOT => */ STRLIT,
+ /* STAR => */ STRLIT,
+ /* HASH => */ STRLIT,
+ /* COLON => */ STRLIT,
+ /* FS => */ STRLIT,
+ /* BS => */ STRLIT,
+ /* SQ => */ STRLIT,
+ /* DQ => */ STRLIT,
+ /* SP => */ STRLIT_BS_WS,
+ /* NL => */ STRLIT
+ )
+ ST(STRLIT_BS_WS, // ugh this really does suck
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ ERROR | FWD,
+ /* u => */ ERROR | FWD,
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ ERROR | FWD,
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ STRLIT_BS_WS,
+ /* NL => */ STRLIT
+ )
+ ST(CHRLIT,
+ /* OTHER => */ CHRLIT,
+ /* LU => */ CHRLIT,
+ /* u => */ CHRLIT,
+ /* DIGIT => */ CHRLIT,
+ /* 8 => */ CHRLIT,
+ /* OP1 => */ CHRLIT,
+ /* MIGHTEQ => */ CHRLIT,
+ /* PLUS => */ CHRLIT,
+ /* MINUS => */ CHRLIT,
+ /* EQ => */ CHRLIT,
+ /* LT => */ CHRLIT,
+ /* GT => */ CHRLIT,
+ /* AND => */ CHRLIT,
+ /* OR => */ CHRLIT,
+ /* DOT => */ CHRLIT,
+ /* STAR => */ CHRLIT,
+ /* HASH => */ CHRLIT,
+ /* COLON => */ CHRLIT,
+ /* FS => */ CHRLIT,
+ /* BS => */ CHRLIT_BS,
+ /* SQ => */ WS,
+ /* DQ => */ CHRLIT,
+ /* SP => */ CHRLIT,
+ /* NL => */ ERROR | FWD
+ )
+ ST(CHRLIT_BS,
+ /* OTHER => */ CHRLIT,
+ /* LU => */ CHRLIT,
+ /* u => */ CHRLIT,
+ /* DIGIT => */ CHRLIT,
+ /* 8 => */ CHRLIT,
+ /* OP1 => */ CHRLIT,
+ /* MIGHTEQ => */ CHRLIT,
+ /* PLUS => */ CHRLIT,
+ /* MINUS => */ CHRLIT,
+ /* EQ => */ CHRLIT,
+ /* LT => */ CHRLIT,
+ /* GT => */ CHRLIT,
+ /* AND => */ CHRLIT,
+ /* OR => */ CHRLIT,
+ /* DOT => */ CHRLIT,
+ /* STAR => */ CHRLIT,
+ /* HASH => */ CHRLIT,
+ /* COLON => */ CHRLIT,
+ /* FS => */ CHRLIT,
+ /* BS => */ CHRLIT,
+ /* SQ => */ CHRLIT,
+ /* DQ => */ CHRLIT,
+ /* SP => */ CHRLIT_BS_WS,
+ /* NL => */ CHRLIT
+ )
+ ST(CHRLIT_BS_WS, // ugh this really does also suck
+ /* OTHER => */ ERROR | FWD,
+ /* LU => */ ERROR | FWD,
+ /* u => */ ERROR | FWD,
+ /* DIGIT => */ ERROR | FWD,
+ /* 8 => */ ERROR | FWD,
+ /* OP1 => */ ERROR | FWD,
+ /* MIGHTEQ => */ ERROR | FWD,
+ /* PLUS => */ ERROR | FWD,
+ /* MINUS => */ ERROR | FWD,
+ /* EQ => */ ERROR | FWD,
+ /* LT => */ ERROR | FWD,
+ /* GT => */ ERROR | FWD,
+ /* AND => */ ERROR | FWD,
+ /* OR => */ ERROR | FWD,
+ /* DOT => */ ERROR | FWD,
+ /* STAR => */ ERROR | FWD,
+ /* HASH => */ ERROR | FWD,
+ /* COLON => */ ERROR | FWD,
+ /* FS => */ ERROR | FWD,
+ /* BS => */ ERROR | FWD,
+ /* SQ => */ ERROR | FWD,
+ /* DQ => */ ERROR | FWD,
+ /* SP => */ CHRLIT_BS_WS,
+ /* NL => */ CHRLIT
+ )
+ ST(ERROR, // dead end, caught outside the branchless main loop
+ /* OTHER => */ ERROR,
+ /* LU => */ ERROR,
+ /* u => */ ERROR,
+ /* DIGIT => */ ERROR,
+ /* 8 => */ ERROR,
+ /* OP1 => */ ERROR,
+ /* MIGHTEQ => */ ERROR,
+ /* PLUS => */ ERROR,
+ /* MINUS => */ ERROR,
+ /* EQ => */ ERROR,
+ /* LT => */ ERROR,
+ /* GT => */ ERROR,
+ /* AND => */ ERROR,
+ /* OR => */ ERROR,
+ /* DOT => */ ERROR,
+ /* STAR => */ ERROR,
+ /* HASH => */ ERROR,
+ /* COLON => */ ERROR,
+ /* FS => */ ERROR,
+ /* BS => */ ERROR,
+ /* SQ => */ ERROR,
+ /* DQ => */ ERROR,
+ /* SP => */ ERROR,
+ /* NL => */ ERROR
+ )
+};
+// }}}
+
+static const unsigned char statetoks[] = {
+ // comments here cover corresponding tokens. stuff in parentheses doesn't
+ // matter (could have any value)
+ CLEX_TOK_EOL, // BOL, (BOL_BS)
+ CLEX_TOK_OP1, // OP1, (WS)
+ CLEX_TOK_OP2, // OP2, (WS_BS)
+ CLEX_TOK_IDENT, // LITPFX, (LITPFX_BS)
+ CLEX_TOK_IDENT, // uPFX, (uPFX_BS)
+ CLEX_TOK_IDENT, // WORD, (WORD_BS)
+ CLEX_TOK_NUM, // NUM, (NUM_BS)
+ CLEX_TOK_OP1, // MIGHTEQ, (MIGHTEQ_BS)
+ CLEX_TOK_OP1, // MINUS, (MINUS_BS)
+ CLEX_TOK_OP1, // PLUS, (PLUS_BS)
+ CLEX_TOK_OP1, // AND, (AND_BS)
+ CLEX_TOK_OP1, // OR, (OR_BS)
+ CLEX_TOK_OP1, // HASH, (HASH_BS)
+ CLEX_TOK_OP1, // COLON, (COLON_BS)
+ CLEX_TOK_OP1, // FS, (FS_BS)
+ CLEX_TOK_OP1, // LT, (LT_BS)
+ CLEX_TOK_OP1, // GT, (GT_BS)
+ CLEX_TOK_OP1, // DOT, (DOT_BS)
+ CLEX_TOK_SHIFT, // SHIFT, (SHIFT_BS)
+ CLEX_TOK_LNCOMM, // LNCOMM, (LNCOMM_BS)
+ CLEX_TOK_BLKCOMM, // BLKCOMM, (BLKSTAR)
+ CLEX_TOK_OPEQ, // OPEQ, (BLKSTAR_BS)
+ CLEX_TOK_SHIFTEQ, // SHIFTEQ, (STRLIT_BS_WS)
+ CLEX_TOK_STR, // STRLIT, (STRLIT_BS)
+ CLEX_TOK_CHAR, // CHRLIT, (CHRLIT_BS)
+ 200 // (ERROR), (CHRLIT_BS_WS) // completely arbitrary value!
+};
+
+// Uncomment and LSP-inspect this for an idea of how much space is used:
+//enum { TABLESPACE = sizeof(classes) + sizeof(trans) + sizeof(statetoks) };
+// => 1786 bytes.
+
+// somewhat inefficient number formatter - size matters more than speed here.
+// also assumes there's extra space in the buffer, because there always is.
+static cold int err_fmtnum(char *out, unsigned int n) {
+ int i = 0, j = 10;
+ do {
+ // explicitly optimised division by 10 to make sure we NEVER have a
+ // divide instruction because integer division is evil.
+ unsigned int div10 = n * 3435973837ull >> 35;
+ // apparently doing * 10 here emits a more efficient lea-based shift-add
+ // thingy than attempting to shift-add by hand, at least with clang.
+ unsigned int remainder = n - div10 * 10;
+ out[--j] = '0' + remainder;
+ n = div10;
+ } while (n);
+ do out[i++] = out[j++]; while (j < 10);
+ return i;
+}
+
+static cold struct linecol {
+ unsigned int ln, col;
+} getlinecol(const char *buf, unsigned int off) {
+ // we don't store line and col for every token as it wastes a lot of memory.
+ // instead we can simply re-scan and count the newlines.
+ // FIXME: grapheme widths for columns? horrendous, but technically correct!
+ // if not that, then probably at least count utf8 codepoints.
+ unsigned int ln = 1, col = 0;
+ for (unsigned int i = 0;; ++i, ++col) {
+ if (buf[i] == '\n') { ++ln; col = 0; }
+ if (i == off) break;
+ }
+ return (struct linecol) {ln, col};
+}
+
+static cold char *err_putprefix(char *restrict out, const char *f,
+ const char *restrict buf, unsigned int off) {
+ struct linecol lc = getlinecol(buf, off);
+ // "filename:line:col: "
+ while (*f) *out++ = *f++;
+ *out++ = ':';
+ out += err_fmtnum(out, lc.ln);
+ *out++ = ':';
+ out += err_fmtnum(out, lc.col);
+ *out++ = ':';
+ *out++ = ' ';
+ return out;
+}
+
+// if we have clang we can avoid even doing a library call here. otherwise we
+// still only depend on memcpy which is pretty reasonable. we could also in
+// theory do some manual word-wise copying of the strings below, but eww.
+#ifdef __clang__
+#define copy __builtin_memcpy_inline
+#else
+#define copy memcpy
+#endif
+
+static cold void err(struct clex *c, const char *p, unsigned int off, int state,
+ int prevtok, const char *f) {
+ if (state == STATE_ERROR) {
+ // all transitions to ERROR have FWD, so we get a dummy token at the
+ // point where the error was detected. if this token points at a
+ // newline, then we have an unterminated literal. if it does *not* point
+ // at a newline, we have a non-newline after a backslash. in the latter
+ // case, backtrack to the backslash to report it as unexpected. in the
+ // former case, move back one character to point to the end of the line.
+ if (p[off--] != '\n') while (p[off] != '\\') --off;
+ }
+ // format the message as f:ln:col: msg. pretty verbose due to not using
+ // stdio or any other convenient formatting library, but also pretty simple
+ // IMPORTANT: clex_memreq() must be recalculated after changing any of this!
+ char *msg = (char *)c->tokoffs, *msgp = msg;
+ c->err = msg;
+ msgp = err_putprefix(msgp, f, p, off);
+ static const char strblock[78] =
+ "unterminated " // + 0, len 13
+ "string" // +13, len 6
+ "character" // +19, len 9
+ " literal" // +28, len 8
+ "block comment" // +36, len 13
+ "expected" // +49, len 8
+ " line after" // +57, len 11
+ " backslash"; // +68, len 10
+ // minor code size trick: write a little too much first, then replace parts
+ copy(msgp, strblock, 19); // "unterminated string"
+ if (state == STATE_ERROR) {
+ // note: ERROR can only follow another token-producing state. prevtok
+ // could otherwise contain garbage from one of the dummy slots (see
+ // dotok() below) but in this scenario that will never be the case.
+ if (prevtok == CLEX_TOK_CHAR) {
+ // s/string /character literal/
+ copy(msgp + 13, strblock + 19, 9 + 8);
+ msgp += 13 + 9 + 8;
+ }
+ else if (prevtok == CLEX_TOK_STR) {
+ // append "literal" -> "
+ copy(msgp + 19, strblock + 29, 8);
+ msgp += 13 + 6 + 8;
+ }
+ else /* prevtok == 200 (backslash case above) */ {
+ // s/terminated/expected/ -> unexpected
+ copy(msgp + 2, strblock + 49, 8);
+ // append " backslash" -> "unexpected backslash"
+ copy(msgp + 10, strblock + 68, 10);
+ msgp += 10 + 10;
+ }
+ }
+ else if (state == STATE_BLKCOMM) {
+ // s/string/block comment/ -> "unterminated block comment"
+ copy(msgp + 13, strblock + 36, 13);
+ msgp += 13 + 13;
+ }
+ else /* state == STATE_BS */ {
+ // s/unterminated string/expected line after backslash/
+ copy(msgp, strblock + 49, 8 + 11 + 10);
+ msgp += 8 + 11 + 10;
+ }
+ *msgp = '\0';
+ c->errlen = msgp - msg;
+}
+
+static inline void dotok(struct clex *c, const unsigned char *restrict p,
+ unsigned int sz, const char *restrict f) {
+ unsigned char *toks = c->toks;
+ unsigned int *tokoffs = c->tokoffs;
+ int state = STATE_BOL;
+ unsigned int off = 0;
+ unsigned int ntoks = 1; // skip first slot, see below
+ for (; off != sz; ++off) {
+ unsigned char state_trans = trans[classes[p[off]] + state];
+ unsigned int lower7 = state_trans & 127;
+ unsigned int off_adj = off - (state_trans >> 7); // see FIXPOS
+ unsigned int fwdbit = state_trans & 1; // see FWD
+ // bit 2 (low state bit): set -> 1; unset -> -1.
+ // => only even-indexed states (which are terminal) update the type
+ // note: this means we have to waste the bottom two slots. oh well!
+ unsigned int idxmask = (state_trans | -3u) + 2;
+ // set the starting offset for the *next* token, that way if we're not
+ // starting a new one we don't clobber the existing position.
+ tokoffs[ntoks + 1] = off_adj;
+ // if FWD bit is set then we have a new token; at this point we increase
+ // ntoks meaning the first token will be at position 2. hence the waste.
+ ntoks += fwdbit; // advance if we have a new token
+ state = lower7 >> 1; // middle 6 bits are our new state
+ toks[ntoks & idxmask] = statetoks[lower7 >> 2];
+ }
+ // hide the wasted slots from the caller
+ c->ntoks = ntoks - 2; c->toks += 2; c->tokoffs += 2;
+ if_cold (state != STATE_BOL) {
+ if (f) {
+ err(c, (char *)p, tokoffs[ntoks], state, toks[ntoks - 1], f);
+ }
+ else {
+ c->err = "syntax error";
+ c->errlen = 12;
+ }
+ }
+}
+
+struct clex clex(const char *restrict buf, unsigned int sz,
+ void *restrict outmem, const char *restrict filename) {
+ struct clex c = {
+ .tokoffs = outmem,
+ .toks = (unsigned char *)(c.tokoffs + sz)
+ };
+ if_cold (sz != 0 && buf[sz - 1] != '\n') {
+ c.err = "input is not a valid text file (must end with newline)";
+ return c;
+ }
+ dotok(&c, (unsigned char *)buf, sz, filename);
+ return c;
+}
+
+static inline bool ishex(char c) {
+ char lower = c | 32;
+ return c >= '0' & c <= '9' | lower >= 'a' & lower <= 'f';
+}
+static inline int hexval(char c) {
+ unsigned char u = c;
+ return (u & 15) + (u >> 6) * 9; // assumes ishex(c)
+}
+
+static inline int utf8len_ucs2(int codepoint) { // assumes codepoint <= 0xFFFF
+ if (codepoint > 0x7FF) return 3;
+ // would be weird to use \u for ascii, so do this branch last
+ if (codepoint <= 0x7F) return 1;
+ return 2;
+}
+static inline int utf8len(int codepoint) { // assumes codepoint <= 0x10FFFF
+ if (codepoint <= 0xFFFF) { // most likely use for \U over \u
+ if (codepoint <= 0x7FF) {
+ if (codepoint <= 0x7F) return 1; // should still be least likely
+ return 2;
+ }
+ return 3;
+ }
+ return 4;
+}
+
+static inline bool isbadcodepoint_ucs2(int codepoint) { // assumes <= 0xFFFF
+ return codepoint >= 0xD800 && codepoint <= 0xDFFF || // surrogates
+ codepoint >= 0xFDD0 && codepoint <= 0xFDEF || // non-characters
+ (codepoint & 0x0FFE) == 0x0FFE; // more non-characters
+}
+static inline bool isbadcodepoint(int codepoint) {
+ return isbadcodepoint_ucs2(codepoint) ||
+ codepoint >= 0x110000 && codepoint <= 0x1FFFFF || // more non-chars
+ codepoint > 0x10FFFF; // max valid codepoint
+}
+
+static struct bsiter_ret { char c; const char *next; } bsiter(const char *p) {
+ char c;
+ // skip line continuations, but don't ignore backslashes mid-line. may need
+ // some backtracking/re-scanning - oh well. this is the slow path.
+ while ((c = *p) == '\\') {
+ const char *q = p;
+ while (classes[(unsigned char)*++q] == CLASS_SP);
+ if (*q != '\n') break; // backslash is significant, return it
+ p = q + 1; // skip newline and keep scanning forward
+ }
+ return (struct bsiter_ret){c, p + 1};
+}
+
+struct clex_ident_validate_ret clex_ident_validate(const struct clex *c,
+ const char *restrict buf, unsigned int idx) {
+ const char *p = buf + c->tokoffs[idx], *start = p;
+ // N.B. file always ends in an EOL, so no need for bounds check on this loop
+ struct clex_ident_validate_ret ret;
+ for (int namelen = 0;;) {
+ struct bsiter_ret iter = bsiter(p); char c = iter.c; p = iter.next;
+ // first 5 entries in CLASSES() are alphanumeric; if it's a symbol or
+ // space, we're past this token...
+ if (classes[(unsigned char)c] > 4 * NSTATES) {
+ if (c == '\\') { // ... unless it's a UCN, or bad syntax
+ const char *pbs = p;
+ iter = bsiter(p); c = iter.c; p = iter.next;
+ if (c == 'u') {
+ char hex[4];
+ iter = bsiter(p); hex[0] = iter.c;
+ const char *p0 = iter.next;
+ for (int i = 0;;) {
+ if_cold (!ishex(hex[i])) {
+ ret.err = CLEX_IDENT_BADUCNHEX4;
+ ret.err_off = p - buf;
+ return ret;
+ }
+ p = iter.next;
+ if (++i == 4) break;
+ iter = bsiter(p); hex[i] = iter.c;
+ }
+ int codepoint = hexval(hex[0]) << 12 | hexval(hex[1]) << 8 |
+ hexval(hex[2]) << 4 | hexval(hex[3]);
+ if_cold (isbadcodepoint_ucs2(codepoint)) {
+ ret.err = CLEX_IDENT_BADCODEPOINT;
+ ret.err_off = p0 - buf;
+ return ret;
+ }
+ namelen += utf8len_ucs2(codepoint);
+ }
+ else if (c == 'U') {
+ char hex[6];
+ iter = bsiter(p); hex[0] = iter.c;
+ const char *p0 = iter.next;
+ for (int i = 0;;) {
+ if_cold (!ishex(hex[i])) {
+ ret.err = CLEX_IDENT_BADUCNHEX6;
+ ret.err_off = p - buf;
+ return ret;
+ }
+ p = iter.next;
+ if (++i == 6) break;
+ iter = bsiter(p); hex[i] = iter.c;
+ }
+ int codepoint = hexval(hex[0]) << 20 | hexval(hex[1]) << 16 |
+ hexval(hex[2]) << 12 | hexval(hex[3] << 8) |
+ hexval(hex[4]) << 4 | hexval(hex[5]);
+ if_cold (isbadcodepoint(codepoint)) {
+ ret.err = CLEX_IDENT_BADCODEPOINT;
+ ret.err_off = p0 - buf;
+ return ret;
+ }
+ namelen += utf8len(codepoint);
+ }
+ else { // we skipped line conts already; must be bad syntax here
+ ret.err = CLEX_IDENT_UNEXPBACKSLASH;
+ ret.err_off = pbs - buf;
+ return ret;
+ }
+ }
+ else {
+ ret.err = 0; ret.ext = p - start; ret.len = namelen;
+ return ret;
+ }
+ }
+ else {
+ ++namelen;
+ }
+ }
+}
+
+cold int clex_ident_errstr(char *restrict out, const char *restrict buf,
+ enum clex_ident_validate_err err, unsigned int err_off,
+ const char *restrict filename) {
+ char *p = out;
+ static const char strblock[81] =
+ "illegal character in 3-digit UCN" // (becomes 4 or 6, see below)
+ "Unicode codepoint"
+ "identifier"
+ "unexpected end of line";
+ if (buf[err_off] == '\n') {
+ p = err_putprefix(p, filename, buf, err_off - 1);
+ copy(p, strblock + 59, 22);
+ p[22] = '\0';
+ return p + 23 - out;
+ }
+ switch (err) {
+ case CLEX_IDENT_BADUCNHEX4:
+ case CLEX_IDENT_BADUCNHEX6:
+ copy(p, strblock, 32);
+ p[21] += err; // 3 -> 4 or 6 (N.B. relies on enum value!!!)
+ p += 32;
+ break;
+ case CLEX_IDENT_BADCODEPOINT:
+ copy(p, strblock, 8); // "illegal "
+ copy(p + 8, strblock + 32, 17); // "Unicode codepoint"
+ p += 25;
+ break;
+ case CLEX_IDENT_UNEXPBACKSLASH:
+ copy(p, strblock, 18); // "illegal character in "
+ copy(p + 18, strblock + 49, 10); // "identifier"
+ p += 28;
+ break;
+ default:
+ unreachable;
+ }
+ *p = '\0';
+ return p - out;
+}
+
+static int utf8put_ucs2(char *out, int codepoint) { // assumes valid UCS2
+ if (codepoint > 0x7FF) {
+ out[0] = 0xE0 | codepoint >> 12;
+ out[1] = 0x80 | codepoint >> 6 & 0x3F;
+ out[2] = 0x80 | codepoint & 0x3F;
+ return 3;
+ }
+ if (codepoint <= 0x7F) {
+ out[0] = codepoint;
+ return 1;
+ }
+ out[0] = 0xC0 | codepoint >> 6;
+ out[1] = 0x80 | codepoint & 0x3F;
+ return 2;
+}
+static int utf8put(char *out, int codepoint) { // assumes valid Unicode
+ if (codepoint <= 0xFFFF) return utf8put_ucs2(out, codepoint);
+ out[0] = 0xF0 | codepoint >> 18;
+ out[1] = 0x80 | codepoint >> 12 & 0x3F;
+ out[2] = 0x80 | codepoint >> 6 & 0x3F;
+ out[3] = 0x80 | codepoint & 0x3F;
+ return 4;
+}
+
+void clex_ident(const struct clex *c, const char *restrict buf,
+ unsigned int idx, char *restrict out) {
+ const char *p = buf + c->tokoffs[idx];
+ struct bsiter_ret iter = bsiter(p); char ch = iter.c; p = iter.next;
+ for (;;) {
+ if (classes[(unsigned char)ch] > 4 * NSTATES) {
+ if (ch == '\\') {
+ iter = bsiter(p); ch = iter.c; p = iter.next;
+ if (ch == 'u') {
+ iter = bsiter(p); char c0 = iter.c; p = iter.next;
+ iter = bsiter(p); char c1 = iter.c; p = iter.next;
+ iter = bsiter(p); char c2 = iter.c; p = iter.next;
+ iter = bsiter(p); char c3 = iter.c; p = iter.next;
+ int codepoint = hexval(c0) << 12 | hexval(c1) << 8 |
+ hexval(c2) << 4 | hexval(c3);
+ out += utf8put_ucs2(out, codepoint);
+ }
+ else /* ch == 'U' */ {
+ iter = bsiter(p); char c0 = iter.c; p = iter.next;
+ iter = bsiter(p); char c1 = iter.c; p = iter.next;
+ iter = bsiter(p); char c2 = iter.c; p = iter.next;
+ iter = bsiter(p); char c3 = iter.c; p = iter.next;
+ iter = bsiter(p); char c4 = iter.c; p = iter.next;
+ iter = bsiter(p); char c5 = iter.c; p = iter.next;
+ int codepoint = hexval(c0) << 20 | hexval(c1) << 16 |
+ hexval(c2) << 12 | (hexval(c3) << 8) |
+ hexval(c4) << 4 | hexval(c5);
+ out += utf8put(out, codepoint);
+ }
+ }
+ else {
+ return;
+ }
+ }
+ else {
+ *out++ = ch;
+ iter = bsiter(p); ch = iter.c; p = iter.next;
+ }
+ }
+}
+
+// vi: sw=4 ts=4 noet tw=80 cc=80 fdm=marker