summaryrefslogtreecommitdiff
path: root/src/3p/chibicc/chibicc.h
diff options
context:
space:
mode:
authorGravatar Michael Smith <mikesmiffy128@gmail.com> 2025-12-13 18:26:51 +0000
committerGravatar Michael Smith <mikesmiffy128@gmail.com> 2026-02-16 19:11:24 +0000
commit40f9d989df2c1ff2f567656ccbdbfc0d97e34a77 (patch)
treec3a0e69b2823adfe1d5ccf0f48d77b288e55740a /src/3p/chibicc/chibicc.h
parentf386f2a1fcae885967278710fe997c56e2183476 (diff)
downloadsst-40f9d989df2c1ff2f567656ccbdbfc0d97e34a77.tar.gz
sst-40f9d989df2c1ff2f567656ccbdbfc0d97e34a77.zip
Switch cmeta from chibicc to a new homegrown lexer
This lexer is being written as a chunklet, not quite quite complete yet as it lacks some functionality to make it generally useful for things, but good enough for the gluegen use case now. So it's in the repo now and we can go ahead and use it for this instead of having this hacked-to-pieces third party thing that's nowhere near as efficient. In the future the goal is to have a decently usable library for any sort of C metaprogramming needs, not just within this project. This design does the data-oriented thing of storing as little about each token as possible (just 5 bytes), and re-lexing specific pieces only when necessary. In some cases, full validation of correct syntax is only possible through this secondary step. Since the use case is metaprogramming and code-generation rather than the reimplementation of Clang, this slight sloppiness in validation doesn't seem too bad. It's been a while since I did a proper performance measurement of this code but in some crude tests I did in the past the primary tokenisation step was running at well over 500MB/s, which is fast enough for me. It's possible that this version is a tiny bit slower due to the added complexity of making UCNs work. UCNs, incidentally, are one of the dumbest features of C by far. However, unlike trigraphs - which this lexer does not handle - UCNs are still in the language, so it's kind of sort of necessary to support them. It's probably still possible to come up with a faster design with some SIMD trickery, but the main loop is currently branchless and the lookup tables are too big for the SIMD lookup things, so that seems kind of hard. A separate SIMD path for whitespace or comment runs seems dubious as it would introduce branch mispredictions everywhere. I also made previous attempts to unroll the main loop and every attempt just made it slower, so I guess code size is a significant factor. Optimising the secondary tokenisation of identifiers (and later numerals and string/character literals once those are handled) is still on the cards, but since that happens less often, I don't know how much difference it'll make. At any rate, in a multithreaded context this thing would already come pretty close to SSD speeds, if open-read-close syscall overhead doesn't get in the way first. I imagine it's fast enough for anyone who hasn't *already* written something faster.
Diffstat (limited to 'src/3p/chibicc/chibicc.h')
-rw-r--r--src/3p/chibicc/chibicc.h264
1 files changed, 0 insertions, 264 deletions
diff --git a/src/3p/chibicc/chibicc.h b/src/3p/chibicc/chibicc.h
deleted file mode 100644
index f3f87ab..0000000
--- a/src/3p/chibicc/chibicc.h
+++ /dev/null
@@ -1,264 +0,0 @@
-// include guards: upstream doesn't have these but we add them so we can cat
-// source files together (or #include them, in particular)
-#ifndef INC_CHIBICC_H
-#define INC_CHIBICC_H
-
-#include <assert.h>
-#include <ctype.h>
-#include <errno.h>
-#include <stdarg.h>
-//#include <stdbool.h>
-#include <stdint.h>
-#include <stdio.h>
-#include <stdlib.h>
-// mike: stdnoreturn means we can't use our noreturn (_Noreturn void)
-// there are no noreturns in tokenize.c anyway, and the ones in this header have
-// been changed to just _Noreturn to avoid any possible conflict
-//#include <stdnoreturn.h>
-#include <string.h>
-
-// exists on all Unixes but normally hidden behind _GNU_SOURCE on Linux.
-// missing entirely on Windows (implemented in 3p/openbsd/asprintf.c for compat)
-int vasprintf(char **str, const char *fmt, va_list ap);
-
-#if !defined(__GNUC__) && !defined(__clang__)
-# define __attribute__(x)
-#endif
-
-typedef struct Type Type;
-typedef struct Member Member;
-typedef struct Node Node;
-typedef struct Hideset Hideset;
-
-//
-// strings.c
-//
-
-typedef struct {
- char **data;
- int capacity;
- int len;
-} StringArray;
-
-void strarray_push(StringArray *arr, char *s);
-
-//
-// tokenize.c
-//
-
-// Token
-typedef enum {
- TK_IDENT, // Identifiers
- TK_PUNCT, // Punctuators
- TK_KEYWORD, // Keywords
- TK_STR, // String literals
- TK_NUM, // Numeric literals
- TK_PP_NUM, // Preprocessing numbers
- TK_EOF, // End-of-file markers
-} TokenKind;
-
-typedef struct {
- char *name;
- int file_no;
- char *contents;
-
- // For #line directive
- char *display_name;
- int line_delta;
-} File;
-
-// Token type
-typedef struct Token Token;
-struct Token {
- TokenKind kind; // Token kind
- Token *next; // Next token
- int64_t val; // If kind is TK_NUM, its value
- long double fval; // If kind is TK_NUM, its value
- char *loc; // Token location
- int len; // Token length
- Type *ty; // Used if TK_NUM or TK_STR
- char *str; // String literal contents including terminating '\0'
-
- File *file; // Source location
- char *filename; // Filename
- int line_no; // Line number
- int line_delta; // Line number
- bool at_bol; // True if this token is at beginning of line
- bool has_space; // True if this token follows a space character
- Hideset *hideset; // For macro expansion
- Token *origin; // If this is expanded from a macro, the original token
-};
-
-_Noreturn void error(char *fmt, ...) __attribute__((format(printf, 1, 2)));
-_Noreturn void error_at(char *loc, char *fmt, ...) __attribute__((format(printf, 2, 3)));
-_Noreturn void error_tok(Token *tok, char *fmt, ...) __attribute__((format(printf, 2, 3)));
-void warn_tok(Token *tok, char *fmt, ...) __attribute__((format(printf, 2, 3)));
-bool equal(const Token *tok, const char *op);
-Token *skip(Token *tok, char *op);
-bool consume(Token **rest, Token *tok, char *str);
-void convert_pp_tokens(Token *tok);
-File **get_input_files(void);
-File *new_file(char *name, int file_no, char *contents);
-Token *tokenize_string_literal(Token *tok, Type *basety);
-Token *tokenize(File *file);
-//Token *tokenize_file(char *filename);
-Token *tokenize_buf(const char *name, char *p);
-
-// note: replacing memstream-based format with asprintf version. moved down here
-// as error() is declared above.
-//char *format(char *fmt, ...) __attribute__((format(printf, 1, 2)));
-__attribute__((format(printf, 1, 2)))
-static inline char *format(const char *fmt, ...) {
- char *ret;
- va_list va;
- va_start(va, fmt);
- if (vasprintf(&ret, fmt, va) == -1) error("couldn't allocate memory");
- va_end(va);
- return ret;
-}
-
-//
-// type.c
-//
-
-typedef enum {
- TY_VOID,
- TY_BOOL,
- TY_CHAR,
- TY_SHORT,
- TY_INT,
- TY_LONG,
- TY_FLOAT,
- TY_DOUBLE,
- TY_LDOUBLE,
- TY_ENUM,
- TY_PTR,
- TY_FUNC,
- TY_ARRAY,
- TY_VLA, // variable-length array
- TY_STRUCT,
- TY_UNION,
-} TypeKind;
-
-struct Type {
- TypeKind kind;
- int size; // sizeof() value
- int align; // alignment
- bool is_unsigned; // unsigned or signed
- bool is_atomic; // true if _Atomic
- Type *origin; // for type compatibility check
-
- // Pointer-to or array-of type. We intentionally use the same member
- // to represent pointer/array duality in C.
- //
- // In many contexts in which a pointer is expected, we examine this
- // member instead of "kind" member to determine whether a type is a
- // pointer or not. That means in many contexts "array of T" is
- // naturally handled as if it were "pointer to T", as required by
- // the C spec.
- Type *base;
-
- // Declaration
- Token *name;
- Token *name_pos;
-
- // Array
- int array_len;
-
- // Variable-length array
- //Node *vla_len; // # of elements
- //Obj *vla_size; // sizeof() value
-
- // Struct
- Member *members;
- bool is_flexible;
- bool is_packed;
-
- // Function type
- Type *return_ty;
- Type *params;
- bool is_variadic;
- Type *next;
-};
-
-// Struct member
-struct Member {
- Member *next;
- Type *ty;
- Token *tok; // for error message
- Token *name;
- int idx;
- int align;
- int offset;
-
- // Bitfield
- bool is_bitfield;
- int bit_offset;
- int bit_width;
-};
-
-extern Type *ty_void;
-extern Type *ty_bool;
-
-extern Type *ty_char;
-extern Type *ty_short;
-extern Type *ty_int;
-extern Type *ty_long;
-
-extern Type *ty_uchar;
-extern Type *ty_ushort;
-extern Type *ty_uint;
-extern Type *ty_ulong;
-
-extern Type *ty_float;
-extern Type *ty_double;
-extern Type *ty_ldouble;
-
-bool is_integer(Type *ty);
-bool is_flonum(Type *ty);
-bool is_numeric(Type *ty);
-bool is_compatible(Type *t1, Type *t2);
-Type *copy_type(Type *ty);
-Type *pointer_to(Type *base);
-Type *func_type(Type *return_ty);
-Type *array_of(Type *base, int size);
-Type *vla_of(Type *base, Node *expr);
-Type *enum_type(void);
-Type *struct_type(void);
-void add_type(Node *node);
-
-//
-// unicode.c
-//
-
-int encode_utf8(char *buf, uint32_t c);
-uint32_t decode_utf8(char **new_pos, char *p);
-bool is_ident1(uint32_t c);
-bool is_ident2(uint32_t c);
-int display_width(char *p, int len);
-
-//
-// hashmap.c
-//
-
-typedef struct {
- char *key;
- int keylen;
- void *val;
-} HashEntry;
-
-typedef struct {
- HashEntry *buckets;
- int capacity;
- int used;
-} HashMap;
-
-void *hashmap_get(HashMap *map, char *key);
-void *hashmap_get2(HashMap *map, char *key, int keylen);
-void hashmap_put(HashMap *map, char *key, void *val);
-void hashmap_put2(HashMap *map, char *key, int keylen, void *val);
-void hashmap_delete(HashMap *map, char *key);
-void hashmap_delete2(HashMap *map, char *key, int keylen);
-void hashmap_test(void);
-
-#endif