From 40f9d989df2c1ff2f567656ccbdbfc0d97e34a77 Mon Sep 17 00:00:00 2001 From: Michael Smith Date: Sat, 13 Dec 2025 18:26:51 +0000 Subject: Switch cmeta from chibicc to a new homegrown lexer This lexer is being written as a chunklet, not quite quite complete yet as it lacks some functionality to make it generally useful for things, but good enough for the gluegen use case now. So it's in the repo now and we can go ahead and use it for this instead of having this hacked-to-pieces third party thing that's nowhere near as efficient. In the future the goal is to have a decently usable library for any sort of C metaprogramming needs, not just within this project. This design does the data-oriented thing of storing as little about each token as possible (just 5 bytes), and re-lexing specific pieces only when necessary. In some cases, full validation of correct syntax is only possible through this secondary step. Since the use case is metaprogramming and code-generation rather than the reimplementation of Clang, this slight sloppiness in validation doesn't seem too bad. It's been a while since I did a proper performance measurement of this code but in some crude tests I did in the past the primary tokenisation step was running at well over 500MB/s, which is fast enough for me. It's possible that this version is a tiny bit slower due to the added complexity of making UCNs work. UCNs, incidentally, are one of the dumbest features of C by far. However, unlike trigraphs - which this lexer does not handle - UCNs are still in the language, so it's kind of sort of necessary to support them. It's probably still possible to come up with a faster design with some SIMD trickery, but the main loop is currently branchless and the lookup tables are too big for the SIMD lookup things, so that seems kind of hard. A separate SIMD path for whitespace or comment runs seems dubious as it would introduce branch mispredictions everywhere. I also made previous attempts to unroll the main loop and every attempt just made it slower, so I guess code size is a significant factor. Optimising the secondary tokenisation of identifiers (and later numerals and string/character literals once those are handled) is still on the cards, but since that happens less often, I don't know how much difference it'll make. At any rate, in a multithreaded context this thing would already come pretty close to SSD speeds, if open-read-close syscall overhead doesn't get in the way first. I imagine it's fast enough for anyone who hasn't *already* written something faster. --- src/3p/chibicc/LICENSE | 21 -- src/3p/chibicc/chibicc.h | 264 ---------------- src/3p/chibicc/hashmap.c | 137 -------- src/3p/chibicc/strings.c | 17 - src/3p/chibicc/tokenize.c | 785 ---------------------------------------------- src/3p/chibicc/unicode.c | 189 ----------- 6 files changed, 1413 deletions(-) delete mode 100644 src/3p/chibicc/LICENSE delete mode 100644 src/3p/chibicc/chibicc.h delete mode 100644 src/3p/chibicc/hashmap.c delete mode 100644 src/3p/chibicc/strings.c delete mode 100644 src/3p/chibicc/tokenize.c delete mode 100644 src/3p/chibicc/unicode.c (limited to 'src/3p/chibicc') diff --git a/src/3p/chibicc/LICENSE b/src/3p/chibicc/LICENSE deleted file mode 100644 index 2d1fd94..0000000 --- a/src/3p/chibicc/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2019 Rui Ueyama - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/src/3p/chibicc/chibicc.h b/src/3p/chibicc/chibicc.h deleted file mode 100644 index f3f87ab..0000000 --- a/src/3p/chibicc/chibicc.h +++ /dev/null @@ -1,264 +0,0 @@ -// include guards: upstream doesn't have these but we add them so we can cat -// source files together (or #include them, in particular) -#ifndef INC_CHIBICC_H -#define INC_CHIBICC_H - -#include -#include -#include -#include -//#include -#include -#include -#include -// mike: stdnoreturn means we can't use our noreturn (_Noreturn void) -// there are no noreturns in tokenize.c anyway, and the ones in this header have -// been changed to just _Noreturn to avoid any possible conflict -//#include -#include - -// exists on all Unixes but normally hidden behind _GNU_SOURCE on Linux. -// missing entirely on Windows (implemented in 3p/openbsd/asprintf.c for compat) -int vasprintf(char **str, const char *fmt, va_list ap); - -#if !defined(__GNUC__) && !defined(__clang__) -# define __attribute__(x) -#endif - -typedef struct Type Type; -typedef struct Member Member; -typedef struct Node Node; -typedef struct Hideset Hideset; - -// -// strings.c -// - -typedef struct { - char **data; - int capacity; - int len; -} StringArray; - -void strarray_push(StringArray *arr, char *s); - -// -// tokenize.c -// - -// Token -typedef enum { - TK_IDENT, // Identifiers - TK_PUNCT, // Punctuators - TK_KEYWORD, // Keywords - TK_STR, // String literals - TK_NUM, // Numeric literals - TK_PP_NUM, // Preprocessing numbers - TK_EOF, // End-of-file markers -} TokenKind; - -typedef struct { - char *name; - int file_no; - char *contents; - - // For #line directive - char *display_name; - int line_delta; -} File; - -// Token type -typedef struct Token Token; -struct Token { - TokenKind kind; // Token kind - Token *next; // Next token - int64_t val; // If kind is TK_NUM, its value - long double fval; // If kind is TK_NUM, its value - char *loc; // Token location - int len; // Token length - Type *ty; // Used if TK_NUM or TK_STR - char *str; // String literal contents including terminating '\0' - - File *file; // Source location - char *filename; // Filename - int line_no; // Line number - int line_delta; // Line number - bool at_bol; // True if this token is at beginning of line - bool has_space; // True if this token follows a space character - Hideset *hideset; // For macro expansion - Token *origin; // If this is expanded from a macro, the original token -}; - -_Noreturn void error(char *fmt, ...) __attribute__((format(printf, 1, 2))); -_Noreturn void error_at(char *loc, char *fmt, ...) __attribute__((format(printf, 2, 3))); -_Noreturn void error_tok(Token *tok, char *fmt, ...) __attribute__((format(printf, 2, 3))); -void warn_tok(Token *tok, char *fmt, ...) __attribute__((format(printf, 2, 3))); -bool equal(const Token *tok, const char *op); -Token *skip(Token *tok, char *op); -bool consume(Token **rest, Token *tok, char *str); -void convert_pp_tokens(Token *tok); -File **get_input_files(void); -File *new_file(char *name, int file_no, char *contents); -Token *tokenize_string_literal(Token *tok, Type *basety); -Token *tokenize(File *file); -//Token *tokenize_file(char *filename); -Token *tokenize_buf(const char *name, char *p); - -// note: replacing memstream-based format with asprintf version. moved down here -// as error() is declared above. -//char *format(char *fmt, ...) __attribute__((format(printf, 1, 2))); -__attribute__((format(printf, 1, 2))) -static inline char *format(const char *fmt, ...) { - char *ret; - va_list va; - va_start(va, fmt); - if (vasprintf(&ret, fmt, va) == -1) error("couldn't allocate memory"); - va_end(va); - return ret; -} - -// -// type.c -// - -typedef enum { - TY_VOID, - TY_BOOL, - TY_CHAR, - TY_SHORT, - TY_INT, - TY_LONG, - TY_FLOAT, - TY_DOUBLE, - TY_LDOUBLE, - TY_ENUM, - TY_PTR, - TY_FUNC, - TY_ARRAY, - TY_VLA, // variable-length array - TY_STRUCT, - TY_UNION, -} TypeKind; - -struct Type { - TypeKind kind; - int size; // sizeof() value - int align; // alignment - bool is_unsigned; // unsigned or signed - bool is_atomic; // true if _Atomic - Type *origin; // for type compatibility check - - // Pointer-to or array-of type. We intentionally use the same member - // to represent pointer/array duality in C. - // - // In many contexts in which a pointer is expected, we examine this - // member instead of "kind" member to determine whether a type is a - // pointer or not. That means in many contexts "array of T" is - // naturally handled as if it were "pointer to T", as required by - // the C spec. - Type *base; - - // Declaration - Token *name; - Token *name_pos; - - // Array - int array_len; - - // Variable-length array - //Node *vla_len; // # of elements - //Obj *vla_size; // sizeof() value - - // Struct - Member *members; - bool is_flexible; - bool is_packed; - - // Function type - Type *return_ty; - Type *params; - bool is_variadic; - Type *next; -}; - -// Struct member -struct Member { - Member *next; - Type *ty; - Token *tok; // for error message - Token *name; - int idx; - int align; - int offset; - - // Bitfield - bool is_bitfield; - int bit_offset; - int bit_width; -}; - -extern Type *ty_void; -extern Type *ty_bool; - -extern Type *ty_char; -extern Type *ty_short; -extern Type *ty_int; -extern Type *ty_long; - -extern Type *ty_uchar; -extern Type *ty_ushort; -extern Type *ty_uint; -extern Type *ty_ulong; - -extern Type *ty_float; -extern Type *ty_double; -extern Type *ty_ldouble; - -bool is_integer(Type *ty); -bool is_flonum(Type *ty); -bool is_numeric(Type *ty); -bool is_compatible(Type *t1, Type *t2); -Type *copy_type(Type *ty); -Type *pointer_to(Type *base); -Type *func_type(Type *return_ty); -Type *array_of(Type *base, int size); -Type *vla_of(Type *base, Node *expr); -Type *enum_type(void); -Type *struct_type(void); -void add_type(Node *node); - -// -// unicode.c -// - -int encode_utf8(char *buf, uint32_t c); -uint32_t decode_utf8(char **new_pos, char *p); -bool is_ident1(uint32_t c); -bool is_ident2(uint32_t c); -int display_width(char *p, int len); - -// -// hashmap.c -// - -typedef struct { - char *key; - int keylen; - void *val; -} HashEntry; - -typedef struct { - HashEntry *buckets; - int capacity; - int used; -} HashMap; - -void *hashmap_get(HashMap *map, char *key); -void *hashmap_get2(HashMap *map, char *key, int keylen); -void hashmap_put(HashMap *map, char *key, void *val); -void hashmap_put2(HashMap *map, char *key, int keylen, void *val); -void hashmap_delete(HashMap *map, char *key); -void hashmap_delete2(HashMap *map, char *key, int keylen); -void hashmap_test(void); - -#endif diff --git a/src/3p/chibicc/hashmap.c b/src/3p/chibicc/hashmap.c deleted file mode 100644 index 2090274..0000000 --- a/src/3p/chibicc/hashmap.c +++ /dev/null @@ -1,137 +0,0 @@ -// This is an implementation of the open-addressing hash table. - -#include "chibicc.h" - -// mike: moved from chibicc.h and also renamed, to avoid conflicts with langext.h -#define chibi_unreachable() error("internal error at %s:%d", __FILE__, __LINE__) - -// Initial hash bucket size -#define INIT_SIZE 16 - -// Rehash if the usage exceeds 70%. -#define HIGH_WATERMARK 70 - -// We'll keep the usage below 50% after rehashing. -#define LOW_WATERMARK 50 - -// Represents a deleted hash entry -#define TOMBSTONE ((void *)-1) - -static uint64_t fnv_hash(char *s, int len) { - uint64_t hash = 0xcbf29ce484222325; - for (int i = 0; i < len; i++) { - hash *= 0x100000001b3; - hash ^= (unsigned char)s[i]; - } - return hash; -} - -// Make room for new entires in a given hashmap by removing -// tombstones and possibly extending the bucket size. -static void rehash(HashMap *map) { - // Compute the size of the new hashmap. - int nkeys = 0; - for (int i = 0; i < map->capacity; i++) - if (map->buckets[i].key && map->buckets[i].key != TOMBSTONE) - nkeys++; - - int cap = map->capacity; - while ((nkeys * 100) / cap >= LOW_WATERMARK) - cap = cap * 2; - assert(cap > 0); - - // Create a new hashmap and copy all key-values. - HashMap map2 = {0}; - map2.buckets = calloc(cap, sizeof(HashEntry)); - map2.capacity = cap; - - for (int i = 0; i < map->capacity; i++) { - HashEntry *ent = &map->buckets[i]; - if (ent->key && ent->key != TOMBSTONE) - hashmap_put2(&map2, ent->key, ent->keylen, ent->val); - } - - assert(map2.used == nkeys); - *map = map2; -} - -static bool match(HashEntry *ent, char *key, int keylen) { - return ent->key && ent->key != TOMBSTONE && - ent->keylen == keylen && memcmp(ent->key, key, keylen) == 0; -} - -static HashEntry *get_entry(HashMap *map, char *key, int keylen) { - if (!map->buckets) - return NULL; - - uint64_t hash = fnv_hash(key, keylen); - - for (int i = 0; i < map->capacity; i++) { - HashEntry *ent = &map->buckets[(hash + i) % map->capacity]; - if (match(ent, key, keylen)) - return ent; - if (ent->key == NULL) - return NULL; - } - chibi_unreachable(); -} - -static HashEntry *get_or_insert_entry(HashMap *map, char *key, int keylen) { - if (!map->buckets) { - map->buckets = calloc(INIT_SIZE, sizeof(HashEntry)); - map->capacity = INIT_SIZE; - } else if ((map->used * 100) / map->capacity >= HIGH_WATERMARK) { - rehash(map); - } - - uint64_t hash = fnv_hash(key, keylen); - - for (int i = 0; i < map->capacity; i++) { - HashEntry *ent = &map->buckets[(hash + i) % map->capacity]; - - if (match(ent, key, keylen)) - return ent; - - if (ent->key == TOMBSTONE) { - ent->key = key; - ent->keylen = keylen; - return ent; - } - - if (ent->key == NULL) { - ent->key = key; - ent->keylen = keylen; - map->used++; - return ent; - } - } - chibi_unreachable(); -} - -void *hashmap_get(HashMap *map, char *key) { - return hashmap_get2(map, key, strlen(key)); -} - -void *hashmap_get2(HashMap *map, char *key, int keylen) { - HashEntry *ent = get_entry(map, key, keylen); - return ent ? ent->val : NULL; -} - -void hashmap_put(HashMap *map, char *key, void *val) { - hashmap_put2(map, key, strlen(key), val); -} - -void hashmap_put2(HashMap *map, char *key, int keylen, void *val) { - HashEntry *ent = get_or_insert_entry(map, key, keylen); - ent->val = val; -} - -void hashmap_delete(HashMap *map, char *key) { - hashmap_delete2(map, key, strlen(key)); -} - -void hashmap_delete2(HashMap *map, char *key, int keylen) { - HashEntry *ent = get_entry(map, key, keylen); - if (ent) - ent->key = TOMBSTONE; -} diff --git a/src/3p/chibicc/strings.c b/src/3p/chibicc/strings.c deleted file mode 100644 index 0538fef..0000000 --- a/src/3p/chibicc/strings.c +++ /dev/null @@ -1,17 +0,0 @@ -#include "chibicc.h" - -void strarray_push(StringArray *arr, char *s) { - if (!arr->data) { - arr->data = calloc(8, sizeof(char *)); - arr->capacity = 8; - } - - if (arr->capacity == arr->len) { - arr->data = realloc(arr->data, sizeof(char *) * arr->capacity * 2); - arr->capacity *= 2; - for (int i = arr->len; i < arr->capacity; i++) - arr->data[i] = NULL; - } - - arr->data[arr->len++] = s; -} diff --git a/src/3p/chibicc/tokenize.c b/src/3p/chibicc/tokenize.c deleted file mode 100644 index 6738206..0000000 --- a/src/3p/chibicc/tokenize.c +++ /dev/null @@ -1,785 +0,0 @@ -#include "chibicc.h" - -#ifdef _WIN32 -#define strncasecmp _strnicmp -#endif - -// Input file -static File *current_file; - -// A list of all input files. -static File **input_files; - -// True if the current position is at the beginning of a line -static bool at_bol; - -// True if the current position follows a space character -static bool has_space; - -// Reports an error and exit. -void error(char *fmt, ...) { - va_list ap; - va_start(ap, fmt); - fprintf(stderr, "cmeta: chibicc: "); - vfprintf(stderr, fmt, ap); - fprintf(stderr, "\n"); - exit(1); -} - -// Reports an error message in the following format. -// -// foo.c:10: x = y + 1; -// ^ -static void verror_at(char *filename, char *input, int line_no, - char *loc, char *fmt, va_list ap) { - // Find a line containing `loc`. - char *line = loc; - while (input < line && line[-1] != '\n') - line--; - - char *end = loc; - while (*end && *end != '\n') - end++; - - // Print out the line. - int indent = fprintf(stderr, "%s:%d: ", filename, line_no); - fprintf(stderr, "%.*s\n", (int)(end - line), line); - - // Show the error message. - int pos = display_width(line, loc - line) + indent; - - fprintf(stderr, "%*s", pos, ""); // print pos spaces. - fprintf(stderr, "^ "); - vfprintf(stderr, fmt, ap); - fprintf(stderr, "\n"); -} - -void error_at(char *loc, char *fmt, ...) { - int line_no = 1; - for (char *p = current_file->contents; p < loc; p++) - if (*p == '\n') - line_no++; - - va_list ap; - va_start(ap, fmt); - verror_at(current_file->name, current_file->contents, line_no, loc, fmt, ap); - exit(1); -} - -void error_tok(Token *tok, char *fmt, ...) { - va_list ap; - va_start(ap, fmt); - verror_at(tok->file->name, tok->file->contents, tok->line_no, tok->loc, fmt, ap); - exit(1); -} - -void warn_tok(Token *tok, char *fmt, ...) { - va_list ap; - va_start(ap, fmt); - verror_at(tok->file->name, tok->file->contents, tok->line_no, tok->loc, fmt, ap); - va_end(ap); -} - -// Consumes the current token if it matches `op`. -bool equal(const Token *tok, const char *op) { - return memcmp(tok->loc, op, tok->len) == 0 && op[tok->len] == '\0'; -} - -// Ensure that the current token is `op`. -Token *skip(Token *tok, char *op) { - if (!equal(tok, op)) - error_tok(tok, "expected '%s'", op); - return tok->next; -} - -bool consume(Token **rest, Token *tok, char *str) { - if (equal(tok, str)) { - *rest = tok->next; - return true; - } - *rest = tok; - return false; -} - -// Create a new token. -static Token *new_token(TokenKind kind, char *start, char *end) { - Token *tok = calloc(1, sizeof(Token)); - tok->kind = kind; - tok->loc = start; - tok->len = end - start; - tok->file = current_file; - tok->filename = current_file->display_name; - tok->at_bol = at_bol; - tok->has_space = has_space; - - at_bol = has_space = false; - return tok; -} - -static bool startswith(char *p, char *q) { - return strncmp(p, q, strlen(q)) == 0; -} - -// Read an identifier and returns the length of it. -// If p does not point to a valid identifier, 0 is returned. -static int read_ident(char *start) { - char *p = start; - uint32_t c = decode_utf8(&p, p); - if (!is_ident1(c)) - return 0; - - for (;;) { - char *q; - c = decode_utf8(&q, p); - if (!is_ident2(c)) - return p - start; - p = q; - } -} - -static int from_hex(char c) { - if ('0' <= c && c <= '9') - return c - '0'; - if ('a' <= c && c <= 'f') - return c - 'a' + 10; - return c - 'A' + 10; -} - -// Read a punctuator token from p and returns its length. -static int read_punct(char *p) { - static char *kw[] = { - "<<=", ">>=", "...", "==", "!=", "<=", ">=", "->", "+=", - "-=", "*=", "/=", "++", "--", "%=", "&=", "|=", "^=", "&&", - "||", "<<", ">>", "##", - }; - - for (int i = 0; i < sizeof(kw) / sizeof(*kw); i++) - if (startswith(p, kw[i])) - return strlen(kw[i]); - - return ispunct(*p) ? 1 : 0; -} - -static bool is_keyword(Token *tok) { - static HashMap map; - - if (map.capacity == 0) { - static char *kw[] = { - "return", "if", "else", "for", "while", "int", "sizeof", "char", - "struct", "union", "short", "long", "void", "typedef", "_Bool", - "enum", "static", "goto", "break", "continue", "switch", "case", - "default", "extern", "_Alignof", "_Alignas", "do", "signed", - "unsigned", "const", "volatile", "auto", "register", "restrict", - "__restrict", "__restrict__", "_Noreturn", "float", "double", - "typeof", "asm", "_Thread_local", "__thread", "_Atomic", - "__attribute__", - }; - - for (int i = 0; i < sizeof(kw) / sizeof(*kw); i++) - hashmap_put(&map, kw[i], (void *)1); - } - - return hashmap_get2(&map, tok->loc, tok->len); -} - -static int read_escaped_char(char **new_pos, char *p) { - if ('0' <= *p && *p <= '7') { - // Read an octal number. - int c = *p++ - '0'; - if ('0' <= *p && *p <= '7') { - c = (c << 3) + (*p++ - '0'); - if ('0' <= *p && *p <= '7') - c = (c << 3) + (*p++ - '0'); - } - *new_pos = p; - return c; - } - - if (*p == 'x') { - // Read a hexadecimal number. - p++; - if (!isxdigit(*p)) - error_at(p, "invalid hex escape sequence"); - - int c = 0; - for (; isxdigit(*p); p++) - c = (c << 4) + from_hex(*p); - *new_pos = p; - return c; - } - - *new_pos = p + 1; - - // Escape sequences are defined using themselves here. E.g. - // '\n' is implemented using '\n'. This tautological definition - // works because the compiler that compiles our compiler knows - // what '\n' actually is. In other words, we "inherit" the ASCII - // code of '\n' from the compiler that compiles our compiler, - // so we don't have to teach the actual code here. - // - // This fact has huge implications not only for the correctness - // of the compiler but also for the security of the generated code. - // For more info, read "Reflections on Trusting Trust" by Ken Thompson. - // https://github.com/rui314/chibicc/wiki/thompson1984.pdf - switch (*p) { - case 'a': return '\a'; - case 'b': return '\b'; - case 't': return '\t'; - case 'n': return '\n'; - case 'v': return '\v'; - case 'f': return '\f'; - case 'r': return '\r'; - // [GNU] \e for the ASCII escape character is a GNU C extension. - case 'e': return 27; - default: return *p; - } -} - -// Find a closing double-quote. -static char *string_literal_end(char *p) { - char *start = p; - for (; *p != '"'; p++) { - if (*p == '\n' || *p == '\0') - error_at(start, "unclosed string literal"); - if (*p == '\\') - p++; - } - return p; -} - -static Token *read_string_literal(char *start, char *quote) { - char *end = string_literal_end(quote + 1); - char *buf = calloc(1, end - quote); - int len = 0; - - for (char *p = quote + 1; p < end;) { - if (*p == '\\') - buf[len++] = read_escaped_char(&p, p + 1); - else - buf[len++] = *p++; - } - - Token *tok = new_token(TK_STR, start, end + 1); - tok->ty = array_of(ty_char, len + 1); - tok->str = buf; - return tok; -} - -// Read a UTF-8-encoded string literal and transcode it in UTF-16. -// -// UTF-16 is yet another variable-width encoding for Unicode. Code -// points smaller than U+10000 are encoded in 2 bytes. Code points -// equal to or larger than that are encoded in 4 bytes. Each 2 bytes -// in the 4 byte sequence is called "surrogate", and a 4 byte sequence -// is called a "surrogate pair". -static Token *read_utf16_string_literal(char *start, char *quote) { - char *end = string_literal_end(quote + 1); - uint16_t *buf = calloc(2, end - start); - int len = 0; - - for (char *p = quote + 1; p < end;) { - if (*p == '\\') { - buf[len++] = read_escaped_char(&p, p + 1); - continue; - } - - uint32_t c = decode_utf8(&p, p); - if (c < 0x10000) { - // Encode a code point in 2 bytes. - buf[len++] = c; - } else { - // Encode a code point in 4 bytes. - c -= 0x10000; - buf[len++] = 0xd800 + ((c >> 10) & 0x3ff); - buf[len++] = 0xdc00 + (c & 0x3ff); - } - } - - Token *tok = new_token(TK_STR, start, end + 1); - tok->ty = array_of(ty_ushort, len + 1); - tok->str = (char *)buf; - return tok; -} - -// Read a UTF-8-encoded string literal and transcode it in UTF-32. -// -// UTF-32 is a fixed-width encoding for Unicode. Each code point is -// encoded in 4 bytes. -static Token *read_utf32_string_literal(char *start, char *quote, Type *ty) { - char *end = string_literal_end(quote + 1); - uint32_t *buf = calloc(4, end - quote); - int len = 0; - - for (char *p = quote + 1; p < end;) { - if (*p == '\\') - buf[len++] = read_escaped_char(&p, p + 1); - else - buf[len++] = decode_utf8(&p, p); - } - - Token *tok = new_token(TK_STR, start, end + 1); - tok->ty = array_of(ty, len + 1); - tok->str = (char *)buf; - return tok; -} - -static Token *read_char_literal(char *start, char *quote, Type *ty) { - char *p = quote + 1; - if (*p == '\0') - error_at(start, "unclosed char literal"); - - int c; - if (*p == '\\') - c = read_escaped_char(&p, p + 1); - else - c = decode_utf8(&p, p); - - char *end = p; - for (; *end != '\''; ++end) { - if (!*end || *end == '\n') - error_at(p, "unclosed char literal"); - } - - Token *tok = new_token(TK_NUM, start, end + 1); - tok->val = c; - tok->ty = ty; - return tok; -} - -static bool convert_pp_int(Token *tok) { - char *p = tok->loc; - - // Read a binary, octal, decimal or hexadecimal number. - int base = 10; - if (!strncasecmp(p, "0x", 2) && isxdigit(p[2])) { - p += 2; - base = 16; - } else if (!strncasecmp(p, "0b", 2) && (p[2] == '0' || p[2] == '1')) { - p += 2; - base = 2; - } else if (*p == '0') { - base = 8; - } - - int64_t val = strtoul(p, &p, base); - - // Read U, L or LL suffixes. - bool l = false; - bool u = false; - - if (startswith(p, "LLU") || startswith(p, "LLu") || - startswith(p, "llU") || startswith(p, "llu") || - startswith(p, "ULL") || startswith(p, "Ull") || - startswith(p, "uLL") || startswith(p, "ull")) { - p += 3; - l = u = true; - } else if (!strncasecmp(p, "lu", 2) || !strncasecmp(p, "ul", 2)) { - p += 2; - l = u = true; - } else if (startswith(p, "LL") || startswith(p, "ll")) { - p += 2; - l = true; - } else if (*p == 'L' || *p == 'l') { - p++; - l = true; - } else if (*p == 'U' || *p == 'u') { - p++; - u = true; - } - - if (p != tok->loc + tok->len) - return false; - - // Infer a type. - Type *ty; - if (base == 10) { - if (l && u) - ty = ty_ulong; - else if (l) - ty = ty_long; - else if (u) - ty = (val >> 32) ? ty_ulong : ty_uint; - else - ty = (val >> 31) ? ty_long : ty_int; - } else { - if (l && u) - ty = ty_ulong; - else if (l) - ty = (val >> 63) ? ty_ulong : ty_long; - else if (u) - ty = (val >> 32) ? ty_ulong : ty_uint; - else if (val >> 63) - ty = ty_ulong; - else if (val >> 32) - ty = ty_long; - else if (val >> 31) - ty = ty_uint; - else - ty = ty_int; - } - - tok->kind = TK_NUM; - tok->val = val; - tok->ty = ty; - return true; -} - -// The definition of the numeric literal at the preprocessing stage -// is more relaxed than the definition of that at the later stages. -// In order to handle that, a numeric literal is tokenized as a -// "pp-number" token first and then converted to a regular number -// token after preprocessing. -// -// This function converts a pp-number token to a regular number token. -static void convert_pp_number(Token *tok) { - // Try to parse as an integer constant. - if (convert_pp_int(tok)) - return; - - // If it's not an integer, it must be a floating point constant. - char *end; - long double val = strtold(tok->loc, &end); - - Type *ty; - if (*end == 'f' || *end == 'F') { - ty = ty_float; - end++; - } else if (*end == 'l' || *end == 'L') { - ty = ty_ldouble; - end++; - } else { - ty = ty_double; - } - - if (tok->loc + tok->len != end) - error_tok(tok, "invalid numeric constant"); - - tok->kind = TK_NUM; - tok->fval = val; - tok->ty = ty; -} - -void convert_pp_tokens(Token *tok) { - for (Token *t = tok; t->kind != TK_EOF; t = t->next) { - if (is_keyword(t)) - t->kind = TK_KEYWORD; - else if (t->kind == TK_PP_NUM) - convert_pp_number(t); - } -} - -// Initialize line info for all tokens. -static void add_line_numbers(Token *tok) { - char *p = current_file->contents; - int n = 1; - - do { - if (p == tok->loc) { - tok->line_no = n; - tok = tok->next; - } - if (*p == '\n') - n++; - } while (*p++); -} - -Token *tokenize_string_literal(Token *tok, Type *basety) { - Token *t; - if (basety->size == 2) - t = read_utf16_string_literal(tok->loc, tok->loc); - else - t = read_utf32_string_literal(tok->loc, tok->loc, basety); - t->next = tok->next; - return t; -} - -// Tokenize a given string and returns new tokens. -Token *tokenize(File *file) { - current_file = file; - - char *p = file->contents; - Token head = {0}; - Token *cur = &head; - - at_bol = true; - has_space = false; - - while (*p) { - // Skip line comments. - if (startswith(p, "//")) { - p += 2; - while (*p != '\n') - p++; - has_space = true; - continue; - } - - // Skip block comments. - if (startswith(p, "/*")) { - char *q = strstr(p + 2, "*/"); - if (!q) - error_at(p, "unclosed block comment"); - p = q + 2; - has_space = true; - continue; - } - - // Skip newline. - if (*p == '\n') { - p++; - at_bol = true; - has_space = false; - continue; - } - - // Skip whitespace characters. - if (isspace(*p)) { - p++; - has_space = true; - continue; - } - - // Numeric literal - if (isdigit(*p) || (*p == '.' && isdigit(p[1]))) { - char *q = p++; - for (;;) { - if (p[0] && p[1] && strchr("eEpP", p[0]) && strchr("+-", p[1])) - p += 2; - else if (isalnum(*p) || *p == '.') - p++; - else - break; - } - cur = cur->next = new_token(TK_PP_NUM, q, p); - continue; - } - - // String literal - if (*p == '"') { - cur = cur->next = read_string_literal(p, p); - p += cur->len; - continue; - } - - // UTF-8 string literal - if (startswith(p, "u8\"")) { - cur = cur->next = read_string_literal(p, p + 2); - p += cur->len; - continue; - } - - // UTF-16 string literal - if (startswith(p, "u\"")) { - cur = cur->next = read_utf16_string_literal(p, p + 1); - p += cur->len; - continue; - } - - // Wide string literal - if (startswith(p, "L\"")) { - cur = cur->next = read_utf32_string_literal(p, p + 1, ty_int); - p += cur->len; - continue; - } - - // UTF-32 string literal - if (startswith(p, "U\"")) { - cur = cur->next = read_utf32_string_literal(p, p + 1, ty_uint); - p += cur->len; - continue; - } - - // Character literal - if (*p == '\'') { - cur = cur->next = read_char_literal(p, p, ty_int); - cur->val = (char)cur->val; - p += cur->len; - continue; - } - - // UTF-16 character literal - if (startswith(p, "u'")) { - cur = cur->next = read_char_literal(p, p + 1, ty_ushort); - cur->val &= 0xffff; - p += cur->len; - continue; - } - - // Wide character literal - if (startswith(p, "L'")) { - cur = cur->next = read_char_literal(p, p + 1, ty_int); - p += cur->len; - continue; - } - - // UTF-32 character literal - if (startswith(p, "U'")) { - cur = cur->next = read_char_literal(p, p + 1, ty_uint); - p += cur->len; - continue; - } - - // Identifier or keyword - int ident_len = read_ident(p); - if (ident_len) { - cur = cur->next = new_token(TK_IDENT, p, p + ident_len); - p += cur->len; - continue; - } - - // Punctuators - int punct_len = read_punct(p); - if (punct_len) { - cur = cur->next = new_token(TK_PUNCT, p, p + punct_len); - p += cur->len; - continue; - } - - error_at(p, "invalid token"); - } - - cur = cur->next = new_token(TK_EOF, p, p); - add_line_numbers(head.next); - return head.next; -} - -// Returns the contents of a given file. -/* this function is a bit wonkily implemented, and relies on a non-windows - * thing, so we have our own in src/cmeta.c. -static char *read_file(char *path) { - FILE *fp; - - if (strcmp(path, "-") == 0) { - // By convention, read from stdin if a given filename is "-". - fp = stdin; - } else { - fp = fopen(path, "r"); - if (!fp) - return NULL; - } - - char *buf; - size_t buflen; - FILE *out = open_memstream(&buf, &buflen); - - // Read the entire file. - for (;;) { - char buf2[4096]; - int n = fread(buf2, 1, sizeof(buf2), fp); - if (n == 0) - break; - fwrite(buf2, 1, n, out); - } - - if (fp != stdin) - fclose(fp); - - // Make sure that the last line is properly terminated with '\n'. - fflush(out); - if (buflen == 0 || buf[buflen - 1] != '\n') - fputc('\n', out); - fputc('\0', out); - fclose(out); - return buf; -} -*/ - -File **get_input_files(void) { - return input_files; -} - -File *new_file(char *name, int file_no, char *contents) { - File *file = calloc(1, sizeof(File)); - file->name = name; - file->display_name = name; - file->file_no = file_no; - file->contents = contents; - return file; -} - -// Removes backslashes followed by a newline. -static void remove_backslash_newline(char *p) { - int i = 0, j = 0; - - // We want to keep the number of newline characters so that - // the logical line number matches the physical one. - // This counter maintain the number of newlines we have removed. - int n = 0; - - while (p[i]) { - if (p[i] == '\\' && p[i + 1] == '\n') { - i += 2; - n++; - } else if (p[i] == '\n') { - p[j++] = p[i++]; - for (; n > 0; n--) - p[j++] = '\n'; - } else { - p[j++] = p[i++]; - } - } - - for (; n > 0; n--) - p[j++] = '\n'; - p[j] = '\0'; -} - -static uint32_t read_universal_char(char *p, int len) { - uint32_t c = 0; - for (int i = 0; i < len; i++) { - if (!isxdigit(p[i])) - return 0; - c = (c << 4) | from_hex(p[i]); - } - return c; -} - -// Replace \u or \U escape sequences with corresponding UTF-8 bytes. -static void convert_universal_chars(char *p) { - char *q = p; - - while (*p) { - if (startswith(p, "\\u")) { - uint32_t c = read_universal_char(p + 2, 4); - if (c) { - p += 6; - q += encode_utf8(q, c); - } else { - *q++ = *p++; - } - } else if (startswith(p, "\\U")) { - uint32_t c = read_universal_char(p + 2, 8); - if (c) { - p += 10; - q += encode_utf8(q, c); - } else { - *q++ = *p++; - } - } else if (p[0] == '\\') { - *q++ = *p++; - *q++ = *p++; - } else { - *q++ = *p++; - } - } - - *q = '\0'; -} - -// NOTE modified API from upstream -Token *tokenize_buf(const char *name, char *p) { - remove_backslash_newline(p); - convert_universal_chars(p); - - // Save the filename for assembler .file directive. - static int file_no; - File *file = new_file((char *)name, file_no + 1, p); - - // Save the filename for assembler .file directive. - input_files = realloc(input_files, sizeof(char *) * (file_no + 2)); - input_files[file_no] = file; - input_files[file_no + 1] = NULL; - file_no++; - - return tokenize(file); -} diff --git a/src/3p/chibicc/unicode.c b/src/3p/chibicc/unicode.c deleted file mode 100644 index 6db1ad7..0000000 --- a/src/3p/chibicc/unicode.c +++ /dev/null @@ -1,189 +0,0 @@ -#include "chibicc.h" - -// Encode a given character in UTF-8. -int encode_utf8(char *buf, uint32_t c) { - if (c <= 0x7F) { - buf[0] = c; - return 1; - } - - if (c <= 0x7FF) { - buf[0] = 0xC0 | (c >> 6); - buf[1] = 0x80 | ((c >> 6) & 0x3F); - return 2; - } - - if (c <= 0xFFFF) { - buf[0] = 0xE0 | (c >> 12); - buf[1] = 0x80 | ((c >> 6) & 0x3F); - buf[2] = 0x80 | (c & 0x3F); - return 3; - } - - buf[0] = 0xF0 | (c >> 18); - buf[1] = 0x80 | ((c >> 12) & 0x3F); - buf[2] = 0x80 | ((c >> 6) & 0x3F); - buf[3] = 0x80 | (c & 0x3F); - return 4; -} - -// Read a UTF-8-encoded Unicode code point from a source file. -// We assume that source files are always in UTF-8. -// -// UTF-8 is a variable-width encoding in which one code point is -// encoded in one to four bytes. One byte UTF-8 code points are -// identical to ASCII. Non-ASCII characters are encoded using more -// than one byte. -uint32_t decode_utf8(char **new_pos, char *p) { - if ((unsigned char)*p < 128) { - *new_pos = p + 1; - return *p; - } - - char *start = p; - int len; - uint32_t c; - - if ((unsigned char)*p >= 0xF0) { - len = 4; - c = *p & 7; - } else if ((unsigned char)*p >= 0xE0) { - len = 3; - c = *p & 15; - } else if ((unsigned char)*p >= 0xC0) { - len = 2; - c = *p & 31; - } else { - error_at(start, "invalid UTF-8 sequence"); - } - - for (int i = 1; i < len; i++) { - if ((unsigned char)p[i] >> 6 != 2) - error_at(start, "invalid UTF-8 sequence"); - c = (c << 6) | (p[i] & 63); - } - - *new_pos = p + len; - return c; -} - -static bool in_range(uint32_t *range, uint32_t c) { - for (int i = 0; range[i] != -1; i += 2) - if (range[i] <= c && c <= range[i + 1]) - return true; - return false; -} - -// [https://www.sigbus.info/n1570#D] C11 allows not only ASCII but -// some multibyte characters in certan Unicode ranges to be used in an -// identifier. -// -// This function returns true if a given character is acceptable as -// the first character of an identifier. -// -// For example, ¾ (U+00BE) is a valid identifier because characters in -// 0x00BE-0x00C0 are allowed, while neither ⟘ (U+27D8) nor ' ' -// (U+3000, full-width space) are allowed because they are out of range. -bool is_ident1(uint32_t c) { - static uint32_t range[] = { - '_', '_', 'a', 'z', 'A', 'Z', '$', '$', - 0x00A8, 0x00A8, 0x00AA, 0x00AA, 0x00AD, 0x00AD, 0x00AF, 0x00AF, - 0x00B2, 0x00B5, 0x00B7, 0x00BA, 0x00BC, 0x00BE, 0x00C0, 0x00D6, - 0x00D8, 0x00F6, 0x00F8, 0x00FF, 0x0100, 0x02FF, 0x0370, 0x167F, - 0x1681, 0x180D, 0x180F, 0x1DBF, 0x1E00, 0x1FFF, 0x200B, 0x200D, - 0x202A, 0x202E, 0x203F, 0x2040, 0x2054, 0x2054, 0x2060, 0x206F, - 0x2070, 0x20CF, 0x2100, 0x218F, 0x2460, 0x24FF, 0x2776, 0x2793, - 0x2C00, 0x2DFF, 0x2E80, 0x2FFF, 0x3004, 0x3007, 0x3021, 0x302F, - 0x3031, 0x303F, 0x3040, 0xD7FF, 0xF900, 0xFD3D, 0xFD40, 0xFDCF, - 0xFDF0, 0xFE1F, 0xFE30, 0xFE44, 0xFE47, 0xFFFD, - 0x10000, 0x1FFFD, 0x20000, 0x2FFFD, 0x30000, 0x3FFFD, 0x40000, 0x4FFFD, - 0x50000, 0x5FFFD, 0x60000, 0x6FFFD, 0x70000, 0x7FFFD, 0x80000, 0x8FFFD, - 0x90000, 0x9FFFD, 0xA0000, 0xAFFFD, 0xB0000, 0xBFFFD, 0xC0000, 0xCFFFD, - 0xD0000, 0xDFFFD, 0xE0000, 0xEFFFD, -1, - }; - - return in_range(range, c); -} - -// Returns true if a given character is acceptable as a non-first -// character of an identifier. -bool is_ident2(uint32_t c) { - static uint32_t range[] = { - '0', '9', '$', '$', 0x0300, 0x036F, 0x1DC0, 0x1DFF, 0x20D0, 0x20FF, - 0xFE20, 0xFE2F, -1, - }; - - return is_ident1(c) || in_range(range, c); -} - -// Returns the number of columns needed to display a given -// character in a fixed-width font. -// -// Based on https://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c -static int char_width(uint32_t c) { - static uint32_t range1[] = { - 0x0000, 0x001F, 0x007f, 0x00a0, 0x0300, 0x036F, 0x0483, 0x0486, - 0x0488, 0x0489, 0x0591, 0x05BD, 0x05BF, 0x05BF, 0x05C1, 0x05C2, - 0x05C4, 0x05C5, 0x05C7, 0x05C7, 0x0600, 0x0603, 0x0610, 0x0615, - 0x064B, 0x065E, 0x0670, 0x0670, 0x06D6, 0x06E4, 0x06E7, 0x06E8, - 0x06EA, 0x06ED, 0x070F, 0x070F, 0x0711, 0x0711, 0x0730, 0x074A, - 0x07A6, 0x07B0, 0x07EB, 0x07F3, 0x0901, 0x0902, 0x093C, 0x093C, - 0x0941, 0x0948, 0x094D, 0x094D, 0x0951, 0x0954, 0x0962, 0x0963, - 0x0981, 0x0981, 0x09BC, 0x09BC, 0x09C1, 0x09C4, 0x09CD, 0x09CD, - 0x09E2, 0x09E3, 0x0A01, 0x0A02, 0x0A3C, 0x0A3C, 0x0A41, 0x0A42, - 0x0A47, 0x0A48, 0x0A4B, 0x0A4D, 0x0A70, 0x0A71, 0x0A81, 0x0A82, - 0x0ABC, 0x0ABC, 0x0AC1, 0x0AC5, 0x0AC7, 0x0AC8, 0x0ACD, 0x0ACD, - 0x0AE2, 0x0AE3, 0x0B01, 0x0B01, 0x0B3C, 0x0B3C, 0x0B3F, 0x0B3F, - 0x0B41, 0x0B43, 0x0B4D, 0x0B4D, 0x0B56, 0x0B56, 0x0B82, 0x0B82, - 0x0BC0, 0x0BC0, 0x0BCD, 0x0BCD, 0x0C3E, 0x0C40, 0x0C46, 0x0C48, - 0x0C4A, 0x0C4D, 0x0C55, 0x0C56, 0x0CBC, 0x0CBC, 0x0CBF, 0x0CBF, - 0x0CC6, 0x0CC6, 0x0CCC, 0x0CCD, 0x0CE2, 0x0CE3, 0x0D41, 0x0D43, - 0x0D4D, 0x0D4D, 0x0DCA, 0x0DCA, 0x0DD2, 0x0DD4, 0x0DD6, 0x0DD6, - 0x0E31, 0x0E31, 0x0E34, 0x0E3A, 0x0E47, 0x0E4E, 0x0EB1, 0x0EB1, - 0x0EB4, 0x0EB9, 0x0EBB, 0x0EBC, 0x0EC8, 0x0ECD, 0x0F18, 0x0F19, - 0x0F35, 0x0F35, 0x0F37, 0x0F37, 0x0F39, 0x0F39, 0x0F71, 0x0F7E, - 0x0F80, 0x0F84, 0x0F86, 0x0F87, 0x0F90, 0x0F97, 0x0F99, 0x0FBC, - 0x0FC6, 0x0FC6, 0x102D, 0x1030, 0x1032, 0x1032, 0x1036, 0x1037, - 0x1039, 0x1039, 0x1058, 0x1059, 0x1160, 0x11FF, 0x135F, 0x135F, - 0x1712, 0x1714, 0x1732, 0x1734, 0x1752, 0x1753, 0x1772, 0x1773, - 0x17B4, 0x17B5, 0x17B7, 0x17BD, 0x17C6, 0x17C6, 0x17C9, 0x17D3, - 0x17DD, 0x17DD, 0x180B, 0x180D, 0x18A9, 0x18A9, 0x1920, 0x1922, - 0x1927, 0x1928, 0x1932, 0x1932, 0x1939, 0x193B, 0x1A17, 0x1A18, - 0x1B00, 0x1B03, 0x1B34, 0x1B34, 0x1B36, 0x1B3A, 0x1B3C, 0x1B3C, - 0x1B42, 0x1B42, 0x1B6B, 0x1B73, 0x1DC0, 0x1DCA, 0x1DFE, 0x1DFF, - 0x200B, 0x200F, 0x202A, 0x202E, 0x2060, 0x2063, 0x206A, 0x206F, - 0x20D0, 0x20EF, 0x302A, 0x302F, 0x3099, 0x309A, 0xA806, 0xA806, - 0xA80B, 0xA80B, 0xA825, 0xA826, 0xFB1E, 0xFB1E, 0xFE00, 0xFE0F, - 0xFE20, 0xFE23, 0xFEFF, 0xFEFF, 0xFFF9, 0xFFFB, 0x10A01, 0x10A03, - 0x10A05, 0x10A06, 0x10A0C, 0x10A0F, 0x10A38, 0x10A3A, 0x10A3F, 0x10A3F, - 0x1D167, 0x1D169, 0x1D173, 0x1D182, 0x1D185, 0x1D18B, 0x1D1AA, 0x1D1AD, - 0x1D242, 0x1D244, 0xE0001, 0xE0001, 0xE0020, 0xE007F, 0xE0100, 0xE01EF, - -1, - }; - - if (in_range(range1, c)) - return 0; - - static uint32_t range2[] = { - 0x1100, 0x115F, 0x2329, 0x2329, 0x232A, 0x232A, 0x2E80, 0x303E, - 0x3040, 0xA4CF, 0xAC00, 0xD7A3, 0xF900, 0xFAFF, 0xFE10, 0xFE19, - 0xFE30, 0xFE6F, 0xFF00, 0xFF60, 0xFFE0, 0xFFE6, 0x1F000, 0x1F644, - 0x20000, 0x2FFFD, 0x30000, 0x3FFFD, -1, - }; - - if (in_range(range2, c)) - return 2; - return 1; -} - -// Returns the number of columns needed to display a given -// string in a fixed-width font. -int display_width(char *p, int len) { - char *start = p; - int w = 0; - while (p - start < len) { - uint32_t c = decode_utf8(&p, p); - w += char_width(c); - } - return w; -} -- cgit v1.2.3-54-g00ecf