summaryrefslogtreecommitdiff
path: root/src/3p/chibicc
diff options
context:
space:
mode:
Diffstat (limited to 'src/3p/chibicc')
-rw-r--r--src/3p/chibicc/LICENSE21
-rw-r--r--src/3p/chibicc/chibicc.h264
-rw-r--r--src/3p/chibicc/hashmap.c137
-rw-r--r--src/3p/chibicc/strings.c17
-rw-r--r--src/3p/chibicc/tokenize.c785
-rw-r--r--src/3p/chibicc/unicode.c189
6 files changed, 0 insertions, 1413 deletions
diff --git a/src/3p/chibicc/LICENSE b/src/3p/chibicc/LICENSE
deleted file mode 100644
index 2d1fd94..0000000
--- a/src/3p/chibicc/LICENSE
+++ /dev/null
@@ -1,21 +0,0 @@
-MIT License
-
-Copyright (c) 2019 Rui Ueyama
-
-Permission is hereby granted, free of charge, to any person obtaining a copy
-of this software and associated documentation files (the "Software"), to deal
-in the Software without restriction, including without limitation the rights
-to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-copies of the Software, and to permit persons to whom the Software is
-furnished to do so, subject to the following conditions:
-
-The above copyright notice and this permission notice shall be included in all
-copies or substantial portions of the Software.
-
-THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-SOFTWARE.
diff --git a/src/3p/chibicc/chibicc.h b/src/3p/chibicc/chibicc.h
deleted file mode 100644
index f3f87ab..0000000
--- a/src/3p/chibicc/chibicc.h
+++ /dev/null
@@ -1,264 +0,0 @@
-// include guards: upstream doesn't have these but we add them so we can cat
-// source files together (or #include them, in particular)
-#ifndef INC_CHIBICC_H
-#define INC_CHIBICC_H
-
-#include <assert.h>
-#include <ctype.h>
-#include <errno.h>
-#include <stdarg.h>
-//#include <stdbool.h>
-#include <stdint.h>
-#include <stdio.h>
-#include <stdlib.h>
-// mike: stdnoreturn means we can't use our noreturn (_Noreturn void)
-// there are no noreturns in tokenize.c anyway, and the ones in this header have
-// been changed to just _Noreturn to avoid any possible conflict
-//#include <stdnoreturn.h>
-#include <string.h>
-
-// exists on all Unixes but normally hidden behind _GNU_SOURCE on Linux.
-// missing entirely on Windows (implemented in 3p/openbsd/asprintf.c for compat)
-int vasprintf(char **str, const char *fmt, va_list ap);
-
-#if !defined(__GNUC__) && !defined(__clang__)
-# define __attribute__(x)
-#endif
-
-typedef struct Type Type;
-typedef struct Member Member;
-typedef struct Node Node;
-typedef struct Hideset Hideset;
-
-//
-// strings.c
-//
-
-typedef struct {
- char **data;
- int capacity;
- int len;
-} StringArray;
-
-void strarray_push(StringArray *arr, char *s);
-
-//
-// tokenize.c
-//
-
-// Token
-typedef enum {
- TK_IDENT, // Identifiers
- TK_PUNCT, // Punctuators
- TK_KEYWORD, // Keywords
- TK_STR, // String literals
- TK_NUM, // Numeric literals
- TK_PP_NUM, // Preprocessing numbers
- TK_EOF, // End-of-file markers
-} TokenKind;
-
-typedef struct {
- char *name;
- int file_no;
- char *contents;
-
- // For #line directive
- char *display_name;
- int line_delta;
-} File;
-
-// Token type
-typedef struct Token Token;
-struct Token {
- TokenKind kind; // Token kind
- Token *next; // Next token
- int64_t val; // If kind is TK_NUM, its value
- long double fval; // If kind is TK_NUM, its value
- char *loc; // Token location
- int len; // Token length
- Type *ty; // Used if TK_NUM or TK_STR
- char *str; // String literal contents including terminating '\0'
-
- File *file; // Source location
- char *filename; // Filename
- int line_no; // Line number
- int line_delta; // Line number
- bool at_bol; // True if this token is at beginning of line
- bool has_space; // True if this token follows a space character
- Hideset *hideset; // For macro expansion
- Token *origin; // If this is expanded from a macro, the original token
-};
-
-_Noreturn void error(char *fmt, ...) __attribute__((format(printf, 1, 2)));
-_Noreturn void error_at(char *loc, char *fmt, ...) __attribute__((format(printf, 2, 3)));
-_Noreturn void error_tok(Token *tok, char *fmt, ...) __attribute__((format(printf, 2, 3)));
-void warn_tok(Token *tok, char *fmt, ...) __attribute__((format(printf, 2, 3)));
-bool equal(const Token *tok, const char *op);
-Token *skip(Token *tok, char *op);
-bool consume(Token **rest, Token *tok, char *str);
-void convert_pp_tokens(Token *tok);
-File **get_input_files(void);
-File *new_file(char *name, int file_no, char *contents);
-Token *tokenize_string_literal(Token *tok, Type *basety);
-Token *tokenize(File *file);
-//Token *tokenize_file(char *filename);
-Token *tokenize_buf(const char *name, char *p);
-
-// note: replacing memstream-based format with asprintf version. moved down here
-// as error() is declared above.
-//char *format(char *fmt, ...) __attribute__((format(printf, 1, 2)));
-__attribute__((format(printf, 1, 2)))
-static inline char *format(const char *fmt, ...) {
- char *ret;
- va_list va;
- va_start(va, fmt);
- if (vasprintf(&ret, fmt, va) == -1) error("couldn't allocate memory");
- va_end(va);
- return ret;
-}
-
-//
-// type.c
-//
-
-typedef enum {
- TY_VOID,
- TY_BOOL,
- TY_CHAR,
- TY_SHORT,
- TY_INT,
- TY_LONG,
- TY_FLOAT,
- TY_DOUBLE,
- TY_LDOUBLE,
- TY_ENUM,
- TY_PTR,
- TY_FUNC,
- TY_ARRAY,
- TY_VLA, // variable-length array
- TY_STRUCT,
- TY_UNION,
-} TypeKind;
-
-struct Type {
- TypeKind kind;
- int size; // sizeof() value
- int align; // alignment
- bool is_unsigned; // unsigned or signed
- bool is_atomic; // true if _Atomic
- Type *origin; // for type compatibility check
-
- // Pointer-to or array-of type. We intentionally use the same member
- // to represent pointer/array duality in C.
- //
- // In many contexts in which a pointer is expected, we examine this
- // member instead of "kind" member to determine whether a type is a
- // pointer or not. That means in many contexts "array of T" is
- // naturally handled as if it were "pointer to T", as required by
- // the C spec.
- Type *base;
-
- // Declaration
- Token *name;
- Token *name_pos;
-
- // Array
- int array_len;
-
- // Variable-length array
- //Node *vla_len; // # of elements
- //Obj *vla_size; // sizeof() value
-
- // Struct
- Member *members;
- bool is_flexible;
- bool is_packed;
-
- // Function type
- Type *return_ty;
- Type *params;
- bool is_variadic;
- Type *next;
-};
-
-// Struct member
-struct Member {
- Member *next;
- Type *ty;
- Token *tok; // for error message
- Token *name;
- int idx;
- int align;
- int offset;
-
- // Bitfield
- bool is_bitfield;
- int bit_offset;
- int bit_width;
-};
-
-extern Type *ty_void;
-extern Type *ty_bool;
-
-extern Type *ty_char;
-extern Type *ty_short;
-extern Type *ty_int;
-extern Type *ty_long;
-
-extern Type *ty_uchar;
-extern Type *ty_ushort;
-extern Type *ty_uint;
-extern Type *ty_ulong;
-
-extern Type *ty_float;
-extern Type *ty_double;
-extern Type *ty_ldouble;
-
-bool is_integer(Type *ty);
-bool is_flonum(Type *ty);
-bool is_numeric(Type *ty);
-bool is_compatible(Type *t1, Type *t2);
-Type *copy_type(Type *ty);
-Type *pointer_to(Type *base);
-Type *func_type(Type *return_ty);
-Type *array_of(Type *base, int size);
-Type *vla_of(Type *base, Node *expr);
-Type *enum_type(void);
-Type *struct_type(void);
-void add_type(Node *node);
-
-//
-// unicode.c
-//
-
-int encode_utf8(char *buf, uint32_t c);
-uint32_t decode_utf8(char **new_pos, char *p);
-bool is_ident1(uint32_t c);
-bool is_ident2(uint32_t c);
-int display_width(char *p, int len);
-
-//
-// hashmap.c
-//
-
-typedef struct {
- char *key;
- int keylen;
- void *val;
-} HashEntry;
-
-typedef struct {
- HashEntry *buckets;
- int capacity;
- int used;
-} HashMap;
-
-void *hashmap_get(HashMap *map, char *key);
-void *hashmap_get2(HashMap *map, char *key, int keylen);
-void hashmap_put(HashMap *map, char *key, void *val);
-void hashmap_put2(HashMap *map, char *key, int keylen, void *val);
-void hashmap_delete(HashMap *map, char *key);
-void hashmap_delete2(HashMap *map, char *key, int keylen);
-void hashmap_test(void);
-
-#endif
diff --git a/src/3p/chibicc/hashmap.c b/src/3p/chibicc/hashmap.c
deleted file mode 100644
index 2090274..0000000
--- a/src/3p/chibicc/hashmap.c
+++ /dev/null
@@ -1,137 +0,0 @@
-// This is an implementation of the open-addressing hash table.
-
-#include "chibicc.h"
-
-// mike: moved from chibicc.h and also renamed, to avoid conflicts with langext.h
-#define chibi_unreachable() error("internal error at %s:%d", __FILE__, __LINE__)
-
-// Initial hash bucket size
-#define INIT_SIZE 16
-
-// Rehash if the usage exceeds 70%.
-#define HIGH_WATERMARK 70
-
-// We'll keep the usage below 50% after rehashing.
-#define LOW_WATERMARK 50
-
-// Represents a deleted hash entry
-#define TOMBSTONE ((void *)-1)
-
-static uint64_t fnv_hash(char *s, int len) {
- uint64_t hash = 0xcbf29ce484222325;
- for (int i = 0; i < len; i++) {
- hash *= 0x100000001b3;
- hash ^= (unsigned char)s[i];
- }
- return hash;
-}
-
-// Make room for new entires in a given hashmap by removing
-// tombstones and possibly extending the bucket size.
-static void rehash(HashMap *map) {
- // Compute the size of the new hashmap.
- int nkeys = 0;
- for (int i = 0; i < map->capacity; i++)
- if (map->buckets[i].key && map->buckets[i].key != TOMBSTONE)
- nkeys++;
-
- int cap = map->capacity;
- while ((nkeys * 100) / cap >= LOW_WATERMARK)
- cap = cap * 2;
- assert(cap > 0);
-
- // Create a new hashmap and copy all key-values.
- HashMap map2 = {0};
- map2.buckets = calloc(cap, sizeof(HashEntry));
- map2.capacity = cap;
-
- for (int i = 0; i < map->capacity; i++) {
- HashEntry *ent = &map->buckets[i];
- if (ent->key && ent->key != TOMBSTONE)
- hashmap_put2(&map2, ent->key, ent->keylen, ent->val);
- }
-
- assert(map2.used == nkeys);
- *map = map2;
-}
-
-static bool match(HashEntry *ent, char *key, int keylen) {
- return ent->key && ent->key != TOMBSTONE &&
- ent->keylen == keylen && memcmp(ent->key, key, keylen) == 0;
-}
-
-static HashEntry *get_entry(HashMap *map, char *key, int keylen) {
- if (!map->buckets)
- return NULL;
-
- uint64_t hash = fnv_hash(key, keylen);
-
- for (int i = 0; i < map->capacity; i++) {
- HashEntry *ent = &map->buckets[(hash + i) % map->capacity];
- if (match(ent, key, keylen))
- return ent;
- if (ent->key == NULL)
- return NULL;
- }
- chibi_unreachable();
-}
-
-static HashEntry *get_or_insert_entry(HashMap *map, char *key, int keylen) {
- if (!map->buckets) {
- map->buckets = calloc(INIT_SIZE, sizeof(HashEntry));
- map->capacity = INIT_SIZE;
- } else if ((map->used * 100) / map->capacity >= HIGH_WATERMARK) {
- rehash(map);
- }
-
- uint64_t hash = fnv_hash(key, keylen);
-
- for (int i = 0; i < map->capacity; i++) {
- HashEntry *ent = &map->buckets[(hash + i) % map->capacity];
-
- if (match(ent, key, keylen))
- return ent;
-
- if (ent->key == TOMBSTONE) {
- ent->key = key;
- ent->keylen = keylen;
- return ent;
- }
-
- if (ent->key == NULL) {
- ent->key = key;
- ent->keylen = keylen;
- map->used++;
- return ent;
- }
- }
- chibi_unreachable();
-}
-
-void *hashmap_get(HashMap *map, char *key) {
- return hashmap_get2(map, key, strlen(key));
-}
-
-void *hashmap_get2(HashMap *map, char *key, int keylen) {
- HashEntry *ent = get_entry(map, key, keylen);
- return ent ? ent->val : NULL;
-}
-
-void hashmap_put(HashMap *map, char *key, void *val) {
- hashmap_put2(map, key, strlen(key), val);
-}
-
-void hashmap_put2(HashMap *map, char *key, int keylen, void *val) {
- HashEntry *ent = get_or_insert_entry(map, key, keylen);
- ent->val = val;
-}
-
-void hashmap_delete(HashMap *map, char *key) {
- hashmap_delete2(map, key, strlen(key));
-}
-
-void hashmap_delete2(HashMap *map, char *key, int keylen) {
- HashEntry *ent = get_entry(map, key, keylen);
- if (ent)
- ent->key = TOMBSTONE;
-}
diff --git a/src/3p/chibicc/strings.c b/src/3p/chibicc/strings.c
deleted file mode 100644
index 0538fef..0000000
--- a/src/3p/chibicc/strings.c
+++ /dev/null
@@ -1,17 +0,0 @@
-#include "chibicc.h"
-
-void strarray_push(StringArray *arr, char *s) {
- if (!arr->data) {
- arr->data = calloc(8, sizeof(char *));
- arr->capacity = 8;
- }
-
- if (arr->capacity == arr->len) {
- arr->data = realloc(arr->data, sizeof(char *) * arr->capacity * 2);
- arr->capacity *= 2;
- for (int i = arr->len; i < arr->capacity; i++)
- arr->data[i] = NULL;
- }
-
- arr->data[arr->len++] = s;
-}
diff --git a/src/3p/chibicc/tokenize.c b/src/3p/chibicc/tokenize.c
deleted file mode 100644
index 6738206..0000000
--- a/src/3p/chibicc/tokenize.c
+++ /dev/null
@@ -1,785 +0,0 @@
-#include "chibicc.h"
-
-#ifdef _WIN32
-#define strncasecmp _strnicmp
-#endif
-
-// Input file
-static File *current_file;
-
-// A list of all input files.
-static File **input_files;
-
-// True if the current position is at the beginning of a line
-static bool at_bol;
-
-// True if the current position follows a space character
-static bool has_space;
-
-// Reports an error and exit.
-void error(char *fmt, ...) {
- va_list ap;
- va_start(ap, fmt);
- fprintf(stderr, "cmeta: chibicc: ");
- vfprintf(stderr, fmt, ap);
- fprintf(stderr, "\n");
- exit(1);
-}
-
-// Reports an error message in the following format.
-//
-// foo.c:10: x = y + 1;
-// ^ <error message here>
-static void verror_at(char *filename, char *input, int line_no,
- char *loc, char *fmt, va_list ap) {
- // Find a line containing `loc`.
- char *line = loc;
- while (input < line && line[-1] != '\n')
- line--;
-
- char *end = loc;
- while (*end && *end != '\n')
- end++;
-
- // Print out the line.
- int indent = fprintf(stderr, "%s:%d: ", filename, line_no);
- fprintf(stderr, "%.*s\n", (int)(end - line), line);
-
- // Show the error message.
- int pos = display_width(line, loc - line) + indent;
-
- fprintf(stderr, "%*s", pos, ""); // print pos spaces.
- fprintf(stderr, "^ ");
- vfprintf(stderr, fmt, ap);
- fprintf(stderr, "\n");
-}
-
-void error_at(char *loc, char *fmt, ...) {
- int line_no = 1;
- for (char *p = current_file->contents; p < loc; p++)
- if (*p == '\n')
- line_no++;
-
- va_list ap;
- va_start(ap, fmt);
- verror_at(current_file->name, current_file->contents, line_no, loc, fmt, ap);
- exit(1);
-}
-
-void error_tok(Token *tok, char *fmt, ...) {
- va_list ap;
- va_start(ap, fmt);
- verror_at(tok->file->name, tok->file->contents, tok->line_no, tok->loc, fmt, ap);
- exit(1);
-}
-
-void warn_tok(Token *tok, char *fmt, ...) {
- va_list ap;
- va_start(ap, fmt);
- verror_at(tok->file->name, tok->file->contents, tok->line_no, tok->loc, fmt, ap);
- va_end(ap);
-}
-
-// Consumes the current token if it matches `op`.
-bool equal(const Token *tok, const char *op) {
- return memcmp(tok->loc, op, tok->len) == 0 && op[tok->len] == '\0';
-}
-
-// Ensure that the current token is `op`.
-Token *skip(Token *tok, char *op) {
- if (!equal(tok, op))
- error_tok(tok, "expected '%s'", op);
- return tok->next;
-}
-
-bool consume(Token **rest, Token *tok, char *str) {
- if (equal(tok, str)) {
- *rest = tok->next;
- return true;
- }
- *rest = tok;
- return false;
-}
-
-// Create a new token.
-static Token *new_token(TokenKind kind, char *start, char *end) {
- Token *tok = calloc(1, sizeof(Token));
- tok->kind = kind;
- tok->loc = start;
- tok->len = end - start;
- tok->file = current_file;
- tok->filename = current_file->display_name;
- tok->at_bol = at_bol;
- tok->has_space = has_space;
-
- at_bol = has_space = false;
- return tok;
-}
-
-static bool startswith(char *p, char *q) {
- return strncmp(p, q, strlen(q)) == 0;
-}
-
-// Read an identifier and returns the length of it.
-// If p does not point to a valid identifier, 0 is returned.
-static int read_ident(char *start) {
- char *p = start;
- uint32_t c = decode_utf8(&p, p);
- if (!is_ident1(c))
- return 0;
-
- for (;;) {
- char *q;
- c = decode_utf8(&q, p);
- if (!is_ident2(c))
- return p - start;
- p = q;
- }
-}
-
-static int from_hex(char c) {
- if ('0' <= c && c <= '9')
- return c - '0';
- if ('a' <= c && c <= 'f')
- return c - 'a' + 10;
- return c - 'A' + 10;
-}
-
-// Read a punctuator token from p and returns its length.
-static int read_punct(char *p) {
- static char *kw[] = {
- "<<=", ">>=", "...", "==", "!=", "<=", ">=", "->", "+=",
- "-=", "*=", "/=", "++", "--", "%=", "&=", "|=", "^=", "&&",
- "||", "<<", ">>", "##",
- };
-
- for (int i = 0; i < sizeof(kw) / sizeof(*kw); i++)
- if (startswith(p, kw[i]))
- return strlen(kw[i]);
-
- return ispunct(*p) ? 1 : 0;
-}
-
-static bool is_keyword(Token *tok) {
- static HashMap map;
-
- if (map.capacity == 0) {
- static char *kw[] = {
- "return", "if", "else", "for", "while", "int", "sizeof", "char",
- "struct", "union", "short", "long", "void", "typedef", "_Bool",
- "enum", "static", "goto", "break", "continue", "switch", "case",
- "default", "extern", "_Alignof", "_Alignas", "do", "signed",
- "unsigned", "const", "volatile", "auto", "register", "restrict",
- "__restrict", "__restrict__", "_Noreturn", "float", "double",
- "typeof", "asm", "_Thread_local", "__thread", "_Atomic",
- "__attribute__",
- };
-
- for (int i = 0; i < sizeof(kw) / sizeof(*kw); i++)
- hashmap_put(&map, kw[i], (void *)1);
- }
-
- return hashmap_get2(&map, tok->loc, tok->len);
-}
-
-static int read_escaped_char(char **new_pos, char *p) {
- if ('0' <= *p && *p <= '7') {
- // Read an octal number.
- int c = *p++ - '0';
- if ('0' <= *p && *p <= '7') {
- c = (c << 3) + (*p++ - '0');
- if ('0' <= *p && *p <= '7')
- c = (c << 3) + (*p++ - '0');
- }
- *new_pos = p;
- return c;
- }
-
- if (*p == 'x') {
- // Read a hexadecimal number.
- p++;
- if (!isxdigit(*p))
- error_at(p, "invalid hex escape sequence");
-
- int c = 0;
- for (; isxdigit(*p); p++)
- c = (c << 4) + from_hex(*p);
- *new_pos = p;
- return c;
- }
-
- *new_pos = p + 1;
-
- // Escape sequences are defined using themselves here. E.g.
- // '\n' is implemented using '\n'. This tautological definition
- // works because the compiler that compiles our compiler knows
- // what '\n' actually is. In other words, we "inherit" the ASCII
- // code of '\n' from the compiler that compiles our compiler,
- // so we don't have to teach the actual code here.
- //
- // This fact has huge implications not only for the correctness
- // of the compiler but also for the security of the generated code.
- // For more info, read "Reflections on Trusting Trust" by Ken Thompson.
- // https://github.com/rui314/chibicc/wiki/thompson1984.pdf
- switch (*p) {
- case 'a': return '\a';
- case 'b': return '\b';
- case 't': return '\t';
- case 'n': return '\n';
- case 'v': return '\v';
- case 'f': return '\f';
- case 'r': return '\r';
- // [GNU] \e for the ASCII escape character is a GNU C extension.
- case 'e': return 27;
- default: return *p;
- }
-}
-
-// Find a closing double-quote.
-static char *string_literal_end(char *p) {
- char *start = p;
- for (; *p != '"'; p++) {
- if (*p == '\n' || *p == '\0')
- error_at(start, "unclosed string literal");
- if (*p == '\\')
- p++;
- }
- return p;
-}
-
-static Token *read_string_literal(char *start, char *quote) {
- char *end = string_literal_end(quote + 1);
- char *buf = calloc(1, end - quote);
- int len = 0;
-
- for (char *p = quote + 1; p < end;) {
- if (*p == '\\')
- buf[len++] = read_escaped_char(&p, p + 1);
- else
- buf[len++] = *p++;
- }
-
- Token *tok = new_token(TK_STR, start, end + 1);
- tok->ty = array_of(ty_char, len + 1);
- tok->str = buf;
- return tok;
-}
-
-// Read a UTF-8-encoded string literal and transcode it in UTF-16.
-//
-// UTF-16 is yet another variable-width encoding for Unicode. Code
-// points smaller than U+10000 are encoded in 2 bytes. Code points
-// equal to or larger than that are encoded in 4 bytes. Each 2 bytes
-// in the 4 byte sequence is called "surrogate", and a 4 byte sequence
-// is called a "surrogate pair".
-static Token *read_utf16_string_literal(char *start, char *quote) {
- char *end = string_literal_end(quote + 1);
- uint16_t *buf = calloc(2, end - start);
- int len = 0;
-
- for (char *p = quote + 1; p < end;) {
- if (*p == '\\') {
- buf[len++] = read_escaped_char(&p, p + 1);
- continue;
- }
-
- uint32_t c = decode_utf8(&p, p);
- if (c < 0x10000) {
- // Encode a code point in 2 bytes.
- buf[len++] = c;
- } else {
- // Encode a code point in 4 bytes.
- c -= 0x10000;
- buf[len++] = 0xd800 + ((c >> 10) & 0x3ff);
- buf[len++] = 0xdc00 + (c & 0x3ff);
- }
- }
-
- Token *tok = new_token(TK_STR, start, end + 1);
- tok->ty = array_of(ty_ushort, len + 1);
- tok->str = (char *)buf;
- return tok;
-}
-
-// Read a UTF-8-encoded string literal and transcode it in UTF-32.
-//
-// UTF-32 is a fixed-width encoding for Unicode. Each code point is
-// encoded in 4 bytes.
-static Token *read_utf32_string_literal(char *start, char *quote, Type *ty) {
- char *end = string_literal_end(quote + 1);
- uint32_t *buf = calloc(4, end - quote);
- int len = 0;
-
- for (char *p = quote + 1; p < end;) {
- if (*p == '\\')
- buf[len++] = read_escaped_char(&p, p + 1);
- else
- buf[len++] = decode_utf8(&p, p);
- }
-
- Token *tok = new_token(TK_STR, start, end + 1);
- tok->ty = array_of(ty, len + 1);
- tok->str = (char *)buf;
- return tok;
-}
-
-static Token *read_char_literal(char *start, char *quote, Type *ty) {
- char *p = quote + 1;
- if (*p == '\0')
- error_at(start, "unclosed char literal");
-
- int c;
- if (*p == '\\')
- c = read_escaped_char(&p, p + 1);
- else
- c = decode_utf8(&p, p);
-
- char *end = p;
- for (; *end != '\''; ++end) {
- if (!*end || *end == '\n')
- error_at(p, "unclosed char literal");
- }
-
- Token *tok = new_token(TK_NUM, start, end + 1);
- tok->val = c;
- tok->ty = ty;
- return tok;
-}
-
-static bool convert_pp_int(Token *tok) {
- char *p = tok->loc;
-
- // Read a binary, octal, decimal or hexadecimal number.
- int base = 10;
- if (!strncasecmp(p, "0x", 2) && isxdigit(p[2])) {
- p += 2;
- base = 16;
- } else if (!strncasecmp(p, "0b", 2) && (p[2] == '0' || p[2] == '1')) {
- p += 2;
- base = 2;
- } else if (*p == '0') {
- base = 8;
- }
-
- int64_t val = strtoul(p, &p, base);
-
- // Read U, L or LL suffixes.
- bool l = false;
- bool u = false;
-
- if (startswith(p, "LLU") || startswith(p, "LLu") ||
- startswith(p, "llU") || startswith(p, "llu") ||
- startswith(p, "ULL") || startswith(p, "Ull") ||
- startswith(p, "uLL") || startswith(p, "ull")) {
- p += 3;
- l = u = true;
- } else if (!strncasecmp(p, "lu", 2) || !strncasecmp(p, "ul", 2)) {
- p += 2;
- l = u = true;
- } else if (startswith(p, "LL") || startswith(p, "ll")) {
- p += 2;
- l = true;
- } else if (*p == 'L' || *p == 'l') {
- p++;
- l = true;
- } else if (*p == 'U' || *p == 'u') {
- p++;
- u = true;
- }
-
- if (p != tok->loc + tok->len)
- return false;
-
- // Infer a type.
- Type *ty;
- if (base == 10) {
- if (l && u)
- ty = ty_ulong;
- else if (l)
- ty = ty_long;
- else if (u)
- ty = (val >> 32) ? ty_ulong : ty_uint;
- else
- ty = (val >> 31) ? ty_long : ty_int;
- } else {
- if (l && u)
- ty = ty_ulong;
- else if (l)
- ty = (val >> 63) ? ty_ulong : ty_long;
- else if (u)
- ty = (val >> 32) ? ty_ulong : ty_uint;
- else if (val >> 63)
- ty = ty_ulong;
- else if (val >> 32)
- ty = ty_long;
- else if (val >> 31)
- ty = ty_uint;
- else
- ty = ty_int;
- }
-
- tok->kind = TK_NUM;
- tok->val = val;
- tok->ty = ty;
- return true;
-}
-
-// The definition of the numeric literal at the preprocessing stage
-// is more relaxed than the definition of that at the later stages.
-// In order to handle that, a numeric literal is tokenized as a
-// "pp-number" token first and then converted to a regular number
-// token after preprocessing.
-//
-// This function converts a pp-number token to a regular number token.
-static void convert_pp_number(Token *tok) {
- // Try to parse as an integer constant.
- if (convert_pp_int(tok))
- return;
-
- // If it's not an integer, it must be a floating point constant.
- char *end;
- long double val = strtold(tok->loc, &end);
-
- Type *ty;
- if (*end == 'f' || *end == 'F') {
- ty = ty_float;
- end++;
- } else if (*end == 'l' || *end == 'L') {
- ty = ty_ldouble;
- end++;
- } else {
- ty = ty_double;
- }
-
- if (tok->loc + tok->len != end)
- error_tok(tok, "invalid numeric constant");
-
- tok->kind = TK_NUM;
- tok->fval = val;
- tok->ty = ty;
-}
-
-void convert_pp_tokens(Token *tok) {
- for (Token *t = tok; t->kind != TK_EOF; t = t->next) {
- if (is_keyword(t))
- t->kind = TK_KEYWORD;
- else if (t->kind == TK_PP_NUM)
- convert_pp_number(t);
- }
-}
-
-// Initialize line info for all tokens.
-static void add_line_numbers(Token *tok) {
- char *p = current_file->contents;
- int n = 1;
-
- do {
- if (p == tok->loc) {
- tok->line_no = n;
- tok = tok->next;
- }
- if (*p == '\n')
- n++;
- } while (*p++);
-}
-
-Token *tokenize_string_literal(Token *tok, Type *basety) {
- Token *t;
- if (basety->size == 2)
- t = read_utf16_string_literal(tok->loc, tok->loc);
- else
- t = read_utf32_string_literal(tok->loc, tok->loc, basety);
- t->next = tok->next;
- return t;
-}
-
-// Tokenize a given string and returns new tokens.
-Token *tokenize(File *file) {
- current_file = file;
-
- char *p = file->contents;
- Token head = {0};
- Token *cur = &head;
-
- at_bol = true;
- has_space = false;
-
- while (*p) {
- // Skip line comments.
- if (startswith(p, "//")) {
- p += 2;
- while (*p != '\n')
- p++;
- has_space = true;
- continue;
- }
-
- // Skip block comments.
- if (startswith(p, "/*")) {
- char *q = strstr(p + 2, "*/");
- if (!q)
- error_at(p, "unclosed block comment");
- p = q + 2;
- has_space = true;
- continue;
- }
-
- // Skip newline.
- if (*p == '\n') {
- p++;
- at_bol = true;
- has_space = false;
- continue;
- }
-
- // Skip whitespace characters.
- if (isspace(*p)) {
- p++;
- has_space = true;
- continue;
- }
-
- // Numeric literal
- if (isdigit(*p) || (*p == '.' && isdigit(p[1]))) {
- char *q = p++;
- for (;;) {
- if (p[0] && p[1] && strchr("eEpP", p[0]) && strchr("+-", p[1]))
- p += 2;
- else if (isalnum(*p) || *p == '.')
- p++;
- else
- break;
- }
- cur = cur->next = new_token(TK_PP_NUM, q, p);
- continue;
- }
-
- // String literal
- if (*p == '"') {
- cur = cur->next = read_string_literal(p, p);
- p += cur->len;
- continue;
- }
-
- // UTF-8 string literal
- if (startswith(p, "u8\"")) {
- cur = cur->next = read_string_literal(p, p + 2);
- p += cur->len;
- continue;
- }
-
- // UTF-16 string literal
- if (startswith(p, "u\"")) {
- cur = cur->next = read_utf16_string_literal(p, p + 1);
- p += cur->len;
- continue;
- }
-
- // Wide string literal
- if (startswith(p, "L\"")) {
- cur = cur->next = read_utf32_string_literal(p, p + 1, ty_int);
- p += cur->len;
- continue;
- }
-
- // UTF-32 string literal
- if (startswith(p, "U\"")) {
- cur = cur->next = read_utf32_string_literal(p, p + 1, ty_uint);
- p += cur->len;
- continue;
- }
-
- // Character literal
- if (*p == '\'') {
- cur = cur->next = read_char_literal(p, p, ty_int);
- cur->val = (char)cur->val;
- p += cur->len;
- continue;
- }
-
- // UTF-16 character literal
- if (startswith(p, "u'")) {
- cur = cur->next = read_char_literal(p, p + 1, ty_ushort);
- cur->val &= 0xffff;
- p += cur->len;
- continue;
- }
-
- // Wide character literal
- if (startswith(p, "L'")) {
- cur = cur->next = read_char_literal(p, p + 1, ty_int);
- p += cur->len;
- continue;
- }
-
- // UTF-32 character literal
- if (startswith(p, "U'")) {
- cur = cur->next = read_char_literal(p, p + 1, ty_uint);
- p += cur->len;
- continue;
- }
-
- // Identifier or keyword
- int ident_len = read_ident(p);
- if (ident_len) {
- cur = cur->next = new_token(TK_IDENT, p, p + ident_len);
- p += cur->len;
- continue;
- }
-
- // Punctuators
- int punct_len = read_punct(p);
- if (punct_len) {
- cur = cur->next = new_token(TK_PUNCT, p, p + punct_len);
- p += cur->len;
- continue;
- }
-
- error_at(p, "invalid token");
- }
-
- cur = cur->next = new_token(TK_EOF, p, p);
- add_line_numbers(head.next);
- return head.next;
-}
-
-// Returns the contents of a given file.
-/* this function is a bit wonkily implemented, and relies on a non-windows
- * thing, so we have our own in src/cmeta.c.
-static char *read_file(char *path) {
- FILE *fp;
-
- if (strcmp(path, "-") == 0) {
- // By convention, read from stdin if a given filename is "-".
- fp = stdin;
- } else {
- fp = fopen(path, "r");
- if (!fp)
- return NULL;
- }
-
- char *buf;
- size_t buflen;
- FILE *out = open_memstream(&buf, &buflen);
-
- // Read the entire file.
- for (;;) {
- char buf2[4096];
- int n = fread(buf2, 1, sizeof(buf2), fp);
- if (n == 0)
- break;
- fwrite(buf2, 1, n, out);
- }
-
- if (fp != stdin)
- fclose(fp);
-
- // Make sure that the last line is properly terminated with '\n'.
- fflush(out);
- if (buflen == 0 || buf[buflen - 1] != '\n')
- fputc('\n', out);
- fputc('\0', out);
- fclose(out);
- return buf;
-}
-*/
-
-File **get_input_files(void) {
- return input_files;
-}
-
-File *new_file(char *name, int file_no, char *contents) {
- File *file = calloc(1, sizeof(File));
- file->name = name;
- file->display_name = name;
- file->file_no = file_no;
- file->contents = contents;
- return file;
-}
-
-// Removes backslashes followed by a newline.
-static void remove_backslash_newline(char *p) {
- int i = 0, j = 0;
-
- // We want to keep the number of newline characters so that
- // the logical line number matches the physical one.
- // This counter maintain the number of newlines we have removed.
- int n = 0;
-
- while (p[i]) {
- if (p[i] == '\\' && p[i + 1] == '\n') {
- i += 2;
- n++;
- } else if (p[i] == '\n') {
- p[j++] = p[i++];
- for (; n > 0; n--)
- p[j++] = '\n';
- } else {
- p[j++] = p[i++];
- }
- }
-
- for (; n > 0; n--)
- p[j++] = '\n';
- p[j] = '\0';
-}
-
-static uint32_t read_universal_char(char *p, int len) {
- uint32_t c = 0;
- for (int i = 0; i < len; i++) {
- if (!isxdigit(p[i]))
- return 0;
- c = (c << 4) | from_hex(p[i]);
- }
- return c;
-}
-
-// Replace \u or \U escape sequences with corresponding UTF-8 bytes.
-static void convert_universal_chars(char *p) {
- char *q = p;
-
- while (*p) {
- if (startswith(p, "\\u")) {
- uint32_t c = read_universal_char(p + 2, 4);
- if (c) {
- p += 6;
- q += encode_utf8(q, c);
- } else {
- *q++ = *p++;
- }
- } else if (startswith(p, "\\U")) {
- uint32_t c = read_universal_char(p + 2, 8);
- if (c) {
- p += 10;
- q += encode_utf8(q, c);
- } else {
- *q++ = *p++;
- }
- } else if (p[0] == '\\') {
- *q++ = *p++;
- *q++ = *p++;
- } else {
- *q++ = *p++;
- }
- }
-
- *q = '\0';
-}
-
-// NOTE modified API from upstream
-Token *tokenize_buf(const char *name, char *p) {
- remove_backslash_newline(p);
- convert_universal_chars(p);
-
- // Save the filename for assembler .file directive.
- static int file_no;
- File *file = new_file((char *)name, file_no + 1, p);
-
- // Save the filename for assembler .file directive.
- input_files = realloc(input_files, sizeof(char *) * (file_no + 2));
- input_files[file_no] = file;
- input_files[file_no + 1] = NULL;
- file_no++;
-
- return tokenize(file);
-}
diff --git a/src/3p/chibicc/unicode.c b/src/3p/chibicc/unicode.c
deleted file mode 100644
index 6db1ad7..0000000
--- a/src/3p/chibicc/unicode.c
+++ /dev/null
@@ -1,189 +0,0 @@
-#include "chibicc.h"
-
-// Encode a given character in UTF-8.
-int encode_utf8(char *buf, uint32_t c) {
- if (c <= 0x7F) {
- buf[0] = c;
- return 1;
- }
-
- if (c <= 0x7FF) {
- buf[0] = 0xC0 | (c >> 6);
- buf[1] = 0x80 | ((c >> 6) & 0x3F);
- return 2;
- }
-
- if (c <= 0xFFFF) {
- buf[0] = 0xE0 | (c >> 12);
- buf[1] = 0x80 | ((c >> 6) & 0x3F);
- buf[2] = 0x80 | (c & 0x3F);
- return 3;
- }
-
- buf[0] = 0xF0 | (c >> 18);
- buf[1] = 0x80 | ((c >> 12) & 0x3F);
- buf[2] = 0x80 | ((c >> 6) & 0x3F);
- buf[3] = 0x80 | (c & 0x3F);
- return 4;
-}
-
-// Read a UTF-8-encoded Unicode code point from a source file.
-// We assume that source files are always in UTF-8.
-//
-// UTF-8 is a variable-width encoding in which one code point is
-// encoded in one to four bytes. One byte UTF-8 code points are
-// identical to ASCII. Non-ASCII characters are encoded using more
-// than one byte.
-uint32_t decode_utf8(char **new_pos, char *p) {
- if ((unsigned char)*p < 128) {
- *new_pos = p + 1;
- return *p;
- }
-
- char *start = p;
- int len;
- uint32_t c;
-
- if ((unsigned char)*p >= 0xF0) {
- len = 4;
- c = *p & 7;
- } else if ((unsigned char)*p >= 0xE0) {
- len = 3;
- c = *p & 15;
- } else if ((unsigned char)*p >= 0xC0) {
- len = 2;
- c = *p & 31;
- } else {
- error_at(start, "invalid UTF-8 sequence");
- }
-
- for (int i = 1; i < len; i++) {
- if ((unsigned char)p[i] >> 6 != 2)
- error_at(start, "invalid UTF-8 sequence");
- c = (c << 6) | (p[i] & 63);
- }
-
- *new_pos = p + len;
- return c;
-}
-
-static bool in_range(uint32_t *range, uint32_t c) {
- for (int i = 0; range[i] != -1; i += 2)
- if (range[i] <= c && c <= range[i + 1])
- return true;
- return false;
-}
-
-// [https://www.sigbus.info/n1570#D] C11 allows not only ASCII but
-// some multibyte characters in certan Unicode ranges to be used in an
-// identifier.
-//
-// This function returns true if a given character is acceptable as
-// the first character of an identifier.
-//
-// For example, ¾ (U+00BE) is a valid identifier because characters in
-// 0x00BE-0x00C0 are allowed, while neither ⟘ (U+27D8) nor ' '
-// (U+3000, full-width space) are allowed because they are out of range.
-bool is_ident1(uint32_t c) {
- static uint32_t range[] = {
- '_', '_', 'a', 'z', 'A', 'Z', '$', '$',
- 0x00A8, 0x00A8, 0x00AA, 0x00AA, 0x00AD, 0x00AD, 0x00AF, 0x00AF,
- 0x00B2, 0x00B5, 0x00B7, 0x00BA, 0x00BC, 0x00BE, 0x00C0, 0x00D6,
- 0x00D8, 0x00F6, 0x00F8, 0x00FF, 0x0100, 0x02FF, 0x0370, 0x167F,
- 0x1681, 0x180D, 0x180F, 0x1DBF, 0x1E00, 0x1FFF, 0x200B, 0x200D,
- 0x202A, 0x202E, 0x203F, 0x2040, 0x2054, 0x2054, 0x2060, 0x206F,
- 0x2070, 0x20CF, 0x2100, 0x218F, 0x2460, 0x24FF, 0x2776, 0x2793,
- 0x2C00, 0x2DFF, 0x2E80, 0x2FFF, 0x3004, 0x3007, 0x3021, 0x302F,
- 0x3031, 0x303F, 0x3040, 0xD7FF, 0xF900, 0xFD3D, 0xFD40, 0xFDCF,
- 0xFDF0, 0xFE1F, 0xFE30, 0xFE44, 0xFE47, 0xFFFD,
- 0x10000, 0x1FFFD, 0x20000, 0x2FFFD, 0x30000, 0x3FFFD, 0x40000, 0x4FFFD,
- 0x50000, 0x5FFFD, 0x60000, 0x6FFFD, 0x70000, 0x7FFFD, 0x80000, 0x8FFFD,
- 0x90000, 0x9FFFD, 0xA0000, 0xAFFFD, 0xB0000, 0xBFFFD, 0xC0000, 0xCFFFD,
- 0xD0000, 0xDFFFD, 0xE0000, 0xEFFFD, -1,
- };
-
- return in_range(range, c);
-}
-
-// Returns true if a given character is acceptable as a non-first
-// character of an identifier.
-bool is_ident2(uint32_t c) {
- static uint32_t range[] = {
- '0', '9', '$', '$', 0x0300, 0x036F, 0x1DC0, 0x1DFF, 0x20D0, 0x20FF,
- 0xFE20, 0xFE2F, -1,
- };
-
- return is_ident1(c) || in_range(range, c);
-}
-
-// Returns the number of columns needed to display a given
-// character in a fixed-width font.
-//
-// Based on https://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c
-static int char_width(uint32_t c) {
- static uint32_t range1[] = {
- 0x0000, 0x001F, 0x007f, 0x00a0, 0x0300, 0x036F, 0x0483, 0x0486,
- 0x0488, 0x0489, 0x0591, 0x05BD, 0x05BF, 0x05BF, 0x05C1, 0x05C2,
- 0x05C4, 0x05C5, 0x05C7, 0x05C7, 0x0600, 0x0603, 0x0610, 0x0615,
- 0x064B, 0x065E, 0x0670, 0x0670, 0x06D6, 0x06E4, 0x06E7, 0x06E8,
- 0x06EA, 0x06ED, 0x070F, 0x070F, 0x0711, 0x0711, 0x0730, 0x074A,
- 0x07A6, 0x07B0, 0x07EB, 0x07F3, 0x0901, 0x0902, 0x093C, 0x093C,
- 0x0941, 0x0948, 0x094D, 0x094D, 0x0951, 0x0954, 0x0962, 0x0963,
- 0x0981, 0x0981, 0x09BC, 0x09BC, 0x09C1, 0x09C4, 0x09CD, 0x09CD,
- 0x09E2, 0x09E3, 0x0A01, 0x0A02, 0x0A3C, 0x0A3C, 0x0A41, 0x0A42,
- 0x0A47, 0x0A48, 0x0A4B, 0x0A4D, 0x0A70, 0x0A71, 0x0A81, 0x0A82,
- 0x0ABC, 0x0ABC, 0x0AC1, 0x0AC5, 0x0AC7, 0x0AC8, 0x0ACD, 0x0ACD,
- 0x0AE2, 0x0AE3, 0x0B01, 0x0B01, 0x0B3C, 0x0B3C, 0x0B3F, 0x0B3F,
- 0x0B41, 0x0B43, 0x0B4D, 0x0B4D, 0x0B56, 0x0B56, 0x0B82, 0x0B82,
- 0x0BC0, 0x0BC0, 0x0BCD, 0x0BCD, 0x0C3E, 0x0C40, 0x0C46, 0x0C48,
- 0x0C4A, 0x0C4D, 0x0C55, 0x0C56, 0x0CBC, 0x0CBC, 0x0CBF, 0x0CBF,
- 0x0CC6, 0x0CC6, 0x0CCC, 0x0CCD, 0x0CE2, 0x0CE3, 0x0D41, 0x0D43,
- 0x0D4D, 0x0D4D, 0x0DCA, 0x0DCA, 0x0DD2, 0x0DD4, 0x0DD6, 0x0DD6,
- 0x0E31, 0x0E31, 0x0E34, 0x0E3A, 0x0E47, 0x0E4E, 0x0EB1, 0x0EB1,
- 0x0EB4, 0x0EB9, 0x0EBB, 0x0EBC, 0x0EC8, 0x0ECD, 0x0F18, 0x0F19,
- 0x0F35, 0x0F35, 0x0F37, 0x0F37, 0x0F39, 0x0F39, 0x0F71, 0x0F7E,
- 0x0F80, 0x0F84, 0x0F86, 0x0F87, 0x0F90, 0x0F97, 0x0F99, 0x0FBC,
- 0x0FC6, 0x0FC6, 0x102D, 0x1030, 0x1032, 0x1032, 0x1036, 0x1037,
- 0x1039, 0x1039, 0x1058, 0x1059, 0x1160, 0x11FF, 0x135F, 0x135F,
- 0x1712, 0x1714, 0x1732, 0x1734, 0x1752, 0x1753, 0x1772, 0x1773,
- 0x17B4, 0x17B5, 0x17B7, 0x17BD, 0x17C6, 0x17C6, 0x17C9, 0x17D3,
- 0x17DD, 0x17DD, 0x180B, 0x180D, 0x18A9, 0x18A9, 0x1920, 0x1922,
- 0x1927, 0x1928, 0x1932, 0x1932, 0x1939, 0x193B, 0x1A17, 0x1A18,
- 0x1B00, 0x1B03, 0x1B34, 0x1B34, 0x1B36, 0x1B3A, 0x1B3C, 0x1B3C,
- 0x1B42, 0x1B42, 0x1B6B, 0x1B73, 0x1DC0, 0x1DCA, 0x1DFE, 0x1DFF,
- 0x200B, 0x200F, 0x202A, 0x202E, 0x2060, 0x2063, 0x206A, 0x206F,
- 0x20D0, 0x20EF, 0x302A, 0x302F, 0x3099, 0x309A, 0xA806, 0xA806,
- 0xA80B, 0xA80B, 0xA825, 0xA826, 0xFB1E, 0xFB1E, 0xFE00, 0xFE0F,
- 0xFE20, 0xFE23, 0xFEFF, 0xFEFF, 0xFFF9, 0xFFFB, 0x10A01, 0x10A03,
- 0x10A05, 0x10A06, 0x10A0C, 0x10A0F, 0x10A38, 0x10A3A, 0x10A3F, 0x10A3F,
- 0x1D167, 0x1D169, 0x1D173, 0x1D182, 0x1D185, 0x1D18B, 0x1D1AA, 0x1D1AD,
- 0x1D242, 0x1D244, 0xE0001, 0xE0001, 0xE0020, 0xE007F, 0xE0100, 0xE01EF,
- -1,
- };
-
- if (in_range(range1, c))
- return 0;
-
- static uint32_t range2[] = {
- 0x1100, 0x115F, 0x2329, 0x2329, 0x232A, 0x232A, 0x2E80, 0x303E,
- 0x3040, 0xA4CF, 0xAC00, 0xD7A3, 0xF900, 0xFAFF, 0xFE10, 0xFE19,
- 0xFE30, 0xFE6F, 0xFF00, 0xFF60, 0xFFE0, 0xFFE6, 0x1F000, 0x1F644,
- 0x20000, 0x2FFFD, 0x30000, 0x3FFFD, -1,
- };
-
- if (in_range(range2, c))
- return 2;
- return 1;
-}
-
-// Returns the number of columns needed to display a given
-// string in a fixed-width font.
-int display_width(char *p, int len) {
- char *start = p;
- int w = 0;
- while (p - start < len) {
- uint32_t c = decode_utf8(&p, p);
- w += char_width(c);
- }
- return w;
-}