diff options
| author | 2025-12-29 16:28:28 +0000 | |
|---|---|---|
| committer | 2026-02-16 19:11:24 +0000 | |
| commit | 2088b61f3d6053d3879dcfcbe91c00157d050e16 (patch) | |
| tree | 5633150ddabf0380577338bccab8b972f71f0013 | |
| parent | 8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45 (diff) | |
| download | sst-2088b61f3d6053d3879dcfcbe91c00157d050e16.tar.gz sst-2088b61f3d6053d3879dcfcbe91c00157d050e16.zip | |
Make x86 library a little bit smaller again
| -rw-r--r-- | src/chunklets/README-x86 | 9 | ||||
| -rw-r--r-- | src/chunklets/x86.c | 155 |
2 files changed, 127 insertions, 37 deletions
diff --git a/src/chunklets/README-x86 b/src/chunklets/README-x86 index cbfcb5d..e280b51 100644 --- a/src/chunklets/README-x86 +++ b/src/chunklets/README-x86 @@ -30,13 +30,14 @@ Note that GCC and Clang will generally give the best-performing output. Once the .c file is built, the public header can be consumed by virtually any C or C++ compiler, as well as probably most half-decent FFIs. -Note that the .c source file is probably C++-compatible at the moment, but this -is not guaranteed, so it's best to compile it as a C source. The header will -work fine from either language. +Note that the .c source file is not C++-compatible, only the header is. The +source file relies on array designated initialisers, which are only standardised +in C. Some C++ compilers may still allow this with a warning, but the most +straightforward thing would be to compile the source using a C compiler. == API usage == -See documentation comments in x86.h for a basic idea. Some *pro tips*: +See documentation comments in x86.h for a basic idea. == OS compatibility == diff --git a/src/chunklets/x86.c b/src/chunklets/x86.c index 23d52ae..506a6ec 100644 --- a/src/chunklets/x86.c +++ b/src/chunklets/x86.c @@ -41,6 +41,84 @@ static int mrmsib(const unsigned char *p, int addrlen) { return 1; // note: include the mrm itself in the byte count } +enum { + CLASS_UNKNOWN, + + CLASS_PFX, + CLASS_NO, + CLASS_I8, + CLASS_IW, + CLASS_IWI, + CLASS_I16, + CLASS_MRM, + CLASS_MRMI8, + CLASS_MRMIW, + CLASS_ENTER, + CLASS_CRAZY8, + CLASS_CRAZYW, + CLASS_2BYTE, + + CLASS2B_NO = 1, + CLASS2B_IW, + CLASS2B_MRM, + CLASS2B_MRMI8 +}; + +// note: could theoretically shrink these tables down to 128B each although to +// do it at compile time without the use of _BitInt(4) would be kind of tricky. +static const unsigned char classtab[256] = { +#define TABLEENT(name, val) [val] = CLASS_PFX, + X86_PREFIXES(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_NO, + X86_OPS_1BYTE_NO(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_I8, + X86_OPS_1BYTE_I8(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_IW, + X86_OPS_1BYTE_IW(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_IWI, + X86_OPS_1BYTE_IWI(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_I16, + X86_OPS_1BYTE_I16(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_MRM, + X86_OPS_1BYTE_MRM(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_MRMI8, + X86_OPS_1BYTE_MRM_I8(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS_MRMIW, + X86_OPS_1BYTE_MRM_IW(TABLEENT) +#undef TABLEENT + [X86_ENTER] = CLASS_ENTER, + [X86_CRAZY8] = CLASS_CRAZY8, + [X86_CRAZYW] = CLASS_CRAZYW, + [X86_2BYTE] = CLASS_2BYTE +}; + +static const unsigned char classtab_2b[256] = { + // we don't support any 3 byte ops for now; implement if ever needed... + [X86_3BYTE1] = CLASS_UNKNOWN, + [X86_3BYTE2] = CLASS_UNKNOWN, + [X86_3DNOW] = CLASS_UNKNOWN, +#define TABLEENT(name, val) [val] = CLASS2B_NO, + X86_OPS_2BYTE_NO(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS2B_IW, + X86_OPS_2BYTE_IW(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS2B_MRM, + X86_OPS_2BYTE_MRM(TABLEENT) +#undef TABLEENT +#define TABLEENT(name, val) [val] = CLASS2B_MRMI8, + X86_OPS_2BYTE_MRM_I8(TABLEENT) +#undef TABLEENT +}; + #ifndef X86_DISABLE_SIZE_OPT #if defined(__clang__) __attribute((minsize)) @@ -52,48 +130,59 @@ int x86_len(const unsigned char *insn) { #define CASES(name, _) case name: int pfxlen = 0, addrlen = 4, operandlen = 4; -p: switch (*insn) { - case X86_PFX_ADSZ: addrlen = 2; goto P; // bit dumb sorry - case X86_PFX_OPSZ: operandlen = 2; -P: X86_SEG_PREFIXES(CASES) - case X86_PFX_LOCK: case X86_PFX_REPN: case X86_PFX_REP: - // instruction can only be 15 bytes. this could go over, oh well, - // just don't want to loop for 8 million years - if (++pfxlen == 14) return -1; - ++insn; - goto p; - } - - switch (*insn) { - X86_OPS_1BYTE_NO(CASES) return pfxlen + 1; - X86_OPS_1BYTE_I8(CASES) operandlen = 1; - X86_OPS_1BYTE_IW(CASES) return pfxlen + 1 + operandlen; - X86_OPS_1BYTE_IWI(CASES) return pfxlen + 1 + addrlen; - X86_OPS_1BYTE_I16(CASES) return pfxlen + 3; - X86_OPS_1BYTE_MRM(CASES) return pfxlen + 1 + mrmsib(insn + 1, addrlen); - X86_OPS_1BYTE_MRM_I8(CASES) operandlen = 1; - X86_OPS_1BYTE_MRM_IW(CASES) +p: switch (classtab[*insn]) { + case CLASS_UNKNOWN: return -1; + case CLASS_PFX: + switch (*insn) { + case X86_PFX_ADSZ: addrlen = 2; goto P; // bit dumb sorry + case X86_PFX_OPSZ: operandlen = 2; +P: X86_SEG_PREFIXES(CASES) + case X86_PFX_LOCK: case X86_PFX_REPN: case X86_PFX_REP: + // instruction can only be 15 bytes. this could go over, oh + // well, just don't want to loop for 8 million years + if (++pfxlen == 14) return -1; + ++insn; + goto p; + } + case CLASS_NO: return pfxlen + 1; + case CLASS_I8: operandlen = 1; + case CLASS_IW: return pfxlen + 1 + operandlen; + case CLASS_IWI: return pfxlen + 1 + addrlen; + case CLASS_I16: return pfxlen + 3; + case CLASS_MRM: return pfxlen + 1 + mrmsib(insn + 1, addrlen); + case CLASS_MRMI8: operandlen = 1; + case CLASS_MRMIW: return pfxlen + 1 + operandlen + mrmsib(insn + 1, addrlen); - case X86_ENTER: return pfxlen + 4; - case X86_CRAZY8: operandlen = 1; - case X86_CRAZYW: + case CLASS_ENTER: return pfxlen + 4; + case CLASS_CRAZY8: operandlen = 1; + case CLASS_CRAZYW: if ((insn[1] & 0x38) >= 0x10) operandlen = 0; return pfxlen + 1 + operandlen + mrmsib(insn + 1, addrlen); - case X86_2BYTE: ++insn; goto b2; + case CLASS_2BYTE: ++insn; goto b2; } +#if defined(__GNUC__) || defined(__clang__) + __builtin_unreachable(); +#elif defined(_MSC_VER) + __assume(0); +#else return -1; +#endif -b2: switch (*insn) { - // we don't support any 3 byte ops for now, implement if ever needed... - case X86_3BYTE1: case X86_3BYTE2: case X86_3DNOW: return -1; - X86_OPS_2BYTE_NO(CASES) return pfxlen + 2; - X86_OPS_2BYTE_IW(CASES) return pfxlen + 2 + operandlen; - X86_OPS_2BYTE_MRM(CASES) return pfxlen + 2 + mrmsib(insn + 1, addrlen); - X86_OPS_2BYTE_MRM_I8(CASES) operandlen = 1; +b2: switch (classtab_2b[*insn]) { + case CLASS_UNKNOWN: return -1; + case CLASS2B_NO: return pfxlen + 2; + case CLASS2B_IW: return pfxlen + 2 + operandlen; + case CLASS2B_MRM: return pfxlen + 2 + mrmsib(insn + 1, addrlen); + case CLASS2B_MRMI8: operandlen = 1; return pfxlen + 2 + operandlen + mrmsib(insn + 1, addrlen); } - +#if defined(__GNUC__) || defined(__clang__) + __builtin_unreachable(); +#elif defined(_MSC_VER) + __assume(0); +#else return -1; +#endif #undef CASES } |
