summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorGravatar Michael Smith <mikesmiffy128@gmail.com> 2025-12-29 16:28:28 +0000
committerGravatar Michael Smith <mikesmiffy128@gmail.com> 2026-02-16 19:11:24 +0000
commit2088b61f3d6053d3879dcfcbe91c00157d050e16 (patch)
tree5633150ddabf0380577338bccab8b972f71f0013
parent8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45 (diff)
downloadsst-2088b61f3d6053d3879dcfcbe91c00157d050e16.tar.gz
sst-2088b61f3d6053d3879dcfcbe91c00157d050e16.zip
Make x86 library a little bit smaller again
-rw-r--r--src/chunklets/README-x869
-rw-r--r--src/chunklets/x86.c155
2 files changed, 127 insertions, 37 deletions
diff --git a/src/chunklets/README-x86 b/src/chunklets/README-x86
index cbfcb5d..e280b51 100644
--- a/src/chunklets/README-x86
+++ b/src/chunklets/README-x86
@@ -30,13 +30,14 @@ Note that GCC and Clang will generally give the best-performing output.
Once the .c file is built, the public header can be consumed by virtually any C
or C++ compiler, as well as probably most half-decent FFIs.
-Note that the .c source file is probably C++-compatible at the moment, but this
-is not guaranteed, so it's best to compile it as a C source. The header will
-work fine from either language.
+Note that the .c source file is not C++-compatible, only the header is. The
+source file relies on array designated initialisers, which are only standardised
+in C. Some C++ compilers may still allow this with a warning, but the most
+straightforward thing would be to compile the source using a C compiler.
== API usage ==
-See documentation comments in x86.h for a basic idea. Some *pro tips*:
+See documentation comments in x86.h for a basic idea.
== OS compatibility ==
diff --git a/src/chunklets/x86.c b/src/chunklets/x86.c
index 23d52ae..506a6ec 100644
--- a/src/chunklets/x86.c
+++ b/src/chunklets/x86.c
@@ -41,6 +41,84 @@ static int mrmsib(const unsigned char *p, int addrlen) {
return 1; // note: include the mrm itself in the byte count
}
+enum {
+ CLASS_UNKNOWN,
+
+ CLASS_PFX,
+ CLASS_NO,
+ CLASS_I8,
+ CLASS_IW,
+ CLASS_IWI,
+ CLASS_I16,
+ CLASS_MRM,
+ CLASS_MRMI8,
+ CLASS_MRMIW,
+ CLASS_ENTER,
+ CLASS_CRAZY8,
+ CLASS_CRAZYW,
+ CLASS_2BYTE,
+
+ CLASS2B_NO = 1,
+ CLASS2B_IW,
+ CLASS2B_MRM,
+ CLASS2B_MRMI8
+};
+
+// note: could theoretically shrink these tables down to 128B each although to
+// do it at compile time without the use of _BitInt(4) would be kind of tricky.
+static const unsigned char classtab[256] = {
+#define TABLEENT(name, val) [val] = CLASS_PFX,
+ X86_PREFIXES(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_NO,
+ X86_OPS_1BYTE_NO(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_I8,
+ X86_OPS_1BYTE_I8(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_IW,
+ X86_OPS_1BYTE_IW(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_IWI,
+ X86_OPS_1BYTE_IWI(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_I16,
+ X86_OPS_1BYTE_I16(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_MRM,
+ X86_OPS_1BYTE_MRM(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_MRMI8,
+ X86_OPS_1BYTE_MRM_I8(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS_MRMIW,
+ X86_OPS_1BYTE_MRM_IW(TABLEENT)
+#undef TABLEENT
+ [X86_ENTER] = CLASS_ENTER,
+ [X86_CRAZY8] = CLASS_CRAZY8,
+ [X86_CRAZYW] = CLASS_CRAZYW,
+ [X86_2BYTE] = CLASS_2BYTE
+};
+
+static const unsigned char classtab_2b[256] = {
+ // we don't support any 3 byte ops for now; implement if ever needed...
+ [X86_3BYTE1] = CLASS_UNKNOWN,
+ [X86_3BYTE2] = CLASS_UNKNOWN,
+ [X86_3DNOW] = CLASS_UNKNOWN,
+#define TABLEENT(name, val) [val] = CLASS2B_NO,
+ X86_OPS_2BYTE_NO(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS2B_IW,
+ X86_OPS_2BYTE_IW(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS2B_MRM,
+ X86_OPS_2BYTE_MRM(TABLEENT)
+#undef TABLEENT
+#define TABLEENT(name, val) [val] = CLASS2B_MRMI8,
+ X86_OPS_2BYTE_MRM_I8(TABLEENT)
+#undef TABLEENT
+};
+
#ifndef X86_DISABLE_SIZE_OPT
#if defined(__clang__)
__attribute((minsize))
@@ -52,48 +130,59 @@ int x86_len(const unsigned char *insn) {
#define CASES(name, _) case name:
int pfxlen = 0, addrlen = 4, operandlen = 4;
-p: switch (*insn) {
- case X86_PFX_ADSZ: addrlen = 2; goto P; // bit dumb sorry
- case X86_PFX_OPSZ: operandlen = 2;
-P: X86_SEG_PREFIXES(CASES)
- case X86_PFX_LOCK: case X86_PFX_REPN: case X86_PFX_REP:
- // instruction can only be 15 bytes. this could go over, oh well,
- // just don't want to loop for 8 million years
- if (++pfxlen == 14) return -1;
- ++insn;
- goto p;
- }
-
- switch (*insn) {
- X86_OPS_1BYTE_NO(CASES) return pfxlen + 1;
- X86_OPS_1BYTE_I8(CASES) operandlen = 1;
- X86_OPS_1BYTE_IW(CASES) return pfxlen + 1 + operandlen;
- X86_OPS_1BYTE_IWI(CASES) return pfxlen + 1 + addrlen;
- X86_OPS_1BYTE_I16(CASES) return pfxlen + 3;
- X86_OPS_1BYTE_MRM(CASES) return pfxlen + 1 + mrmsib(insn + 1, addrlen);
- X86_OPS_1BYTE_MRM_I8(CASES) operandlen = 1;
- X86_OPS_1BYTE_MRM_IW(CASES)
+p: switch (classtab[*insn]) {
+ case CLASS_UNKNOWN: return -1;
+ case CLASS_PFX:
+ switch (*insn) {
+ case X86_PFX_ADSZ: addrlen = 2; goto P; // bit dumb sorry
+ case X86_PFX_OPSZ: operandlen = 2;
+P: X86_SEG_PREFIXES(CASES)
+ case X86_PFX_LOCK: case X86_PFX_REPN: case X86_PFX_REP:
+ // instruction can only be 15 bytes. this could go over, oh
+ // well, just don't want to loop for 8 million years
+ if (++pfxlen == 14) return -1;
+ ++insn;
+ goto p;
+ }
+ case CLASS_NO: return pfxlen + 1;
+ case CLASS_I8: operandlen = 1;
+ case CLASS_IW: return pfxlen + 1 + operandlen;
+ case CLASS_IWI: return pfxlen + 1 + addrlen;
+ case CLASS_I16: return pfxlen + 3;
+ case CLASS_MRM: return pfxlen + 1 + mrmsib(insn + 1, addrlen);
+ case CLASS_MRMI8: operandlen = 1;
+ case CLASS_MRMIW:
return pfxlen + 1 + operandlen + mrmsib(insn + 1, addrlen);
- case X86_ENTER: return pfxlen + 4;
- case X86_CRAZY8: operandlen = 1;
- case X86_CRAZYW:
+ case CLASS_ENTER: return pfxlen + 4;
+ case CLASS_CRAZY8: operandlen = 1;
+ case CLASS_CRAZYW:
if ((insn[1] & 0x38) >= 0x10) operandlen = 0;
return pfxlen + 1 + operandlen + mrmsib(insn + 1, addrlen);
- case X86_2BYTE: ++insn; goto b2;
+ case CLASS_2BYTE: ++insn; goto b2;
}
+#if defined(__GNUC__) || defined(__clang__)
+ __builtin_unreachable();
+#elif defined(_MSC_VER)
+ __assume(0);
+#else
return -1;
+#endif
-b2: switch (*insn) {
- // we don't support any 3 byte ops for now, implement if ever needed...
- case X86_3BYTE1: case X86_3BYTE2: case X86_3DNOW: return -1;
- X86_OPS_2BYTE_NO(CASES) return pfxlen + 2;
- X86_OPS_2BYTE_IW(CASES) return pfxlen + 2 + operandlen;
- X86_OPS_2BYTE_MRM(CASES) return pfxlen + 2 + mrmsib(insn + 1, addrlen);
- X86_OPS_2BYTE_MRM_I8(CASES) operandlen = 1;
+b2: switch (classtab_2b[*insn]) {
+ case CLASS_UNKNOWN: return -1;
+ case CLASS2B_NO: return pfxlen + 2;
+ case CLASS2B_IW: return pfxlen + 2 + operandlen;
+ case CLASS2B_MRM: return pfxlen + 2 + mrmsib(insn + 1, addrlen);
+ case CLASS2B_MRMI8: operandlen = 1;
return pfxlen + 2 + operandlen + mrmsib(insn + 1, addrlen);
}
-
+#if defined(__GNUC__) || defined(__clang__)
+ __builtin_unreachable();
+#elif defined(_MSC_VER)
+ __assume(0);
+#else
return -1;
+#endif
#undef CASES
}