summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorGravatar Michael Smith <mikesmiffy128@gmail.com> 2025-12-27 03:05:05 +0000
committerGravatar Michael Smith <mikesmiffy128@gmail.com> 2026-02-16 19:11:24 +0000
commit8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45 (patch)
tree63a8ea6f645b997fd7943a58a002e66244fa49ac
parente7943fb834cbcaec585ed68e67b5f17ceda504f7 (diff)
downloadsst-8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45.tar.gz
sst-8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45.zip
Do away with memcpy/memset/memcmp
Now everything is either explicitly a rep thing, or explicitly some other small run of reasonably efficient instructions. No library calls or function call overhead of any kind. Code size remains nice and small. Ban a few other bad libc functions while we're at it. More can be added later if we decide that's a useful idea. This will require some discipline and workarounds to cope with Clang's tendency to shove mem* calls in places you don't want them even when you tell it not to. Well, c'est la vie. I'm a little nervous that the __builtin_memmove() usage in shuntvars() could still one day emit actual library calls. It appears we're not importing memmove() from anywhere at the moment, though. So I guess it's good enough for now. Also I had to get rid of alloca() in con_.c but that was kind of bad anyway, so that's fine.
-rwxr-xr-xcompile8
-rw-r--r--compile.bat8
-rw-r--r--src/1.h37
-rw-r--r--src/ac.c16
-rw-r--r--src/build/gluegen.c3
-rw-r--r--src/con_.c16
-rw-r--r--src/con_.h7
-rw-r--r--src/democustom.c4
-rw-r--r--src/demorec.c13
-rw-r--r--src/fixes.c4
-rw-r--r--src/gameinfo.c6
-rw-r--r--src/hexcolour.c11
-rw-r--r--src/hook.c11
-rw-r--r--src/l4d1democompat.c3
-rw-r--r--src/l4daddon.c10
-rw-r--r--src/l4dmm.c2
-rw-r--r--src/langext.h3
-rw-r--r--src/mem.h155
-rw-r--r--src/os.c18
-rw-r--r--src/os.h5
-rw-r--r--src/portalcolours.c10
-rw-r--r--src/portalisg.c4
-rw-r--r--src/sst.c8
-rw-r--r--src/wincrt.c71
24 files changed, 291 insertions, 142 deletions
diff --git a/compile b/compile
index ca8fa0c..3ede107 100755
--- a/compile
+++ b/compile
@@ -28,8 +28,8 @@ if [ "$dbg" = 1 ]; then
cflags="-O0 -g3 -masm=intel -fsanitize-trap=undefined -DSST_DBG"
ldflags="-O0 -g3"
else
- cflags="-O2 -fvisibility=hidden -masm=intel"
- ldflags="-O2 -s"
+ cflags="-O2 -fvisibility=hidden -masm=intel -ffreestanding"
+ ldflags="-O2 -s -ffreestanding"
fi
objs=
@@ -40,8 +40,8 @@ cc() {
# ugly annoying special case
if [ "$_mn" = " -DMODULE_NAME=con_" ]; then _mn=" -DMODULE_NAME=con"
elif [ "$_mn" = "-DMODULE_NAME=sst" ]; then _mn=; fi
- $CC -c -flto -fpic -fno-ident $cflags $warnings -I.build/include \
- $stdflags$_mn -o ".build/${_bn%%.c}.o" "src/$1"
+ $CC -c -flto -fpic -fno-ident $cflags $warnings -include src/1.h \
+ -I.build/include $stdflags$_mn -o ".build/${_bn%%.c}.o" "src/$1"
}
ld() {
diff --git a/compile.bat b/compile.bat
index 2958186..10be904 100644
--- a/compile.bat
+++ b/compile.bat
@@ -29,8 +29,8 @@ if "%dbg%"=="1" (
set cflags=-O0 -g3 -masm=intel -fsanitize-trap=undefined -DSST_DBG
set ldflags=-O0 -g3
) else (
- set cflags=-O2 -masm=intel
- set ldflags=-O2
+ set cflags=-O2 -masm=intel -ffreestanding
+ set ldflags=-O2 -ffreestanding
)
set objs=
@@ -46,8 +46,8 @@ set dmodname= -DMODULE_NAME=%basename%
if "%dmodname%"==" -DMODULE_NAME=con_" set dmodname= -DMODULE_NAME=con
if "%dmodname%"==" -DMODULE_NAME=sst" set dmodname=
set objs=%objs% .build/%basename%%ext%.o
-%CC% -c -flto -mno-stack-arg-probe %cflags% %warnings% %stdflags% -I.build/include ^
--D_DLL%dmodname% -o .build/%basename%%ext%.o %1 || goto :end
+%CC% -c -flto -mno-stack-arg-probe %cflags% %warnings% %stdflags% -include src/1.h ^
+-I.build/include -D_DLL%dmodname% -o .build/%basename%%ext%.o %1 || goto :end
goto :eof
:main
diff --git a/src/1.h b/src/1.h
new file mode 100644
index 0000000..f356bc3
--- /dev/null
+++ b/src/1.h
@@ -0,0 +1,37 @@
+/* This file is dedicated to the public domain. */
+
+// This is a special header force-included in every source file.
+// Currently it's just used to ban certain functions from the codebase.
+// Ideally this could be done using the poison thing but that triggers a bunch
+// of anger when including Windows.h. Of course. So we just use #defines.
+
+#ifndef SST_DBG // we wrap these in debug mode, so leave them unbanned there.
+// These should not be used explicitly. Use mem_copy, mem_set and mem_cmp from
+// mem.h instead. Those are optimised for both code size and speed on x86.
+#define memcpy please_use_mem_copy_from_mem_dot_h_instead_of_memcpy
+#define memset please_use_mem_set_from_mem_dot_h_instead_of_memset
+#define memcmp please_use_mem_cmp_from_mem_dot_h_instead_of_memcmp
+#endif
+
+// TODO(opt): implement this if and when we ever need it (unlikely really)
+#define memmove sorry_memmove_needs_an_equivalent_in_mem_dot_h_not_done_yet
+
+// These functions are impossible to use correctly and probably have better
+// alternatives already within the codebase. Or, if not, then better
+// alternatives can be written easily.
+#define strcpy strcpy_is_broken_and_should_never_be_used
+#define strcat strcat_is_broken_and_should_never_be_used
+#define strncpy strncpy_is_broken_and_should_never_be_used
+#define strncat strncat_is_broken_and_should_never_be_used
+#define strtok strtok_is_broken_and_slow_and_should_never_be_used
+#define strtok_r strtok_r_is_broken_and_slow_and_should_never_be_used
+// Windows headers throw a fit when redefining these, lol. The following two
+// effectively prevent their use anyway since they're inline functions in
+// stdio.h. Dunno about on Linux, but we compile first on Windows anyway innit.
+//#define sprintf sprintf_is_dangerous_and_slow_and_should_never_be_used
+//#define snprintf snprintf_is_dangerous_and_slow_and_should_never_be_used
+#define vsprintf vsprintf_is_dangerous_and_slow_and_should_never_be_used
+// XXX: have some straggling vsnprintf usage I'd like to get rid of...
+//#define vsnprintf vsnprintf_is_dangerous_and_slow_and_should_never_be_used
+
+// vi: sw=4 ts=4 noet tw=80 cc=80
diff --git a/src/ac.c b/src/ac.c
index f5a2de5..fc1d318 100644
--- a/src/ac.c
+++ b/src/ac.c
@@ -111,11 +111,11 @@ static ssize __stdcall kproc(int code, usize wp, ssize lp) {
// something like the following, but with a proper abstraction...
//uchar buf[28 + 16], *p = buf;
//msg_putasz4(p, 2); p += 1;
- // msg_putssz5(p, 8); memcpy(p + 1, "FakeKey", 7); p += 8;
+ // msg_putssz5(p, 8); mem_copy(p + 1, "FakeKey", 7); p += 8;
// msg_putmsz4(p, 2); p += 1;
- // msg_putssz5(p, 3); memcpy(p + 1, "vk", 2); p += 3;
+ // msg_putssz5(p, 3); mem_copy(p + 1, "vk", 2); p += 3;
// p += msg_putu32(p, data->vkCode);
- // msg_putssz5(p, 3); memcpy(p + 1, "scan", 4); p += 5;
+ // msg_putssz5(p, 3); mem_copy(p + 1, "scan", 4); p += 5;
// p += msg_putu32(p, data->scanCode);
//++keybox->nonce;
//// append mac at end of message
@@ -297,14 +297,14 @@ static void hookdest_Key_Event(struct inputevent *ev) {
// TODO(rta): do something interesting with button data
//uchar buf[28], *p = buf;
//msg_putasz4(p, 2); p += 1;
- // msg_putssz5(p, 8); memcpy(p + 1, "KeyInput", 8); p += 9;
+ // msg_putssz5(p, 8); mem_copy(p + 1, "KeyInput", 8); p += 9;
// msg_putmsz4(p, 2); p += 1;
- // msg_putssz5(p, 3); memcpy(p + 1, "key", 3); p += 4;
+ // msg_putssz5(p, 3); mem_copy(p + 1, "key", 3); p += 4;
// p += msg_puts32(p, ev->data);
- // msg_putssz5(p, 3); memcpy(p + 1, "btn", 3); p += 4;
+ // msg_putssz5(p, 3); mem_copy(p + 1, "btn", 3); p += 4;
// int idx = ev->type - BTNDOWN;
// msg_putssz5(p++, desclen[idx]);
- // memcpy(p, desc[idx], desclen[idx]); p += desclen[idx];
+ // mem_copy(p, desc[idx], desclen[idx]); p += desclen[idx];
}
orig_Key_Event(ev);
}
@@ -423,7 +423,7 @@ INIT {
if (GAMETYPE_MATCHES(L4D)) {
// copy into the keybox so key derivation blake2 gets a nice contiguous
// run of bytes
- memcpy(keybox->lbpub, lbpubkeys[LBPK_L4D], 32);
+ mem_copy(keybox->lbpub, lbpubkeys[LBPK_L4D], 32);
}
hook_commit_Key_Event(h.hookpos, hookdest_Key_Event);
return FEAT_OK;
diff --git a/src/build/gluegen.c b/src/build/gluegen.c
index 33ba600..b0649ca 100644
--- a/src/build/gluegen.c
+++ b/src/build/gluegen.c
@@ -877,8 +877,9 @@ _( "")
_( "static inline void shuntvars() {")
_( "#ifdef _WIN32")
for (int i = 1; i < ncvars; ++i) {
-F( " memmove(&%.*s->v1, &%.*s->v2, sizeof(struct con_var_common));",
+F( " __builtin_memmove(&%.*s->v1, &%.*s->v2, ",
cvar_names[i].len, cvar_names[i].s, cvar_names[i].len, cvar_names[i].s)
+_( " sizeof(struct con_var_common));")
}
_( "#endif")
_( "}")
diff --git a/src/con_.c b/src/con_.c
index 6d5ce03..56ab57f 100644
--- a/src/con_.c
+++ b/src/con_.c
@@ -201,13 +201,13 @@ void VCALLCONV Dispatch_OE(struct con_cmd *this) {
static void ChangeStringValue_common(struct con_var *this,
struct con_var_common *common, char *old, const char *s) {
- memcpy(old, common->strval, common->strlen);
+ mem_copy(old, common->strval, common->strlen);
int len = strlen(s) + 1;
if (len > common->strlen) {
common->strval = extrealloc(common->strval, len);
common->strlen = len;
}
- memcpy(common->strval, s, len);
+ mem_copy(common->strval, s, len);
// callbacks don't matter as far as ABI compat goes (and thank goodness
// because e.g. portal2 randomly adds a *list* of callbacks!?). however we
// do need callbacks for at least one feature, so do our own minimal thing
@@ -215,15 +215,19 @@ static void ChangeStringValue_common(struct con_var *this,
}
static void VCALLCONV ChangeStringValue(struct con_var *this, const char *s,
float oldf) {
- char *old = alloca(this->v2.strlen);
+ char oldbuf[128], *old = oldbuf;
+ if_cold (this->v2.strlen > sizeof(oldbuf)) old = extmalloc(this->v2.strlen);
ChangeStringValue_common(this, &this->v2, old, s);
CallGlobalChangeCallbacks(_con_iface, this, old, oldf);
+ if_cold (old != oldbuf) extfree(old);
}
#ifdef _WIN32
static void VCALLCONV ChangeStringValue_OE(struct con_var *this, const char *s) {
- char *old = alloca(this->v1.strlen);
+ char oldbuf[128], *old = oldbuf;
+ if_cold (this->v1.strlen > sizeof(oldbuf)) old = extmalloc(this->v1.strlen);
ChangeStringValue_common(this, &this->v1, old, s);
CallGlobalChangeCallbacks_OE(_con_iface, this, old);
+ extfree(old);
}
#endif
@@ -418,7 +422,7 @@ void con_regvar(struct con_var *v) {
fudgeflags(&v->base);
struct con_var_common *c = con_getvarcommon(v);
c->strval = extmalloc(c->strlen); // note: _DEF_CVAR() sets strlen member
- memcpy(c->strval, c->defaultval, c->strlen);
+ mem_copy(c->strval, c->defaultval, c->strlen);
RegisterConCommand(_con_iface, v);
}
@@ -578,7 +582,7 @@ bool con_detect(int pluginver) {
if (!find_argcargv()) return false;
if (!find_Con_ColorPrintf()) return false;
// note: _con_colourmsg is already RWX, just overwrite it.
- memcpy((void *)&_con_colourmsg, (void *)&_con_colourmsg_OE,
+ mem_copy((void *)&_con_colourmsg, (void *)&_con_colourmsg_OE,
SELFMOD_LEN);
// NOTE: the default static struct layout is for NE; immediately after
// engineapi init finishes, the generated glue code will shunt
diff --git a/src/con_.h b/src/con_.h
index eb03642..5de5608 100644
--- a/src/con_.h
+++ b/src/con_.h
@@ -19,6 +19,7 @@
#define INC_CON_H
#include "intdefs.h"
+#include "mem.h"
#if defined(__GNUC__) || defined(__clang__)
#define _CON_PRINTF(x, y) __attribute((format(printf, (x), (y))))
@@ -417,10 +418,10 @@ extern struct _con_vtab_iconvar_wrap {
and back to avoid confusing side effects for the caller. */ \
/* XXX: not bothering with the null term here; should we be? */ \
const char *_orig_argv[80]; \
- memcpy(_orig_argv, _con_argv, _orig_argc * sizeof(*argv)); \
- memcpy(_con_argv, argv, argc * sizeof(*argv)); \
+ mem_copy(_orig_argv, _con_argv, _orig_argc * sizeof(*argv)); \
+ mem_copy(_con_argv, argv, argc * sizeof(*argv)); \
_orig_##name##_cb.v1(); \
- memcpy(_con_argv, _orig_argv, _orig_argc * sizeof(*argv)); \
+ mem_copy(_con_argv, _orig_argv, _orig_argc * sizeof(*argv)); \
} \
else { \
_orig_##name##_cb.v1(); \
diff --git a/src/democustom.c b/src/democustom.c
index 05c1d41..5c2ab55 100644
--- a/src/democustom.c
+++ b/src/democustom.c
@@ -72,13 +72,13 @@ static WriteMessages_func WriteMessages = 0;
void democustom_write(const void *buf, int len) {
for (; len > CHUNKSZ; len -= CHUNKSZ) {
createhdr(&bb, CHUNKSZ, false);
- memcpy(bb.buf + (bb.nbits >> 3), buf, CHUNKSZ);
+ mem_copy(bb.buf + (bb.nbits >> 3), buf, CHUNKSZ);
bb.nbits += CHUNKSZ << 3;
WriteMessages(demorecorder, &bb);
bitbuf_reset(&bb);
}
createhdr(&bb, len, true);
- memcpy(bb.buf + (bb.nbits >> 3), buf, len);
+ mem_copy(bb.buf + (bb.nbits >> 3), buf, len);
bb.nbits += len << 3;
WriteMessages(demorecorder, &bb);
bitbuf_reset(&bb);
diff --git a/src/demorec.c b/src/demorec.c
index 6070bee..7a2c798 100644
--- a/src/demorec.c
+++ b/src/demorec.c
@@ -124,17 +124,26 @@ DEF_CCMD_COMPAT_HOOK(record) {
int gdlen = os_strlen(gameinfo_gamedir);
if (gdlen + 1 + argdirlen < PATH_MAX) { // if not, too bad
os_char dir[PATH_MAX], *q = dir;
- memcpy(q, gameinfo_gamedir, gdlen * sizeof(*gameinfo_gamedir));
+ mem_copy(q, gameinfo_gamedir, gdlen * sizeof(*gameinfo_gamedir));
q += gdlen;
*q++ = OS_LIT('/');
- // ascii->wtf16 (probably turns into memcpy() on linux)
+#ifdef _WIN32
+ // ascii->wtf16
for (const char *p = arg; p - arg < argdirlen; ++p, ++q) {
*q = (uchar)*p;
}
+#else
+ // we're -ffreestanding, so don't rely on the compiler turning
+ // the above into a memcpy() call automatically.
+ mem_copy(q, arg, argdirlen);
+ q += argdirlen;
+#endif
*q = OS_LIT('\0');
// this is pretty ugly. the error cases would be way tidier if
// we could use open(O_DIRECTORY), but that's not a thing on
// windows, of course.
+ // TODO(opt): actually, yes it is, as a more low-level NT thing.
+ // revisit, if/when we can be bothered. (not exactly crucial)
struct os_stat s;
static const char *const errpfx = "ERROR: can't record demo: ";
if (os_stat(dir, &s) == -1) {
diff --git a/src/fixes.c b/src/fixes.c
index ac21722..b7e3bcd 100644
--- a/src/fixes.c
+++ b/src/fixes.c
@@ -15,8 +15,6 @@
* PERFORMANCE OF THIS SOFTWARE.
*/
-#include <string.h>
-
#ifdef _WIN32
#include <d3d9.h>
#endif
@@ -216,7 +214,7 @@ static inline void portal1specific() {
void *EyeAngles = mem_offset(clientlib, 0x19D1B0); // in C_PortalPlayer
static const char match[] =
HEXBYTES(56, 8B, F1, E8, 48, 50, EA, FF, 84, C0, 74, 25);
- if (!memcmp(EyeAngles, match, sizeof(match))) {
+ if (!mem_cmp(EyeAngles, match, sizeof(match))) {
char *patch = mem_offset(EyeAngles, 39);
if (patch[0] == 0x75 && patch[1] == 0x08) {
if_hot (os_mprot(patch, 2, PAGE_EXECUTE_READWRITE)) {
diff --git a/src/gameinfo.c b/src/gameinfo.c
index 877ca2a..efa1c2c 100644
--- a/src/gameinfo.c
+++ b/src/gameinfo.c
@@ -14,7 +14,6 @@
* PERFORMANCE OF THIS SOFTWARE.
*/
-#include <string.h>
#ifdef _WIN32
#include <Windows.h> // MultiByteToWideChar()
#endif
@@ -24,6 +23,7 @@
#include "gamedata.h"
#include "gametype.h"
#include "langext.h"
+#include "mem.h"
#include "os.h"
#include "vcall.h"
@@ -77,10 +77,10 @@ bool gameinfo_init() {
int len = GetWindowTextA(gamewin, title, ssizeof(title));
// argh, why did they start doing this, it's so pointless!
// hopefully nobody included these suffixes in their mod names, lol
- if (len > 13 && !memcmp(title + len - 13, " - Direct3D 9", 13)) {
+ if (len > 13 && !mem_cmp(title + len - 13, " - Direct3D 9", 13)) {
title[len - 13] = '\0';
}
- else if (len > 9 && !memcmp(title + len - 9, " - Vulkan", 9)) {
+ else if (len > 9 && !mem_cmp(title + len - 9, " - Vulkan", 9)) {
title[len - 9] = '\0';
}
#else
diff --git a/src/hexcolour.c b/src/hexcolour.c
index f5d2af1..80186aa 100644
--- a/src/hexcolour.c
+++ b/src/hexcolour.c
@@ -14,9 +14,8 @@
* PERFORMANCE OF THIS SOFTWARE.
*/
-#include <string.h>
-
#include "intdefs.h"
+#include "mem.h"
void hexcolour_rgb(uchar out[static 4], const char *s) {
const char *p = s;
@@ -30,7 +29,7 @@ void hexcolour_rgb(uchar out[static 4], const char *s) {
else {
// screw it, just fall back on white, I guess.
// note: this also handles *p == '\0' so we don't overrun the string
- memset(out, 255, 4); // should be a single mov
+ mem_storeu32(out, -1u); // 255, 255, 255, 255
return;
}
// repetitive unrolled nonsense
@@ -41,7 +40,7 @@ void hexcolour_rgb(uchar out[static 4], const char *s) {
*q |= 10 + (*p++ | 32) - 'a';
}
else {
- memset(out, 255, 4); // should be a single mov
+ mem_storeu32(out, -1u); // 255, 255, 255, 255
return;
}
}
@@ -65,7 +64,7 @@ void hexcolour_rgba(uchar out[static 4], const char *s) {
out[3] = 255;
return;
}
- memset(out, 255, 4);
+ mem_storeu32(out, -1u); // 255, 255, 255, 255
return;
}
// even more repetitive unrolled nonsense
@@ -76,7 +75,7 @@ void hexcolour_rgba(uchar out[static 4], const char *s) {
*q |= 10 + (*p++ | 32) - 'a';
}
else {
- memset(out, 255, 4);
+ mem_storeu32(out, -1u); // 255, 255, 255, 255
return;
}
}
diff --git a/src/hook.c b/src/hook.c
index 845d8b1..6738723 100644
--- a/src/hook.c
+++ b/src/hook.c
@@ -15,8 +15,6 @@
* PERFORMANCE OF THIS SOFTWARE.
*/
-#include <string.h>
-
#include "chunklets/x86.h"
#include "hook.h"
#include "intdefs.h"
@@ -55,10 +53,10 @@ struct _hook_prep_ret _hook_prep(uchar *func, uchar *trampoline) {
}
len += ilen;
if (len >= 5) {
- memcpy(trampoline, p, len);
+ mem_copy(trampoline, p, len);
trampoline[len] = X86_JMPIW;
s32 diff = p - (trampoline + 5); // goto the continuation
- memcpy(trampoline + len + 1, &diff, 4);
+ mem_stores32(trampoline + len + 1, diff);
return (struct _hook_prep_ret){func, len, 0};
}
if_cold (p[len] == X86_JMPIW) {
@@ -76,13 +74,14 @@ bool hook_inline_mprot(void *hookpos) {
void _hook_inline_commit(uchar *restrict hookpos, const uchar *restrict target) {
s32 diff = (uchar *)target - (hookpos + 5); // goto the hook target
hookpos[0] = X86_JMPIW;
- memcpy(hookpos + 1, &diff, 4);
+ mem_stores32(hookpos + 1, diff);
}
void _unhook_inline(uchar *trampoline, int len) {
s32 off = mem_loads32(trampoline + len + 1);
uchar *orig = trampoline + off + 5;
- memcpy(orig, trampoline, 5);
+ mem_storeu32(orig, mem_loadu32(trampoline));
+ orig[4] = trampoline[4];
}
// vi: sw=4 ts=4 noet tw=80 cc=80
diff --git a/src/l4d1democompat.c b/src/l4d1democompat.c
index a6d418f..735215c 100644
--- a/src/l4d1democompat.c
+++ b/src/l4d1democompat.c
@@ -127,8 +127,9 @@ static inline ReadDemoHeader_func find_ReadDemoHeader(const uchar *insns) {
static inline void *find_midpoint(ReadDemoHeader_func ReadDemoHeader) {
uchar *insns = (uchar *)ReadDemoHeader;
for (uchar *p = insns; p - insns < 128;) {
+ const u64 HL2DEMO = 0x4F4D4544324C48; // "HL2DEMO\0" ascii, little-endian
if (p[0] == X86_PUSHIW && p[5] == X86_PUSHEBX && p[6] == X86_CALL &&
- !memcmp(mem_loadptr(p + 1), "HL2DEMO", 7)) {
+ !mem_loadu64(mem_loadptr(p + 1)) == HL2DEMO) {
return p + 11;
}
NEXT_INSN(p, "ReadDemoHeader hook midpoint");
diff --git a/src/l4daddon.c b/src/l4daddon.c
index 79c5e72..800232b 100644
--- a/src/l4daddon.c
+++ b/src/l4daddon.c
@@ -121,8 +121,8 @@ static void hookdest_FS_MAFAS(bool disallowaddons, char *mission,
!strncmp(gamemode, last_gamemode, gamemodelen + 1);
}
last_disallowaddons = disallowaddons;
- memcpy(last_mission, mission, missionlen + 1);
- memcpy(last_gamemode, gamemode, gamemodelen + 1);
+ mem_copy(last_mission, mission, missionlen + 1);
+ mem_copy(last_gamemode, gamemode, gamemodelen + 1);
if_hot (canskip) return;
}
else {
@@ -198,8 +198,8 @@ static inline void try_fix_broken_addon_check(uchar *insns) {
// mprot too just so there's no page boundary issues
if_hot (os_mprot(p, 13, PAGE_EXECUTE_READWRITE)) {
broken_addon_check = p; // conditional so END doesn't crash!
- memcpy(orig_broken_addon_check_bytes, broken_addon_check, 13);
- memcpy(broken_addon_check, nops, noplen);
+ mem_copy(orig_broken_addon_check_bytes, broken_addon_check, 13);
+ mem_copy(broken_addon_check, nops, noplen);
}
else {
errmsg_warnsys("couldn't fix broken addon check: "
@@ -243,7 +243,7 @@ END {
unhook_FS_MAFAS();
if_cold (sst_userunloaded) {
if (broken_addon_check) {
- memcpy(broken_addon_check, orig_broken_addon_check_bytes, 13);
+ mem_copy(broken_addon_check, orig_broken_addon_check_bytes, 13);
}
}
}
diff --git a/src/l4dmm.c b/src/l4dmm.c
index 67af36d..343ea0d 100644
--- a/src/l4dmm.c
+++ b/src/l4dmm.c
@@ -95,7 +95,7 @@ const char *l4dmm_curcampaign() {
// reasonable...
usize len = strlen(ret);
if_cold (len > sizeof(campaignbuf) - 1) ret = 0;
- else ret = memcpy(campaignbuf, ret, len + 1);
+ else ret = mem_copy(campaignbuf, ret, len + 1);
}
kvsys_free(kv);
return ret;
diff --git a/src/langext.h b/src/langext.h
index 7e3cbdb..6c57b24 100644
--- a/src/langext.h
+++ b/src/langext.h
@@ -18,6 +18,7 @@
#define assume(x) ((void)(!!(x) || (unreachable, 0)))
#define cold __attribute((__cold__, __noinline__))
#define asm_only __attribute((__naked__)) // N.B.: may not actually work in GCC?
+#define forceinline inline __attribute((__always_inline__))
#else
#define if_hot(x) if (x)
#define if_cold(x) if (x)
@@ -27,12 +28,14 @@
#define assume(x) ((void)(__assume(x), 0))
#define cold __declspec(noinline)
#define asm_only __declspec(naked)
+#define forceinline __forceinline
#else
static inline _Noreturn void _invoke_ub() {}
#define unreachable (_invoke_ub())
#define assume(x) ((void)(!!(x) || (_invoke_ub(), 0)))
#define cold
//#define asm_only // Can't use this without Clang/GCC/MSVC. Too bad.
+#define forceinline inline
#endif
#endif
diff --git a/src/mem.h b/src/mem.h
index 86d310e..e374ca3 100644
--- a/src/mem.h
+++ b/src/mem.h
@@ -17,7 +17,12 @@
#ifndef INC_MEMUTIL_H
#define INC_MEMUTIL_H
+#ifdef SST_DBG
+#include <string.h>
+#endif
+
#include "intdefs.h"
+#include "langext.h"
/* Retrieves an unsigned 32-bit integer from an unaligned pointer. */
static inline u32 mem_loadu32(const void *p) {
@@ -31,26 +36,46 @@ static inline u32 mem_loadu32(const void *p) {
//return (u32)cp[0] | (u32)cp[1] << 8 | (u32)cp[2] << 16 | (u32)cp[3] << 24;
}
-/* Retrieves a signed 32-bit integer from an unaligned pointer. */
-static inline s32 mem_loads32(const void *p) {
- return (s32)mem_loadu32(p);
+/* Writes an unsigned 32-bit integer to an unaligned pointer. */
+static inline void mem_storeu32(const void *p, u32 x) {
+ // Same idea as above.
+ *(u32 *)p = x;
}
+/* Retrieves a signed 32-bit integer from an unaligned pointer. */
+static inline s32 mem_loads32(const void *p) { return (s32)mem_loadu32(p); }
+
+/* Writes a signed 32-bit integer to an unaligned pointer. */
+static inline void mem_stores32(const void *p, s32 x) { mem_storeu32(p, x); }
+
/* Retrieves an unsigned 64-bit integer from an unaligned pointer. */
static inline u64 mem_loadu64(const void *p) {
// this seems not to get butchered as badly in most cases?
return (u64)mem_loadu32(p) | (u64)mem_loadu32((uchar *)p + 4) << 32;
}
-/* Retrieves a signed 64-bit integer from an unaligned pointer. */
-static inline s64 mem_loads64(const void *p) {
- return (s64)mem_loadu64(p);
+/* Writes an unsigned 64-bit integer to an unaligned pointer. */
+static inline void mem_storeu64(const void *p, u64 x) {
+ mem_storeu32(p, x);
+ mem_storeu32((u32 *)p + 1, x >> 32);
}
+/* Retrieves a signed 64-bit integer from an unaligned pointer. */
+static inline s64 mem_loads64(const void *p) { return (s64)mem_loadu64(p); }
+
+/* Writes a signed 64-bit integer to an unaligned pointer. */
+static inline void mem_stores64(const void *p, s64 x) { mem_storeu64(p, x); }
+
/* Retrieves a pointer from an unaligned pointer-to-pointer. */
static inline void *mem_loadptr(const void *p) {
if (sizeof(void *) == 8) return (void *)mem_loadu64(p);
- return (void *)mem_loadu32(p);
+ return (void *)(usize)mem_loadu32(p); // extra cast to prevent warning
+}
+
+/* Writes a pointer to an unaligned pointer-to-pointer. */
+static inline void mem_storeptr(const void *p, const void *x) {
+ if (sizeof(void *) == 8) mem_storeu64(p, (u64)x);
+ else mem_storeu32(p, (u32)(usize)x); // extra cast to prevent warning
}
/* Retrieves a signed size/offset value from an unaligned pointer. */
@@ -58,11 +83,21 @@ static inline ssize mem_loadssize(const void *p) {
return (ssize)mem_loadptr(p);
}
+/* Writes a signed size/offset value to an unaligned pointer. */
+static inline void mem_storessize(const void *p, ssize x) {
+ mem_storeptr(p, (void *)x);
+}
+
/* Retrieves an unsigned size or raw address value from an unaligned pointer. */
static inline usize mem_loadusize(const void *p) {
return (usize)mem_loadptr(p);
}
+/* Writes an unsigned size or raw address value to an unaligned pointer. */
+static inline void mem_storeusize(const void *p, usize x) {
+ mem_storeptr(p, (void *)x);
+}
+
/* Adds a byte count to a pointer and returns a freely-assignable pointer. */
static inline void *mem_offset(const void *p, int off) { return (char *)p + off; }
@@ -71,6 +106,112 @@ static inline ssize mem_diff(const void *p, const void *q) {
return (char *)p - (char *)q;
}
+// Note: the following functions have ifdefs with fallbacks for MVSC and other
+// compilers just in case that code is ever useful somewhere else, but generally
+// the SST codebase can only be built using Clang.
+
+/*
+ * Equivalent to memcpy(), but explicitly generates the most efficient inline
+ * `rep movsb` instruction rather than calling out to a library function.
+ *
+ * Should always be used instead of explicit memcpy() calls. Furthermore, the
+ * compiler is being instructed not to generate automatic memcpy() calls, so it
+ * is the programmer's judgement when to use this. For very small arrays or
+ * structs, simply assigning values is likely faster.
+ *
+ * In debug builds, this just wraps memcpy anyway, to get the CRT debug checks.
+ */
+static forceinline void *mem_copy(void *restrict x, const void *restrict y,
+ unsigned int sz) {
+#if defined(SST_DBG)
+ return memcpy(x, y, sz);
+#elif defined(__GNUC__) || defined(__clang__)
+ void *r = x;
+ __asm volatile (
+ "rep movsb\n"
+ : "+D" (x), "+S" (y), "+c" (sz)
+ :
+ : "memory"
+ );
+ return r;
+#elif defined(_MSC_VER)
+ void __movsb(uchar *, uchar *, usize);
+ __movsb((uchar *)x, (uchar *)y, sz);
+ return x;
+#else
+ char *restrict xb = x; const char *restrict yb = y;
+ for (unsigned int i = 0; i < sz; ++i) xb[i] = yb[i];
+ return x;
+#endif
+}
+
+/*
+ * Equivalent to memset(), but explicitly generates the most efficient inline
+ * `rep stosb` instruction rather than calling out to a library function.
+ *
+ * Should always be used instead of explicit memset() calls. Furthermore, the
+ * compiler is being instructed not to generate automatic memset() calls, so it
+ * is the programmer's judgement whether to use this on structs or small arrays.
+ *
+ * In debug builds, this just wraps memset anyway, to get the CRT debug checks.
+ */
+static forceinline void *mem_set(void *x, int c, unsigned int sz) {
+#if defined(SST_DBG)
+ return memset(x, c, sz);
+#elif defined(__GNUC__) || defined(__clang__)
+ void *r = x;
+ __asm volatile (
+ "rep stosb\n"
+ : "+D" (x), "+c" (sz)
+ : "a"(c)
+ : "memory"
+ );
+ return r;
+#elif defined(_MSC_VER)
+ void __stosb(uchar *, uchar, usize);
+ __stosb((uchar *)x, c, sz);
+ return x;
+#else
+ const unsigned char *xb = x;
+ for (unsigned int i = 0; i < len; ++i) xb[i] = (unsigned char)c;
+ return x;
+#endif
+}
+
+/*
+ * Equivalent to memcmp(), but explicitly generates the most efficient inline
+ * `rep cmpsb` instruction rather than calling out to a library function.
+ *
+ * Should always be used instead of explicit memcmp() calls. Furthermore, the
+ * compiler is being instructed not to generate automatic memcmp() calls, so it
+ * is the programmer's judgement whether to use this on structs or small arrays.
+ *
+ * In debug builds, this just wraps memcmp anyway, to get the CRT debug checks.
+ */
+static forceinline int mem_cmp(const void *restrict x, const void *restrict y,
+ unsigned int sz) {
+#if defined(SST_DBG)
+ return memcmp(x, y, sz);
+#elif defined(__GNUC__) || defined(__clang__)
+ int a, b;
+ __asm volatile (
+ "xor eax, eax\n"
+ "repz cmpsb\n"
+ : "+D" (x), "+S" (y), "+c" (sz), "=@cca"(a), "=@ccb"(b)
+ :
+ : "ax", "memory"
+ );
+ return b - a;
+#else // no msvc intrinsic for this apparently
+ const char *x = x_, *y = y_;
+ for (unsigned int i = 0; i < sz; ++i) {
+ if (x[i] > y[i]) return 1;
+ if (x[i] < y[i]) return -1;
+ }
+ return 0;
+#endif
+}
+
#endif
// vi: sw=4 ts=4 noet tw=80 cc=80
diff --git a/src/os.c b/src/os.c
index 5fb84bc..2bc0e45 100644
--- a/src/os.c
+++ b/src/os.c
@@ -32,6 +32,22 @@
#include "intdefs.h"
#include "langext.h"
+#include "os.h"
+
+// HACK: host tools link in os.c as well. if cross-compiling, we can't use our
+// inline asm mem_copy implementation. just use memcpy if not on 32-bit x86.
+// the host tools will have the C runtime available, unlike SST itself
+#if defined(__i386__) || defined(_M_IX86)
+#include "mem.h"
+#endif
+
+void os_spancopy(os_char *restrict dest, const os_char *restrict src, int n) {
+#if defined(__i386__) || defined(_M_IX86)
+ mem_copy(dest, src, n * sizeof(os_char));
+#else
+ memcpy(dest, src, n * sizeof(os_char));
+#endif
+}
#ifdef _WIN32
@@ -196,7 +212,7 @@ int os_dlfile(void *lib, char *buf, int sz) {
struct link_map *lm = lib;
int ssz = strlen(lm->l_name) + 1;
if_cold (ssz > sz) { errno = ENAMETOOLONG; return -1; }
- memcpy(buf, lm->l_name, ssz);
+ mem_copy(buf, lm->l_name, ssz);
return ssz;
}
#endif
diff --git a/src/os.h b/src/os.h
index 83b5d32..6466e45 100644
--- a/src/os.h
+++ b/src/os.h
@@ -120,10 +120,7 @@ typedef char os_char;
#endif
/* Copies n characters from src to dest, using the OS-specific char type. */
-static inline void os_spancopy(os_char *restrict dest,
- const os_char *restrict src, int n) {
- memcpy(dest, src, n * sizeof(os_char));
-}
+void os_spancopy(os_char *restrict dest, const os_char *restrict src, int n);
/*
* Returns the last error code from an OS function - equivalent to calling
diff --git a/src/portalcolours.c b/src/portalcolours.c
index ff9ab46..41b3a07 100644
--- a/src/portalcolours.c
+++ b/src/portalcolours.c
@@ -14,8 +14,6 @@
* PERFORMANCE OF THIS SOFTWARE.
*/
-#include <string.h>
-
#include "con_.h"
#include "engineapi.h"
#include "errmsg.h"
@@ -81,19 +79,19 @@ static UTIL_Portal_Color_func find_UTIL_Portal_Color(void *base) {
E8, 01, B1, FF, 74, 1E, 83, E8, 01, 8B, 44, 24, 04, 88);
// 5135
void *f = mem_offset(base, 0x1BF090);
- if (!memcmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f;
+ if (!mem_cmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f;
// 4104
f = mem_offset(base, 0x1ADC30);
- if (!memcmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f;
+ if (!mem_cmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f;
// 3420
f = mem_offset(base, 0x1AA810);
- if (!memcmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f;
+ if (!mem_cmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f;
// SteamPipe (7197370) - almost sure to break in a later update!
// TODO(compat): this has indeed been broken for ages.
static const uchar y[] = HEXBYTES(55, 8B, EC, 8B, 45, 0C, 83, E8, 00, 74,
24, 48, 74, 16, 48, 8B, 45, 08, 74, 08, C7, 00, FF, FF);
f = mem_offset(base, 0x234C00);
- if (!memcmp(f, y, sizeof(y))) return (UTIL_Portal_Color_func)f;
+ if (!mem_cmp(f, y, sizeof(y))) return (UTIL_Portal_Color_func)f;
return 0;
}
diff --git a/src/portalisg.c b/src/portalisg.c
index d7851eb..d0f5923 100644
--- a/src/portalisg.c
+++ b/src/portalisg.c
@@ -37,7 +37,9 @@ static con_cmdcbv2 disconnect_cb;
DEF_FEAT_CCMD_HERE(sst_portal_resetisg,
"Remove \"ISG\" state and disconnect from the server", 0) {
// TODO(compat): OE? guess it might work by accident due to cdecl, find out
- disconnect_cb(&(struct con_cmdargs){0});
+ struct con_cmdargs args; // {0} does a memset(), even with -ffreestanding.
+ args.argc = 0; // we only actually have to init this one member.
+ disconnect_cb(&args);
*isg_flag = false;
}
diff --git a/src/sst.c b/src/sst.c
index 070ce12..f5f61bd 100644
--- a/src/sst.c
+++ b/src/sst.c
@@ -180,7 +180,7 @@ DEF_CCMD_HERE(sst_autoload_enable, "Register SST to load on game startup", 0) {
}
}
}
-c: memcpy(r, p + slash + 1, rellen);
+c: mem_copy(r, p + slash + 1, rellen);
#endif
int len = os_strlen(gameinfo_gamedir);
if (len + ssizeof("/addons/" VDFBASENAME ".vdf") > countof(path)) {
@@ -200,9 +200,11 @@ c: memcpy(r, p + slash + 1, rellen);
if_cold (f == -1) { errmsg_errorsys("couldn't open %" fS, path); return; }
#ifdef _WIN32
char buf[19 + PATH_MAX];
- memcpy(buf, "Plugin { file \"", 15);
+ // XXX: with actual calls to memcpy gone, we need this builtin to produce 4
+ // 4-byte movs from this string; can we expose this in a less hideous way?
+ __builtin_memcpy_inline(buf, "Plugin { file \"", 15);
for (int i = 0; i < rellen; ++i) buf[i + 15] = relpath[i];
- memcpy(buf + 15 + rellen, "\" }\n", 4);
+ __builtin_memcpy_inline(buf + 15 + rellen, "\" }\n", 4);
if_cold (os_write(f, buf, rellen + 19) == -1) { // blegh
#else
struct iovec iov[3] = {
diff --git a/src/wincrt.c b/src/wincrt.c
index 9a0326b..5d5883e 100644
--- a/src/wincrt.c
+++ b/src/wincrt.c
@@ -2,72 +2,13 @@
// We get most of the libc functions from ucrtbase.dll, which comes with
// Windows, but for some reason a few of the intrinsic-y things are part of
-// vcruntime, which does *not* come with Windows!!! We can statically link just
-// that part but it adds ~12KiB of random useless bloat to our binary. So, let's
-// just implement the handful of required things here instead. This is only for
-// release/non-debug builds; we want the extra checks in Microsoft's CRT when
-// debugging.
+// vcruntime, which does *not* come with Windows!!!
//
-// Is it actually reasonable to have to do any of this? Of course not.
-
-// Note: these functions have ifdefs with non-asm fallbacks just in case this
-// file is ever useful somewhere else, but generally we assume this codebase
-// will be built with Clang.
-
-int memcmp(const void *restrict x, const void *restrict y, unsigned int sz) {
-#if defined(__GNUC__) || defined(__clang__)
- int a, b;
- __asm volatile (
- "xor eax, eax\n"
- "repz cmpsb\n"
- : "+D" (x), "+S" (y), "+c" (sz), "=@cca"(a), "=@ccb"(b)
- :
- : "ax", "memory"
- );
- return b - a;
-#else
- const char *x = x_, *y = y_;
- for (unsigned int i = 0; i < sz; ++i) {
- if (x[i] > y[i]) return 1;
- if (x[i] < y[i]) return -1;
- }
- return 0;
-#endif
-}
-
-void *memcpy(void *restrict x, const void *restrict y, unsigned int sz) {
-#if defined(__GNUC__) || defined(__clang__)
- void *r = x;
- __asm volatile (
- "rep movsb\n"
- : "+D" (x), "+S" (y), "+c" (sz)
- :
- : "memory"
- );
- return r;
-#else
- char *restrict xb = x; const char *restrict yb = y;
- for (unsigned int i = 0; i < sz; ++i) xb[i] = yb[i];
- return x;
-#endif
-}
-
-void *memset(void *x, int c, unsigned int sz) {
-#if defined(__GNUC__) || defined(__clang__)
- void *r = x;
- __asm volatile (
- "rep stosb\n"
- : "+D" (x), "+c" (sz)
- : "a"(c)
- : "memory"
- );
- return r;
-#else
- const unsigned char *xb = x;
- for (unsigned int i = 0; i < len; ++i) xb[i] = (unsigned char)c;
- return x;
-#endif
-}
+// At this point, we no longer use memcpy/memcmp/memset (preferring to use the
+// x86 single-instruction equivalents and/or explicit builtins/intrinsics).
+//
+// So all we actually have to define here are a couple of dummy symbols to
+// appease the linker.
int __stdcall _DllMainCRTStartup(void *inst, unsigned int reason,
void *reserved) {