diff options
| author | 2025-12-27 03:05:05 +0000 | |
|---|---|---|
| committer | 2026-02-16 19:11:24 +0000 | |
| commit | 8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45 (patch) | |
| tree | 63a8ea6f645b997fd7943a58a002e66244fa49ac | |
| parent | e7943fb834cbcaec585ed68e67b5f17ceda504f7 (diff) | |
| download | sst-8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45.tar.gz sst-8d4b47298b6c9eeb9a3f79a5b80d894f80ed4a45.zip | |
Do away with memcpy/memset/memcmp
Now everything is either explicitly a rep thing, or explicitly some
other small run of reasonably efficient instructions. No library calls
or function call overhead of any kind. Code size remains nice and small.
Ban a few other bad libc functions while we're at it. More can be added
later if we decide that's a useful idea.
This will require some discipline and workarounds to cope with Clang's
tendency to shove mem* calls in places you don't want them even when you
tell it not to. Well, c'est la vie.
I'm a little nervous that the __builtin_memmove() usage in shuntvars()
could still one day emit actual library calls. It appears we're not
importing memmove() from anywhere at the moment, though. So I guess it's
good enough for now.
Also I had to get rid of alloca() in con_.c but that was kind of bad
anyway, so that's fine.
| -rwxr-xr-x | compile | 8 | ||||
| -rw-r--r-- | compile.bat | 8 | ||||
| -rw-r--r-- | src/1.h | 37 | ||||
| -rw-r--r-- | src/ac.c | 16 | ||||
| -rw-r--r-- | src/build/gluegen.c | 3 | ||||
| -rw-r--r-- | src/con_.c | 16 | ||||
| -rw-r--r-- | src/con_.h | 7 | ||||
| -rw-r--r-- | src/democustom.c | 4 | ||||
| -rw-r--r-- | src/demorec.c | 13 | ||||
| -rw-r--r-- | src/fixes.c | 4 | ||||
| -rw-r--r-- | src/gameinfo.c | 6 | ||||
| -rw-r--r-- | src/hexcolour.c | 11 | ||||
| -rw-r--r-- | src/hook.c | 11 | ||||
| -rw-r--r-- | src/l4d1democompat.c | 3 | ||||
| -rw-r--r-- | src/l4daddon.c | 10 | ||||
| -rw-r--r-- | src/l4dmm.c | 2 | ||||
| -rw-r--r-- | src/langext.h | 3 | ||||
| -rw-r--r-- | src/mem.h | 155 | ||||
| -rw-r--r-- | src/os.c | 18 | ||||
| -rw-r--r-- | src/os.h | 5 | ||||
| -rw-r--r-- | src/portalcolours.c | 10 | ||||
| -rw-r--r-- | src/portalisg.c | 4 | ||||
| -rw-r--r-- | src/sst.c | 8 | ||||
| -rw-r--r-- | src/wincrt.c | 71 |
24 files changed, 291 insertions, 142 deletions
@@ -28,8 +28,8 @@ if [ "$dbg" = 1 ]; then cflags="-O0 -g3 -masm=intel -fsanitize-trap=undefined -DSST_DBG" ldflags="-O0 -g3" else - cflags="-O2 -fvisibility=hidden -masm=intel" - ldflags="-O2 -s" + cflags="-O2 -fvisibility=hidden -masm=intel -ffreestanding" + ldflags="-O2 -s -ffreestanding" fi objs= @@ -40,8 +40,8 @@ cc() { # ugly annoying special case if [ "$_mn" = " -DMODULE_NAME=con_" ]; then _mn=" -DMODULE_NAME=con" elif [ "$_mn" = "-DMODULE_NAME=sst" ]; then _mn=; fi - $CC -c -flto -fpic -fno-ident $cflags $warnings -I.build/include \ - $stdflags$_mn -o ".build/${_bn%%.c}.o" "src/$1" + $CC -c -flto -fpic -fno-ident $cflags $warnings -include src/1.h \ + -I.build/include $stdflags$_mn -o ".build/${_bn%%.c}.o" "src/$1" } ld() { diff --git a/compile.bat b/compile.bat index 2958186..10be904 100644 --- a/compile.bat +++ b/compile.bat @@ -29,8 +29,8 @@ if "%dbg%"=="1" ( set cflags=-O0 -g3 -masm=intel -fsanitize-trap=undefined -DSST_DBG
set ldflags=-O0 -g3
) else (
- set cflags=-O2 -masm=intel
- set ldflags=-O2
+ set cflags=-O2 -masm=intel -ffreestanding
+ set ldflags=-O2 -ffreestanding
)
set objs=
@@ -46,8 +46,8 @@ set dmodname= -DMODULE_NAME=%basename% if "%dmodname%"==" -DMODULE_NAME=con_" set dmodname= -DMODULE_NAME=con
if "%dmodname%"==" -DMODULE_NAME=sst" set dmodname=
set objs=%objs% .build/%basename%%ext%.o
-%CC% -c -flto -mno-stack-arg-probe %cflags% %warnings% %stdflags% -I.build/include ^
--D_DLL%dmodname% -o .build/%basename%%ext%.o %1 || goto :end
+%CC% -c -flto -mno-stack-arg-probe %cflags% %warnings% %stdflags% -include src/1.h ^
+-I.build/include -D_DLL%dmodname% -o .build/%basename%%ext%.o %1 || goto :end
goto :eof
:main
@@ -0,0 +1,37 @@ +/* This file is dedicated to the public domain. */ + +// This is a special header force-included in every source file. +// Currently it's just used to ban certain functions from the codebase. +// Ideally this could be done using the poison thing but that triggers a bunch +// of anger when including Windows.h. Of course. So we just use #defines. + +#ifndef SST_DBG // we wrap these in debug mode, so leave them unbanned there. +// These should not be used explicitly. Use mem_copy, mem_set and mem_cmp from +// mem.h instead. Those are optimised for both code size and speed on x86. +#define memcpy please_use_mem_copy_from_mem_dot_h_instead_of_memcpy +#define memset please_use_mem_set_from_mem_dot_h_instead_of_memset +#define memcmp please_use_mem_cmp_from_mem_dot_h_instead_of_memcmp +#endif + +// TODO(opt): implement this if and when we ever need it (unlikely really) +#define memmove sorry_memmove_needs_an_equivalent_in_mem_dot_h_not_done_yet + +// These functions are impossible to use correctly and probably have better +// alternatives already within the codebase. Or, if not, then better +// alternatives can be written easily. +#define strcpy strcpy_is_broken_and_should_never_be_used +#define strcat strcat_is_broken_and_should_never_be_used +#define strncpy strncpy_is_broken_and_should_never_be_used +#define strncat strncat_is_broken_and_should_never_be_used +#define strtok strtok_is_broken_and_slow_and_should_never_be_used +#define strtok_r strtok_r_is_broken_and_slow_and_should_never_be_used +// Windows headers throw a fit when redefining these, lol. The following two +// effectively prevent their use anyway since they're inline functions in +// stdio.h. Dunno about on Linux, but we compile first on Windows anyway innit. +//#define sprintf sprintf_is_dangerous_and_slow_and_should_never_be_used +//#define snprintf snprintf_is_dangerous_and_slow_and_should_never_be_used +#define vsprintf vsprintf_is_dangerous_and_slow_and_should_never_be_used +// XXX: have some straggling vsnprintf usage I'd like to get rid of... +//#define vsnprintf vsnprintf_is_dangerous_and_slow_and_should_never_be_used + +// vi: sw=4 ts=4 noet tw=80 cc=80 @@ -111,11 +111,11 @@ static ssize __stdcall kproc(int code, usize wp, ssize lp) { // something like the following, but with a proper abstraction... //uchar buf[28 + 16], *p = buf; //msg_putasz4(p, 2); p += 1; - // msg_putssz5(p, 8); memcpy(p + 1, "FakeKey", 7); p += 8; + // msg_putssz5(p, 8); mem_copy(p + 1, "FakeKey", 7); p += 8; // msg_putmsz4(p, 2); p += 1; - // msg_putssz5(p, 3); memcpy(p + 1, "vk", 2); p += 3; + // msg_putssz5(p, 3); mem_copy(p + 1, "vk", 2); p += 3; // p += msg_putu32(p, data->vkCode); - // msg_putssz5(p, 3); memcpy(p + 1, "scan", 4); p += 5; + // msg_putssz5(p, 3); mem_copy(p + 1, "scan", 4); p += 5; // p += msg_putu32(p, data->scanCode); //++keybox->nonce; //// append mac at end of message @@ -297,14 +297,14 @@ static void hookdest_Key_Event(struct inputevent *ev) { // TODO(rta): do something interesting with button data //uchar buf[28], *p = buf; //msg_putasz4(p, 2); p += 1; - // msg_putssz5(p, 8); memcpy(p + 1, "KeyInput", 8); p += 9; + // msg_putssz5(p, 8); mem_copy(p + 1, "KeyInput", 8); p += 9; // msg_putmsz4(p, 2); p += 1; - // msg_putssz5(p, 3); memcpy(p + 1, "key", 3); p += 4; + // msg_putssz5(p, 3); mem_copy(p + 1, "key", 3); p += 4; // p += msg_puts32(p, ev->data); - // msg_putssz5(p, 3); memcpy(p + 1, "btn", 3); p += 4; + // msg_putssz5(p, 3); mem_copy(p + 1, "btn", 3); p += 4; // int idx = ev->type - BTNDOWN; // msg_putssz5(p++, desclen[idx]); - // memcpy(p, desc[idx], desclen[idx]); p += desclen[idx]; + // mem_copy(p, desc[idx], desclen[idx]); p += desclen[idx]; } orig_Key_Event(ev); } @@ -423,7 +423,7 @@ INIT { if (GAMETYPE_MATCHES(L4D)) { // copy into the keybox so key derivation blake2 gets a nice contiguous // run of bytes - memcpy(keybox->lbpub, lbpubkeys[LBPK_L4D], 32); + mem_copy(keybox->lbpub, lbpubkeys[LBPK_L4D], 32); } hook_commit_Key_Event(h.hookpos, hookdest_Key_Event); return FEAT_OK; diff --git a/src/build/gluegen.c b/src/build/gluegen.c index 33ba600..b0649ca 100644 --- a/src/build/gluegen.c +++ b/src/build/gluegen.c @@ -877,8 +877,9 @@ _( "") _( "static inline void shuntvars() {") _( "#ifdef _WIN32") for (int i = 1; i < ncvars; ++i) { -F( " memmove(&%.*s->v1, &%.*s->v2, sizeof(struct con_var_common));", +F( " __builtin_memmove(&%.*s->v1, &%.*s->v2, ", cvar_names[i].len, cvar_names[i].s, cvar_names[i].len, cvar_names[i].s) +_( " sizeof(struct con_var_common));") } _( "#endif") _( "}") @@ -201,13 +201,13 @@ void VCALLCONV Dispatch_OE(struct con_cmd *this) { static void ChangeStringValue_common(struct con_var *this, struct con_var_common *common, char *old, const char *s) { - memcpy(old, common->strval, common->strlen); + mem_copy(old, common->strval, common->strlen); int len = strlen(s) + 1; if (len > common->strlen) { common->strval = extrealloc(common->strval, len); common->strlen = len; } - memcpy(common->strval, s, len); + mem_copy(common->strval, s, len); // callbacks don't matter as far as ABI compat goes (and thank goodness // because e.g. portal2 randomly adds a *list* of callbacks!?). however we // do need callbacks for at least one feature, so do our own minimal thing @@ -215,15 +215,19 @@ static void ChangeStringValue_common(struct con_var *this, } static void VCALLCONV ChangeStringValue(struct con_var *this, const char *s, float oldf) { - char *old = alloca(this->v2.strlen); + char oldbuf[128], *old = oldbuf; + if_cold (this->v2.strlen > sizeof(oldbuf)) old = extmalloc(this->v2.strlen); ChangeStringValue_common(this, &this->v2, old, s); CallGlobalChangeCallbacks(_con_iface, this, old, oldf); + if_cold (old != oldbuf) extfree(old); } #ifdef _WIN32 static void VCALLCONV ChangeStringValue_OE(struct con_var *this, const char *s) { - char *old = alloca(this->v1.strlen); + char oldbuf[128], *old = oldbuf; + if_cold (this->v1.strlen > sizeof(oldbuf)) old = extmalloc(this->v1.strlen); ChangeStringValue_common(this, &this->v1, old, s); CallGlobalChangeCallbacks_OE(_con_iface, this, old); + extfree(old); } #endif @@ -418,7 +422,7 @@ void con_regvar(struct con_var *v) { fudgeflags(&v->base); struct con_var_common *c = con_getvarcommon(v); c->strval = extmalloc(c->strlen); // note: _DEF_CVAR() sets strlen member - memcpy(c->strval, c->defaultval, c->strlen); + mem_copy(c->strval, c->defaultval, c->strlen); RegisterConCommand(_con_iface, v); } @@ -578,7 +582,7 @@ bool con_detect(int pluginver) { if (!find_argcargv()) return false; if (!find_Con_ColorPrintf()) return false; // note: _con_colourmsg is already RWX, just overwrite it. - memcpy((void *)&_con_colourmsg, (void *)&_con_colourmsg_OE, + mem_copy((void *)&_con_colourmsg, (void *)&_con_colourmsg_OE, SELFMOD_LEN); // NOTE: the default static struct layout is for NE; immediately after // engineapi init finishes, the generated glue code will shunt @@ -19,6 +19,7 @@ #define INC_CON_H #include "intdefs.h" +#include "mem.h" #if defined(__GNUC__) || defined(__clang__) #define _CON_PRINTF(x, y) __attribute((format(printf, (x), (y)))) @@ -417,10 +418,10 @@ extern struct _con_vtab_iconvar_wrap { and back to avoid confusing side effects for the caller. */ \ /* XXX: not bothering with the null term here; should we be? */ \ const char *_orig_argv[80]; \ - memcpy(_orig_argv, _con_argv, _orig_argc * sizeof(*argv)); \ - memcpy(_con_argv, argv, argc * sizeof(*argv)); \ + mem_copy(_orig_argv, _con_argv, _orig_argc * sizeof(*argv)); \ + mem_copy(_con_argv, argv, argc * sizeof(*argv)); \ _orig_##name##_cb.v1(); \ - memcpy(_con_argv, _orig_argv, _orig_argc * sizeof(*argv)); \ + mem_copy(_con_argv, _orig_argv, _orig_argc * sizeof(*argv)); \ } \ else { \ _orig_##name##_cb.v1(); \ diff --git a/src/democustom.c b/src/democustom.c index 05c1d41..5c2ab55 100644 --- a/src/democustom.c +++ b/src/democustom.c @@ -72,13 +72,13 @@ static WriteMessages_func WriteMessages = 0; void democustom_write(const void *buf, int len) { for (; len > CHUNKSZ; len -= CHUNKSZ) { createhdr(&bb, CHUNKSZ, false); - memcpy(bb.buf + (bb.nbits >> 3), buf, CHUNKSZ); + mem_copy(bb.buf + (bb.nbits >> 3), buf, CHUNKSZ); bb.nbits += CHUNKSZ << 3; WriteMessages(demorecorder, &bb); bitbuf_reset(&bb); } createhdr(&bb, len, true); - memcpy(bb.buf + (bb.nbits >> 3), buf, len); + mem_copy(bb.buf + (bb.nbits >> 3), buf, len); bb.nbits += len << 3; WriteMessages(demorecorder, &bb); bitbuf_reset(&bb); diff --git a/src/demorec.c b/src/demorec.c index 6070bee..7a2c798 100644 --- a/src/demorec.c +++ b/src/demorec.c @@ -124,17 +124,26 @@ DEF_CCMD_COMPAT_HOOK(record) { int gdlen = os_strlen(gameinfo_gamedir); if (gdlen + 1 + argdirlen < PATH_MAX) { // if not, too bad os_char dir[PATH_MAX], *q = dir; - memcpy(q, gameinfo_gamedir, gdlen * sizeof(*gameinfo_gamedir)); + mem_copy(q, gameinfo_gamedir, gdlen * sizeof(*gameinfo_gamedir)); q += gdlen; *q++ = OS_LIT('/'); - // ascii->wtf16 (probably turns into memcpy() on linux) +#ifdef _WIN32 + // ascii->wtf16 for (const char *p = arg; p - arg < argdirlen; ++p, ++q) { *q = (uchar)*p; } +#else + // we're -ffreestanding, so don't rely on the compiler turning + // the above into a memcpy() call automatically. + mem_copy(q, arg, argdirlen); + q += argdirlen; +#endif *q = OS_LIT('\0'); // this is pretty ugly. the error cases would be way tidier if // we could use open(O_DIRECTORY), but that's not a thing on // windows, of course. + // TODO(opt): actually, yes it is, as a more low-level NT thing. + // revisit, if/when we can be bothered. (not exactly crucial) struct os_stat s; static const char *const errpfx = "ERROR: can't record demo: "; if (os_stat(dir, &s) == -1) { diff --git a/src/fixes.c b/src/fixes.c index ac21722..b7e3bcd 100644 --- a/src/fixes.c +++ b/src/fixes.c @@ -15,8 +15,6 @@ * PERFORMANCE OF THIS SOFTWARE. */ -#include <string.h> - #ifdef _WIN32 #include <d3d9.h> #endif @@ -216,7 +214,7 @@ static inline void portal1specific() { void *EyeAngles = mem_offset(clientlib, 0x19D1B0); // in C_PortalPlayer static const char match[] = HEXBYTES(56, 8B, F1, E8, 48, 50, EA, FF, 84, C0, 74, 25); - if (!memcmp(EyeAngles, match, sizeof(match))) { + if (!mem_cmp(EyeAngles, match, sizeof(match))) { char *patch = mem_offset(EyeAngles, 39); if (patch[0] == 0x75 && patch[1] == 0x08) { if_hot (os_mprot(patch, 2, PAGE_EXECUTE_READWRITE)) { diff --git a/src/gameinfo.c b/src/gameinfo.c index 877ca2a..efa1c2c 100644 --- a/src/gameinfo.c +++ b/src/gameinfo.c @@ -14,7 +14,6 @@ * PERFORMANCE OF THIS SOFTWARE. */ -#include <string.h> #ifdef _WIN32 #include <Windows.h> // MultiByteToWideChar() #endif @@ -24,6 +23,7 @@ #include "gamedata.h" #include "gametype.h" #include "langext.h" +#include "mem.h" #include "os.h" #include "vcall.h" @@ -77,10 +77,10 @@ bool gameinfo_init() { int len = GetWindowTextA(gamewin, title, ssizeof(title)); // argh, why did they start doing this, it's so pointless! // hopefully nobody included these suffixes in their mod names, lol - if (len > 13 && !memcmp(title + len - 13, " - Direct3D 9", 13)) { + if (len > 13 && !mem_cmp(title + len - 13, " - Direct3D 9", 13)) { title[len - 13] = '\0'; } - else if (len > 9 && !memcmp(title + len - 9, " - Vulkan", 9)) { + else if (len > 9 && !mem_cmp(title + len - 9, " - Vulkan", 9)) { title[len - 9] = '\0'; } #else diff --git a/src/hexcolour.c b/src/hexcolour.c index f5d2af1..80186aa 100644 --- a/src/hexcolour.c +++ b/src/hexcolour.c @@ -14,9 +14,8 @@ * PERFORMANCE OF THIS SOFTWARE. */ -#include <string.h> - #include "intdefs.h" +#include "mem.h" void hexcolour_rgb(uchar out[static 4], const char *s) { const char *p = s; @@ -30,7 +29,7 @@ void hexcolour_rgb(uchar out[static 4], const char *s) { else { // screw it, just fall back on white, I guess. // note: this also handles *p == '\0' so we don't overrun the string - memset(out, 255, 4); // should be a single mov + mem_storeu32(out, -1u); // 255, 255, 255, 255 return; } // repetitive unrolled nonsense @@ -41,7 +40,7 @@ void hexcolour_rgb(uchar out[static 4], const char *s) { *q |= 10 + (*p++ | 32) - 'a'; } else { - memset(out, 255, 4); // should be a single mov + mem_storeu32(out, -1u); // 255, 255, 255, 255 return; } } @@ -65,7 +64,7 @@ void hexcolour_rgba(uchar out[static 4], const char *s) { out[3] = 255; return; } - memset(out, 255, 4); + mem_storeu32(out, -1u); // 255, 255, 255, 255 return; } // even more repetitive unrolled nonsense @@ -76,7 +75,7 @@ void hexcolour_rgba(uchar out[static 4], const char *s) { *q |= 10 + (*p++ | 32) - 'a'; } else { - memset(out, 255, 4); + mem_storeu32(out, -1u); // 255, 255, 255, 255 return; } } @@ -15,8 +15,6 @@ * PERFORMANCE OF THIS SOFTWARE. */ -#include <string.h> - #include "chunklets/x86.h" #include "hook.h" #include "intdefs.h" @@ -55,10 +53,10 @@ struct _hook_prep_ret _hook_prep(uchar *func, uchar *trampoline) { } len += ilen; if (len >= 5) { - memcpy(trampoline, p, len); + mem_copy(trampoline, p, len); trampoline[len] = X86_JMPIW; s32 diff = p - (trampoline + 5); // goto the continuation - memcpy(trampoline + len + 1, &diff, 4); + mem_stores32(trampoline + len + 1, diff); return (struct _hook_prep_ret){func, len, 0}; } if_cold (p[len] == X86_JMPIW) { @@ -76,13 +74,14 @@ bool hook_inline_mprot(void *hookpos) { void _hook_inline_commit(uchar *restrict hookpos, const uchar *restrict target) { s32 diff = (uchar *)target - (hookpos + 5); // goto the hook target hookpos[0] = X86_JMPIW; - memcpy(hookpos + 1, &diff, 4); + mem_stores32(hookpos + 1, diff); } void _unhook_inline(uchar *trampoline, int len) { s32 off = mem_loads32(trampoline + len + 1); uchar *orig = trampoline + off + 5; - memcpy(orig, trampoline, 5); + mem_storeu32(orig, mem_loadu32(trampoline)); + orig[4] = trampoline[4]; } // vi: sw=4 ts=4 noet tw=80 cc=80 diff --git a/src/l4d1democompat.c b/src/l4d1democompat.c index a6d418f..735215c 100644 --- a/src/l4d1democompat.c +++ b/src/l4d1democompat.c @@ -127,8 +127,9 @@ static inline ReadDemoHeader_func find_ReadDemoHeader(const uchar *insns) { static inline void *find_midpoint(ReadDemoHeader_func ReadDemoHeader) { uchar *insns = (uchar *)ReadDemoHeader; for (uchar *p = insns; p - insns < 128;) { + const u64 HL2DEMO = 0x4F4D4544324C48; // "HL2DEMO\0" ascii, little-endian if (p[0] == X86_PUSHIW && p[5] == X86_PUSHEBX && p[6] == X86_CALL && - !memcmp(mem_loadptr(p + 1), "HL2DEMO", 7)) { + !mem_loadu64(mem_loadptr(p + 1)) == HL2DEMO) { return p + 11; } NEXT_INSN(p, "ReadDemoHeader hook midpoint"); diff --git a/src/l4daddon.c b/src/l4daddon.c index 79c5e72..800232b 100644 --- a/src/l4daddon.c +++ b/src/l4daddon.c @@ -121,8 +121,8 @@ static void hookdest_FS_MAFAS(bool disallowaddons, char *mission, !strncmp(gamemode, last_gamemode, gamemodelen + 1); } last_disallowaddons = disallowaddons; - memcpy(last_mission, mission, missionlen + 1); - memcpy(last_gamemode, gamemode, gamemodelen + 1); + mem_copy(last_mission, mission, missionlen + 1); + mem_copy(last_gamemode, gamemode, gamemodelen + 1); if_hot (canskip) return; } else { @@ -198,8 +198,8 @@ static inline void try_fix_broken_addon_check(uchar *insns) { // mprot too just so there's no page boundary issues if_hot (os_mprot(p, 13, PAGE_EXECUTE_READWRITE)) { broken_addon_check = p; // conditional so END doesn't crash! - memcpy(orig_broken_addon_check_bytes, broken_addon_check, 13); - memcpy(broken_addon_check, nops, noplen); + mem_copy(orig_broken_addon_check_bytes, broken_addon_check, 13); + mem_copy(broken_addon_check, nops, noplen); } else { errmsg_warnsys("couldn't fix broken addon check: " @@ -243,7 +243,7 @@ END { unhook_FS_MAFAS(); if_cold (sst_userunloaded) { if (broken_addon_check) { - memcpy(broken_addon_check, orig_broken_addon_check_bytes, 13); + mem_copy(broken_addon_check, orig_broken_addon_check_bytes, 13); } } } diff --git a/src/l4dmm.c b/src/l4dmm.c index 67af36d..343ea0d 100644 --- a/src/l4dmm.c +++ b/src/l4dmm.c @@ -95,7 +95,7 @@ const char *l4dmm_curcampaign() { // reasonable... usize len = strlen(ret); if_cold (len > sizeof(campaignbuf) - 1) ret = 0; - else ret = memcpy(campaignbuf, ret, len + 1); + else ret = mem_copy(campaignbuf, ret, len + 1); } kvsys_free(kv); return ret; diff --git a/src/langext.h b/src/langext.h index 7e3cbdb..6c57b24 100644 --- a/src/langext.h +++ b/src/langext.h @@ -18,6 +18,7 @@ #define assume(x) ((void)(!!(x) || (unreachable, 0))) #define cold __attribute((__cold__, __noinline__)) #define asm_only __attribute((__naked__)) // N.B.: may not actually work in GCC? +#define forceinline inline __attribute((__always_inline__)) #else #define if_hot(x) if (x) #define if_cold(x) if (x) @@ -27,12 +28,14 @@ #define assume(x) ((void)(__assume(x), 0)) #define cold __declspec(noinline) #define asm_only __declspec(naked) +#define forceinline __forceinline #else static inline _Noreturn void _invoke_ub() {} #define unreachable (_invoke_ub()) #define assume(x) ((void)(!!(x) || (_invoke_ub(), 0))) #define cold //#define asm_only // Can't use this without Clang/GCC/MSVC. Too bad. +#define forceinline inline #endif #endif @@ -17,7 +17,12 @@ #ifndef INC_MEMUTIL_H #define INC_MEMUTIL_H +#ifdef SST_DBG +#include <string.h> +#endif + #include "intdefs.h" +#include "langext.h" /* Retrieves an unsigned 32-bit integer from an unaligned pointer. */ static inline u32 mem_loadu32(const void *p) { @@ -31,26 +36,46 @@ static inline u32 mem_loadu32(const void *p) { //return (u32)cp[0] | (u32)cp[1] << 8 | (u32)cp[2] << 16 | (u32)cp[3] << 24; } -/* Retrieves a signed 32-bit integer from an unaligned pointer. */ -static inline s32 mem_loads32(const void *p) { - return (s32)mem_loadu32(p); +/* Writes an unsigned 32-bit integer to an unaligned pointer. */ +static inline void mem_storeu32(const void *p, u32 x) { + // Same idea as above. + *(u32 *)p = x; } +/* Retrieves a signed 32-bit integer from an unaligned pointer. */ +static inline s32 mem_loads32(const void *p) { return (s32)mem_loadu32(p); } + +/* Writes a signed 32-bit integer to an unaligned pointer. */ +static inline void mem_stores32(const void *p, s32 x) { mem_storeu32(p, x); } + /* Retrieves an unsigned 64-bit integer from an unaligned pointer. */ static inline u64 mem_loadu64(const void *p) { // this seems not to get butchered as badly in most cases? return (u64)mem_loadu32(p) | (u64)mem_loadu32((uchar *)p + 4) << 32; } -/* Retrieves a signed 64-bit integer from an unaligned pointer. */ -static inline s64 mem_loads64(const void *p) { - return (s64)mem_loadu64(p); +/* Writes an unsigned 64-bit integer to an unaligned pointer. */ +static inline void mem_storeu64(const void *p, u64 x) { + mem_storeu32(p, x); + mem_storeu32((u32 *)p + 1, x >> 32); } +/* Retrieves a signed 64-bit integer from an unaligned pointer. */ +static inline s64 mem_loads64(const void *p) { return (s64)mem_loadu64(p); } + +/* Writes a signed 64-bit integer to an unaligned pointer. */ +static inline void mem_stores64(const void *p, s64 x) { mem_storeu64(p, x); } + /* Retrieves a pointer from an unaligned pointer-to-pointer. */ static inline void *mem_loadptr(const void *p) { if (sizeof(void *) == 8) return (void *)mem_loadu64(p); - return (void *)mem_loadu32(p); + return (void *)(usize)mem_loadu32(p); // extra cast to prevent warning +} + +/* Writes a pointer to an unaligned pointer-to-pointer. */ +static inline void mem_storeptr(const void *p, const void *x) { + if (sizeof(void *) == 8) mem_storeu64(p, (u64)x); + else mem_storeu32(p, (u32)(usize)x); // extra cast to prevent warning } /* Retrieves a signed size/offset value from an unaligned pointer. */ @@ -58,11 +83,21 @@ static inline ssize mem_loadssize(const void *p) { return (ssize)mem_loadptr(p); } +/* Writes a signed size/offset value to an unaligned pointer. */ +static inline void mem_storessize(const void *p, ssize x) { + mem_storeptr(p, (void *)x); +} + /* Retrieves an unsigned size or raw address value from an unaligned pointer. */ static inline usize mem_loadusize(const void *p) { return (usize)mem_loadptr(p); } +/* Writes an unsigned size or raw address value to an unaligned pointer. */ +static inline void mem_storeusize(const void *p, usize x) { + mem_storeptr(p, (void *)x); +} + /* Adds a byte count to a pointer and returns a freely-assignable pointer. */ static inline void *mem_offset(const void *p, int off) { return (char *)p + off; } @@ -71,6 +106,112 @@ static inline ssize mem_diff(const void *p, const void *q) { return (char *)p - (char *)q; } +// Note: the following functions have ifdefs with fallbacks for MVSC and other +// compilers just in case that code is ever useful somewhere else, but generally +// the SST codebase can only be built using Clang. + +/* + * Equivalent to memcpy(), but explicitly generates the most efficient inline + * `rep movsb` instruction rather than calling out to a library function. + * + * Should always be used instead of explicit memcpy() calls. Furthermore, the + * compiler is being instructed not to generate automatic memcpy() calls, so it + * is the programmer's judgement when to use this. For very small arrays or + * structs, simply assigning values is likely faster. + * + * In debug builds, this just wraps memcpy anyway, to get the CRT debug checks. + */ +static forceinline void *mem_copy(void *restrict x, const void *restrict y, + unsigned int sz) { +#if defined(SST_DBG) + return memcpy(x, y, sz); +#elif defined(__GNUC__) || defined(__clang__) + void *r = x; + __asm volatile ( + "rep movsb\n" + : "+D" (x), "+S" (y), "+c" (sz) + : + : "memory" + ); + return r; +#elif defined(_MSC_VER) + void __movsb(uchar *, uchar *, usize); + __movsb((uchar *)x, (uchar *)y, sz); + return x; +#else + char *restrict xb = x; const char *restrict yb = y; + for (unsigned int i = 0; i < sz; ++i) xb[i] = yb[i]; + return x; +#endif +} + +/* + * Equivalent to memset(), but explicitly generates the most efficient inline + * `rep stosb` instruction rather than calling out to a library function. + * + * Should always be used instead of explicit memset() calls. Furthermore, the + * compiler is being instructed not to generate automatic memset() calls, so it + * is the programmer's judgement whether to use this on structs or small arrays. + * + * In debug builds, this just wraps memset anyway, to get the CRT debug checks. + */ +static forceinline void *mem_set(void *x, int c, unsigned int sz) { +#if defined(SST_DBG) + return memset(x, c, sz); +#elif defined(__GNUC__) || defined(__clang__) + void *r = x; + __asm volatile ( + "rep stosb\n" + : "+D" (x), "+c" (sz) + : "a"(c) + : "memory" + ); + return r; +#elif defined(_MSC_VER) + void __stosb(uchar *, uchar, usize); + __stosb((uchar *)x, c, sz); + return x; +#else + const unsigned char *xb = x; + for (unsigned int i = 0; i < len; ++i) xb[i] = (unsigned char)c; + return x; +#endif +} + +/* + * Equivalent to memcmp(), but explicitly generates the most efficient inline + * `rep cmpsb` instruction rather than calling out to a library function. + * + * Should always be used instead of explicit memcmp() calls. Furthermore, the + * compiler is being instructed not to generate automatic memcmp() calls, so it + * is the programmer's judgement whether to use this on structs or small arrays. + * + * In debug builds, this just wraps memcmp anyway, to get the CRT debug checks. + */ +static forceinline int mem_cmp(const void *restrict x, const void *restrict y, + unsigned int sz) { +#if defined(SST_DBG) + return memcmp(x, y, sz); +#elif defined(__GNUC__) || defined(__clang__) + int a, b; + __asm volatile ( + "xor eax, eax\n" + "repz cmpsb\n" + : "+D" (x), "+S" (y), "+c" (sz), "=@cca"(a), "=@ccb"(b) + : + : "ax", "memory" + ); + return b - a; +#else // no msvc intrinsic for this apparently + const char *x = x_, *y = y_; + for (unsigned int i = 0; i < sz; ++i) { + if (x[i] > y[i]) return 1; + if (x[i] < y[i]) return -1; + } + return 0; +#endif +} + #endif // vi: sw=4 ts=4 noet tw=80 cc=80 @@ -32,6 +32,22 @@ #include "intdefs.h" #include "langext.h" +#include "os.h" + +// HACK: host tools link in os.c as well. if cross-compiling, we can't use our +// inline asm mem_copy implementation. just use memcpy if not on 32-bit x86. +// the host tools will have the C runtime available, unlike SST itself +#if defined(__i386__) || defined(_M_IX86) +#include "mem.h" +#endif + +void os_spancopy(os_char *restrict dest, const os_char *restrict src, int n) { +#if defined(__i386__) || defined(_M_IX86) + mem_copy(dest, src, n * sizeof(os_char)); +#else + memcpy(dest, src, n * sizeof(os_char)); +#endif +} #ifdef _WIN32 @@ -196,7 +212,7 @@ int os_dlfile(void *lib, char *buf, int sz) { struct link_map *lm = lib; int ssz = strlen(lm->l_name) + 1; if_cold (ssz > sz) { errno = ENAMETOOLONG; return -1; } - memcpy(buf, lm->l_name, ssz); + mem_copy(buf, lm->l_name, ssz); return ssz; } #endif @@ -120,10 +120,7 @@ typedef char os_char; #endif /* Copies n characters from src to dest, using the OS-specific char type. */ -static inline void os_spancopy(os_char *restrict dest, - const os_char *restrict src, int n) { - memcpy(dest, src, n * sizeof(os_char)); -} +void os_spancopy(os_char *restrict dest, const os_char *restrict src, int n); /* * Returns the last error code from an OS function - equivalent to calling diff --git a/src/portalcolours.c b/src/portalcolours.c index ff9ab46..41b3a07 100644 --- a/src/portalcolours.c +++ b/src/portalcolours.c @@ -14,8 +14,6 @@ * PERFORMANCE OF THIS SOFTWARE. */ -#include <string.h> - #include "con_.h" #include "engineapi.h" #include "errmsg.h" @@ -81,19 +79,19 @@ static UTIL_Portal_Color_func find_UTIL_Portal_Color(void *base) { E8, 01, B1, FF, 74, 1E, 83, E8, 01, 8B, 44, 24, 04, 88); // 5135 void *f = mem_offset(base, 0x1BF090); - if (!memcmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f; + if (!mem_cmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f; // 4104 f = mem_offset(base, 0x1ADC30); - if (!memcmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f; + if (!mem_cmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f; // 3420 f = mem_offset(base, 0x1AA810); - if (!memcmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f; + if (!mem_cmp(f, x, sizeof(x))) return (UTIL_Portal_Color_func)f; // SteamPipe (7197370) - almost sure to break in a later update! // TODO(compat): this has indeed been broken for ages. static const uchar y[] = HEXBYTES(55, 8B, EC, 8B, 45, 0C, 83, E8, 00, 74, 24, 48, 74, 16, 48, 8B, 45, 08, 74, 08, C7, 00, FF, FF); f = mem_offset(base, 0x234C00); - if (!memcmp(f, y, sizeof(y))) return (UTIL_Portal_Color_func)f; + if (!mem_cmp(f, y, sizeof(y))) return (UTIL_Portal_Color_func)f; return 0; } diff --git a/src/portalisg.c b/src/portalisg.c index d7851eb..d0f5923 100644 --- a/src/portalisg.c +++ b/src/portalisg.c @@ -37,7 +37,9 @@ static con_cmdcbv2 disconnect_cb; DEF_FEAT_CCMD_HERE(sst_portal_resetisg, "Remove \"ISG\" state and disconnect from the server", 0) { // TODO(compat): OE? guess it might work by accident due to cdecl, find out - disconnect_cb(&(struct con_cmdargs){0}); + struct con_cmdargs args; // {0} does a memset(), even with -ffreestanding. + args.argc = 0; // we only actually have to init this one member. + disconnect_cb(&args); *isg_flag = false; } @@ -180,7 +180,7 @@ DEF_CCMD_HERE(sst_autoload_enable, "Register SST to load on game startup", 0) { } } } -c: memcpy(r, p + slash + 1, rellen); +c: mem_copy(r, p + slash + 1, rellen); #endif int len = os_strlen(gameinfo_gamedir); if (len + ssizeof("/addons/" VDFBASENAME ".vdf") > countof(path)) { @@ -200,9 +200,11 @@ c: memcpy(r, p + slash + 1, rellen); if_cold (f == -1) { errmsg_errorsys("couldn't open %" fS, path); return; } #ifdef _WIN32 char buf[19 + PATH_MAX]; - memcpy(buf, "Plugin { file \"", 15); + // XXX: with actual calls to memcpy gone, we need this builtin to produce 4 + // 4-byte movs from this string; can we expose this in a less hideous way? + __builtin_memcpy_inline(buf, "Plugin { file \"", 15); for (int i = 0; i < rellen; ++i) buf[i + 15] = relpath[i]; - memcpy(buf + 15 + rellen, "\" }\n", 4); + __builtin_memcpy_inline(buf + 15 + rellen, "\" }\n", 4); if_cold (os_write(f, buf, rellen + 19) == -1) { // blegh #else struct iovec iov[3] = { diff --git a/src/wincrt.c b/src/wincrt.c index 9a0326b..5d5883e 100644 --- a/src/wincrt.c +++ b/src/wincrt.c @@ -2,72 +2,13 @@ // We get most of the libc functions from ucrtbase.dll, which comes with // Windows, but for some reason a few of the intrinsic-y things are part of -// vcruntime, which does *not* come with Windows!!! We can statically link just -// that part but it adds ~12KiB of random useless bloat to our binary. So, let's -// just implement the handful of required things here instead. This is only for -// release/non-debug builds; we want the extra checks in Microsoft's CRT when -// debugging. +// vcruntime, which does *not* come with Windows!!! // -// Is it actually reasonable to have to do any of this? Of course not. - -// Note: these functions have ifdefs with non-asm fallbacks just in case this -// file is ever useful somewhere else, but generally we assume this codebase -// will be built with Clang. - -int memcmp(const void *restrict x, const void *restrict y, unsigned int sz) { -#if defined(__GNUC__) || defined(__clang__) - int a, b; - __asm volatile ( - "xor eax, eax\n" - "repz cmpsb\n" - : "+D" (x), "+S" (y), "+c" (sz), "=@cca"(a), "=@ccb"(b) - : - : "ax", "memory" - ); - return b - a; -#else - const char *x = x_, *y = y_; - for (unsigned int i = 0; i < sz; ++i) { - if (x[i] > y[i]) return 1; - if (x[i] < y[i]) return -1; - } - return 0; -#endif -} - -void *memcpy(void *restrict x, const void *restrict y, unsigned int sz) { -#if defined(__GNUC__) || defined(__clang__) - void *r = x; - __asm volatile ( - "rep movsb\n" - : "+D" (x), "+S" (y), "+c" (sz) - : - : "memory" - ); - return r; -#else - char *restrict xb = x; const char *restrict yb = y; - for (unsigned int i = 0; i < sz; ++i) xb[i] = yb[i]; - return x; -#endif -} - -void *memset(void *x, int c, unsigned int sz) { -#if defined(__GNUC__) || defined(__clang__) - void *r = x; - __asm volatile ( - "rep stosb\n" - : "+D" (x), "+c" (sz) - : "a"(c) - : "memory" - ); - return r; -#else - const unsigned char *xb = x; - for (unsigned int i = 0; i < len; ++i) xb[i] = (unsigned char)c; - return x; -#endif -} +// At this point, we no longer use memcpy/memcmp/memset (preferring to use the +// x86 single-instruction equivalents and/or explicit builtins/intrinsics). +// +// So all we actually have to define here are a couple of dummy symbols to +// appease the linker. int __stdcall _DllMainCRTStartup(void *inst, unsigned int reason, void *reserved) { |
