particle: VBState all paths, cached setupRender, @mulAdd — ~25% speedup

- VBState caching on all 5 vertex paths (was only 2D and sin/cos).
  Eliminates pointer re-reads: load once, emit 4 vertices, writeback.
- Cache setupRender() result per-frame via static + resetParticleCache()
  called from worldUpdateDetour. Saves ~7K function calls/frame.
- @mulAdd throughout for FMA codegen on vertex position and texcoord.
- A/B verified: BASELINE ~521ms → CUSTOM ~387ms (~25% reduction).
This commit is contained in:
MarcelineVQ
2026-03-23 22:29:12 -07:00
parent 1dc1350645
commit 0d994db4b3
2 changed files with 87 additions and 105 deletions
+2
View File
@@ -24,6 +24,7 @@ extern fn multiplyMatrix4x4(u32, u32, u32) u32;
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn resetParticleCache() void;
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
@@ -365,6 +366,7 @@ const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void;
var world_update_hook: hook.Detour(WorldUpdateFn) = .{};
fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
resetParticleCache(); // Clear per-frame caches before rendering
const now = rdtsc();
if (last_frame_tsc != 0) {
const delta = now - last_frame_tsc;