particle: VBState all paths, cached setupRender, @mulAdd — ~25% speedup
- VBState caching on all 5 vertex paths (was only 2D and sin/cos). Eliminates pointer re-reads: load once, emit 4 vertices, writeback. - Cache setupRender() result per-frame via static + resetParticleCache() called from worldUpdateDetour. Saves ~7K function calls/frame. - @mulAdd throughout for FMA codegen on vertex position and texcoord. - A/B verified: BASELINE ~521ms → CUSTOM ~387ms (~25% reduction).
This commit is contained in:
@@ -24,6 +24,7 @@ extern fn multiplyMatrix4x4(u32, u32, u32) u32;
|
||||
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
|
||||
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn resetParticleCache() void;
|
||||
|
||||
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
|
||||
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
|
||||
@@ -365,6 +366,7 @@ const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void;
|
||||
var world_update_hook: hook.Detour(WorldUpdateFn) = .{};
|
||||
|
||||
fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
|
||||
resetParticleCache(); // Clear per-frame caches before rendering
|
||||
const now = rdtsc();
|
||||
if (last_frame_tsc != 0) {
|
||||
const delta = now - last_frame_tsc;
|
||||
|
||||
Reference in New Issue
Block a user