From 1f318a856532fa7847ce88d91fda7fe0ab976ad3 Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Wed, 25 Mar 2026 14:46:56 -0700 Subject: [PATCH] perf: GUID lookup cache (97% hit, 2.6x) FindObjectByGUID (0x464890): 4096-entry direct-mapped cache with validate-on-hit. 97% hit rate at 10K+ calls/frame, reducing from 0.8% to 0.3% frame time. Validates cached pointers by checking GUID at obj+0x30/+0x34 on every hit. AddToSpatialGrid (0x6816F0): SSE rewrite attempted, no measurable gain (memory-bound linked list ops dominate). Not shipped. --- src/bench/main.zig | 123 +++++++++++++++++++++++++-- src/performance/cull_sse.zig | 103 ++++++++++++++++++++++ src/performance/weirdperformance.zig | 34 ++++++++ src/transform44/transform44.zig | 48 ++++++++++- 4 files changed, 302 insertions(+), 6 deletions(-) diff --git a/src/bench/main.zig b/src/bench/main.zig index 157661e..fa310dd 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -63,6 +63,7 @@ extern fn benchComputeOutcodes(u32, u32, u32, u32) void; extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; +extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void; // ========================================================================= // Infrastructure @@ -237,14 +238,14 @@ pub fn main() void { print("{s:>30} {s:>10} {s:>10} {s}\n", .{ "function", "original", "sse", "status" }); print("{s}\n", .{"-" ** 72}); - // Full performCollisionDetection (SSE vs original x87) + // AddToSpatialGrid -- linked list requires game state, A/B test in-game only + // bench_addToSpatialGrid(); + + if (false) { // disabled: not testing these right now bench_collisionDetection(); - - // ray_triangle_intersection_indexed_int (SSE vs original x87) bench_rayTriIndexedInt(); - - // UpdateEntityAndChunksPositions (SSE vs original x87) bench_entityUpdate(); + } if (false) { // disabled: not working on these right now @@ -2739,6 +2740,118 @@ fn bench_rayTriIndexedInt() void { report("rayTriIndexedInt", orig_cyc, sse_cyc, ok); } +fn bench_addToSpatialGrid() void { + const NOBJS = 20; + + // Map globals needed by AddToSpatialGrid + _ = mapZeroed(0x810000, 0x1000); // g_grid_scale at 0x810174 + _ = mapZeroed(0x868000, 0x1000); // g_grid_offset at 0x86861C + _ = mapZeroed(0xC7B000, 0x5000); // grid array at 0xC7BD40 through coeffs at 0xC7CFC4 + _ = mapZeroed(0x687000, 0x1000); // CalculateLinkedListOffset at 0x6876B0 (in .text, already mapped) + + // Set up spatial coefficients (view-like dot product) + @as(*align(1) f32, @ptrFromInt(0xC7CFB8)).* = 0.5; // coeff a + @as(*align(1) f32, @ptrFromInt(0xC7CFBC)).* = 0.3; // coeff b + @as(*align(1) f32, @ptrFromInt(0xC7CFC0)).* = 0.7; // coeff c + @as(*align(1) f32, @ptrFromInt(0xC7CFC4)).* = 1.0; // coeff d + @as(*align(1) f32, @ptrFromInt(0x810174)).* = 0.5; // grid scale + @as(*align(1) f32, @ptrFromInt(0x86861C)).* = 0.0; // grid offset + + // Set up grid buckets: each bucket at grid_base + idx*0x6C + // Bucket+0x18 = node offset within object, Bucket+0x1C = list head + // We need dummy list heads for each bucket + const grid_base: u32 = 0xC7BD40; + for (0..32) |bi| { + const bucket = grid_base + @as(u32, @intCast(bi)) * 0x6C; + @as(*align(1) u32, @ptrFromInt(bucket + 0x18)).* = 0x200; // node offset within object + // List head: point to a dummy node (use a region in the bucket itself) + const head_node = bucket + 0x20; + @as(*align(1) u32, @ptrFromInt(bucket + 0x1C)).* = head_node; + // Sentinel head: next=self, prev=self (empty doubly-linked list) + @as(*align(1) u32, @ptrFromInt(head_node)).* = head_node; + @as(*align(1) u32, @ptrFromInt(head_node + 4)).* = head_node; + } + + // Build fake objects with varied positions (different grid indices) + // Each object needs: floats at +0x5C,+0x60,+0x64,+0x68, and node space at +0x200 + var obj_bufs: [NOBJS][0x210]u8 align(16) = [_][0x210]u8{[_]u8{0} ** 0x210} ** NOBJS; + var objs: [NOBJS]u32 = undefined; + + var seed: u32 = 0x98765432; + for (0..NOBJS) |oi| { + objs[oi] = @intFromPtr(&obj_bufs[oi]); + const o = objs[oi]; + // Position that maps to different grid buckets + seed = seed *% 1103515245 +% 12345; + const fx: f32 = @as(f32, @floatFromInt(@as(i16, @bitCast(@as(u16, @truncate(seed >> 16)))))) * 0.0001; + seed = seed *% 1103515245 +% 12345; + const fy: f32 = @as(f32, @floatFromInt(@as(i16, @bitCast(@as(u16, @truncate(seed >> 16)))))) * 0.0001; + seed = seed *% 1103515245 +% 12345; + const fz: f32 = @as(f32, @floatFromInt(@as(i16, @bitCast(@as(u16, @truncate(seed >> 16)))))) * 0.0001; + @as(*align(1) f32, @ptrFromInt(o + 0x5C)).* = fx; + @as(*align(1) f32, @ptrFromInt(o + 0x60)).* = fy; + @as(*align(1) f32, @ptrFromInt(o + 0x64)).* = fz; + @as(*align(1) f32, @ptrFromInt(o + 0x68)).* = 0.5; // depth offset + } + + const orig_fn = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6816F0)); + + // Reset grid state between runs + const resetGrid = struct { + fn f(os: *[NOBJS]u32) void { + // Clear all node pointers in objects + for (os) |o| { + @as(*align(1) u32, @ptrFromInt(o + 0x200)).* = 0; + @as(*align(1) u32, @ptrFromInt(o + 0x204)).* = 0; + } + // Reset bucket heads + for (0..32) |bi| { + const bucket = @as(u32, 0xC7BD40) + @as(u32, @intCast(bi)) * 0x6C; + @as(*align(1) u32, @ptrFromInt(bucket + 0x1C)).* = bucket + 0x20; + @as(*align(1) u32, @ptrFromInt(bucket + 0x20)).* = 0; + @as(*align(1) u32, @ptrFromInt(bucket + 0x24)).* = 0; + } + } + }.f; + + _ = orig_fn; + resetGrid(&objs); + // Debug: check mapped memory and first object + print(" obj0=0x{x} grid_base=0x{x} bucket0_head=0x{x}\n", .{ + objs[0], grid_base, @as(*align(1) u32, @ptrFromInt(grid_base + 0x1C)).*, + }); + for (0..NOBJS) |oi| { + @call(.never_tail, addToSpatialGridSSE, .{objs[oi]}); + print(" obj {d} OK\n", .{oi}); + } + var sse_heads: [32]u32 = undefined; + for (0..32) |bi| sse_heads[bi] = @as(*align(1) u32, @ptrFromInt(grid_base + @as(u32, @intCast(bi)) * 0x6C + 0x1C)).*; + + // Verify objects landed in valid buckets (non-zero heads for buckets with objects) + var ok = true; + var populated: u32 = 0; + for (0..32) |bi| { + if (sse_heads[bi] != grid_base + @as(u32, @intCast(bi)) * 0x6C + 0x20) populated += 1; + } + if (populated == 0) { print(" no buckets populated!\n", .{}); ok = false; } + + // Benchmark SSE only -- re-insert same objects (they get re-linked each call) + // No resetGrid needed: the function unlinks before re-inserting + var best: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t0 = rdtsc(); + for (0..ITERS) |_| { + for (0..NOBJS) |oi| @call(.never_tail, addToSpatialGridSSE, .{objs[oi]}); + } + t0 = rdtsc() - t0; + if (t0 < best) best = t0; + } + + const per_call = best / ITERS / NOBJS; + const status: [*:0]const u8 = if (ok) "OK" else "MISMATCH"; + print("{s:>30}: {d} cyc/call ({d} objects) {s}\n", .{ "AddToSpatialGrid", per_call, NOBJS, status }); +} + fn bench_entityUpdate() void { // Map .bss pages for globals the entity update reads/writes _ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510 diff --git a/src/performance/cull_sse.zig b/src/performance/cull_sse.zig index c82da92..f7147dd 100644 --- a/src/performance/cull_sse.zig +++ b/src/performance/cull_sse.zig @@ -34,6 +34,109 @@ export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, co } } +// ============================================================================= +// FindObjectByGUID (0x464890) cache +// stdcall(guidLow, guidHigh) -> objectPtr. RET 0x8. +// Direct-mapped cache: hash the 64-bit GUID, check cached result still valid. +// ============================================================================= + +const GUID_CACHE_BITS = 8; +const GUID_CACHE_SIZE = 1 << GUID_CACHE_BITS; +const GUID_CACHE_MASK = GUID_CACHE_SIZE - 1; + +const GuidCacheEntry = struct { + guid_lo: u32 = 0, + guid_hi: u32 = 0, + result: u32 = 0, +}; + +var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID_CACHE_SIZE; + +const origFindObjectByGUID = @as(*const fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32, @ptrFromInt(0x464890)); + +export fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 { + const hash = (guid_lo ^ (guid_hi *% 0x9E3779B9)) & GUID_CACHE_MASK; + const entry = &guid_cache[hash]; + + if (entry.guid_lo == guid_lo and entry.guid_hi == guid_hi and entry.result != 0) { + // Validate: object at cached address still has this GUID + const obj = entry.result; + if (readU32(obj + 0x30) == guid_lo and readU32(obj + 0x34) == guid_hi) { + return obj; + } + } + + // Cache miss or stale: call original + const result = @call(.never_tail, origFindObjectByGUID, .{ guid_lo, guid_hi }); + + // Store in cache (even if result is 0 -- avoids repeated misses for deleted objects) + entry.* = .{ .guid_lo = guid_lo, .guid_hi = guid_hi, .result = result }; + + return result; +} + +// ============================================================================= +// AddToSpatialGrid (0x6816F0) +// __fastcall(ECX=objectPtr), RET +// Computes grid bucket index via dot product + scale + round, then +// inserts object into the bucket's linked list. +// ============================================================================= + +// CalculateLinkedListOffset: thiscall(ECX=node, stack=direction) -> ptr +const CalculateLinkedListOffset = @as(*const fn (u32, i32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x6876B0)); + +// Grid globals +const g_grid_base: u32 = 0xC7BD40; // grid array base (stride 0x6C per bucket) +const g_grid_scale: *const f32 = @ptrFromInt(0x810174); +const g_grid_offset: *const f32 = @ptrFromInt(0x86861C); +// Dot product coefficients at 0xC7CFB8..C7CFC4 (same as entpos view coeffs but different address) +const g_spatial_coeffs: u32 = 0xC7CFB8; + +export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void { + // Dot product: coeff_a * obj[0x5C] + coeff_b * obj[0x60] + coeff_c * obj[0x64] + coeff_d + const depth = readF32(g_spatial_coeffs) * readF32(obj + 0x5C) + + readF32(g_spatial_coeffs + 4) * readF32(obj + 0x60) + + readF32(g_spatial_coeffs + 8) * readF32(obj + 0x64) + + readF32(g_spatial_coeffs + 12) - readF32(obj + 0x68); + + // Grid index: round(depth * scale - offset), clamped to [0, 31] + const scaled = depth * g_grid_scale.* - g_grid_offset.*; + + // Round to nearest (matches x87 FISTP with default rounding mode) + // @round returns f32, then convert to int + const rounded = @round(scaled); + var idx: i32 = @intFromFloat(rounded); + + if (idx < 0) { + idx = 0; + } else if (idx >= 0x20) { + return; + } + + // Bucket layout: grid_base + idx * 0x6C + // Bucket+0x18 = offset to node within object + // Bucket+0x1C = list head pointer + const bucket = g_grid_base + @as(u32, @bitCast(idx)) * 0x6C; + const node_offset = readU32(bucket + 0x18); + const node = node_offset + obj; + + // If already in a list, unlink first + if (readU32(node) != 0) { + const prev = @call(.never_tail, CalculateLinkedListOffset, .{ node, -1 }); + @as(*align(1) u32, @ptrFromInt(prev)).* = readU32(node); + @as(*align(1) u32, @ptrFromInt(readU32(node) + 4)).* = readU32(node + 4); + @as(*align(1) u32, @ptrFromInt(node)).* = 0; + @as(*align(1) u32, @ptrFromInt(node + 4)).* = 0; + } + + // Insert at head of bucket list + const head = readU32(bucket + 0x1C); + @as(*align(1) u32, @ptrFromInt(node)).* = head; + @as(*align(1) u32, @ptrFromInt(node + 4)).* = readU32(head + 4); + @as(*align(1) u32, @ptrFromInt(head + 4)).* = obj; + @as(*align(1) u32, @ptrFromInt(bucket + 0x1C)).* = node; +} + // ============================================================================= // ray_triangle_intersection_indexed_int (0x7C2C40) // Same Moller-Trumbore as _indexed_ushort but indices are int* not u16*. diff --git a/src/performance/weirdperformance.zig b/src/performance/weirdperformance.zig index 90732ae..0a4fe11 100644 --- a/src/performance/weirdperformance.zig +++ b/src/performance/weirdperformance.zig @@ -133,6 +133,36 @@ fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyo return ret; } +// ============================================================================= +// FindObjectByGUID cache (0x464890) — direct-mapped, validate-on-hit +// ============================================================================= + +const FindGuidFn = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +var findguid_hook: hook.Detour(FindGuidFn) = .{}; + +const GUID_CACHE_BITS = 12; +const GUID_CACHE_SIZE = 1 << GUID_CACHE_BITS; +const GUID_CACHE_MASK = GUID_CACHE_SIZE - 1; +const GuidCacheEntry = struct { guid_lo: u32 = 0, guid_hi: u32 = 0, result: u32 = 0 }; +var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID_CACHE_SIZE; + +fn findguidDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + // stdcall(2): c=guidLow, d=guidHigh (a,b unused fastcall reg args) + const hash = (c ^ (d *% 0x9E3779B9)) & GUID_CACHE_MASK; + const entry = &guid_cache[hash]; + + if (entry.guid_lo == c and entry.guid_hi == d and entry.result != 0) { + const obj = entry.result; + if (hook.readMem(u32, obj + 0x30) == c and hook.readMem(u32, obj + 0x34) == d) { + return @ptrFromInt(obj); + } + } + + const ret = findguid_hook.callOriginal(.{ a, b, c, d }); + entry.* = .{ .guid_lo = c, .guid_hi = d, .result = @intFromPtr(ret) }; + return ret; +} + // ============================================================================= // OnWorldUpdate hook (0x482EA0) — per-frame cache reset // ============================================================================= @@ -263,6 +293,9 @@ pub fn installHooks() void { // Glyph cache if (glyph_hook.attach(0x5CA2D0, &glyphDetour) == .ok) installed += 1; + // GUID lookup cache + if (findguid_hook.attach(0x464890, &findguidDetour) == .ok) installed += 1; + // Per-frame cache reset if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1; @@ -286,6 +319,7 @@ pub fn removeHooks() void { transform_hook.detach(); particle_hook.detach(); glyph_hook.detach(); + findguid_hook.detach(); world_update_hook.detach(); log.close(); mod_mutex.release(&g_mutex); diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index 1b1933c..e572c01 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -30,6 +30,8 @@ extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; +extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void; +extern fn findObjectByGUID_Cached(u32, u32) callconv(.{ .x86_stdcall = .{} }) u32; extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern var stride_info: [8]u32; // exported from particle_sse.zig @@ -1092,8 +1094,38 @@ fn cbiterDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc prof.cbiter_calls +|= 1; return ret; } +const GUID_CACHE_BITS = 12; +const GUID_CACHE_SIZE = 1 << GUID_CACHE_BITS; +const GUID_CACHE_MASK = GUID_CACHE_SIZE - 1; +const GuidCacheEntry = struct { guid_lo: u32 = 0, guid_hi: u32 = 0, result: u32 = 0 }; +var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID_CACHE_SIZE; +var guid_cache_hits: u64 = 0; +var guid_cache_misses: u64 = 0; + fn findguidDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + // stdcall(2): c=guidLow, d=guidHigh const s = rdtsc(); + if (AB_OTHER_HOOKS and ab_use_custom) { + const hash = (c ^ (d *% 0x9E3779B9)) & GUID_CACHE_MASK; + const entry = &guid_cache[hash]; + + if (entry.guid_lo == c and entry.guid_hi == d and entry.result != 0) { + const obj = entry.result; + if (hook.readMem(u32, obj + 0x30) == c and hook.readMem(u32, obj + 0x34) == d) { + guid_cache_hits +|= 1; + prof.findguid_cycles +|= rdtsc() - s; + prof.findguid_calls +|= 1; + return @ptrFromInt(obj); + } + } + + guid_cache_misses +|= 1; + const ret = findguid_hook.callOriginal(.{ a, b, c, d }); + entry.* = .{ .guid_lo = c, .guid_hi = d, .result = @intFromPtr(ret) }; + prof.findguid_cycles +|= rdtsc() - s; + prof.findguid_calls +|= 1; + return ret; + } const ret = findguid_hook.callOriginal(.{ a, b, c, d }); prof.findguid_cycles +|= rdtsc() - s; prof.findguid_calls +|= 1; @@ -1533,6 +1565,19 @@ fn dumpStats() void { }); } + // GUID cache stats (custom mode only) + const guid_total = guid_cache_hits +| guid_cache_misses; + if (guid_total > 0) { + const ghit_pct = pct(guid_cache_hits, guid_total); + log.fmt(" guid_cache: {d}.{d}% hit ({d}hit/{d}miss)\n", .{ + ghit_pct / 10, ghit_pct % 10, + guid_cache_hits, + guid_cache_misses, + }); + guid_cache_hits = 0; + guid_cache_misses = 0; + } + // Dump particle VB stride info (once) if (debug_vertex_count != 0) { log.fmt(" partsetup_debug: verts={d} maxSprites={d} fmt={d} dataPtr=0x{x}\n", .{ @@ -1641,7 +1686,8 @@ pub fn installHooks() void { // _ = colldet_hook.attach(0x6b88e0, &colldetDetour); _ = activep_hook.attach(0x7b5a10, &activepDetour); _ = cbiter_hook.attach(0x404130, &cbiterDetour); - _ = findguid_hook.attach(0x464890, &findguidDetour); + // findguid graduated to weirdperformance GUID cache + // _ = findguid_hook.attach(0x464890, &findguidDetour); // _ = raytri2_hook.attach(0x632700, &raytri2Detour); _ = drawbatch_hook.attach(0x70cb30, &drawbatchDetour); _ = findlua_hook.attach(0x702000, &findluaDetour);