perf: GUID lookup cache (97% hit, 2.6x)

FindObjectByGUID (0x464890): 4096-entry direct-mapped cache with
validate-on-hit. 97% hit rate at 10K+ calls/frame, reducing from
0.8% to 0.3% frame time. Validates cached pointers by checking
GUID at obj+0x30/+0x34 on every hit.

AddToSpatialGrid (0x6816F0): SSE rewrite attempted, no measurable
gain (memory-bound linked list ops dominate). Not shipped.
This commit is contained in:
MarcelineVQ
2026-03-25 14:46:56 -07:00
parent ad33eb9b90
commit 1f318a8565
4 changed files with 302 additions and 6 deletions
+118 -5
View File
@@ -63,6 +63,7 @@ extern fn benchComputeOutcodes(u32, u32, u32, u32) void;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void;
// =========================================================================
// Infrastructure
@@ -237,14 +238,14 @@ pub fn main() void {
print("{s:>30} {s:>10} {s:>10} {s}\n", .{ "function", "original", "sse", "status" });
print("{s}\n", .{"-" ** 72});
// Full performCollisionDetection (SSE vs original x87)
// AddToSpatialGrid -- linked list requires game state, A/B test in-game only
// bench_addToSpatialGrid();
if (false) { // disabled: not testing these right now
bench_collisionDetection();
// ray_triangle_intersection_indexed_int (SSE vs original x87)
bench_rayTriIndexedInt();
// UpdateEntityAndChunksPositions (SSE vs original x87)
bench_entityUpdate();
}
if (false) { // disabled: not working on these right now
@@ -2739,6 +2740,118 @@ fn bench_rayTriIndexedInt() void {
report("rayTriIndexedInt", orig_cyc, sse_cyc, ok);
}
fn bench_addToSpatialGrid() void {
const NOBJS = 20;
// Map globals needed by AddToSpatialGrid
_ = mapZeroed(0x810000, 0x1000); // g_grid_scale at 0x810174
_ = mapZeroed(0x868000, 0x1000); // g_grid_offset at 0x86861C
_ = mapZeroed(0xC7B000, 0x5000); // grid array at 0xC7BD40 through coeffs at 0xC7CFC4
_ = mapZeroed(0x687000, 0x1000); // CalculateLinkedListOffset at 0x6876B0 (in .text, already mapped)
// Set up spatial coefficients (view-like dot product)
@as(*align(1) f32, @ptrFromInt(0xC7CFB8)).* = 0.5; // coeff a
@as(*align(1) f32, @ptrFromInt(0xC7CFBC)).* = 0.3; // coeff b
@as(*align(1) f32, @ptrFromInt(0xC7CFC0)).* = 0.7; // coeff c
@as(*align(1) f32, @ptrFromInt(0xC7CFC4)).* = 1.0; // coeff d
@as(*align(1) f32, @ptrFromInt(0x810174)).* = 0.5; // grid scale
@as(*align(1) f32, @ptrFromInt(0x86861C)).* = 0.0; // grid offset
// Set up grid buckets: each bucket at grid_base + idx*0x6C
// Bucket+0x18 = node offset within object, Bucket+0x1C = list head
// We need dummy list heads for each bucket
const grid_base: u32 = 0xC7BD40;
for (0..32) |bi| {
const bucket = grid_base + @as(u32, @intCast(bi)) * 0x6C;
@as(*align(1) u32, @ptrFromInt(bucket + 0x18)).* = 0x200; // node offset within object
// List head: point to a dummy node (use a region in the bucket itself)
const head_node = bucket + 0x20;
@as(*align(1) u32, @ptrFromInt(bucket + 0x1C)).* = head_node;
// Sentinel head: next=self, prev=self (empty doubly-linked list)
@as(*align(1) u32, @ptrFromInt(head_node)).* = head_node;
@as(*align(1) u32, @ptrFromInt(head_node + 4)).* = head_node;
}
// Build fake objects with varied positions (different grid indices)
// Each object needs: floats at +0x5C,+0x60,+0x64,+0x68, and node space at +0x200
var obj_bufs: [NOBJS][0x210]u8 align(16) = [_][0x210]u8{[_]u8{0} ** 0x210} ** NOBJS;
var objs: [NOBJS]u32 = undefined;
var seed: u32 = 0x98765432;
for (0..NOBJS) |oi| {
objs[oi] = @intFromPtr(&obj_bufs[oi]);
const o = objs[oi];
// Position that maps to different grid buckets
seed = seed *% 1103515245 +% 12345;
const fx: f32 = @as(f32, @floatFromInt(@as(i16, @bitCast(@as(u16, @truncate(seed >> 16)))))) * 0.0001;
seed = seed *% 1103515245 +% 12345;
const fy: f32 = @as(f32, @floatFromInt(@as(i16, @bitCast(@as(u16, @truncate(seed >> 16)))))) * 0.0001;
seed = seed *% 1103515245 +% 12345;
const fz: f32 = @as(f32, @floatFromInt(@as(i16, @bitCast(@as(u16, @truncate(seed >> 16)))))) * 0.0001;
@as(*align(1) f32, @ptrFromInt(o + 0x5C)).* = fx;
@as(*align(1) f32, @ptrFromInt(o + 0x60)).* = fy;
@as(*align(1) f32, @ptrFromInt(o + 0x64)).* = fz;
@as(*align(1) f32, @ptrFromInt(o + 0x68)).* = 0.5; // depth offset
}
const orig_fn = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6816F0));
// Reset grid state between runs
const resetGrid = struct {
fn f(os: *[NOBJS]u32) void {
// Clear all node pointers in objects
for (os) |o| {
@as(*align(1) u32, @ptrFromInt(o + 0x200)).* = 0;
@as(*align(1) u32, @ptrFromInt(o + 0x204)).* = 0;
}
// Reset bucket heads
for (0..32) |bi| {
const bucket = @as(u32, 0xC7BD40) + @as(u32, @intCast(bi)) * 0x6C;
@as(*align(1) u32, @ptrFromInt(bucket + 0x1C)).* = bucket + 0x20;
@as(*align(1) u32, @ptrFromInt(bucket + 0x20)).* = 0;
@as(*align(1) u32, @ptrFromInt(bucket + 0x24)).* = 0;
}
}
}.f;
_ = orig_fn;
resetGrid(&objs);
// Debug: check mapped memory and first object
print(" obj0=0x{x} grid_base=0x{x} bucket0_head=0x{x}\n", .{
objs[0], grid_base, @as(*align(1) u32, @ptrFromInt(grid_base + 0x1C)).*,
});
for (0..NOBJS) |oi| {
@call(.never_tail, addToSpatialGridSSE, .{objs[oi]});
print(" obj {d} OK\n", .{oi});
}
var sse_heads: [32]u32 = undefined;
for (0..32) |bi| sse_heads[bi] = @as(*align(1) u32, @ptrFromInt(grid_base + @as(u32, @intCast(bi)) * 0x6C + 0x1C)).*;
// Verify objects landed in valid buckets (non-zero heads for buckets with objects)
var ok = true;
var populated: u32 = 0;
for (0..32) |bi| {
if (sse_heads[bi] != grid_base + @as(u32, @intCast(bi)) * 0x6C + 0x20) populated += 1;
}
if (populated == 0) { print(" no buckets populated!\n", .{}); ok = false; }
// Benchmark SSE only -- re-insert same objects (they get re-linked each call)
// No resetGrid needed: the function unlinks before re-inserting
var best: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t0 = rdtsc();
for (0..ITERS) |_| {
for (0..NOBJS) |oi| @call(.never_tail, addToSpatialGridSSE, .{objs[oi]});
}
t0 = rdtsc() - t0;
if (t0 < best) best = t0;
}
const per_call = best / ITERS / NOBJS;
const status: [*:0]const u8 = if (ok) "OK" else "MISMATCH";
print("{s:>30}: {d} cyc/call ({d} objects) {s}\n", .{ "AddToSpatialGrid", per_call, NOBJS, status });
}
fn bench_entityUpdate() void {
// Map .bss pages for globals the entity update reads/writes
_ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510
+103
View File
@@ -34,6 +34,109 @@ export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, co
}
}
// =============================================================================
// FindObjectByGUID (0x464890) cache
// stdcall(guidLow, guidHigh) -> objectPtr. RET 0x8.
// Direct-mapped cache: hash the 64-bit GUID, check cached result still valid.
// =============================================================================
const GUID_CACHE_BITS = 8;
const GUID_CACHE_SIZE = 1 << GUID_CACHE_BITS;
const GUID_CACHE_MASK = GUID_CACHE_SIZE - 1;
const GuidCacheEntry = struct {
guid_lo: u32 = 0,
guid_hi: u32 = 0,
result: u32 = 0,
};
var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID_CACHE_SIZE;
const origFindObjectByGUID = @as(*const fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32, @ptrFromInt(0x464890));
export fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
const hash = (guid_lo ^ (guid_hi *% 0x9E3779B9)) & GUID_CACHE_MASK;
const entry = &guid_cache[hash];
if (entry.guid_lo == guid_lo and entry.guid_hi == guid_hi and entry.result != 0) {
// Validate: object at cached address still has this GUID
const obj = entry.result;
if (readU32(obj + 0x30) == guid_lo and readU32(obj + 0x34) == guid_hi) {
return obj;
}
}
// Cache miss or stale: call original
const result = @call(.never_tail, origFindObjectByGUID, .{ guid_lo, guid_hi });
// Store in cache (even if result is 0 -- avoids repeated misses for deleted objects)
entry.* = .{ .guid_lo = guid_lo, .guid_hi = guid_hi, .result = result };
return result;
}
// =============================================================================
// AddToSpatialGrid (0x6816F0)
// __fastcall(ECX=objectPtr), RET
// Computes grid bucket index via dot product + scale + round, then
// inserts object into the bucket's linked list.
// =============================================================================
// CalculateLinkedListOffset: thiscall(ECX=node, stack=direction) -> ptr
const CalculateLinkedListOffset = @as(*const fn (u32, i32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x6876B0));
// Grid globals
const g_grid_base: u32 = 0xC7BD40; // grid array base (stride 0x6C per bucket)
const g_grid_scale: *const f32 = @ptrFromInt(0x810174);
const g_grid_offset: *const f32 = @ptrFromInt(0x86861C);
// Dot product coefficients at 0xC7CFB8..C7CFC4 (same as entpos view coeffs but different address)
const g_spatial_coeffs: u32 = 0xC7CFB8;
export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void {
// Dot product: coeff_a * obj[0x5C] + coeff_b * obj[0x60] + coeff_c * obj[0x64] + coeff_d
const depth = readF32(g_spatial_coeffs) * readF32(obj + 0x5C) +
readF32(g_spatial_coeffs + 4) * readF32(obj + 0x60) +
readF32(g_spatial_coeffs + 8) * readF32(obj + 0x64) +
readF32(g_spatial_coeffs + 12) - readF32(obj + 0x68);
// Grid index: round(depth * scale - offset), clamped to [0, 31]
const scaled = depth * g_grid_scale.* - g_grid_offset.*;
// Round to nearest (matches x87 FISTP with default rounding mode)
// @round returns f32, then convert to int
const rounded = @round(scaled);
var idx: i32 = @intFromFloat(rounded);
if (idx < 0) {
idx = 0;
} else if (idx >= 0x20) {
return;
}
// Bucket layout: grid_base + idx * 0x6C
// Bucket+0x18 = offset to node within object
// Bucket+0x1C = list head pointer
const bucket = g_grid_base + @as(u32, @bitCast(idx)) * 0x6C;
const node_offset = readU32(bucket + 0x18);
const node = node_offset + obj;
// If already in a list, unlink first
if (readU32(node) != 0) {
const prev = @call(.never_tail, CalculateLinkedListOffset, .{ node, -1 });
@as(*align(1) u32, @ptrFromInt(prev)).* = readU32(node);
@as(*align(1) u32, @ptrFromInt(readU32(node) + 4)).* = readU32(node + 4);
@as(*align(1) u32, @ptrFromInt(node)).* = 0;
@as(*align(1) u32, @ptrFromInt(node + 4)).* = 0;
}
// Insert at head of bucket list
const head = readU32(bucket + 0x1C);
@as(*align(1) u32, @ptrFromInt(node)).* = head;
@as(*align(1) u32, @ptrFromInt(node + 4)).* = readU32(head + 4);
@as(*align(1) u32, @ptrFromInt(head + 4)).* = obj;
@as(*align(1) u32, @ptrFromInt(bucket + 0x1C)).* = node;
}
// =============================================================================
// ray_triangle_intersection_indexed_int (0x7C2C40)
// Same Moller-Trumbore as _indexed_ushort but indices are int* not u16*.
+34
View File
@@ -133,6 +133,36 @@ fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyo
return ret;
}
// =============================================================================
// FindObjectByGUID cache (0x464890) — direct-mapped, validate-on-hit
// =============================================================================
const FindGuidFn = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque;
var findguid_hook: hook.Detour(FindGuidFn) = .{};
const GUID_CACHE_BITS = 12;
const GUID_CACHE_SIZE = 1 << GUID_CACHE_BITS;
const GUID_CACHE_MASK = GUID_CACHE_SIZE - 1;
const GuidCacheEntry = struct { guid_lo: u32 = 0, guid_hi: u32 = 0, result: u32 = 0 };
var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID_CACHE_SIZE;
fn findguidDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
// stdcall(2): c=guidLow, d=guidHigh (a,b unused fastcall reg args)
const hash = (c ^ (d *% 0x9E3779B9)) & GUID_CACHE_MASK;
const entry = &guid_cache[hash];
if (entry.guid_lo == c and entry.guid_hi == d and entry.result != 0) {
const obj = entry.result;
if (hook.readMem(u32, obj + 0x30) == c and hook.readMem(u32, obj + 0x34) == d) {
return @ptrFromInt(obj);
}
}
const ret = findguid_hook.callOriginal(.{ a, b, c, d });
entry.* = .{ .guid_lo = c, .guid_hi = d, .result = @intFromPtr(ret) };
return ret;
}
// =============================================================================
// OnWorldUpdate hook (0x482EA0) — per-frame cache reset
// =============================================================================
@@ -263,6 +293,9 @@ pub fn installHooks() void {
// Glyph cache
if (glyph_hook.attach(0x5CA2D0, &glyphDetour) == .ok) installed += 1;
// GUID lookup cache
if (findguid_hook.attach(0x464890, &findguidDetour) == .ok) installed += 1;
// Per-frame cache reset
if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1;
@@ -286,6 +319,7 @@ pub fn removeHooks() void {
transform_hook.detach();
particle_hook.detach();
glyph_hook.detach();
findguid_hook.detach();
world_update_hook.detach();
log.close();
mod_mutex.release(&g_mutex);
+47 -1
View File
@@ -30,6 +30,8 @@ extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn findObjectByGUID_Cached(u32, u32) callconv(.{ .x86_stdcall = .{} }) u32;
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern var stride_info: [8]u32; // exported from particle_sse.zig
@@ -1092,8 +1094,38 @@ fn cbiterDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc
prof.cbiter_calls +|= 1;
return ret;
}
const GUID_CACHE_BITS = 12;
const GUID_CACHE_SIZE = 1 << GUID_CACHE_BITS;
const GUID_CACHE_MASK = GUID_CACHE_SIZE - 1;
const GuidCacheEntry = struct { guid_lo: u32 = 0, guid_hi: u32 = 0, result: u32 = 0 };
var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID_CACHE_SIZE;
var guid_cache_hits: u64 = 0;
var guid_cache_misses: u64 = 0;
fn findguidDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
// stdcall(2): c=guidLow, d=guidHigh
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const hash = (c ^ (d *% 0x9E3779B9)) & GUID_CACHE_MASK;
const entry = &guid_cache[hash];
if (entry.guid_lo == c and entry.guid_hi == d and entry.result != 0) {
const obj = entry.result;
if (hook.readMem(u32, obj + 0x30) == c and hook.readMem(u32, obj + 0x34) == d) {
guid_cache_hits +|= 1;
prof.findguid_cycles +|= rdtsc() - s;
prof.findguid_calls +|= 1;
return @ptrFromInt(obj);
}
}
guid_cache_misses +|= 1;
const ret = findguid_hook.callOriginal(.{ a, b, c, d });
entry.* = .{ .guid_lo = c, .guid_hi = d, .result = @intFromPtr(ret) };
prof.findguid_cycles +|= rdtsc() - s;
prof.findguid_calls +|= 1;
return ret;
}
const ret = findguid_hook.callOriginal(.{ a, b, c, d });
prof.findguid_cycles +|= rdtsc() - s;
prof.findguid_calls +|= 1;
@@ -1533,6 +1565,19 @@ fn dumpStats() void {
});
}
// GUID cache stats (custom mode only)
const guid_total = guid_cache_hits +| guid_cache_misses;
if (guid_total > 0) {
const ghit_pct = pct(guid_cache_hits, guid_total);
log.fmt(" guid_cache: {d}.{d}% hit ({d}hit/{d}miss)\n", .{
ghit_pct / 10, ghit_pct % 10,
guid_cache_hits,
guid_cache_misses,
});
guid_cache_hits = 0;
guid_cache_misses = 0;
}
// Dump particle VB stride info (once)
if (debug_vertex_count != 0) {
log.fmt(" partsetup_debug: verts={d} maxSprites={d} fmt={d} dataPtr=0x{x}\n", .{
@@ -1641,7 +1686,8 @@ pub fn installHooks() void {
// _ = colldet_hook.attach(0x6b88e0, &colldetDetour);
_ = activep_hook.attach(0x7b5a10, &activepDetour);
_ = cbiter_hook.attach(0x404130, &cbiterDetour);
_ = findguid_hook.attach(0x464890, &findguidDetour);
// findguid graduated to weirdperformance GUID cache
// _ = findguid_hook.attach(0x464890, &findguidDetour);
// _ = raytri2_hook.attach(0x632700, &raytri2Detour);
_ = drawbatch_hook.attach(0x70cb30, &drawbatchDetour);
_ = findlua_hook.attach(0x702000, &findluaDetour);