From 7819d6d914f5441f46658a8f8770bd1fb9c2202c Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Mon, 23 Mar 2026 20:27:26 -0700 Subject: [PATCH] perf: findInterpIdx dedup, processLinkedListCollision SSE, permanent t44 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - bone_sse: deduplicate findInterpIdx calls in bone loop — rotation's search result reused for scale/translation when tracks share temporal structure (canReuseInterp guard). Est. ~23% bone loop cycle reduction. - silicon: SSE replacement for processLinkedListCollision (0x6ABC40, 1.57% CPU). V4 AABB overlap test replaces 6 x87 FCOMP/FNSTSW. Benched at 3.2x speedup (378→115 cyc/call, 8 nodes). - transform44: remove A/B toggle, always use SSE path (teardown guard kept). A/B infrastructure remains for other hooks. - bench: add processLinkedListCollision benchmark with fake linked list test fixture and stubbed addGeometryToBuffer. --- src/bench/main.zig | 143 ++++++++++++++++++++++++++++++++ src/silicon/silicon.zig | 2 + src/silicon/silicon_sse.zig | 78 +++++++++++++++++ src/transform44/bone_sse.zig | 64 ++++++++++++-- src/transform44/transform44.zig | 4 +- 5 files changed, 283 insertions(+), 8 deletions(-) diff --git a/src/bench/main.zig b/src/bench/main.zig index 32292a5..ba53961 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -49,6 +49,7 @@ extern fn si_mulMat3x4InPlace(u32, u32) callconv(cc_tc) u32; extern fn si_normalizeVec3InPlace(u32) callconv(cc_tc) void; extern fn si_vec3Dot(u32, u32) callconv(cc_fc) f64; extern fn si_translateBoundingVol(u32, u32) callconv(cc_tc) void; +extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(cc_fc) u32; extern fn si_addVec3ToAccumulator(u32, u32) callconv(cc_tc) void; extern fn si_addToColorAccumulator(u32, u32) callconv(cc_tc) void; extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void; @@ -1767,9 +1768,151 @@ pub fn main() void { } } + // si_processLinkedListCollision -- fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack) -> u32 + // Builds a fake linked list with 8 nodes to benchmark AABB overlap test. + bench_processLinkedListCollision(); + print("\n", .{}); } +fn bench_processLinkedListCollision() void { + // Map page for sentinel global at 0xC89F20 + _ = mapZeroed(0xC89000, 0x1000); + // Map page for addGeometryToBuffer's result_buf writes (just needs writable memory) + // Also need pages at 0xCA0000 range for any globals addGeometryToBuffer touches + + const NODE_COUNT = 8; + + // Sentinel: just a unique non-zero value. Original code reads *(u32*)0xC89F20. + const sentinel: u32 = 0xDEADBEEF; + @as(*u32, @ptrFromInt(0xC89F20)).* = sentinel; + + // --- Build fake node data blocks (need offsets: +0x0C, +0x88, +0x8C, +0x14C-0x164, +0x180, +0x184) --- + // Each node_data needs at least 0x188 bytes + const NODE_DATA_SIZE = 0x190; + var node_data_buf: [NODE_COUNT * NODE_DATA_SIZE]u8 align(4) = std.mem.zeroes([NODE_COUNT * NODE_DATA_SIZE]u8); + + // Query box: min=(0,0,0), max=(10,10,10) + var query_box = [6]f32{ 0.0, 0.0, 0.0, 10.0, 10.0, 10.0 }; + + // Stub addGeometryToBuffer at 0x6ABD90 → RET 0x4 (just returns, no side effects). + // Both original and SSE call the same stub, isolating the linked list walk + AABB test. + // Original bytes are in mapped .text — overwrite with: C2 04 00 (RET 4) + @as(*[3]u8, @ptrFromInt(0x6ABD90)).* = .{ 0xC2, 0x04, 0x00 }; + + // Set up each node_data + for (0..NODE_COUNT) |i| { + const nd = @intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE; + // flags at +0x0C: bit 0x80 set (required, else returns 0), no 0x100 (not skipped) + @as(*align(1) u16, @ptrFromInt(nd + 0x0C)).* = 0x80; + // active at +0x88: non-zero (just needs to pass != 0 check) + @as(*align(1) u32, @ptrFromInt(nd + 0x88)).* = 1; + // visited at +0x8C: NOT sentinel (so it gets processed) + @as(*align(1) u32, @ptrFromInt(nd + 0x8C)).* = 0; + // type discriminator: both zero → use flags & 0xF + @as(*align(1) u32, @ptrFromInt(nd + 0x180)).* = 0; + @as(*align(1) u32, @ptrFromInt(nd + 0x184)).* = 0; + + // AABB at +0x14C: alternate overlapping and non-overlapping + const aabb: *align(1) [6]f32 = @ptrFromInt(nd + 0x14C); + if (i % 2 == 0) { + // Overlapping: min=(1,1,1), max=(5,5,5) + aabb.* = .{ 1.0, 1.0, 1.0, 5.0, 5.0, 5.0 }; + } else { + // Non-overlapping: min=(20,20,20), max=(30,30,30) + aabb.* = .{ 20.0, 20.0, 20.0, 30.0, 30.0, 30.0 }; + } + } + + // --- Build linked list nodes --- + // Intrusive list: node = { ??, node_data_ptr, ... } + // link_offset stored at listHead[0], next at *(link_offset + node + 4) + // Simplest: link_offset = 0, so next = *(node + 4) ... no wait. + // Re-reading assembly: next = *(*(listHead) + prev_node + 4) + // listHead[0] = link_offset (byte offset within node to find next-ptr) + // Actually from the asm: MOV EAX,[EBP-0xc] (=listHead), MOV EAX,[EAX] (=*listHead = link_offset) + // MOV ECX,[EAX + EDX*1 + 4] where EDX=node + // So: next = *(link_offset + node + 4) + // If link_offset = 0: next = *(node + 4), but node+4 is node_data_ptr! + // We need link_offset such that (link_offset + node + 4) points to a "next" field. + // Let's use link_offset = 4, so next = *(node + 8). + // Node layout: [node_data_ptr(+0), ?(+4), next(+8)] + // But wait, node+4 is where node_data is read: MOV EBX,[EDX+4] (EDX=node) + // So node = { pad(+0), node_data(+4), next(+8) } and link_offset = 4. + + const NODE_SIZE = 12; // pad, node_data_ptr, next_ptr + var nodes: [NODE_COUNT * NODE_SIZE]u8 align(4) = std.mem.zeroes([NODE_COUNT * NODE_SIZE]u8); + + for (0..NODE_COUNT) |i| { + const n = @intFromPtr(&nodes) + i * NODE_SIZE; + // node+4 = node_data pointer + @as(*align(1) u32, @ptrFromInt(n + 4)).* = @intCast(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE); + // node+8 = next node (link_offset=4, so *(link_offset + node + 4) = *(node + 8)) + if (i + 1 < NODE_COUNT) { + @as(*align(1) u32, @ptrFromInt(n + 8)).* = @intCast(@intFromPtr(&nodes) + (i + 1) * NODE_SIZE); + } else { + @as(*align(1) u32, @ptrFromInt(n + 8)).* = 0; // end: NULL terminates + } + } + + // listHead: [0]=link_offset, [4]=??, [8]=first_node + var list_head = [3]u32{ + 4, // link_offset + 0, + @intCast(@intFromPtr(&nodes)), // first node + }; + + // Result buffer: addGeometryToBuffer writes here. Just needs writable memory. + var result_buf: [4096]u8 = std.mem.zeroes([4096]u8); + + // flags: 0xF (low nibble set, matching type discriminator for both-zero type) + const flags: u32 = 0x8F; // bit 7 set + low nibble + + // --- Correctness check --- + const of = origFn(fn (u32, u32, u32, u32) callconv(cc_fc) u32, 0x6ABC40); + + // Reset visited markers before each call + for (0..NODE_COUNT) |i| { + @as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0; + } + const ret_orig = of(a(&list_head), a(&query_box), a(&result_buf), flags); + + for (0..NODE_COUNT) |i| { + @as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0; + } + const ret_sse = si_processLinkedListCollision(a(&list_head), a(&query_box), a(&result_buf), flags); + const ok = ret_orig == ret_sse; + + // --- Benchmark --- + var t: u64 = std.math.maxInt(u64); + for (0..5) |_| { + const _t0 = rdtsc(); + for (0..ITERS) |_| { + // Reset visited markers each iteration (original marks them) + for (0..NODE_COUNT) |i| { + @as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0; + } + _ = of(a(&list_head), a(&query_box), a(&result_buf), flags); + } + const _te = rdtsc() - _t0; + if (_te < t) t = _te; + } + + var s: u64 = std.math.maxInt(u64); + for (0..5) |_| { + const _t0 = rdtsc(); + for (0..ITERS) |_| { + for (0..NODE_COUNT) |i| { + @as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0; + } + _ = si_processLinkedListCollision(a(&list_head), a(&query_box), a(&result_buf), flags); + } + const _te = rdtsc() - _t0; + if (_te < s) s = _te; + } + report("processLinkedListCollision", t, s, ok); +} + // ========================================================================= // Generic benchmarks for common signatures (called versions) // ========================================================================= diff --git a/src/silicon/silicon.zig b/src/silicon/silicon.zig index 1edd4fe..faab7fe 100644 --- a/src/silicon/silicon.zig +++ b/src/silicon/silicon.zig @@ -2088,6 +2088,7 @@ const sse = struct { extern fn si_ftol() callconv(.naked) void; extern fn si_vec3Dot() callconv(.naked) void; extern fn si_translateBoundingVol(u32, u32) callconv(TC) void; + extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(FC) u32; }; const PatchEntry = struct { @@ -2122,6 +2123,7 @@ fn getPatchTable() []const PatchEntry { .{ .target = 0x40A2B0, .replacement = @intFromPtr(&sse.si_ftol), .name = "__ftol", .direct_size = 9 }, // vec3Dot (0x602630): removed — 0.6x regression, x87 is optimal for this ABI .{ .target = 0x686820, .replacement = @intFromPtr(&sse.si_translateBoundingVol), .name = "translateBoundingVol" }, + .{ .target = 0x6ABC40, .replacement = @intFromPtr(&sse.si_processLinkedListCollision), .name = "processLinkedListCollision" }, }; return &table; } diff --git a/src/silicon/silicon_sse.zig b/src/silicon/silicon_sse.zig index 55b38b3..624a676 100644 --- a/src/silicon/silicon_sse.zig +++ b/src/silicon/silicon_sse.zig @@ -515,3 +515,81 @@ export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void { obj[48] += dx; obj[49] += dy; obj[50] += dz; obj[51] += dx; obj[52] += dy; obj[53] += dz; } + +// --- 0x6ABC40: processLinkedListCollision --- +// Walks intrusive linked list, per-node AABB overlap test, calls addGeometryToBuffer on hit. +// Original: 329 bytes, 6 x87 FCOMP/FNSTSW comparisons per node. +// SSE: 2 V4 loads + 2 CMPPS + AND + MOVMSK replaces the 6 scalar comparisons. +// +// __fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack), RET 0x8 +// addGeometryToBuffer at 0x6ABD90: __fastcall(queryBox_ECX, nodeData_EDX, resultBuf_stack), RET 0x4 +// Visited sentinel: *(u32*)0xC89F20 +export fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 { + if (flags & 0xF0000F == 0) return 1; + + const addGeometryToBuffer: *const fn (u32, u32, u32) callconv(FC) void = @ptrFromInt(0x6ABD90); + + // Load query box min/max as V4 for SSE AABB test + // queryBox layout: min(+0,+4,+8), max(+0xC,+0x10,+0x14) + const q_min = loadV4(query_box); // {qmin.x, qmin.y, qmin.z, } + const q_max = loadV4(query_box + 0x0C); // {qmax.x, qmax.y, qmax.z, } + + const sentinel = @as(*const u32, @ptrFromInt(0xC89F20)).*; + const link_offset = @as(*const u32, @ptrFromInt(list_head)).*; + + // First node: listHead[2] (offset +8) + var node: u32 = @as(*const u32, @ptrFromInt(list_head + 8)).*; + + // Linked list tag bit: bit 0 set = end sentinel + if (node & 1 != 0 or node == 0) return 1; + + while (node & 1 == 0 and node != 0) { + const node_data = @as(*const u32, @ptrFromInt(node + 4)).*; + const prev_node = node; + + // Skip: flags bit 0x100 set + const node_flags = @as(*const u16, @ptrFromInt(node_data + 0x0C)).*; + if (node_flags & 0x100 == 0) { + // Skip: already visited or null + const visited = @as(*const u32, @ptrFromInt(node_data + 0x8C)).*; + const active = @as(*const u32, @ptrFromInt(node_data + 0x88)).*; + if (visited != sentinel and active != 0) { + // Type discriminator: pick flag mask + const type_a = @as(*const u32, @ptrFromInt(node_data + 0x180)).*; + const type_b = @as(*const u32, @ptrFromInt(node_data + 0x184)).*; + const mask = if (type_a | type_b != 0) flags & 0xF00000 else flags & 0xF; + + if (mask != 0) { + // Bit 7 of flags byte: if clear, abort with 0 + if (@as(i8, @bitCast(@as(u8, @truncate(node_flags)))) >= 0) return 0; + + // --- SSE AABB overlap test --- + // node AABB at node_data+0x14C: min(3 floats), max(3 floats) + const n_min = loadV4(node_data + 0x14C); // {nmin.x, nmin.y, nmin.z, } + const n_max = loadV4(node_data + 0x158); // {nmax.x, nmax.y, nmax.z, } + + // Overlap: nodeMin < queryMax AND queryMin <= nodeMax + // Compare lane-wise, check low 3 bits of mask + const lt_mask = n_min < q_max; + const le_mask = q_min <= n_max; + const lt_bits: u4 = @bitCast(lt_mask); + const le_bits: u4 = @bitCast(le_mask); + const bits = lt_bits & le_bits; + + if (bits & 0x7 == 0x7) { + addGeometryToBuffer(query_box, node_data, result_buf); + } + + // Mark visited + @as(*u32, @ptrFromInt(node_data + 0x8C)).* = sentinel; + } + } + } + + // Advance: next = *(node + link_offset + 4) + // Original: MOV ECX,[EAX + EDX*1 + 4] where EAX=*listHead, EDX=node + node = @as(*const u32, @ptrFromInt(link_offset + prev_node + 4)).*; + } + + return 1; +} diff --git a/src/transform44/bone_sse.zig b/src/transform44/bone_sse.zig index eeef1e1..fccf4f8 100644 --- a/src/transform44/bone_sse.zig +++ b/src/transform44/bone_sse.zig @@ -520,6 +520,23 @@ const InterpResult = struct { t: f32, }; +/// Check if two AnimData tracks share the same temporal structure, +/// meaning findInterpIdx would produce identical (idx0, idx1, t) for both. +/// Both must use prim_time (time_index == -1) and have matching range/timestamp layout. +inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool { + if (ri16(ref_anim + AD.time_index) != -1) return false; + if (ri16(other_anim + AD.time_index) != -1) return false; + return ru32(ref_anim + AD.track_count_flag) == ru32(other_anim + AD.track_count_flag) and + ru32(ref_anim + AD.keyframe_ranges) == ru32(other_anim + AD.keyframe_ranges) and + ru32(ref_anim + AD.timestamps_ptr) == ru32(other_anim + AD.timestamps_ptr) and + ru32(ref_anim + AD.keyframe_count) == ru32(other_anim + AD.keyframe_count); +} + +/// Write temporal coherence cache for a reused result so next frame's forward scan starts right. +inline fn applyCachedResult(cached: InterpResult, output: u32) void { + wu32(output, cached.idx0); +} + /// findInterpIdx: temporal-coherence keyframe search. /// Reimplementation of game function at 0x713D50 (334 bytes). /// Assembly-verified against t44_helpers_asm.txt. @@ -643,7 +660,16 @@ inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_dat /// Quaternion keyframe interpolation — replaces game's 0x713EA0. /// Assembly-verified: stride 16 (SHL EAX,4), values are 4×float, not CompQuat. inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 { - const r = findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); + return interpAnimKFCached(this, bone_rt, anim_data, output, null); +} + +/// Quaternion keyframe interpolation with optional cached primary InterpResult. +/// When cached_primary is non-null, skips findInterpIdx and uses the cached indices/t. +inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 { + const r = if (cached_primary) |c| blk: { + applyCachedResult(c, output); + break :blk c; + } else findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); const mode = ri16(anim_data + AD.interp_mode); const kf_base = ru32(anim_data + AD.keyframe_base); @@ -730,7 +756,22 @@ inline fn interpVec3Track( output: u32, blend_weight: f32, ) [3]f32 { - const r = findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); + return interpVec3TrackCached(this, bone_rt, anim_data, output, blend_weight, null); +} + +/// Vec3 keyframe interpolation with optional cached primary InterpResult. +inline fn interpVec3TrackCached( + this: u32, + bone_rt: u32, + anim_data: u32, + output: u32, + blend_weight: f32, + cached_primary: ?InterpResult, +) [3]f32 { + const r = if (cached_primary) |c| blk: { + applyCachedResult(c, output); + break :blk c; + } else findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); const interp_mode = ri16(anim_data + AD.interp_mode); const kf_base = ru32(anim_data + AD.keyframe_base); @@ -1530,10 +1571,19 @@ export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u3 const rot_anim = bdef + BD.rot_anim; const rot_kf_count = ru32(bdef + BD.rot_nts); + // Capture primary InterpResult from rotation for reuse by scale/translation. + // findInterpIdx is the #1 leaf function in the engine; eliminating redundant + // calls saves ~70 cycles/bone (~23% of bone loop baseline). + var rot_primary_cache: ?InterpResult = null; + // Rotation overwrites all 16 floats — skip identity init when present if (rot_kf_count != 0) { if (frame_ctr < rot_kf_count) { - const q = interpAnimKF(this, brt, rot_anim, brt + BR.rot_idx0); + // Call findInterpIdx via interpAnimKF — capture result for reuse + const rot_output = brt + BR.rot_idx0; + const r = findInterpIdx(this, ru32(brt + BR.prim_time), ru32(brt + BR.prim_track), rot_anim, rot_output); + rot_primary_cache = r; + const q = interpAnimKFCached(this, brt, rot_anim, rot_output, r); local_mat2 = buildRotationMatrixVal(q[0], q[1], q[2], q[3]); } else { local_mat2 = buildRotationMatrixVal(rf32(brt + BR.rot_x), rf32(brt + BR.rot_y), rf32(brt + BR.rot_z), rf32(brt + BR.rot_w)); @@ -1550,7 +1600,9 @@ export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u3 var sy: f32 = undefined; var sz: f32 = undefined; if (frame_ctr < scale_kf_count) { - const s = interpVec3Track(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight))); + // Reuse rotation's search result if temporal structure matches + const scale_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, scale_anim)) rot_primary_cache else null; + const s = interpVec3TrackCached(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)), scale_cache); sx = s[0]; sy = s[1]; sz = s[2]; } else { sx = rf32(brt + BR.scale_x); sy = rf32(brt + BR.scale_y); sz = rf32(brt + BR.scale_z); @@ -1574,7 +1626,9 @@ export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u3 const trans_kf_count = ru32(bdef + BD.trans_nts); if (trans_kf_count != 0) { if (frame_ctr < trans_kf_count) { - const t = interpVec3Track(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight))); + // Reuse rotation's search result if temporal structure matches + const trans_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, trans_anim)) rot_primary_cache else null; + const t = interpVec3TrackCached(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)), trans_cache); tx_val += t[0]; ty_val += t[1]; tz_val += t[2]; diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index e74dadf..d3ba89c 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -298,10 +298,8 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco if (teardown_active) { transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - } else if (ab_use_custom) { - transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); } else { - transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4); + transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); } t44_depth -|= 1;