perf: findInterpIdx dedup, processLinkedListCollision SSE, permanent t44

- bone_sse: deduplicate findInterpIdx calls in bone loop — rotation's
  search result reused for scale/translation when tracks share temporal
  structure (canReuseInterp guard). Est. ~23% bone loop cycle reduction.
- silicon: SSE replacement for processLinkedListCollision (0x6ABC40,
  1.57% CPU). V4 AABB overlap test replaces 6 x87 FCOMP/FNSTSW.
  Benched at 3.2x speedup (378→115 cyc/call, 8 nodes).
- transform44: remove A/B toggle, always use SSE path (teardown guard
  kept). A/B infrastructure remains for other hooks.
- bench: add processLinkedListCollision benchmark with fake linked list
  test fixture and stubbed addGeometryToBuffer.
This commit is contained in:
MarcelineVQ
2026-03-23 20:27:26 -07:00
parent b0b70a744e
commit 7819d6d914
5 changed files with 283 additions and 8 deletions
+143
View File
@@ -49,6 +49,7 @@ extern fn si_mulMat3x4InPlace(u32, u32) callconv(cc_tc) u32;
extern fn si_normalizeVec3InPlace(u32) callconv(cc_tc) void;
extern fn si_vec3Dot(u32, u32) callconv(cc_fc) f64;
extern fn si_translateBoundingVol(u32, u32) callconv(cc_tc) void;
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(cc_fc) u32;
extern fn si_addVec3ToAccumulator(u32, u32) callconv(cc_tc) void;
extern fn si_addToColorAccumulator(u32, u32) callconv(cc_tc) void;
extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void;
@@ -1767,9 +1768,151 @@ pub fn main() void {
}
}
// si_processLinkedListCollision -- fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack) -> u32
// Builds a fake linked list with 8 nodes to benchmark AABB overlap test.
bench_processLinkedListCollision();
print("\n", .{});
}
fn bench_processLinkedListCollision() void {
// Map page for sentinel global at 0xC89F20
_ = mapZeroed(0xC89000, 0x1000);
// Map page for addGeometryToBuffer's result_buf writes (just needs writable memory)
// Also need pages at 0xCA0000 range for any globals addGeometryToBuffer touches
const NODE_COUNT = 8;
// Sentinel: just a unique non-zero value. Original code reads *(u32*)0xC89F20.
const sentinel: u32 = 0xDEADBEEF;
@as(*u32, @ptrFromInt(0xC89F20)).* = sentinel;
// --- Build fake node data blocks (need offsets: +0x0C, +0x88, +0x8C, +0x14C-0x164, +0x180, +0x184) ---
// Each node_data needs at least 0x188 bytes
const NODE_DATA_SIZE = 0x190;
var node_data_buf: [NODE_COUNT * NODE_DATA_SIZE]u8 align(4) = std.mem.zeroes([NODE_COUNT * NODE_DATA_SIZE]u8);
// Query box: min=(0,0,0), max=(10,10,10)
var query_box = [6]f32{ 0.0, 0.0, 0.0, 10.0, 10.0, 10.0 };
// Stub addGeometryToBuffer at 0x6ABD90 → RET 0x4 (just returns, no side effects).
// Both original and SSE call the same stub, isolating the linked list walk + AABB test.
// Original bytes are in mapped .text — overwrite with: C2 04 00 (RET 4)
@as(*[3]u8, @ptrFromInt(0x6ABD90)).* = .{ 0xC2, 0x04, 0x00 };
// Set up each node_data
for (0..NODE_COUNT) |i| {
const nd = @intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE;
// flags at +0x0C: bit 0x80 set (required, else returns 0), no 0x100 (not skipped)
@as(*align(1) u16, @ptrFromInt(nd + 0x0C)).* = 0x80;
// active at +0x88: non-zero (just needs to pass != 0 check)
@as(*align(1) u32, @ptrFromInt(nd + 0x88)).* = 1;
// visited at +0x8C: NOT sentinel (so it gets processed)
@as(*align(1) u32, @ptrFromInt(nd + 0x8C)).* = 0;
// type discriminator: both zero → use flags & 0xF
@as(*align(1) u32, @ptrFromInt(nd + 0x180)).* = 0;
@as(*align(1) u32, @ptrFromInt(nd + 0x184)).* = 0;
// AABB at +0x14C: alternate overlapping and non-overlapping
const aabb: *align(1) [6]f32 = @ptrFromInt(nd + 0x14C);
if (i % 2 == 0) {
// Overlapping: min=(1,1,1), max=(5,5,5)
aabb.* = .{ 1.0, 1.0, 1.0, 5.0, 5.0, 5.0 };
} else {
// Non-overlapping: min=(20,20,20), max=(30,30,30)
aabb.* = .{ 20.0, 20.0, 20.0, 30.0, 30.0, 30.0 };
}
}
// --- Build linked list nodes ---
// Intrusive list: node = { ??, node_data_ptr, ... }
// link_offset stored at listHead[0], next at *(link_offset + node + 4)
// Simplest: link_offset = 0, so next = *(node + 4) ... no wait.
// Re-reading assembly: next = *(*(listHead) + prev_node + 4)
// listHead[0] = link_offset (byte offset within node to find next-ptr)
// Actually from the asm: MOV EAX,[EBP-0xc] (=listHead), MOV EAX,[EAX] (=*listHead = link_offset)
// MOV ECX,[EAX + EDX*1 + 4] where EDX=node
// So: next = *(link_offset + node + 4)
// If link_offset = 0: next = *(node + 4), but node+4 is node_data_ptr!
// We need link_offset such that (link_offset + node + 4) points to a "next" field.
// Let's use link_offset = 4, so next = *(node + 8).
// Node layout: [node_data_ptr(+0), ?(+4), next(+8)]
// But wait, node+4 is where node_data is read: MOV EBX,[EDX+4] (EDX=node)
// So node = { pad(+0), node_data(+4), next(+8) } and link_offset = 4.
const NODE_SIZE = 12; // pad, node_data_ptr, next_ptr
var nodes: [NODE_COUNT * NODE_SIZE]u8 align(4) = std.mem.zeroes([NODE_COUNT * NODE_SIZE]u8);
for (0..NODE_COUNT) |i| {
const n = @intFromPtr(&nodes) + i * NODE_SIZE;
// node+4 = node_data pointer
@as(*align(1) u32, @ptrFromInt(n + 4)).* = @intCast(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE);
// node+8 = next node (link_offset=4, so *(link_offset + node + 4) = *(node + 8))
if (i + 1 < NODE_COUNT) {
@as(*align(1) u32, @ptrFromInt(n + 8)).* = @intCast(@intFromPtr(&nodes) + (i + 1) * NODE_SIZE);
} else {
@as(*align(1) u32, @ptrFromInt(n + 8)).* = 0; // end: NULL terminates
}
}
// listHead: [0]=link_offset, [4]=??, [8]=first_node
var list_head = [3]u32{
4, // link_offset
0,
@intCast(@intFromPtr(&nodes)), // first node
};
// Result buffer: addGeometryToBuffer writes here. Just needs writable memory.
var result_buf: [4096]u8 = std.mem.zeroes([4096]u8);
// flags: 0xF (low nibble set, matching type discriminator for both-zero type)
const flags: u32 = 0x8F; // bit 7 set + low nibble
// --- Correctness check ---
const of = origFn(fn (u32, u32, u32, u32) callconv(cc_fc) u32, 0x6ABC40);
// Reset visited markers before each call
for (0..NODE_COUNT) |i| {
@as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0;
}
const ret_orig = of(a(&list_head), a(&query_box), a(&result_buf), flags);
for (0..NODE_COUNT) |i| {
@as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0;
}
const ret_sse = si_processLinkedListCollision(a(&list_head), a(&query_box), a(&result_buf), flags);
const ok = ret_orig == ret_sse;
// --- Benchmark ---
var t: u64 = std.math.maxInt(u64);
for (0..5) |_| {
const _t0 = rdtsc();
for (0..ITERS) |_| {
// Reset visited markers each iteration (original marks them)
for (0..NODE_COUNT) |i| {
@as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0;
}
_ = of(a(&list_head), a(&query_box), a(&result_buf), flags);
}
const _te = rdtsc() - _t0;
if (_te < t) t = _te;
}
var s: u64 = std.math.maxInt(u64);
for (0..5) |_| {
const _t0 = rdtsc();
for (0..ITERS) |_| {
for (0..NODE_COUNT) |i| {
@as(*align(1) u32, @ptrFromInt(@intFromPtr(&node_data_buf) + i * NODE_DATA_SIZE + 0x8C)).* = 0;
}
_ = si_processLinkedListCollision(a(&list_head), a(&query_box), a(&result_buf), flags);
}
const _te = rdtsc() - _t0;
if (_te < s) s = _te;
}
report("processLinkedListCollision", t, s, ok);
}
// =========================================================================
// Generic benchmarks for common signatures (called versions)
// =========================================================================
+2
View File
@@ -2088,6 +2088,7 @@ const sse = struct {
extern fn si_ftol() callconv(.naked) void;
extern fn si_vec3Dot() callconv(.naked) void;
extern fn si_translateBoundingVol(u32, u32) callconv(TC) void;
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(FC) u32;
};
const PatchEntry = struct {
@@ -2122,6 +2123,7 @@ fn getPatchTable() []const PatchEntry {
.{ .target = 0x40A2B0, .replacement = @intFromPtr(&sse.si_ftol), .name = "__ftol", .direct_size = 9 },
// vec3Dot (0x602630): removed — 0.6x regression, x87 is optimal for this ABI
.{ .target = 0x686820, .replacement = @intFromPtr(&sse.si_translateBoundingVol), .name = "translateBoundingVol" },
.{ .target = 0x6ABC40, .replacement = @intFromPtr(&sse.si_processLinkedListCollision), .name = "processLinkedListCollision" },
};
return &table;
}
+78
View File
@@ -515,3 +515,81 @@ export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
obj[48] += dx; obj[49] += dy; obj[50] += dz;
obj[51] += dx; obj[52] += dy; obj[53] += dz;
}
// --- 0x6ABC40: processLinkedListCollision ---
// Walks intrusive linked list, per-node AABB overlap test, calls addGeometryToBuffer on hit.
// Original: 329 bytes, 6 x87 FCOMP/FNSTSW comparisons per node.
// SSE: 2 V4 loads + 2 CMPPS + AND + MOVMSK replaces the 6 scalar comparisons.
//
// __fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack), RET 0x8
// addGeometryToBuffer at 0x6ABD90: __fastcall(queryBox_ECX, nodeData_EDX, resultBuf_stack), RET 0x4
// Visited sentinel: *(u32*)0xC89F20
export fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 {
if (flags & 0xF0000F == 0) return 1;
const addGeometryToBuffer: *const fn (u32, u32, u32) callconv(FC) void = @ptrFromInt(0x6ABD90);
// Load query box min/max as V4 for SSE AABB test
// queryBox layout: min(+0,+4,+8), max(+0xC,+0x10,+0x14)
const q_min = loadV4(query_box); // {qmin.x, qmin.y, qmin.z, <garbage>}
const q_max = loadV4(query_box + 0x0C); // {qmax.x, qmax.y, qmax.z, <garbage>}
const sentinel = @as(*const u32, @ptrFromInt(0xC89F20)).*;
const link_offset = @as(*const u32, @ptrFromInt(list_head)).*;
// First node: listHead[2] (offset +8)
var node: u32 = @as(*const u32, @ptrFromInt(list_head + 8)).*;
// Linked list tag bit: bit 0 set = end sentinel
if (node & 1 != 0 or node == 0) return 1;
while (node & 1 == 0 and node != 0) {
const node_data = @as(*const u32, @ptrFromInt(node + 4)).*;
const prev_node = node;
// Skip: flags bit 0x100 set
const node_flags = @as(*const u16, @ptrFromInt(node_data + 0x0C)).*;
if (node_flags & 0x100 == 0) {
// Skip: already visited or null
const visited = @as(*const u32, @ptrFromInt(node_data + 0x8C)).*;
const active = @as(*const u32, @ptrFromInt(node_data + 0x88)).*;
if (visited != sentinel and active != 0) {
// Type discriminator: pick flag mask
const type_a = @as(*const u32, @ptrFromInt(node_data + 0x180)).*;
const type_b = @as(*const u32, @ptrFromInt(node_data + 0x184)).*;
const mask = if (type_a | type_b != 0) flags & 0xF00000 else flags & 0xF;
if (mask != 0) {
// Bit 7 of flags byte: if clear, abort with 0
if (@as(i8, @bitCast(@as(u8, @truncate(node_flags)))) >= 0) return 0;
// --- SSE AABB overlap test ---
// node AABB at node_data+0x14C: min(3 floats), max(3 floats)
const n_min = loadV4(node_data + 0x14C); // {nmin.x, nmin.y, nmin.z, <nmax.x>}
const n_max = loadV4(node_data + 0x158); // {nmax.x, nmax.y, nmax.z, <garbage>}
// Overlap: nodeMin < queryMax AND queryMin <= nodeMax
// Compare lane-wise, check low 3 bits of mask
const lt_mask = n_min < q_max;
const le_mask = q_min <= n_max;
const lt_bits: u4 = @bitCast(lt_mask);
const le_bits: u4 = @bitCast(le_mask);
const bits = lt_bits & le_bits;
if (bits & 0x7 == 0x7) {
addGeometryToBuffer(query_box, node_data, result_buf);
}
// Mark visited
@as(*u32, @ptrFromInt(node_data + 0x8C)).* = sentinel;
}
}
}
// Advance: next = *(node + link_offset + 4)
// Original: MOV ECX,[EAX + EDX*1 + 4] where EAX=*listHead, EDX=node
node = @as(*const u32, @ptrFromInt(link_offset + prev_node + 4)).*;
}
return 1;
}
+59 -5
View File
@@ -520,6 +520,23 @@ const InterpResult = struct {
t: f32,
};
/// Check if two AnimData tracks share the same temporal structure,
/// meaning findInterpIdx would produce identical (idx0, idx1, t) for both.
/// Both must use prim_time (time_index == -1) and have matching range/timestamp layout.
inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
if (ri16(ref_anim + AD.time_index) != -1) return false;
if (ri16(other_anim + AD.time_index) != -1) return false;
return ru32(ref_anim + AD.track_count_flag) == ru32(other_anim + AD.track_count_flag) and
ru32(ref_anim + AD.keyframe_ranges) == ru32(other_anim + AD.keyframe_ranges) and
ru32(ref_anim + AD.timestamps_ptr) == ru32(other_anim + AD.timestamps_ptr) and
ru32(ref_anim + AD.keyframe_count) == ru32(other_anim + AD.keyframe_count);
}
/// Write temporal coherence cache for a reused result so next frame's forward scan starts right.
inline fn applyCachedResult(cached: InterpResult, output: u32) void {
wu32(output, cached.idx0);
}
/// findInterpIdx: temporal-coherence keyframe search.
/// Reimplementation of game function at 0x713D50 (334 bytes).
/// Assembly-verified against t44_helpers_asm.txt.
@@ -643,7 +660,16 @@ inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_dat
/// Quaternion keyframe interpolation — replaces game's 0x713EA0.
/// Assembly-verified: stride 16 (SHL EAX,4), values are 4×float, not CompQuat.
inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 {
const r = findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output);
return interpAnimKFCached(this, bone_rt, anim_data, output, null);
}
/// Quaternion keyframe interpolation with optional cached primary InterpResult.
/// When cached_primary is non-null, skips findInterpIdx and uses the cached indices/t.
inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 {
const r = if (cached_primary) |c| blk: {
applyCachedResult(c, output);
break :blk c;
} else findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output);
const mode = ri16(anim_data + AD.interp_mode);
const kf_base = ru32(anim_data + AD.keyframe_base);
@@ -730,7 +756,22 @@ inline fn interpVec3Track(
output: u32,
blend_weight: f32,
) [3]f32 {
const r = findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output);
return interpVec3TrackCached(this, bone_rt, anim_data, output, blend_weight, null);
}
/// Vec3 keyframe interpolation with optional cached primary InterpResult.
inline fn interpVec3TrackCached(
this: u32,
bone_rt: u32,
anim_data: u32,
output: u32,
blend_weight: f32,
cached_primary: ?InterpResult,
) [3]f32 {
const r = if (cached_primary) |c| blk: {
applyCachedResult(c, output);
break :blk c;
} else findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output);
const interp_mode = ri16(anim_data + AD.interp_mode);
const kf_base = ru32(anim_data + AD.keyframe_base);
@@ -1530,10 +1571,19 @@ export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u3
const rot_anim = bdef + BD.rot_anim;
const rot_kf_count = ru32(bdef + BD.rot_nts);
// Capture primary InterpResult from rotation for reuse by scale/translation.
// findInterpIdx is the #1 leaf function in the engine; eliminating redundant
// calls saves ~70 cycles/bone (~23% of bone loop baseline).
var rot_primary_cache: ?InterpResult = null;
// Rotation overwrites all 16 floats — skip identity init when present
if (rot_kf_count != 0) {
if (frame_ctr < rot_kf_count) {
const q = interpAnimKF(this, brt, rot_anim, brt + BR.rot_idx0);
// Call findInterpIdx via interpAnimKF — capture result for reuse
const rot_output = brt + BR.rot_idx0;
const r = findInterpIdx(this, ru32(brt + BR.prim_time), ru32(brt + BR.prim_track), rot_anim, rot_output);
rot_primary_cache = r;
const q = interpAnimKFCached(this, brt, rot_anim, rot_output, r);
local_mat2 = buildRotationMatrixVal(q[0], q[1], q[2], q[3]);
} else {
local_mat2 = buildRotationMatrixVal(rf32(brt + BR.rot_x), rf32(brt + BR.rot_y), rf32(brt + BR.rot_z), rf32(brt + BR.rot_w));
@@ -1550,7 +1600,9 @@ export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u3
var sy: f32 = undefined;
var sz: f32 = undefined;
if (frame_ctr < scale_kf_count) {
const s = interpVec3Track(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)));
// Reuse rotation's search result if temporal structure matches
const scale_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, scale_anim)) rot_primary_cache else null;
const s = interpVec3TrackCached(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)), scale_cache);
sx = s[0]; sy = s[1]; sz = s[2];
} else {
sx = rf32(brt + BR.scale_x); sy = rf32(brt + BR.scale_y); sz = rf32(brt + BR.scale_z);
@@ -1574,7 +1626,9 @@ export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u3
const trans_kf_count = ru32(bdef + BD.trans_nts);
if (trans_kf_count != 0) {
if (frame_ctr < trans_kf_count) {
const t = interpVec3Track(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)));
// Reuse rotation's search result if temporal structure matches
const trans_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, trans_anim)) rot_primary_cache else null;
const t = interpVec3TrackCached(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)), trans_cache);
tx_val += t[0];
ty_val += t[1];
tz_val += t[2];
+1 -3
View File
@@ -298,10 +298,8 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
if (teardown_active) {
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
} else if (ab_use_custom) {
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
} else {
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
}
t44_depth -|= 1;