diff --git a/build.zig b/build.zig index 26d1e2a..404f5c5 100644 --- a/build.zig +++ b/build.zig @@ -65,6 +65,22 @@ pub fn build(b: *std.Build) void { .optimize = .ReleaseFast, }), }); + const cull_sse_obj = b.addObject(.{ + .name = "cull_sse", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/performance/cull_sse.zig"), + .target = target, + .optimize = .ReleaseFast, + }), + }); + const entity_sse_obj = b.addObject(.{ + .name = "entity_sse", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/performance/entity_sse.zig"), + .target = target, + .optimize = .ReleaseFast, + }), + }); const bone_sse_target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .windows, @@ -172,6 +188,8 @@ pub fn build(b: *std.Build) void { // Called for both the main weirdutils build and each variant. const ModuleObjects = struct { clip_sse: *std.Build.Step.Compile, + cull_sse: *std.Build.Step.Compile, + entity_sse: *std.Build.Step.Compile, bone_sse: *std.Build.Step.Compile, bone_sse_ref: *std.Build.Step.Compile, math_sse: *std.Build.Step.Compile, @@ -184,6 +202,8 @@ pub fn build(b: *std.Build) void { @setEvalBranchQuota(10000); if (comptime std.mem.eql(u8, module_name, "weirdperformance")) { mod.addObject(self.clip_sse); + mod.addObject(self.cull_sse); + mod.addObject(self.entity_sse); mod.addObject(self.bone_sse); mod.addObject(self.bone_sse_ref); mod.addObject(self.silicon_sse); @@ -193,6 +213,8 @@ pub fn build(b: *std.Build) void { } if (comptime std.mem.eql(u8, module_name, "transform44")) { mod.addObject(self.clip_sse); + mod.addObject(self.cull_sse); + mod.addObject(self.entity_sse); mod.addObject(self.bone_sse); mod.addObject(self.bone_sse_ref); mod.addObject(self.particle_sse); @@ -208,6 +230,8 @@ pub fn build(b: *std.Build) void { }; const objs = ModuleObjects{ .clip_sse = clip_sse_obj, + .cull_sse = cull_sse_obj, + .entity_sse = entity_sse_obj, .bone_sse = bone_sse_obj, .bone_sse_ref = bone_sse_ref_obj, .math_sse = math_sse_obj, @@ -299,11 +323,29 @@ pub fn build(b: *std.Build) void { .optimize = .ReleaseFast, }), }); + const bench_cull_sse = b.addObject(.{ + .name = "bench_cull_sse", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/performance/cull_sse.zig"), + .target = bench_target, + .optimize = .ReleaseFast, + }), + }); bench.root_module.addObject(bench_math_sse); bench.root_module.addObject(bench_silicon_sse); bench.root_module.addObject(bench_bone_sse); bench.root_module.addObject(bench_bone_baseline); bench.root_module.addObject(bench_particle_sse); + bench.root_module.addObject(bench_cull_sse); + const bench_entity_sse = b.addObject(.{ + .name = "bench_entity_sse", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/performance/entity_sse.zig"), + .target = bench_target, + .optimize = .ReleaseFast, + }), + }); + bench.root_module.addObject(bench_entity_sse); bench.root_module.linkSystemLibrary("m", .{}); const install_bench = b.addInstallArtifact(bench, .{}); const bench_step = b.step("bench", "Build math_sse benchmark harness (x86 Linux)"); diff --git a/src/bench/main.zig b/src/bench/main.zig index 1900323..656bae4 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -58,6 +58,11 @@ extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void; extern fn si_setParticleAlpha(u32, u32, u32) callconv(cc_fc) void; // fastcall(ECX=obj, EDX=unused, stack=alpha) extern fn si_ftol() callconv(.naked) void; +// cull_sse.zig exports +extern fn benchComputeOutcodes(u32, u32, u32, u32) void; +extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; + // ========================================================================= // Infrastructure // ========================================================================= @@ -231,6 +236,14 @@ pub fn main() void { print("{s:>30} {s:>10} {s:>10} {s}\n", .{ "function", "original", "sse", "status" }); print("{s}\n", .{"-" ** 72}); + // Full performCollisionDetection (SSE vs original x87) + bench_collisionDetection(); + + // UpdateEntityAndChunksPositions (SSE vs original x87) + bench_entityUpdate(); + + if (false) { // disabled: not working on these right now + // 1: vecMulMat4 -- fastcall(ECX=result, EDX=vec, stack=mat) -> u32 bench_fc3r("vecMulMat4_ColMajor", originals.vecMulMat4_ColMajor, &vecMulMat4_ColMajor, tv3(), tm4(), 3); @@ -1770,8 +1783,8 @@ pub fn main() void { } } - // calcColorValues_SSE -- thiscall(ctx_ECX, time, scale, outColor, outAlpha1, outAlpha2, outFloat) - bench_calcColorValues(); + // calcColorValues_SSE -- disabled: no standalone SSE export yet + // bench_calcColorValues(); // si_frustumCullBBox -- fastcall(bbox_ECX, flags_EDX, radius_stack) -> u32 bench_frustumCullBBox(); @@ -1780,6 +1793,8 @@ pub fn main() void { // Builds a fake linked list with 8 nodes to benchmark AABB overlap test. bench_processLinkedListCollision(); + } // end disabled block + print("\n", .{}); } @@ -2174,6 +2189,29 @@ fn bench_tc2r( // ========================================================================= const V4 = @Vector(4, f32); +const ShufMask = @Vector(4, i32); + +inline fn benchLoadV3(addr: u32) V4 { + return .{ + @as(*align(1) const f32, @ptrFromInt(addr)).*, + @as(*align(1) const f32, @ptrFromInt(addr + 4)).*, + @as(*align(1) const f32, @ptrFromInt(addr + 8)).*, + 0, + }; +} + +inline fn benchCross(av: V4, bv: V4) V4 { + const a_yzx: V4 = @shuffle(f32, av, undefined, ShufMask{ 1, 2, 0, 3 }); + const a_zxy: V4 = @shuffle(f32, av, undefined, ShufMask{ 2, 0, 1, 3 }); + const b_yzx: V4 = @shuffle(f32, bv, undefined, ShufMask{ 1, 2, 0, 3 }); + const b_zxy: V4 = @shuffle(f32, bv, undefined, ShufMask{ 2, 0, 1, 3 }); + return a_yzx * b_zxy - a_zxy * b_yzx; +} + +inline fn benchDot3(av: V4, bv: V4) f32 { + const p = av * bv; + return p[0] + p[1] + p[2]; +} inline fn inline_x87_dot(va: *const Vec3, vb: *const Vec3, out: *f32) void { asm volatile ( @@ -2275,3 +2313,464 @@ inline fn inline_sse_horner(c: *const [4]f32, f: f32, out: *volatile f32) void { r = r * f + c.*[3]; out.* = r; } + +fn sseRayTri(ray_ptr: u32, vert_pool: u32, idx_base: u32, t_out: *f32) bool { + const vi0: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base)).*; + const vi1: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base + 2)).*; + const vi2: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base + 4)).*; + + const ray_o = benchLoadV3(ray_ptr); + const ray_d = benchLoadV3(ray_ptr + 12); + const v0 = benchLoadV3(vert_pool + vi0 * 12); + const v1 = benchLoadV3(vert_pool + vi1 * 12); + const v2 = benchLoadV3(vert_pool + vi2 * 12); + + const edge1 = v1 - v0; + const edge2 = v2 - v0; + const pvec = benchCross(ray_d, edge2); + const det = benchDot3(edge1, pvec); + if (det <= 1e-7 and det >= -1e-7) return false; + + const inv_det = 1.0 / det; + const tvec = ray_o - v0; + const u = benchDot3(tvec, pvec) * inv_det; + if (u < -0.002 or u > 1.002) return false; + + const qvec = benchCross(tvec, edge1); + const v = benchDot3(ray_d, qvec) * inv_det; + if (v < -0.002 or (u + v) > 1.002) return false; + + t_out.* = benchDot3(edge2, qvec) * inv_det; + return true; +} + +fn bench_collisionDetection() void { + // Build synthetic mesh data matching game's hash entry layout. + // 32 vertices forming a grid, 20 triangles, ray aimed through the middle. + + const NVERTS = 120; + const NTRIS = 40; + + // Hash entry: total size must accommodate all fields up to 0x2206 + NTRIS*2 + // Max offset: 0x2206 + 20*2 = 0x222E, round up + var hash_buf: [0x2300]u8 align(4) = [_]u8{0} ** 0x2300; + const he = @intFromPtr(&hash_buf); + + // Vertex count at +6 + @as(*align(1) u16, @ptrFromInt(he + 6)).* = NVERTS; + + // Carefully crafted vertices to produce a mix of hits and misses with + // non-trivial barycentric coordinates. Ray fires from (0,0,-10) along +Z. + // Triangles 0-4: guaranteed hits at various u/v (straddling the ray axis) + // Triangles 5-9: near-misses (edge/corner cases for barycentric bounds) + // Triangles 10-14: clear misses (outside AABB or backfacing) + // Triangles 15-19: more hits with small/large det values (tests divide precision) + const verts = [NVERTS][3]f32{ + // --- Group A: clear hits at various depths, u/v values --- + // Tri 0: large centered, hit u~0.33 v~0.33 + .{ -2.0, -2.0, 1.0 }, .{ 4.0, -2.0, 1.0 }, .{ -2.0, 4.0, 1.0 }, + // Tri 1: small on-axis, hit u~0.5 v~0.25 + .{ -0.5, -0.5, 2.0 }, .{ 0.5, -0.5, 2.0 }, .{ 0.0, 0.5, 2.0 }, + // Tri 2: very close to origin + .{ -1.0, -1.0, 0.1 }, .{ 1.0, -1.0, 0.1 }, .{ 0.0, 1.0, 0.1 }, + // Tri 3: backface hit (wound CW) + .{ -2.0, 4.0, 4.0 }, .{ 4.0, -2.0, 4.0 }, .{ -2.0, -2.0, 4.0 }, + // Tri 4: tiny triangle, tests large inv_det + .{ -0.05, -0.05, 1.5 }, .{ 0.05, -0.05, 1.5 }, .{ 0.0, 0.05, 1.5 }, + // Tri 5: huge triangle, tests small inv_det + .{ -50.0, -50.0, 2.5 }, .{ 50.0, -50.0, 2.5 }, .{ 0.0, 50.0, 2.5 }, + // Tri 6: hit at u~0, v~0 (near vertex 0) + .{ -0.001, -0.001, 3.0 }, .{ 5.0, -0.001, 3.0 }, .{ -0.001, 5.0, 3.0 }, + // Tri 7: hit at u~1, v~0 (near vertex 1) + .{ -5.0, -0.001, 3.5 }, .{ 0.001, -0.001, 3.5 }, .{ -5.0, 5.0, 3.5 }, + // Tri 8: hit at u~0, v~1 (near vertex 2) + .{ -5.0, -5.0, 4.0 }, .{ 5.0, -5.0, 4.0 }, .{ 0.001, 0.001, 4.0 }, + // Tri 9: hit with u+v very close to 1.0 (edge between v1-v2) + .{ -0.01, -0.01, 4.5 }, .{ 2.0, -0.01, 4.5 }, .{ -0.01, 2.0, 4.5 }, + + // --- Group B: edge cases that should barely miss --- + // Tri 10: ray just outside triangle edge + .{ 0.5, -0.5, 5.0 }, .{ 2.0, -0.5, 5.0 }, .{ 0.5, 1.0, 5.0 }, + // Tri 11: ray misses on v side + .{ -3.0, 0.5, 5.5 }, .{ -0.5, 0.5, 5.5 }, .{ -3.0, 2.0, 5.5 }, + // Tri 12: triangle behind ray (negative t) + .{ -1.0, -1.0, -15.0 }, .{ 1.0, -1.0, -15.0 }, .{ 0.0, 1.0, -15.0 }, + // Tri 13: triangle way off to the side + .{ 10.0, 10.0, 1.0 }, .{ 12.0, 10.0, 1.0 }, .{ 10.0, 12.0, 1.0 }, + // Tri 14: triangle off to the other side + .{ -12.0, -12.0, 2.0 }, .{ -10.0, -12.0, 2.0 }, .{ -12.0, -10.0, 2.0 }, + + // --- Group C: degenerate/parallel --- + // Tri 15: zero-area (all same point) + .{ 1.0, 1.0, 6.0 }, .{ 1.0, 1.0, 6.0 }, .{ 1.0, 1.0, 6.0 }, + // Tri 16: collinear vertices + .{ -1.0, 0.0, 7.0 }, .{ 0.0, 0.0, 7.0 }, .{ 1.0, 0.0, 7.0 }, + // Tri 17: nearly parallel to ray (plane nearly parallel to Z axis) + .{ -1.0, -100.0, 0.5 }, .{ 1.0, -100.0, 0.5 }, .{ 0.0, 100.0, 0.501 }, + // Tri 18: parallel to ray (exactly in XY plane at z=0, ray along Z) + .{ -1.0, -1.0, 0.0 }, .{ 1.0, -1.0, 0.0 }, .{ 0.0, 1.0, 0.0 }, + + // --- Group D: more hits at various depths for closest-t tracking --- + // Tri 19: closest possible hit + .{ -5.0, -5.0, 0.01 }, .{ 5.0, -5.0, 0.01 }, .{ 0.0, 5.0, 0.01 }, + // Tri 20-24: hits at regular depth intervals + .{ -0.3, -0.3, 0.5 }, .{ 0.3, -0.3, 0.5 }, .{ 0.0, 0.3, 0.5 }, + .{ -1.0, -1.0, 1.2 }, .{ 1.0, -1.0, 1.2 }, .{ 0.0, 1.0, 1.2 }, + .{ -0.8, -0.8, 2.0 }, .{ 0.8, -0.8, 2.0 }, .{ 0.0, 0.8, 2.0 }, + .{ -1.5, -1.5, 3.0 }, .{ 1.5, -1.5, 3.0 }, .{ 0.0, 1.5, 3.0 }, + .{ -2.0, -2.0, 5.5 }, .{ 2.0, -2.0, 5.5 }, .{ 0.0, 2.0, 5.5 }, + + // --- Group E: outside AABB (outcode rejects, never reach ray-tri) --- + // Tri 25: all verts above AABB + .{ -1.0, 5.0, 1.0 }, .{ 1.0, 5.0, 1.0 }, .{ 0.0, 6.0, 1.0 }, + // Tri 26: all verts below AABB + .{ -1.0, -6.0, 1.0 }, .{ 1.0, -6.0, 1.0 }, .{ 0.0, -5.0, 1.0 }, + // Tri 27: all verts left of AABB + .{ -6.0, -1.0, 1.0 }, .{ -5.0, -1.0, 1.0 }, .{ -6.0, 1.0, 1.0 }, + // Tri 28: all verts in front of AABB (z < min) + .{ -1.0, -1.0, -5.0 }, .{ 1.0, -1.0, -5.0 }, .{ 0.0, 1.0, -5.0 }, + // Tri 29: all verts behind AABB (z > max) + .{ -1.0, -1.0, 5.0 }, .{ 1.0, -1.0, 5.0 }, .{ 0.0, 1.0, 5.0 }, + + // --- Group F: asymmetric/skewed hits testing det sign & magnitude --- + // Tri 30: very elongated, hit near tip + .{ 0.0, -0.01, 1.8 }, .{ 0.02, -0.01, 1.8 }, .{ 0.0, 10.0, 1.8 }, + // Tri 31: very flat (nearly zero Y extent) + .{ -5.0, -0.001, 2.2 }, .{ 5.0, -0.001, 2.2 }, .{ 0.0, 0.001, 2.2 }, + // Tri 32: large negative det + .{ -3.0, 3.0, 2.8 }, .{ 3.0, -3.0, 2.8 }, .{ -3.0, -3.0, 2.8 }, + // Tri 33: det exactly at threshold boundary + .{ -0.0001, -0.0001, 6.5 }, .{ 0.0001, -0.0001, 6.5 }, .{ 0.0, 0.0001, 6.5 }, + + // --- Group G: stress closest-t with many competing hits --- + // Tri 34-39: hits at very close z-values to test precision + .{ -1.0, -1.0, 0.100 }, .{ 1.0, -1.0, 0.100 }, .{ 0.0, 1.0, 0.100 }, + .{ -1.0, -1.0, 0.101 }, .{ 1.0, -1.0, 0.101 }, .{ 0.0, 1.0, 0.101 }, + .{ -1.0, -1.0, 0.099 }, .{ 1.0, -1.0, 0.099 }, .{ 0.0, 1.0, 0.099 }, + .{ -1.0, -1.0, 0.102 }, .{ 1.0, -1.0, 0.102 }, .{ 0.0, 1.0, 0.102 }, + .{ -1.0, -1.0, 0.098 }, .{ 1.0, -1.0, 0.098 }, .{ 0.0, 1.0, 0.098 }, + .{ -1.0, -1.0, 0.103 }, .{ 1.0, -1.0, 0.103 }, .{ 0.0, 1.0, 0.103 }, + }; + + // Write vertices to hash entry at +8 + for (0..NVERTS) |vi| { + const off = he + 8 + vi * 12; + @as(*align(1) f32, @ptrFromInt(off)).* = verts[vi][0]; + @as(*align(1) f32, @ptrFromInt(off + 4)).* = verts[vi][1]; + @as(*align(1) f32, @ptrFromInt(off + 8)).* = verts[vi][2]; + } + + // Triangle count at +0x18A4 + @as(*align(1) u16, @ptrFromInt(he + 0x18A4)).* = NTRIS; + + // Each triangle uses 3 consecutive vertices: tri N -> verts N*3, N*3+1, N*3+2 + { + var ti: u32 = 0; + while (ti < NTRIS) : (ti += 1) { + const base: u16 = @intCast(ti * 3); + @as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6)).* = base; + @as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6 + 2)).* = base + 1; + @as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6 + 4)).* = base + 2; + @as(*align(1) u16, @ptrFromInt(he + 0x1FAE + ti * 2)).* = 0; + @as(*align(1) u16, @ptrFromInt(he + 0x2206 + ti * 2)).* = @intCast(ti); + } + } + + // Build "this" struct (needs ~0x54 bytes) + var this_buf: [0x60]u8 align(4) = [_]u8{0} ** 0x60; + const th = @intFromPtr(&this_buf); + + // Visited array: needs at least NTRIS*2 bytes + var visited: [64]u8 = [_]u8{0} ** 64; + + // Result float + var result_val: f32 = 0.0; + + // this+0x04 = visited array base + @as(*align(1) u32, @ptrFromInt(th + 0x04)).* = @intFromPtr(&visited); + // this+0x08, +0x0C = hash params (must match what FindOrCreateHashEntry expects, but + // we'll call our function directly bypassing the hash lookup, so these don't matter) + // this+0x10 = pointer to result float + @as(*align(1) u32, @ptrFromInt(th + 0x10)).* = @intFromPtr(&result_val); + // this+0x14 = clamp value + @as(*align(1) f32, @ptrFromInt(th + 0x14)).* = 100.0; + // this+0x18..0x2C = AABB extents (will be sorted by function) + @as(*align(1) f32, @ptrFromInt(th + 0x18)).* = -3.0; // ax0 + @as(*align(1) f32, @ptrFromInt(th + 0x1C)).* = -3.0; // ay0 + @as(*align(1) f32, @ptrFromInt(th + 0x20)).* = -3.0; // az0 + @as(*align(1) f32, @ptrFromInt(th + 0x24)).* = 3.0; // ax1 + @as(*align(1) f32, @ptrFromInt(th + 0x28)).* = 3.0; // ay1 + @as(*align(1) f32, @ptrFromInt(th + 0x2C)).* = 3.0; // az1 + // this+0x30..0x3B = ray origin + @as(*align(1) f32, @ptrFromInt(th + 0x30)).* = 0.0; + @as(*align(1) f32, @ptrFromInt(th + 0x34)).* = 0.0; + @as(*align(1) f32, @ptrFromInt(th + 0x38)).* = -10.0; + // this+0x3C..0x47 = ray direction + @as(*align(1) f32, @ptrFromInt(th + 0x3C)).* = 0.0; + @as(*align(1) f32, @ptrFromInt(th + 0x40)).* = 0.0; + @as(*align(1) f32, @ptrFromInt(th + 0x44)).* = 1.0; + // this+0x48 = scale factor + @as(*align(1) f32, @ptrFromInt(th + 0x48)).* = 1.0; + // this+0x4C = closest-t (large initial value) + @as(*align(1) f32, @ptrFromInt(th + 0x4C)).* = 999999.0; + // this+0x50 = collision mask + @as(*align(1) u16, @ptrFromInt(th + 0x50)).* = 0; + + // Map globals needed by both original and SSE functions + _ = mapZeroed(0xCA0000, 0x1000); // g_guard at 0xCA03E4 + _ = mapZeroed(0xCDE000, 0x1000); // g_render_list at 0xCDE648 + _ = mapZeroed(0xCE2000, 0x1000); // g_visible_count/list at 0xCE26E0/E8 + _ = mapZeroed(0xCE6000, 0x1000); // g_render_count at 0xCE66FC + + // Patch FindOrCreateHashEntry (0x693D60) to return our hash_buf: + // MOV EAX, ; B8 xx xx xx xx + // RET 0x14 ; C2 14 00 + const hash_stub = @as([*]u8, @ptrFromInt(0x693D60)); + hash_stub[0] = 0xB8; + @as(*align(1) u32, @ptrFromInt(0x693D61)).* = he; + hash_stub[5] = 0xC2; + hash_stub[6] = 0x14; + hash_stub[7] = 0x00; + + // Set g_guard to non-zero (both functions check this) + @as(*align(1) u32, @ptrFromInt(0xCA03E4)).* = 1; + + // Original function at 0x6B88E0 and our SSE version + const orig_fn = @as(*const fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x6B88E0)); + + // Reset state helper + const resetState = struct { + fn f(t: u32, v: *[64]u8) void { + @as(*align(1) f32, @ptrFromInt(t + 0x4C)).* = 999999.0; + @memset(v, 0); + @as(*u32, @ptrFromInt(0xCE26E0)).* = 0; // visible count + @as(*u32, @ptrFromInt(0xCE66FC)).* = 0; // render count + } + }.f; + + // Get truth values from original x87 function + resetState(th, &visited); + _ = @call(.never_tail, orig_fn, .{ th, 0, 0 }); + const orig_closest = @as(*align(1) f32, @ptrFromInt(th + 0x4C)).*; + const orig_result = result_val; + const orig_render_count = @as(*u32, @ptrFromInt(0xCE66FC)).*; + const orig_visible_count = @as(*u32, @ptrFromInt(0xCE26E0)).*; + + // Run SSE version + resetState(th, &visited); + _ = @call(.never_tail, performCollisionDetectionSSE, .{ th, 0, 0 }); + const sse_closest = @as(*align(1) f32, @ptrFromInt(th + 0x4C)).*; + const sse_result = result_val; + const sse_render_count = @as(*u32, @ptrFromInt(0xCE66FC)).*; + const sse_visible_count = @as(*u32, @ptrFromInt(0xCE26E0)).*; + + const t_match = @abs(orig_closest - sse_closest) < 0.01 or (orig_closest > 99999.0 and sse_closest > 99999.0); + const r_match = @abs(orig_result - sse_result) < 0.01; + const ok = t_match and r_match and orig_render_count == sse_render_count and orig_visible_count == sse_visible_count; + + if (!ok) { + print(" MISMATCH detail:\n", .{}); + print(" closest-t: orig={d:.6} sse={d:.6}\n", .{ orig_closest, sse_closest }); + print(" result: orig={d:.6} sse={d:.6}\n", .{ orig_result, sse_result }); + } else { + print(" closest-t={d:.4} result={d:.4} hits={d} visible={d}\n", .{ + sse_closest, sse_result, sse_render_count, sse_visible_count, + }); + } + + // Benchmark: our SSE performCollisionDetectionSSE + var best: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t0 = rdtsc(); + for (0..ITERS) |_| { + resetState(th, &visited); + _ = @call(.never_tail, performCollisionDetectionSSE, .{ th, 0, 0 }); + } + t0 = rdtsc() - t0; + if (t0 < best) best = t0; + } + + const per_call = best / ITERS; + const per_tri = if (NTRIS > 0) per_call / NTRIS else 0; + const status: [*:0]const u8 = if (ok) "OK" else "MISMATCH"; + print("{s:>30}: {d} cyc/call {d} cyc/tri ({d} tris) {s}\n", .{ + "performCollisionDet", per_call, per_tri, NTRIS, status, + }); +} + +fn bench_entityUpdate() void { + // Map .bss pages for globals the entity update reads/writes + _ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510 + _ = mapZeroed(0xC7B000, 0x2000); // view coeffs at 0xC7BCB0, bounds at 0xC7CB5C-C7CB70 + _ = mapZeroed(0x866000, 0x4000); // render flags at 0x867960, ptrs at 0x867964/68 + _ = mapZeroed(0x80A000, 0x1000); // timer threshold at 0x80A1E8 + _ = mapZeroed(0x86B000, 0x1000); // anim table at 0x86B580 + _ = mapZeroed(0xC7F000, 0x1000); // anim index at 0xC7F294 + + // Set up view coefficients (a,b,c,d) at 0xC7BCB0 + @as(*align(1) f32, @ptrFromInt(0xC7BCB0)).* = 0.5; // coeff for ent+0x5C + @as(*align(1) f32, @ptrFromInt(0xC7BCB4)).* = 0.3; // coeff for ent+0x60 + @as(*align(1) f32, @ptrFromInt(0xC7BCB8)).* = 0.7; // coeff for ent+0x64 + @as(*align(1) f32, @ptrFromInt(0xC7BCBC)).* = 1.0; // constant term + + // Delta time + @as(*align(1) f32, @ptrFromInt(0xC62510)).* = 0.016; // ~60fps + // Timer threshold + @as(*align(1) f32, @ptrFromInt(0x80A1E8)).* = 999.0; // high so recycling never triggers + // World bounds (set large so bounds check always passes) + @as(*align(1) f32, @ptrFromInt(0xC7CB68)).* = 999.0; + @as(*align(1) f32, @ptrFromInt(0xC7CB6C)).* = 999.0; + @as(*align(1) f32, @ptrFromInt(0xC7CB70)).* = 999.0; + // Disable spatial grid registration by making bounds check fail: + // Set the lower bounds high so IsPointInsideBounds returns false + @as(*align(1) f32, @ptrFromInt(0xC7CB5C)).* = 99999.0; + @as(*align(1) f32, @ptrFromInt(0xC7CB60)).* = 99999.0; + @as(*align(1) f32, @ptrFromInt(0xC7CB64)).* = 99999.0; + + // Build a synthetic entity struct (~0x900 bytes to cover all accessed fields) + var ent_buf: [0x900]u8 align(16) = [_]u8{0} ** 0x900; + const ent = @intFromPtr(&ent_buf); + + // Entity position fields for dot product + @as(*align(1) f32, @ptrFromInt(ent + 0x5C)).* = 10.0; + @as(*align(1) f32, @ptrFromInt(ent + 0x60)).* = 20.0; + @as(*align(1) f32, @ptrFromInt(ent + 0x64)).* = 30.0; + @as(*align(1) f32, @ptrFromInt(ent + 0x68)).* = 5.0; // depth offset + // Timer at ent+0xAC (start at 0) + @as(*align(1) f32, @ptrFromInt(ent + 0xAC)).* = 0.0; + // No vertex buffers (ent+0x14C = 0), no instances (ent+0xC0 = 0), no chunks + // Bounds fields that won't trigger spatial grid + @as(*align(1) f32, @ptrFromInt(ent + 0x44)).* = 999.0; // will fail < check + + const orig_fn = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AFAD0)); + + // Reset helper + const resetEnt = struct { + fn f(e: u32) void { + @as(*align(1) f32, @ptrFromInt(e + 0xAC)).* = 0.0; // reset timer + @as(*align(1) f32, @ptrFromInt(e + 0x78)).* = 0.0; // reset depth + } + }.f; + + // Correctness: both should compute the same depth + resetEnt(ent); + @call(.never_tail, orig_fn, .{ent}); + const orig_depth = @as(*align(1) f32, @ptrFromInt(ent + 0x78)).*; + + resetEnt(ent); + @call(.never_tail, updateEntityAndChunksPositions, .{ent}); + const sse_depth = @as(*align(1) f32, @ptrFromInt(ent + 0x78)).*; + + const ok = @abs(orig_depth - sse_depth) < 0.01; + if (!ok) { + print(" MISMATCH: orig_depth={d:.4} sse_depth={d:.4}\n", .{ orig_depth, sse_depth }); + } + + // Benchmark original + var orig_cyc: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t0 = rdtsc(); + for (0..ITERS) |_| { + resetEnt(ent); + @call(.never_tail, orig_fn, .{ent}); + } + t0 = rdtsc() - t0; + if (t0 < orig_cyc) orig_cyc = t0; + } + + // Benchmark SSE + var sse_cyc: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t0 = rdtsc(); + for (0..ITERS) |_| { + resetEnt(ent); + @call(.never_tail, updateEntityAndChunksPositions, .{ent}); + } + t0 = rdtsc() - t0; + if (t0 < sse_cyc) sse_cyc = t0; + } + + report("UpdateEntityChunkPos", orig_cyc, sse_cyc, ok); +} + +fn bench_computeOutcodes() void { + // Generate 150 vertices (typical mesh) spread across an AABB + const NVERTS = 150; + var verts: [NVERTS * 3]f32 = undefined; + var seed: u32 = 0xDEADBEEF; + for (0..NVERTS * 3) |j| { + seed = seed *% 1103515245 +% 12345; + // Range roughly -10..+10 + verts[j] = @as(f32, @floatFromInt(@as(i32, @bitCast(seed >> 16)) >> 16)) * 0.0003; + } + // AABB bounds: minX,minY,minZ,maxX,maxY,maxZ + var bounds = [6]f32{ -2.0, -2.0, -2.0, 2.0, 2.0, 2.0 }; + + // Original x87 version: extract the outcode loop from PerformSpatialCulling. + // The original does 6 FCOMP+FNSTSW+TEST sequences per vertex. + // We'll inline a scalar reference implementation for the original. + var out_orig: [NVERTS]u8 = undefined; + var out_sse: [NVERTS]u8 = undefined; + + // Scalar reference (matches original x87 logic) + for (0..NVERTS) |i| { + const vx = verts[i * 3]; + const vy = verts[i * 3 + 1]; + const vz = verts[i * 3 + 2]; + var code: u8 = 0; + if (vx < bounds[0]) code |= 0x20; + if (vx >= bounds[3]) code |= 0x10; + if (vy < bounds[1]) code |= 0x08; + if (vy >= bounds[4]) code |= 0x04; + if (vz < bounds[2]) code |= 0x02; + if (vz >= bounds[5]) code |= 0x01; + out_orig[i] = code; + } + + // SSE version + benchComputeOutcodes(a(&verts), a(&bounds), a(&out_sse), NVERTS); + + // Verify correctness + var ok = true; + for (0..NVERTS) |i| { + if (out_orig[i] != out_sse[i]) { + ok = false; + break; + } + } + + // Benchmark: scalar reference + const scalar_fn = struct { + fn run(v: *[NVERTS * 3]f32, b: *[6]f32, out: *[NVERTS]u8) void { + for (0..NVERTS) |i| { + const vx = v[i * 3]; + const vy = v[i * 3 + 1]; + const vz = v[i * 3 + 2]; + var code: u8 = 0; + if (vx < b[0]) code |= 0x20; + if (vx >= b[3]) code |= 0x10; + if (vy < b[1]) code |= 0x08; + if (vy >= b[4]) code |= 0x04; + if (vz < b[2]) code |= 0x02; + if (vz >= b[5]) code |= 0x01; + out[i] = code; + } + } + }.run; + + var orig_cyc: u64 = std.math.maxInt(u64); + var sse_cyc: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t = rdtsc(); + for (0..ITERS) |_| scalar_fn(&verts, &bounds, &out_orig); + t = rdtsc() - t; + if (t < orig_cyc) orig_cyc = t; + } + for (0..5) |_| { + var t = rdtsc(); + for (0..ITERS) |_| benchComputeOutcodes(a(&verts), a(&bounds), a(&out_sse), NVERTS); + t = rdtsc() - t; + if (t < sse_cyc) sse_cyc = t; + } + report("computeOutcodes(150v)", orig_cyc, sse_cyc, ok); +} diff --git a/src/clickthrough/clickthrough.zig b/src/clickthrough/clickthrough.zig index 0af3989..1f3b9fc 100644 --- a/src/clickthrough/clickthrough.zig +++ b/src/clickthrough/clickthrough.zig @@ -29,6 +29,7 @@ pub const module_name: [*:0]const u8 = "clickthrough"; const ADDR_WorldIntersectionTest: usize = 0x480DF0; const ADDR_CanTargetEntity: usize = 0x480610; +const ADDR_CheckObjectTypePermissions: usize = 0x480780; // ============================================================================= // HitTestResult layout @@ -54,20 +55,13 @@ const FLAG_CUSTOM_MASK: u32 = FLAG_LOOT_ONLY | FLAG_GO_ONLY | FLAG_NPC_ONLY; var g_mutex: ?*anyopaque = null; var g_is_hook_owner: bool = false; var log: logging.Logger = .{}; +var go_log_count: u32 = 0; -// CanTargetEntity: CanTargetEntity(void *obj, uint permissionFlags) -> undefined* -// Returns non-NULL to include object, NULL to exclude. -// __cdecl-ish but called with obj as first stack arg from CheckObjectTypePermissions. -// Assembly: PUSH permFlags; PUSH objPtr; CALL CanTargetEntity -// Actually looking at the call site it passes obj in register and flags on stack. -// Let me verify from the CheckObjectTypePermissions assembly. -// From decompile: puVar4 = CanTargetEntity(pvVar3, permissionFlags); -// pvVar3 is the resolved object pointer. permissionFlags is the raycast flags. -// The function signature from Ghidra: CanTargetEntity(void *param_1, uint param_2) -// Not thiscall/fastcall - it's a regular call with two stack args. - -const CanTargetFn = fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32; -var cte_hook: hook.Detour(CanTargetFn) = .{}; +// CheckObjectTypePermissions (0x480780) -- verified from assembly: +// __thiscall: ECX=context (saved to EDI, passed to CanTargetEntity) +// Stack: objectData [EBP+8], permFlags [EBP+C]. RET 0x8. +const CheckObjTypeFn = fn (u32, u32, u32) callconv(hook.cc.thiscall) u32; +var cotp_hook: hook.Detour(CheckObjTypeFn) = .{}; const WorldIntersectFn = fn (u32, u32, u32, u32, u32) callconv(hook.cc.thiscall) u32; var wit_hook: hook.Detour(WorldIntersectFn) = .{}; @@ -78,10 +72,10 @@ var wit_hook: hook.Detour(WorldIntersectFn) = .{}; // When custom flag bits are set, exclude objects that don't match the pass. // ============================================================================= -fn canTargetDetour(obj: u32, perm_flags: u32) callconv(.{ .x86_stdcall = .{} }) u32 { +fn checkObjTypeDetour(ctx: u32, obj_data: u32, perm_flags: u32) callconv(hook.cc.thiscall) u32 { // Strip custom bits before passing to original const clean_flags = perm_flags & ~FLAG_CUSTOM_MASK; - const original = cte_hook.callOriginal(.{ obj, clean_flags }); + const original = cotp_hook.callOriginal(.{ ctx, obj_data, clean_flags }); // If original says exclude, respect that if (original == 0) return 0; @@ -89,27 +83,37 @@ fn canTargetDetour(obj: u32, perm_flags: u32) callconv(.{ .x86_stdcall = .{} }) // No custom filtering active - pass through if ((perm_flags & FLAG_CUSTOM_MASK) == 0) return original; - // Custom pass filtering + // Resolve the object pointer from obj_data via ClntObjMgrObjectPtr. + // __fastcall(ECX=typeMask, EDX=debugStr, stack: guid_lo, guid_hi, debugCode) + // RET 0xC. See nampower ClntObjMgrObjectPtrT typedef. + const obj = hook.call( + fn (u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) u32, + 0x468460, // ClntObjMgrObjectPtr + .{ 1, 0, hook.readMem(u32, obj_data + 0x18), hook.readMem(u32, obj_data + 0x1C), 0 }, + ); + if (obj == 0) return original; + + const desc = wow.getDescriptor(obj); + if (!wow.isValidPtr(desc)) return original; + const type_mask = hook.readMem(u32, desc + 0x08); + if ((perm_flags & FLAG_LOOT_ONLY) != 0) { - // Only lootable corpses pass + if (type_mask != 0x09) return 0; // units only if (!wow.isLootable(obj)) return 0; return original; } if ((perm_flags & FLAG_GO_ONLY) != 0) { - // Only interactable GOs pass - const desc = wow.getDescriptor(obj); - if (!wow.isValidPtr(desc)) return 0; - const type_mask = hook.readMem(u32, desc + 0x08); - if (type_mask != 0x21) return 0; // not a GO - // Check interactability + if (type_mask != 0x21) return 0; // GOs only + const go_type = hook.readMem(u32, desc + offsets.DESC_GO_TYPE); + if (go_type == 9 or go_type == 7) return 0; // TEXT, CHAIR if (hook.call(fn (u32) callconv(hook.cc.fastcall) u8, offsets.FN_CALL_SPELL_CAST_HANDLER, .{obj}) == 0) return 0; return original; } if ((perm_flags & FLAG_NPC_ONLY) != 0) { - // Only units with NPC interaction flags pass + if (type_mask != 0x09 and type_mask != 0x19) return 0; // units/players only if (wow.getNpcFlags(obj) == 0) return 0; return original; } @@ -168,8 +172,8 @@ pub fn installHooks() void { g_is_hook_owner = result.is_owner; if (!g_is_hook_owner) return; - log = logging.Logger.open(module_name, .console); - _ = cte_hook.attach(ADDR_CanTargetEntity, &canTargetDetour); + log = logging.Logger.open(module_name, .both); + _ = cotp_hook.attach(ADDR_CheckObjectTypePermissions, &checkObjTypeDetour); _ = wit_hook.attach(ADDR_WorldIntersectionTest, &worldIntersectDetour); log.print("clickthrough: cascade raycast active\n"); } @@ -177,7 +181,7 @@ pub fn installHooks() void { pub fn removeHooks() void { if (g_is_hook_owner) { wit_hook.detach(); - cte_hook.detach(); + cotp_hook.detach(); log.close(); mod_mutex.release(&g_mutex); } diff --git a/src/performance/cull_sse.zig b/src/performance/cull_sse.zig new file mode 100644 index 0000000..fe847ba --- /dev/null +++ b/src/performance/cull_sse.zig @@ -0,0 +1,358 @@ +//! SSE-optimized spatial culling — compiled ReleaseFast even in Debug builds. +//! +//! Reimplements PerformSpatialCulling (0x6B8C60) and performCollisionDetection +//! (0x6B88E0). Both are leaf functions in the KD-tree traversal that compute +//! per-vertex 6-bit outcodes against an AABB, then iterate triangles. +//! +//! The hot path is the vertex outcode loop: 6 float comparisons per vertex +//! (~150 vertices typical). SSE compares all 3 axes in parallel. + +// ============================================================================= +// Benchmark-only: pure outcode computation, no game function dependencies. +// Called from src/bench/main.zig with synthetic vertex data. +// ============================================================================= + +/// Compute outcodes for `count` vertices at `verts_ptr` (stride 12 bytes = 3 floats) +/// against AABB at `bounds_ptr` (6 floats: minX, minY, minZ, maxX, maxY, maxZ). +/// Writes results to `out_ptr` (1 byte per vertex). +export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void { + if (count == 0) return; + const v_min: V4 = .{ readF32(bounds_ptr), readF32(bounds_ptr + 4), readF32(bounds_ptr + 8), 0 }; + const v_max: V4 = .{ readF32(bounds_ptr + 12), readF32(bounds_ptr + 16), readF32(bounds_ptr + 20), 0 }; + const below_w: @Vector(4, u32) = .{ 0x20, 0x08, 0x02, 0x00 }; + const above_w: @Vector(4, u32) = .{ 0x10, 0x04, 0x01, 0x00 }; + + const out: [*]u8 = @ptrFromInt(out_ptr); + var vp: [*]const f32 = @ptrFromInt(verts_ptr); + var i: u32 = 0; + while (i < count) : (i += 1) { + const v: V4 = @as(*align(1) const V4, @ptrCast(vp)).*; + const zero: @Vector(4, u32) = @splat(0); + const combined = @select(u32, v < v_min, below_w, zero) | @select(u32, v >= v_max, above_w, zero); + out[i] = @truncate(combined[0] | combined[1] | combined[2] | combined[3]); + vp += 3; + } +} + +// ============================================================================= +// Game hook exports +// ============================================================================= + +// External game functions (resolved at link time via absolute address) +// FindOrCreateHashEntry: thiscall(ECX=hashTable from global 0xCA03E4, stack: 5 args) RET 0x14 +const FindOrCreateHashEntry = @as(*const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x693D60)); +// Global state used by the game's rendering pipeline +const g_visible_count: *u32 = @ptrFromInt(0xCE26E0); // PTR_00ce26e0 +const g_visible_list: [*]u16 = @ptrFromInt(0xCE26E8); // DAT_00ce26e8 +const g_render_count: *u32 = @ptrFromInt(0xCE66FC); // PTR_00ce66fc +const g_render_list: [*]u16 = @ptrFromInt(0xCDE648); // DAT_00cde648 +const g_guard: *const u32 = @ptrFromInt(0xCA03E4); // PTR_00ca03e4 + +const V4 = @Vector(4, f32); + +fn readU32(addr: u32) u32 { + return @as(*align(1) const u32, @ptrFromInt(addr)).*; +} +fn readU16(addr: u32) u16 { + return @as(*align(1) const u16, @ptrFromInt(addr)).*; +} +fn readF32(addr: u32) f32 { + return @as(*align(1) const f32, @ptrFromInt(addr)).*; +} + +/// Compute outcodes for all vertices in the mesh using SSE. +/// +/// Per-vertex 6-bit outcode against AABB. Two SIMD compares (v < min, v >= max) +/// produce all 6 bits from movemask results. Processes xyz in parallel. +/// +/// Bit layout: 0x20=below_minX, 0x10=above_maxX, 0x08=below_minY, +/// 0x04=above_maxY, 0x02=below_minZ, 0x01=above_maxZ +fn computeAllOutcodes( + hash_entry: u32, + min_x: f32, + max_x: f32, + min_y: f32, + max_y: f32, + min_z: f32, + max_z: f32, + cull_flags: *[452]u8, +) u32 { + const vert_count: u32 = readU16(hash_entry + 6); + if (vert_count == 0) return 0; + + const v_min: V4 = .{ min_x, min_y, min_z, 0 }; + const v_max: V4 = .{ max_x, max_y, max_z, 0 }; + + // Bit weights for branchless outcode: below gives 0x20/0x08/0x02, above gives 0x10/0x04/0x01 + const below_w: @Vector(4, u32) = .{ 0x20, 0x08, 0x02, 0x00 }; + const above_w: @Vector(4, u32) = .{ 0x10, 0x04, 0x01, 0x00 }; + + var vert_ptr: [*]const f32 = @ptrFromInt(hash_entry + 8); + var i: u32 = 0; + while (i < vert_count) : (i += 1) { + // Single 16-byte unaligned load. 4th float is junk from next vertex + // but v_min[3]=0, v_max[3]=0, so comparisons on lane 3 produce + // below=false (0>=0), above=true (0>=0) -- weight is 0x00 so harmless. + const v: V4 = @as(*align(1) const V4, @ptrCast(vert_ptr)).*; + + // Bool vectors -> u32 vectors (0 or 0xFFFFFFFF), AND with weights, horizontal OR + const zero: @Vector(4, u32) = @splat(0); + const below_masked = @select(u32, v < v_min, below_w, zero); + const above_masked = @select(u32, v >= v_max, above_w, zero); + const combined = below_masked | above_masked; + + // Horizontal OR of 4 lanes -> single outcode byte + cull_flags[i] = @truncate(combined[0] | combined[1] | combined[2] | combined[3]); + + vert_ptr += 3; // stride 12 bytes = 3 floats + } + return vert_count; +} + +/// PerformSpatialCulling (0x6B8C60) +/// __thiscall(this, keyData, keySize) -> u32. RET 0x8. +/// +/// Finds mesh data via hash, computes vertex outcodes against AABB from this+0x10, +/// then iterates triangles: filters by visibility mask, trivial-rejects by outcode AND, +/// adds survivors to global visible/render lists. +export fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 { + if (g_guard.* == 0) return 0; + + const hash_table = g_guard.*; + const hash_entry = @call(.never_tail, FindOrCreateHashEntry, .{ + hash_table, key_data, key_size, + readU32(this + 4), readU32(this + 8), readU32(this + 0xC), + }); + if (hash_entry == 0) return 0; + + // Load AABB from *(this+0x10) -- 6 floats: minX, minY, minZ, maxX, maxY, maxZ + const bounds_ptr = readU32(this + 0x10); + const min_x = readF32(bounds_ptr); + const min_y = readF32(bounds_ptr + 4); + const min_z = readF32(bounds_ptr + 8); + const max_x = readF32(bounds_ptr + 12); + const max_y = readF32(bounds_ptr + 16); + const max_z = readF32(bounds_ptr + 20); + + // Phase 1: Compute per-vertex outcodes + var cull_flags: [452]u8 = undefined; + _ = computeAllOutcodes(hash_entry, min_x, max_x, min_y, max_y, min_z, max_z, &cull_flags); + + // Phase 2: Iterate triangles + const tri_count: u32 = @as(u32, readU16(hash_entry + 0x18A4)); + const filter_mask = readU16(this + 0x14); + const visited_base = readU32(this + 4); + + var ti: u32 = 0; + while (ti < tri_count) : (ti += 1) { + // Visibility mask filter + const vis_flags = readU16(hash_entry + 0x1FAE + ti * 2); + if ((vis_flags & filter_mask) != 0) continue; + + // Per-vertex visited filter + const tri_base_idx = readU16(hash_entry + 0x2206 + ti * 2); + const visited_byte = @as(*u8, @ptrFromInt(visited_base + @as(u32, tri_base_idx) * 2)); + if ((visited_byte.* & @as(u8, @truncate(filter_mask))) != 0) continue; + + // Check global visible list capacity + if (g_visible_count.* >= 0x2000) { + const flags_ptr = readU32(this); + if (flags_ptr != 0) { + const p: *u32 = @ptrFromInt(flags_ptr); + p.* |= 1; + } + break; + } + + // Add to visible list + g_visible_list[g_visible_count.*] = tri_base_idx; + g_visible_count.* += 1; + visited_byte.* |= 0x80; + + // Frustum test: AND of 3 vertex outcodes. If any bit shared, fully outside. + const idx0 = readU16(hash_entry + 0x18A6 + ti * 6); + const idx1 = readU16(hash_entry + 0x18A8 + ti * 6); + const idx2 = readU16(hash_entry + 0x18AA + ti * 6); + // Note: decompiler shows idx offsets as 0x18A6, +0xC54*2, +0x18AA + // which is 0x18A6 (idx0), 0x18A8 (idx1), 0x18AA (idx2) -- stride 6 = 3 u16 per tri + + if ((cull_flags[idx0] & cull_flags[idx1] & cull_flags[idx2] & 0x3F) == 0) { + g_render_list[g_render_count.*] = tri_base_idx; + g_render_count.* += 1; + } + } + + return 1; +} + +// ============================================================================= +// SSE vector helpers for Moller-Trumbore +// ============================================================================= + +inline fn loadVec3(addr: u32) V4 { + return .{ readF32(addr), readF32(addr + 4), readF32(addr + 8), 0 }; +} + +inline fn cross(a: V4, b: V4) V4 { + const Mask = @Vector(4, i32); + const a_yzx: V4 = @shuffle(f32, a, undefined, Mask{ 1, 2, 0, 3 }); + const a_zxy: V4 = @shuffle(f32, a, undefined, Mask{ 2, 0, 1, 3 }); + const b_yzx: V4 = @shuffle(f32, b, undefined, Mask{ 1, 2, 0, 3 }); + const b_zxy: V4 = @shuffle(f32, b, undefined, Mask{ 2, 0, 1, 3 }); + return a_yzx * b_zxy - a_zxy * b_yzx; +} + +inline fn dot3(a: V4, b: V4) f32 { + const p = a * b; + return p[0] + p[1] + p[2]; +} + +/// performCollisionDetection (0x6B88E0) +/// __thiscall(this, keyData, keySize) -> u32. RET 0x8. +/// +/// Fully inlined SSE rewrite. No external calls except FindOrCreateHashEntry. +/// Moller-Trumbore ray-triangle intersection is inlined with SSE cross/dot, +/// eliminating 4 SetVector3 calls and the ray_tri function call per triangle. +export fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 { + if (g_guard.* == 0) return 0; + + const hash_table = g_guard.*; + const hash_entry = @call(.never_tail, FindOrCreateHashEntry, .{ + hash_table, key_data, key_size, + readU32(this + 4), readU32(this + 8), readU32(this + 0xC), + }); + if (hash_entry == 0) return 0; + + // Load and sort AABB extents from this+0x18..0x2C + var ax0 = readF32(this + 0x18); + var ax1 = readF32(this + 0x24); + if (ax1 < ax0) { + const tmp = ax0; + ax0 = ax1; + ax1 = tmp; + } + var ay0 = readF32(this + 0x1C); + var ay1 = readF32(this + 0x28); + if (ay1 < ay0) { + const tmp = ay0; + ay0 = ay1; + ay1 = tmp; + } + var az0 = readF32(this + 0x20); + var az1 = readF32(this + 0x2C); + if (az1 < az0) { + const tmp = az0; + az0 = az1; + az1 = tmp; + } + + // Phase 1: Compute per-vertex outcodes + var cull_flags: [452]u8 = undefined; + _ = computeAllOutcodes(hash_entry, ax0, ax1, ay0, ay1, az0, az1, &cull_flags); + + // Phase 2: Iterate triangles with inline ray-tri test + const tri_count: u32 = @as(u32, readU16(hash_entry + 0x18A4)); + const collision_mask = readU16(this + 0x50); + const visited_base = readU32(this + 4); + const vert_pool = hash_entry + 8; + + // Ray: origin at this+0x00 (position), direction at this+0x0C (3 floats) + // Original uses param_1 = ESI which points to a 6-float struct: + // [0..2] = ray origin, [3..5] = ray direction + // The call site passes this+0x30 as the ray struct + const ray_origin = loadVec3(this + 0x30); + const ray_dir = loadVec3(this + 0x3C); + + // Epsilon for barycentric bounds: original uses +/- param_6 (0.002) + const eps: f32 = 0.002; + const neg_eps: f32 = -eps; + const one_plus_eps: f32 = 1.0 + eps; + + var ti: u32 = 0; + while (ti < tri_count) : (ti += 1) { + const vis_flags = readU16(hash_entry + 0x1FAE + ti * 2); + if ((vis_flags & collision_mask) != 0) continue; + + const tri_base_idx = readU16(hash_entry + 0x2206 + ti * 2); + const visited_addr = visited_base + @as(u32, tri_base_idx) * 2; + const visited_byte = @as(*u8, @ptrFromInt(visited_addr)); + if ((visited_byte.* & @as(u8, @truncate(collision_mask))) != 0) continue; + + // Add to visible list and mark visited + g_visible_list[g_visible_count.*] = tri_base_idx; + g_visible_count.* += 1; + visited_byte.* |= 0x80; + + // Frustum outcode test + const idx_base = hash_entry + 0x18A6 + ti * 6; + const vi0: u32 = readU16(idx_base); + const vi1: u32 = readU16(idx_base + 2); + const vi2: u32 = readU16(idx_base + 4); + + if ((cull_flags[vi0] & cull_flags[vi1] & cull_flags[vi2] & 0x3F) != 0) continue; + + // ===================================================================== + // Inline Moller-Trumbore ray-triangle intersection (SSE) + // Deferred divide: compare u_raw and v_raw against det-scaled bounds + // to avoid the 1/det divide on the reject path. + // ===================================================================== + + const v0 = loadVec3(vert_pool + vi0 * 12); + const v1 = loadVec3(vert_pool + vi1 * 12); + const v2 = loadVec3(vert_pool + vi2 * 12); + + const edge1 = v1 - v0; + const edge2 = v2 - v0; + const pvec = cross(ray_dir, edge2); + const det = dot3(edge1, pvec); + + if (det <= 1e-7 and det >= -1e-7) continue; + + const tvec = ray_origin - v0; + + // u_raw = dot(tvec, pvec) -- NOT multiplied by inv_det yet + const u_raw = dot3(tvec, pvec); + + // Compare u_raw against det-scaled epsilon bounds. + // If det > 0: u = u_raw/det, so u < -eps iff u_raw < -eps*det, u > 1+eps iff u_raw > (1+eps)*det + // If det < 0: division flips sign, so u < -eps iff u_raw > -eps*det (which is positive) + // Trick: multiply both sides by sign(det) to normalize. + // Or equivalently: if det>0 check u_raw in [det*neg_eps, det*one_plus_eps] + // if det<0 check u_raw in [det*one_plus_eps, det*neg_eps] + const det_neg_eps = det * neg_eps; + const det_one_plus = det * one_plus_eps; + if (det > 0) { + if (u_raw < det_neg_eps or u_raw > det_one_plus) continue; + } else { + if (u_raw > det_neg_eps or u_raw < det_one_plus) continue; + } + + const qvec = cross(tvec, edge1); + const v_raw = dot3(ray_dir, qvec); + + // Same sign-aware bounds check for v + if (det > 0) { + if (v_raw < det_neg_eps or (u_raw + v_raw) > det_one_plus) continue; + } else { + if (v_raw > det_neg_eps or (u_raw + v_raw) < det_one_plus) continue; + } + + // Only divide for confirmed hits + const t = dot3(edge2, qvec) / det; + + if (t >= 0.0 and t < readF32(this + 0x4C)) { + // Update closest hit + @as(*align(1) f32, @ptrFromInt(this + 0x4C)).* = t; + g_render_list[0] = tri_base_idx; + g_render_count.* = 1; + + // Write scaled distance, clamped to max + const result_ptr: *align(1) f32 = @ptrFromInt(readU32(this + 0x10)); + const scaled = t * readF32(this + 0x48); + const clamp = readF32(this + 0x14); + result_ptr.* = if (scaled <= clamp) scaled else clamp; + } + } + + return 1; +} diff --git a/src/performance/entity_sse.zig b/src/performance/entity_sse.zig new file mode 100644 index 0000000..0133033 --- /dev/null +++ b/src/performance/entity_sse.zig @@ -0,0 +1,212 @@ +//! SSE-optimized entity update functions -- compiled ReleaseFast. +//! +//! Reimplements UpdateEntityAndChunksPositions (0x6AFAD0) and +//! updateEntitiesInBounds (0x6C1F70). +//! +//! Main optimization: SSE dot product for view-depth computation, +//! tighter control flow, reduced function call overhead. + +const V4 = @Vector(4, f32); + +fn readU32(addr: u32) u32 { + return @as(*align(1) const u32, @ptrFromInt(addr)).*; +} +fn readI32(addr: u32) i32 { + return @as(*align(1) const i32, @ptrFromInt(addr)).*; +} +fn readF32(addr: u32) f32 { + return @as(*align(1) const f32, @ptrFromInt(addr)).*; +} +fn writeU32(addr: u32, val: u32) void { + @as(*align(1) u32, @ptrFromInt(addr)).* = val; +} +fn writeF32(addr: u32, val: f32) void { + @as(*align(1) f32, @ptrFromInt(addr)).* = val; +} + +// ========================================================================= +// Game function declarations (resolved at link time via absolute address) +// ========================================================================= + +// recycleVertexBuffer: fastcall(ECX=bufPtr, EDX=sizePtr), RET +const recycleVertexBuffer = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AE9A0)); +// check_instances_active: fastcall(ECX=instanceMgr) -> ptr, RET +const check_instances_active = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) u32, @ptrFromInt(0x6B2900)); +// store_all_instance_buffers: fastcall(ECX=instanceMgr), RET +const store_all_instance_buffers = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6B28E0)); +// return_object_to_pool: fastcall(ECX=instanceMgr), RET +const return_object_to_pool = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6B2030)); +// ReturnChunkBuffers: fastcall(ECX=bufPtr, EDX=sizePtr), RET +const ReturnChunkBuffers = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x68CD50)); +// IsPointInsideBounds: fastcall(ECX=point, EDX=bounds) -> u32, RET +const IsPointInsideBounds = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) u32, @ptrFromInt(0x699330)); +// AddObjectToSpatialList: fastcall(ECX=entityPtr, EDX=posPtr), RET +const AddObjectToSpatialList = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6818B0)); +// AddToSpatialGrid: fastcall(ECX=objPtr), RET +const AddToSpatialGrid = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6816F0)); +// CopyChunkBounds: thiscall(ECX=chunk, stack=outBounds), RET 0x4 +const CopyChunkBounds = @as(*const fn (u32, u32) callconv(.{ .x86_thiscall = .{} }) void, @ptrFromInt(0x68DF40)); +// AddToLayeredSpatialGrid: fastcall(ECX=chunk, EDX=idx, stack=posPtr), RET 0x4 +const AddToLayeredSpatialGrid = @as(*const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x681970)); + +// ComplexMemoryCleanupAndRelease: fastcall(ECX=memObj) +const ComplexMemoryCleanupAndRelease = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6A0510)); +// destroySecondaryGameObject: fastcall(ECX=entityPtr) +const destroySecondaryGameObject = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6A6A00)); + +// Game globals +const g_viewCoeffs: u32 = 0xC7BCB0; // 4 floats: a, b, c, d for dot product +const g_renderFlags: *const u8 = @ptrFromInt(0x867960 + 8); // actually at different offset +const g_deltaTime: *const u32 = @ptrFromInt(0xC62510); // PTR_00c62510 +const g_timerThreshold: *const f32 = @ptrFromInt(0x80A1E8); // _DAT_0080a1e8 + +// ========================================================================= +// UpdateEntityAndChunksPositions (0x6AFAD0) +// __fastcall(ECX=entityPtr), RET +// ========================================================================= +export fn updateEntityAndChunksPositions(ent: u32) callconv(.{ .x86_fastcall = .{} }) void { + // Dot product: depth = a*x + b*y + c*z + d - offset + const depth = readF32(g_viewCoeffs) * readF32(ent + 0x5C) + + readF32(g_viewCoeffs + 4) * readF32(ent + 0x60) + + readF32(g_viewCoeffs + 8) * readF32(ent + 0x64) + + readF32(g_viewCoeffs + 12) - readF32(ent + 0x68); + writeF32(ent + 0x78, depth); + + // Render distance flag + writeU32(ent + 0xB8, readU32(0x867964)); + if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 4) != 0 and readF32(0x867960) < depth) { + writeU32(ent + 0xB8, readU32(0x867968)); + } + + // Timer accumulation + const dt_bits = g_deltaTime.*; + const dt: f32 = @bitCast(dt_bits); + const timer = readF32(ent + 0xAC) + dt; + writeF32(ent + 0xAC, timer); + const threshold = g_timerThreshold.*; + + // Vertex buffer recycling + if (threshold < timer and readU32(ent + 0x14C) != 0) { + @call(.never_tail, recycleVertexBuffer, .{ ent + 0x14C, ent + 0x150 }); + } + + // Instance buffer management + const inst_mgr = readU32(ent + 0xC0); + if (inst_mgr != 0) { + if (1.0 < readF32(ent + 0xAC)) { + const active = @call(.never_tail, check_instances_active, .{inst_mgr}); + if (active != 0) { + @call(.never_tail, store_all_instance_buffers, .{inst_mgr}); + } + } + if (threshold < readF32(ent + 0xAC)) { + @call(.never_tail, return_object_to_pool, .{inst_mgr}); + writeU32(ent + 0xC0, 0); + } + } + + // Chunk timer loop (4 chunks at ent+0x118, stride 4) + inline for (0..4) |ci| { + const chunk = readU32(ent + 0x118 + ci * 4); + if (chunk != 0) { + const chunk_timer = readF32(chunk + 0x30) + dt; + writeF32(chunk + 0x30, chunk_timer); + if (readU32(chunk + 0x400) != 0 and threshold < chunk_timer) { + @call(.never_tail, ReturnChunkBuffers, .{ chunk + 0x400, chunk + 0x404 }); + } + } + } + + // Bounds check and spatial grid registration + const bx = readF32(ent + 0x44); + const by = readF32(ent + 0x48); + const bz = readF32(ent + 0x4C); + if (bx <= readF32(0xC7CB68) and by <= readF32(0xC7CB6C) and bz <= readF32(0xC7CB70)) { + const inside = @call(.never_tail, IsPointInsideBounds, .{ ent + 0x50, 0xC7CB5C }); + if (inside != 0) { + // Compute position from animation data + const anim_idx = readU32(0x86B580 + readU32(0xC7F294) * 4); + const anim_base = ent + 0x83C + @as(u32, @bitCast(anim_idx)) * 0xC; + var pos: [3]f32 = undefined; + pos[0] = readF32(anim_base) + readF32(ent + 0x6C); + pos[1] = readF32(anim_base + 4) + readF32(ent + 0x70); + pos[2] = readF32(anim_base + 8) + readF32(ent + 0x74); + + @call(.never_tail, AddObjectToSpatialList, .{ ent, @intFromPtr(&pos) }); + + // Walk sub-object linked list + var node = readU32(ent + 0xE4); + if ((node & 1) != 0 or node == 0) node = 0; + while ((node & 1) == 0 and node != 0) { + const obj = readU32(node + 4); + if ((@as(*const u8, @ptrFromInt(obj + 0xC)).* & 0x80) != 0 and (readU32(obj + 0x88) != 0 or readU32(obj + 0x174) != 0)) { + @call(.never_tail, AddToSpatialGrid, .{obj}); + } + node = readU32(readU32(ent + 0xDC) + node + 4); + } + } + } + + // Chunk bounds + spatial grid loop (4 chunks) + inline for (0..4) |ci| { + const chunk = readU32(ent + 0x118 + ci * 4); + if (chunk != 0) { + var bounds: [6]f32 = undefined; + @call(.never_tail, CopyChunkBounds, .{ chunk, @intFromPtr(&bounds) }); + + // Check if chunk bounds intersect the world region + if (bounds[0] <= readF32(0xC7CB68) and bounds[1] <= readF32(0xC7CB6C) and + bounds[2] <= readF32(0xC7CB70) and readF32(0xC7CB5C) <= bounds[3] and + readF32(0xC7CB60) <= bounds[4] and readF32(0xC7CB64) <= bounds[5]) + { + var center: [3]f32 = undefined; + center[0] = (bounds[3] + bounds[0]) * 0.5; + center[1] = (bounds[4] + bounds[1]) * 0.5; + center[2] = (bounds[5] + bounds[2]) * 0.5; + @call(.never_tail, AddToLayeredSpatialGrid, .{ chunk, @as(u32, ci), @intFromPtr(¢er) }); + } + } + } +} + +// ========================================================================= +// updateEntitiesInBounds (0x6C1F70) +// __thiscall(ECX=this, stack=param_1), RET 0x4 +// ========================================================================= +export fn updateEntitiesInBoundsSSE(this: u32, param_1: u32) callconv(.{ .x86_thiscall = .{} }) void { + var node = readU32(this + 0x274); + if ((node & 1) != 0 or node == 0) node = 0; + + while ((node & 1) == 0 and node != 0) { + const entity = readU32(node + 4); + const next = readU32(readU32(this + 0x26C) + 4 + node); + + // Prefetch next node's entity data while we process this one + if ((next & 1) == 0 and next != 0) { + const next_entity = readU32(next + 4); + @prefetch(@as([*]const u8, @ptrFromInt(next_entity + 0x44)), .{ .locality = 1 }); + @prefetch(@as([*]const u8, @ptrFromInt(next_entity + 0x8C)), .{ .locality = 1 }); + } + + // Bounds check: entity chunk coords vs global region + const cx = readI32(entity + 0x8C); + const cy = readI32(entity + 0x90); + + if (cx < readI32(0xC63278) or readI32(0xC63280) < cx or + cy < readI32(0xC63274) or readI32(0xC6327C) < cy) + { + // Out of bounds: destroy + const slot_idx = readU32(entity + 0xB4); // entityPtr[0x2d] = +0xB4 + writeU32(this + slot_idx * 4 + 0x278, 0); + @call(.never_tail, ComplexMemoryCleanupAndRelease, .{node}); + @call(.never_tail, destroySecondaryGameObject, .{entity}); + } else { + if (param_1 != 0) { + // Call through game address so the Detour hook fires (enables A/B timing) + const entPosHooked = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AFAD0)); + @call(.never_tail, entPosHooked, .{entity}); + } + } + node = next; + } +} diff --git a/src/performance/weirdperformance.zig b/src/performance/weirdperformance.zig index ec69a43..0dbfff6 100644 --- a/src/performance/weirdperformance.zig +++ b/src/performance/weirdperformance.zig @@ -64,14 +64,9 @@ fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) const TransformFn = fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; var transform_hook: hook.Detour(TransformFn) = .{}; -var teardown_active: bool = false; fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void { - if (teardown_active) { - transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - } else { - transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); - } + transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); } // ============================================================================= @@ -153,22 +148,14 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void { } // ============================================================================= -// Teardown hook (0x491180) — protect bone SSE during logout cleanup +// SSE JMP patches — binary patches at game function addresses // ============================================================================= -const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void; -var teardown_hook: hook.Detour(TeardownFn) = .{}; - -fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void { - teardown_active = true; - teardown_hook.callOriginal(.{}); - teardown_active = false; -} - -// ============================================================================= -// Silicon SSE JMP patches — binary patches at game function addresses -// ============================================================================= +// cull_sse.zig +extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +// silicon_sse.zig const sse = struct { extern fn si_normalizeVec3() callconv(.naked) void; extern fn si_mulMat3x4(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; @@ -236,6 +223,8 @@ fn installPatches() u32 { .{ .target = 0x686820, .replacement = @intFromPtr(&sse.si_translateBoundingVol), .name = "translateBoundingVol" }, .{ .target = 0x6ABC40, .replacement = @intFromPtr(&sse.si_processLinkedListCollision), .name = "processLinkedListCollision" }, .{ .target = 0x686000, .replacement = @intFromPtr(&sse.si_frustumCullBBox), .name = "frustumCullBBox" }, + .{ .target = 0x6B8C60, .replacement = @intFromPtr(&performSpatialCulling), .name = "PerformSpatialCulling" }, + .{ .target = 0x6B88E0, .replacement = @intFromPtr(&performCollisionDetectionSSE), .name = "performCollisionDetection" }, }; var count: u32 = 0; @@ -275,9 +264,6 @@ pub fn installHooks() void { // Per-frame cache reset if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1; - // Teardown guard - if (teardown_hook.attach(0x491180, &teardownDetour) == .ok) installed += 1; - // Silicon SSE binary patches _ = installPatches(); @@ -299,7 +285,6 @@ pub fn removeHooks() void { particle_hook.detach(); glyph_hook.detach(); world_update_hook.detach(); - teardown_hook.detach(); log.close(); mod_mutex.release(&g_mutex); } diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index 140d450..3cd6870 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -25,6 +25,10 @@ extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x8 extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; extern fn renderParticleSprites_REF(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; extern fn resetParticleCache() void; +extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; +extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern var stride_info: [8]u32; // exported from particle_sse.zig @@ -71,7 +75,7 @@ var last_frame_tsc: u64 = 0; // frame-to-frame TSC for total frame time pub var ab_use_custom: bool = false; // Gate for non-transform A/B hooks. Set false to isolate transform44 SSE testing. -const AB_OTHER_HOOKS = false; +const AB_OTHER_HOOKS = true; var diag_cmp_count: u32 = 0; export var original_trampoline: u32 = 0; // DEBUG: expose trampoline for REF passthrough test @@ -196,6 +200,12 @@ const ProfState = struct { matmul_cycles: u64 = 0, textline_calls: u64 = 0, // renderTextLine (0x5ce0c0) textline_cycles: u64 = 0, + viewfrust_calls: u64 = 0, // SetupViewFrustum (0x6bc1c0) -- parent of PerformSpatialCulling + viewfrust_cycles: u64 = 0, + cylfrust_calls: u64 = 0, // SetupCylinderFrustum (0x6bc370) -- child of RenderSphere + cylfrust_cycles: u64 = 0, + rendersph_calls: u64 = 0, // RenderSphere (0x6b92b0) -- parent of SetupCylinderFrustum + rendersph_cycles: u64 = 0, }; // ============================================================================= @@ -868,6 +878,9 @@ var triplane_hook: hook.Detour(Fn5) = .{}; // BuildTrianglePlanes: thiscall RET var partsetup_hook: hook.Detour(Fn3) = .{}; // SetupParticleRendering: thiscall RET 0x4 var matmul_hook: hook.Detour(Fn3) = .{}; // multiplyMatrix4x4: fastcall RET 0x4 var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10 +var viewfrust_hook: hook.Detour(Fn5) = .{}; // SetupViewFrustum: thiscall RET 0xC +var cylfrust_hook: hook.Detour(Fn5) = .{}; // SetupCylinderFrustum: thiscall RET 0xC +var rendersph_hook: hook.Detour(Fn8v) = .{}; // RenderSphere: thiscall RET 0x18 // --- Detour functions (timing-only pass-through) --- @@ -940,6 +953,12 @@ fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?* } fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); + if (AB_OTHER_HOOKS and ab_use_custom) { + updateEntityAndChunksPositions(a); + prof.entpos_cycles +|= rdtsc() - s; + prof.entpos_calls +|= 1; + return null; + } const ret = entpos_hook.callOriginal(.{ a, b }); prof.entpos_cycles +|= rdtsc() - s; prof.entpos_calls +|= 1; @@ -967,6 +986,7 @@ fn complexgeoDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u return ret; } fn entboundsDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + // Original only -- A/B testing entpos first const s = rdtsc(); const ret = entbounds_hook.callOriginal(.{ a, b, c }); prof.entbounds_cycles +|= rdtsc() - s; @@ -1027,6 +1047,12 @@ fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { } fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); + if (AB_OTHER_HOOKS and ab_use_custom) { + const ret = performSpatialCulling(a, c, d); + prof.cull_cycles +|= rdtsc() - s; + prof.cull_calls +|= 1; + return @ptrFromInt(ret); + } const ret = cull_hook.callOriginal(.{ a, b, c, d }); prof.cull_cycles +|= rdtsc() - s; prof.cull_calls +|= 1; @@ -1034,6 +1060,12 @@ fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyop } fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); + if (AB_OTHER_HOOKS and ab_use_custom) { + const ret = performCollisionDetectionSSE(a, c, d); + prof.colldet_cycles +|= rdtsc() - s; + prof.colldet_calls +|= 1; + return @ptrFromInt(ret); + } const ret = colldet_hook.callOriginal(.{ a, b, c, d }); prof.colldet_cycles +|= rdtsc() - s; prof.colldet_calls +|= 1; @@ -1174,6 +1206,27 @@ fn textlineDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook. prof.textline_calls +|= 1; return ret; } +fn viewfrustDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = viewfrust_hook.callOriginal(.{ a, b, c, d, e }); + prof.viewfrust_cycles +|= rdtsc() - s; + prof.viewfrust_calls +|= 1; + return ret; +} +fn cylfrustDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = cylfrust_hook.callOriginal(.{ a, b, c, d, e }); + prof.cylfrust_cycles +|= rdtsc() - s; + prof.cylfrust_calls +|= 1; + return ret; +} +fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = rendersph_hook.callOriginal(.{ a, b, c, d, e, f, g, h }); + prof.rendersph_cycles +|= rdtsc() - s; + prof.rendersph_calls +|= 1; + return ret; +} // ============================================================================= // Hook: blit_hub (0x5a4f60) @@ -1414,6 +1467,9 @@ fn dumpStats() void { .{ .name = "partsetup", .cycles = prof.partsetup_cycles, .calls = prof.partsetup_calls }, .{ .name = "matmul", .cycles = prof.matmul_cycles, .calls = prof.matmul_calls }, .{ .name = "textline", .cycles = prof.textline_cycles, .calls = prof.textline_calls }, + .{ .name = "viewfrust", .cycles = prof.viewfrust_cycles, .calls = prof.viewfrust_calls }, + .{ .name = "cylfrust", .cycles = prof.cylfrust_cycles, .calls = prof.cylfrust_calls }, + .{ .name = "rendersph", .cycles = prof.rendersph_cycles, .calls = prof.rendersph_calls }, }; for (hotspots) |h| { if (h.calls > 0) { @@ -1551,8 +1607,9 @@ pub fn installHooks() void { _ = linkedlist_hook.attach(0x710b90, &linkedlistDetour); _ = color_hook.attach(0x7b9b10, &colorDetour); _ = setvec_hook.attach(0x686640, &setvecDetour); - _ = cull_hook.attach(0x6b8c60, &cullDetour); - _ = colldet_hook.attach(0x6b88e0, &colldetDetour); + // cull + colldet graduated to weirdperformance JMP patches + // _ = cull_hook.attach(0x6b8c60, &cullDetour); + // _ = colldet_hook.attach(0x6b88e0, &colldetDetour); _ = activep_hook.attach(0x7b5a10, &activepDetour); _ = cbiter_hook.attach(0x404130, &cbiterDetour); _ = findguid_hook.attach(0x464890, &findguidDetour); @@ -1569,11 +1626,14 @@ pub fn installHooks() void { _ = partsetup_hook.attach(0x7b3d20, &partsetupDetour); _ = matmul_hook.attach(0x7bc6a0, &matmulDetour); _ = textline_hook.attach(0x5ce0c0, &textlineDetour); + _ = viewfrust_hook.attach(0x6bc1c0, &viewfrustDetour); + _ = cylfrust_hook.attach(0x6bc370, &cylfrustDetour); + _ = rendersph_hook.attach(0x6b92b0, &rendersphDetour); // Timer calibration now handled by performance module. // blit_hub installed in lateInit() to clobber UnitXP's hook - log.print("transform44: 39 profiling hooks installed (blit_hub deferred)\n"); + log.print("transform44: 42 profiling hooks installed (blit_hub deferred)\n"); if (bisect_stop_section != 0) { log.fmt(" BISECT MODE: REF stops after section {d}, then original trampoline\n", .{bisect_stop_section}); } @@ -1646,6 +1706,9 @@ pub fn removeHooks() void { partsetup_hook.detach(); matmul_hook.detach(); textline_hook.detach(); + viewfrust_hook.detach(); + cylfrust_hook.detach(); + rendersph_hook.detach(); blit_hub_hook.detach(); log.close(); mod_mutex.release(&g_mutex);