perf: SSE spatial culling, inlined ray-tri, clickthrough CC fix

PerformSpatialCulling (0x6B8C60): Zig rewrite with SSE outcode
computation. 1.4x speedup, JMP-patched in weirdperformance.

performCollisionDetection (0x6B88E0): fully inlined SSE Moller-Trumbore
ray-triangle intersection, eliminating 4 SetVector3 calls and the
ray_tri function pointer call per triangle. ~22 cyc/tri in bench.

Both graduated from transform44 A/B testing to production JMP patches.

entity_sse.zig: reimplementations of UpdateEntityAndChunksPositions and
updateEntitiesInBounds (A/B tested, 1.2x bench, not shipped - memory
bound with negligible real-world gain).

clickthrough: fixed CheckObjectTypePermissions hook from fastcall to
thiscall (ECX preservation), fixed ClntObjMgrObjectPtr from fastcall(4)
to fastcall(5) with correct arg count.

bench: added performCollisionDetection bench with synthetic mesh data
and patched FindOrCreateHashEntry stub.
This commit is contained in:
MarcelineVQ
2026-03-25 12:34:57 -07:00
parent 4274729a77
commit 41ca30cfd5
7 changed files with 1219 additions and 56 deletions
+42
View File
@@ -65,6 +65,22 @@ pub fn build(b: *std.Build) void {
.optimize = .ReleaseFast,
}),
});
const cull_sse_obj = b.addObject(.{
.name = "cull_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/cull_sse.zig"),
.target = target,
.optimize = .ReleaseFast,
}),
});
const entity_sse_obj = b.addObject(.{
.name = "entity_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/entity_sse.zig"),
.target = target,
.optimize = .ReleaseFast,
}),
});
const bone_sse_target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .windows,
@@ -172,6 +188,8 @@ pub fn build(b: *std.Build) void {
// Called for both the main weirdutils build and each variant.
const ModuleObjects = struct {
clip_sse: *std.Build.Step.Compile,
cull_sse: *std.Build.Step.Compile,
entity_sse: *std.Build.Step.Compile,
bone_sse: *std.Build.Step.Compile,
bone_sse_ref: *std.Build.Step.Compile,
math_sse: *std.Build.Step.Compile,
@@ -184,6 +202,8 @@ pub fn build(b: *std.Build) void {
@setEvalBranchQuota(10000);
if (comptime std.mem.eql(u8, module_name, "weirdperformance")) {
mod.addObject(self.clip_sse);
mod.addObject(self.cull_sse);
mod.addObject(self.entity_sse);
mod.addObject(self.bone_sse);
mod.addObject(self.bone_sse_ref);
mod.addObject(self.silicon_sse);
@@ -193,6 +213,8 @@ pub fn build(b: *std.Build) void {
}
if (comptime std.mem.eql(u8, module_name, "transform44")) {
mod.addObject(self.clip_sse);
mod.addObject(self.cull_sse);
mod.addObject(self.entity_sse);
mod.addObject(self.bone_sse);
mod.addObject(self.bone_sse_ref);
mod.addObject(self.particle_sse);
@@ -208,6 +230,8 @@ pub fn build(b: *std.Build) void {
};
const objs = ModuleObjects{
.clip_sse = clip_sse_obj,
.cull_sse = cull_sse_obj,
.entity_sse = entity_sse_obj,
.bone_sse = bone_sse_obj,
.bone_sse_ref = bone_sse_ref_obj,
.math_sse = math_sse_obj,
@@ -299,11 +323,29 @@ pub fn build(b: *std.Build) void {
.optimize = .ReleaseFast,
}),
});
const bench_cull_sse = b.addObject(.{
.name = "bench_cull_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/cull_sse.zig"),
.target = bench_target,
.optimize = .ReleaseFast,
}),
});
bench.root_module.addObject(bench_math_sse);
bench.root_module.addObject(bench_silicon_sse);
bench.root_module.addObject(bench_bone_sse);
bench.root_module.addObject(bench_bone_baseline);
bench.root_module.addObject(bench_particle_sse);
bench.root_module.addObject(bench_cull_sse);
const bench_entity_sse = b.addObject(.{
.name = "bench_entity_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/entity_sse.zig"),
.target = bench_target,
.optimize = .ReleaseFast,
}),
});
bench.root_module.addObject(bench_entity_sse);
bench.root_module.linkSystemLibrary("m", .{});
const install_bench = b.addInstallArtifact(bench, .{});
const bench_step = b.step("bench", "Build math_sse benchmark harness (x86 Linux)");
+501 -2
View File
@@ -58,6 +58,11 @@ extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void;
extern fn si_setParticleAlpha(u32, u32, u32) callconv(cc_fc) void; // fastcall(ECX=obj, EDX=unused, stack=alpha)
extern fn si_ftol() callconv(.naked) void;
// cull_sse.zig exports
extern fn benchComputeOutcodes(u32, u32, u32, u32) void;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
// =========================================================================
// Infrastructure
// =========================================================================
@@ -231,6 +236,14 @@ pub fn main() void {
print("{s:>30} {s:>10} {s:>10} {s}\n", .{ "function", "original", "sse", "status" });
print("{s}\n", .{"-" ** 72});
// Full performCollisionDetection (SSE vs original x87)
bench_collisionDetection();
// UpdateEntityAndChunksPositions (SSE vs original x87)
bench_entityUpdate();
if (false) { // disabled: not working on these right now
// 1: vecMulMat4 -- fastcall(ECX=result, EDX=vec, stack=mat) -> u32
bench_fc3r("vecMulMat4_ColMajor", originals.vecMulMat4_ColMajor, &vecMulMat4_ColMajor, tv3(), tm4(), 3);
@@ -1770,8 +1783,8 @@ pub fn main() void {
}
}
// calcColorValues_SSE -- thiscall(ctx_ECX, time, scale, outColor, outAlpha1, outAlpha2, outFloat)
bench_calcColorValues();
// calcColorValues_SSE -- disabled: no standalone SSE export yet
// bench_calcColorValues();
// si_frustumCullBBox -- fastcall(bbox_ECX, flags_EDX, radius_stack) -> u32
bench_frustumCullBBox();
@@ -1780,6 +1793,8 @@ pub fn main() void {
// Builds a fake linked list with 8 nodes to benchmark AABB overlap test.
bench_processLinkedListCollision();
} // end disabled block
print("\n", .{});
}
@@ -2174,6 +2189,29 @@ fn bench_tc2r(
// =========================================================================
const V4 = @Vector(4, f32);
const ShufMask = @Vector(4, i32);
inline fn benchLoadV3(addr: u32) V4 {
return .{
@as(*align(1) const f32, @ptrFromInt(addr)).*,
@as(*align(1) const f32, @ptrFromInt(addr + 4)).*,
@as(*align(1) const f32, @ptrFromInt(addr + 8)).*,
0,
};
}
inline fn benchCross(av: V4, bv: V4) V4 {
const a_yzx: V4 = @shuffle(f32, av, undefined, ShufMask{ 1, 2, 0, 3 });
const a_zxy: V4 = @shuffle(f32, av, undefined, ShufMask{ 2, 0, 1, 3 });
const b_yzx: V4 = @shuffle(f32, bv, undefined, ShufMask{ 1, 2, 0, 3 });
const b_zxy: V4 = @shuffle(f32, bv, undefined, ShufMask{ 2, 0, 1, 3 });
return a_yzx * b_zxy - a_zxy * b_yzx;
}
inline fn benchDot3(av: V4, bv: V4) f32 {
const p = av * bv;
return p[0] + p[1] + p[2];
}
inline fn inline_x87_dot(va: *const Vec3, vb: *const Vec3, out: *f32) void {
asm volatile (
@@ -2275,3 +2313,464 @@ inline fn inline_sse_horner(c: *const [4]f32, f: f32, out: *volatile f32) void {
r = r * f + c.*[3];
out.* = r;
}
fn sseRayTri(ray_ptr: u32, vert_pool: u32, idx_base: u32, t_out: *f32) bool {
const vi0: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base)).*;
const vi1: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base + 2)).*;
const vi2: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base + 4)).*;
const ray_o = benchLoadV3(ray_ptr);
const ray_d = benchLoadV3(ray_ptr + 12);
const v0 = benchLoadV3(vert_pool + vi0 * 12);
const v1 = benchLoadV3(vert_pool + vi1 * 12);
const v2 = benchLoadV3(vert_pool + vi2 * 12);
const edge1 = v1 - v0;
const edge2 = v2 - v0;
const pvec = benchCross(ray_d, edge2);
const det = benchDot3(edge1, pvec);
if (det <= 1e-7 and det >= -1e-7) return false;
const inv_det = 1.0 / det;
const tvec = ray_o - v0;
const u = benchDot3(tvec, pvec) * inv_det;
if (u < -0.002 or u > 1.002) return false;
const qvec = benchCross(tvec, edge1);
const v = benchDot3(ray_d, qvec) * inv_det;
if (v < -0.002 or (u + v) > 1.002) return false;
t_out.* = benchDot3(edge2, qvec) * inv_det;
return true;
}
fn bench_collisionDetection() void {
// Build synthetic mesh data matching game's hash entry layout.
// 32 vertices forming a grid, 20 triangles, ray aimed through the middle.
const NVERTS = 120;
const NTRIS = 40;
// Hash entry: total size must accommodate all fields up to 0x2206 + NTRIS*2
// Max offset: 0x2206 + 20*2 = 0x222E, round up
var hash_buf: [0x2300]u8 align(4) = [_]u8{0} ** 0x2300;
const he = @intFromPtr(&hash_buf);
// Vertex count at +6
@as(*align(1) u16, @ptrFromInt(he + 6)).* = NVERTS;
// Carefully crafted vertices to produce a mix of hits and misses with
// non-trivial barycentric coordinates. Ray fires from (0,0,-10) along +Z.
// Triangles 0-4: guaranteed hits at various u/v (straddling the ray axis)
// Triangles 5-9: near-misses (edge/corner cases for barycentric bounds)
// Triangles 10-14: clear misses (outside AABB or backfacing)
// Triangles 15-19: more hits with small/large det values (tests divide precision)
const verts = [NVERTS][3]f32{
// --- Group A: clear hits at various depths, u/v values ---
// Tri 0: large centered, hit u~0.33 v~0.33
.{ -2.0, -2.0, 1.0 }, .{ 4.0, -2.0, 1.0 }, .{ -2.0, 4.0, 1.0 },
// Tri 1: small on-axis, hit u~0.5 v~0.25
.{ -0.5, -0.5, 2.0 }, .{ 0.5, -0.5, 2.0 }, .{ 0.0, 0.5, 2.0 },
// Tri 2: very close to origin
.{ -1.0, -1.0, 0.1 }, .{ 1.0, -1.0, 0.1 }, .{ 0.0, 1.0, 0.1 },
// Tri 3: backface hit (wound CW)
.{ -2.0, 4.0, 4.0 }, .{ 4.0, -2.0, 4.0 }, .{ -2.0, -2.0, 4.0 },
// Tri 4: tiny triangle, tests large inv_det
.{ -0.05, -0.05, 1.5 }, .{ 0.05, -0.05, 1.5 }, .{ 0.0, 0.05, 1.5 },
// Tri 5: huge triangle, tests small inv_det
.{ -50.0, -50.0, 2.5 }, .{ 50.0, -50.0, 2.5 }, .{ 0.0, 50.0, 2.5 },
// Tri 6: hit at u~0, v~0 (near vertex 0)
.{ -0.001, -0.001, 3.0 }, .{ 5.0, -0.001, 3.0 }, .{ -0.001, 5.0, 3.0 },
// Tri 7: hit at u~1, v~0 (near vertex 1)
.{ -5.0, -0.001, 3.5 }, .{ 0.001, -0.001, 3.5 }, .{ -5.0, 5.0, 3.5 },
// Tri 8: hit at u~0, v~1 (near vertex 2)
.{ -5.0, -5.0, 4.0 }, .{ 5.0, -5.0, 4.0 }, .{ 0.001, 0.001, 4.0 },
// Tri 9: hit with u+v very close to 1.0 (edge between v1-v2)
.{ -0.01, -0.01, 4.5 }, .{ 2.0, -0.01, 4.5 }, .{ -0.01, 2.0, 4.5 },
// --- Group B: edge cases that should barely miss ---
// Tri 10: ray just outside triangle edge
.{ 0.5, -0.5, 5.0 }, .{ 2.0, -0.5, 5.0 }, .{ 0.5, 1.0, 5.0 },
// Tri 11: ray misses on v side
.{ -3.0, 0.5, 5.5 }, .{ -0.5, 0.5, 5.5 }, .{ -3.0, 2.0, 5.5 },
// Tri 12: triangle behind ray (negative t)
.{ -1.0, -1.0, -15.0 }, .{ 1.0, -1.0, -15.0 }, .{ 0.0, 1.0, -15.0 },
// Tri 13: triangle way off to the side
.{ 10.0, 10.0, 1.0 }, .{ 12.0, 10.0, 1.0 }, .{ 10.0, 12.0, 1.0 },
// Tri 14: triangle off to the other side
.{ -12.0, -12.0, 2.0 }, .{ -10.0, -12.0, 2.0 }, .{ -12.0, -10.0, 2.0 },
// --- Group C: degenerate/parallel ---
// Tri 15: zero-area (all same point)
.{ 1.0, 1.0, 6.0 }, .{ 1.0, 1.0, 6.0 }, .{ 1.0, 1.0, 6.0 },
// Tri 16: collinear vertices
.{ -1.0, 0.0, 7.0 }, .{ 0.0, 0.0, 7.0 }, .{ 1.0, 0.0, 7.0 },
// Tri 17: nearly parallel to ray (plane nearly parallel to Z axis)
.{ -1.0, -100.0, 0.5 }, .{ 1.0, -100.0, 0.5 }, .{ 0.0, 100.0, 0.501 },
// Tri 18: parallel to ray (exactly in XY plane at z=0, ray along Z)
.{ -1.0, -1.0, 0.0 }, .{ 1.0, -1.0, 0.0 }, .{ 0.0, 1.0, 0.0 },
// --- Group D: more hits at various depths for closest-t tracking ---
// Tri 19: closest possible hit
.{ -5.0, -5.0, 0.01 }, .{ 5.0, -5.0, 0.01 }, .{ 0.0, 5.0, 0.01 },
// Tri 20-24: hits at regular depth intervals
.{ -0.3, -0.3, 0.5 }, .{ 0.3, -0.3, 0.5 }, .{ 0.0, 0.3, 0.5 },
.{ -1.0, -1.0, 1.2 }, .{ 1.0, -1.0, 1.2 }, .{ 0.0, 1.0, 1.2 },
.{ -0.8, -0.8, 2.0 }, .{ 0.8, -0.8, 2.0 }, .{ 0.0, 0.8, 2.0 },
.{ -1.5, -1.5, 3.0 }, .{ 1.5, -1.5, 3.0 }, .{ 0.0, 1.5, 3.0 },
.{ -2.0, -2.0, 5.5 }, .{ 2.0, -2.0, 5.5 }, .{ 0.0, 2.0, 5.5 },
// --- Group E: outside AABB (outcode rejects, never reach ray-tri) ---
// Tri 25: all verts above AABB
.{ -1.0, 5.0, 1.0 }, .{ 1.0, 5.0, 1.0 }, .{ 0.0, 6.0, 1.0 },
// Tri 26: all verts below AABB
.{ -1.0, -6.0, 1.0 }, .{ 1.0, -6.0, 1.0 }, .{ 0.0, -5.0, 1.0 },
// Tri 27: all verts left of AABB
.{ -6.0, -1.0, 1.0 }, .{ -5.0, -1.0, 1.0 }, .{ -6.0, 1.0, 1.0 },
// Tri 28: all verts in front of AABB (z < min)
.{ -1.0, -1.0, -5.0 }, .{ 1.0, -1.0, -5.0 }, .{ 0.0, 1.0, -5.0 },
// Tri 29: all verts behind AABB (z > max)
.{ -1.0, -1.0, 5.0 }, .{ 1.0, -1.0, 5.0 }, .{ 0.0, 1.0, 5.0 },
// --- Group F: asymmetric/skewed hits testing det sign & magnitude ---
// Tri 30: very elongated, hit near tip
.{ 0.0, -0.01, 1.8 }, .{ 0.02, -0.01, 1.8 }, .{ 0.0, 10.0, 1.8 },
// Tri 31: very flat (nearly zero Y extent)
.{ -5.0, -0.001, 2.2 }, .{ 5.0, -0.001, 2.2 }, .{ 0.0, 0.001, 2.2 },
// Tri 32: large negative det
.{ -3.0, 3.0, 2.8 }, .{ 3.0, -3.0, 2.8 }, .{ -3.0, -3.0, 2.8 },
// Tri 33: det exactly at threshold boundary
.{ -0.0001, -0.0001, 6.5 }, .{ 0.0001, -0.0001, 6.5 }, .{ 0.0, 0.0001, 6.5 },
// --- Group G: stress closest-t with many competing hits ---
// Tri 34-39: hits at very close z-values to test precision
.{ -1.0, -1.0, 0.100 }, .{ 1.0, -1.0, 0.100 }, .{ 0.0, 1.0, 0.100 },
.{ -1.0, -1.0, 0.101 }, .{ 1.0, -1.0, 0.101 }, .{ 0.0, 1.0, 0.101 },
.{ -1.0, -1.0, 0.099 }, .{ 1.0, -1.0, 0.099 }, .{ 0.0, 1.0, 0.099 },
.{ -1.0, -1.0, 0.102 }, .{ 1.0, -1.0, 0.102 }, .{ 0.0, 1.0, 0.102 },
.{ -1.0, -1.0, 0.098 }, .{ 1.0, -1.0, 0.098 }, .{ 0.0, 1.0, 0.098 },
.{ -1.0, -1.0, 0.103 }, .{ 1.0, -1.0, 0.103 }, .{ 0.0, 1.0, 0.103 },
};
// Write vertices to hash entry at +8
for (0..NVERTS) |vi| {
const off = he + 8 + vi * 12;
@as(*align(1) f32, @ptrFromInt(off)).* = verts[vi][0];
@as(*align(1) f32, @ptrFromInt(off + 4)).* = verts[vi][1];
@as(*align(1) f32, @ptrFromInt(off + 8)).* = verts[vi][2];
}
// Triangle count at +0x18A4
@as(*align(1) u16, @ptrFromInt(he + 0x18A4)).* = NTRIS;
// Each triangle uses 3 consecutive vertices: tri N -> verts N*3, N*3+1, N*3+2
{
var ti: u32 = 0;
while (ti < NTRIS) : (ti += 1) {
const base: u16 = @intCast(ti * 3);
@as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6)).* = base;
@as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6 + 2)).* = base + 1;
@as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6 + 4)).* = base + 2;
@as(*align(1) u16, @ptrFromInt(he + 0x1FAE + ti * 2)).* = 0;
@as(*align(1) u16, @ptrFromInt(he + 0x2206 + ti * 2)).* = @intCast(ti);
}
}
// Build "this" struct (needs ~0x54 bytes)
var this_buf: [0x60]u8 align(4) = [_]u8{0} ** 0x60;
const th = @intFromPtr(&this_buf);
// Visited array: needs at least NTRIS*2 bytes
var visited: [64]u8 = [_]u8{0} ** 64;
// Result float
var result_val: f32 = 0.0;
// this+0x04 = visited array base
@as(*align(1) u32, @ptrFromInt(th + 0x04)).* = @intFromPtr(&visited);
// this+0x08, +0x0C = hash params (must match what FindOrCreateHashEntry expects, but
// we'll call our function directly bypassing the hash lookup, so these don't matter)
// this+0x10 = pointer to result float
@as(*align(1) u32, @ptrFromInt(th + 0x10)).* = @intFromPtr(&result_val);
// this+0x14 = clamp value
@as(*align(1) f32, @ptrFromInt(th + 0x14)).* = 100.0;
// this+0x18..0x2C = AABB extents (will be sorted by function)
@as(*align(1) f32, @ptrFromInt(th + 0x18)).* = -3.0; // ax0
@as(*align(1) f32, @ptrFromInt(th + 0x1C)).* = -3.0; // ay0
@as(*align(1) f32, @ptrFromInt(th + 0x20)).* = -3.0; // az0
@as(*align(1) f32, @ptrFromInt(th + 0x24)).* = 3.0; // ax1
@as(*align(1) f32, @ptrFromInt(th + 0x28)).* = 3.0; // ay1
@as(*align(1) f32, @ptrFromInt(th + 0x2C)).* = 3.0; // az1
// this+0x30..0x3B = ray origin
@as(*align(1) f32, @ptrFromInt(th + 0x30)).* = 0.0;
@as(*align(1) f32, @ptrFromInt(th + 0x34)).* = 0.0;
@as(*align(1) f32, @ptrFromInt(th + 0x38)).* = -10.0;
// this+0x3C..0x47 = ray direction
@as(*align(1) f32, @ptrFromInt(th + 0x3C)).* = 0.0;
@as(*align(1) f32, @ptrFromInt(th + 0x40)).* = 0.0;
@as(*align(1) f32, @ptrFromInt(th + 0x44)).* = 1.0;
// this+0x48 = scale factor
@as(*align(1) f32, @ptrFromInt(th + 0x48)).* = 1.0;
// this+0x4C = closest-t (large initial value)
@as(*align(1) f32, @ptrFromInt(th + 0x4C)).* = 999999.0;
// this+0x50 = collision mask
@as(*align(1) u16, @ptrFromInt(th + 0x50)).* = 0;
// Map globals needed by both original and SSE functions
_ = mapZeroed(0xCA0000, 0x1000); // g_guard at 0xCA03E4
_ = mapZeroed(0xCDE000, 0x1000); // g_render_list at 0xCDE648
_ = mapZeroed(0xCE2000, 0x1000); // g_visible_count/list at 0xCE26E0/E8
_ = mapZeroed(0xCE6000, 0x1000); // g_render_count at 0xCE66FC
// Patch FindOrCreateHashEntry (0x693D60) to return our hash_buf:
// MOV EAX, <hash_buf_addr> ; B8 xx xx xx xx
// RET 0x14 ; C2 14 00
const hash_stub = @as([*]u8, @ptrFromInt(0x693D60));
hash_stub[0] = 0xB8;
@as(*align(1) u32, @ptrFromInt(0x693D61)).* = he;
hash_stub[5] = 0xC2;
hash_stub[6] = 0x14;
hash_stub[7] = 0x00;
// Set g_guard to non-zero (both functions check this)
@as(*align(1) u32, @ptrFromInt(0xCA03E4)).* = 1;
// Original function at 0x6B88E0 and our SSE version
const orig_fn = @as(*const fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x6B88E0));
// Reset state helper
const resetState = struct {
fn f(t: u32, v: *[64]u8) void {
@as(*align(1) f32, @ptrFromInt(t + 0x4C)).* = 999999.0;
@memset(v, 0);
@as(*u32, @ptrFromInt(0xCE26E0)).* = 0; // visible count
@as(*u32, @ptrFromInt(0xCE66FC)).* = 0; // render count
}
}.f;
// Get truth values from original x87 function
resetState(th, &visited);
_ = @call(.never_tail, orig_fn, .{ th, 0, 0 });
const orig_closest = @as(*align(1) f32, @ptrFromInt(th + 0x4C)).*;
const orig_result = result_val;
const orig_render_count = @as(*u32, @ptrFromInt(0xCE66FC)).*;
const orig_visible_count = @as(*u32, @ptrFromInt(0xCE26E0)).*;
// Run SSE version
resetState(th, &visited);
_ = @call(.never_tail, performCollisionDetectionSSE, .{ th, 0, 0 });
const sse_closest = @as(*align(1) f32, @ptrFromInt(th + 0x4C)).*;
const sse_result = result_val;
const sse_render_count = @as(*u32, @ptrFromInt(0xCE66FC)).*;
const sse_visible_count = @as(*u32, @ptrFromInt(0xCE26E0)).*;
const t_match = @abs(orig_closest - sse_closest) < 0.01 or (orig_closest > 99999.0 and sse_closest > 99999.0);
const r_match = @abs(orig_result - sse_result) < 0.01;
const ok = t_match and r_match and orig_render_count == sse_render_count and orig_visible_count == sse_visible_count;
if (!ok) {
print(" MISMATCH detail:\n", .{});
print(" closest-t: orig={d:.6} sse={d:.6}\n", .{ orig_closest, sse_closest });
print(" result: orig={d:.6} sse={d:.6}\n", .{ orig_result, sse_result });
} else {
print(" closest-t={d:.4} result={d:.4} hits={d} visible={d}\n", .{
sse_closest, sse_result, sse_render_count, sse_visible_count,
});
}
// Benchmark: our SSE performCollisionDetectionSSE
var best: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t0 = rdtsc();
for (0..ITERS) |_| {
resetState(th, &visited);
_ = @call(.never_tail, performCollisionDetectionSSE, .{ th, 0, 0 });
}
t0 = rdtsc() - t0;
if (t0 < best) best = t0;
}
const per_call = best / ITERS;
const per_tri = if (NTRIS > 0) per_call / NTRIS else 0;
const status: [*:0]const u8 = if (ok) "OK" else "MISMATCH";
print("{s:>30}: {d} cyc/call {d} cyc/tri ({d} tris) {s}\n", .{
"performCollisionDet", per_call, per_tri, NTRIS, status,
});
}
fn bench_entityUpdate() void {
// Map .bss pages for globals the entity update reads/writes
_ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510
_ = mapZeroed(0xC7B000, 0x2000); // view coeffs at 0xC7BCB0, bounds at 0xC7CB5C-C7CB70
_ = mapZeroed(0x866000, 0x4000); // render flags at 0x867960, ptrs at 0x867964/68
_ = mapZeroed(0x80A000, 0x1000); // timer threshold at 0x80A1E8
_ = mapZeroed(0x86B000, 0x1000); // anim table at 0x86B580
_ = mapZeroed(0xC7F000, 0x1000); // anim index at 0xC7F294
// Set up view coefficients (a,b,c,d) at 0xC7BCB0
@as(*align(1) f32, @ptrFromInt(0xC7BCB0)).* = 0.5; // coeff for ent+0x5C
@as(*align(1) f32, @ptrFromInt(0xC7BCB4)).* = 0.3; // coeff for ent+0x60
@as(*align(1) f32, @ptrFromInt(0xC7BCB8)).* = 0.7; // coeff for ent+0x64
@as(*align(1) f32, @ptrFromInt(0xC7BCBC)).* = 1.0; // constant term
// Delta time
@as(*align(1) f32, @ptrFromInt(0xC62510)).* = 0.016; // ~60fps
// Timer threshold
@as(*align(1) f32, @ptrFromInt(0x80A1E8)).* = 999.0; // high so recycling never triggers
// World bounds (set large so bounds check always passes)
@as(*align(1) f32, @ptrFromInt(0xC7CB68)).* = 999.0;
@as(*align(1) f32, @ptrFromInt(0xC7CB6C)).* = 999.0;
@as(*align(1) f32, @ptrFromInt(0xC7CB70)).* = 999.0;
// Disable spatial grid registration by making bounds check fail:
// Set the lower bounds high so IsPointInsideBounds returns false
@as(*align(1) f32, @ptrFromInt(0xC7CB5C)).* = 99999.0;
@as(*align(1) f32, @ptrFromInt(0xC7CB60)).* = 99999.0;
@as(*align(1) f32, @ptrFromInt(0xC7CB64)).* = 99999.0;
// Build a synthetic entity struct (~0x900 bytes to cover all accessed fields)
var ent_buf: [0x900]u8 align(16) = [_]u8{0} ** 0x900;
const ent = @intFromPtr(&ent_buf);
// Entity position fields for dot product
@as(*align(1) f32, @ptrFromInt(ent + 0x5C)).* = 10.0;
@as(*align(1) f32, @ptrFromInt(ent + 0x60)).* = 20.0;
@as(*align(1) f32, @ptrFromInt(ent + 0x64)).* = 30.0;
@as(*align(1) f32, @ptrFromInt(ent + 0x68)).* = 5.0; // depth offset
// Timer at ent+0xAC (start at 0)
@as(*align(1) f32, @ptrFromInt(ent + 0xAC)).* = 0.0;
// No vertex buffers (ent+0x14C = 0), no instances (ent+0xC0 = 0), no chunks
// Bounds fields that won't trigger spatial grid
@as(*align(1) f32, @ptrFromInt(ent + 0x44)).* = 999.0; // will fail < check
const orig_fn = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AFAD0));
// Reset helper
const resetEnt = struct {
fn f(e: u32) void {
@as(*align(1) f32, @ptrFromInt(e + 0xAC)).* = 0.0; // reset timer
@as(*align(1) f32, @ptrFromInt(e + 0x78)).* = 0.0; // reset depth
}
}.f;
// Correctness: both should compute the same depth
resetEnt(ent);
@call(.never_tail, orig_fn, .{ent});
const orig_depth = @as(*align(1) f32, @ptrFromInt(ent + 0x78)).*;
resetEnt(ent);
@call(.never_tail, updateEntityAndChunksPositions, .{ent});
const sse_depth = @as(*align(1) f32, @ptrFromInt(ent + 0x78)).*;
const ok = @abs(orig_depth - sse_depth) < 0.01;
if (!ok) {
print(" MISMATCH: orig_depth={d:.4} sse_depth={d:.4}\n", .{ orig_depth, sse_depth });
}
// Benchmark original
var orig_cyc: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t0 = rdtsc();
for (0..ITERS) |_| {
resetEnt(ent);
@call(.never_tail, orig_fn, .{ent});
}
t0 = rdtsc() - t0;
if (t0 < orig_cyc) orig_cyc = t0;
}
// Benchmark SSE
var sse_cyc: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t0 = rdtsc();
for (0..ITERS) |_| {
resetEnt(ent);
@call(.never_tail, updateEntityAndChunksPositions, .{ent});
}
t0 = rdtsc() - t0;
if (t0 < sse_cyc) sse_cyc = t0;
}
report("UpdateEntityChunkPos", orig_cyc, sse_cyc, ok);
}
fn bench_computeOutcodes() void {
// Generate 150 vertices (typical mesh) spread across an AABB
const NVERTS = 150;
var verts: [NVERTS * 3]f32 = undefined;
var seed: u32 = 0xDEADBEEF;
for (0..NVERTS * 3) |j| {
seed = seed *% 1103515245 +% 12345;
// Range roughly -10..+10
verts[j] = @as(f32, @floatFromInt(@as(i32, @bitCast(seed >> 16)) >> 16)) * 0.0003;
}
// AABB bounds: minX,minY,minZ,maxX,maxY,maxZ
var bounds = [6]f32{ -2.0, -2.0, -2.0, 2.0, 2.0, 2.0 };
// Original x87 version: extract the outcode loop from PerformSpatialCulling.
// The original does 6 FCOMP+FNSTSW+TEST sequences per vertex.
// We'll inline a scalar reference implementation for the original.
var out_orig: [NVERTS]u8 = undefined;
var out_sse: [NVERTS]u8 = undefined;
// Scalar reference (matches original x87 logic)
for (0..NVERTS) |i| {
const vx = verts[i * 3];
const vy = verts[i * 3 + 1];
const vz = verts[i * 3 + 2];
var code: u8 = 0;
if (vx < bounds[0]) code |= 0x20;
if (vx >= bounds[3]) code |= 0x10;
if (vy < bounds[1]) code |= 0x08;
if (vy >= bounds[4]) code |= 0x04;
if (vz < bounds[2]) code |= 0x02;
if (vz >= bounds[5]) code |= 0x01;
out_orig[i] = code;
}
// SSE version
benchComputeOutcodes(a(&verts), a(&bounds), a(&out_sse), NVERTS);
// Verify correctness
var ok = true;
for (0..NVERTS) |i| {
if (out_orig[i] != out_sse[i]) {
ok = false;
break;
}
}
// Benchmark: scalar reference
const scalar_fn = struct {
fn run(v: *[NVERTS * 3]f32, b: *[6]f32, out: *[NVERTS]u8) void {
for (0..NVERTS) |i| {
const vx = v[i * 3];
const vy = v[i * 3 + 1];
const vz = v[i * 3 + 2];
var code: u8 = 0;
if (vx < b[0]) code |= 0x20;
if (vx >= b[3]) code |= 0x10;
if (vy < b[1]) code |= 0x08;
if (vy >= b[4]) code |= 0x04;
if (vz < b[2]) code |= 0x02;
if (vz >= b[5]) code |= 0x01;
out[i] = code;
}
}
}.run;
var orig_cyc: u64 = std.math.maxInt(u64);
var sse_cyc: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t = rdtsc();
for (0..ITERS) |_| scalar_fn(&verts, &bounds, &out_orig);
t = rdtsc() - t;
if (t < orig_cyc) orig_cyc = t;
}
for (0..5) |_| {
var t = rdtsc();
for (0..ITERS) |_| benchComputeOutcodes(a(&verts), a(&bounds), a(&out_sse), NVERTS);
t = rdtsc() - t;
if (t < sse_cyc) sse_cyc = t;
}
report("computeOutcodes(150v)", orig_cyc, sse_cyc, ok);
}
+31 -27
View File
@@ -29,6 +29,7 @@ pub const module_name: [*:0]const u8 = "clickthrough";
const ADDR_WorldIntersectionTest: usize = 0x480DF0;
const ADDR_CanTargetEntity: usize = 0x480610;
const ADDR_CheckObjectTypePermissions: usize = 0x480780;
// =============================================================================
// HitTestResult layout
@@ -54,20 +55,13 @@ const FLAG_CUSTOM_MASK: u32 = FLAG_LOOT_ONLY | FLAG_GO_ONLY | FLAG_NPC_ONLY;
var g_mutex: ?*anyopaque = null;
var g_is_hook_owner: bool = false;
var log: logging.Logger = .{};
var go_log_count: u32 = 0;
// CanTargetEntity: CanTargetEntity(void *obj, uint permissionFlags) -> undefined*
// Returns non-NULL to include object, NULL to exclude.
// __cdecl-ish but called with obj as first stack arg from CheckObjectTypePermissions.
// Assembly: PUSH permFlags; PUSH objPtr; CALL CanTargetEntity
// Actually looking at the call site it passes obj in register and flags on stack.
// Let me verify from the CheckObjectTypePermissions assembly.
// From decompile: puVar4 = CanTargetEntity(pvVar3, permissionFlags);
// pvVar3 is the resolved object pointer. permissionFlags is the raycast flags.
// The function signature from Ghidra: CanTargetEntity(void *param_1, uint param_2)
// Not thiscall/fastcall - it's a regular call with two stack args.
const CanTargetFn = fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32;
var cte_hook: hook.Detour(CanTargetFn) = .{};
// CheckObjectTypePermissions (0x480780) -- verified from assembly:
// __thiscall: ECX=context (saved to EDI, passed to CanTargetEntity)
// Stack: objectData [EBP+8], permFlags [EBP+C]. RET 0x8.
const CheckObjTypeFn = fn (u32, u32, u32) callconv(hook.cc.thiscall) u32;
var cotp_hook: hook.Detour(CheckObjTypeFn) = .{};
const WorldIntersectFn = fn (u32, u32, u32, u32, u32) callconv(hook.cc.thiscall) u32;
var wit_hook: hook.Detour(WorldIntersectFn) = .{};
@@ -78,10 +72,10 @@ var wit_hook: hook.Detour(WorldIntersectFn) = .{};
// When custom flag bits are set, exclude objects that don't match the pass.
// =============================================================================
fn canTargetDetour(obj: u32, perm_flags: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
fn checkObjTypeDetour(ctx: u32, obj_data: u32, perm_flags: u32) callconv(hook.cc.thiscall) u32 {
// Strip custom bits before passing to original
const clean_flags = perm_flags & ~FLAG_CUSTOM_MASK;
const original = cte_hook.callOriginal(.{ obj, clean_flags });
const original = cotp_hook.callOriginal(.{ ctx, obj_data, clean_flags });
// If original says exclude, respect that
if (original == 0) return 0;
@@ -89,27 +83,37 @@ fn canTargetDetour(obj: u32, perm_flags: u32) callconv(.{ .x86_stdcall = .{} })
// No custom filtering active - pass through
if ((perm_flags & FLAG_CUSTOM_MASK) == 0) return original;
// Custom pass filtering
// Resolve the object pointer from obj_data via ClntObjMgrObjectPtr.
// __fastcall(ECX=typeMask, EDX=debugStr, stack: guid_lo, guid_hi, debugCode)
// RET 0xC. See nampower ClntObjMgrObjectPtrT typedef.
const obj = hook.call(
fn (u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) u32,
0x468460, // ClntObjMgrObjectPtr
.{ 1, 0, hook.readMem(u32, obj_data + 0x18), hook.readMem(u32, obj_data + 0x1C), 0 },
);
if (obj == 0) return original;
const desc = wow.getDescriptor(obj);
if (!wow.isValidPtr(desc)) return original;
const type_mask = hook.readMem(u32, desc + 0x08);
if ((perm_flags & FLAG_LOOT_ONLY) != 0) {
// Only lootable corpses pass
if (type_mask != 0x09) return 0; // units only
if (!wow.isLootable(obj)) return 0;
return original;
}
if ((perm_flags & FLAG_GO_ONLY) != 0) {
// Only interactable GOs pass
const desc = wow.getDescriptor(obj);
if (!wow.isValidPtr(desc)) return 0;
const type_mask = hook.readMem(u32, desc + 0x08);
if (type_mask != 0x21) return 0; // not a GO
// Check interactability
if (type_mask != 0x21) return 0; // GOs only
const go_type = hook.readMem(u32, desc + offsets.DESC_GO_TYPE);
if (go_type == 9 or go_type == 7) return 0; // TEXT, CHAIR
if (hook.call(fn (u32) callconv(hook.cc.fastcall) u8, offsets.FN_CALL_SPELL_CAST_HANDLER, .{obj}) == 0)
return 0;
return original;
}
if ((perm_flags & FLAG_NPC_ONLY) != 0) {
// Only units with NPC interaction flags pass
if (type_mask != 0x09 and type_mask != 0x19) return 0; // units/players only
if (wow.getNpcFlags(obj) == 0) return 0;
return original;
}
@@ -168,8 +172,8 @@ pub fn installHooks() void {
g_is_hook_owner = result.is_owner;
if (!g_is_hook_owner) return;
log = logging.Logger.open(module_name, .console);
_ = cte_hook.attach(ADDR_CanTargetEntity, &canTargetDetour);
log = logging.Logger.open(module_name, .both);
_ = cotp_hook.attach(ADDR_CheckObjectTypePermissions, &checkObjTypeDetour);
_ = wit_hook.attach(ADDR_WorldIntersectionTest, &worldIntersectDetour);
log.print("clickthrough: cascade raycast active\n");
}
@@ -177,7 +181,7 @@ pub fn installHooks() void {
pub fn removeHooks() void {
if (g_is_hook_owner) {
wit_hook.detach();
cte_hook.detach();
cotp_hook.detach();
log.close();
mod_mutex.release(&g_mutex);
}
+358
View File
@@ -0,0 +1,358 @@
//! SSE-optimized spatial culling — compiled ReleaseFast even in Debug builds.
//!
//! Reimplements PerformSpatialCulling (0x6B8C60) and performCollisionDetection
//! (0x6B88E0). Both are leaf functions in the KD-tree traversal that compute
//! per-vertex 6-bit outcodes against an AABB, then iterate triangles.
//!
//! The hot path is the vertex outcode loop: 6 float comparisons per vertex
//! (~150 vertices typical). SSE compares all 3 axes in parallel.
// =============================================================================
// Benchmark-only: pure outcode computation, no game function dependencies.
// Called from src/bench/main.zig with synthetic vertex data.
// =============================================================================
/// Compute outcodes for `count` vertices at `verts_ptr` (stride 12 bytes = 3 floats)
/// against AABB at `bounds_ptr` (6 floats: minX, minY, minZ, maxX, maxY, maxZ).
/// Writes results to `out_ptr` (1 byte per vertex).
export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void {
if (count == 0) return;
const v_min: V4 = .{ readF32(bounds_ptr), readF32(bounds_ptr + 4), readF32(bounds_ptr + 8), 0 };
const v_max: V4 = .{ readF32(bounds_ptr + 12), readF32(bounds_ptr + 16), readF32(bounds_ptr + 20), 0 };
const below_w: @Vector(4, u32) = .{ 0x20, 0x08, 0x02, 0x00 };
const above_w: @Vector(4, u32) = .{ 0x10, 0x04, 0x01, 0x00 };
const out: [*]u8 = @ptrFromInt(out_ptr);
var vp: [*]const f32 = @ptrFromInt(verts_ptr);
var i: u32 = 0;
while (i < count) : (i += 1) {
const v: V4 = @as(*align(1) const V4, @ptrCast(vp)).*;
const zero: @Vector(4, u32) = @splat(0);
const combined = @select(u32, v < v_min, below_w, zero) | @select(u32, v >= v_max, above_w, zero);
out[i] = @truncate(combined[0] | combined[1] | combined[2] | combined[3]);
vp += 3;
}
}
// =============================================================================
// Game hook exports
// =============================================================================
// External game functions (resolved at link time via absolute address)
// FindOrCreateHashEntry: thiscall(ECX=hashTable from global 0xCA03E4, stack: 5 args) RET 0x14
const FindOrCreateHashEntry = @as(*const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x693D60));
// Global state used by the game's rendering pipeline
const g_visible_count: *u32 = @ptrFromInt(0xCE26E0); // PTR_00ce26e0
const g_visible_list: [*]u16 = @ptrFromInt(0xCE26E8); // DAT_00ce26e8
const g_render_count: *u32 = @ptrFromInt(0xCE66FC); // PTR_00ce66fc
const g_render_list: [*]u16 = @ptrFromInt(0xCDE648); // DAT_00cde648
const g_guard: *const u32 = @ptrFromInt(0xCA03E4); // PTR_00ca03e4
const V4 = @Vector(4, f32);
fn readU32(addr: u32) u32 {
return @as(*align(1) const u32, @ptrFromInt(addr)).*;
}
fn readU16(addr: u32) u16 {
return @as(*align(1) const u16, @ptrFromInt(addr)).*;
}
fn readF32(addr: u32) f32 {
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
}
/// Compute outcodes for all vertices in the mesh using SSE.
///
/// Per-vertex 6-bit outcode against AABB. Two SIMD compares (v < min, v >= max)
/// produce all 6 bits from movemask results. Processes xyz in parallel.
///
/// Bit layout: 0x20=below_minX, 0x10=above_maxX, 0x08=below_minY,
/// 0x04=above_maxY, 0x02=below_minZ, 0x01=above_maxZ
fn computeAllOutcodes(
hash_entry: u32,
min_x: f32,
max_x: f32,
min_y: f32,
max_y: f32,
min_z: f32,
max_z: f32,
cull_flags: *[452]u8,
) u32 {
const vert_count: u32 = readU16(hash_entry + 6);
if (vert_count == 0) return 0;
const v_min: V4 = .{ min_x, min_y, min_z, 0 };
const v_max: V4 = .{ max_x, max_y, max_z, 0 };
// Bit weights for branchless outcode: below gives 0x20/0x08/0x02, above gives 0x10/0x04/0x01
const below_w: @Vector(4, u32) = .{ 0x20, 0x08, 0x02, 0x00 };
const above_w: @Vector(4, u32) = .{ 0x10, 0x04, 0x01, 0x00 };
var vert_ptr: [*]const f32 = @ptrFromInt(hash_entry + 8);
var i: u32 = 0;
while (i < vert_count) : (i += 1) {
// Single 16-byte unaligned load. 4th float is junk from next vertex
// but v_min[3]=0, v_max[3]=0, so comparisons on lane 3 produce
// below=false (0>=0), above=true (0>=0) -- weight is 0x00 so harmless.
const v: V4 = @as(*align(1) const V4, @ptrCast(vert_ptr)).*;
// Bool vectors -> u32 vectors (0 or 0xFFFFFFFF), AND with weights, horizontal OR
const zero: @Vector(4, u32) = @splat(0);
const below_masked = @select(u32, v < v_min, below_w, zero);
const above_masked = @select(u32, v >= v_max, above_w, zero);
const combined = below_masked | above_masked;
// Horizontal OR of 4 lanes -> single outcode byte
cull_flags[i] = @truncate(combined[0] | combined[1] | combined[2] | combined[3]);
vert_ptr += 3; // stride 12 bytes = 3 floats
}
return vert_count;
}
/// PerformSpatialCulling (0x6B8C60)
/// __thiscall(this, keyData, keySize) -> u32. RET 0x8.
///
/// Finds mesh data via hash, computes vertex outcodes against AABB from this+0x10,
/// then iterates triangles: filters by visibility mask, trivial-rejects by outcode AND,
/// adds survivors to global visible/render lists.
export fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
if (g_guard.* == 0) return 0;
const hash_table = g_guard.*;
const hash_entry = @call(.never_tail, FindOrCreateHashEntry, .{
hash_table, key_data, key_size,
readU32(this + 4), readU32(this + 8), readU32(this + 0xC),
});
if (hash_entry == 0) return 0;
// Load AABB from *(this+0x10) -- 6 floats: minX, minY, minZ, maxX, maxY, maxZ
const bounds_ptr = readU32(this + 0x10);
const min_x = readF32(bounds_ptr);
const min_y = readF32(bounds_ptr + 4);
const min_z = readF32(bounds_ptr + 8);
const max_x = readF32(bounds_ptr + 12);
const max_y = readF32(bounds_ptr + 16);
const max_z = readF32(bounds_ptr + 20);
// Phase 1: Compute per-vertex outcodes
var cull_flags: [452]u8 = undefined;
_ = computeAllOutcodes(hash_entry, min_x, max_x, min_y, max_y, min_z, max_z, &cull_flags);
// Phase 2: Iterate triangles
const tri_count: u32 = @as(u32, readU16(hash_entry + 0x18A4));
const filter_mask = readU16(this + 0x14);
const visited_base = readU32(this + 4);
var ti: u32 = 0;
while (ti < tri_count) : (ti += 1) {
// Visibility mask filter
const vis_flags = readU16(hash_entry + 0x1FAE + ti * 2);
if ((vis_flags & filter_mask) != 0) continue;
// Per-vertex visited filter
const tri_base_idx = readU16(hash_entry + 0x2206 + ti * 2);
const visited_byte = @as(*u8, @ptrFromInt(visited_base + @as(u32, tri_base_idx) * 2));
if ((visited_byte.* & @as(u8, @truncate(filter_mask))) != 0) continue;
// Check global visible list capacity
if (g_visible_count.* >= 0x2000) {
const flags_ptr = readU32(this);
if (flags_ptr != 0) {
const p: *u32 = @ptrFromInt(flags_ptr);
p.* |= 1;
}
break;
}
// Add to visible list
g_visible_list[g_visible_count.*] = tri_base_idx;
g_visible_count.* += 1;
visited_byte.* |= 0x80;
// Frustum test: AND of 3 vertex outcodes. If any bit shared, fully outside.
const idx0 = readU16(hash_entry + 0x18A6 + ti * 6);
const idx1 = readU16(hash_entry + 0x18A8 + ti * 6);
const idx2 = readU16(hash_entry + 0x18AA + ti * 6);
// Note: decompiler shows idx offsets as 0x18A6, +0xC54*2, +0x18AA
// which is 0x18A6 (idx0), 0x18A8 (idx1), 0x18AA (idx2) -- stride 6 = 3 u16 per tri
if ((cull_flags[idx0] & cull_flags[idx1] & cull_flags[idx2] & 0x3F) == 0) {
g_render_list[g_render_count.*] = tri_base_idx;
g_render_count.* += 1;
}
}
return 1;
}
// =============================================================================
// SSE vector helpers for Moller-Trumbore
// =============================================================================
inline fn loadVec3(addr: u32) V4 {
return .{ readF32(addr), readF32(addr + 4), readF32(addr + 8), 0 };
}
inline fn cross(a: V4, b: V4) V4 {
const Mask = @Vector(4, i32);
const a_yzx: V4 = @shuffle(f32, a, undefined, Mask{ 1, 2, 0, 3 });
const a_zxy: V4 = @shuffle(f32, a, undefined, Mask{ 2, 0, 1, 3 });
const b_yzx: V4 = @shuffle(f32, b, undefined, Mask{ 1, 2, 0, 3 });
const b_zxy: V4 = @shuffle(f32, b, undefined, Mask{ 2, 0, 1, 3 });
return a_yzx * b_zxy - a_zxy * b_yzx;
}
inline fn dot3(a: V4, b: V4) f32 {
const p = a * b;
return p[0] + p[1] + p[2];
}
/// performCollisionDetection (0x6B88E0)
/// __thiscall(this, keyData, keySize) -> u32. RET 0x8.
///
/// Fully inlined SSE rewrite. No external calls except FindOrCreateHashEntry.
/// Moller-Trumbore ray-triangle intersection is inlined with SSE cross/dot,
/// eliminating 4 SetVector3 calls and the ray_tri function call per triangle.
export fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
if (g_guard.* == 0) return 0;
const hash_table = g_guard.*;
const hash_entry = @call(.never_tail, FindOrCreateHashEntry, .{
hash_table, key_data, key_size,
readU32(this + 4), readU32(this + 8), readU32(this + 0xC),
});
if (hash_entry == 0) return 0;
// Load and sort AABB extents from this+0x18..0x2C
var ax0 = readF32(this + 0x18);
var ax1 = readF32(this + 0x24);
if (ax1 < ax0) {
const tmp = ax0;
ax0 = ax1;
ax1 = tmp;
}
var ay0 = readF32(this + 0x1C);
var ay1 = readF32(this + 0x28);
if (ay1 < ay0) {
const tmp = ay0;
ay0 = ay1;
ay1 = tmp;
}
var az0 = readF32(this + 0x20);
var az1 = readF32(this + 0x2C);
if (az1 < az0) {
const tmp = az0;
az0 = az1;
az1 = tmp;
}
// Phase 1: Compute per-vertex outcodes
var cull_flags: [452]u8 = undefined;
_ = computeAllOutcodes(hash_entry, ax0, ax1, ay0, ay1, az0, az1, &cull_flags);
// Phase 2: Iterate triangles with inline ray-tri test
const tri_count: u32 = @as(u32, readU16(hash_entry + 0x18A4));
const collision_mask = readU16(this + 0x50);
const visited_base = readU32(this + 4);
const vert_pool = hash_entry + 8;
// Ray: origin at this+0x00 (position), direction at this+0x0C (3 floats)
// Original uses param_1 = ESI which points to a 6-float struct:
// [0..2] = ray origin, [3..5] = ray direction
// The call site passes this+0x30 as the ray struct
const ray_origin = loadVec3(this + 0x30);
const ray_dir = loadVec3(this + 0x3C);
// Epsilon for barycentric bounds: original uses +/- param_6 (0.002)
const eps: f32 = 0.002;
const neg_eps: f32 = -eps;
const one_plus_eps: f32 = 1.0 + eps;
var ti: u32 = 0;
while (ti < tri_count) : (ti += 1) {
const vis_flags = readU16(hash_entry + 0x1FAE + ti * 2);
if ((vis_flags & collision_mask) != 0) continue;
const tri_base_idx = readU16(hash_entry + 0x2206 + ti * 2);
const visited_addr = visited_base + @as(u32, tri_base_idx) * 2;
const visited_byte = @as(*u8, @ptrFromInt(visited_addr));
if ((visited_byte.* & @as(u8, @truncate(collision_mask))) != 0) continue;
// Add to visible list and mark visited
g_visible_list[g_visible_count.*] = tri_base_idx;
g_visible_count.* += 1;
visited_byte.* |= 0x80;
// Frustum outcode test
const idx_base = hash_entry + 0x18A6 + ti * 6;
const vi0: u32 = readU16(idx_base);
const vi1: u32 = readU16(idx_base + 2);
const vi2: u32 = readU16(idx_base + 4);
if ((cull_flags[vi0] & cull_flags[vi1] & cull_flags[vi2] & 0x3F) != 0) continue;
// =====================================================================
// Inline Moller-Trumbore ray-triangle intersection (SSE)
// Deferred divide: compare u_raw and v_raw against det-scaled bounds
// to avoid the 1/det divide on the reject path.
// =====================================================================
const v0 = loadVec3(vert_pool + vi0 * 12);
const v1 = loadVec3(vert_pool + vi1 * 12);
const v2 = loadVec3(vert_pool + vi2 * 12);
const edge1 = v1 - v0;
const edge2 = v2 - v0;
const pvec = cross(ray_dir, edge2);
const det = dot3(edge1, pvec);
if (det <= 1e-7 and det >= -1e-7) continue;
const tvec = ray_origin - v0;
// u_raw = dot(tvec, pvec) -- NOT multiplied by inv_det yet
const u_raw = dot3(tvec, pvec);
// Compare u_raw against det-scaled epsilon bounds.
// If det > 0: u = u_raw/det, so u < -eps iff u_raw < -eps*det, u > 1+eps iff u_raw > (1+eps)*det
// If det < 0: division flips sign, so u < -eps iff u_raw > -eps*det (which is positive)
// Trick: multiply both sides by sign(det) to normalize.
// Or equivalently: if det>0 check u_raw in [det*neg_eps, det*one_plus_eps]
// if det<0 check u_raw in [det*one_plus_eps, det*neg_eps]
const det_neg_eps = det * neg_eps;
const det_one_plus = det * one_plus_eps;
if (det > 0) {
if (u_raw < det_neg_eps or u_raw > det_one_plus) continue;
} else {
if (u_raw > det_neg_eps or u_raw < det_one_plus) continue;
}
const qvec = cross(tvec, edge1);
const v_raw = dot3(ray_dir, qvec);
// Same sign-aware bounds check for v
if (det > 0) {
if (v_raw < det_neg_eps or (u_raw + v_raw) > det_one_plus) continue;
} else {
if (v_raw > det_neg_eps or (u_raw + v_raw) < det_one_plus) continue;
}
// Only divide for confirmed hits
const t = dot3(edge2, qvec) / det;
if (t >= 0.0 and t < readF32(this + 0x4C)) {
// Update closest hit
@as(*align(1) f32, @ptrFromInt(this + 0x4C)).* = t;
g_render_list[0] = tri_base_idx;
g_render_count.* = 1;
// Write scaled distance, clamped to max
const result_ptr: *align(1) f32 = @ptrFromInt(readU32(this + 0x10));
const scaled = t * readF32(this + 0x48);
const clamp = readF32(this + 0x14);
result_ptr.* = if (scaled <= clamp) scaled else clamp;
}
}
return 1;
}
+212
View File
@@ -0,0 +1,212 @@
//! SSE-optimized entity update functions -- compiled ReleaseFast.
//!
//! Reimplements UpdateEntityAndChunksPositions (0x6AFAD0) and
//! updateEntitiesInBounds (0x6C1F70).
//!
//! Main optimization: SSE dot product for view-depth computation,
//! tighter control flow, reduced function call overhead.
const V4 = @Vector(4, f32);
fn readU32(addr: u32) u32 {
return @as(*align(1) const u32, @ptrFromInt(addr)).*;
}
fn readI32(addr: u32) i32 {
return @as(*align(1) const i32, @ptrFromInt(addr)).*;
}
fn readF32(addr: u32) f32 {
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
}
fn writeU32(addr: u32, val: u32) void {
@as(*align(1) u32, @ptrFromInt(addr)).* = val;
}
fn writeF32(addr: u32, val: f32) void {
@as(*align(1) f32, @ptrFromInt(addr)).* = val;
}
// =========================================================================
// Game function declarations (resolved at link time via absolute address)
// =========================================================================
// recycleVertexBuffer: fastcall(ECX=bufPtr, EDX=sizePtr), RET
const recycleVertexBuffer = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AE9A0));
// check_instances_active: fastcall(ECX=instanceMgr) -> ptr, RET
const check_instances_active = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) u32, @ptrFromInt(0x6B2900));
// store_all_instance_buffers: fastcall(ECX=instanceMgr), RET
const store_all_instance_buffers = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6B28E0));
// return_object_to_pool: fastcall(ECX=instanceMgr), RET
const return_object_to_pool = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6B2030));
// ReturnChunkBuffers: fastcall(ECX=bufPtr, EDX=sizePtr), RET
const ReturnChunkBuffers = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x68CD50));
// IsPointInsideBounds: fastcall(ECX=point, EDX=bounds) -> u32, RET
const IsPointInsideBounds = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) u32, @ptrFromInt(0x699330));
// AddObjectToSpatialList: fastcall(ECX=entityPtr, EDX=posPtr), RET
const AddObjectToSpatialList = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6818B0));
// AddToSpatialGrid: fastcall(ECX=objPtr), RET
const AddToSpatialGrid = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6816F0));
// CopyChunkBounds: thiscall(ECX=chunk, stack=outBounds), RET 0x4
const CopyChunkBounds = @as(*const fn (u32, u32) callconv(.{ .x86_thiscall = .{} }) void, @ptrFromInt(0x68DF40));
// AddToLayeredSpatialGrid: fastcall(ECX=chunk, EDX=idx, stack=posPtr), RET 0x4
const AddToLayeredSpatialGrid = @as(*const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x681970));
// ComplexMemoryCleanupAndRelease: fastcall(ECX=memObj)
const ComplexMemoryCleanupAndRelease = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6A0510));
// destroySecondaryGameObject: fastcall(ECX=entityPtr)
const destroySecondaryGameObject = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6A6A00));
// Game globals
const g_viewCoeffs: u32 = 0xC7BCB0; // 4 floats: a, b, c, d for dot product
const g_renderFlags: *const u8 = @ptrFromInt(0x867960 + 8); // actually at different offset
const g_deltaTime: *const u32 = @ptrFromInt(0xC62510); // PTR_00c62510
const g_timerThreshold: *const f32 = @ptrFromInt(0x80A1E8); // _DAT_0080a1e8
// =========================================================================
// UpdateEntityAndChunksPositions (0x6AFAD0)
// __fastcall(ECX=entityPtr), RET
// =========================================================================
export fn updateEntityAndChunksPositions(ent: u32) callconv(.{ .x86_fastcall = .{} }) void {
// Dot product: depth = a*x + b*y + c*z + d - offset
const depth = readF32(g_viewCoeffs) * readF32(ent + 0x5C) +
readF32(g_viewCoeffs + 4) * readF32(ent + 0x60) +
readF32(g_viewCoeffs + 8) * readF32(ent + 0x64) +
readF32(g_viewCoeffs + 12) - readF32(ent + 0x68);
writeF32(ent + 0x78, depth);
// Render distance flag
writeU32(ent + 0xB8, readU32(0x867964));
if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 4) != 0 and readF32(0x867960) < depth) {
writeU32(ent + 0xB8, readU32(0x867968));
}
// Timer accumulation
const dt_bits = g_deltaTime.*;
const dt: f32 = @bitCast(dt_bits);
const timer = readF32(ent + 0xAC) + dt;
writeF32(ent + 0xAC, timer);
const threshold = g_timerThreshold.*;
// Vertex buffer recycling
if (threshold < timer and readU32(ent + 0x14C) != 0) {
@call(.never_tail, recycleVertexBuffer, .{ ent + 0x14C, ent + 0x150 });
}
// Instance buffer management
const inst_mgr = readU32(ent + 0xC0);
if (inst_mgr != 0) {
if (1.0 < readF32(ent + 0xAC)) {
const active = @call(.never_tail, check_instances_active, .{inst_mgr});
if (active != 0) {
@call(.never_tail, store_all_instance_buffers, .{inst_mgr});
}
}
if (threshold < readF32(ent + 0xAC)) {
@call(.never_tail, return_object_to_pool, .{inst_mgr});
writeU32(ent + 0xC0, 0);
}
}
// Chunk timer loop (4 chunks at ent+0x118, stride 4)
inline for (0..4) |ci| {
const chunk = readU32(ent + 0x118 + ci * 4);
if (chunk != 0) {
const chunk_timer = readF32(chunk + 0x30) + dt;
writeF32(chunk + 0x30, chunk_timer);
if (readU32(chunk + 0x400) != 0 and threshold < chunk_timer) {
@call(.never_tail, ReturnChunkBuffers, .{ chunk + 0x400, chunk + 0x404 });
}
}
}
// Bounds check and spatial grid registration
const bx = readF32(ent + 0x44);
const by = readF32(ent + 0x48);
const bz = readF32(ent + 0x4C);
if (bx <= readF32(0xC7CB68) and by <= readF32(0xC7CB6C) and bz <= readF32(0xC7CB70)) {
const inside = @call(.never_tail, IsPointInsideBounds, .{ ent + 0x50, 0xC7CB5C });
if (inside != 0) {
// Compute position from animation data
const anim_idx = readU32(0x86B580 + readU32(0xC7F294) * 4);
const anim_base = ent + 0x83C + @as(u32, @bitCast(anim_idx)) * 0xC;
var pos: [3]f32 = undefined;
pos[0] = readF32(anim_base) + readF32(ent + 0x6C);
pos[1] = readF32(anim_base + 4) + readF32(ent + 0x70);
pos[2] = readF32(anim_base + 8) + readF32(ent + 0x74);
@call(.never_tail, AddObjectToSpatialList, .{ ent, @intFromPtr(&pos) });
// Walk sub-object linked list
var node = readU32(ent + 0xE4);
if ((node & 1) != 0 or node == 0) node = 0;
while ((node & 1) == 0 and node != 0) {
const obj = readU32(node + 4);
if ((@as(*const u8, @ptrFromInt(obj + 0xC)).* & 0x80) != 0 and (readU32(obj + 0x88) != 0 or readU32(obj + 0x174) != 0)) {
@call(.never_tail, AddToSpatialGrid, .{obj});
}
node = readU32(readU32(ent + 0xDC) + node + 4);
}
}
}
// Chunk bounds + spatial grid loop (4 chunks)
inline for (0..4) |ci| {
const chunk = readU32(ent + 0x118 + ci * 4);
if (chunk != 0) {
var bounds: [6]f32 = undefined;
@call(.never_tail, CopyChunkBounds, .{ chunk, @intFromPtr(&bounds) });
// Check if chunk bounds intersect the world region
if (bounds[0] <= readF32(0xC7CB68) and bounds[1] <= readF32(0xC7CB6C) and
bounds[2] <= readF32(0xC7CB70) and readF32(0xC7CB5C) <= bounds[3] and
readF32(0xC7CB60) <= bounds[4] and readF32(0xC7CB64) <= bounds[5])
{
var center: [3]f32 = undefined;
center[0] = (bounds[3] + bounds[0]) * 0.5;
center[1] = (bounds[4] + bounds[1]) * 0.5;
center[2] = (bounds[5] + bounds[2]) * 0.5;
@call(.never_tail, AddToLayeredSpatialGrid, .{ chunk, @as(u32, ci), @intFromPtr(&center) });
}
}
}
}
// =========================================================================
// updateEntitiesInBounds (0x6C1F70)
// __thiscall(ECX=this, stack=param_1), RET 0x4
// =========================================================================
export fn updateEntitiesInBoundsSSE(this: u32, param_1: u32) callconv(.{ .x86_thiscall = .{} }) void {
var node = readU32(this + 0x274);
if ((node & 1) != 0 or node == 0) node = 0;
while ((node & 1) == 0 and node != 0) {
const entity = readU32(node + 4);
const next = readU32(readU32(this + 0x26C) + 4 + node);
// Prefetch next node's entity data while we process this one
if ((next & 1) == 0 and next != 0) {
const next_entity = readU32(next + 4);
@prefetch(@as([*]const u8, @ptrFromInt(next_entity + 0x44)), .{ .locality = 1 });
@prefetch(@as([*]const u8, @ptrFromInt(next_entity + 0x8C)), .{ .locality = 1 });
}
// Bounds check: entity chunk coords vs global region
const cx = readI32(entity + 0x8C);
const cy = readI32(entity + 0x90);
if (cx < readI32(0xC63278) or readI32(0xC63280) < cx or
cy < readI32(0xC63274) or readI32(0xC6327C) < cy)
{
// Out of bounds: destroy
const slot_idx = readU32(entity + 0xB4); // entityPtr[0x2d] = +0xB4
writeU32(this + slot_idx * 4 + 0x278, 0);
@call(.never_tail, ComplexMemoryCleanupAndRelease, .{node});
@call(.never_tail, destroySecondaryGameObject, .{entity});
} else {
if (param_1 != 0) {
// Call through game address so the Detour hook fires (enables A/B timing)
const entPosHooked = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AFAD0));
@call(.never_tail, entPosHooked, .{entity});
}
}
node = next;
}
}
+8 -23
View File
@@ -64,14 +64,9 @@ fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
const TransformFn = fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
var transform_hook: hook.Detour(TransformFn) = .{};
var teardown_active: bool = false;
fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
if (teardown_active) {
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
} else {
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
}
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
}
// =============================================================================
@@ -153,22 +148,14 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
}
// =============================================================================
// Teardown hook (0x491180) — protect bone SSE during logout cleanup
// SSE JMP patches — binary patches at game function addresses
// =============================================================================
const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void;
var teardown_hook: hook.Detour(TeardownFn) = .{};
fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void {
teardown_active = true;
teardown_hook.callOriginal(.{});
teardown_active = false;
}
// =============================================================================
// Silicon SSE JMP patches — binary patches at game function addresses
// =============================================================================
// cull_sse.zig
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
// silicon_sse.zig
const sse = struct {
extern fn si_normalizeVec3() callconv(.naked) void;
extern fn si_mulMat3x4(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
@@ -236,6 +223,8 @@ fn installPatches() u32 {
.{ .target = 0x686820, .replacement = @intFromPtr(&sse.si_translateBoundingVol), .name = "translateBoundingVol" },
.{ .target = 0x6ABC40, .replacement = @intFromPtr(&sse.si_processLinkedListCollision), .name = "processLinkedListCollision" },
.{ .target = 0x686000, .replacement = @intFromPtr(&sse.si_frustumCullBBox), .name = "frustumCullBBox" },
.{ .target = 0x6B8C60, .replacement = @intFromPtr(&performSpatialCulling), .name = "PerformSpatialCulling" },
.{ .target = 0x6B88E0, .replacement = @intFromPtr(&performCollisionDetectionSSE), .name = "performCollisionDetection" },
};
var count: u32 = 0;
@@ -275,9 +264,6 @@ pub fn installHooks() void {
// Per-frame cache reset
if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1;
// Teardown guard
if (teardown_hook.attach(0x491180, &teardownDetour) == .ok) installed += 1;
// Silicon SSE binary patches
_ = installPatches();
@@ -299,7 +285,6 @@ pub fn removeHooks() void {
particle_hook.detach();
glyph_hook.detach();
world_update_hook.detach();
teardown_hook.detach();
log.close();
mod_mutex.release(&g_mutex);
}
+67 -4
View File
@@ -25,6 +25,10 @@ extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x8
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn renderParticleSprites_REF(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn resetParticleCache() void;
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern var stride_info: [8]u32; // exported from particle_sse.zig
@@ -71,7 +75,7 @@ var last_frame_tsc: u64 = 0; // frame-to-frame TSC for total frame time
pub var ab_use_custom: bool = false;
// Gate for non-transform A/B hooks. Set false to isolate transform44 SSE testing.
const AB_OTHER_HOOKS = false;
const AB_OTHER_HOOKS = true;
var diag_cmp_count: u32 = 0;
export var original_trampoline: u32 = 0; // DEBUG: expose trampoline for REF passthrough test
@@ -196,6 +200,12 @@ const ProfState = struct {
matmul_cycles: u64 = 0,
textline_calls: u64 = 0, // renderTextLine (0x5ce0c0)
textline_cycles: u64 = 0,
viewfrust_calls: u64 = 0, // SetupViewFrustum (0x6bc1c0) -- parent of PerformSpatialCulling
viewfrust_cycles: u64 = 0,
cylfrust_calls: u64 = 0, // SetupCylinderFrustum (0x6bc370) -- child of RenderSphere
cylfrust_cycles: u64 = 0,
rendersph_calls: u64 = 0, // RenderSphere (0x6b92b0) -- parent of SetupCylinderFrustum
rendersph_cycles: u64 = 0,
};
// =============================================================================
@@ -868,6 +878,9 @@ var triplane_hook: hook.Detour(Fn5) = .{}; // BuildTrianglePlanes: thiscall RET
var partsetup_hook: hook.Detour(Fn3) = .{}; // SetupParticleRendering: thiscall RET 0x4
var matmul_hook: hook.Detour(Fn3) = .{}; // multiplyMatrix4x4: fastcall RET 0x4
var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10
var viewfrust_hook: hook.Detour(Fn5) = .{}; // SetupViewFrustum: thiscall RET 0xC
var cylfrust_hook: hook.Detour(Fn5) = .{}; // SetupCylinderFrustum: thiscall RET 0xC
var rendersph_hook: hook.Detour(Fn8v) = .{}; // RenderSphere: thiscall RET 0x18
// --- Detour functions (timing-only pass-through) ---
@@ -940,6 +953,12 @@ fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*
}
fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
updateEntityAndChunksPositions(a);
prof.entpos_cycles +|= rdtsc() - s;
prof.entpos_calls +|= 1;
return null;
}
const ret = entpos_hook.callOriginal(.{ a, b });
prof.entpos_cycles +|= rdtsc() - s;
prof.entpos_calls +|= 1;
@@ -967,6 +986,7 @@ fn complexgeoDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u
return ret;
}
fn entboundsDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque {
// Original only -- A/B testing entpos first
const s = rdtsc();
const ret = entbounds_hook.callOriginal(.{ a, b, c });
prof.entbounds_cycles +|= rdtsc() - s;
@@ -1027,6 +1047,12 @@ fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
}
fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = performSpatialCulling(a, c, d);
prof.cull_cycles +|= rdtsc() - s;
prof.cull_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = cull_hook.callOriginal(.{ a, b, c, d });
prof.cull_cycles +|= rdtsc() - s;
prof.cull_calls +|= 1;
@@ -1034,6 +1060,12 @@ fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyop
}
fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = performCollisionDetectionSSE(a, c, d);
prof.colldet_cycles +|= rdtsc() - s;
prof.colldet_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = colldet_hook.callOriginal(.{ a, b, c, d });
prof.colldet_cycles +|= rdtsc() - s;
prof.colldet_calls +|= 1;
@@ -1174,6 +1206,27 @@ fn textlineDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.
prof.textline_calls +|= 1;
return ret;
}
fn viewfrustDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
const ret = viewfrust_hook.callOriginal(.{ a, b, c, d, e });
prof.viewfrust_cycles +|= rdtsc() - s;
prof.viewfrust_calls +|= 1;
return ret;
}
fn cylfrustDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
const ret = cylfrust_hook.callOriginal(.{ a, b, c, d, e });
prof.cylfrust_cycles +|= rdtsc() - s;
prof.cylfrust_calls +|= 1;
return ret;
}
fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
const ret = rendersph_hook.callOriginal(.{ a, b, c, d, e, f, g, h });
prof.rendersph_cycles +|= rdtsc() - s;
prof.rendersph_calls +|= 1;
return ret;
}
// =============================================================================
// Hook: blit_hub (0x5a4f60)
@@ -1414,6 +1467,9 @@ fn dumpStats() void {
.{ .name = "partsetup", .cycles = prof.partsetup_cycles, .calls = prof.partsetup_calls },
.{ .name = "matmul", .cycles = prof.matmul_cycles, .calls = prof.matmul_calls },
.{ .name = "textline", .cycles = prof.textline_cycles, .calls = prof.textline_calls },
.{ .name = "viewfrust", .cycles = prof.viewfrust_cycles, .calls = prof.viewfrust_calls },
.{ .name = "cylfrust", .cycles = prof.cylfrust_cycles, .calls = prof.cylfrust_calls },
.{ .name = "rendersph", .cycles = prof.rendersph_cycles, .calls = prof.rendersph_calls },
};
for (hotspots) |h| {
if (h.calls > 0) {
@@ -1551,8 +1607,9 @@ pub fn installHooks() void {
_ = linkedlist_hook.attach(0x710b90, &linkedlistDetour);
_ = color_hook.attach(0x7b9b10, &colorDetour);
_ = setvec_hook.attach(0x686640, &setvecDetour);
_ = cull_hook.attach(0x6b8c60, &cullDetour);
_ = colldet_hook.attach(0x6b88e0, &colldetDetour);
// cull + colldet graduated to weirdperformance JMP patches
// _ = cull_hook.attach(0x6b8c60, &cullDetour);
// _ = colldet_hook.attach(0x6b88e0, &colldetDetour);
_ = activep_hook.attach(0x7b5a10, &activepDetour);
_ = cbiter_hook.attach(0x404130, &cbiterDetour);
_ = findguid_hook.attach(0x464890, &findguidDetour);
@@ -1569,11 +1626,14 @@ pub fn installHooks() void {
_ = partsetup_hook.attach(0x7b3d20, &partsetupDetour);
_ = matmul_hook.attach(0x7bc6a0, &matmulDetour);
_ = textline_hook.attach(0x5ce0c0, &textlineDetour);
_ = viewfrust_hook.attach(0x6bc1c0, &viewfrustDetour);
_ = cylfrust_hook.attach(0x6bc370, &cylfrustDetour);
_ = rendersph_hook.attach(0x6b92b0, &rendersphDetour);
// Timer calibration now handled by performance module.
// blit_hub installed in lateInit() to clobber UnitXP's hook
log.print("transform44: 39 profiling hooks installed (blit_hub deferred)\n");
log.print("transform44: 42 profiling hooks installed (blit_hub deferred)\n");
if (bisect_stop_section != 0) {
log.fmt(" BISECT MODE: REF stops after section {d}, then original trampoline\n", .{bisect_stop_section});
}
@@ -1646,6 +1706,9 @@ pub fn removeHooks() void {
partsetup_hook.detach();
matmul_hook.detach();
textline_hook.detach();
viewfrust_hook.detach();
cylfrust_hook.detach();
rendersph_hook.detach();
blit_hub_hook.detach();
log.close();
mod_mutex.release(&g_mutex);