perf: SSE spatial culling, inlined ray-tri, clickthrough CC fix
PerformSpatialCulling (0x6B8C60): Zig rewrite with SSE outcode computation. 1.4x speedup, JMP-patched in weirdperformance. performCollisionDetection (0x6B88E0): fully inlined SSE Moller-Trumbore ray-triangle intersection, eliminating 4 SetVector3 calls and the ray_tri function pointer call per triangle. ~22 cyc/tri in bench. Both graduated from transform44 A/B testing to production JMP patches. entity_sse.zig: reimplementations of UpdateEntityAndChunksPositions and updateEntitiesInBounds (A/B tested, 1.2x bench, not shipped - memory bound with negligible real-world gain). clickthrough: fixed CheckObjectTypePermissions hook from fastcall to thiscall (ECX preservation), fixed ClntObjMgrObjectPtr from fastcall(4) to fastcall(5) with correct arg count. bench: added performCollisionDetection bench with synthetic mesh data and patched FindOrCreateHashEntry stub.
This commit is contained in:
@@ -65,6 +65,22 @@ pub fn build(b: *std.Build) void {
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const cull_sse_obj = b.addObject(.{
|
||||
.name = "cull_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/cull_sse.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const entity_sse_obj = b.addObject(.{
|
||||
.name = "entity_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/entity_sse.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bone_sse_target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .windows,
|
||||
@@ -172,6 +188,8 @@ pub fn build(b: *std.Build) void {
|
||||
// Called for both the main weirdutils build and each variant.
|
||||
const ModuleObjects = struct {
|
||||
clip_sse: *std.Build.Step.Compile,
|
||||
cull_sse: *std.Build.Step.Compile,
|
||||
entity_sse: *std.Build.Step.Compile,
|
||||
bone_sse: *std.Build.Step.Compile,
|
||||
bone_sse_ref: *std.Build.Step.Compile,
|
||||
math_sse: *std.Build.Step.Compile,
|
||||
@@ -184,6 +202,8 @@ pub fn build(b: *std.Build) void {
|
||||
@setEvalBranchQuota(10000);
|
||||
if (comptime std.mem.eql(u8, module_name, "weirdperformance")) {
|
||||
mod.addObject(self.clip_sse);
|
||||
mod.addObject(self.cull_sse);
|
||||
mod.addObject(self.entity_sse);
|
||||
mod.addObject(self.bone_sse);
|
||||
mod.addObject(self.bone_sse_ref);
|
||||
mod.addObject(self.silicon_sse);
|
||||
@@ -193,6 +213,8 @@ pub fn build(b: *std.Build) void {
|
||||
}
|
||||
if (comptime std.mem.eql(u8, module_name, "transform44")) {
|
||||
mod.addObject(self.clip_sse);
|
||||
mod.addObject(self.cull_sse);
|
||||
mod.addObject(self.entity_sse);
|
||||
mod.addObject(self.bone_sse);
|
||||
mod.addObject(self.bone_sse_ref);
|
||||
mod.addObject(self.particle_sse);
|
||||
@@ -208,6 +230,8 @@ pub fn build(b: *std.Build) void {
|
||||
};
|
||||
const objs = ModuleObjects{
|
||||
.clip_sse = clip_sse_obj,
|
||||
.cull_sse = cull_sse_obj,
|
||||
.entity_sse = entity_sse_obj,
|
||||
.bone_sse = bone_sse_obj,
|
||||
.bone_sse_ref = bone_sse_ref_obj,
|
||||
.math_sse = math_sse_obj,
|
||||
@@ -299,11 +323,29 @@ pub fn build(b: *std.Build) void {
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bench_cull_sse = b.addObject(.{
|
||||
.name = "bench_cull_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/cull_sse.zig"),
|
||||
.target = bench_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
bench.root_module.addObject(bench_math_sse);
|
||||
bench.root_module.addObject(bench_silicon_sse);
|
||||
bench.root_module.addObject(bench_bone_sse);
|
||||
bench.root_module.addObject(bench_bone_baseline);
|
||||
bench.root_module.addObject(bench_particle_sse);
|
||||
bench.root_module.addObject(bench_cull_sse);
|
||||
const bench_entity_sse = b.addObject(.{
|
||||
.name = "bench_entity_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/entity_sse.zig"),
|
||||
.target = bench_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
bench.root_module.addObject(bench_entity_sse);
|
||||
bench.root_module.linkSystemLibrary("m", .{});
|
||||
const install_bench = b.addInstallArtifact(bench, .{});
|
||||
const bench_step = b.step("bench", "Build math_sse benchmark harness (x86 Linux)");
|
||||
|
||||
+501
-2
@@ -58,6 +58,11 @@ extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void;
|
||||
extern fn si_setParticleAlpha(u32, u32, u32) callconv(cc_fc) void; // fastcall(ECX=obj, EDX=unused, stack=alpha)
|
||||
extern fn si_ftol() callconv(.naked) void;
|
||||
|
||||
// cull_sse.zig exports
|
||||
extern fn benchComputeOutcodes(u32, u32, u32, u32) void;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
|
||||
|
||||
// =========================================================================
|
||||
// Infrastructure
|
||||
// =========================================================================
|
||||
@@ -231,6 +236,14 @@ pub fn main() void {
|
||||
print("{s:>30} {s:>10} {s:>10} {s}\n", .{ "function", "original", "sse", "status" });
|
||||
print("{s}\n", .{"-" ** 72});
|
||||
|
||||
// Full performCollisionDetection (SSE vs original x87)
|
||||
bench_collisionDetection();
|
||||
|
||||
// UpdateEntityAndChunksPositions (SSE vs original x87)
|
||||
bench_entityUpdate();
|
||||
|
||||
if (false) { // disabled: not working on these right now
|
||||
|
||||
// 1: vecMulMat4 -- fastcall(ECX=result, EDX=vec, stack=mat) -> u32
|
||||
bench_fc3r("vecMulMat4_ColMajor", originals.vecMulMat4_ColMajor, &vecMulMat4_ColMajor, tv3(), tm4(), 3);
|
||||
|
||||
@@ -1770,8 +1783,8 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
|
||||
// calcColorValues_SSE -- thiscall(ctx_ECX, time, scale, outColor, outAlpha1, outAlpha2, outFloat)
|
||||
bench_calcColorValues();
|
||||
// calcColorValues_SSE -- disabled: no standalone SSE export yet
|
||||
// bench_calcColorValues();
|
||||
|
||||
// si_frustumCullBBox -- fastcall(bbox_ECX, flags_EDX, radius_stack) -> u32
|
||||
bench_frustumCullBBox();
|
||||
@@ -1780,6 +1793,8 @@ pub fn main() void {
|
||||
// Builds a fake linked list with 8 nodes to benchmark AABB overlap test.
|
||||
bench_processLinkedListCollision();
|
||||
|
||||
} // end disabled block
|
||||
|
||||
print("\n", .{});
|
||||
}
|
||||
|
||||
@@ -2174,6 +2189,29 @@ fn bench_tc2r(
|
||||
// =========================================================================
|
||||
|
||||
const V4 = @Vector(4, f32);
|
||||
const ShufMask = @Vector(4, i32);
|
||||
|
||||
inline fn benchLoadV3(addr: u32) V4 {
|
||||
return .{
|
||||
@as(*align(1) const f32, @ptrFromInt(addr)).*,
|
||||
@as(*align(1) const f32, @ptrFromInt(addr + 4)).*,
|
||||
@as(*align(1) const f32, @ptrFromInt(addr + 8)).*,
|
||||
0,
|
||||
};
|
||||
}
|
||||
|
||||
inline fn benchCross(av: V4, bv: V4) V4 {
|
||||
const a_yzx: V4 = @shuffle(f32, av, undefined, ShufMask{ 1, 2, 0, 3 });
|
||||
const a_zxy: V4 = @shuffle(f32, av, undefined, ShufMask{ 2, 0, 1, 3 });
|
||||
const b_yzx: V4 = @shuffle(f32, bv, undefined, ShufMask{ 1, 2, 0, 3 });
|
||||
const b_zxy: V4 = @shuffle(f32, bv, undefined, ShufMask{ 2, 0, 1, 3 });
|
||||
return a_yzx * b_zxy - a_zxy * b_yzx;
|
||||
}
|
||||
|
||||
inline fn benchDot3(av: V4, bv: V4) f32 {
|
||||
const p = av * bv;
|
||||
return p[0] + p[1] + p[2];
|
||||
}
|
||||
|
||||
inline fn inline_x87_dot(va: *const Vec3, vb: *const Vec3, out: *f32) void {
|
||||
asm volatile (
|
||||
@@ -2275,3 +2313,464 @@ inline fn inline_sse_horner(c: *const [4]f32, f: f32, out: *volatile f32) void {
|
||||
r = r * f + c.*[3];
|
||||
out.* = r;
|
||||
}
|
||||
|
||||
fn sseRayTri(ray_ptr: u32, vert_pool: u32, idx_base: u32, t_out: *f32) bool {
|
||||
const vi0: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base)).*;
|
||||
const vi1: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base + 2)).*;
|
||||
const vi2: u32 = @as(*align(1) const u16, @ptrFromInt(idx_base + 4)).*;
|
||||
|
||||
const ray_o = benchLoadV3(ray_ptr);
|
||||
const ray_d = benchLoadV3(ray_ptr + 12);
|
||||
const v0 = benchLoadV3(vert_pool + vi0 * 12);
|
||||
const v1 = benchLoadV3(vert_pool + vi1 * 12);
|
||||
const v2 = benchLoadV3(vert_pool + vi2 * 12);
|
||||
|
||||
const edge1 = v1 - v0;
|
||||
const edge2 = v2 - v0;
|
||||
const pvec = benchCross(ray_d, edge2);
|
||||
const det = benchDot3(edge1, pvec);
|
||||
if (det <= 1e-7 and det >= -1e-7) return false;
|
||||
|
||||
const inv_det = 1.0 / det;
|
||||
const tvec = ray_o - v0;
|
||||
const u = benchDot3(tvec, pvec) * inv_det;
|
||||
if (u < -0.002 or u > 1.002) return false;
|
||||
|
||||
const qvec = benchCross(tvec, edge1);
|
||||
const v = benchDot3(ray_d, qvec) * inv_det;
|
||||
if (v < -0.002 or (u + v) > 1.002) return false;
|
||||
|
||||
t_out.* = benchDot3(edge2, qvec) * inv_det;
|
||||
return true;
|
||||
}
|
||||
|
||||
fn bench_collisionDetection() void {
|
||||
// Build synthetic mesh data matching game's hash entry layout.
|
||||
// 32 vertices forming a grid, 20 triangles, ray aimed through the middle.
|
||||
|
||||
const NVERTS = 120;
|
||||
const NTRIS = 40;
|
||||
|
||||
// Hash entry: total size must accommodate all fields up to 0x2206 + NTRIS*2
|
||||
// Max offset: 0x2206 + 20*2 = 0x222E, round up
|
||||
var hash_buf: [0x2300]u8 align(4) = [_]u8{0} ** 0x2300;
|
||||
const he = @intFromPtr(&hash_buf);
|
||||
|
||||
// Vertex count at +6
|
||||
@as(*align(1) u16, @ptrFromInt(he + 6)).* = NVERTS;
|
||||
|
||||
// Carefully crafted vertices to produce a mix of hits and misses with
|
||||
// non-trivial barycentric coordinates. Ray fires from (0,0,-10) along +Z.
|
||||
// Triangles 0-4: guaranteed hits at various u/v (straddling the ray axis)
|
||||
// Triangles 5-9: near-misses (edge/corner cases for barycentric bounds)
|
||||
// Triangles 10-14: clear misses (outside AABB or backfacing)
|
||||
// Triangles 15-19: more hits with small/large det values (tests divide precision)
|
||||
const verts = [NVERTS][3]f32{
|
||||
// --- Group A: clear hits at various depths, u/v values ---
|
||||
// Tri 0: large centered, hit u~0.33 v~0.33
|
||||
.{ -2.0, -2.0, 1.0 }, .{ 4.0, -2.0, 1.0 }, .{ -2.0, 4.0, 1.0 },
|
||||
// Tri 1: small on-axis, hit u~0.5 v~0.25
|
||||
.{ -0.5, -0.5, 2.0 }, .{ 0.5, -0.5, 2.0 }, .{ 0.0, 0.5, 2.0 },
|
||||
// Tri 2: very close to origin
|
||||
.{ -1.0, -1.0, 0.1 }, .{ 1.0, -1.0, 0.1 }, .{ 0.0, 1.0, 0.1 },
|
||||
// Tri 3: backface hit (wound CW)
|
||||
.{ -2.0, 4.0, 4.0 }, .{ 4.0, -2.0, 4.0 }, .{ -2.0, -2.0, 4.0 },
|
||||
// Tri 4: tiny triangle, tests large inv_det
|
||||
.{ -0.05, -0.05, 1.5 }, .{ 0.05, -0.05, 1.5 }, .{ 0.0, 0.05, 1.5 },
|
||||
// Tri 5: huge triangle, tests small inv_det
|
||||
.{ -50.0, -50.0, 2.5 }, .{ 50.0, -50.0, 2.5 }, .{ 0.0, 50.0, 2.5 },
|
||||
// Tri 6: hit at u~0, v~0 (near vertex 0)
|
||||
.{ -0.001, -0.001, 3.0 }, .{ 5.0, -0.001, 3.0 }, .{ -0.001, 5.0, 3.0 },
|
||||
// Tri 7: hit at u~1, v~0 (near vertex 1)
|
||||
.{ -5.0, -0.001, 3.5 }, .{ 0.001, -0.001, 3.5 }, .{ -5.0, 5.0, 3.5 },
|
||||
// Tri 8: hit at u~0, v~1 (near vertex 2)
|
||||
.{ -5.0, -5.0, 4.0 }, .{ 5.0, -5.0, 4.0 }, .{ 0.001, 0.001, 4.0 },
|
||||
// Tri 9: hit with u+v very close to 1.0 (edge between v1-v2)
|
||||
.{ -0.01, -0.01, 4.5 }, .{ 2.0, -0.01, 4.5 }, .{ -0.01, 2.0, 4.5 },
|
||||
|
||||
// --- Group B: edge cases that should barely miss ---
|
||||
// Tri 10: ray just outside triangle edge
|
||||
.{ 0.5, -0.5, 5.0 }, .{ 2.0, -0.5, 5.0 }, .{ 0.5, 1.0, 5.0 },
|
||||
// Tri 11: ray misses on v side
|
||||
.{ -3.0, 0.5, 5.5 }, .{ -0.5, 0.5, 5.5 }, .{ -3.0, 2.0, 5.5 },
|
||||
// Tri 12: triangle behind ray (negative t)
|
||||
.{ -1.0, -1.0, -15.0 }, .{ 1.0, -1.0, -15.0 }, .{ 0.0, 1.0, -15.0 },
|
||||
// Tri 13: triangle way off to the side
|
||||
.{ 10.0, 10.0, 1.0 }, .{ 12.0, 10.0, 1.0 }, .{ 10.0, 12.0, 1.0 },
|
||||
// Tri 14: triangle off to the other side
|
||||
.{ -12.0, -12.0, 2.0 }, .{ -10.0, -12.0, 2.0 }, .{ -12.0, -10.0, 2.0 },
|
||||
|
||||
// --- Group C: degenerate/parallel ---
|
||||
// Tri 15: zero-area (all same point)
|
||||
.{ 1.0, 1.0, 6.0 }, .{ 1.0, 1.0, 6.0 }, .{ 1.0, 1.0, 6.0 },
|
||||
// Tri 16: collinear vertices
|
||||
.{ -1.0, 0.0, 7.0 }, .{ 0.0, 0.0, 7.0 }, .{ 1.0, 0.0, 7.0 },
|
||||
// Tri 17: nearly parallel to ray (plane nearly parallel to Z axis)
|
||||
.{ -1.0, -100.0, 0.5 }, .{ 1.0, -100.0, 0.5 }, .{ 0.0, 100.0, 0.501 },
|
||||
// Tri 18: parallel to ray (exactly in XY plane at z=0, ray along Z)
|
||||
.{ -1.0, -1.0, 0.0 }, .{ 1.0, -1.0, 0.0 }, .{ 0.0, 1.0, 0.0 },
|
||||
|
||||
// --- Group D: more hits at various depths for closest-t tracking ---
|
||||
// Tri 19: closest possible hit
|
||||
.{ -5.0, -5.0, 0.01 }, .{ 5.0, -5.0, 0.01 }, .{ 0.0, 5.0, 0.01 },
|
||||
// Tri 20-24: hits at regular depth intervals
|
||||
.{ -0.3, -0.3, 0.5 }, .{ 0.3, -0.3, 0.5 }, .{ 0.0, 0.3, 0.5 },
|
||||
.{ -1.0, -1.0, 1.2 }, .{ 1.0, -1.0, 1.2 }, .{ 0.0, 1.0, 1.2 },
|
||||
.{ -0.8, -0.8, 2.0 }, .{ 0.8, -0.8, 2.0 }, .{ 0.0, 0.8, 2.0 },
|
||||
.{ -1.5, -1.5, 3.0 }, .{ 1.5, -1.5, 3.0 }, .{ 0.0, 1.5, 3.0 },
|
||||
.{ -2.0, -2.0, 5.5 }, .{ 2.0, -2.0, 5.5 }, .{ 0.0, 2.0, 5.5 },
|
||||
|
||||
// --- Group E: outside AABB (outcode rejects, never reach ray-tri) ---
|
||||
// Tri 25: all verts above AABB
|
||||
.{ -1.0, 5.0, 1.0 }, .{ 1.0, 5.0, 1.0 }, .{ 0.0, 6.0, 1.0 },
|
||||
// Tri 26: all verts below AABB
|
||||
.{ -1.0, -6.0, 1.0 }, .{ 1.0, -6.0, 1.0 }, .{ 0.0, -5.0, 1.0 },
|
||||
// Tri 27: all verts left of AABB
|
||||
.{ -6.0, -1.0, 1.0 }, .{ -5.0, -1.0, 1.0 }, .{ -6.0, 1.0, 1.0 },
|
||||
// Tri 28: all verts in front of AABB (z < min)
|
||||
.{ -1.0, -1.0, -5.0 }, .{ 1.0, -1.0, -5.0 }, .{ 0.0, 1.0, -5.0 },
|
||||
// Tri 29: all verts behind AABB (z > max)
|
||||
.{ -1.0, -1.0, 5.0 }, .{ 1.0, -1.0, 5.0 }, .{ 0.0, 1.0, 5.0 },
|
||||
|
||||
// --- Group F: asymmetric/skewed hits testing det sign & magnitude ---
|
||||
// Tri 30: very elongated, hit near tip
|
||||
.{ 0.0, -0.01, 1.8 }, .{ 0.02, -0.01, 1.8 }, .{ 0.0, 10.0, 1.8 },
|
||||
// Tri 31: very flat (nearly zero Y extent)
|
||||
.{ -5.0, -0.001, 2.2 }, .{ 5.0, -0.001, 2.2 }, .{ 0.0, 0.001, 2.2 },
|
||||
// Tri 32: large negative det
|
||||
.{ -3.0, 3.0, 2.8 }, .{ 3.0, -3.0, 2.8 }, .{ -3.0, -3.0, 2.8 },
|
||||
// Tri 33: det exactly at threshold boundary
|
||||
.{ -0.0001, -0.0001, 6.5 }, .{ 0.0001, -0.0001, 6.5 }, .{ 0.0, 0.0001, 6.5 },
|
||||
|
||||
// --- Group G: stress closest-t with many competing hits ---
|
||||
// Tri 34-39: hits at very close z-values to test precision
|
||||
.{ -1.0, -1.0, 0.100 }, .{ 1.0, -1.0, 0.100 }, .{ 0.0, 1.0, 0.100 },
|
||||
.{ -1.0, -1.0, 0.101 }, .{ 1.0, -1.0, 0.101 }, .{ 0.0, 1.0, 0.101 },
|
||||
.{ -1.0, -1.0, 0.099 }, .{ 1.0, -1.0, 0.099 }, .{ 0.0, 1.0, 0.099 },
|
||||
.{ -1.0, -1.0, 0.102 }, .{ 1.0, -1.0, 0.102 }, .{ 0.0, 1.0, 0.102 },
|
||||
.{ -1.0, -1.0, 0.098 }, .{ 1.0, -1.0, 0.098 }, .{ 0.0, 1.0, 0.098 },
|
||||
.{ -1.0, -1.0, 0.103 }, .{ 1.0, -1.0, 0.103 }, .{ 0.0, 1.0, 0.103 },
|
||||
};
|
||||
|
||||
// Write vertices to hash entry at +8
|
||||
for (0..NVERTS) |vi| {
|
||||
const off = he + 8 + vi * 12;
|
||||
@as(*align(1) f32, @ptrFromInt(off)).* = verts[vi][0];
|
||||
@as(*align(1) f32, @ptrFromInt(off + 4)).* = verts[vi][1];
|
||||
@as(*align(1) f32, @ptrFromInt(off + 8)).* = verts[vi][2];
|
||||
}
|
||||
|
||||
// Triangle count at +0x18A4
|
||||
@as(*align(1) u16, @ptrFromInt(he + 0x18A4)).* = NTRIS;
|
||||
|
||||
// Each triangle uses 3 consecutive vertices: tri N -> verts N*3, N*3+1, N*3+2
|
||||
{
|
||||
var ti: u32 = 0;
|
||||
while (ti < NTRIS) : (ti += 1) {
|
||||
const base: u16 = @intCast(ti * 3);
|
||||
@as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6)).* = base;
|
||||
@as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6 + 2)).* = base + 1;
|
||||
@as(*align(1) u16, @ptrFromInt(he + 0x18A6 + ti * 6 + 4)).* = base + 2;
|
||||
@as(*align(1) u16, @ptrFromInt(he + 0x1FAE + ti * 2)).* = 0;
|
||||
@as(*align(1) u16, @ptrFromInt(he + 0x2206 + ti * 2)).* = @intCast(ti);
|
||||
}
|
||||
}
|
||||
|
||||
// Build "this" struct (needs ~0x54 bytes)
|
||||
var this_buf: [0x60]u8 align(4) = [_]u8{0} ** 0x60;
|
||||
const th = @intFromPtr(&this_buf);
|
||||
|
||||
// Visited array: needs at least NTRIS*2 bytes
|
||||
var visited: [64]u8 = [_]u8{0} ** 64;
|
||||
|
||||
// Result float
|
||||
var result_val: f32 = 0.0;
|
||||
|
||||
// this+0x04 = visited array base
|
||||
@as(*align(1) u32, @ptrFromInt(th + 0x04)).* = @intFromPtr(&visited);
|
||||
// this+0x08, +0x0C = hash params (must match what FindOrCreateHashEntry expects, but
|
||||
// we'll call our function directly bypassing the hash lookup, so these don't matter)
|
||||
// this+0x10 = pointer to result float
|
||||
@as(*align(1) u32, @ptrFromInt(th + 0x10)).* = @intFromPtr(&result_val);
|
||||
// this+0x14 = clamp value
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x14)).* = 100.0;
|
||||
// this+0x18..0x2C = AABB extents (will be sorted by function)
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x18)).* = -3.0; // ax0
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x1C)).* = -3.0; // ay0
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x20)).* = -3.0; // az0
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x24)).* = 3.0; // ax1
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x28)).* = 3.0; // ay1
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x2C)).* = 3.0; // az1
|
||||
// this+0x30..0x3B = ray origin
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x30)).* = 0.0;
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x34)).* = 0.0;
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x38)).* = -10.0;
|
||||
// this+0x3C..0x47 = ray direction
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x3C)).* = 0.0;
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x40)).* = 0.0;
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x44)).* = 1.0;
|
||||
// this+0x48 = scale factor
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x48)).* = 1.0;
|
||||
// this+0x4C = closest-t (large initial value)
|
||||
@as(*align(1) f32, @ptrFromInt(th + 0x4C)).* = 999999.0;
|
||||
// this+0x50 = collision mask
|
||||
@as(*align(1) u16, @ptrFromInt(th + 0x50)).* = 0;
|
||||
|
||||
// Map globals needed by both original and SSE functions
|
||||
_ = mapZeroed(0xCA0000, 0x1000); // g_guard at 0xCA03E4
|
||||
_ = mapZeroed(0xCDE000, 0x1000); // g_render_list at 0xCDE648
|
||||
_ = mapZeroed(0xCE2000, 0x1000); // g_visible_count/list at 0xCE26E0/E8
|
||||
_ = mapZeroed(0xCE6000, 0x1000); // g_render_count at 0xCE66FC
|
||||
|
||||
// Patch FindOrCreateHashEntry (0x693D60) to return our hash_buf:
|
||||
// MOV EAX, <hash_buf_addr> ; B8 xx xx xx xx
|
||||
// RET 0x14 ; C2 14 00
|
||||
const hash_stub = @as([*]u8, @ptrFromInt(0x693D60));
|
||||
hash_stub[0] = 0xB8;
|
||||
@as(*align(1) u32, @ptrFromInt(0x693D61)).* = he;
|
||||
hash_stub[5] = 0xC2;
|
||||
hash_stub[6] = 0x14;
|
||||
hash_stub[7] = 0x00;
|
||||
|
||||
// Set g_guard to non-zero (both functions check this)
|
||||
@as(*align(1) u32, @ptrFromInt(0xCA03E4)).* = 1;
|
||||
|
||||
// Original function at 0x6B88E0 and our SSE version
|
||||
const orig_fn = @as(*const fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x6B88E0));
|
||||
|
||||
// Reset state helper
|
||||
const resetState = struct {
|
||||
fn f(t: u32, v: *[64]u8) void {
|
||||
@as(*align(1) f32, @ptrFromInt(t + 0x4C)).* = 999999.0;
|
||||
@memset(v, 0);
|
||||
@as(*u32, @ptrFromInt(0xCE26E0)).* = 0; // visible count
|
||||
@as(*u32, @ptrFromInt(0xCE66FC)).* = 0; // render count
|
||||
}
|
||||
}.f;
|
||||
|
||||
// Get truth values from original x87 function
|
||||
resetState(th, &visited);
|
||||
_ = @call(.never_tail, orig_fn, .{ th, 0, 0 });
|
||||
const orig_closest = @as(*align(1) f32, @ptrFromInt(th + 0x4C)).*;
|
||||
const orig_result = result_val;
|
||||
const orig_render_count = @as(*u32, @ptrFromInt(0xCE66FC)).*;
|
||||
const orig_visible_count = @as(*u32, @ptrFromInt(0xCE26E0)).*;
|
||||
|
||||
// Run SSE version
|
||||
resetState(th, &visited);
|
||||
_ = @call(.never_tail, performCollisionDetectionSSE, .{ th, 0, 0 });
|
||||
const sse_closest = @as(*align(1) f32, @ptrFromInt(th + 0x4C)).*;
|
||||
const sse_result = result_val;
|
||||
const sse_render_count = @as(*u32, @ptrFromInt(0xCE66FC)).*;
|
||||
const sse_visible_count = @as(*u32, @ptrFromInt(0xCE26E0)).*;
|
||||
|
||||
const t_match = @abs(orig_closest - sse_closest) < 0.01 or (orig_closest > 99999.0 and sse_closest > 99999.0);
|
||||
const r_match = @abs(orig_result - sse_result) < 0.01;
|
||||
const ok = t_match and r_match and orig_render_count == sse_render_count and orig_visible_count == sse_visible_count;
|
||||
|
||||
if (!ok) {
|
||||
print(" MISMATCH detail:\n", .{});
|
||||
print(" closest-t: orig={d:.6} sse={d:.6}\n", .{ orig_closest, sse_closest });
|
||||
print(" result: orig={d:.6} sse={d:.6}\n", .{ orig_result, sse_result });
|
||||
} else {
|
||||
print(" closest-t={d:.4} result={d:.4} hits={d} visible={d}\n", .{
|
||||
sse_closest, sse_result, sse_render_count, sse_visible_count,
|
||||
});
|
||||
}
|
||||
|
||||
// Benchmark: our SSE performCollisionDetectionSSE
|
||||
var best: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
var t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
resetState(th, &visited);
|
||||
_ = @call(.never_tail, performCollisionDetectionSSE, .{ th, 0, 0 });
|
||||
}
|
||||
t0 = rdtsc() - t0;
|
||||
if (t0 < best) best = t0;
|
||||
}
|
||||
|
||||
const per_call = best / ITERS;
|
||||
const per_tri = if (NTRIS > 0) per_call / NTRIS else 0;
|
||||
const status: [*:0]const u8 = if (ok) "OK" else "MISMATCH";
|
||||
print("{s:>30}: {d} cyc/call {d} cyc/tri ({d} tris) {s}\n", .{
|
||||
"performCollisionDet", per_call, per_tri, NTRIS, status,
|
||||
});
|
||||
}
|
||||
|
||||
fn bench_entityUpdate() void {
|
||||
// Map .bss pages for globals the entity update reads/writes
|
||||
_ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510
|
||||
_ = mapZeroed(0xC7B000, 0x2000); // view coeffs at 0xC7BCB0, bounds at 0xC7CB5C-C7CB70
|
||||
_ = mapZeroed(0x866000, 0x4000); // render flags at 0x867960, ptrs at 0x867964/68
|
||||
_ = mapZeroed(0x80A000, 0x1000); // timer threshold at 0x80A1E8
|
||||
_ = mapZeroed(0x86B000, 0x1000); // anim table at 0x86B580
|
||||
_ = mapZeroed(0xC7F000, 0x1000); // anim index at 0xC7F294
|
||||
|
||||
// Set up view coefficients (a,b,c,d) at 0xC7BCB0
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7BCB0)).* = 0.5; // coeff for ent+0x5C
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7BCB4)).* = 0.3; // coeff for ent+0x60
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7BCB8)).* = 0.7; // coeff for ent+0x64
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7BCBC)).* = 1.0; // constant term
|
||||
|
||||
// Delta time
|
||||
@as(*align(1) f32, @ptrFromInt(0xC62510)).* = 0.016; // ~60fps
|
||||
// Timer threshold
|
||||
@as(*align(1) f32, @ptrFromInt(0x80A1E8)).* = 999.0; // high so recycling never triggers
|
||||
// World bounds (set large so bounds check always passes)
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7CB68)).* = 999.0;
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7CB6C)).* = 999.0;
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7CB70)).* = 999.0;
|
||||
// Disable spatial grid registration by making bounds check fail:
|
||||
// Set the lower bounds high so IsPointInsideBounds returns false
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7CB5C)).* = 99999.0;
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7CB60)).* = 99999.0;
|
||||
@as(*align(1) f32, @ptrFromInt(0xC7CB64)).* = 99999.0;
|
||||
|
||||
// Build a synthetic entity struct (~0x900 bytes to cover all accessed fields)
|
||||
var ent_buf: [0x900]u8 align(16) = [_]u8{0} ** 0x900;
|
||||
const ent = @intFromPtr(&ent_buf);
|
||||
|
||||
// Entity position fields for dot product
|
||||
@as(*align(1) f32, @ptrFromInt(ent + 0x5C)).* = 10.0;
|
||||
@as(*align(1) f32, @ptrFromInt(ent + 0x60)).* = 20.0;
|
||||
@as(*align(1) f32, @ptrFromInt(ent + 0x64)).* = 30.0;
|
||||
@as(*align(1) f32, @ptrFromInt(ent + 0x68)).* = 5.0; // depth offset
|
||||
// Timer at ent+0xAC (start at 0)
|
||||
@as(*align(1) f32, @ptrFromInt(ent + 0xAC)).* = 0.0;
|
||||
// No vertex buffers (ent+0x14C = 0), no instances (ent+0xC0 = 0), no chunks
|
||||
// Bounds fields that won't trigger spatial grid
|
||||
@as(*align(1) f32, @ptrFromInt(ent + 0x44)).* = 999.0; // will fail < check
|
||||
|
||||
const orig_fn = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AFAD0));
|
||||
|
||||
// Reset helper
|
||||
const resetEnt = struct {
|
||||
fn f(e: u32) void {
|
||||
@as(*align(1) f32, @ptrFromInt(e + 0xAC)).* = 0.0; // reset timer
|
||||
@as(*align(1) f32, @ptrFromInt(e + 0x78)).* = 0.0; // reset depth
|
||||
}
|
||||
}.f;
|
||||
|
||||
// Correctness: both should compute the same depth
|
||||
resetEnt(ent);
|
||||
@call(.never_tail, orig_fn, .{ent});
|
||||
const orig_depth = @as(*align(1) f32, @ptrFromInt(ent + 0x78)).*;
|
||||
|
||||
resetEnt(ent);
|
||||
@call(.never_tail, updateEntityAndChunksPositions, .{ent});
|
||||
const sse_depth = @as(*align(1) f32, @ptrFromInt(ent + 0x78)).*;
|
||||
|
||||
const ok = @abs(orig_depth - sse_depth) < 0.01;
|
||||
if (!ok) {
|
||||
print(" MISMATCH: orig_depth={d:.4} sse_depth={d:.4}\n", .{ orig_depth, sse_depth });
|
||||
}
|
||||
|
||||
// Benchmark original
|
||||
var orig_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
var t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
resetEnt(ent);
|
||||
@call(.never_tail, orig_fn, .{ent});
|
||||
}
|
||||
t0 = rdtsc() - t0;
|
||||
if (t0 < orig_cyc) orig_cyc = t0;
|
||||
}
|
||||
|
||||
// Benchmark SSE
|
||||
var sse_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
var t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
resetEnt(ent);
|
||||
@call(.never_tail, updateEntityAndChunksPositions, .{ent});
|
||||
}
|
||||
t0 = rdtsc() - t0;
|
||||
if (t0 < sse_cyc) sse_cyc = t0;
|
||||
}
|
||||
|
||||
report("UpdateEntityChunkPos", orig_cyc, sse_cyc, ok);
|
||||
}
|
||||
|
||||
fn bench_computeOutcodes() void {
|
||||
// Generate 150 vertices (typical mesh) spread across an AABB
|
||||
const NVERTS = 150;
|
||||
var verts: [NVERTS * 3]f32 = undefined;
|
||||
var seed: u32 = 0xDEADBEEF;
|
||||
for (0..NVERTS * 3) |j| {
|
||||
seed = seed *% 1103515245 +% 12345;
|
||||
// Range roughly -10..+10
|
||||
verts[j] = @as(f32, @floatFromInt(@as(i32, @bitCast(seed >> 16)) >> 16)) * 0.0003;
|
||||
}
|
||||
// AABB bounds: minX,minY,minZ,maxX,maxY,maxZ
|
||||
var bounds = [6]f32{ -2.0, -2.0, -2.0, 2.0, 2.0, 2.0 };
|
||||
|
||||
// Original x87 version: extract the outcode loop from PerformSpatialCulling.
|
||||
// The original does 6 FCOMP+FNSTSW+TEST sequences per vertex.
|
||||
// We'll inline a scalar reference implementation for the original.
|
||||
var out_orig: [NVERTS]u8 = undefined;
|
||||
var out_sse: [NVERTS]u8 = undefined;
|
||||
|
||||
// Scalar reference (matches original x87 logic)
|
||||
for (0..NVERTS) |i| {
|
||||
const vx = verts[i * 3];
|
||||
const vy = verts[i * 3 + 1];
|
||||
const vz = verts[i * 3 + 2];
|
||||
var code: u8 = 0;
|
||||
if (vx < bounds[0]) code |= 0x20;
|
||||
if (vx >= bounds[3]) code |= 0x10;
|
||||
if (vy < bounds[1]) code |= 0x08;
|
||||
if (vy >= bounds[4]) code |= 0x04;
|
||||
if (vz < bounds[2]) code |= 0x02;
|
||||
if (vz >= bounds[5]) code |= 0x01;
|
||||
out_orig[i] = code;
|
||||
}
|
||||
|
||||
// SSE version
|
||||
benchComputeOutcodes(a(&verts), a(&bounds), a(&out_sse), NVERTS);
|
||||
|
||||
// Verify correctness
|
||||
var ok = true;
|
||||
for (0..NVERTS) |i| {
|
||||
if (out_orig[i] != out_sse[i]) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Benchmark: scalar reference
|
||||
const scalar_fn = struct {
|
||||
fn run(v: *[NVERTS * 3]f32, b: *[6]f32, out: *[NVERTS]u8) void {
|
||||
for (0..NVERTS) |i| {
|
||||
const vx = v[i * 3];
|
||||
const vy = v[i * 3 + 1];
|
||||
const vz = v[i * 3 + 2];
|
||||
var code: u8 = 0;
|
||||
if (vx < b[0]) code |= 0x20;
|
||||
if (vx >= b[3]) code |= 0x10;
|
||||
if (vy < b[1]) code |= 0x08;
|
||||
if (vy >= b[4]) code |= 0x04;
|
||||
if (vz < b[2]) code |= 0x02;
|
||||
if (vz >= b[5]) code |= 0x01;
|
||||
out[i] = code;
|
||||
}
|
||||
}
|
||||
}.run;
|
||||
|
||||
var orig_cyc: u64 = std.math.maxInt(u64);
|
||||
var sse_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
var t = rdtsc();
|
||||
for (0..ITERS) |_| scalar_fn(&verts, &bounds, &out_orig);
|
||||
t = rdtsc() - t;
|
||||
if (t < orig_cyc) orig_cyc = t;
|
||||
}
|
||||
for (0..5) |_| {
|
||||
var t = rdtsc();
|
||||
for (0..ITERS) |_| benchComputeOutcodes(a(&verts), a(&bounds), a(&out_sse), NVERTS);
|
||||
t = rdtsc() - t;
|
||||
if (t < sse_cyc) sse_cyc = t;
|
||||
}
|
||||
report("computeOutcodes(150v)", orig_cyc, sse_cyc, ok);
|
||||
}
|
||||
|
||||
@@ -29,6 +29,7 @@ pub const module_name: [*:0]const u8 = "clickthrough";
|
||||
|
||||
const ADDR_WorldIntersectionTest: usize = 0x480DF0;
|
||||
const ADDR_CanTargetEntity: usize = 0x480610;
|
||||
const ADDR_CheckObjectTypePermissions: usize = 0x480780;
|
||||
|
||||
// =============================================================================
|
||||
// HitTestResult layout
|
||||
@@ -54,20 +55,13 @@ const FLAG_CUSTOM_MASK: u32 = FLAG_LOOT_ONLY | FLAG_GO_ONLY | FLAG_NPC_ONLY;
|
||||
var g_mutex: ?*anyopaque = null;
|
||||
var g_is_hook_owner: bool = false;
|
||||
var log: logging.Logger = .{};
|
||||
var go_log_count: u32 = 0;
|
||||
|
||||
// CanTargetEntity: CanTargetEntity(void *obj, uint permissionFlags) -> undefined*
|
||||
// Returns non-NULL to include object, NULL to exclude.
|
||||
// __cdecl-ish but called with obj as first stack arg from CheckObjectTypePermissions.
|
||||
// Assembly: PUSH permFlags; PUSH objPtr; CALL CanTargetEntity
|
||||
// Actually looking at the call site it passes obj in register and flags on stack.
|
||||
// Let me verify from the CheckObjectTypePermissions assembly.
|
||||
// From decompile: puVar4 = CanTargetEntity(pvVar3, permissionFlags);
|
||||
// pvVar3 is the resolved object pointer. permissionFlags is the raycast flags.
|
||||
// The function signature from Ghidra: CanTargetEntity(void *param_1, uint param_2)
|
||||
// Not thiscall/fastcall - it's a regular call with two stack args.
|
||||
|
||||
const CanTargetFn = fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32;
|
||||
var cte_hook: hook.Detour(CanTargetFn) = .{};
|
||||
// CheckObjectTypePermissions (0x480780) -- verified from assembly:
|
||||
// __thiscall: ECX=context (saved to EDI, passed to CanTargetEntity)
|
||||
// Stack: objectData [EBP+8], permFlags [EBP+C]. RET 0x8.
|
||||
const CheckObjTypeFn = fn (u32, u32, u32) callconv(hook.cc.thiscall) u32;
|
||||
var cotp_hook: hook.Detour(CheckObjTypeFn) = .{};
|
||||
|
||||
const WorldIntersectFn = fn (u32, u32, u32, u32, u32) callconv(hook.cc.thiscall) u32;
|
||||
var wit_hook: hook.Detour(WorldIntersectFn) = .{};
|
||||
@@ -78,10 +72,10 @@ var wit_hook: hook.Detour(WorldIntersectFn) = .{};
|
||||
// When custom flag bits are set, exclude objects that don't match the pass.
|
||||
// =============================================================================
|
||||
|
||||
fn canTargetDetour(obj: u32, perm_flags: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
|
||||
fn checkObjTypeDetour(ctx: u32, obj_data: u32, perm_flags: u32) callconv(hook.cc.thiscall) u32 {
|
||||
// Strip custom bits before passing to original
|
||||
const clean_flags = perm_flags & ~FLAG_CUSTOM_MASK;
|
||||
const original = cte_hook.callOriginal(.{ obj, clean_flags });
|
||||
const original = cotp_hook.callOriginal(.{ ctx, obj_data, clean_flags });
|
||||
|
||||
// If original says exclude, respect that
|
||||
if (original == 0) return 0;
|
||||
@@ -89,27 +83,37 @@ fn canTargetDetour(obj: u32, perm_flags: u32) callconv(.{ .x86_stdcall = .{} })
|
||||
// No custom filtering active - pass through
|
||||
if ((perm_flags & FLAG_CUSTOM_MASK) == 0) return original;
|
||||
|
||||
// Custom pass filtering
|
||||
// Resolve the object pointer from obj_data via ClntObjMgrObjectPtr.
|
||||
// __fastcall(ECX=typeMask, EDX=debugStr, stack: guid_lo, guid_hi, debugCode)
|
||||
// RET 0xC. See nampower ClntObjMgrObjectPtrT typedef.
|
||||
const obj = hook.call(
|
||||
fn (u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) u32,
|
||||
0x468460, // ClntObjMgrObjectPtr
|
||||
.{ 1, 0, hook.readMem(u32, obj_data + 0x18), hook.readMem(u32, obj_data + 0x1C), 0 },
|
||||
);
|
||||
if (obj == 0) return original;
|
||||
|
||||
const desc = wow.getDescriptor(obj);
|
||||
if (!wow.isValidPtr(desc)) return original;
|
||||
const type_mask = hook.readMem(u32, desc + 0x08);
|
||||
|
||||
if ((perm_flags & FLAG_LOOT_ONLY) != 0) {
|
||||
// Only lootable corpses pass
|
||||
if (type_mask != 0x09) return 0; // units only
|
||||
if (!wow.isLootable(obj)) return 0;
|
||||
return original;
|
||||
}
|
||||
|
||||
if ((perm_flags & FLAG_GO_ONLY) != 0) {
|
||||
// Only interactable GOs pass
|
||||
const desc = wow.getDescriptor(obj);
|
||||
if (!wow.isValidPtr(desc)) return 0;
|
||||
const type_mask = hook.readMem(u32, desc + 0x08);
|
||||
if (type_mask != 0x21) return 0; // not a GO
|
||||
// Check interactability
|
||||
if (type_mask != 0x21) return 0; // GOs only
|
||||
const go_type = hook.readMem(u32, desc + offsets.DESC_GO_TYPE);
|
||||
if (go_type == 9 or go_type == 7) return 0; // TEXT, CHAIR
|
||||
if (hook.call(fn (u32) callconv(hook.cc.fastcall) u8, offsets.FN_CALL_SPELL_CAST_HANDLER, .{obj}) == 0)
|
||||
return 0;
|
||||
return original;
|
||||
}
|
||||
|
||||
if ((perm_flags & FLAG_NPC_ONLY) != 0) {
|
||||
// Only units with NPC interaction flags pass
|
||||
if (type_mask != 0x09 and type_mask != 0x19) return 0; // units/players only
|
||||
if (wow.getNpcFlags(obj) == 0) return 0;
|
||||
return original;
|
||||
}
|
||||
@@ -168,8 +172,8 @@ pub fn installHooks() void {
|
||||
g_is_hook_owner = result.is_owner;
|
||||
if (!g_is_hook_owner) return;
|
||||
|
||||
log = logging.Logger.open(module_name, .console);
|
||||
_ = cte_hook.attach(ADDR_CanTargetEntity, &canTargetDetour);
|
||||
log = logging.Logger.open(module_name, .both);
|
||||
_ = cotp_hook.attach(ADDR_CheckObjectTypePermissions, &checkObjTypeDetour);
|
||||
_ = wit_hook.attach(ADDR_WorldIntersectionTest, &worldIntersectDetour);
|
||||
log.print("clickthrough: cascade raycast active\n");
|
||||
}
|
||||
@@ -177,7 +181,7 @@ pub fn installHooks() void {
|
||||
pub fn removeHooks() void {
|
||||
if (g_is_hook_owner) {
|
||||
wit_hook.detach();
|
||||
cte_hook.detach();
|
||||
cotp_hook.detach();
|
||||
log.close();
|
||||
mod_mutex.release(&g_mutex);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,358 @@
|
||||
//! SSE-optimized spatial culling — compiled ReleaseFast even in Debug builds.
|
||||
//!
|
||||
//! Reimplements PerformSpatialCulling (0x6B8C60) and performCollisionDetection
|
||||
//! (0x6B88E0). Both are leaf functions in the KD-tree traversal that compute
|
||||
//! per-vertex 6-bit outcodes against an AABB, then iterate triangles.
|
||||
//!
|
||||
//! The hot path is the vertex outcode loop: 6 float comparisons per vertex
|
||||
//! (~150 vertices typical). SSE compares all 3 axes in parallel.
|
||||
|
||||
// =============================================================================
|
||||
// Benchmark-only: pure outcode computation, no game function dependencies.
|
||||
// Called from src/bench/main.zig with synthetic vertex data.
|
||||
// =============================================================================
|
||||
|
||||
/// Compute outcodes for `count` vertices at `verts_ptr` (stride 12 bytes = 3 floats)
|
||||
/// against AABB at `bounds_ptr` (6 floats: minX, minY, minZ, maxX, maxY, maxZ).
|
||||
/// Writes results to `out_ptr` (1 byte per vertex).
|
||||
export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void {
|
||||
if (count == 0) return;
|
||||
const v_min: V4 = .{ readF32(bounds_ptr), readF32(bounds_ptr + 4), readF32(bounds_ptr + 8), 0 };
|
||||
const v_max: V4 = .{ readF32(bounds_ptr + 12), readF32(bounds_ptr + 16), readF32(bounds_ptr + 20), 0 };
|
||||
const below_w: @Vector(4, u32) = .{ 0x20, 0x08, 0x02, 0x00 };
|
||||
const above_w: @Vector(4, u32) = .{ 0x10, 0x04, 0x01, 0x00 };
|
||||
|
||||
const out: [*]u8 = @ptrFromInt(out_ptr);
|
||||
var vp: [*]const f32 = @ptrFromInt(verts_ptr);
|
||||
var i: u32 = 0;
|
||||
while (i < count) : (i += 1) {
|
||||
const v: V4 = @as(*align(1) const V4, @ptrCast(vp)).*;
|
||||
const zero: @Vector(4, u32) = @splat(0);
|
||||
const combined = @select(u32, v < v_min, below_w, zero) | @select(u32, v >= v_max, above_w, zero);
|
||||
out[i] = @truncate(combined[0] | combined[1] | combined[2] | combined[3]);
|
||||
vp += 3;
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Game hook exports
|
||||
// =============================================================================
|
||||
|
||||
// External game functions (resolved at link time via absolute address)
|
||||
// FindOrCreateHashEntry: thiscall(ECX=hashTable from global 0xCA03E4, stack: 5 args) RET 0x14
|
||||
const FindOrCreateHashEntry = @as(*const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32, @ptrFromInt(0x693D60));
|
||||
// Global state used by the game's rendering pipeline
|
||||
const g_visible_count: *u32 = @ptrFromInt(0xCE26E0); // PTR_00ce26e0
|
||||
const g_visible_list: [*]u16 = @ptrFromInt(0xCE26E8); // DAT_00ce26e8
|
||||
const g_render_count: *u32 = @ptrFromInt(0xCE66FC); // PTR_00ce66fc
|
||||
const g_render_list: [*]u16 = @ptrFromInt(0xCDE648); // DAT_00cde648
|
||||
const g_guard: *const u32 = @ptrFromInt(0xCA03E4); // PTR_00ca03e4
|
||||
|
||||
const V4 = @Vector(4, f32);
|
||||
|
||||
fn readU32(addr: u32) u32 {
|
||||
return @as(*align(1) const u32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
fn readU16(addr: u32) u16 {
|
||||
return @as(*align(1) const u16, @ptrFromInt(addr)).*;
|
||||
}
|
||||
fn readF32(addr: u32) f32 {
|
||||
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
/// Compute outcodes for all vertices in the mesh using SSE.
|
||||
///
|
||||
/// Per-vertex 6-bit outcode against AABB. Two SIMD compares (v < min, v >= max)
|
||||
/// produce all 6 bits from movemask results. Processes xyz in parallel.
|
||||
///
|
||||
/// Bit layout: 0x20=below_minX, 0x10=above_maxX, 0x08=below_minY,
|
||||
/// 0x04=above_maxY, 0x02=below_minZ, 0x01=above_maxZ
|
||||
fn computeAllOutcodes(
|
||||
hash_entry: u32,
|
||||
min_x: f32,
|
||||
max_x: f32,
|
||||
min_y: f32,
|
||||
max_y: f32,
|
||||
min_z: f32,
|
||||
max_z: f32,
|
||||
cull_flags: *[452]u8,
|
||||
) u32 {
|
||||
const vert_count: u32 = readU16(hash_entry + 6);
|
||||
if (vert_count == 0) return 0;
|
||||
|
||||
const v_min: V4 = .{ min_x, min_y, min_z, 0 };
|
||||
const v_max: V4 = .{ max_x, max_y, max_z, 0 };
|
||||
|
||||
// Bit weights for branchless outcode: below gives 0x20/0x08/0x02, above gives 0x10/0x04/0x01
|
||||
const below_w: @Vector(4, u32) = .{ 0x20, 0x08, 0x02, 0x00 };
|
||||
const above_w: @Vector(4, u32) = .{ 0x10, 0x04, 0x01, 0x00 };
|
||||
|
||||
var vert_ptr: [*]const f32 = @ptrFromInt(hash_entry + 8);
|
||||
var i: u32 = 0;
|
||||
while (i < vert_count) : (i += 1) {
|
||||
// Single 16-byte unaligned load. 4th float is junk from next vertex
|
||||
// but v_min[3]=0, v_max[3]=0, so comparisons on lane 3 produce
|
||||
// below=false (0>=0), above=true (0>=0) -- weight is 0x00 so harmless.
|
||||
const v: V4 = @as(*align(1) const V4, @ptrCast(vert_ptr)).*;
|
||||
|
||||
// Bool vectors -> u32 vectors (0 or 0xFFFFFFFF), AND with weights, horizontal OR
|
||||
const zero: @Vector(4, u32) = @splat(0);
|
||||
const below_masked = @select(u32, v < v_min, below_w, zero);
|
||||
const above_masked = @select(u32, v >= v_max, above_w, zero);
|
||||
const combined = below_masked | above_masked;
|
||||
|
||||
// Horizontal OR of 4 lanes -> single outcode byte
|
||||
cull_flags[i] = @truncate(combined[0] | combined[1] | combined[2] | combined[3]);
|
||||
|
||||
vert_ptr += 3; // stride 12 bytes = 3 floats
|
||||
}
|
||||
return vert_count;
|
||||
}
|
||||
|
||||
/// PerformSpatialCulling (0x6B8C60)
|
||||
/// __thiscall(this, keyData, keySize) -> u32. RET 0x8.
|
||||
///
|
||||
/// Finds mesh data via hash, computes vertex outcodes against AABB from this+0x10,
|
||||
/// then iterates triangles: filters by visibility mask, trivial-rejects by outcode AND,
|
||||
/// adds survivors to global visible/render lists.
|
||||
export fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
if (g_guard.* == 0) return 0;
|
||||
|
||||
const hash_table = g_guard.*;
|
||||
const hash_entry = @call(.never_tail, FindOrCreateHashEntry, .{
|
||||
hash_table, key_data, key_size,
|
||||
readU32(this + 4), readU32(this + 8), readU32(this + 0xC),
|
||||
});
|
||||
if (hash_entry == 0) return 0;
|
||||
|
||||
// Load AABB from *(this+0x10) -- 6 floats: minX, minY, minZ, maxX, maxY, maxZ
|
||||
const bounds_ptr = readU32(this + 0x10);
|
||||
const min_x = readF32(bounds_ptr);
|
||||
const min_y = readF32(bounds_ptr + 4);
|
||||
const min_z = readF32(bounds_ptr + 8);
|
||||
const max_x = readF32(bounds_ptr + 12);
|
||||
const max_y = readF32(bounds_ptr + 16);
|
||||
const max_z = readF32(bounds_ptr + 20);
|
||||
|
||||
// Phase 1: Compute per-vertex outcodes
|
||||
var cull_flags: [452]u8 = undefined;
|
||||
_ = computeAllOutcodes(hash_entry, min_x, max_x, min_y, max_y, min_z, max_z, &cull_flags);
|
||||
|
||||
// Phase 2: Iterate triangles
|
||||
const tri_count: u32 = @as(u32, readU16(hash_entry + 0x18A4));
|
||||
const filter_mask = readU16(this + 0x14);
|
||||
const visited_base = readU32(this + 4);
|
||||
|
||||
var ti: u32 = 0;
|
||||
while (ti < tri_count) : (ti += 1) {
|
||||
// Visibility mask filter
|
||||
const vis_flags = readU16(hash_entry + 0x1FAE + ti * 2);
|
||||
if ((vis_flags & filter_mask) != 0) continue;
|
||||
|
||||
// Per-vertex visited filter
|
||||
const tri_base_idx = readU16(hash_entry + 0x2206 + ti * 2);
|
||||
const visited_byte = @as(*u8, @ptrFromInt(visited_base + @as(u32, tri_base_idx) * 2));
|
||||
if ((visited_byte.* & @as(u8, @truncate(filter_mask))) != 0) continue;
|
||||
|
||||
// Check global visible list capacity
|
||||
if (g_visible_count.* >= 0x2000) {
|
||||
const flags_ptr = readU32(this);
|
||||
if (flags_ptr != 0) {
|
||||
const p: *u32 = @ptrFromInt(flags_ptr);
|
||||
p.* |= 1;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
// Add to visible list
|
||||
g_visible_list[g_visible_count.*] = tri_base_idx;
|
||||
g_visible_count.* += 1;
|
||||
visited_byte.* |= 0x80;
|
||||
|
||||
// Frustum test: AND of 3 vertex outcodes. If any bit shared, fully outside.
|
||||
const idx0 = readU16(hash_entry + 0x18A6 + ti * 6);
|
||||
const idx1 = readU16(hash_entry + 0x18A8 + ti * 6);
|
||||
const idx2 = readU16(hash_entry + 0x18AA + ti * 6);
|
||||
// Note: decompiler shows idx offsets as 0x18A6, +0xC54*2, +0x18AA
|
||||
// which is 0x18A6 (idx0), 0x18A8 (idx1), 0x18AA (idx2) -- stride 6 = 3 u16 per tri
|
||||
|
||||
if ((cull_flags[idx0] & cull_flags[idx1] & cull_flags[idx2] & 0x3F) == 0) {
|
||||
g_render_list[g_render_count.*] = tri_base_idx;
|
||||
g_render_count.* += 1;
|
||||
}
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// SSE vector helpers for Moller-Trumbore
|
||||
// =============================================================================
|
||||
|
||||
inline fn loadVec3(addr: u32) V4 {
|
||||
return .{ readF32(addr), readF32(addr + 4), readF32(addr + 8), 0 };
|
||||
}
|
||||
|
||||
inline fn cross(a: V4, b: V4) V4 {
|
||||
const Mask = @Vector(4, i32);
|
||||
const a_yzx: V4 = @shuffle(f32, a, undefined, Mask{ 1, 2, 0, 3 });
|
||||
const a_zxy: V4 = @shuffle(f32, a, undefined, Mask{ 2, 0, 1, 3 });
|
||||
const b_yzx: V4 = @shuffle(f32, b, undefined, Mask{ 1, 2, 0, 3 });
|
||||
const b_zxy: V4 = @shuffle(f32, b, undefined, Mask{ 2, 0, 1, 3 });
|
||||
return a_yzx * b_zxy - a_zxy * b_yzx;
|
||||
}
|
||||
|
||||
inline fn dot3(a: V4, b: V4) f32 {
|
||||
const p = a * b;
|
||||
return p[0] + p[1] + p[2];
|
||||
}
|
||||
|
||||
/// performCollisionDetection (0x6B88E0)
|
||||
/// __thiscall(this, keyData, keySize) -> u32. RET 0x8.
|
||||
///
|
||||
/// Fully inlined SSE rewrite. No external calls except FindOrCreateHashEntry.
|
||||
/// Moller-Trumbore ray-triangle intersection is inlined with SSE cross/dot,
|
||||
/// eliminating 4 SetVector3 calls and the ray_tri function call per triangle.
|
||||
export fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
if (g_guard.* == 0) return 0;
|
||||
|
||||
const hash_table = g_guard.*;
|
||||
const hash_entry = @call(.never_tail, FindOrCreateHashEntry, .{
|
||||
hash_table, key_data, key_size,
|
||||
readU32(this + 4), readU32(this + 8), readU32(this + 0xC),
|
||||
});
|
||||
if (hash_entry == 0) return 0;
|
||||
|
||||
// Load and sort AABB extents from this+0x18..0x2C
|
||||
var ax0 = readF32(this + 0x18);
|
||||
var ax1 = readF32(this + 0x24);
|
||||
if (ax1 < ax0) {
|
||||
const tmp = ax0;
|
||||
ax0 = ax1;
|
||||
ax1 = tmp;
|
||||
}
|
||||
var ay0 = readF32(this + 0x1C);
|
||||
var ay1 = readF32(this + 0x28);
|
||||
if (ay1 < ay0) {
|
||||
const tmp = ay0;
|
||||
ay0 = ay1;
|
||||
ay1 = tmp;
|
||||
}
|
||||
var az0 = readF32(this + 0x20);
|
||||
var az1 = readF32(this + 0x2C);
|
||||
if (az1 < az0) {
|
||||
const tmp = az0;
|
||||
az0 = az1;
|
||||
az1 = tmp;
|
||||
}
|
||||
|
||||
// Phase 1: Compute per-vertex outcodes
|
||||
var cull_flags: [452]u8 = undefined;
|
||||
_ = computeAllOutcodes(hash_entry, ax0, ax1, ay0, ay1, az0, az1, &cull_flags);
|
||||
|
||||
// Phase 2: Iterate triangles with inline ray-tri test
|
||||
const tri_count: u32 = @as(u32, readU16(hash_entry + 0x18A4));
|
||||
const collision_mask = readU16(this + 0x50);
|
||||
const visited_base = readU32(this + 4);
|
||||
const vert_pool = hash_entry + 8;
|
||||
|
||||
// Ray: origin at this+0x00 (position), direction at this+0x0C (3 floats)
|
||||
// Original uses param_1 = ESI which points to a 6-float struct:
|
||||
// [0..2] = ray origin, [3..5] = ray direction
|
||||
// The call site passes this+0x30 as the ray struct
|
||||
const ray_origin = loadVec3(this + 0x30);
|
||||
const ray_dir = loadVec3(this + 0x3C);
|
||||
|
||||
// Epsilon for barycentric bounds: original uses +/- param_6 (0.002)
|
||||
const eps: f32 = 0.002;
|
||||
const neg_eps: f32 = -eps;
|
||||
const one_plus_eps: f32 = 1.0 + eps;
|
||||
|
||||
var ti: u32 = 0;
|
||||
while (ti < tri_count) : (ti += 1) {
|
||||
const vis_flags = readU16(hash_entry + 0x1FAE + ti * 2);
|
||||
if ((vis_flags & collision_mask) != 0) continue;
|
||||
|
||||
const tri_base_idx = readU16(hash_entry + 0x2206 + ti * 2);
|
||||
const visited_addr = visited_base + @as(u32, tri_base_idx) * 2;
|
||||
const visited_byte = @as(*u8, @ptrFromInt(visited_addr));
|
||||
if ((visited_byte.* & @as(u8, @truncate(collision_mask))) != 0) continue;
|
||||
|
||||
// Add to visible list and mark visited
|
||||
g_visible_list[g_visible_count.*] = tri_base_idx;
|
||||
g_visible_count.* += 1;
|
||||
visited_byte.* |= 0x80;
|
||||
|
||||
// Frustum outcode test
|
||||
const idx_base = hash_entry + 0x18A6 + ti * 6;
|
||||
const vi0: u32 = readU16(idx_base);
|
||||
const vi1: u32 = readU16(idx_base + 2);
|
||||
const vi2: u32 = readU16(idx_base + 4);
|
||||
|
||||
if ((cull_flags[vi0] & cull_flags[vi1] & cull_flags[vi2] & 0x3F) != 0) continue;
|
||||
|
||||
// =====================================================================
|
||||
// Inline Moller-Trumbore ray-triangle intersection (SSE)
|
||||
// Deferred divide: compare u_raw and v_raw against det-scaled bounds
|
||||
// to avoid the 1/det divide on the reject path.
|
||||
// =====================================================================
|
||||
|
||||
const v0 = loadVec3(vert_pool + vi0 * 12);
|
||||
const v1 = loadVec3(vert_pool + vi1 * 12);
|
||||
const v2 = loadVec3(vert_pool + vi2 * 12);
|
||||
|
||||
const edge1 = v1 - v0;
|
||||
const edge2 = v2 - v0;
|
||||
const pvec = cross(ray_dir, edge2);
|
||||
const det = dot3(edge1, pvec);
|
||||
|
||||
if (det <= 1e-7 and det >= -1e-7) continue;
|
||||
|
||||
const tvec = ray_origin - v0;
|
||||
|
||||
// u_raw = dot(tvec, pvec) -- NOT multiplied by inv_det yet
|
||||
const u_raw = dot3(tvec, pvec);
|
||||
|
||||
// Compare u_raw against det-scaled epsilon bounds.
|
||||
// If det > 0: u = u_raw/det, so u < -eps iff u_raw < -eps*det, u > 1+eps iff u_raw > (1+eps)*det
|
||||
// If det < 0: division flips sign, so u < -eps iff u_raw > -eps*det (which is positive)
|
||||
// Trick: multiply both sides by sign(det) to normalize.
|
||||
// Or equivalently: if det>0 check u_raw in [det*neg_eps, det*one_plus_eps]
|
||||
// if det<0 check u_raw in [det*one_plus_eps, det*neg_eps]
|
||||
const det_neg_eps = det * neg_eps;
|
||||
const det_one_plus = det * one_plus_eps;
|
||||
if (det > 0) {
|
||||
if (u_raw < det_neg_eps or u_raw > det_one_plus) continue;
|
||||
} else {
|
||||
if (u_raw > det_neg_eps or u_raw < det_one_plus) continue;
|
||||
}
|
||||
|
||||
const qvec = cross(tvec, edge1);
|
||||
const v_raw = dot3(ray_dir, qvec);
|
||||
|
||||
// Same sign-aware bounds check for v
|
||||
if (det > 0) {
|
||||
if (v_raw < det_neg_eps or (u_raw + v_raw) > det_one_plus) continue;
|
||||
} else {
|
||||
if (v_raw > det_neg_eps or (u_raw + v_raw) < det_one_plus) continue;
|
||||
}
|
||||
|
||||
// Only divide for confirmed hits
|
||||
const t = dot3(edge2, qvec) / det;
|
||||
|
||||
if (t >= 0.0 and t < readF32(this + 0x4C)) {
|
||||
// Update closest hit
|
||||
@as(*align(1) f32, @ptrFromInt(this + 0x4C)).* = t;
|
||||
g_render_list[0] = tri_base_idx;
|
||||
g_render_count.* = 1;
|
||||
|
||||
// Write scaled distance, clamped to max
|
||||
const result_ptr: *align(1) f32 = @ptrFromInt(readU32(this + 0x10));
|
||||
const scaled = t * readF32(this + 0x48);
|
||||
const clamp = readF32(this + 0x14);
|
||||
result_ptr.* = if (scaled <= clamp) scaled else clamp;
|
||||
}
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
//! SSE-optimized entity update functions -- compiled ReleaseFast.
|
||||
//!
|
||||
//! Reimplements UpdateEntityAndChunksPositions (0x6AFAD0) and
|
||||
//! updateEntitiesInBounds (0x6C1F70).
|
||||
//!
|
||||
//! Main optimization: SSE dot product for view-depth computation,
|
||||
//! tighter control flow, reduced function call overhead.
|
||||
|
||||
const V4 = @Vector(4, f32);
|
||||
|
||||
fn readU32(addr: u32) u32 {
|
||||
return @as(*align(1) const u32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
fn readI32(addr: u32) i32 {
|
||||
return @as(*align(1) const i32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
fn readF32(addr: u32) f32 {
|
||||
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
fn writeU32(addr: u32, val: u32) void {
|
||||
@as(*align(1) u32, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
fn writeF32(addr: u32, val: f32) void {
|
||||
@as(*align(1) f32, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Game function declarations (resolved at link time via absolute address)
|
||||
// =========================================================================
|
||||
|
||||
// recycleVertexBuffer: fastcall(ECX=bufPtr, EDX=sizePtr), RET
|
||||
const recycleVertexBuffer = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AE9A0));
|
||||
// check_instances_active: fastcall(ECX=instanceMgr) -> ptr, RET
|
||||
const check_instances_active = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) u32, @ptrFromInt(0x6B2900));
|
||||
// store_all_instance_buffers: fastcall(ECX=instanceMgr), RET
|
||||
const store_all_instance_buffers = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6B28E0));
|
||||
// return_object_to_pool: fastcall(ECX=instanceMgr), RET
|
||||
const return_object_to_pool = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6B2030));
|
||||
// ReturnChunkBuffers: fastcall(ECX=bufPtr, EDX=sizePtr), RET
|
||||
const ReturnChunkBuffers = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x68CD50));
|
||||
// IsPointInsideBounds: fastcall(ECX=point, EDX=bounds) -> u32, RET
|
||||
const IsPointInsideBounds = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) u32, @ptrFromInt(0x699330));
|
||||
// AddObjectToSpatialList: fastcall(ECX=entityPtr, EDX=posPtr), RET
|
||||
const AddObjectToSpatialList = @as(*const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6818B0));
|
||||
// AddToSpatialGrid: fastcall(ECX=objPtr), RET
|
||||
const AddToSpatialGrid = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6816F0));
|
||||
// CopyChunkBounds: thiscall(ECX=chunk, stack=outBounds), RET 0x4
|
||||
const CopyChunkBounds = @as(*const fn (u32, u32) callconv(.{ .x86_thiscall = .{} }) void, @ptrFromInt(0x68DF40));
|
||||
// AddToLayeredSpatialGrid: fastcall(ECX=chunk, EDX=idx, stack=posPtr), RET 0x4
|
||||
const AddToLayeredSpatialGrid = @as(*const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x681970));
|
||||
|
||||
// ComplexMemoryCleanupAndRelease: fastcall(ECX=memObj)
|
||||
const ComplexMemoryCleanupAndRelease = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6A0510));
|
||||
// destroySecondaryGameObject: fastcall(ECX=entityPtr)
|
||||
const destroySecondaryGameObject = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6A6A00));
|
||||
|
||||
// Game globals
|
||||
const g_viewCoeffs: u32 = 0xC7BCB0; // 4 floats: a, b, c, d for dot product
|
||||
const g_renderFlags: *const u8 = @ptrFromInt(0x867960 + 8); // actually at different offset
|
||||
const g_deltaTime: *const u32 = @ptrFromInt(0xC62510); // PTR_00c62510
|
||||
const g_timerThreshold: *const f32 = @ptrFromInt(0x80A1E8); // _DAT_0080a1e8
|
||||
|
||||
// =========================================================================
|
||||
// UpdateEntityAndChunksPositions (0x6AFAD0)
|
||||
// __fastcall(ECX=entityPtr), RET
|
||||
// =========================================================================
|
||||
export fn updateEntityAndChunksPositions(ent: u32) callconv(.{ .x86_fastcall = .{} }) void {
|
||||
// Dot product: depth = a*x + b*y + c*z + d - offset
|
||||
const depth = readF32(g_viewCoeffs) * readF32(ent + 0x5C) +
|
||||
readF32(g_viewCoeffs + 4) * readF32(ent + 0x60) +
|
||||
readF32(g_viewCoeffs + 8) * readF32(ent + 0x64) +
|
||||
readF32(g_viewCoeffs + 12) - readF32(ent + 0x68);
|
||||
writeF32(ent + 0x78, depth);
|
||||
|
||||
// Render distance flag
|
||||
writeU32(ent + 0xB8, readU32(0x867964));
|
||||
if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 4) != 0 and readF32(0x867960) < depth) {
|
||||
writeU32(ent + 0xB8, readU32(0x867968));
|
||||
}
|
||||
|
||||
// Timer accumulation
|
||||
const dt_bits = g_deltaTime.*;
|
||||
const dt: f32 = @bitCast(dt_bits);
|
||||
const timer = readF32(ent + 0xAC) + dt;
|
||||
writeF32(ent + 0xAC, timer);
|
||||
const threshold = g_timerThreshold.*;
|
||||
|
||||
// Vertex buffer recycling
|
||||
if (threshold < timer and readU32(ent + 0x14C) != 0) {
|
||||
@call(.never_tail, recycleVertexBuffer, .{ ent + 0x14C, ent + 0x150 });
|
||||
}
|
||||
|
||||
// Instance buffer management
|
||||
const inst_mgr = readU32(ent + 0xC0);
|
||||
if (inst_mgr != 0) {
|
||||
if (1.0 < readF32(ent + 0xAC)) {
|
||||
const active = @call(.never_tail, check_instances_active, .{inst_mgr});
|
||||
if (active != 0) {
|
||||
@call(.never_tail, store_all_instance_buffers, .{inst_mgr});
|
||||
}
|
||||
}
|
||||
if (threshold < readF32(ent + 0xAC)) {
|
||||
@call(.never_tail, return_object_to_pool, .{inst_mgr});
|
||||
writeU32(ent + 0xC0, 0);
|
||||
}
|
||||
}
|
||||
|
||||
// Chunk timer loop (4 chunks at ent+0x118, stride 4)
|
||||
inline for (0..4) |ci| {
|
||||
const chunk = readU32(ent + 0x118 + ci * 4);
|
||||
if (chunk != 0) {
|
||||
const chunk_timer = readF32(chunk + 0x30) + dt;
|
||||
writeF32(chunk + 0x30, chunk_timer);
|
||||
if (readU32(chunk + 0x400) != 0 and threshold < chunk_timer) {
|
||||
@call(.never_tail, ReturnChunkBuffers, .{ chunk + 0x400, chunk + 0x404 });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Bounds check and spatial grid registration
|
||||
const bx = readF32(ent + 0x44);
|
||||
const by = readF32(ent + 0x48);
|
||||
const bz = readF32(ent + 0x4C);
|
||||
if (bx <= readF32(0xC7CB68) and by <= readF32(0xC7CB6C) and bz <= readF32(0xC7CB70)) {
|
||||
const inside = @call(.never_tail, IsPointInsideBounds, .{ ent + 0x50, 0xC7CB5C });
|
||||
if (inside != 0) {
|
||||
// Compute position from animation data
|
||||
const anim_idx = readU32(0x86B580 + readU32(0xC7F294) * 4);
|
||||
const anim_base = ent + 0x83C + @as(u32, @bitCast(anim_idx)) * 0xC;
|
||||
var pos: [3]f32 = undefined;
|
||||
pos[0] = readF32(anim_base) + readF32(ent + 0x6C);
|
||||
pos[1] = readF32(anim_base + 4) + readF32(ent + 0x70);
|
||||
pos[2] = readF32(anim_base + 8) + readF32(ent + 0x74);
|
||||
|
||||
@call(.never_tail, AddObjectToSpatialList, .{ ent, @intFromPtr(&pos) });
|
||||
|
||||
// Walk sub-object linked list
|
||||
var node = readU32(ent + 0xE4);
|
||||
if ((node & 1) != 0 or node == 0) node = 0;
|
||||
while ((node & 1) == 0 and node != 0) {
|
||||
const obj = readU32(node + 4);
|
||||
if ((@as(*const u8, @ptrFromInt(obj + 0xC)).* & 0x80) != 0 and (readU32(obj + 0x88) != 0 or readU32(obj + 0x174) != 0)) {
|
||||
@call(.never_tail, AddToSpatialGrid, .{obj});
|
||||
}
|
||||
node = readU32(readU32(ent + 0xDC) + node + 4);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Chunk bounds + spatial grid loop (4 chunks)
|
||||
inline for (0..4) |ci| {
|
||||
const chunk = readU32(ent + 0x118 + ci * 4);
|
||||
if (chunk != 0) {
|
||||
var bounds: [6]f32 = undefined;
|
||||
@call(.never_tail, CopyChunkBounds, .{ chunk, @intFromPtr(&bounds) });
|
||||
|
||||
// Check if chunk bounds intersect the world region
|
||||
if (bounds[0] <= readF32(0xC7CB68) and bounds[1] <= readF32(0xC7CB6C) and
|
||||
bounds[2] <= readF32(0xC7CB70) and readF32(0xC7CB5C) <= bounds[3] and
|
||||
readF32(0xC7CB60) <= bounds[4] and readF32(0xC7CB64) <= bounds[5])
|
||||
{
|
||||
var center: [3]f32 = undefined;
|
||||
center[0] = (bounds[3] + bounds[0]) * 0.5;
|
||||
center[1] = (bounds[4] + bounds[1]) * 0.5;
|
||||
center[2] = (bounds[5] + bounds[2]) * 0.5;
|
||||
@call(.never_tail, AddToLayeredSpatialGrid, .{ chunk, @as(u32, ci), @intFromPtr(¢er) });
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// updateEntitiesInBounds (0x6C1F70)
|
||||
// __thiscall(ECX=this, stack=param_1), RET 0x4
|
||||
// =========================================================================
|
||||
export fn updateEntitiesInBoundsSSE(this: u32, param_1: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||||
var node = readU32(this + 0x274);
|
||||
if ((node & 1) != 0 or node == 0) node = 0;
|
||||
|
||||
while ((node & 1) == 0 and node != 0) {
|
||||
const entity = readU32(node + 4);
|
||||
const next = readU32(readU32(this + 0x26C) + 4 + node);
|
||||
|
||||
// Prefetch next node's entity data while we process this one
|
||||
if ((next & 1) == 0 and next != 0) {
|
||||
const next_entity = readU32(next + 4);
|
||||
@prefetch(@as([*]const u8, @ptrFromInt(next_entity + 0x44)), .{ .locality = 1 });
|
||||
@prefetch(@as([*]const u8, @ptrFromInt(next_entity + 0x8C)), .{ .locality = 1 });
|
||||
}
|
||||
|
||||
// Bounds check: entity chunk coords vs global region
|
||||
const cx = readI32(entity + 0x8C);
|
||||
const cy = readI32(entity + 0x90);
|
||||
|
||||
if (cx < readI32(0xC63278) or readI32(0xC63280) < cx or
|
||||
cy < readI32(0xC63274) or readI32(0xC6327C) < cy)
|
||||
{
|
||||
// Out of bounds: destroy
|
||||
const slot_idx = readU32(entity + 0xB4); // entityPtr[0x2d] = +0xB4
|
||||
writeU32(this + slot_idx * 4 + 0x278, 0);
|
||||
@call(.never_tail, ComplexMemoryCleanupAndRelease, .{node});
|
||||
@call(.never_tail, destroySecondaryGameObject, .{entity});
|
||||
} else {
|
||||
if (param_1 != 0) {
|
||||
// Call through game address so the Detour hook fires (enables A/B timing)
|
||||
const entPosHooked = @as(*const fn (u32) callconv(.{ .x86_fastcall = .{} }) void, @ptrFromInt(0x6AFAD0));
|
||||
@call(.never_tail, entPosHooked, .{entity});
|
||||
}
|
||||
}
|
||||
node = next;
|
||||
}
|
||||
}
|
||||
@@ -64,14 +64,9 @@ fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
|
||||
const TransformFn = fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
var transform_hook: hook.Detour(TransformFn) = .{};
|
||||
var teardown_active: bool = false;
|
||||
|
||||
fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||||
if (teardown_active) {
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
} else {
|
||||
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
|
||||
}
|
||||
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
@@ -153,22 +148,14 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Teardown hook (0x491180) — protect bone SSE during logout cleanup
|
||||
// SSE JMP patches — binary patches at game function addresses
|
||||
// =============================================================================
|
||||
|
||||
const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void;
|
||||
var teardown_hook: hook.Detour(TeardownFn) = .{};
|
||||
|
||||
fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void {
|
||||
teardown_active = true;
|
||||
teardown_hook.callOriginal(.{});
|
||||
teardown_active = false;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Silicon SSE JMP patches — binary patches at game function addresses
|
||||
// =============================================================================
|
||||
// cull_sse.zig
|
||||
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
|
||||
// silicon_sse.zig
|
||||
const sse = struct {
|
||||
extern fn si_normalizeVec3() callconv(.naked) void;
|
||||
extern fn si_mulMat3x4(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
@@ -236,6 +223,8 @@ fn installPatches() u32 {
|
||||
.{ .target = 0x686820, .replacement = @intFromPtr(&sse.si_translateBoundingVol), .name = "translateBoundingVol" },
|
||||
.{ .target = 0x6ABC40, .replacement = @intFromPtr(&sse.si_processLinkedListCollision), .name = "processLinkedListCollision" },
|
||||
.{ .target = 0x686000, .replacement = @intFromPtr(&sse.si_frustumCullBBox), .name = "frustumCullBBox" },
|
||||
.{ .target = 0x6B8C60, .replacement = @intFromPtr(&performSpatialCulling), .name = "PerformSpatialCulling" },
|
||||
.{ .target = 0x6B88E0, .replacement = @intFromPtr(&performCollisionDetectionSSE), .name = "performCollisionDetection" },
|
||||
};
|
||||
|
||||
var count: u32 = 0;
|
||||
@@ -275,9 +264,6 @@ pub fn installHooks() void {
|
||||
// Per-frame cache reset
|
||||
if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1;
|
||||
|
||||
// Teardown guard
|
||||
if (teardown_hook.attach(0x491180, &teardownDetour) == .ok) installed += 1;
|
||||
|
||||
// Silicon SSE binary patches
|
||||
_ = installPatches();
|
||||
|
||||
@@ -299,7 +285,6 @@ pub fn removeHooks() void {
|
||||
particle_hook.detach();
|
||||
glyph_hook.detach();
|
||||
world_update_hook.detach();
|
||||
teardown_hook.detach();
|
||||
log.close();
|
||||
mod_mutex.release(&g_mutex);
|
||||
}
|
||||
|
||||
@@ -25,6 +25,10 @@ extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x8
|
||||
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn renderParticleSprites_REF(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn resetParticleCache() void;
|
||||
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
|
||||
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern var stride_info: [8]u32; // exported from particle_sse.zig
|
||||
@@ -71,7 +75,7 @@ var last_frame_tsc: u64 = 0; // frame-to-frame TSC for total frame time
|
||||
pub var ab_use_custom: bool = false;
|
||||
|
||||
// Gate for non-transform A/B hooks. Set false to isolate transform44 SSE testing.
|
||||
const AB_OTHER_HOOKS = false;
|
||||
const AB_OTHER_HOOKS = true;
|
||||
var diag_cmp_count: u32 = 0;
|
||||
export var original_trampoline: u32 = 0; // DEBUG: expose trampoline for REF passthrough test
|
||||
|
||||
@@ -196,6 +200,12 @@ const ProfState = struct {
|
||||
matmul_cycles: u64 = 0,
|
||||
textline_calls: u64 = 0, // renderTextLine (0x5ce0c0)
|
||||
textline_cycles: u64 = 0,
|
||||
viewfrust_calls: u64 = 0, // SetupViewFrustum (0x6bc1c0) -- parent of PerformSpatialCulling
|
||||
viewfrust_cycles: u64 = 0,
|
||||
cylfrust_calls: u64 = 0, // SetupCylinderFrustum (0x6bc370) -- child of RenderSphere
|
||||
cylfrust_cycles: u64 = 0,
|
||||
rendersph_calls: u64 = 0, // RenderSphere (0x6b92b0) -- parent of SetupCylinderFrustum
|
||||
rendersph_cycles: u64 = 0,
|
||||
};
|
||||
|
||||
// =============================================================================
|
||||
@@ -868,6 +878,9 @@ var triplane_hook: hook.Detour(Fn5) = .{}; // BuildTrianglePlanes: thiscall RET
|
||||
var partsetup_hook: hook.Detour(Fn3) = .{}; // SetupParticleRendering: thiscall RET 0x4
|
||||
var matmul_hook: hook.Detour(Fn3) = .{}; // multiplyMatrix4x4: fastcall RET 0x4
|
||||
var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10
|
||||
var viewfrust_hook: hook.Detour(Fn5) = .{}; // SetupViewFrustum: thiscall RET 0xC
|
||||
var cylfrust_hook: hook.Detour(Fn5) = .{}; // SetupCylinderFrustum: thiscall RET 0xC
|
||||
var rendersph_hook: hook.Detour(Fn8v) = .{}; // RenderSphere: thiscall RET 0x18
|
||||
|
||||
// --- Detour functions (timing-only pass-through) ---
|
||||
|
||||
@@ -940,6 +953,12 @@ fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*
|
||||
}
|
||||
fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
updateEntityAndChunksPositions(a);
|
||||
prof.entpos_cycles +|= rdtsc() - s;
|
||||
prof.entpos_calls +|= 1;
|
||||
return null;
|
||||
}
|
||||
const ret = entpos_hook.callOriginal(.{ a, b });
|
||||
prof.entpos_cycles +|= rdtsc() - s;
|
||||
prof.entpos_calls +|= 1;
|
||||
@@ -967,6 +986,7 @@ fn complexgeoDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u
|
||||
return ret;
|
||||
}
|
||||
fn entboundsDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
// Original only -- A/B testing entpos first
|
||||
const s = rdtsc();
|
||||
const ret = entbounds_hook.callOriginal(.{ a, b, c });
|
||||
prof.entbounds_cycles +|= rdtsc() - s;
|
||||
@@ -1027,6 +1047,12 @@ fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
}
|
||||
fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = performSpatialCulling(a, c, d);
|
||||
prof.cull_cycles +|= rdtsc() - s;
|
||||
prof.cull_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = cull_hook.callOriginal(.{ a, b, c, d });
|
||||
prof.cull_cycles +|= rdtsc() - s;
|
||||
prof.cull_calls +|= 1;
|
||||
@@ -1034,6 +1060,12 @@ fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyop
|
||||
}
|
||||
fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = performCollisionDetectionSSE(a, c, d);
|
||||
prof.colldet_cycles +|= rdtsc() - s;
|
||||
prof.colldet_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = colldet_hook.callOriginal(.{ a, b, c, d });
|
||||
prof.colldet_cycles +|= rdtsc() - s;
|
||||
prof.colldet_calls +|= 1;
|
||||
@@ -1174,6 +1206,27 @@ fn textlineDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.
|
||||
prof.textline_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
fn viewfrustDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
const ret = viewfrust_hook.callOriginal(.{ a, b, c, d, e });
|
||||
prof.viewfrust_cycles +|= rdtsc() - s;
|
||||
prof.viewfrust_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
fn cylfrustDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
const ret = cylfrust_hook.callOriginal(.{ a, b, c, d, e });
|
||||
prof.cylfrust_cycles +|= rdtsc() - s;
|
||||
prof.cylfrust_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
const ret = rendersph_hook.callOriginal(.{ a, b, c, d, e, f, g, h });
|
||||
prof.rendersph_cycles +|= rdtsc() - s;
|
||||
prof.rendersph_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Hook: blit_hub (0x5a4f60)
|
||||
@@ -1414,6 +1467,9 @@ fn dumpStats() void {
|
||||
.{ .name = "partsetup", .cycles = prof.partsetup_cycles, .calls = prof.partsetup_calls },
|
||||
.{ .name = "matmul", .cycles = prof.matmul_cycles, .calls = prof.matmul_calls },
|
||||
.{ .name = "textline", .cycles = prof.textline_cycles, .calls = prof.textline_calls },
|
||||
.{ .name = "viewfrust", .cycles = prof.viewfrust_cycles, .calls = prof.viewfrust_calls },
|
||||
.{ .name = "cylfrust", .cycles = prof.cylfrust_cycles, .calls = prof.cylfrust_calls },
|
||||
.{ .name = "rendersph", .cycles = prof.rendersph_cycles, .calls = prof.rendersph_calls },
|
||||
};
|
||||
for (hotspots) |h| {
|
||||
if (h.calls > 0) {
|
||||
@@ -1551,8 +1607,9 @@ pub fn installHooks() void {
|
||||
_ = linkedlist_hook.attach(0x710b90, &linkedlistDetour);
|
||||
_ = color_hook.attach(0x7b9b10, &colorDetour);
|
||||
_ = setvec_hook.attach(0x686640, &setvecDetour);
|
||||
_ = cull_hook.attach(0x6b8c60, &cullDetour);
|
||||
_ = colldet_hook.attach(0x6b88e0, &colldetDetour);
|
||||
// cull + colldet graduated to weirdperformance JMP patches
|
||||
// _ = cull_hook.attach(0x6b8c60, &cullDetour);
|
||||
// _ = colldet_hook.attach(0x6b88e0, &colldetDetour);
|
||||
_ = activep_hook.attach(0x7b5a10, &activepDetour);
|
||||
_ = cbiter_hook.attach(0x404130, &cbiterDetour);
|
||||
_ = findguid_hook.attach(0x464890, &findguidDetour);
|
||||
@@ -1569,11 +1626,14 @@ pub fn installHooks() void {
|
||||
_ = partsetup_hook.attach(0x7b3d20, &partsetupDetour);
|
||||
_ = matmul_hook.attach(0x7bc6a0, &matmulDetour);
|
||||
_ = textline_hook.attach(0x5ce0c0, &textlineDetour);
|
||||
_ = viewfrust_hook.attach(0x6bc1c0, &viewfrustDetour);
|
||||
_ = cylfrust_hook.attach(0x6bc370, &cylfrustDetour);
|
||||
_ = rendersph_hook.attach(0x6b92b0, &rendersphDetour);
|
||||
|
||||
// Timer calibration now handled by performance module.
|
||||
|
||||
// blit_hub installed in lateInit() to clobber UnitXP's hook
|
||||
log.print("transform44: 39 profiling hooks installed (blit_hub deferred)\n");
|
||||
log.print("transform44: 42 profiling hooks installed (blit_hub deferred)\n");
|
||||
if (bisect_stop_section != 0) {
|
||||
log.fmt(" BISECT MODE: REF stops after section {d}, then original trampoline\n", .{bisect_stop_section});
|
||||
}
|
||||
@@ -1646,6 +1706,9 @@ pub fn removeHooks() void {
|
||||
partsetup_hook.detach();
|
||||
matmul_hook.detach();
|
||||
textline_hook.detach();
|
||||
viewfrust_hook.detach();
|
||||
cylfrust_hook.detach();
|
||||
rendersph_hook.detach();
|
||||
blit_hub_hook.detach();
|
||||
log.close();
|
||||
mod_mutex.release(&g_mutex);
|
||||
|
||||
Reference in New Issue
Block a user