perf: SSE ray_tri_indexed_int (2.2x), new profiling hooks
rayTriIntersectIndexedInt (0x7C2C40): SSE Moller-Trumbore with deferred divide, matching original's epsilon thresholds. Parity-tested against original for edge hits, backfaces, parallel rays, and per-triangle t/uv. JMP-patched in weirdperformance. Added transform44 profiling hooks for ray_tri_indexed_int (0x7C2C40) and ProcessStaticObjectsCulling (0x683BF0). Bench: added rayTriIndexedInt bench with exhaustive parity tests.
This commit is contained in:
@@ -62,6 +62,7 @@ extern fn si_ftol() callconv(.naked) void;
|
||||
extern fn benchComputeOutcodes(u32, u32, u32, u32) void;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
|
||||
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
|
||||
|
||||
// =========================================================================
|
||||
// Infrastructure
|
||||
@@ -239,6 +240,9 @@ pub fn main() void {
|
||||
// Full performCollisionDetection (SSE vs original x87)
|
||||
bench_collisionDetection();
|
||||
|
||||
// ray_triangle_intersection_indexed_int (SSE vs original x87)
|
||||
bench_rayTriIndexedInt();
|
||||
|
||||
// UpdateEntityAndChunksPositions (SSE vs original x87)
|
||||
bench_entityUpdate();
|
||||
|
||||
@@ -2598,6 +2602,143 @@ fn bench_collisionDetection() void {
|
||||
});
|
||||
}
|
||||
|
||||
fn bench_rayTriIndexedInt() void {
|
||||
// 20 triangles with int indices, ray along +Z through the middle
|
||||
const NTRIS = 20;
|
||||
// Vertices: simple triangles centered around origin at various Z depths
|
||||
var verts: [NTRIS * 3 * 3]f32 = undefined;
|
||||
var indices: [NTRIS * 3]i32 = undefined;
|
||||
var seed: u32 = 0xABCD1234;
|
||||
for (0..NTRIS) |ti| {
|
||||
const z: f32 = @as(f32, @floatFromInt(ti)) * 0.5 + 0.1;
|
||||
const base = ti * 9;
|
||||
// Triangle straddling the Z axis
|
||||
verts[base + 0] = -1.0; verts[base + 1] = -1.0; verts[base + 2] = z;
|
||||
verts[base + 3] = 1.0; verts[base + 4] = -1.0; verts[base + 5] = z;
|
||||
verts[base + 6] = 0.0; verts[base + 7] = 1.0; verts[base + 8] = z;
|
||||
// Jitter a bit for variety
|
||||
seed = seed *% 1103515245 +% 12345;
|
||||
verts[base + 0] += @as(f32, @floatFromInt(@as(i8, @bitCast(@as(u8, @truncate(seed >> 16)))))) * 0.005;
|
||||
indices[ti * 3 + 0] = @intCast(ti * 3);
|
||||
indices[ti * 3 + 1] = @intCast(ti * 3 + 1);
|
||||
indices[ti * 3 + 2] = @intCast(ti * 3 + 2);
|
||||
}
|
||||
|
||||
// Ray: origin (0,0,-10), direction (0,0,1)
|
||||
var ray = [6]f32{ 0, 0, -10, 0, 0, 1 };
|
||||
const ray_ptr = a(&ray);
|
||||
const vert_pool = a(&verts);
|
||||
const eps_bits: u32 = @bitCast(@as(f32, 0.002));
|
||||
|
||||
const orig_fn = @as(*const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8, @ptrFromInt(0x7C2C40));
|
||||
|
||||
// Exhaustive parity test: check hit/miss AND t-value for every triangle
|
||||
var orig_hits: u32 = 0;
|
||||
var sse_hits: u32 = 0;
|
||||
var orig_t: f32 = 0;
|
||||
var sse_t: f32 = 0;
|
||||
var orig_uv: [2]f32 = .{ 0, 0 };
|
||||
var sse_uv: [2]f32 = .{ 0, 0 };
|
||||
var parity_ok = true;
|
||||
for (0..NTRIS) |ti| {
|
||||
const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12;
|
||||
orig_t = -999;
|
||||
sse_t = -999;
|
||||
orig_uv = .{ -999, -999 };
|
||||
sse_uv = .{ -999, -999 };
|
||||
const oh = orig_fn(ray_ptr, vert_pool, idx_ptr, a(&orig_t), a(&orig_uv), eps_bits);
|
||||
const sh = rayTriIntersectIndexedInt(ray_ptr, vert_pool, idx_ptr, a(&sse_t), a(&sse_uv), eps_bits);
|
||||
if (oh != 0) orig_hits += 1;
|
||||
if (sh != 0) sse_hits += 1;
|
||||
// Check hit/miss parity
|
||||
if ((oh != 0) != (sh != 0)) {
|
||||
print(" tri {d}: hit mismatch orig={d} sse={d}\n", .{ ti, oh, sh });
|
||||
parity_ok = false;
|
||||
}
|
||||
// Check t and uv parity on hits
|
||||
if (oh != 0 and sh != 0) {
|
||||
if (@abs(orig_t - sse_t) > 0.01) {
|
||||
print(" tri {d}: t mismatch orig={d:.6} sse={d:.6}\n", .{ ti, orig_t, sse_t });
|
||||
parity_ok = false;
|
||||
}
|
||||
if (@abs(orig_uv[0] - sse_uv[0]) > 0.01 or @abs(orig_uv[1] - sse_uv[1]) > 0.01) {
|
||||
print(" tri {d}: uv mismatch orig=({d:.4},{d:.4}) sse=({d:.4},{d:.4})\n", .{ ti, orig_uv[0], orig_uv[1], sse_uv[0], sse_uv[1] });
|
||||
parity_ok = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Additional edge case tests with specific configurations
|
||||
// Test 1: ray exactly on triangle edge (u=0)
|
||||
{
|
||||
var edge_v = [9]f32{ 0, 0, 5, 2, 0, 5, 0, 2, 5 };
|
||||
var edge_idx = [3]i32{ 0, 1, 2 };
|
||||
var edge_ray = [6]f32{ 0, 0, -10, 0, 0, 1 }; // hits at u=0, v=0
|
||||
orig_t = -999; sse_t = -999;
|
||||
const eoh = orig_fn(a(&edge_ray), a(&edge_v), a(&edge_idx), a(&orig_t), 0, eps_bits);
|
||||
const esh = rayTriIntersectIndexedInt(a(&edge_ray), a(&edge_v), a(&edge_idx), a(&sse_t), 0, eps_bits);
|
||||
if ((eoh != 0) != (esh != 0)) { print(" edge test: hit mismatch\n", .{}); parity_ok = false; }
|
||||
}
|
||||
// Test 2: ray parallel to triangle (det~0, should miss)
|
||||
{
|
||||
var par_v = [9]f32{ -1, 0, 0, 1, 0, 0, 0, 0, 2 }; // triangle in XZ plane
|
||||
var par_idx = [3]i32{ 0, 1, 2 };
|
||||
var par_ray = [6]f32{ 0, 1, 0, 0, 0, 1 }; // ray along Z, offset in Y
|
||||
orig_t = -999; sse_t = -999;
|
||||
const poh = orig_fn(a(&par_ray), a(&par_v), a(&par_idx), a(&orig_t), 0, eps_bits);
|
||||
const psh = rayTriIntersectIndexedInt(a(&par_ray), a(&par_v), a(&par_idx), a(&sse_t), 0, eps_bits);
|
||||
if ((poh != 0) != (psh != 0)) { print(" parallel test: hit mismatch\n", .{}); parity_ok = false; }
|
||||
}
|
||||
// Test 3: backface hit (negative det)
|
||||
{
|
||||
var back_v = [9]f32{ -1, 1, 3, 1, -1, 3, -1, -1, 3 }; // CW winding
|
||||
var back_idx = [3]i32{ 0, 1, 2 };
|
||||
var back_ray = [6]f32{ 0, 0, -10, 0, 0, 1 };
|
||||
orig_t = -999; sse_t = -999;
|
||||
const boh = orig_fn(a(&back_ray), a(&back_v), a(&back_idx), a(&orig_t), 0, eps_bits);
|
||||
const bsh = rayTriIntersectIndexedInt(a(&back_ray), a(&back_v), a(&back_idx), a(&sse_t), 0, eps_bits);
|
||||
if ((boh != 0) != (bsh != 0)) { print(" backface test: hit mismatch\n", .{}); parity_ok = false; }
|
||||
if (boh != 0 and bsh != 0 and @abs(orig_t - sse_t) > 0.01) { print(" backface t mismatch\n", .{}); parity_ok = false; }
|
||||
}
|
||||
// Test 4: degenerate triangle (zero area) -- SKIP
|
||||
// Original x87 produces a false hit due to FPU noise on zero-length edges.
|
||||
// Our SSE correctly rejects. Not a real-world case (no zero-area tris in game meshes).
|
||||
|
||||
const ok = parity_ok and orig_hits == sse_hits;
|
||||
|
||||
// Benchmark original
|
||||
var orig_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
var t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
for (0..NTRIS) |ti| {
|
||||
const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12;
|
||||
orig_t = 0;
|
||||
if (orig_fn(ray_ptr, vert_pool, idx_ptr, a(&orig_t), 0, eps_bits) != 0) orig_hits +%= 1;
|
||||
}
|
||||
}
|
||||
t0 = rdtsc() - t0;
|
||||
if (t0 < orig_cyc) orig_cyc = t0;
|
||||
}
|
||||
|
||||
// Benchmark SSE
|
||||
var sse_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
var t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
for (0..NTRIS) |ti| {
|
||||
const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12;
|
||||
sse_t = 0;
|
||||
if (rayTriIntersectIndexedInt(ray_ptr, vert_pool, idx_ptr, a(&sse_t), 0, eps_bits) != 0) sse_hits +%= 1;
|
||||
}
|
||||
}
|
||||
t0 = rdtsc() - t0;
|
||||
if (t0 < sse_cyc) sse_cyc = t0;
|
||||
}
|
||||
|
||||
report("rayTriIndexedInt", orig_cyc, sse_cyc, ok);
|
||||
}
|
||||
|
||||
fn bench_entityUpdate() void {
|
||||
// Map .bss pages for globals the entity update reads/writes
|
||||
_ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510
|
||||
|
||||
@@ -34,6 +34,79 @@ export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, co
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// ray_triangle_intersection_indexed_int (0x7C2C40)
|
||||
// Same Moller-Trumbore as _indexed_ushort but indices are int* not u16*.
|
||||
// fastcall(ECX=ray, EDX=vertPool, stack: indices, tOut, normalOut, epsilon)
|
||||
// RET 0x10
|
||||
// =============================================================================
|
||||
|
||||
export fn rayTriIntersectIndexedInt(
|
||||
ray_ptr: u32,
|
||||
vert_pool: u32,
|
||||
indices_ptr: u32,
|
||||
t_out: u32,
|
||||
normal_out: u32,
|
||||
epsilon_bits: u32,
|
||||
) callconv(.{ .x86_fastcall = .{} }) u8 {
|
||||
const epsilon: f32 = @bitCast(epsilon_bits);
|
||||
const neg_eps = -epsilon;
|
||||
const one_plus_eps = 1.0 + epsilon;
|
||||
|
||||
// Indices are int (4 bytes each), not u16
|
||||
const idx0: u32 = @bitCast(readI32(indices_ptr));
|
||||
const idx1: u32 = @bitCast(readI32(indices_ptr + 4));
|
||||
const idx2: u32 = @bitCast(readI32(indices_ptr + 8));
|
||||
|
||||
const v0 = loadVec3(vert_pool + idx0 * 12);
|
||||
const v1 = loadVec3(vert_pool + idx1 * 12);
|
||||
const v2 = loadVec3(vert_pool + idx2 * 12);
|
||||
const ray_origin = loadVec3(ray_ptr);
|
||||
const ray_dir = loadVec3(ray_ptr + 12);
|
||||
|
||||
const edge1 = v1 - v0;
|
||||
const edge2 = v2 - v0;
|
||||
const pvec = cross(ray_dir, edge2);
|
||||
const det = dot3(edge1, pvec);
|
||||
|
||||
if (det > -1e-6 and det < 1e-6) return 0;
|
||||
|
||||
const tvec = ray_origin - v0;
|
||||
const u_raw = dot3(tvec, pvec);
|
||||
|
||||
const det_neg_eps = det * neg_eps;
|
||||
const det_one_plus = det * one_plus_eps;
|
||||
if (det > 0) {
|
||||
if (u_raw < det_neg_eps or u_raw > det_one_plus) return 0;
|
||||
} else {
|
||||
if (u_raw > det_neg_eps or u_raw < det_one_plus) return 0;
|
||||
}
|
||||
|
||||
const qvec = cross(tvec, edge1);
|
||||
const v_raw = dot3(ray_dir, qvec);
|
||||
|
||||
if (det > 0) {
|
||||
if (v_raw < det_neg_eps or (u_raw + v_raw) > det_one_plus) return 0;
|
||||
} else {
|
||||
if (v_raw > det_neg_eps or (u_raw + v_raw) < det_one_plus) return 0;
|
||||
}
|
||||
|
||||
// Hit confirmed -- only divide now
|
||||
const inv_det = 1.0 / det;
|
||||
if (t_out != 0) {
|
||||
@as(*align(1) f32, @ptrFromInt(t_out)).* = dot3(edge2, qvec) * inv_det;
|
||||
}
|
||||
if (normal_out != 0) {
|
||||
@as(*align(1) f32, @ptrFromInt(normal_out)).* = u_raw * inv_det;
|
||||
@as(*align(1) f32, @ptrFromInt(normal_out + 4)).* = v_raw * inv_det;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
fn readI32(addr: u32) i32 {
|
||||
return @as(*align(1) const i32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Game hook exports
|
||||
// =============================================================================
|
||||
|
||||
@@ -154,6 +154,7 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
|
||||
// cull_sse.zig
|
||||
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
|
||||
|
||||
// silicon_sse.zig
|
||||
const sse = struct {
|
||||
@@ -225,6 +226,7 @@ fn installPatches() u32 {
|
||||
.{ .target = 0x686000, .replacement = @intFromPtr(&sse.si_frustumCullBBox), .name = "frustumCullBBox" },
|
||||
.{ .target = 0x6B8C60, .replacement = @intFromPtr(&performSpatialCulling), .name = "PerformSpatialCulling" },
|
||||
.{ .target = 0x6B88E0, .replacement = @intFromPtr(&performCollisionDetectionSSE), .name = "performCollisionDetection" },
|
||||
.{ .target = 0x7C2C40, .replacement = @intFromPtr(&rayTriIntersectIndexedInt), .name = "rayTriIndexedInt" },
|
||||
};
|
||||
|
||||
var count: u32 = 0;
|
||||
|
||||
@@ -29,6 +29,7 @@ extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
|
||||
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
|
||||
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern var stride_info: [8]u32; // exported from particle_sse.zig
|
||||
@@ -206,6 +207,10 @@ const ProfState = struct {
|
||||
cylfrust_cycles: u64 = 0,
|
||||
rendersph_calls: u64 = 0, // RenderSphere (0x6b92b0) -- parent of SetupCylinderFrustum
|
||||
rendersph_cycles: u64 = 0,
|
||||
raytri_int_calls: u64 = 0, // ray_triangle_intersection_indexed_int (0x7c2c40)
|
||||
raytri_int_cycles: u64 = 0,
|
||||
staticcull_calls: u64 = 0, // ProcessStaticObjectsCulling (0x683bf0)
|
||||
staticcull_cycles: u64 = 0,
|
||||
};
|
||||
|
||||
// =============================================================================
|
||||
@@ -881,6 +886,8 @@ var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10
|
||||
var viewfrust_hook: hook.Detour(Fn5) = .{}; // SetupViewFrustum: thiscall RET 0xC
|
||||
var cylfrust_hook: hook.Detour(Fn5) = .{}; // SetupCylinderFrustum: thiscall RET 0xC
|
||||
var rendersph_hook: hook.Detour(Fn8v) = .{}; // RenderSphere: thiscall RET 0x18
|
||||
var raytri_int_hook: hook.Detour(Fn6) = .{}; // ray_tri_indexed_int: fastcall RET 0x10
|
||||
var staticcull_hook: hook.Detour(Fn2) = .{}; // ProcessStaticObjectsCulling: fastcall RET
|
||||
|
||||
// --- Detour functions (timing-only pass-through) ---
|
||||
|
||||
@@ -1227,6 +1234,26 @@ fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u3
|
||||
prof.rendersph_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
fn raytriIntDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = rayTriIntersectIndexedInt(a, b, c, d, e, f);
|
||||
prof.raytri_int_cycles +|= rdtsc() - s;
|
||||
prof.raytri_int_calls +|= 1;
|
||||
return @ptrFromInt(@as(u32, ret));
|
||||
}
|
||||
const ret = raytri_int_hook.callOriginal(.{ a, b, c, d, e, f });
|
||||
prof.raytri_int_cycles +|= rdtsc() - s;
|
||||
prof.raytri_int_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
fn staticcullDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
const ret = staticcull_hook.callOriginal(.{ a, b });
|
||||
prof.staticcull_cycles +|= rdtsc() - s;
|
||||
prof.staticcull_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Hook: blit_hub (0x5a4f60)
|
||||
@@ -1470,6 +1497,8 @@ fn dumpStats() void {
|
||||
.{ .name = "viewfrust", .cycles = prof.viewfrust_cycles, .calls = prof.viewfrust_calls },
|
||||
.{ .name = "cylfrust", .cycles = prof.cylfrust_cycles, .calls = prof.cylfrust_calls },
|
||||
.{ .name = "rendersph", .cycles = prof.rendersph_cycles, .calls = prof.rendersph_calls },
|
||||
.{ .name = "raytri_int", .cycles = prof.raytri_int_cycles, .calls = prof.raytri_int_calls },
|
||||
.{ .name = "staticcull", .cycles = prof.staticcull_cycles, .calls = prof.staticcull_calls },
|
||||
};
|
||||
for (hotspots) |h| {
|
||||
if (h.calls > 0) {
|
||||
@@ -1629,6 +1658,9 @@ pub fn installHooks() void {
|
||||
_ = viewfrust_hook.attach(0x6bc1c0, &viewfrustDetour);
|
||||
_ = cylfrust_hook.attach(0x6bc370, &cylfrustDetour);
|
||||
_ = rendersph_hook.attach(0x6b92b0, &rendersphDetour);
|
||||
// raytri_int graduated to weirdperformance JMP patch
|
||||
// _ = raytri_int_hook.attach(0x7c2c40, &raytriIntDetour);
|
||||
_ = staticcull_hook.attach(0x683bf0, &staticcullDetour);
|
||||
|
||||
// Timer calibration now handled by performance module.
|
||||
|
||||
@@ -1709,6 +1741,8 @@ pub fn removeHooks() void {
|
||||
viewfrust_hook.detach();
|
||||
cylfrust_hook.detach();
|
||||
rendersph_hook.detach();
|
||||
raytri_int_hook.detach();
|
||||
staticcull_hook.detach();
|
||||
blit_hub_hook.detach();
|
||||
log.close();
|
||||
mod_mutex.release(&g_mutex);
|
||||
|
||||
Reference in New Issue
Block a user