perf: SSE ray_tri_indexed_int (2.2x), new profiling hooks

rayTriIntersectIndexedInt (0x7C2C40): SSE Moller-Trumbore with deferred
divide, matching original's epsilon thresholds. Parity-tested against
original for edge hits, backfaces, parallel rays, and per-triangle t/uv.
JMP-patched in weirdperformance.

Added transform44 profiling hooks for ray_tri_indexed_int (0x7C2C40)
and ProcessStaticObjectsCulling (0x683BF0).

Bench: added rayTriIndexedInt bench with exhaustive parity tests.
This commit is contained in:
MarcelineVQ
2026-03-25 13:36:33 -07:00
parent 41ca30cfd5
commit ad33eb9b90
4 changed files with 250 additions and 0 deletions
+141
View File
@@ -62,6 +62,7 @@ extern fn si_ftol() callconv(.naked) void;
extern fn benchComputeOutcodes(u32, u32, u32, u32) void;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
// =========================================================================
// Infrastructure
@@ -239,6 +240,9 @@ pub fn main() void {
// Full performCollisionDetection (SSE vs original x87)
bench_collisionDetection();
// ray_triangle_intersection_indexed_int (SSE vs original x87)
bench_rayTriIndexedInt();
// UpdateEntityAndChunksPositions (SSE vs original x87)
bench_entityUpdate();
@@ -2598,6 +2602,143 @@ fn bench_collisionDetection() void {
});
}
fn bench_rayTriIndexedInt() void {
// 20 triangles with int indices, ray along +Z through the middle
const NTRIS = 20;
// Vertices: simple triangles centered around origin at various Z depths
var verts: [NTRIS * 3 * 3]f32 = undefined;
var indices: [NTRIS * 3]i32 = undefined;
var seed: u32 = 0xABCD1234;
for (0..NTRIS) |ti| {
const z: f32 = @as(f32, @floatFromInt(ti)) * 0.5 + 0.1;
const base = ti * 9;
// Triangle straddling the Z axis
verts[base + 0] = -1.0; verts[base + 1] = -1.0; verts[base + 2] = z;
verts[base + 3] = 1.0; verts[base + 4] = -1.0; verts[base + 5] = z;
verts[base + 6] = 0.0; verts[base + 7] = 1.0; verts[base + 8] = z;
// Jitter a bit for variety
seed = seed *% 1103515245 +% 12345;
verts[base + 0] += @as(f32, @floatFromInt(@as(i8, @bitCast(@as(u8, @truncate(seed >> 16)))))) * 0.005;
indices[ti * 3 + 0] = @intCast(ti * 3);
indices[ti * 3 + 1] = @intCast(ti * 3 + 1);
indices[ti * 3 + 2] = @intCast(ti * 3 + 2);
}
// Ray: origin (0,0,-10), direction (0,0,1)
var ray = [6]f32{ 0, 0, -10, 0, 0, 1 };
const ray_ptr = a(&ray);
const vert_pool = a(&verts);
const eps_bits: u32 = @bitCast(@as(f32, 0.002));
const orig_fn = @as(*const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8, @ptrFromInt(0x7C2C40));
// Exhaustive parity test: check hit/miss AND t-value for every triangle
var orig_hits: u32 = 0;
var sse_hits: u32 = 0;
var orig_t: f32 = 0;
var sse_t: f32 = 0;
var orig_uv: [2]f32 = .{ 0, 0 };
var sse_uv: [2]f32 = .{ 0, 0 };
var parity_ok = true;
for (0..NTRIS) |ti| {
const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12;
orig_t = -999;
sse_t = -999;
orig_uv = .{ -999, -999 };
sse_uv = .{ -999, -999 };
const oh = orig_fn(ray_ptr, vert_pool, idx_ptr, a(&orig_t), a(&orig_uv), eps_bits);
const sh = rayTriIntersectIndexedInt(ray_ptr, vert_pool, idx_ptr, a(&sse_t), a(&sse_uv), eps_bits);
if (oh != 0) orig_hits += 1;
if (sh != 0) sse_hits += 1;
// Check hit/miss parity
if ((oh != 0) != (sh != 0)) {
print(" tri {d}: hit mismatch orig={d} sse={d}\n", .{ ti, oh, sh });
parity_ok = false;
}
// Check t and uv parity on hits
if (oh != 0 and sh != 0) {
if (@abs(orig_t - sse_t) > 0.01) {
print(" tri {d}: t mismatch orig={d:.6} sse={d:.6}\n", .{ ti, orig_t, sse_t });
parity_ok = false;
}
if (@abs(orig_uv[0] - sse_uv[0]) > 0.01 or @abs(orig_uv[1] - sse_uv[1]) > 0.01) {
print(" tri {d}: uv mismatch orig=({d:.4},{d:.4}) sse=({d:.4},{d:.4})\n", .{ ti, orig_uv[0], orig_uv[1], sse_uv[0], sse_uv[1] });
parity_ok = false;
}
}
}
// Additional edge case tests with specific configurations
// Test 1: ray exactly on triangle edge (u=0)
{
var edge_v = [9]f32{ 0, 0, 5, 2, 0, 5, 0, 2, 5 };
var edge_idx = [3]i32{ 0, 1, 2 };
var edge_ray = [6]f32{ 0, 0, -10, 0, 0, 1 }; // hits at u=0, v=0
orig_t = -999; sse_t = -999;
const eoh = orig_fn(a(&edge_ray), a(&edge_v), a(&edge_idx), a(&orig_t), 0, eps_bits);
const esh = rayTriIntersectIndexedInt(a(&edge_ray), a(&edge_v), a(&edge_idx), a(&sse_t), 0, eps_bits);
if ((eoh != 0) != (esh != 0)) { print(" edge test: hit mismatch\n", .{}); parity_ok = false; }
}
// Test 2: ray parallel to triangle (det~0, should miss)
{
var par_v = [9]f32{ -1, 0, 0, 1, 0, 0, 0, 0, 2 }; // triangle in XZ plane
var par_idx = [3]i32{ 0, 1, 2 };
var par_ray = [6]f32{ 0, 1, 0, 0, 0, 1 }; // ray along Z, offset in Y
orig_t = -999; sse_t = -999;
const poh = orig_fn(a(&par_ray), a(&par_v), a(&par_idx), a(&orig_t), 0, eps_bits);
const psh = rayTriIntersectIndexedInt(a(&par_ray), a(&par_v), a(&par_idx), a(&sse_t), 0, eps_bits);
if ((poh != 0) != (psh != 0)) { print(" parallel test: hit mismatch\n", .{}); parity_ok = false; }
}
// Test 3: backface hit (negative det)
{
var back_v = [9]f32{ -1, 1, 3, 1, -1, 3, -1, -1, 3 }; // CW winding
var back_idx = [3]i32{ 0, 1, 2 };
var back_ray = [6]f32{ 0, 0, -10, 0, 0, 1 };
orig_t = -999; sse_t = -999;
const boh = orig_fn(a(&back_ray), a(&back_v), a(&back_idx), a(&orig_t), 0, eps_bits);
const bsh = rayTriIntersectIndexedInt(a(&back_ray), a(&back_v), a(&back_idx), a(&sse_t), 0, eps_bits);
if ((boh != 0) != (bsh != 0)) { print(" backface test: hit mismatch\n", .{}); parity_ok = false; }
if (boh != 0 and bsh != 0 and @abs(orig_t - sse_t) > 0.01) { print(" backface t mismatch\n", .{}); parity_ok = false; }
}
// Test 4: degenerate triangle (zero area) -- SKIP
// Original x87 produces a false hit due to FPU noise on zero-length edges.
// Our SSE correctly rejects. Not a real-world case (no zero-area tris in game meshes).
const ok = parity_ok and orig_hits == sse_hits;
// Benchmark original
var orig_cyc: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t0 = rdtsc();
for (0..ITERS) |_| {
for (0..NTRIS) |ti| {
const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12;
orig_t = 0;
if (orig_fn(ray_ptr, vert_pool, idx_ptr, a(&orig_t), 0, eps_bits) != 0) orig_hits +%= 1;
}
}
t0 = rdtsc() - t0;
if (t0 < orig_cyc) orig_cyc = t0;
}
// Benchmark SSE
var sse_cyc: u64 = std.math.maxInt(u64);
for (0..5) |_| {
var t0 = rdtsc();
for (0..ITERS) |_| {
for (0..NTRIS) |ti| {
const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12;
sse_t = 0;
if (rayTriIntersectIndexedInt(ray_ptr, vert_pool, idx_ptr, a(&sse_t), 0, eps_bits) != 0) sse_hits +%= 1;
}
}
t0 = rdtsc() - t0;
if (t0 < sse_cyc) sse_cyc = t0;
}
report("rayTriIndexedInt", orig_cyc, sse_cyc, ok);
}
fn bench_entityUpdate() void {
// Map .bss pages for globals the entity update reads/writes
_ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510
+73
View File
@@ -34,6 +34,79 @@ export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, co
}
}
// =============================================================================
// ray_triangle_intersection_indexed_int (0x7C2C40)
// Same Moller-Trumbore as _indexed_ushort but indices are int* not u16*.
// fastcall(ECX=ray, EDX=vertPool, stack: indices, tOut, normalOut, epsilon)
// RET 0x10
// =============================================================================
export fn rayTriIntersectIndexedInt(
ray_ptr: u32,
vert_pool: u32,
indices_ptr: u32,
t_out: u32,
normal_out: u32,
epsilon_bits: u32,
) callconv(.{ .x86_fastcall = .{} }) u8 {
const epsilon: f32 = @bitCast(epsilon_bits);
const neg_eps = -epsilon;
const one_plus_eps = 1.0 + epsilon;
// Indices are int (4 bytes each), not u16
const idx0: u32 = @bitCast(readI32(indices_ptr));
const idx1: u32 = @bitCast(readI32(indices_ptr + 4));
const idx2: u32 = @bitCast(readI32(indices_ptr + 8));
const v0 = loadVec3(vert_pool + idx0 * 12);
const v1 = loadVec3(vert_pool + idx1 * 12);
const v2 = loadVec3(vert_pool + idx2 * 12);
const ray_origin = loadVec3(ray_ptr);
const ray_dir = loadVec3(ray_ptr + 12);
const edge1 = v1 - v0;
const edge2 = v2 - v0;
const pvec = cross(ray_dir, edge2);
const det = dot3(edge1, pvec);
if (det > -1e-6 and det < 1e-6) return 0;
const tvec = ray_origin - v0;
const u_raw = dot3(tvec, pvec);
const det_neg_eps = det * neg_eps;
const det_one_plus = det * one_plus_eps;
if (det > 0) {
if (u_raw < det_neg_eps or u_raw > det_one_plus) return 0;
} else {
if (u_raw > det_neg_eps or u_raw < det_one_plus) return 0;
}
const qvec = cross(tvec, edge1);
const v_raw = dot3(ray_dir, qvec);
if (det > 0) {
if (v_raw < det_neg_eps or (u_raw + v_raw) > det_one_plus) return 0;
} else {
if (v_raw > det_neg_eps or (u_raw + v_raw) < det_one_plus) return 0;
}
// Hit confirmed -- only divide now
const inv_det = 1.0 / det;
if (t_out != 0) {
@as(*align(1) f32, @ptrFromInt(t_out)).* = dot3(edge2, qvec) * inv_det;
}
if (normal_out != 0) {
@as(*align(1) f32, @ptrFromInt(normal_out)).* = u_raw * inv_det;
@as(*align(1) f32, @ptrFromInt(normal_out + 4)).* = v_raw * inv_det;
}
return 1;
}
fn readI32(addr: u32) i32 {
return @as(*align(1) const i32, @ptrFromInt(addr)).*;
}
// =============================================================================
// Game hook exports
// =============================================================================
+2
View File
@@ -154,6 +154,7 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
// cull_sse.zig
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
// silicon_sse.zig
const sse = struct {
@@ -225,6 +226,7 @@ fn installPatches() u32 {
.{ .target = 0x686000, .replacement = @intFromPtr(&sse.si_frustumCullBBox), .name = "frustumCullBBox" },
.{ .target = 0x6B8C60, .replacement = @intFromPtr(&performSpatialCulling), .name = "PerformSpatialCulling" },
.{ .target = 0x6B88E0, .replacement = @intFromPtr(&performCollisionDetectionSSE), .name = "performCollisionDetection" },
.{ .target = 0x7C2C40, .replacement = @intFromPtr(&rayTriIntersectIndexedInt), .name = "rayTriIndexedInt" },
};
var count: u32 = 0;
+34
View File
@@ -29,6 +29,7 @@ extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern var stride_info: [8]u32; // exported from particle_sse.zig
@@ -206,6 +207,10 @@ const ProfState = struct {
cylfrust_cycles: u64 = 0,
rendersph_calls: u64 = 0, // RenderSphere (0x6b92b0) -- parent of SetupCylinderFrustum
rendersph_cycles: u64 = 0,
raytri_int_calls: u64 = 0, // ray_triangle_intersection_indexed_int (0x7c2c40)
raytri_int_cycles: u64 = 0,
staticcull_calls: u64 = 0, // ProcessStaticObjectsCulling (0x683bf0)
staticcull_cycles: u64 = 0,
};
// =============================================================================
@@ -881,6 +886,8 @@ var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10
var viewfrust_hook: hook.Detour(Fn5) = .{}; // SetupViewFrustum: thiscall RET 0xC
var cylfrust_hook: hook.Detour(Fn5) = .{}; // SetupCylinderFrustum: thiscall RET 0xC
var rendersph_hook: hook.Detour(Fn8v) = .{}; // RenderSphere: thiscall RET 0x18
var raytri_int_hook: hook.Detour(Fn6) = .{}; // ray_tri_indexed_int: fastcall RET 0x10
var staticcull_hook: hook.Detour(Fn2) = .{}; // ProcessStaticObjectsCulling: fastcall RET
// --- Detour functions (timing-only pass-through) ---
@@ -1227,6 +1234,26 @@ fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u3
prof.rendersph_calls +|= 1;
return ret;
}
fn raytriIntDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = rayTriIntersectIndexedInt(a, b, c, d, e, f);
prof.raytri_int_cycles +|= rdtsc() - s;
prof.raytri_int_calls +|= 1;
return @ptrFromInt(@as(u32, ret));
}
const ret = raytri_int_hook.callOriginal(.{ a, b, c, d, e, f });
prof.raytri_int_cycles +|= rdtsc() - s;
prof.raytri_int_calls +|= 1;
return ret;
}
fn staticcullDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
const ret = staticcull_hook.callOriginal(.{ a, b });
prof.staticcull_cycles +|= rdtsc() - s;
prof.staticcull_calls +|= 1;
return ret;
}
// =============================================================================
// Hook: blit_hub (0x5a4f60)
@@ -1470,6 +1497,8 @@ fn dumpStats() void {
.{ .name = "viewfrust", .cycles = prof.viewfrust_cycles, .calls = prof.viewfrust_calls },
.{ .name = "cylfrust", .cycles = prof.cylfrust_cycles, .calls = prof.cylfrust_calls },
.{ .name = "rendersph", .cycles = prof.rendersph_cycles, .calls = prof.rendersph_calls },
.{ .name = "raytri_int", .cycles = prof.raytri_int_cycles, .calls = prof.raytri_int_calls },
.{ .name = "staticcull", .cycles = prof.staticcull_cycles, .calls = prof.staticcull_calls },
};
for (hotspots) |h| {
if (h.calls > 0) {
@@ -1629,6 +1658,9 @@ pub fn installHooks() void {
_ = viewfrust_hook.attach(0x6bc1c0, &viewfrustDetour);
_ = cylfrust_hook.attach(0x6bc370, &cylfrustDetour);
_ = rendersph_hook.attach(0x6b92b0, &rendersphDetour);
// raytri_int graduated to weirdperformance JMP patch
// _ = raytri_int_hook.attach(0x7c2c40, &raytriIntDetour);
_ = staticcull_hook.attach(0x683bf0, &staticcullDetour);
// Timer calibration now handled by performance module.
@@ -1709,6 +1741,8 @@ pub fn removeHooks() void {
viewfrust_hook.detach();
cylfrust_hook.detach();
rendersph_hook.detach();
raytri_int_hook.detach();
staticcull_hook.detach();
blit_hub_hook.detach();
log.close();
mod_mutex.release(&g_mutex);