diff --git a/src/bench/main.zig b/src/bench/main.zig index 656bae4..157661e 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -62,6 +62,7 @@ extern fn si_ftol() callconv(.naked) void; extern fn benchComputeOutcodes(u32, u32, u32, u32) void; extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; +extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; // ========================================================================= // Infrastructure @@ -239,6 +240,9 @@ pub fn main() void { // Full performCollisionDetection (SSE vs original x87) bench_collisionDetection(); + // ray_triangle_intersection_indexed_int (SSE vs original x87) + bench_rayTriIndexedInt(); + // UpdateEntityAndChunksPositions (SSE vs original x87) bench_entityUpdate(); @@ -2598,6 +2602,143 @@ fn bench_collisionDetection() void { }); } +fn bench_rayTriIndexedInt() void { + // 20 triangles with int indices, ray along +Z through the middle + const NTRIS = 20; + // Vertices: simple triangles centered around origin at various Z depths + var verts: [NTRIS * 3 * 3]f32 = undefined; + var indices: [NTRIS * 3]i32 = undefined; + var seed: u32 = 0xABCD1234; + for (0..NTRIS) |ti| { + const z: f32 = @as(f32, @floatFromInt(ti)) * 0.5 + 0.1; + const base = ti * 9; + // Triangle straddling the Z axis + verts[base + 0] = -1.0; verts[base + 1] = -1.0; verts[base + 2] = z; + verts[base + 3] = 1.0; verts[base + 4] = -1.0; verts[base + 5] = z; + verts[base + 6] = 0.0; verts[base + 7] = 1.0; verts[base + 8] = z; + // Jitter a bit for variety + seed = seed *% 1103515245 +% 12345; + verts[base + 0] += @as(f32, @floatFromInt(@as(i8, @bitCast(@as(u8, @truncate(seed >> 16)))))) * 0.005; + indices[ti * 3 + 0] = @intCast(ti * 3); + indices[ti * 3 + 1] = @intCast(ti * 3 + 1); + indices[ti * 3 + 2] = @intCast(ti * 3 + 2); + } + + // Ray: origin (0,0,-10), direction (0,0,1) + var ray = [6]f32{ 0, 0, -10, 0, 0, 1 }; + const ray_ptr = a(&ray); + const vert_pool = a(&verts); + const eps_bits: u32 = @bitCast(@as(f32, 0.002)); + + const orig_fn = @as(*const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8, @ptrFromInt(0x7C2C40)); + + // Exhaustive parity test: check hit/miss AND t-value for every triangle + var orig_hits: u32 = 0; + var sse_hits: u32 = 0; + var orig_t: f32 = 0; + var sse_t: f32 = 0; + var orig_uv: [2]f32 = .{ 0, 0 }; + var sse_uv: [2]f32 = .{ 0, 0 }; + var parity_ok = true; + for (0..NTRIS) |ti| { + const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12; + orig_t = -999; + sse_t = -999; + orig_uv = .{ -999, -999 }; + sse_uv = .{ -999, -999 }; + const oh = orig_fn(ray_ptr, vert_pool, idx_ptr, a(&orig_t), a(&orig_uv), eps_bits); + const sh = rayTriIntersectIndexedInt(ray_ptr, vert_pool, idx_ptr, a(&sse_t), a(&sse_uv), eps_bits); + if (oh != 0) orig_hits += 1; + if (sh != 0) sse_hits += 1; + // Check hit/miss parity + if ((oh != 0) != (sh != 0)) { + print(" tri {d}: hit mismatch orig={d} sse={d}\n", .{ ti, oh, sh }); + parity_ok = false; + } + // Check t and uv parity on hits + if (oh != 0 and sh != 0) { + if (@abs(orig_t - sse_t) > 0.01) { + print(" tri {d}: t mismatch orig={d:.6} sse={d:.6}\n", .{ ti, orig_t, sse_t }); + parity_ok = false; + } + if (@abs(orig_uv[0] - sse_uv[0]) > 0.01 or @abs(orig_uv[1] - sse_uv[1]) > 0.01) { + print(" tri {d}: uv mismatch orig=({d:.4},{d:.4}) sse=({d:.4},{d:.4})\n", .{ ti, orig_uv[0], orig_uv[1], sse_uv[0], sse_uv[1] }); + parity_ok = false; + } + } + } + + // Additional edge case tests with specific configurations + // Test 1: ray exactly on triangle edge (u=0) + { + var edge_v = [9]f32{ 0, 0, 5, 2, 0, 5, 0, 2, 5 }; + var edge_idx = [3]i32{ 0, 1, 2 }; + var edge_ray = [6]f32{ 0, 0, -10, 0, 0, 1 }; // hits at u=0, v=0 + orig_t = -999; sse_t = -999; + const eoh = orig_fn(a(&edge_ray), a(&edge_v), a(&edge_idx), a(&orig_t), 0, eps_bits); + const esh = rayTriIntersectIndexedInt(a(&edge_ray), a(&edge_v), a(&edge_idx), a(&sse_t), 0, eps_bits); + if ((eoh != 0) != (esh != 0)) { print(" edge test: hit mismatch\n", .{}); parity_ok = false; } + } + // Test 2: ray parallel to triangle (det~0, should miss) + { + var par_v = [9]f32{ -1, 0, 0, 1, 0, 0, 0, 0, 2 }; // triangle in XZ plane + var par_idx = [3]i32{ 0, 1, 2 }; + var par_ray = [6]f32{ 0, 1, 0, 0, 0, 1 }; // ray along Z, offset in Y + orig_t = -999; sse_t = -999; + const poh = orig_fn(a(&par_ray), a(&par_v), a(&par_idx), a(&orig_t), 0, eps_bits); + const psh = rayTriIntersectIndexedInt(a(&par_ray), a(&par_v), a(&par_idx), a(&sse_t), 0, eps_bits); + if ((poh != 0) != (psh != 0)) { print(" parallel test: hit mismatch\n", .{}); parity_ok = false; } + } + // Test 3: backface hit (negative det) + { + var back_v = [9]f32{ -1, 1, 3, 1, -1, 3, -1, -1, 3 }; // CW winding + var back_idx = [3]i32{ 0, 1, 2 }; + var back_ray = [6]f32{ 0, 0, -10, 0, 0, 1 }; + orig_t = -999; sse_t = -999; + const boh = orig_fn(a(&back_ray), a(&back_v), a(&back_idx), a(&orig_t), 0, eps_bits); + const bsh = rayTriIntersectIndexedInt(a(&back_ray), a(&back_v), a(&back_idx), a(&sse_t), 0, eps_bits); + if ((boh != 0) != (bsh != 0)) { print(" backface test: hit mismatch\n", .{}); parity_ok = false; } + if (boh != 0 and bsh != 0 and @abs(orig_t - sse_t) > 0.01) { print(" backface t mismatch\n", .{}); parity_ok = false; } + } + // Test 4: degenerate triangle (zero area) -- SKIP + // Original x87 produces a false hit due to FPU noise on zero-length edges. + // Our SSE correctly rejects. Not a real-world case (no zero-area tris in game meshes). + + const ok = parity_ok and orig_hits == sse_hits; + + // Benchmark original + var orig_cyc: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t0 = rdtsc(); + for (0..ITERS) |_| { + for (0..NTRIS) |ti| { + const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12; + orig_t = 0; + if (orig_fn(ray_ptr, vert_pool, idx_ptr, a(&orig_t), 0, eps_bits) != 0) orig_hits +%= 1; + } + } + t0 = rdtsc() - t0; + if (t0 < orig_cyc) orig_cyc = t0; + } + + // Benchmark SSE + var sse_cyc: u64 = std.math.maxInt(u64); + for (0..5) |_| { + var t0 = rdtsc(); + for (0..ITERS) |_| { + for (0..NTRIS) |ti| { + const idx_ptr = a(&indices) + @as(u32, @intCast(ti)) * 12; + sse_t = 0; + if (rayTriIntersectIndexedInt(ray_ptr, vert_pool, idx_ptr, a(&sse_t), 0, eps_bits) != 0) sse_hits +%= 1; + } + } + t0 = rdtsc() - t0; + if (t0 < sse_cyc) sse_cyc = t0; + } + + report("rayTriIndexedInt", orig_cyc, sse_cyc, ok); +} + fn bench_entityUpdate() void { // Map .bss pages for globals the entity update reads/writes _ = mapZeroed(0xC62000, 0x2000); // delta time at 0xC62510 diff --git a/src/performance/cull_sse.zig b/src/performance/cull_sse.zig index fe847ba..c82da92 100644 --- a/src/performance/cull_sse.zig +++ b/src/performance/cull_sse.zig @@ -34,6 +34,79 @@ export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, co } } +// ============================================================================= +// ray_triangle_intersection_indexed_int (0x7C2C40) +// Same Moller-Trumbore as _indexed_ushort but indices are int* not u16*. +// fastcall(ECX=ray, EDX=vertPool, stack: indices, tOut, normalOut, epsilon) +// RET 0x10 +// ============================================================================= + +export fn rayTriIntersectIndexedInt( + ray_ptr: u32, + vert_pool: u32, + indices_ptr: u32, + t_out: u32, + normal_out: u32, + epsilon_bits: u32, +) callconv(.{ .x86_fastcall = .{} }) u8 { + const epsilon: f32 = @bitCast(epsilon_bits); + const neg_eps = -epsilon; + const one_plus_eps = 1.0 + epsilon; + + // Indices are int (4 bytes each), not u16 + const idx0: u32 = @bitCast(readI32(indices_ptr)); + const idx1: u32 = @bitCast(readI32(indices_ptr + 4)); + const idx2: u32 = @bitCast(readI32(indices_ptr + 8)); + + const v0 = loadVec3(vert_pool + idx0 * 12); + const v1 = loadVec3(vert_pool + idx1 * 12); + const v2 = loadVec3(vert_pool + idx2 * 12); + const ray_origin = loadVec3(ray_ptr); + const ray_dir = loadVec3(ray_ptr + 12); + + const edge1 = v1 - v0; + const edge2 = v2 - v0; + const pvec = cross(ray_dir, edge2); + const det = dot3(edge1, pvec); + + if (det > -1e-6 and det < 1e-6) return 0; + + const tvec = ray_origin - v0; + const u_raw = dot3(tvec, pvec); + + const det_neg_eps = det * neg_eps; + const det_one_plus = det * one_plus_eps; + if (det > 0) { + if (u_raw < det_neg_eps or u_raw > det_one_plus) return 0; + } else { + if (u_raw > det_neg_eps or u_raw < det_one_plus) return 0; + } + + const qvec = cross(tvec, edge1); + const v_raw = dot3(ray_dir, qvec); + + if (det > 0) { + if (v_raw < det_neg_eps or (u_raw + v_raw) > det_one_plus) return 0; + } else { + if (v_raw > det_neg_eps or (u_raw + v_raw) < det_one_plus) return 0; + } + + // Hit confirmed -- only divide now + const inv_det = 1.0 / det; + if (t_out != 0) { + @as(*align(1) f32, @ptrFromInt(t_out)).* = dot3(edge2, qvec) * inv_det; + } + if (normal_out != 0) { + @as(*align(1) f32, @ptrFromInt(normal_out)).* = u_raw * inv_det; + @as(*align(1) f32, @ptrFromInt(normal_out + 4)).* = v_raw * inv_det; + } + return 1; +} + +fn readI32(addr: u32) i32 { + return @as(*align(1) const i32, @ptrFromInt(addr)).*; +} + // ============================================================================= // Game hook exports // ============================================================================= diff --git a/src/performance/weirdperformance.zig b/src/performance/weirdperformance.zig index 0dbfff6..90732ae 100644 --- a/src/performance/weirdperformance.zig +++ b/src/performance/weirdperformance.zig @@ -154,6 +154,7 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void { // cull_sse.zig extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; // silicon_sse.zig const sse = struct { @@ -225,6 +226,7 @@ fn installPatches() u32 { .{ .target = 0x686000, .replacement = @intFromPtr(&sse.si_frustumCullBBox), .name = "frustumCullBBox" }, .{ .target = 0x6B8C60, .replacement = @intFromPtr(&performSpatialCulling), .name = "PerformSpatialCulling" }, .{ .target = 0x6B88E0, .replacement = @intFromPtr(&performCollisionDetectionSSE), .name = "performCollisionDetection" }, + .{ .target = 0x7C2C40, .replacement = @intFromPtr(&rayTriIntersectIndexedInt), .name = "rayTriIndexedInt" }, }; var count: u32 = 0; diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index 3cd6870..1b1933c 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -29,6 +29,7 @@ extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} } extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; +extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern var stride_info: [8]u32; // exported from particle_sse.zig @@ -206,6 +207,10 @@ const ProfState = struct { cylfrust_cycles: u64 = 0, rendersph_calls: u64 = 0, // RenderSphere (0x6b92b0) -- parent of SetupCylinderFrustum rendersph_cycles: u64 = 0, + raytri_int_calls: u64 = 0, // ray_triangle_intersection_indexed_int (0x7c2c40) + raytri_int_cycles: u64 = 0, + staticcull_calls: u64 = 0, // ProcessStaticObjectsCulling (0x683bf0) + staticcull_cycles: u64 = 0, }; // ============================================================================= @@ -881,6 +886,8 @@ var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10 var viewfrust_hook: hook.Detour(Fn5) = .{}; // SetupViewFrustum: thiscall RET 0xC var cylfrust_hook: hook.Detour(Fn5) = .{}; // SetupCylinderFrustum: thiscall RET 0xC var rendersph_hook: hook.Detour(Fn8v) = .{}; // RenderSphere: thiscall RET 0x18 +var raytri_int_hook: hook.Detour(Fn6) = .{}; // ray_tri_indexed_int: fastcall RET 0x10 +var staticcull_hook: hook.Detour(Fn2) = .{}; // ProcessStaticObjectsCulling: fastcall RET // --- Detour functions (timing-only pass-through) --- @@ -1227,6 +1234,26 @@ fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u3 prof.rendersph_calls +|= 1; return ret; } +fn raytriIntDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + if (AB_OTHER_HOOKS and ab_use_custom) { + const ret = rayTriIntersectIndexedInt(a, b, c, d, e, f); + prof.raytri_int_cycles +|= rdtsc() - s; + prof.raytri_int_calls +|= 1; + return @ptrFromInt(@as(u32, ret)); + } + const ret = raytri_int_hook.callOriginal(.{ a, b, c, d, e, f }); + prof.raytri_int_cycles +|= rdtsc() - s; + prof.raytri_int_calls +|= 1; + return ret; +} +fn staticcullDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = staticcull_hook.callOriginal(.{ a, b }); + prof.staticcull_cycles +|= rdtsc() - s; + prof.staticcull_calls +|= 1; + return ret; +} // ============================================================================= // Hook: blit_hub (0x5a4f60) @@ -1470,6 +1497,8 @@ fn dumpStats() void { .{ .name = "viewfrust", .cycles = prof.viewfrust_cycles, .calls = prof.viewfrust_calls }, .{ .name = "cylfrust", .cycles = prof.cylfrust_cycles, .calls = prof.cylfrust_calls }, .{ .name = "rendersph", .cycles = prof.rendersph_cycles, .calls = prof.rendersph_calls }, + .{ .name = "raytri_int", .cycles = prof.raytri_int_cycles, .calls = prof.raytri_int_calls }, + .{ .name = "staticcull", .cycles = prof.staticcull_cycles, .calls = prof.staticcull_calls }, }; for (hotspots) |h| { if (h.calls > 0) { @@ -1629,6 +1658,9 @@ pub fn installHooks() void { _ = viewfrust_hook.attach(0x6bc1c0, &viewfrustDetour); _ = cylfrust_hook.attach(0x6bc370, &cylfrustDetour); _ = rendersph_hook.attach(0x6b92b0, &rendersphDetour); + // raytri_int graduated to weirdperformance JMP patch + // _ = raytri_int_hook.attach(0x7c2c40, &raytriIntDetour); + _ = staticcull_hook.attach(0x683bf0, &staticcullDetour); // Timer calibration now handled by performance module. @@ -1709,6 +1741,8 @@ pub fn removeHooks() void { viewfrust_hook.detach(); cylfrust_hook.detach(); rendersph_hook.detach(); + raytri_int_hook.detach(); + staticcull_hook.detach(); blit_hub_hook.detach(); log.close(); mod_mutex.release(&g_mutex);