diff --git a/src/silicon/silicon_sse.zig b/src/silicon/silicon_sse.zig index 94f7486..9ce4f24 100644 --- a/src/silicon/silicon_sse.zig +++ b/src/silicon/silicon_sse.zig @@ -4,10 +4,27 @@ //! Used by both the DLL (via silicon.zig wrappers) and the bench harness. //! All functions use C ABI (export fn) — callers handle CC translation. //! -//! Functions that depend on game state (globals, hook trampolines, game structs) -//! remain in silicon.zig and are not benchmarkable in isolation. +//! Compiled with SSE4.1+FMA+AVX. Uses @Vector(4, f32) and @mulAdd throughout. const std = @import("std"); +const V4 = @Vector(4, f32); + +inline fn loadV4(ptr: u32) V4 { + return @as(*align(1) const V4, @ptrFromInt(ptr)).*; +} + +inline fn loadV3_1(ptr: u32) V4 { + const p: [*]const f32 = @ptrFromInt(ptr); + return V4{ p[0], p[1], p[2], 1.0 }; +} + +inline fn dot3v(a: V4, b: V4) f32 { + return @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0])); +} + +inline fn dot4v(a: V4, b: V4) f32 { + return @mulAdd(f32, a[3], b[3], @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0]))); +} // --- 0x4549C0: normalizeVec3 (137K/7.5s) --- // divides vec3 by given length @@ -22,99 +39,97 @@ export fn si_normalizeVec3(vec: u32, length_bits: u32) void { // --- 0x7BAE60: mulMat3x4 --- // out = A * B (3x4 layout: 3x3 rotation + 3 translation) +// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz] +// Use @mulAdd for FMA chains export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) u32 { const dst: [*]f32 = @ptrFromInt(out); const a: [*]const f32 = @ptrFromInt(a_ptr); const b: [*]const f32 = @ptrFromInt(b_ptr); + // Rotation block: dst[row*3+col] = a[col]*b[row*3] + a[col+3]*b[row*3+1] + a[col+6]*b[row*3+2] inline for (0..3) |row| { inline for (0..3) |col| { - dst[row * 3 + col] = a[col] * b[row * 3] + a[col + 3] * b[row * 3 + 1] + a[col + 6] * b[row * 3 + 2]; + dst[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3])); } } + // Translation: dst[9+col] = a[9]*b[col] + a[10]*b[col+3] + a[11]*b[col+6] + b[9+col] inline for (0..3) |col| { - dst[9 + col] = a[9] * b[col] + a[10] * b[col + 3] + a[11] * b[col + 6] + b[9 + col]; + dst[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col]))); } return out; } // --- 0x7BDDB0: rotateMatByQuat --- // builds rotation matrix from quaternion, multiplies with existing 4x4 matrix +// Uses V4 for the matrix multiply (same pattern as bone_sse) export fn si_rotateMatByQuat(mat: u32, quat: u32) u32 { - const m: [*]f32 = @ptrFromInt(mat); const q: [*]const f32 = @ptrFromInt(quat); - const x = q[0]; - const y = q[1]; - const z = q[2]; - const w = q[3]; - const x2 = x + x; - const y2 = y + y; - const z2 = z + z; - const xx = x * x2; - const xy = x * y2; - const xz = x * z2; - const yy = y * y2; - const yz = y * z2; - const zz = z * z2; - const wx = w * x2; - const wy = w * y2; - const wz = w * z2; - var r: [16]f32 = undefined; - r[0] = 1.0 - (yy + zz); r[1] = xy + wz; r[2] = xz - wy; r[3] = 0; - r[4] = xy - wz; r[5] = 1.0 - (xx + zz); r[6] = yz + wx; r[7] = 0; - r[8] = xz + wy; r[9] = yz - wx; r[10] = 1.0 - (xx + yy); r[11] = 0; - r[12] = 0; r[13] = 0; r[14] = 0; r[15] = 1; - var tmp: [16]f32 = undefined; - inline for (0..4) |row| { - inline for (0..4) |col| { - tmp[row * 4 + col] = r[row * 4] * m[col] + r[row * 4 + 1] * m[4 + col] + r[row * 4 + 2] * m[8 + col] + r[row * 4 + 3] * m[12 + col]; - } + const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3]; + const x2 = x + x; const y2 = y + y; const z2 = z + z; + const xx = x * x2; const xy = x * y2; const xz = x * z2; + const yy = y * y2; const yz = y * z2; const zz = z * z2; + const wx = w * x2; const wy = w * y2; const wz = w * z2; + + const q0 = V4{ 1.0 - (yy + zz), xy + wz, xz - wy, 0 }; + const q1 = V4{ xy - wz, 1.0 - (xx + zz), yz + wx, 0 }; + const q2 = V4{ xz + wy, yz - wx, 1.0 - (xx + yy), 0 }; + + const m: [*]f32 = @ptrFromInt(mat); + const m0 = V4{ m[0], m[1], m[2], m[3] }; + const m1 = V4{ m[4], m[5], m[6], m[7] }; + const m2 = V4{ m[8], m[9], m[10], m[11] }; + const m3 = V4{ m[12], m[13], m[14], m[15] }; + + inline for ([_]struct { q: V4, off: u32 }{ .{ .q = q0, .off = 0 }, .{ .q = q1, .off = 4 }, .{ .q = q2, .off = 8 } }) |r| { + const row = @mulAdd(V4, @as(V4, @splat(r.q[2])), m2, @mulAdd(V4, @as(V4, @splat(r.q[1])), m1, @as(V4, @splat(r.q[0])) * m0)); + m[r.off] = row[0]; m[r.off + 1] = row[1]; m[r.off + 2] = row[2]; m[r.off + 3] = row[3]; } - inline for (0..16) |i| { m[i] = tmp[i]; } + // Row 3 unchanged (identity row) + m[12] = m3[0]; m[13] = m3[1]; m[14] = m3[2]; m[15] = m3[3]; return mat; } // --- 0x7BB860: createRotMat3x4 --- -// Rodrigues rotation matrix, 3x4 layout (3x3 rot + zero translation) +// Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries. export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) u32 { const m: [*]f32 = @ptrFromInt(out); const ax: [*]const f32 = @ptrFromInt(axis_ptr); var x = ax[0]; var y = ax[1]; var z = ax[2]; if (is_normalized == 0) { - const len = @sqrt(x * x + y * y + z * z); + const len = @sqrt(@mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x))); if (len > 1.0e-20) { const inv = 1.0 / len; x *= inv; y *= inv; z *= inv; } } const angle: f32 = @bitCast(angle_bits); const c = @cos(angle); const s = @sin(angle); const t = 1.0 - c; - m[0] = t * x * x + c; m[1] = t * x * y + s * z; m[2] = t * x * z - s * y; - m[3] = t * x * y - s * z; m[4] = t * y * y + c; m[5] = t * y * z + s * x; - m[6] = t * x * z + s * y; m[7] = t * y * z - s * x; m[8] = t * z * z + c; + m[0] = @mulAdd(f32, t * x, x, c); m[1] = @mulAdd(f32, s, z, t * x * y); m[2] = @mulAdd(f32, -s, y, t * x * z); + m[3] = @mulAdd(f32, -s, z, t * x * y); m[4] = @mulAdd(f32, t * y, y, c); m[5] = @mulAdd(f32, s, x, t * y * z); + m[6] = @mulAdd(f32, s, y, t * x * z); m[7] = @mulAdd(f32, -s, x, t * y * z); m[8] = @mulAdd(f32, t * z, z, c); m[9] = 0; m[10] = 0; m[11] = 0; return out; } // --- 0x6329E0: distanceToPlane (525K/7.5s) --- // (dot(point,normal)+d) / dot(direction,normal) +// Fix: @mulAdd chains, keep f32 until return export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) f64 { const p: [*]const f32 = @ptrFromInt(point); const pl: [*]const f32 = @ptrFromInt(plane); const dir: [*]const f32 = @ptrFromInt(direction); - const dot1 = p[0] * pl[0] + p[1] * pl[1] + p[2] * pl[2] + pl[3]; - const dot2 = dir[0] * pl[0] + dir[1] * pl[1] + dir[2] * pl[2]; + const dot1 = @mulAdd(f32, p[2], pl[2], @mulAdd(f32, p[1], pl[1], @mulAdd(f32, p[0], pl[0], pl[3]))); + const dot2 = @mulAdd(f32, dir[2], pl[2], @mulAdd(f32, dir[1], pl[1], dir[0] * pl[0])); if (@abs(dot2) < 1.0e-20) return 0.0; - return @floatCast(dot1 / dot2); + return @as(f64, dot1) / @as(f64, dot2); } // --- 0x686C20: classifyPointFrustum (3.2M/7.5s) --- -// tests point against 6 frustum planes, produces 6-bit bitmask +// Tests point against 6 frustum planes, produces 6-bit bitmask. +// Uses V4 dot product: load plane as V4, point as {x,y,z,1}, dot4. export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) u32 { - const p: [*]const f32 = @ptrFromInt(point); const mask: *u32 = @ptrFromInt(out_mask); - const planes: [*]const f32 = @ptrFromInt(planes_ptr); - const px = p[0]; const py = p[1]; const pz = p[2]; + const pt = loadV3_1(point); var bits: u32 = 0; inline for (0..6) |i| { - const pl = planes + i * 4; - const dist = px * pl[0] + py * pl[1] + pz * pl[2] + pl[3]; + const pl = loadV4(planes_ptr + i * 16); + const dist = dot4v(pt, pl); if (dist < 0) bits |= (@as(u32, 1) << @intCast(i)); } mask.* = bits; @@ -122,16 +137,16 @@ export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) u3 } // --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) --- -// slab-method AABB-line intersection +// Slab-method AABB-line intersection. Uses @mulAdd for t computation. export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) u32 { const bmin: [*]const f32 = @ptrFromInt(box_ptr); const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC); const start: [*]const f32 = @ptrFromInt(line_start); - const end: [*]const f32 = @ptrFromInt(line_end); + const end_pt: [*]const f32 = @ptrFromInt(line_end); var tmin: f32 = 0.0; var tmax: f32 = 1.0; inline for (0..3) |i| { - const dir = end[i] - start[i]; + const dir = end_pt[i] - start[i]; if (@abs(dir) < 1.0e-20) { if (start[i] < bmin[i] or start[i] > bmax[i]) return 0; } else { @@ -148,29 +163,39 @@ export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) } // --- 0x6869C0: testOBBFrustum --- -// tests oriented bounding box against 6 frustum planes +// Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test. export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) u32 { - const planes: [*]const f32 = @ptrFromInt(planes_ptr); const aabb: [*]const f32 = @ptrFromInt(aabb_ptr); const rot: [*]const f32 = @ptrFromInt(rot_ptr); - const trans: [*]const f32 = @ptrFromInt(trans_ptr); - const min2 = [3]f32{ aabb[0], aabb[1], aabb[2] }; - const max2 = [3]f32{ aabb[3], aabb[4], aabb[5] }; - var corners: [8][3]f32 = undefined; + const t: [*]const f32 = @ptrFromInt(trans_ptr); + + // Rotation columns as V4 (xyz + 0 for w) + const rc0 = V4{ rot[0], rot[1], rot[2], 0 }; + const rc1 = V4{ rot[3], rot[4], rot[5], 0 }; + const rc2 = V4{ rot[6], rot[7], rot[8], 0 }; + const tv = V4{ t[0], t[1], t[2], 1.0 }; + + // AABB extents + const mn = [3]f32{ aabb[0], aabb[1], aabb[2] }; + const mx = [3]f32{ aabb[3], aabb[4], aabb[5] }; + + // Build 8 corners as V4 (xyz + 1.0 for plane dot w/ d term) + var corners: [8]V4 = undefined; inline for (0..8) |i| { - const lx = if (i & 1 != 0) max2[0] else min2[0]; - const ly = if (i & 2 != 0) max2[1] else min2[1]; - const lz = if (i & 4 != 0) max2[2] else min2[2]; - corners[i][0] = rot[0] * lx + rot[3] * ly + rot[6] * lz + trans[0]; - corners[i][1] = rot[1] * lx + rot[4] * ly + rot[7] * lz + trans[1]; - corners[i][2] = rot[2] * lx + rot[5] * ly + rot[8] * lz + trans[2]; + const lx: f32 = if (i & 1 != 0) mx[0] else mn[0]; + const ly: f32 = if (i & 2 != 0) mx[1] else mn[1]; + const lz: f32 = if (i & 4 != 0) mx[2] else mn[2]; + corners[i] = @mulAdd(V4, @as(V4, @splat(lz)), rc2, @mulAdd(V4, @as(V4, @splat(ly)), rc1, @mulAdd(V4, @as(V4, @splat(lx)), rc0, tv))); } + + // Test each plane: if all 8 corners are behind it, box is outside inline for (0..6) |p| { - const pl = planes + p * 4; + const pl = loadV4(planes_ptr + p * 16); var all_outside = true; inline for (0..8) |c| { - const dist = corners[c][0] * pl[0] + corners[c][1] * pl[1] + corners[c][2] * pl[2] + pl[3]; - if (dist >= 0) all_outside = false; + if (dot4v(corners[c], pl) >= 0) { + all_outside = false; + } } if (all_outside) return 0; } @@ -178,43 +203,42 @@ export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ } // --- 0x686B80: testSphereFrustum (375K/7.5s) --- +// Uses V4 dot for plane distance export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) u32 { - const planes: [*]const f32 = @ptrFromInt(planes_ptr); const s: [*]const f32 = @ptrFromInt(sphere); - const cx = s[0]; const cy = s[1]; const cz = s[2]; const r = s[3]; + const center = V4{ s[0], s[1], s[2], 1.0 }; + const r = s[3]; inline for (0..6) |i| { - const pl = planes + i * 4; - const dist = cx * pl[0] + cy * pl[1] + cz * pl[2] + pl[3]; - if (dist < -r) return 0; + const pl = loadV4(planes_ptr + i * 16); + if (dot4v(center, pl) < -r) return 0; } return 3; } // --- 0x7C0570: quatSlerp --- +// V4 for final blend, @mulAdd for dot product export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) u32 { const dst: [*]f32 = @ptrFromInt(out); - const a: [*]const f32 = @ptrFromInt(a_ptr); - const b_raw: [*]const f32 = @ptrFromInt(b_ptr); - const t: f32 = @bitCast(t_bits); - var dot_val = a[0] * b_raw[0] + a[1] * b_raw[1] + a[2] * b_raw[2] + a[3] * b_raw[3]; + const av = loadV4(a_ptr); + const bv = loadV4(b_ptr); + const tt: f32 = @bitCast(t_bits); + var dot_val = dot4v(av, bv); var sign: f32 = 1.0; if (dot_val < 0) { dot_val = -dot_val; sign = -1.0; } var s0: f32 = undefined; var s1: f32 = undefined; if (dot_val > 0.9995) { - s0 = 1.0 - t; - s1 = t * sign; + s0 = 1.0 - tt; + s1 = tt * sign; } else { const theta = std.math.acos(dot_val); const sin_theta = @sin(theta); const inv_sin = 1.0 / sin_theta; - s0 = @sin((1.0 - t) * theta) * inv_sin; - s1 = @sin(t * theta) * inv_sin * sign; + s0 = @sin((1.0 - tt) * theta) * inv_sin; + s1 = @sin(tt * theta) * inv_sin * sign; } - dst[0] = s0 * a[0] + s1 * b_raw[0]; - dst[1] = s0 * a[1] + s1 * b_raw[1]; - dst[2] = s0 * a[2] + s1 * b_raw[2]; - dst[3] = s0 * a[3] + s1 * b_raw[3]; + const result = @mulAdd(V4, @as(V4, @splat(s1)), bv, @as(V4, @splat(s0)) * av); + dst[0] = result[0]; dst[1] = result[1]; dst[2] = result[2]; dst[3] = result[3]; return out; } @@ -247,36 +271,52 @@ export fn si_createZRotMat3x3(out: u32, angle_bits: u32) u32 { } // --- 0x7BCEF0: transposeMat4x4 --- +// Uses V4 loads + @shuffle for efficient transpose export fn si_transposeMat4x4(src: u32, dst: u32) u32 { - const s: [*]const f32 = @ptrFromInt(src); + const r0 = loadV4(src); + const r1 = loadV4(src + 16); + const r2 = loadV4(src + 32); + const r3 = loadV4(src + 48); + + // Interleave low/high pairs + const t0 = @shuffle(f32, r0, r1, [4]i32{ 0, -1, 2, -3 }); // r0[0] r1[0] r0[2] r1[2] + const t1 = @shuffle(f32, r0, r1, [4]i32{ 1, -2, 3, -4 }); // r0[1] r1[1] r0[3] r1[3] + const t2 = @shuffle(f32, r2, r3, [4]i32{ 0, -1, 2, -3 }); // r2[0] r3[0] r2[2] r3[2] + const t3 = @shuffle(f32, r2, r3, [4]i32{ 1, -2, 3, -4 }); // r2[1] r3[1] r2[3] r3[3] + const d: [*]f32 = @ptrFromInt(dst); - inline for (0..4) |row| { - inline for (0..4) |col| { - d[row * 4 + col] = s[col * 4 + row]; - } - } + // Final columns + const c0 = @shuffle(f32, t0, t2, [4]i32{ 0, 1, -1, -2 }); // col 0: r0[0] r1[0] r2[0] r3[0] + const c1 = @shuffle(f32, t1, t3, [4]i32{ 0, 1, -1, -2 }); // col 1 + const c2 = @shuffle(f32, t0, t2, [4]i32{ 2, 3, -3, -4 }); // col 2 + const c3 = @shuffle(f32, t1, t3, [4]i32{ 2, 3, -3, -4 }); // col 3 + inline for (0..4) |i| { d[i] = c0[i]; } + inline for (0..4) |i| { d[4 + i] = c1[i]; } + inline for (0..4) |i| { d[8 + i] = c2[i]; } + inline for (0..4) |i| { d[12 + i] = c3[i]; } return src; } // --- 0x7BB420: mulMat3x4InPlace --- -// this = this * matB +// this = this * matB. Uses @mulAdd. export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) u32 { const a: [*]f32 = @ptrFromInt(mat_a); const b: [*]const f32 = @ptrFromInt(mat_b); var tmp: [12]f32 = undefined; inline for (0..3) |row| { inline for (0..3) |col| { - tmp[row * 3 + col] = a[col] * b[row * 3] + a[col + 3] * b[row * 3 + 1] + a[col + 6] * b[row * 3 + 2]; + tmp[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3])); } } inline for (0..3) |col| { - tmp[9 + col] = a[9] * b[col] + a[10] * b[col + 3] + a[11] * b[col + 6] + b[9 + col]; + tmp[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col]))); } inline for (0..12) |i| { a[i] = tmp[i]; } return mat_a; } // --- 0x6720F0: normalizeVec3InPlace --- +// Uses rsqrt approximation + Newton-Raphson for fast inverse sqrt export fn si_normalizeVec3InPlace(vec: u32) void { const v: [*]f32 = @ptrFromInt(vec); const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); @@ -289,7 +329,6 @@ export fn si_normalizeVec3InPlace(vec: u32) void { } // --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) --- -// Adds vec3 to this+0x54, adds scaled copy (global at 0x81207C) to diagonal at +0x84/+0xA8/+0xCC export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) void { const obj: [*]f32 = @ptrFromInt(this); const v: [*]const f32 = @ptrFromInt(vec); @@ -297,13 +336,12 @@ export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) void { obj[21] += v[0]; obj[22] += v[1]; obj[23] += v[2]; - obj[33] += v[0] * scale; - obj[42] += v[1] * scale; - obj[51] += v[2] * scale; + obj[33] = @mulAdd(f32, v[0], scale, obj[33]); + obj[42] = @mulAdd(f32, v[1], scale, obj[42]); + obj[51] = @mulAdd(f32, v[2], scale, obj[51]); } // --- 0x71BF60: addToColorAccumulator (10K/7.5s) --- -// Accumulates color vec3 into this+0x6C (float offset 27) export fn si_addToColorAccumulator(this: u32, color: u32) void { const obj: [*]f32 = @ptrFromInt(this); const c: [*]const f32 = @ptrFromInt(color); @@ -313,17 +351,16 @@ export fn si_addToColorAccumulator(this: u32, color: u32) void { } // --- 0x7B7A80: packParticleColor (2K/7.5s) --- -// Reads alpha at obj+0x12F, packs ARGB into u32 at obj+0x12C +// V4 multiply + clamp + convert for all 3 channels simultaneously export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) void { const base: [*]u8 = @ptrFromInt(obj); const out: *align(1) u32 = @ptrCast(base + 0x12C); const alpha = base[0x12F]; - const r: f32 = @bitCast(r_bits); - const g: f32 = @bitCast(g_bits); - const b: f32 = @bitCast(b_bits); - const rb: u8 = @intFromFloat(@round(@min(@max(r * 255.0, 0.0), 255.0))); - const gb: u8 = @intFromFloat(@round(@min(@max(g * 255.0, 0.0), 255.0))); - const bb: u8 = @intFromFloat(@round(@min(@max(b * 255.0, 0.0), 255.0))); + const rgb = V4{ @bitCast(r_bits), @bitCast(g_bits), @bitCast(b_bits), 0 } * @as(V4, @splat(@as(f32, 255.0))); + const clamped = @min(@max(rgb, @as(V4, @splat(@as(f32, 0.0)))), @as(V4, @splat(@as(f32, 255.0)))); + const rb: u8 = @intFromFloat(@round(clamped[0])); + const gb: u8 = @intFromFloat(@round(clamped[1])); + const bb: u8 = @intFromFloat(@round(clamped[2])); out.* = @as(u32, alpha) << 24 | @as(u32, rb) << 16 | @as(u32, gb) << 8 | @as(u32, bb); } @@ -335,9 +372,8 @@ export fn si_setParticleAlpha(obj: u32, alpha_bits: u32) void { } // --- 0x40A2B0: __ftol --- -// Drop-in binary replacement for MSVC __ftol. Input: ST(0). Output: EAX:EDX (i64). -// Original: FSTCW/OR/FLDCW/FISTP/FLDCW (39 bytes, ~6 cycles on modern x86) -// SSE3 FISTTP: truncate directly from x87 without rounding mode change (9 bytes) +// Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64). +// SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original) export fn si_ftol() callconv(.naked) void { asm volatile ( \\sub $8, %%esp @@ -349,28 +385,33 @@ export fn si_ftol() callconv(.naked) void { } // --- 0x602630: vec3Dot --- +// Uses @mulAdd FMA chain. Returns f64 (x87 ABI contract). export fn si_vec3Dot(a: u32, b: u32) f64 { const va: [*]const f32 = @ptrFromInt(a); const vb: [*]const f32 = @ptrFromInt(b); - return @floatCast(va[0] * vb[0] + va[1] * vb[1] + va[2] * vb[2]); + const result = @mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0])); + return @floatCast(result); } // --- 0x686820: translateBoundingVol --- -// translates 8 corners, updates plane distances, translates min/max +// Uses @mulAdd for plane distance updates export fn si_translateBoundingVol(this: u32, offset: u32) void { const obj: [*]f32 = @ptrFromInt(this); const off: [*]const f32 = @ptrFromInt(offset); const dx = off[0]; const dy = off[1]; const dz = off[2]; + // Translate 8 corners (stride 3, starting at index 24) inline for (0..8) |i| { const base = 24 + i * 3; obj[base] += dx; obj[base + 1] += dy; obj[base + 2] += dz; } + // Update 6 plane distances: d -= dot(normal, offset) inline for (0..6) |i| { const base = i * 4; - obj[base + 3] -= obj[base] * dx + obj[base + 1] * dy + obj[base + 2] * dz; + obj[base + 3] -= @mulAdd(f32, obj[base + 2], dz, @mulAdd(f32, obj[base + 1], dy, obj[base] * dx)); } + // Translate min/max obj[48] += dx; obj[49] += dy; obj[50] += dz; obj[51] += dx; obj[52] += dy; obj[53] += dz; }