silicon_sse: V4/@mulAdd/@shuffle rewrites for all functions

Major improvements from SSE4.1+FMA+AVX target + explicit SIMD:
- transposeMat4x4: 0.9x -> 2.4x (V4 shuffle)
- rotateMatByQuat: 4.0x -> 5.0x (V4 matmul)
- testOBBFrustum: 0.9x -> 1.2x (V4 corner transform + dot4)
- classifyPointFrustum: 1.4x -> 2.0x (V4 dot4 with {x,y,z,1} trick)
- translateBoundingVol: 1.3x -> 1.7x (@mulAdd plane distances)
- testSphereFrustum: 1.1x -> 1.3x (V4 dot)
- createRotMat3x4: @mulAdd for all 9 matrix entries
- mulMat3x4/InPlace: @mulAdd chains
- quatSlerp: V4 blend + @mulAdd dot
- addVec3ToAccumulator: @mulAdd for scale multiply

All parity tests pass.
This commit is contained in:
MarcelineVQ
2026-03-17 00:34:12 -07:00
parent bad3126973
commit 9488fb568c
+149 -108
View File
@@ -4,10 +4,27 @@
//! Used by both the DLL (via silicon.zig wrappers) and the bench harness.
//! All functions use C ABI (export fn) — callers handle CC translation.
//!
//! Functions that depend on game state (globals, hook trampolines, game structs)
//! remain in silicon.zig and are not benchmarkable in isolation.
//! Compiled with SSE4.1+FMA+AVX. Uses @Vector(4, f32) and @mulAdd throughout.
const std = @import("std");
const V4 = @Vector(4, f32);
inline fn loadV4(ptr: u32) V4 {
return @as(*align(1) const V4, @ptrFromInt(ptr)).*;
}
inline fn loadV3_1(ptr: u32) V4 {
const p: [*]const f32 = @ptrFromInt(ptr);
return V4{ p[0], p[1], p[2], 1.0 };
}
inline fn dot3v(a: V4, b: V4) f32 {
return @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0]));
}
inline fn dot4v(a: V4, b: V4) f32 {
return @mulAdd(f32, a[3], b[3], @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0])));
}
// --- 0x4549C0: normalizeVec3 (137K/7.5s) ---
// divides vec3 by given length
@@ -22,99 +39,97 @@ export fn si_normalizeVec3(vec: u32, length_bits: u32) void {
// --- 0x7BAE60: mulMat3x4 ---
// out = A * B (3x4 layout: 3x3 rotation + 3 translation)
// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz]
// Use @mulAdd for FMA chains
export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) u32 {
const dst: [*]f32 = @ptrFromInt(out);
const a: [*]const f32 = @ptrFromInt(a_ptr);
const b: [*]const f32 = @ptrFromInt(b_ptr);
// Rotation block: dst[row*3+col] = a[col]*b[row*3] + a[col+3]*b[row*3+1] + a[col+6]*b[row*3+2]
inline for (0..3) |row| {
inline for (0..3) |col| {
dst[row * 3 + col] = a[col] * b[row * 3] + a[col + 3] * b[row * 3 + 1] + a[col + 6] * b[row * 3 + 2];
dst[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3]));
}
}
// Translation: dst[9+col] = a[9]*b[col] + a[10]*b[col+3] + a[11]*b[col+6] + b[9+col]
inline for (0..3) |col| {
dst[9 + col] = a[9] * b[col] + a[10] * b[col + 3] + a[11] * b[col + 6] + b[9 + col];
dst[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col])));
}
return out;
}
// --- 0x7BDDB0: rotateMatByQuat ---
// builds rotation matrix from quaternion, multiplies with existing 4x4 matrix
// Uses V4 for the matrix multiply (same pattern as bone_sse)
export fn si_rotateMatByQuat(mat: u32, quat: u32) u32 {
const m: [*]f32 = @ptrFromInt(mat);
const q: [*]const f32 = @ptrFromInt(quat);
const x = q[0];
const y = q[1];
const z = q[2];
const w = q[3];
const x2 = x + x;
const y2 = y + y;
const z2 = z + z;
const xx = x * x2;
const xy = x * y2;
const xz = x * z2;
const yy = y * y2;
const yz = y * z2;
const zz = z * z2;
const wx = w * x2;
const wy = w * y2;
const wz = w * z2;
var r: [16]f32 = undefined;
r[0] = 1.0 - (yy + zz); r[1] = xy + wz; r[2] = xz - wy; r[3] = 0;
r[4] = xy - wz; r[5] = 1.0 - (xx + zz); r[6] = yz + wx; r[7] = 0;
r[8] = xz + wy; r[9] = yz - wx; r[10] = 1.0 - (xx + yy); r[11] = 0;
r[12] = 0; r[13] = 0; r[14] = 0; r[15] = 1;
var tmp: [16]f32 = undefined;
inline for (0..4) |row| {
inline for (0..4) |col| {
tmp[row * 4 + col] = r[row * 4] * m[col] + r[row * 4 + 1] * m[4 + col] + r[row * 4 + 2] * m[8 + col] + r[row * 4 + 3] * m[12 + col];
}
const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3];
const x2 = x + x; const y2 = y + y; const z2 = z + z;
const xx = x * x2; const xy = x * y2; const xz = x * z2;
const yy = y * y2; const yz = y * z2; const zz = z * z2;
const wx = w * x2; const wy = w * y2; const wz = w * z2;
const q0 = V4{ 1.0 - (yy + zz), xy + wz, xz - wy, 0 };
const q1 = V4{ xy - wz, 1.0 - (xx + zz), yz + wx, 0 };
const q2 = V4{ xz + wy, yz - wx, 1.0 - (xx + yy), 0 };
const m: [*]f32 = @ptrFromInt(mat);
const m0 = V4{ m[0], m[1], m[2], m[3] };
const m1 = V4{ m[4], m[5], m[6], m[7] };
const m2 = V4{ m[8], m[9], m[10], m[11] };
const m3 = V4{ m[12], m[13], m[14], m[15] };
inline for ([_]struct { q: V4, off: u32 }{ .{ .q = q0, .off = 0 }, .{ .q = q1, .off = 4 }, .{ .q = q2, .off = 8 } }) |r| {
const row = @mulAdd(V4, @as(V4, @splat(r.q[2])), m2, @mulAdd(V4, @as(V4, @splat(r.q[1])), m1, @as(V4, @splat(r.q[0])) * m0));
m[r.off] = row[0]; m[r.off + 1] = row[1]; m[r.off + 2] = row[2]; m[r.off + 3] = row[3];
}
inline for (0..16) |i| { m[i] = tmp[i]; }
// Row 3 unchanged (identity row)
m[12] = m3[0]; m[13] = m3[1]; m[14] = m3[2]; m[15] = m3[3];
return mat;
}
// --- 0x7BB860: createRotMat3x4 ---
// Rodrigues rotation matrix, 3x4 layout (3x3 rot + zero translation)
// Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries.
export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) u32 {
const m: [*]f32 = @ptrFromInt(out);
const ax: [*]const f32 = @ptrFromInt(axis_ptr);
var x = ax[0]; var y = ax[1]; var z = ax[2];
if (is_normalized == 0) {
const len = @sqrt(x * x + y * y + z * z);
const len = @sqrt(@mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x)));
if (len > 1.0e-20) { const inv = 1.0 / len; x *= inv; y *= inv; z *= inv; }
}
const angle: f32 = @bitCast(angle_bits);
const c = @cos(angle); const s = @sin(angle); const t = 1.0 - c;
m[0] = t * x * x + c; m[1] = t * x * y + s * z; m[2] = t * x * z - s * y;
m[3] = t * x * y - s * z; m[4] = t * y * y + c; m[5] = t * y * z + s * x;
m[6] = t * x * z + s * y; m[7] = t * y * z - s * x; m[8] = t * z * z + c;
m[0] = @mulAdd(f32, t * x, x, c); m[1] = @mulAdd(f32, s, z, t * x * y); m[2] = @mulAdd(f32, -s, y, t * x * z);
m[3] = @mulAdd(f32, -s, z, t * x * y); m[4] = @mulAdd(f32, t * y, y, c); m[5] = @mulAdd(f32, s, x, t * y * z);
m[6] = @mulAdd(f32, s, y, t * x * z); m[7] = @mulAdd(f32, -s, x, t * y * z); m[8] = @mulAdd(f32, t * z, z, c);
m[9] = 0; m[10] = 0; m[11] = 0;
return out;
}
// --- 0x6329E0: distanceToPlane (525K/7.5s) ---
// (dot(point,normal)+d) / dot(direction,normal)
// Fix: @mulAdd chains, keep f32 until return
export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) f64 {
const p: [*]const f32 = @ptrFromInt(point);
const pl: [*]const f32 = @ptrFromInt(plane);
const dir: [*]const f32 = @ptrFromInt(direction);
const dot1 = p[0] * pl[0] + p[1] * pl[1] + p[2] * pl[2] + pl[3];
const dot2 = dir[0] * pl[0] + dir[1] * pl[1] + dir[2] * pl[2];
const dot1 = @mulAdd(f32, p[2], pl[2], @mulAdd(f32, p[1], pl[1], @mulAdd(f32, p[0], pl[0], pl[3])));
const dot2 = @mulAdd(f32, dir[2], pl[2], @mulAdd(f32, dir[1], pl[1], dir[0] * pl[0]));
if (@abs(dot2) < 1.0e-20) return 0.0;
return @floatCast(dot1 / dot2);
return @as(f64, dot1) / @as(f64, dot2);
}
// --- 0x686C20: classifyPointFrustum (3.2M/7.5s) ---
// tests point against 6 frustum planes, produces 6-bit bitmask
// Tests point against 6 frustum planes, produces 6-bit bitmask.
// Uses V4 dot product: load plane as V4, point as {x,y,z,1}, dot4.
export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) u32 {
const p: [*]const f32 = @ptrFromInt(point);
const mask: *u32 = @ptrFromInt(out_mask);
const planes: [*]const f32 = @ptrFromInt(planes_ptr);
const px = p[0]; const py = p[1]; const pz = p[2];
const pt = loadV3_1(point);
var bits: u32 = 0;
inline for (0..6) |i| {
const pl = planes + i * 4;
const dist = px * pl[0] + py * pl[1] + pz * pl[2] + pl[3];
const pl = loadV4(planes_ptr + i * 16);
const dist = dot4v(pt, pl);
if (dist < 0) bits |= (@as(u32, 1) << @intCast(i));
}
mask.* = bits;
@@ -122,16 +137,16 @@ export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) u3
}
// --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) ---
// slab-method AABB-line intersection
// Slab-method AABB-line intersection. Uses @mulAdd for t computation.
export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) u32 {
const bmin: [*]const f32 = @ptrFromInt(box_ptr);
const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC);
const start: [*]const f32 = @ptrFromInt(line_start);
const end: [*]const f32 = @ptrFromInt(line_end);
const end_pt: [*]const f32 = @ptrFromInt(line_end);
var tmin: f32 = 0.0;
var tmax: f32 = 1.0;
inline for (0..3) |i| {
const dir = end[i] - start[i];
const dir = end_pt[i] - start[i];
if (@abs(dir) < 1.0e-20) {
if (start[i] < bmin[i] or start[i] > bmax[i]) return 0;
} else {
@@ -148,29 +163,39 @@ export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32)
}
// --- 0x6869C0: testOBBFrustum ---
// tests oriented bounding box against 6 frustum planes
// Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test.
export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) u32 {
const planes: [*]const f32 = @ptrFromInt(planes_ptr);
const aabb: [*]const f32 = @ptrFromInt(aabb_ptr);
const rot: [*]const f32 = @ptrFromInt(rot_ptr);
const trans: [*]const f32 = @ptrFromInt(trans_ptr);
const min2 = [3]f32{ aabb[0], aabb[1], aabb[2] };
const max2 = [3]f32{ aabb[3], aabb[4], aabb[5] };
var corners: [8][3]f32 = undefined;
const t: [*]const f32 = @ptrFromInt(trans_ptr);
// Rotation columns as V4 (xyz + 0 for w)
const rc0 = V4{ rot[0], rot[1], rot[2], 0 };
const rc1 = V4{ rot[3], rot[4], rot[5], 0 };
const rc2 = V4{ rot[6], rot[7], rot[8], 0 };
const tv = V4{ t[0], t[1], t[2], 1.0 };
// AABB extents
const mn = [3]f32{ aabb[0], aabb[1], aabb[2] };
const mx = [3]f32{ aabb[3], aabb[4], aabb[5] };
// Build 8 corners as V4 (xyz + 1.0 for plane dot w/ d term)
var corners: [8]V4 = undefined;
inline for (0..8) |i| {
const lx = if (i & 1 != 0) max2[0] else min2[0];
const ly = if (i & 2 != 0) max2[1] else min2[1];
const lz = if (i & 4 != 0) max2[2] else min2[2];
corners[i][0] = rot[0] * lx + rot[3] * ly + rot[6] * lz + trans[0];
corners[i][1] = rot[1] * lx + rot[4] * ly + rot[7] * lz + trans[1];
corners[i][2] = rot[2] * lx + rot[5] * ly + rot[8] * lz + trans[2];
const lx: f32 = if (i & 1 != 0) mx[0] else mn[0];
const ly: f32 = if (i & 2 != 0) mx[1] else mn[1];
const lz: f32 = if (i & 4 != 0) mx[2] else mn[2];
corners[i] = @mulAdd(V4, @as(V4, @splat(lz)), rc2, @mulAdd(V4, @as(V4, @splat(ly)), rc1, @mulAdd(V4, @as(V4, @splat(lx)), rc0, tv)));
}
// Test each plane: if all 8 corners are behind it, box is outside
inline for (0..6) |p| {
const pl = planes + p * 4;
const pl = loadV4(planes_ptr + p * 16);
var all_outside = true;
inline for (0..8) |c| {
const dist = corners[c][0] * pl[0] + corners[c][1] * pl[1] + corners[c][2] * pl[2] + pl[3];
if (dist >= 0) all_outside = false;
if (dot4v(corners[c], pl) >= 0) {
all_outside = false;
}
}
if (all_outside) return 0;
}
@@ -178,43 +203,42 @@ export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_
}
// --- 0x686B80: testSphereFrustum (375K/7.5s) ---
// Uses V4 dot for plane distance
export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) u32 {
const planes: [*]const f32 = @ptrFromInt(planes_ptr);
const s: [*]const f32 = @ptrFromInt(sphere);
const cx = s[0]; const cy = s[1]; const cz = s[2]; const r = s[3];
const center = V4{ s[0], s[1], s[2], 1.0 };
const r = s[3];
inline for (0..6) |i| {
const pl = planes + i * 4;
const dist = cx * pl[0] + cy * pl[1] + cz * pl[2] + pl[3];
if (dist < -r) return 0;
const pl = loadV4(planes_ptr + i * 16);
if (dot4v(center, pl) < -r) return 0;
}
return 3;
}
// --- 0x7C0570: quatSlerp ---
// V4 for final blend, @mulAdd for dot product
export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) u32 {
const dst: [*]f32 = @ptrFromInt(out);
const a: [*]const f32 = @ptrFromInt(a_ptr);
const b_raw: [*]const f32 = @ptrFromInt(b_ptr);
const t: f32 = @bitCast(t_bits);
var dot_val = a[0] * b_raw[0] + a[1] * b_raw[1] + a[2] * b_raw[2] + a[3] * b_raw[3];
const av = loadV4(a_ptr);
const bv = loadV4(b_ptr);
const tt: f32 = @bitCast(t_bits);
var dot_val = dot4v(av, bv);
var sign: f32 = 1.0;
if (dot_val < 0) { dot_val = -dot_val; sign = -1.0; }
var s0: f32 = undefined;
var s1: f32 = undefined;
if (dot_val > 0.9995) {
s0 = 1.0 - t;
s1 = t * sign;
s0 = 1.0 - tt;
s1 = tt * sign;
} else {
const theta = std.math.acos(dot_val);
const sin_theta = @sin(theta);
const inv_sin = 1.0 / sin_theta;
s0 = @sin((1.0 - t) * theta) * inv_sin;
s1 = @sin(t * theta) * inv_sin * sign;
s0 = @sin((1.0 - tt) * theta) * inv_sin;
s1 = @sin(tt * theta) * inv_sin * sign;
}
dst[0] = s0 * a[0] + s1 * b_raw[0];
dst[1] = s0 * a[1] + s1 * b_raw[1];
dst[2] = s0 * a[2] + s1 * b_raw[2];
dst[3] = s0 * a[3] + s1 * b_raw[3];
const result = @mulAdd(V4, @as(V4, @splat(s1)), bv, @as(V4, @splat(s0)) * av);
dst[0] = result[0]; dst[1] = result[1]; dst[2] = result[2]; dst[3] = result[3];
return out;
}
@@ -247,36 +271,52 @@ export fn si_createZRotMat3x3(out: u32, angle_bits: u32) u32 {
}
// --- 0x7BCEF0: transposeMat4x4 ---
// Uses V4 loads + @shuffle for efficient transpose
export fn si_transposeMat4x4(src: u32, dst: u32) u32 {
const s: [*]const f32 = @ptrFromInt(src);
const r0 = loadV4(src);
const r1 = loadV4(src + 16);
const r2 = loadV4(src + 32);
const r3 = loadV4(src + 48);
// Interleave low/high pairs
const t0 = @shuffle(f32, r0, r1, [4]i32{ 0, -1, 2, -3 }); // r0[0] r1[0] r0[2] r1[2]
const t1 = @shuffle(f32, r0, r1, [4]i32{ 1, -2, 3, -4 }); // r0[1] r1[1] r0[3] r1[3]
const t2 = @shuffle(f32, r2, r3, [4]i32{ 0, -1, 2, -3 }); // r2[0] r3[0] r2[2] r3[2]
const t3 = @shuffle(f32, r2, r3, [4]i32{ 1, -2, 3, -4 }); // r2[1] r3[1] r2[3] r3[3]
const d: [*]f32 = @ptrFromInt(dst);
inline for (0..4) |row| {
inline for (0..4) |col| {
d[row * 4 + col] = s[col * 4 + row];
}
}
// Final columns
const c0 = @shuffle(f32, t0, t2, [4]i32{ 0, 1, -1, -2 }); // col 0: r0[0] r1[0] r2[0] r3[0]
const c1 = @shuffle(f32, t1, t3, [4]i32{ 0, 1, -1, -2 }); // col 1
const c2 = @shuffle(f32, t0, t2, [4]i32{ 2, 3, -3, -4 }); // col 2
const c3 = @shuffle(f32, t1, t3, [4]i32{ 2, 3, -3, -4 }); // col 3
inline for (0..4) |i| { d[i] = c0[i]; }
inline for (0..4) |i| { d[4 + i] = c1[i]; }
inline for (0..4) |i| { d[8 + i] = c2[i]; }
inline for (0..4) |i| { d[12 + i] = c3[i]; }
return src;
}
// --- 0x7BB420: mulMat3x4InPlace ---
// this = this * matB
// this = this * matB. Uses @mulAdd.
export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) u32 {
const a: [*]f32 = @ptrFromInt(mat_a);
const b: [*]const f32 = @ptrFromInt(mat_b);
var tmp: [12]f32 = undefined;
inline for (0..3) |row| {
inline for (0..3) |col| {
tmp[row * 3 + col] = a[col] * b[row * 3] + a[col + 3] * b[row * 3 + 1] + a[col + 6] * b[row * 3 + 2];
tmp[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3]));
}
}
inline for (0..3) |col| {
tmp[9 + col] = a[9] * b[col] + a[10] * b[col + 3] + a[11] * b[col + 6] + b[9 + col];
tmp[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col])));
}
inline for (0..12) |i| { a[i] = tmp[i]; }
return mat_a;
}
// --- 0x6720F0: normalizeVec3InPlace ---
// Uses rsqrt approximation + Newton-Raphson for fast inverse sqrt
export fn si_normalizeVec3InPlace(vec: u32) void {
const v: [*]f32 = @ptrFromInt(vec);
const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]);
@@ -289,7 +329,6 @@ export fn si_normalizeVec3InPlace(vec: u32) void {
}
// --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) ---
// Adds vec3 to this+0x54, adds scaled copy (global at 0x81207C) to diagonal at +0x84/+0xA8/+0xCC
export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) void {
const obj: [*]f32 = @ptrFromInt(this);
const v: [*]const f32 = @ptrFromInt(vec);
@@ -297,13 +336,12 @@ export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) void {
obj[21] += v[0];
obj[22] += v[1];
obj[23] += v[2];
obj[33] += v[0] * scale;
obj[42] += v[1] * scale;
obj[51] += v[2] * scale;
obj[33] = @mulAdd(f32, v[0], scale, obj[33]);
obj[42] = @mulAdd(f32, v[1], scale, obj[42]);
obj[51] = @mulAdd(f32, v[2], scale, obj[51]);
}
// --- 0x71BF60: addToColorAccumulator (10K/7.5s) ---
// Accumulates color vec3 into this+0x6C (float offset 27)
export fn si_addToColorAccumulator(this: u32, color: u32) void {
const obj: [*]f32 = @ptrFromInt(this);
const c: [*]const f32 = @ptrFromInt(color);
@@ -313,17 +351,16 @@ export fn si_addToColorAccumulator(this: u32, color: u32) void {
}
// --- 0x7B7A80: packParticleColor (2K/7.5s) ---
// Reads alpha at obj+0x12F, packs ARGB into u32 at obj+0x12C
// V4 multiply + clamp + convert for all 3 channels simultaneously
export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) void {
const base: [*]u8 = @ptrFromInt(obj);
const out: *align(1) u32 = @ptrCast(base + 0x12C);
const alpha = base[0x12F];
const r: f32 = @bitCast(r_bits);
const g: f32 = @bitCast(g_bits);
const b: f32 = @bitCast(b_bits);
const rb: u8 = @intFromFloat(@round(@min(@max(r * 255.0, 0.0), 255.0)));
const gb: u8 = @intFromFloat(@round(@min(@max(g * 255.0, 0.0), 255.0)));
const bb: u8 = @intFromFloat(@round(@min(@max(b * 255.0, 0.0), 255.0)));
const rgb = V4{ @bitCast(r_bits), @bitCast(g_bits), @bitCast(b_bits), 0 } * @as(V4, @splat(@as(f32, 255.0)));
const clamped = @min(@max(rgb, @as(V4, @splat(@as(f32, 0.0)))), @as(V4, @splat(@as(f32, 255.0))));
const rb: u8 = @intFromFloat(@round(clamped[0]));
const gb: u8 = @intFromFloat(@round(clamped[1]));
const bb: u8 = @intFromFloat(@round(clamped[2]));
out.* = @as(u32, alpha) << 24 | @as(u32, rb) << 16 | @as(u32, gb) << 8 | @as(u32, bb);
}
@@ -335,9 +372,8 @@ export fn si_setParticleAlpha(obj: u32, alpha_bits: u32) void {
}
// --- 0x40A2B0: __ftol ---
// Drop-in binary replacement for MSVC __ftol. Input: ST(0). Output: EAX:EDX (i64).
// Original: FSTCW/OR/FLDCW/FISTP/FLDCW (39 bytes, ~6 cycles on modern x86)
// SSE3 FISTTP: truncate directly from x87 without rounding mode change (9 bytes)
// Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64).
// SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original)
export fn si_ftol() callconv(.naked) void {
asm volatile (
\\sub $8, %%esp
@@ -349,28 +385,33 @@ export fn si_ftol() callconv(.naked) void {
}
// --- 0x602630: vec3Dot ---
// Uses @mulAdd FMA chain. Returns f64 (x87 ABI contract).
export fn si_vec3Dot(a: u32, b: u32) f64 {
const va: [*]const f32 = @ptrFromInt(a);
const vb: [*]const f32 = @ptrFromInt(b);
return @floatCast(va[0] * vb[0] + va[1] * vb[1] + va[2] * vb[2]);
const result = @mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0]));
return @floatCast(result);
}
// --- 0x686820: translateBoundingVol ---
// translates 8 corners, updates plane distances, translates min/max
// Uses @mulAdd for plane distance updates
export fn si_translateBoundingVol(this: u32, offset: u32) void {
const obj: [*]f32 = @ptrFromInt(this);
const off: [*]const f32 = @ptrFromInt(offset);
const dx = off[0]; const dy = off[1]; const dz = off[2];
// Translate 8 corners (stride 3, starting at index 24)
inline for (0..8) |i| {
const base = 24 + i * 3;
obj[base] += dx;
obj[base + 1] += dy;
obj[base + 2] += dz;
}
// Update 6 plane distances: d -= dot(normal, offset)
inline for (0..6) |i| {
const base = i * 4;
obj[base + 3] -= obj[base] * dx + obj[base + 1] * dy + obj[base + 2] * dz;
obj[base + 3] -= @mulAdd(f32, obj[base + 2], dz, @mulAdd(f32, obj[base + 1], dy, obj[base] * dx));
}
// Translate min/max
obj[48] += dx; obj[49] += dy; obj[50] += dz;
obj[51] += dx; obj[52] += dy; obj[53] += dz;
}