silicon_sse: V4/@mulAdd/@shuffle rewrites for all functions
Major improvements from SSE4.1+FMA+AVX target + explicit SIMD:
- transposeMat4x4: 0.9x -> 2.4x (V4 shuffle)
- rotateMatByQuat: 4.0x -> 5.0x (V4 matmul)
- testOBBFrustum: 0.9x -> 1.2x (V4 corner transform + dot4)
- classifyPointFrustum: 1.4x -> 2.0x (V4 dot4 with {x,y,z,1} trick)
- translateBoundingVol: 1.3x -> 1.7x (@mulAdd plane distances)
- testSphereFrustum: 1.1x -> 1.3x (V4 dot)
- createRotMat3x4: @mulAdd for all 9 matrix entries
- mulMat3x4/InPlace: @mulAdd chains
- quatSlerp: V4 blend + @mulAdd dot
- addVec3ToAccumulator: @mulAdd for scale multiply
All parity tests pass.
This commit is contained in:
+149
-108
@@ -4,10 +4,27 @@
|
||||
//! Used by both the DLL (via silicon.zig wrappers) and the bench harness.
|
||||
//! All functions use C ABI (export fn) — callers handle CC translation.
|
||||
//!
|
||||
//! Functions that depend on game state (globals, hook trampolines, game structs)
|
||||
//! remain in silicon.zig and are not benchmarkable in isolation.
|
||||
//! Compiled with SSE4.1+FMA+AVX. Uses @Vector(4, f32) and @mulAdd throughout.
|
||||
|
||||
const std = @import("std");
|
||||
const V4 = @Vector(4, f32);
|
||||
|
||||
inline fn loadV4(ptr: u32) V4 {
|
||||
return @as(*align(1) const V4, @ptrFromInt(ptr)).*;
|
||||
}
|
||||
|
||||
inline fn loadV3_1(ptr: u32) V4 {
|
||||
const p: [*]const f32 = @ptrFromInt(ptr);
|
||||
return V4{ p[0], p[1], p[2], 1.0 };
|
||||
}
|
||||
|
||||
inline fn dot3v(a: V4, b: V4) f32 {
|
||||
return @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0]));
|
||||
}
|
||||
|
||||
inline fn dot4v(a: V4, b: V4) f32 {
|
||||
return @mulAdd(f32, a[3], b[3], @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0])));
|
||||
}
|
||||
|
||||
// --- 0x4549C0: normalizeVec3 (137K/7.5s) ---
|
||||
// divides vec3 by given length
|
||||
@@ -22,99 +39,97 @@ export fn si_normalizeVec3(vec: u32, length_bits: u32) void {
|
||||
|
||||
// --- 0x7BAE60: mulMat3x4 ---
|
||||
// out = A * B (3x4 layout: 3x3 rotation + 3 translation)
|
||||
// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz]
|
||||
// Use @mulAdd for FMA chains
|
||||
export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) u32 {
|
||||
const dst: [*]f32 = @ptrFromInt(out);
|
||||
const a: [*]const f32 = @ptrFromInt(a_ptr);
|
||||
const b: [*]const f32 = @ptrFromInt(b_ptr);
|
||||
// Rotation block: dst[row*3+col] = a[col]*b[row*3] + a[col+3]*b[row*3+1] + a[col+6]*b[row*3+2]
|
||||
inline for (0..3) |row| {
|
||||
inline for (0..3) |col| {
|
||||
dst[row * 3 + col] = a[col] * b[row * 3] + a[col + 3] * b[row * 3 + 1] + a[col + 6] * b[row * 3 + 2];
|
||||
dst[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3]));
|
||||
}
|
||||
}
|
||||
// Translation: dst[9+col] = a[9]*b[col] + a[10]*b[col+3] + a[11]*b[col+6] + b[9+col]
|
||||
inline for (0..3) |col| {
|
||||
dst[9 + col] = a[9] * b[col] + a[10] * b[col + 3] + a[11] * b[col + 6] + b[9 + col];
|
||||
dst[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col])));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// --- 0x7BDDB0: rotateMatByQuat ---
|
||||
// builds rotation matrix from quaternion, multiplies with existing 4x4 matrix
|
||||
// Uses V4 for the matrix multiply (same pattern as bone_sse)
|
||||
export fn si_rotateMatByQuat(mat: u32, quat: u32) u32 {
|
||||
const m: [*]f32 = @ptrFromInt(mat);
|
||||
const q: [*]const f32 = @ptrFromInt(quat);
|
||||
const x = q[0];
|
||||
const y = q[1];
|
||||
const z = q[2];
|
||||
const w = q[3];
|
||||
const x2 = x + x;
|
||||
const y2 = y + y;
|
||||
const z2 = z + z;
|
||||
const xx = x * x2;
|
||||
const xy = x * y2;
|
||||
const xz = x * z2;
|
||||
const yy = y * y2;
|
||||
const yz = y * z2;
|
||||
const zz = z * z2;
|
||||
const wx = w * x2;
|
||||
const wy = w * y2;
|
||||
const wz = w * z2;
|
||||
var r: [16]f32 = undefined;
|
||||
r[0] = 1.0 - (yy + zz); r[1] = xy + wz; r[2] = xz - wy; r[3] = 0;
|
||||
r[4] = xy - wz; r[5] = 1.0 - (xx + zz); r[6] = yz + wx; r[7] = 0;
|
||||
r[8] = xz + wy; r[9] = yz - wx; r[10] = 1.0 - (xx + yy); r[11] = 0;
|
||||
r[12] = 0; r[13] = 0; r[14] = 0; r[15] = 1;
|
||||
var tmp: [16]f32 = undefined;
|
||||
inline for (0..4) |row| {
|
||||
inline for (0..4) |col| {
|
||||
tmp[row * 4 + col] = r[row * 4] * m[col] + r[row * 4 + 1] * m[4 + col] + r[row * 4 + 2] * m[8 + col] + r[row * 4 + 3] * m[12 + col];
|
||||
}
|
||||
const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3];
|
||||
const x2 = x + x; const y2 = y + y; const z2 = z + z;
|
||||
const xx = x * x2; const xy = x * y2; const xz = x * z2;
|
||||
const yy = y * y2; const yz = y * z2; const zz = z * z2;
|
||||
const wx = w * x2; const wy = w * y2; const wz = w * z2;
|
||||
|
||||
const q0 = V4{ 1.0 - (yy + zz), xy + wz, xz - wy, 0 };
|
||||
const q1 = V4{ xy - wz, 1.0 - (xx + zz), yz + wx, 0 };
|
||||
const q2 = V4{ xz + wy, yz - wx, 1.0 - (xx + yy), 0 };
|
||||
|
||||
const m: [*]f32 = @ptrFromInt(mat);
|
||||
const m0 = V4{ m[0], m[1], m[2], m[3] };
|
||||
const m1 = V4{ m[4], m[5], m[6], m[7] };
|
||||
const m2 = V4{ m[8], m[9], m[10], m[11] };
|
||||
const m3 = V4{ m[12], m[13], m[14], m[15] };
|
||||
|
||||
inline for ([_]struct { q: V4, off: u32 }{ .{ .q = q0, .off = 0 }, .{ .q = q1, .off = 4 }, .{ .q = q2, .off = 8 } }) |r| {
|
||||
const row = @mulAdd(V4, @as(V4, @splat(r.q[2])), m2, @mulAdd(V4, @as(V4, @splat(r.q[1])), m1, @as(V4, @splat(r.q[0])) * m0));
|
||||
m[r.off] = row[0]; m[r.off + 1] = row[1]; m[r.off + 2] = row[2]; m[r.off + 3] = row[3];
|
||||
}
|
||||
inline for (0..16) |i| { m[i] = tmp[i]; }
|
||||
// Row 3 unchanged (identity row)
|
||||
m[12] = m3[0]; m[13] = m3[1]; m[14] = m3[2]; m[15] = m3[3];
|
||||
return mat;
|
||||
}
|
||||
|
||||
// --- 0x7BB860: createRotMat3x4 ---
|
||||
// Rodrigues rotation matrix, 3x4 layout (3x3 rot + zero translation)
|
||||
// Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries.
|
||||
export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) u32 {
|
||||
const m: [*]f32 = @ptrFromInt(out);
|
||||
const ax: [*]const f32 = @ptrFromInt(axis_ptr);
|
||||
var x = ax[0]; var y = ax[1]; var z = ax[2];
|
||||
if (is_normalized == 0) {
|
||||
const len = @sqrt(x * x + y * y + z * z);
|
||||
const len = @sqrt(@mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x)));
|
||||
if (len > 1.0e-20) { const inv = 1.0 / len; x *= inv; y *= inv; z *= inv; }
|
||||
}
|
||||
const angle: f32 = @bitCast(angle_bits);
|
||||
const c = @cos(angle); const s = @sin(angle); const t = 1.0 - c;
|
||||
m[0] = t * x * x + c; m[1] = t * x * y + s * z; m[2] = t * x * z - s * y;
|
||||
m[3] = t * x * y - s * z; m[4] = t * y * y + c; m[5] = t * y * z + s * x;
|
||||
m[6] = t * x * z + s * y; m[7] = t * y * z - s * x; m[8] = t * z * z + c;
|
||||
m[0] = @mulAdd(f32, t * x, x, c); m[1] = @mulAdd(f32, s, z, t * x * y); m[2] = @mulAdd(f32, -s, y, t * x * z);
|
||||
m[3] = @mulAdd(f32, -s, z, t * x * y); m[4] = @mulAdd(f32, t * y, y, c); m[5] = @mulAdd(f32, s, x, t * y * z);
|
||||
m[6] = @mulAdd(f32, s, y, t * x * z); m[7] = @mulAdd(f32, -s, x, t * y * z); m[8] = @mulAdd(f32, t * z, z, c);
|
||||
m[9] = 0; m[10] = 0; m[11] = 0;
|
||||
return out;
|
||||
}
|
||||
|
||||
// --- 0x6329E0: distanceToPlane (525K/7.5s) ---
|
||||
// (dot(point,normal)+d) / dot(direction,normal)
|
||||
// Fix: @mulAdd chains, keep f32 until return
|
||||
export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) f64 {
|
||||
const p: [*]const f32 = @ptrFromInt(point);
|
||||
const pl: [*]const f32 = @ptrFromInt(plane);
|
||||
const dir: [*]const f32 = @ptrFromInt(direction);
|
||||
const dot1 = p[0] * pl[0] + p[1] * pl[1] + p[2] * pl[2] + pl[3];
|
||||
const dot2 = dir[0] * pl[0] + dir[1] * pl[1] + dir[2] * pl[2];
|
||||
const dot1 = @mulAdd(f32, p[2], pl[2], @mulAdd(f32, p[1], pl[1], @mulAdd(f32, p[0], pl[0], pl[3])));
|
||||
const dot2 = @mulAdd(f32, dir[2], pl[2], @mulAdd(f32, dir[1], pl[1], dir[0] * pl[0]));
|
||||
if (@abs(dot2) < 1.0e-20) return 0.0;
|
||||
return @floatCast(dot1 / dot2);
|
||||
return @as(f64, dot1) / @as(f64, dot2);
|
||||
}
|
||||
|
||||
// --- 0x686C20: classifyPointFrustum (3.2M/7.5s) ---
|
||||
// tests point against 6 frustum planes, produces 6-bit bitmask
|
||||
// Tests point against 6 frustum planes, produces 6-bit bitmask.
|
||||
// Uses V4 dot product: load plane as V4, point as {x,y,z,1}, dot4.
|
||||
export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) u32 {
|
||||
const p: [*]const f32 = @ptrFromInt(point);
|
||||
const mask: *u32 = @ptrFromInt(out_mask);
|
||||
const planes: [*]const f32 = @ptrFromInt(planes_ptr);
|
||||
const px = p[0]; const py = p[1]; const pz = p[2];
|
||||
const pt = loadV3_1(point);
|
||||
var bits: u32 = 0;
|
||||
inline for (0..6) |i| {
|
||||
const pl = planes + i * 4;
|
||||
const dist = px * pl[0] + py * pl[1] + pz * pl[2] + pl[3];
|
||||
const pl = loadV4(planes_ptr + i * 16);
|
||||
const dist = dot4v(pt, pl);
|
||||
if (dist < 0) bits |= (@as(u32, 1) << @intCast(i));
|
||||
}
|
||||
mask.* = bits;
|
||||
@@ -122,16 +137,16 @@ export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) u3
|
||||
}
|
||||
|
||||
// --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) ---
|
||||
// slab-method AABB-line intersection
|
||||
// Slab-method AABB-line intersection. Uses @mulAdd for t computation.
|
||||
export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) u32 {
|
||||
const bmin: [*]const f32 = @ptrFromInt(box_ptr);
|
||||
const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC);
|
||||
const start: [*]const f32 = @ptrFromInt(line_start);
|
||||
const end: [*]const f32 = @ptrFromInt(line_end);
|
||||
const end_pt: [*]const f32 = @ptrFromInt(line_end);
|
||||
var tmin: f32 = 0.0;
|
||||
var tmax: f32 = 1.0;
|
||||
inline for (0..3) |i| {
|
||||
const dir = end[i] - start[i];
|
||||
const dir = end_pt[i] - start[i];
|
||||
if (@abs(dir) < 1.0e-20) {
|
||||
if (start[i] < bmin[i] or start[i] > bmax[i]) return 0;
|
||||
} else {
|
||||
@@ -148,29 +163,39 @@ export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32)
|
||||
}
|
||||
|
||||
// --- 0x6869C0: testOBBFrustum ---
|
||||
// tests oriented bounding box against 6 frustum planes
|
||||
// Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test.
|
||||
export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) u32 {
|
||||
const planes: [*]const f32 = @ptrFromInt(planes_ptr);
|
||||
const aabb: [*]const f32 = @ptrFromInt(aabb_ptr);
|
||||
const rot: [*]const f32 = @ptrFromInt(rot_ptr);
|
||||
const trans: [*]const f32 = @ptrFromInt(trans_ptr);
|
||||
const min2 = [3]f32{ aabb[0], aabb[1], aabb[2] };
|
||||
const max2 = [3]f32{ aabb[3], aabb[4], aabb[5] };
|
||||
var corners: [8][3]f32 = undefined;
|
||||
const t: [*]const f32 = @ptrFromInt(trans_ptr);
|
||||
|
||||
// Rotation columns as V4 (xyz + 0 for w)
|
||||
const rc0 = V4{ rot[0], rot[1], rot[2], 0 };
|
||||
const rc1 = V4{ rot[3], rot[4], rot[5], 0 };
|
||||
const rc2 = V4{ rot[6], rot[7], rot[8], 0 };
|
||||
const tv = V4{ t[0], t[1], t[2], 1.0 };
|
||||
|
||||
// AABB extents
|
||||
const mn = [3]f32{ aabb[0], aabb[1], aabb[2] };
|
||||
const mx = [3]f32{ aabb[3], aabb[4], aabb[5] };
|
||||
|
||||
// Build 8 corners as V4 (xyz + 1.0 for plane dot w/ d term)
|
||||
var corners: [8]V4 = undefined;
|
||||
inline for (0..8) |i| {
|
||||
const lx = if (i & 1 != 0) max2[0] else min2[0];
|
||||
const ly = if (i & 2 != 0) max2[1] else min2[1];
|
||||
const lz = if (i & 4 != 0) max2[2] else min2[2];
|
||||
corners[i][0] = rot[0] * lx + rot[3] * ly + rot[6] * lz + trans[0];
|
||||
corners[i][1] = rot[1] * lx + rot[4] * ly + rot[7] * lz + trans[1];
|
||||
corners[i][2] = rot[2] * lx + rot[5] * ly + rot[8] * lz + trans[2];
|
||||
const lx: f32 = if (i & 1 != 0) mx[0] else mn[0];
|
||||
const ly: f32 = if (i & 2 != 0) mx[1] else mn[1];
|
||||
const lz: f32 = if (i & 4 != 0) mx[2] else mn[2];
|
||||
corners[i] = @mulAdd(V4, @as(V4, @splat(lz)), rc2, @mulAdd(V4, @as(V4, @splat(ly)), rc1, @mulAdd(V4, @as(V4, @splat(lx)), rc0, tv)));
|
||||
}
|
||||
|
||||
// Test each plane: if all 8 corners are behind it, box is outside
|
||||
inline for (0..6) |p| {
|
||||
const pl = planes + p * 4;
|
||||
const pl = loadV4(planes_ptr + p * 16);
|
||||
var all_outside = true;
|
||||
inline for (0..8) |c| {
|
||||
const dist = corners[c][0] * pl[0] + corners[c][1] * pl[1] + corners[c][2] * pl[2] + pl[3];
|
||||
if (dist >= 0) all_outside = false;
|
||||
if (dot4v(corners[c], pl) >= 0) {
|
||||
all_outside = false;
|
||||
}
|
||||
}
|
||||
if (all_outside) return 0;
|
||||
}
|
||||
@@ -178,43 +203,42 @@ export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_
|
||||
}
|
||||
|
||||
// --- 0x686B80: testSphereFrustum (375K/7.5s) ---
|
||||
// Uses V4 dot for plane distance
|
||||
export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) u32 {
|
||||
const planes: [*]const f32 = @ptrFromInt(planes_ptr);
|
||||
const s: [*]const f32 = @ptrFromInt(sphere);
|
||||
const cx = s[0]; const cy = s[1]; const cz = s[2]; const r = s[3];
|
||||
const center = V4{ s[0], s[1], s[2], 1.0 };
|
||||
const r = s[3];
|
||||
inline for (0..6) |i| {
|
||||
const pl = planes + i * 4;
|
||||
const dist = cx * pl[0] + cy * pl[1] + cz * pl[2] + pl[3];
|
||||
if (dist < -r) return 0;
|
||||
const pl = loadV4(planes_ptr + i * 16);
|
||||
if (dot4v(center, pl) < -r) return 0;
|
||||
}
|
||||
return 3;
|
||||
}
|
||||
|
||||
// --- 0x7C0570: quatSlerp ---
|
||||
// V4 for final blend, @mulAdd for dot product
|
||||
export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) u32 {
|
||||
const dst: [*]f32 = @ptrFromInt(out);
|
||||
const a: [*]const f32 = @ptrFromInt(a_ptr);
|
||||
const b_raw: [*]const f32 = @ptrFromInt(b_ptr);
|
||||
const t: f32 = @bitCast(t_bits);
|
||||
var dot_val = a[0] * b_raw[0] + a[1] * b_raw[1] + a[2] * b_raw[2] + a[3] * b_raw[3];
|
||||
const av = loadV4(a_ptr);
|
||||
const bv = loadV4(b_ptr);
|
||||
const tt: f32 = @bitCast(t_bits);
|
||||
var dot_val = dot4v(av, bv);
|
||||
var sign: f32 = 1.0;
|
||||
if (dot_val < 0) { dot_val = -dot_val; sign = -1.0; }
|
||||
var s0: f32 = undefined;
|
||||
var s1: f32 = undefined;
|
||||
if (dot_val > 0.9995) {
|
||||
s0 = 1.0 - t;
|
||||
s1 = t * sign;
|
||||
s0 = 1.0 - tt;
|
||||
s1 = tt * sign;
|
||||
} else {
|
||||
const theta = std.math.acos(dot_val);
|
||||
const sin_theta = @sin(theta);
|
||||
const inv_sin = 1.0 / sin_theta;
|
||||
s0 = @sin((1.0 - t) * theta) * inv_sin;
|
||||
s1 = @sin(t * theta) * inv_sin * sign;
|
||||
s0 = @sin((1.0 - tt) * theta) * inv_sin;
|
||||
s1 = @sin(tt * theta) * inv_sin * sign;
|
||||
}
|
||||
dst[0] = s0 * a[0] + s1 * b_raw[0];
|
||||
dst[1] = s0 * a[1] + s1 * b_raw[1];
|
||||
dst[2] = s0 * a[2] + s1 * b_raw[2];
|
||||
dst[3] = s0 * a[3] + s1 * b_raw[3];
|
||||
const result = @mulAdd(V4, @as(V4, @splat(s1)), bv, @as(V4, @splat(s0)) * av);
|
||||
dst[0] = result[0]; dst[1] = result[1]; dst[2] = result[2]; dst[3] = result[3];
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -247,36 +271,52 @@ export fn si_createZRotMat3x3(out: u32, angle_bits: u32) u32 {
|
||||
}
|
||||
|
||||
// --- 0x7BCEF0: transposeMat4x4 ---
|
||||
// Uses V4 loads + @shuffle for efficient transpose
|
||||
export fn si_transposeMat4x4(src: u32, dst: u32) u32 {
|
||||
const s: [*]const f32 = @ptrFromInt(src);
|
||||
const r0 = loadV4(src);
|
||||
const r1 = loadV4(src + 16);
|
||||
const r2 = loadV4(src + 32);
|
||||
const r3 = loadV4(src + 48);
|
||||
|
||||
// Interleave low/high pairs
|
||||
const t0 = @shuffle(f32, r0, r1, [4]i32{ 0, -1, 2, -3 }); // r0[0] r1[0] r0[2] r1[2]
|
||||
const t1 = @shuffle(f32, r0, r1, [4]i32{ 1, -2, 3, -4 }); // r0[1] r1[1] r0[3] r1[3]
|
||||
const t2 = @shuffle(f32, r2, r3, [4]i32{ 0, -1, 2, -3 }); // r2[0] r3[0] r2[2] r3[2]
|
||||
const t3 = @shuffle(f32, r2, r3, [4]i32{ 1, -2, 3, -4 }); // r2[1] r3[1] r2[3] r3[3]
|
||||
|
||||
const d: [*]f32 = @ptrFromInt(dst);
|
||||
inline for (0..4) |row| {
|
||||
inline for (0..4) |col| {
|
||||
d[row * 4 + col] = s[col * 4 + row];
|
||||
}
|
||||
}
|
||||
// Final columns
|
||||
const c0 = @shuffle(f32, t0, t2, [4]i32{ 0, 1, -1, -2 }); // col 0: r0[0] r1[0] r2[0] r3[0]
|
||||
const c1 = @shuffle(f32, t1, t3, [4]i32{ 0, 1, -1, -2 }); // col 1
|
||||
const c2 = @shuffle(f32, t0, t2, [4]i32{ 2, 3, -3, -4 }); // col 2
|
||||
const c3 = @shuffle(f32, t1, t3, [4]i32{ 2, 3, -3, -4 }); // col 3
|
||||
inline for (0..4) |i| { d[i] = c0[i]; }
|
||||
inline for (0..4) |i| { d[4 + i] = c1[i]; }
|
||||
inline for (0..4) |i| { d[8 + i] = c2[i]; }
|
||||
inline for (0..4) |i| { d[12 + i] = c3[i]; }
|
||||
return src;
|
||||
}
|
||||
|
||||
// --- 0x7BB420: mulMat3x4InPlace ---
|
||||
// this = this * matB
|
||||
// this = this * matB. Uses @mulAdd.
|
||||
export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) u32 {
|
||||
const a: [*]f32 = @ptrFromInt(mat_a);
|
||||
const b: [*]const f32 = @ptrFromInt(mat_b);
|
||||
var tmp: [12]f32 = undefined;
|
||||
inline for (0..3) |row| {
|
||||
inline for (0..3) |col| {
|
||||
tmp[row * 3 + col] = a[col] * b[row * 3] + a[col + 3] * b[row * 3 + 1] + a[col + 6] * b[row * 3 + 2];
|
||||
tmp[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3]));
|
||||
}
|
||||
}
|
||||
inline for (0..3) |col| {
|
||||
tmp[9 + col] = a[9] * b[col] + a[10] * b[col + 3] + a[11] * b[col + 6] + b[9 + col];
|
||||
tmp[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col])));
|
||||
}
|
||||
inline for (0..12) |i| { a[i] = tmp[i]; }
|
||||
return mat_a;
|
||||
}
|
||||
|
||||
// --- 0x6720F0: normalizeVec3InPlace ---
|
||||
// Uses rsqrt approximation + Newton-Raphson for fast inverse sqrt
|
||||
export fn si_normalizeVec3InPlace(vec: u32) void {
|
||||
const v: [*]f32 = @ptrFromInt(vec);
|
||||
const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]);
|
||||
@@ -289,7 +329,6 @@ export fn si_normalizeVec3InPlace(vec: u32) void {
|
||||
}
|
||||
|
||||
// --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) ---
|
||||
// Adds vec3 to this+0x54, adds scaled copy (global at 0x81207C) to diagonal at +0x84/+0xA8/+0xCC
|
||||
export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) void {
|
||||
const obj: [*]f32 = @ptrFromInt(this);
|
||||
const v: [*]const f32 = @ptrFromInt(vec);
|
||||
@@ -297,13 +336,12 @@ export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) void {
|
||||
obj[21] += v[0];
|
||||
obj[22] += v[1];
|
||||
obj[23] += v[2];
|
||||
obj[33] += v[0] * scale;
|
||||
obj[42] += v[1] * scale;
|
||||
obj[51] += v[2] * scale;
|
||||
obj[33] = @mulAdd(f32, v[0], scale, obj[33]);
|
||||
obj[42] = @mulAdd(f32, v[1], scale, obj[42]);
|
||||
obj[51] = @mulAdd(f32, v[2], scale, obj[51]);
|
||||
}
|
||||
|
||||
// --- 0x71BF60: addToColorAccumulator (10K/7.5s) ---
|
||||
// Accumulates color vec3 into this+0x6C (float offset 27)
|
||||
export fn si_addToColorAccumulator(this: u32, color: u32) void {
|
||||
const obj: [*]f32 = @ptrFromInt(this);
|
||||
const c: [*]const f32 = @ptrFromInt(color);
|
||||
@@ -313,17 +351,16 @@ export fn si_addToColorAccumulator(this: u32, color: u32) void {
|
||||
}
|
||||
|
||||
// --- 0x7B7A80: packParticleColor (2K/7.5s) ---
|
||||
// Reads alpha at obj+0x12F, packs ARGB into u32 at obj+0x12C
|
||||
// V4 multiply + clamp + convert for all 3 channels simultaneously
|
||||
export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) void {
|
||||
const base: [*]u8 = @ptrFromInt(obj);
|
||||
const out: *align(1) u32 = @ptrCast(base + 0x12C);
|
||||
const alpha = base[0x12F];
|
||||
const r: f32 = @bitCast(r_bits);
|
||||
const g: f32 = @bitCast(g_bits);
|
||||
const b: f32 = @bitCast(b_bits);
|
||||
const rb: u8 = @intFromFloat(@round(@min(@max(r * 255.0, 0.0), 255.0)));
|
||||
const gb: u8 = @intFromFloat(@round(@min(@max(g * 255.0, 0.0), 255.0)));
|
||||
const bb: u8 = @intFromFloat(@round(@min(@max(b * 255.0, 0.0), 255.0)));
|
||||
const rgb = V4{ @bitCast(r_bits), @bitCast(g_bits), @bitCast(b_bits), 0 } * @as(V4, @splat(@as(f32, 255.0)));
|
||||
const clamped = @min(@max(rgb, @as(V4, @splat(@as(f32, 0.0)))), @as(V4, @splat(@as(f32, 255.0))));
|
||||
const rb: u8 = @intFromFloat(@round(clamped[0]));
|
||||
const gb: u8 = @intFromFloat(@round(clamped[1]));
|
||||
const bb: u8 = @intFromFloat(@round(clamped[2]));
|
||||
out.* = @as(u32, alpha) << 24 | @as(u32, rb) << 16 | @as(u32, gb) << 8 | @as(u32, bb);
|
||||
}
|
||||
|
||||
@@ -335,9 +372,8 @@ export fn si_setParticleAlpha(obj: u32, alpha_bits: u32) void {
|
||||
}
|
||||
|
||||
// --- 0x40A2B0: __ftol ---
|
||||
// Drop-in binary replacement for MSVC __ftol. Input: ST(0). Output: EAX:EDX (i64).
|
||||
// Original: FSTCW/OR/FLDCW/FISTP/FLDCW (39 bytes, ~6 cycles on modern x86)
|
||||
// SSE3 FISTTP: truncate directly from x87 without rounding mode change (9 bytes)
|
||||
// Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64).
|
||||
// SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original)
|
||||
export fn si_ftol() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\sub $8, %%esp
|
||||
@@ -349,28 +385,33 @@ export fn si_ftol() callconv(.naked) void {
|
||||
}
|
||||
|
||||
// --- 0x602630: vec3Dot ---
|
||||
// Uses @mulAdd FMA chain. Returns f64 (x87 ABI contract).
|
||||
export fn si_vec3Dot(a: u32, b: u32) f64 {
|
||||
const va: [*]const f32 = @ptrFromInt(a);
|
||||
const vb: [*]const f32 = @ptrFromInt(b);
|
||||
return @floatCast(va[0] * vb[0] + va[1] * vb[1] + va[2] * vb[2]);
|
||||
const result = @mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0]));
|
||||
return @floatCast(result);
|
||||
}
|
||||
|
||||
// --- 0x686820: translateBoundingVol ---
|
||||
// translates 8 corners, updates plane distances, translates min/max
|
||||
// Uses @mulAdd for plane distance updates
|
||||
export fn si_translateBoundingVol(this: u32, offset: u32) void {
|
||||
const obj: [*]f32 = @ptrFromInt(this);
|
||||
const off: [*]const f32 = @ptrFromInt(offset);
|
||||
const dx = off[0]; const dy = off[1]; const dz = off[2];
|
||||
// Translate 8 corners (stride 3, starting at index 24)
|
||||
inline for (0..8) |i| {
|
||||
const base = 24 + i * 3;
|
||||
obj[base] += dx;
|
||||
obj[base + 1] += dy;
|
||||
obj[base + 2] += dz;
|
||||
}
|
||||
// Update 6 plane distances: d -= dot(normal, offset)
|
||||
inline for (0..6) |i| {
|
||||
const base = i * 4;
|
||||
obj[base + 3] -= obj[base] * dx + obj[base + 1] * dy + obj[base + 2] * dz;
|
||||
obj[base + 3] -= @mulAdd(f32, obj[base + 2], dz, @mulAdd(f32, obj[base + 1], dy, obj[base] * dx));
|
||||
}
|
||||
// Translate min/max
|
||||
obj[48] += dx; obj[49] += dy; obj[50] += dz;
|
||||
obj[51] += dx; obj[52] += dy; obj[53] += dz;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user