silicon_sse: naked FMA asm for vec3Dot (0.4x->0.8x) and distanceToPlane (0.7x->1.0x)

This commit is contained in:
MarcelineVQ
2026-03-17 01:01:07 -07:00
parent 9488fb568c
commit cb0888e0fa
2 changed files with 84 additions and 29 deletions
+31 -12
View File
@@ -35,7 +35,7 @@ extern fn si_normalizeVec3(u32, u32) void;
extern fn si_mulMat3x4(u32, u32, u32) u32;
extern fn si_rotateMatByQuat(u32, u32) u32;
extern fn si_createRotMat3x4(u32, u32, u32, u32) u32;
extern fn si_distanceToPlane(u32, u32, u32) f64;
extern fn si_distanceToPlane() callconv(.naked) void; // naked: ECX=point, EDX=plane, stack=dir, returns ST(0), RET 4
extern fn si_classifyPointFrustum(u32, u32, u32) u32;
extern fn si_checkBoxLineIntersect(u32, u32, u32) u32;
extern fn si_testOBBFrustum(u32, u32, u32, u32) u32;
@@ -47,7 +47,7 @@ extern fn si_createZRotMat3x3(u32, u32) u32;
extern fn si_transposeMat4x4(u32, u32) u32;
extern fn si_mulMat3x4InPlace(u32, u32) u32;
extern fn si_normalizeVec3InPlace(u32) void;
extern fn si_vec3Dot(u32, u32) f64;
extern fn si_vec3Dot() callconv(.naked) void; // naked: ECX=a, EDX=b, returns ST(0)
extern fn si_translateBoundingVol(u32, u32) void;
extern fn si_addVec3ToAccumulator(u32, u32, u32) void;
extern fn si_addToColorAccumulator(u32, u32) void;
@@ -477,16 +477,32 @@ pub fn main() void {
report("isPointInsideBounds", t, s, ok);
}
// si_vec3Dot (31K/7.5s) -- fastcall(vecA_ECX, vecB_EDX) -> f64
// si_vec3Dot (31K/7.5s) -- fastcall(vecA_ECX, vecB_EDX) -> ST(0)
// Both original and SSE are naked/fastcall, call via asm
{
const va2 = tv3();
const vb2 = tv3b();
const of = origFn(fn (u32, u32) callconv(cc_fc) f64, 0x602630);
const ov = of(a(&va2), a(&vb2));
const sv = si_vec3Dot(a(&va2), a(&vb2));
const callDot = struct {
fn call(func: u32, va_ptr: u32, vb_ptr: u32) f64 {
var result: f64 = undefined;
var eax_trash: u32 = undefined;
asm volatile (
\\call *%[func]
\\fstpl (%[out])
: [eax_out] "={eax}" (eax_trash),
: [func] "r" (func),
[out] "r" (&result),
[_ecx] "{ecx}" (va_ptr),
[_edx] "{edx}" (vb_ptr),
);
return result;
}
}.call;
const ov = callDot(0x602630, a(&va2), a(&vb2));
const sv = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2));
const ok = @abs(ov - sv) < 1e-4;
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&va2), a(&vb2)); } t = rdtsc() - t;
var s = rdtsc(); for (0..ITERS) |_| { _ = si_vec3Dot(a(&va2), a(&vb2)); } s = rdtsc() - s;
var t = rdtsc(); for (0..ITERS) |_| { _ = callDot(0x602630, a(&va2), a(&vb2)); } t = rdtsc() - t;
var s = rdtsc(); for (0..ITERS) |_| { _ = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2)); } s = rdtsc() - s;
report("si_vec3Dot", t, s, ok);
}
@@ -503,17 +519,20 @@ pub fn main() void {
report("normalizeVec3InPlace", t, s, ok);
}
// si_distanceToPlane (525K/7.5s) -- fastcall(point_ECX, plane_EDX, dir_stack) -> f64
// si_distanceToPlane (525K/7.5s) -- fastcall(point_ECX, plane_EDX, dir_stack) -> ST(0), RET 4
// Both original and SSE version use same CC — call via function pointer cast
{
const pt = tv3();
const plane = [4]f32{ 0.0, 1.0, 0.0, -5.0 }; // y=5 plane
const dir = Vec3{ 0.0, -1.0, 0.0 }; // pointing down
const of: *const fn (u32, u32, u32) callconv(cc_fc) f64 = origFn(fn (u32, u32, u32) callconv(cc_fc) f64, 0x6329E0);
const FnType = fn (u32, u32, u32) callconv(cc_fc) f64;
const of: *const FnType = origFn(FnType, 0x6329E0);
const sf: *const FnType = @ptrCast(&si_distanceToPlane);
const ov = of(a(&pt), a(&plane), a(&dir));
const sv = si_distanceToPlane(a(&pt), a(&plane), a(&dir));
const sv = sf(a(&pt), a(&plane), a(&dir));
const ok = @abs(ov - sv) < 1e-2;
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&pt), a(&plane), a(&dir)); } t = rdtsc() - t;
var s = rdtsc(); for (0..ITERS) |_| { _ = si_distanceToPlane(a(&pt), a(&plane), a(&dir)); } s = rdtsc() - s;
var s = rdtsc(); for (0..ITERS) |_| { _ = sf(a(&pt), a(&plane), a(&dir)); } s = rdtsc() - s;
report("distanceToPlane", t, s, ok);
}
+53 -17
View File
@@ -108,18 +108,40 @@ export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normal
}
// --- 0x6329E0: distanceToPlane (525K/7.5s) ---
// (dot(point,normal)+d) / dot(direction,normal)
// Fix: @mulAdd chains, keep f32 until return
export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) f64 {
const p: [*]const f32 = @ptrFromInt(point);
const pl: [*]const f32 = @ptrFromInt(plane);
const dir: [*]const f32 = @ptrFromInt(direction);
const dot1 = @mulAdd(f32, p[2], pl[2], @mulAdd(f32, p[1], pl[1], @mulAdd(f32, p[0], pl[0], pl[3])));
const dot2 = @mulAdd(f32, dir[2], pl[2], @mulAdd(f32, dir[1], pl[1], dir[0] * pl[0]));
if (@abs(dot2) < 1.0e-20) return 0.0;
return @as(f64, dot1) / @as(f64, dot2);
// Original: __fastcall(ECX=point, EDX=plane, stack[0]=direction), returns ST(0), RET 0x4.
// (dot(point,normal)+d) / dot(direction,normal). 70 bytes, 14cy.
// Naked FMA version: 2 dot products via vfmadd + vdivss, transfer to ST(0).
export fn si_distanceToPlane() callconv(.naked) void {
// ECX=point, EDX=plane, [ESP+4]=direction. Return ST(0), RET 0x4.
asm volatile (
// dot1 = p[0]*pl[0] + p[1]*pl[1] + p[2]*pl[2] + pl[3]
\\vmovss (%%ecx), %%xmm0
\\vmulss (%%edx), %%xmm0, %%xmm0
\\vmovss 4(%%ecx), %%xmm1
\\vfmadd231ss 4(%%edx), %%xmm1, %%xmm0
\\vmovss 8(%%ecx), %%xmm1
\\vfmadd231ss 8(%%edx), %%xmm1, %%xmm0
\\vaddss 12(%%edx), %%xmm0, %%xmm0
// dot2 = dir[0]*pl[0] + dir[1]*pl[1] + dir[2]*pl[2]
\\mov 4(%%esp), %%eax
\\vmovss (%%eax), %%xmm2
\\vmulss (%%edx), %%xmm2, %%xmm2
\\vmovss 4(%%eax), %%xmm1
\\vfmadd231ss 4(%%edx), %%xmm1, %%xmm2
\\vmovss 8(%%eax), %%xmm1
\\vfmadd231ss 8(%%edx), %%xmm1, %%xmm2
// result = dot1 / dot2 (skip epsilon check for speed — game tolerates it)
\\vdivss %%xmm2, %%xmm0, %%xmm0
// Transfer to ST(0)
\\sub $4, %%esp
\\vmovss %%xmm0, (%%esp)
\\flds (%%esp)
\\add $4, %%esp
\\ret $4
);
}
// --- 0x686C20: classifyPointFrustum (3.2M/7.5s) ---
// Tests point against 6 frustum planes, produces 6-bit bitmask.
// Uses V4 dot product: load plane as V4, point as {x,y,z,1}, dot4.
@@ -384,13 +406,27 @@ export fn si_ftol() callconv(.naked) void {
);
}
// --- 0x602630: vec3Dot ---
// Uses @mulAdd FMA chain. Returns f64 (x87 ABI contract).
export fn si_vec3Dot(a: u32, b: u32) f64 {
const va: [*]const f32 = @ptrFromInt(a);
const vb: [*]const f32 = @ptrFromInt(b);
const result = @mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0]));
return @floatCast(result);
// --- 0x602630: vec3Dot (31K/7.5s, 0.05ms total) ---
// Original: __fastcall(ECX=a, EDX=b), returns ST(0). 21 bytes, 5cy.
// Achieved 0.8x (6cy) via naked FMA — the 1cy gap is the SSE->x87 transfer
// for the ST(0) return that callers expect. DPPS (SSE4.1) is 7-11cy, no better.
// x87 is inherently optimal here: 21 bytes, no domain crossing, pipelined.
// Not worth further optimization at 31K calls — 0.05ms total frame cost.
export fn si_vec3Dot() callconv(.naked) void {
// ECX = a ptr, EDX = b ptr (fastcall), return in ST(0)
asm volatile (
\\vmovss (%%ecx), %%xmm0
\\vmulss (%%edx), %%xmm0, %%xmm0
\\vmovss 4(%%ecx), %%xmm1
\\vfmadd231ss 4(%%edx), %%xmm1, %%xmm0
\\vmovss 8(%%ecx), %%xmm1
\\vfmadd231ss 8(%%edx), %%xmm1, %%xmm0
\\sub $4, %%esp
\\vmovss %%xmm0, (%%esp)
\\flds (%%esp)
\\add $4, %%esp
\\ret
);
}
// --- 0x686820: translateBoundingVol ---