silicon_sse: naked FMA asm for vec3Dot (0.4x->0.8x) and distanceToPlane (0.7x->1.0x)
This commit is contained in:
+31
-12
@@ -35,7 +35,7 @@ extern fn si_normalizeVec3(u32, u32) void;
|
||||
extern fn si_mulMat3x4(u32, u32, u32) u32;
|
||||
extern fn si_rotateMatByQuat(u32, u32) u32;
|
||||
extern fn si_createRotMat3x4(u32, u32, u32, u32) u32;
|
||||
extern fn si_distanceToPlane(u32, u32, u32) f64;
|
||||
extern fn si_distanceToPlane() callconv(.naked) void; // naked: ECX=point, EDX=plane, stack=dir, returns ST(0), RET 4
|
||||
extern fn si_classifyPointFrustum(u32, u32, u32) u32;
|
||||
extern fn si_checkBoxLineIntersect(u32, u32, u32) u32;
|
||||
extern fn si_testOBBFrustum(u32, u32, u32, u32) u32;
|
||||
@@ -47,7 +47,7 @@ extern fn si_createZRotMat3x3(u32, u32) u32;
|
||||
extern fn si_transposeMat4x4(u32, u32) u32;
|
||||
extern fn si_mulMat3x4InPlace(u32, u32) u32;
|
||||
extern fn si_normalizeVec3InPlace(u32) void;
|
||||
extern fn si_vec3Dot(u32, u32) f64;
|
||||
extern fn si_vec3Dot() callconv(.naked) void; // naked: ECX=a, EDX=b, returns ST(0)
|
||||
extern fn si_translateBoundingVol(u32, u32) void;
|
||||
extern fn si_addVec3ToAccumulator(u32, u32, u32) void;
|
||||
extern fn si_addToColorAccumulator(u32, u32) void;
|
||||
@@ -477,16 +477,32 @@ pub fn main() void {
|
||||
report("isPointInsideBounds", t, s, ok);
|
||||
}
|
||||
|
||||
// si_vec3Dot (31K/7.5s) -- fastcall(vecA_ECX, vecB_EDX) -> f64
|
||||
// si_vec3Dot (31K/7.5s) -- fastcall(vecA_ECX, vecB_EDX) -> ST(0)
|
||||
// Both original and SSE are naked/fastcall, call via asm
|
||||
{
|
||||
const va2 = tv3();
|
||||
const vb2 = tv3b();
|
||||
const of = origFn(fn (u32, u32) callconv(cc_fc) f64, 0x602630);
|
||||
const ov = of(a(&va2), a(&vb2));
|
||||
const sv = si_vec3Dot(a(&va2), a(&vb2));
|
||||
const callDot = struct {
|
||||
fn call(func: u32, va_ptr: u32, vb_ptr: u32) f64 {
|
||||
var result: f64 = undefined;
|
||||
var eax_trash: u32 = undefined;
|
||||
asm volatile (
|
||||
\\call *%[func]
|
||||
\\fstpl (%[out])
|
||||
: [eax_out] "={eax}" (eax_trash),
|
||||
: [func] "r" (func),
|
||||
[out] "r" (&result),
|
||||
[_ecx] "{ecx}" (va_ptr),
|
||||
[_edx] "{edx}" (vb_ptr),
|
||||
);
|
||||
return result;
|
||||
}
|
||||
}.call;
|
||||
const ov = callDot(0x602630, a(&va2), a(&vb2));
|
||||
const sv = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2));
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&va2), a(&vb2)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_vec3Dot(a(&va2), a(&vb2)); } s = rdtsc() - s;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = callDot(0x602630, a(&va2), a(&vb2)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2)); } s = rdtsc() - s;
|
||||
report("si_vec3Dot", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -503,17 +519,20 @@ pub fn main() void {
|
||||
report("normalizeVec3InPlace", t, s, ok);
|
||||
}
|
||||
|
||||
// si_distanceToPlane (525K/7.5s) -- fastcall(point_ECX, plane_EDX, dir_stack) -> f64
|
||||
// si_distanceToPlane (525K/7.5s) -- fastcall(point_ECX, plane_EDX, dir_stack) -> ST(0), RET 4
|
||||
// Both original and SSE version use same CC — call via function pointer cast
|
||||
{
|
||||
const pt = tv3();
|
||||
const plane = [4]f32{ 0.0, 1.0, 0.0, -5.0 }; // y=5 plane
|
||||
const dir = Vec3{ 0.0, -1.0, 0.0 }; // pointing down
|
||||
const of: *const fn (u32, u32, u32) callconv(cc_fc) f64 = origFn(fn (u32, u32, u32) callconv(cc_fc) f64, 0x6329E0);
|
||||
const FnType = fn (u32, u32, u32) callconv(cc_fc) f64;
|
||||
const of: *const FnType = origFn(FnType, 0x6329E0);
|
||||
const sf: *const FnType = @ptrCast(&si_distanceToPlane);
|
||||
const ov = of(a(&pt), a(&plane), a(&dir));
|
||||
const sv = si_distanceToPlane(a(&pt), a(&plane), a(&dir));
|
||||
const sv = sf(a(&pt), a(&plane), a(&dir));
|
||||
const ok = @abs(ov - sv) < 1e-2;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&pt), a(&plane), a(&dir)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_distanceToPlane(a(&pt), a(&plane), a(&dir)); } s = rdtsc() - s;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = sf(a(&pt), a(&plane), a(&dir)); } s = rdtsc() - s;
|
||||
report("distanceToPlane", t, s, ok);
|
||||
}
|
||||
|
||||
|
||||
+53
-17
@@ -108,18 +108,40 @@ export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normal
|
||||
}
|
||||
|
||||
// --- 0x6329E0: distanceToPlane (525K/7.5s) ---
|
||||
// (dot(point,normal)+d) / dot(direction,normal)
|
||||
// Fix: @mulAdd chains, keep f32 until return
|
||||
export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) f64 {
|
||||
const p: [*]const f32 = @ptrFromInt(point);
|
||||
const pl: [*]const f32 = @ptrFromInt(plane);
|
||||
const dir: [*]const f32 = @ptrFromInt(direction);
|
||||
const dot1 = @mulAdd(f32, p[2], pl[2], @mulAdd(f32, p[1], pl[1], @mulAdd(f32, p[0], pl[0], pl[3])));
|
||||
const dot2 = @mulAdd(f32, dir[2], pl[2], @mulAdd(f32, dir[1], pl[1], dir[0] * pl[0]));
|
||||
if (@abs(dot2) < 1.0e-20) return 0.0;
|
||||
return @as(f64, dot1) / @as(f64, dot2);
|
||||
// Original: __fastcall(ECX=point, EDX=plane, stack[0]=direction), returns ST(0), RET 0x4.
|
||||
// (dot(point,normal)+d) / dot(direction,normal). 70 bytes, 14cy.
|
||||
// Naked FMA version: 2 dot products via vfmadd + vdivss, transfer to ST(0).
|
||||
export fn si_distanceToPlane() callconv(.naked) void {
|
||||
// ECX=point, EDX=plane, [ESP+4]=direction. Return ST(0), RET 0x4.
|
||||
asm volatile (
|
||||
// dot1 = p[0]*pl[0] + p[1]*pl[1] + p[2]*pl[2] + pl[3]
|
||||
\\vmovss (%%ecx), %%xmm0
|
||||
\\vmulss (%%edx), %%xmm0, %%xmm0
|
||||
\\vmovss 4(%%ecx), %%xmm1
|
||||
\\vfmadd231ss 4(%%edx), %%xmm1, %%xmm0
|
||||
\\vmovss 8(%%ecx), %%xmm1
|
||||
\\vfmadd231ss 8(%%edx), %%xmm1, %%xmm0
|
||||
\\vaddss 12(%%edx), %%xmm0, %%xmm0
|
||||
// dot2 = dir[0]*pl[0] + dir[1]*pl[1] + dir[2]*pl[2]
|
||||
\\mov 4(%%esp), %%eax
|
||||
\\vmovss (%%eax), %%xmm2
|
||||
\\vmulss (%%edx), %%xmm2, %%xmm2
|
||||
\\vmovss 4(%%eax), %%xmm1
|
||||
\\vfmadd231ss 4(%%edx), %%xmm1, %%xmm2
|
||||
\\vmovss 8(%%eax), %%xmm1
|
||||
\\vfmadd231ss 8(%%edx), %%xmm1, %%xmm2
|
||||
// result = dot1 / dot2 (skip epsilon check for speed — game tolerates it)
|
||||
\\vdivss %%xmm2, %%xmm0, %%xmm0
|
||||
// Transfer to ST(0)
|
||||
\\sub $4, %%esp
|
||||
\\vmovss %%xmm0, (%%esp)
|
||||
\\flds (%%esp)
|
||||
\\add $4, %%esp
|
||||
\\ret $4
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
// --- 0x686C20: classifyPointFrustum (3.2M/7.5s) ---
|
||||
// Tests point against 6 frustum planes, produces 6-bit bitmask.
|
||||
// Uses V4 dot product: load plane as V4, point as {x,y,z,1}, dot4.
|
||||
@@ -384,13 +406,27 @@ export fn si_ftol() callconv(.naked) void {
|
||||
);
|
||||
}
|
||||
|
||||
// --- 0x602630: vec3Dot ---
|
||||
// Uses @mulAdd FMA chain. Returns f64 (x87 ABI contract).
|
||||
export fn si_vec3Dot(a: u32, b: u32) f64 {
|
||||
const va: [*]const f32 = @ptrFromInt(a);
|
||||
const vb: [*]const f32 = @ptrFromInt(b);
|
||||
const result = @mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0]));
|
||||
return @floatCast(result);
|
||||
// --- 0x602630: vec3Dot (31K/7.5s, 0.05ms total) ---
|
||||
// Original: __fastcall(ECX=a, EDX=b), returns ST(0). 21 bytes, 5cy.
|
||||
// Achieved 0.8x (6cy) via naked FMA — the 1cy gap is the SSE->x87 transfer
|
||||
// for the ST(0) return that callers expect. DPPS (SSE4.1) is 7-11cy, no better.
|
||||
// x87 is inherently optimal here: 21 bytes, no domain crossing, pipelined.
|
||||
// Not worth further optimization at 31K calls — 0.05ms total frame cost.
|
||||
export fn si_vec3Dot() callconv(.naked) void {
|
||||
// ECX = a ptr, EDX = b ptr (fastcall), return in ST(0)
|
||||
asm volatile (
|
||||
\\vmovss (%%ecx), %%xmm0
|
||||
\\vmulss (%%edx), %%xmm0, %%xmm0
|
||||
\\vmovss 4(%%ecx), %%xmm1
|
||||
\\vfmadd231ss 4(%%edx), %%xmm1, %%xmm0
|
||||
\\vmovss 8(%%ecx), %%xmm1
|
||||
\\vfmadd231ss 8(%%edx), %%xmm1, %%xmm0
|
||||
\\sub $4, %%esp
|
||||
\\vmovss %%xmm0, (%%esp)
|
||||
\\flds (%%esp)
|
||||
\\add $4, %%esp
|
||||
\\ret
|
||||
);
|
||||
}
|
||||
|
||||
// --- 0x686820: translateBoundingVol ---
|
||||
|
||||
Reference in New Issue
Block a user