bench: best-of-5 for all silicon functions; silicon_sse: V4 column mulMat3x4/InPlace (1.8x/1.7x)
This commit is contained in:
+97
-82
@@ -169,6 +169,21 @@ fn report(name: []const u8, orig_cyc: u64, sse_cyc: u64, ok: bool) void {
|
||||
});
|
||||
}
|
||||
|
||||
/// Run a function ITERS times, return best-of-5 cycle count.
|
||||
fn bench5(comptime func: anytype, args: anytype) u64 {
|
||||
var best: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
const t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
const r = @call(.never_inline, func, args);
|
||||
std.mem.doNotOptimizeAway(r);
|
||||
}
|
||||
const elapsed = rdtsc() - t0;
|
||||
if (elapsed < best) best = elapsed;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Calling convention types for original x87 functions (game binary)
|
||||
// =========================================================================
|
||||
@@ -249,8 +264,8 @@ pub fn main() void {
|
||||
_ = of(a(&do), fb);
|
||||
_ = vec3MulAssign(a(&ds), fb);
|
||||
const ok = cmpSlice(&do, &ds);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { do = tmpl; _ = of(a(&do), fb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { ds = tmpl; _ = vec3MulAssign(a(&ds), fb); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { do = tmpl; _ = of(a(&do), fb); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { ds = tmpl; _ = vec3MulAssign(a(&ds), fb); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("vec3MulAssign", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -270,8 +285,8 @@ pub fn main() void {
|
||||
of(a(&mo), fb);
|
||||
scaleMatrix3x3ByScalar(a(&ms), fb);
|
||||
const ok = cmpSlice(&mo, &ms);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { mo = tmpl; of(a(&mo), fb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { ms = tmpl; scaleMatrix3x3ByScalar(a(&ms), fb); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { mo = tmpl; of(a(&mo), fb); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { ms = tmpl; scaleMatrix3x3ByScalar(a(&ms), fb); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("scaleByScalar", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -288,8 +303,8 @@ pub fn main() void {
|
||||
_ = of(a(&ro), a(&axis), ab, 1);
|
||||
_ = createAxisAngleRotMat3x3(a(&rs), a(&axis), ab, 1);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis), ab, 1); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = createAxisAngleRotMat3x3(a(&rs), a(&axis), ab, 1); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis), ab, 1); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = createAxisAngleRotMat3x3(a(&rs), a(&axis), ab, 1); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("rotMat3x3", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -303,8 +318,8 @@ pub fn main() void {
|
||||
_ = of(a(&ro), a(&axis), ab, 1);
|
||||
_ = createAxisAngleRotMat4x4(a(&rs), a(&axis), ab, 1);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis), ab, 1); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = createAxisAngleRotMat4x4(a(&rs), a(&axis), ab, 1); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis), ab, 1); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = createAxisAngleRotMat4x4(a(&rs), a(&axis), ab, 1); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("rotMat4x4", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -319,8 +334,8 @@ pub fn main() void {
|
||||
const ov = of(a(&va), a(&vb));
|
||||
const sv = dotProduct(a(&va), a(&vb));
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&va), a(&vb)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = dotProduct(a(&va), a(&vb)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&va), a(&vb)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = dotProduct(a(&va), a(&vb)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("dotProduct", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -331,8 +346,8 @@ pub fn main() void {
|
||||
const ov = of(a(&v));
|
||||
const sv = squaredMagnitude(a(&v));
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&v)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = squaredMagnitude(a(&v)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&v)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = squaredMagnitude(a(&v)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("squaredMagnitude", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -344,8 +359,8 @@ pub fn main() void {
|
||||
const ov = of(3, a(&coeffs), fb);
|
||||
const sv = evaluatePolynomial(3, a(&coeffs), fb);
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(3, a(&coeffs), fb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = evaluatePolynomial(3, a(&coeffs), fb); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(3, a(&coeffs), fb); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = evaluatePolynomial(3, a(&coeffs), fb); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("evaluatePolynomial", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -360,8 +375,8 @@ pub fn main() void {
|
||||
of(a(&ro), a(&p1), a(&p2), a(&p3));
|
||||
calculatePlaneNormal(a(&rs), a(&p1), a(&p2), a(&p3));
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&ro), a(&p1), a(&p2), a(&p3)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { calculatePlaneNormal(a(&rs), a(&p1), a(&p2), a(&p3)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { of(a(&ro), a(&p1), a(&p2), a(&p3)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { calculatePlaneNormal(a(&rs), a(&p1), a(&p2), a(&p3)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("planeNormal", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -377,8 +392,8 @@ pub fn main() void {
|
||||
of(a(&mat), a(&va), a(&vb), a(&box_in), a(&bo));
|
||||
transformAABox(a(&mat), a(&va), a(&vb), a(&box_in), a(&bs));
|
||||
const ok = cmpSlice(&bo, &bs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { bo = .{ 0, 0, 0, 0, 0, 0 }; of(a(&mat), a(&va), a(&vb), a(&box_in), a(&bo)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { bs = .{ 0, 0, 0, 0, 0, 0 }; transformAABox(a(&mat), a(&va), a(&vb), a(&box_in), a(&bs)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { bo = .{ 0, 0, 0, 0, 0, 0 }; of(a(&mat), a(&va), a(&vb), a(&box_in), a(&bo)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { bs = .{ 0, 0, 0, 0, 0, 0 }; transformAABox(a(&mat), a(&va), a(&vb), a(&box_in), a(&bs)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("transformAABox", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -397,12 +412,12 @@ pub fn main() void {
|
||||
inline_x87_dot(&va2, &vb2, &rx);
|
||||
inline_sse_dot(&va2, &vb2, &rs);
|
||||
const ok = compareF32(rx, rs);
|
||||
var t = rdtsc();
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_x87_dot(&va2, &vb2, &rx);
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_sse_dot(&va2, &vb2, &rs);
|
||||
s = rdtsc() - s;
|
||||
const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("dotProduct(inlined)", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -414,12 +429,12 @@ pub fn main() void {
|
||||
inline_x87_sqmag(&v, &rx);
|
||||
inline_sse_sqmag(&v, &rs);
|
||||
const ok = compareF32(rx, rs);
|
||||
var t = rdtsc();
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_x87_sqmag(&v, &rx);
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_sse_sqmag(&v, &rs);
|
||||
s = rdtsc() - s;
|
||||
const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("squaredMag(inlined)", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -432,12 +447,12 @@ pub fn main() void {
|
||||
inline_x87_v3scale(&vec, &factor, &ro);
|
||||
inline_sse_v3scale(&vec, factor, &rs2);
|
||||
const ok = cmpSlice(&ro, &rs2);
|
||||
var t = rdtsc();
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_x87_v3scale(&vec, &factor, &ro);
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_sse_v3scale(&vec, factor, &rs2);
|
||||
s = rdtsc() - s;
|
||||
const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("vec3MulScalar(inlined)", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -450,12 +465,12 @@ pub fn main() void {
|
||||
inline_x87_horner(&coeffs, &factor, &rx);
|
||||
inline_sse_horner(&coeffs, factor, &rs);
|
||||
const ok = compareF32(rx, rs);
|
||||
var t = rdtsc();
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_x87_horner(&coeffs, &factor, &rx);
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| inline_sse_horner(&coeffs, factor, &rs);
|
||||
s = rdtsc() - s;
|
||||
const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("evalPoly(inlined)", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -547,8 +562,8 @@ pub fn main() void {
|
||||
const ov = callDot(0x602630, a(&va2), a(&vb2));
|
||||
const sv = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2));
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = callDot(0x602630, a(&va2), a(&vb2)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = callDot(0x602630, a(&va2), a(&vb2)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = callDot(@intFromPtr(&si_vec3Dot), a(&va2), a(&vb2)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("si_vec3Dot", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -560,8 +575,8 @@ pub fn main() void {
|
||||
of(a(&vo));
|
||||
si_normalizeVec3InPlace(a(&vs));
|
||||
const ok = cmpSlice(&vo, &vs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { vo = tv3(); of(a(&vo)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { vs = tv3(); si_normalizeVec3InPlace(a(&vs)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { vo = tv3(); of(a(&vo)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { vs = tv3(); si_normalizeVec3InPlace(a(&vs)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("normalizeVec3InPlace", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -577,8 +592,8 @@ pub fn main() void {
|
||||
const ov = of(a(&pt), a(&plane), a(&dir));
|
||||
const sv = sf(a(&pt), a(&plane), a(&dir));
|
||||
const ok = @abs(ov - sv) < 1e-2;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&pt), a(&plane), a(&dir)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = sf(a(&pt), a(&plane), a(&dir)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&pt), a(&plane), a(&dir)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = sf(a(&pt), a(&plane), a(&dir)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("distanceToPlane", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -591,8 +606,8 @@ pub fn main() void {
|
||||
const ov = of(a(&box), a(&ls), a(&le));
|
||||
const sv = si_checkBoxLineIntersect(a(&box), a(&ls), a(&le));
|
||||
const ok = ov == sv;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&box), a(&ls), a(&le)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_checkBoxLineIntersect(a(&box), a(&ls), a(&le)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&box), a(&ls), a(&le)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_checkBoxLineIntersect(a(&box), a(&ls), a(&le)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("checkBoxLineIntersect", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -609,8 +624,8 @@ pub fn main() void {
|
||||
_ = of(a(&planes), a(&pt), a(&mask_o));
|
||||
_ = si_classifyPointFrustum(a(&planes), a(&pt), a(&mask_s));
|
||||
const ok = mask_o == mask_s;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&planes), a(&pt), a(&mask_o)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_classifyPointFrustum(a(&planes), a(&pt), a(&mask_s)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&planes), a(&pt), a(&mask_o)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_classifyPointFrustum(a(&planes), a(&pt), a(&mask_s)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("classifyPointFrustum", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -624,8 +639,8 @@ pub fn main() void {
|
||||
const ov = of(a(&planes), a(&sphere));
|
||||
const sv = si_testSphereFrustum(a(&planes), a(&sphere));
|
||||
const ok = ov == sv;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&planes), a(&sphere)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_testSphereFrustum(a(&planes), a(&sphere)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&planes), a(&sphere)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_testSphereFrustum(a(&planes), a(&sphere)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("testSphereFrustum", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -638,8 +653,8 @@ pub fn main() void {
|
||||
_ = of(a(&src), a(&dst_o));
|
||||
_ = si_transposeMat4x4(a(&src), a(&dst_s));
|
||||
const ok = cmpSlice(&dst_o, &dst_s);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&src), a(&dst_o)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_transposeMat4x4(a(&src), a(&dst_s)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&src), a(&dst_o)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_transposeMat4x4(a(&src), a(&dst_s)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("transposeMat4x4", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -654,8 +669,8 @@ pub fn main() void {
|
||||
_ = of(a(&ro), a(&qa), tb, a(&qb));
|
||||
_ = si_quatSlerp(a(&rs), a(&qa), tb, a(&qb));
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&qa), tb, a(&qb)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_quatSlerp(a(&rs), a(&qa), tb, a(&qb)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&qa), tb, a(&qb)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_quatSlerp(a(&rs), a(&qa), tb, a(&qb)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("quatSlerp", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -668,8 +683,8 @@ pub fn main() void {
|
||||
_ = of(a(&ro), ab2);
|
||||
_ = si_createZRotMat3x3(a(&rs), ab2);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), ab2); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_createZRotMat3x3(a(&rs), ab2); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), ab2); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_createZRotMat3x3(a(&rs), ab2); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("createZRotMat3x3", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -691,8 +706,8 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&ma), a(&mb)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_mulMat3x4(a(&rs), a(&ma), a(&mb)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&ma), a(&mb)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_mulMat3x4(a(&rs), a(&ma), a(&mb)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("mulMat3x4", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -706,8 +721,8 @@ pub fn main() void {
|
||||
_ = si_rotateMatByQuat(a(&ms), a(&quat2));
|
||||
const ok = cmpSlice(&mo, &ms);
|
||||
mo = tm4(); ms = tm4();
|
||||
var t = rdtsc(); for (0..ITERS) |_| { mo = tm4(); _ = of(a(&mo), a(&quat2)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { ms = tm4(); _ = si_rotateMatByQuat(a(&ms), a(&quat2)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { mo = tm4(); _ = of(a(&mo), a(&quat2)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { ms = tm4(); _ = si_rotateMatByQuat(a(&ms), a(&quat2)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("rotateMatByQuat", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -721,8 +736,8 @@ pub fn main() void {
|
||||
_ = of(a(&ro), a(&axis2), ab2, 1);
|
||||
_ = si_createRotMat3x4(a(&rs), a(&axis2), ab2, 1);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis2), ab2, 1); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_createRotMat3x4(a(&rs), a(&axis2), ab2, 1); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis2), ab2, 1); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_createRotMat3x4(a(&rs), a(&axis2), ab2, 1); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("createRotMat3x4", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -744,8 +759,8 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
}
|
||||
var t = rdtsc(); for (0..ITERS) |_| { mo2 = tmpl2; _ = of(a(&mo2), a(&mb2)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { ms2 = tmpl2; _ = si_mulMat3x4InPlace(a(&ms2), a(&mb2)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { mo2 = tmpl2; _ = of(a(&mo2), a(&mb2)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { ms2 = tmpl2; _ = si_mulMat3x4InPlace(a(&ms2), a(&mb2)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("mulMat3x4InPlace", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -760,8 +775,8 @@ pub fn main() void {
|
||||
of(a(&vo), lb);
|
||||
si_normalizeVec3(a(&vs), lb);
|
||||
const ok = cmpSlice(&vo, &vs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { vo = tmpl3; of(a(&vo), lb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { vs = tmpl3; si_normalizeVec3(a(&vs), lb); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { vo = tmpl3; of(a(&vo), lb); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { vs = tmpl3; si_normalizeVec3(a(&vs), lb); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("normalizeVec3", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -777,8 +792,8 @@ pub fn main() void {
|
||||
const ov = of(a(&planes), a(&aabb), a(&rot), a(&trans));
|
||||
const sv = si_testOBBFrustum(a(&planes), a(&aabb), a(&rot), a(&trans));
|
||||
const ok = ov == sv;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&planes), a(&aabb), a(&rot), a(&trans)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = si_testOBBFrustum(a(&planes), a(&aabb), a(&rot), a(&trans)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = of(a(&planes), a(&aabb), a(&rot), a(&trans)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { _ = si_testOBBFrustum(a(&planes), a(&aabb), a(&rot), a(&trans)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("testOBBFrustum", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -794,8 +809,8 @@ pub fn main() void {
|
||||
of(ab2, a(&sin_o), a(&cos_o));
|
||||
si_calculateSinCos(ab2, a(&sin_s), a(&cos_s));
|
||||
const ok = compareF32(sin_o, sin_s) and compareF32(cos_o, cos_s);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(ab2, a(&sin_o), a(&cos_o)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { si_calculateSinCos(ab2, a(&sin_s), a(&cos_s)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { of(ab2, a(&sin_o), a(&cos_o)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { si_calculateSinCos(ab2, a(&sin_s), a(&cos_s)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("calculateSinCos", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -829,8 +844,8 @@ pub fn main() void {
|
||||
var obj_bench_o = obj_o;
|
||||
var obj_bench_s = obj_s;
|
||||
const zero_off = Vec3{ 0.001, -0.001, 0.001 }; // tiny offset to avoid overflow
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&obj_bench_o), a(&zero_off)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { si_translateBoundingVol(a(&obj_bench_s), a(&zero_off)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { of(a(&obj_bench_o), a(&zero_off)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { si_translateBoundingVol(a(&obj_bench_s), a(&zero_off)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("translateBoundingVol", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -843,8 +858,8 @@ pub fn main() void {
|
||||
of(a(&obj_o), a(&color));
|
||||
si_addToColorAccumulator(a(&obj_s), a(&color));
|
||||
const ok = compareF32(obj_o[27], obj_s[27]) and compareF32(obj_o[28], obj_s[28]) and compareF32(obj_o[29], obj_s[29]);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&obj_o), a(&color)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { si_addToColorAccumulator(a(&obj_s), a(&color)); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { of(a(&obj_o), a(&color)); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { si_addToColorAccumulator(a(&obj_s), a(&color)); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("addToColorAccum", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -869,8 +884,8 @@ pub fn main() void {
|
||||
print(" orig bytes: [{x} {x} {x} {x}]\n", .{ obj_o[0x12C], obj_o[0x12D], obj_o[0x12E], obj_o[0x12F] });
|
||||
print(" sse bytes: [{x} {x} {x} {x}]\n", .{ obj_s[0x12C], obj_s[0x12D], obj_s[0x12E], obj_s[0x12F] });
|
||||
}
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&obj_o), 0, rb, gb, bb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { si_packParticleColor(a(&obj_s), rb, gb, bb); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { of(a(&obj_o), 0, rb, gb, bb); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { si_packParticleColor(a(&obj_s), rb, gb, bb); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("packParticleColor", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -883,8 +898,8 @@ pub fn main() void {
|
||||
of(a(&obj_o), 0, ab2);
|
||||
si_setParticleAlpha(a(&obj_s), ab2);
|
||||
const ok = obj_o[0x12F] == obj_s[0x12F];
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&obj_o), 0, ab2); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { si_setParticleAlpha(a(&obj_s), ab2); } s = rdtsc() - s;
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { of(a(&obj_o), 0, ab2); } const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc(); for (0..ITERS) |_| { si_setParticleAlpha(a(&obj_s), ab2); } const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report("setParticleAlpha", t, s, ok);
|
||||
}
|
||||
|
||||
@@ -1842,12 +1857,12 @@ fn bench_fc3r(
|
||||
_ = sse_fn(a(&rs), a(¶m_a), a(¶m_b));
|
||||
const ok = cmpSlice(ro[0..result_len], rs[0..result_len]);
|
||||
|
||||
var t = rdtsc();
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| { _ = of(a(&ro), a(¶m_a), a(¶m_b)); }
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| { _ = sse_fn(a(&rs), a(¶m_a), a(¶m_b)); }
|
||||
s = rdtsc() - s;
|
||||
const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report(name, t, s, ok);
|
||||
}
|
||||
|
||||
@@ -1872,12 +1887,12 @@ fn bench_tc2r(
|
||||
_ = sse_fn(a(&ss), a(¶m));
|
||||
const ok = cmpSlice(@as([*]const f32, @ptrCast(&so))[0..result_len], @as([*]const f32, @ptrCast(&ss))[0..result_len]);
|
||||
|
||||
var t = rdtsc();
|
||||
var t: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| { so = self_init; _ = of(a(&so), a(¶m)); }
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
const _te = rdtsc() - _t0; if (_te < t) t = _te; }
|
||||
var s: u64 = std.math.maxInt(u64); for (0..5) |_| { const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| { ss = self_init; _ = sse_fn(a(&ss), a(¶m)); }
|
||||
s = rdtsc() - s;
|
||||
const _te = rdtsc() - _t0; if (_te < s) s = _te; }
|
||||
report(name, t, s, ok);
|
||||
}
|
||||
|
||||
|
||||
+46
-12
@@ -40,21 +40,41 @@ export fn si_normalizeVec3(vec: u32, length_bits: u32) void {
|
||||
// --- 0x7BAE60: mulMat3x4 ---
|
||||
// out = A * B (3x4 layout: 3x3 rotation + 3 translation)
|
||||
// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz]
|
||||
// Use @mulAdd for FMA chains
|
||||
// V4 per row: broadcast b[row*3+k], multiply with a's columns, accumulate.
|
||||
export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) u32 {
|
||||
const dst: [*]f32 = @ptrFromInt(out);
|
||||
const a: [*]const f32 = @ptrFromInt(a_ptr);
|
||||
const aa: [*]const f32 = @ptrFromInt(a_ptr);
|
||||
const b: [*]const f32 = @ptrFromInt(b_ptr);
|
||||
// Rotation block: dst[row*3+col] = a[col]*b[row*3] + a[col+3]*b[row*3+1] + a[col+6]*b[row*3+2]
|
||||
|
||||
// a is row-major 3x3. a[0..2] = row 0, a[3..5] = row 1, a[6..8] = row 2.
|
||||
// The formula: dst[row*3+col] = a[col]*b[row*3] + a[col+3]*b[row*3+1] + a[col+6]*b[row*3+2]
|
||||
// This means: for each output row, we broadcast b's row elements and dot with a's "columns"
|
||||
// a's "column k" is {a[k], a[k+3], a[k+6]} — that's picking one element from each of a's rows.
|
||||
|
||||
// Load a's columns (stride 3)
|
||||
const ac0 = V4{ aa[0], aa[3], aa[6], aa[9] }; // col 0 of a (+ translation x)
|
||||
const ac1 = V4{ aa[1], aa[4], aa[7], aa[10] }; // col 1 of a (+ translation y)
|
||||
const ac2 = V4{ aa[2], aa[5], aa[8], aa[11] }; // col 2 of a (+ translation z)
|
||||
|
||||
// Rotation: 3 output rows
|
||||
inline for (0..3) |row| {
|
||||
inline for (0..3) |col| {
|
||||
dst[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3]));
|
||||
}
|
||||
const br0: V4 = @splat(b[row * 3]);
|
||||
const br1: V4 = @splat(b[row * 3 + 1]);
|
||||
const br2: V4 = @splat(b[row * 3 + 2]);
|
||||
const result = @mulAdd(V4, ac2, br2, @mulAdd(V4, ac1, br1, ac0 * br0));
|
||||
dst[row * 3] = result[0];
|
||||
dst[row * 3 + 1] = result[1];
|
||||
dst[row * 3 + 2] = result[2];
|
||||
}
|
||||
// Translation: dst[9+col] = a[9]*b[col] + a[10]*b[col+3] + a[11]*b[col+6] + b[9+col]
|
||||
|
||||
// Translation: dst[9+i] = a_trans dot b_col_i + b_trans[i]
|
||||
// = ac0[3]*b[i] + ac1[3]*b[i+3] + ac2[3]*b[i+6] + b[9+i]
|
||||
// Using the V4 approach: broadcast b elements, same ac columns, take lane 3 + add b_trans
|
||||
// Translation: scalar @mulAdd (b's columns don't align for V4)
|
||||
inline for (0..3) |col| {
|
||||
dst[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col])));
|
||||
dst[9 + col] = @mulAdd(f32, aa[11], b[col + 6], @mulAdd(f32, aa[10], b[col + 3], @mulAdd(f32, aa[9], b[col], b[9 + col])));
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -338,19 +358,33 @@ export fn si_transposeMat4x4(src: u32, dst: u32) u32 {
|
||||
}
|
||||
|
||||
// --- 0x7BB420: mulMat3x4InPlace ---
|
||||
// this = this * matB. Uses @mulAdd.
|
||||
// this = this * matB. V4 column approach: load a's columns (stride 3), broadcast b's row elements.
|
||||
export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) u32 {
|
||||
const a: [*]f32 = @ptrFromInt(mat_a);
|
||||
const b: [*]const f32 = @ptrFromInt(mat_b);
|
||||
|
||||
// Load a's columns (stride 3, including translation row)
|
||||
const ac0 = V4{ a[0], a[3], a[6], a[9] };
|
||||
const ac1 = V4{ a[1], a[4], a[7], a[10] };
|
||||
const ac2 = V4{ a[2], a[5], a[8], a[11] };
|
||||
|
||||
// Rotation: 3 output rows via V4 broadcast+FMA
|
||||
var tmp: [12]f32 = undefined;
|
||||
inline for (0..3) |row| {
|
||||
inline for (0..3) |col| {
|
||||
tmp[row * 3 + col] = @mulAdd(f32, a[col + 6], b[row * 3 + 2], @mulAdd(f32, a[col + 3], b[row * 3 + 1], a[col] * b[row * 3]));
|
||||
}
|
||||
const br0: V4 = @splat(b[row * 3]);
|
||||
const br1: V4 = @splat(b[row * 3 + 1]);
|
||||
const br2: V4 = @splat(b[row * 3 + 2]);
|
||||
const result = @mulAdd(V4, ac2, br2, @mulAdd(V4, ac1, br1, ac0 * br0));
|
||||
tmp[row * 3] = result[0];
|
||||
tmp[row * 3 + 1] = result[1];
|
||||
tmp[row * 3 + 2] = result[2];
|
||||
}
|
||||
|
||||
// Translation: scalar @mulAdd
|
||||
inline for (0..3) |col| {
|
||||
tmp[9 + col] = @mulAdd(f32, a[11], b[col + 6], @mulAdd(f32, a[10], b[col + 3], @mulAdd(f32, a[9], b[col], b[9 + col])));
|
||||
}
|
||||
|
||||
inline for (0..12) |i| { a[i] = tmp[i]; }
|
||||
return mat_a;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user