weirdperformance: f64-intermediate bone transform to eliminate spell/doodad z-fighting

Adds bone_sse64.zig as an f64-intermediate port of transformMatrix4x4, used as
the active hook. M2 bone matrices are built and multiplied as [16]f64 and only
narrow to f32 on final store into the bone output buffer -- matching the x87
original's rounding profile (wide intermediates, single f32 store) and keeping
M2 vertex positions aligned with the terrain/projected-texture pipeline.

Also fixes, in both bone_sse (f32) and bone_sse64:

- Pre-billboard tx/ty/tz accumulation order (row 0 = pz+px+py; rows 1/2 = pz+py+px)
- Post-billboard pos_y/pos_z accumulation order (py+pz+px)
- Post-billboard scale-recompute accumulation order (row0 + row2 + row1)
- Billboard types 2/4 normalize using f64 intermediates (load-bearing for camera
  basis vectors -- pure f32 drifted from x87 by a ULP per axis and caused
  particle emitters to jitter on camera motion)

Additional bone_sse64-specific changes:

- Local attachmentRecursion64 that recurses into transformImpl_SSE64 instead of
  bone_sse.transformImpl_SSE, so attached child models stay on the f64 path
- child_padding (this+0x84) computed with f64 intermediates

bone_sse remains the reference f32 implementation; its struct fields, inline
helpers, and section-loop fns are now `pub` so bone_sse64 can share them
(types/interpolation helpers/post-loop loops). Artifact size is unchanged.

build.zig adds bench_bone_sse64 object; src/bench/main.zig runs the new variant
through the same warmup/timing harness and prints SSE vs SSE64 vs BASELINE
cycles plus a parity check.
This commit is contained in:
MarcelineVQ
2026-04-16 22:58:09 -07:00
parent 37c96eb577
commit f500147fc7
5 changed files with 1310 additions and 270 deletions
+51
View File
@@ -1662,6 +1662,7 @@ pub fn main() void {
const ofs = [3]f32{ 0, 0, 0 };
const sb: u32 = @bitCast(@as(f32, 1.0));
const transformImpl_SSE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE" });
const transformImpl_SSE64 = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE64" });
const transformImpl_BASELINE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.c) void, .{ .name = "transformImpl_BASELINE" });
// Pre-set boneKeyframe init flag so we skip the atexit call (Windows CRT, can't run on Linux)
@@ -1716,6 +1717,20 @@ pub fn main() void {
const best_sse = run_bench_fn(transformImpl_SSE, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
const avg_sse = best_sse / T44_ITERS;
// Warmup + bench bone_sse64 (f64-intermediate variant)
for (0..500) |iter| {
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(iter * 2)), .little);
wu(u32, scene_obj[0x40..0x44], 0, .little);
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
}
for (0..500) |iter| {
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(999 - iter * 2)), .little);
wu(u32, scene_obj[0x40..0x44], 0, .little);
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
}
const best_sse64 = run_bench_fn(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
const avg_sse64 = best_sse64 / T44_ITERS;
print(" BASELINE: {d} cycles/call (frozen)\n", .{BASELINE_CYCLES});
print(" SSE: {d} cycles/call", .{avg_sse});
if (avg_sse < BASELINE_CYCLES) {
@@ -1727,6 +1742,25 @@ pub fn main() void {
} else {
print(" (same)\n", .{});
}
print(" SSE64: {d} cycles/call", .{avg_sse64});
if (avg_sse64 < BASELINE_CYCLES) {
const pct = (BASELINE_CYCLES - avg_sse64) * 100 / BASELINE_CYCLES;
print(" (-{d}% vs BASELINE", .{pct});
} else if (avg_sse64 > BASELINE_CYCLES) {
const pct = (avg_sse64 - BASELINE_CYCLES) * 100 / BASELINE_CYCLES;
print(" (+{d}% vs BASELINE", .{pct});
} else {
print(" (same as BASELINE", .{});
}
if (avg_sse64 > avg_sse) {
const pct = (avg_sse64 - avg_sse) * 100 / avg_sse;
print(", +{d}% vs SSE)\n", .{pct});
} else if (avg_sse64 < avg_sse) {
const pct = (avg_sse - avg_sse64) * 100 / avg_sse;
print(", -{d}% vs SSE)\n", .{pct});
} else {
print(", same as SSE)\n", .{});
}
// --- Output parity: run BASELINE then SSE with identical input, compare ALL outputs ---
{
@@ -1792,6 +1826,23 @@ pub fn main() void {
} else {
print(" parity: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs, total_len });
}
// Run SSE64 with same input
reset_and_run(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, &bone_rt, BONE_COUNT);
var diffs64: u32 = 0;
off = 0;
for (bufs) |b| {
for (0..b.len) |i| {
if (b.ptr[i] != snap[off + i]) diffs64 += 1;
}
off += b.len;
}
if (diffs64 == 0) {
print(" parity64: PASS (SSE64 == BASELINE, {d} bytes checked)\n", .{total_len});
} else {
print(" parity64: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs64, total_len });
}
}
}