weirdperformance: f64-intermediate bone transform to eliminate spell/doodad z-fighting
Adds bone_sse64.zig as an f64-intermediate port of transformMatrix4x4, used as the active hook. M2 bone matrices are built and multiplied as [16]f64 and only narrow to f32 on final store into the bone output buffer -- matching the x87 original's rounding profile (wide intermediates, single f32 store) and keeping M2 vertex positions aligned with the terrain/projected-texture pipeline. Also fixes, in both bone_sse (f32) and bone_sse64: - Pre-billboard tx/ty/tz accumulation order (row 0 = pz+px+py; rows 1/2 = pz+py+px) - Post-billboard pos_y/pos_z accumulation order (py+pz+px) - Post-billboard scale-recompute accumulation order (row0 + row2 + row1) - Billboard types 2/4 normalize using f64 intermediates (load-bearing for camera basis vectors -- pure f32 drifted from x87 by a ULP per axis and caused particle emitters to jitter on camera motion) Additional bone_sse64-specific changes: - Local attachmentRecursion64 that recurses into transformImpl_SSE64 instead of bone_sse.transformImpl_SSE, so attached child models stay on the f64 path - child_padding (this+0x84) computed with f64 intermediates bone_sse remains the reference f32 implementation; its struct fields, inline helpers, and section-loop fns are now `pub` so bone_sse64 can share them (types/interpolation helpers/post-loop loops). Artifact size is unchanged. build.zig adds bench_bone_sse64 object; src/bench/main.zig runs the new variant through the same warmup/timing harness and prints SSE vs SSE64 vs BASELINE cycles plus a parity check.
This commit is contained in:
@@ -1662,6 +1662,7 @@ pub fn main() void {
|
||||
const ofs = [3]f32{ 0, 0, 0 };
|
||||
const sb: u32 = @bitCast(@as(f32, 1.0));
|
||||
const transformImpl_SSE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE" });
|
||||
const transformImpl_SSE64 = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE64" });
|
||||
const transformImpl_BASELINE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.c) void, .{ .name = "transformImpl_BASELINE" });
|
||||
|
||||
// Pre-set boneKeyframe init flag so we skip the atexit call (Windows CRT, can't run on Linux)
|
||||
@@ -1716,6 +1717,20 @@ pub fn main() void {
|
||||
const best_sse = run_bench_fn(transformImpl_SSE, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
|
||||
const avg_sse = best_sse / T44_ITERS;
|
||||
|
||||
// Warmup + bench bone_sse64 (f64-intermediate variant)
|
||||
for (0..500) |iter| {
|
||||
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(iter * 2)), .little);
|
||||
wu(u32, scene_obj[0x40..0x44], 0, .little);
|
||||
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
|
||||
}
|
||||
for (0..500) |iter| {
|
||||
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(999 - iter * 2)), .little);
|
||||
wu(u32, scene_obj[0x40..0x44], 0, .little);
|
||||
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
|
||||
}
|
||||
const best_sse64 = run_bench_fn(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
|
||||
const avg_sse64 = best_sse64 / T44_ITERS;
|
||||
|
||||
print(" BASELINE: {d} cycles/call (frozen)\n", .{BASELINE_CYCLES});
|
||||
print(" SSE: {d} cycles/call", .{avg_sse});
|
||||
if (avg_sse < BASELINE_CYCLES) {
|
||||
@@ -1727,6 +1742,25 @@ pub fn main() void {
|
||||
} else {
|
||||
print(" (same)\n", .{});
|
||||
}
|
||||
print(" SSE64: {d} cycles/call", .{avg_sse64});
|
||||
if (avg_sse64 < BASELINE_CYCLES) {
|
||||
const pct = (BASELINE_CYCLES - avg_sse64) * 100 / BASELINE_CYCLES;
|
||||
print(" (-{d}% vs BASELINE", .{pct});
|
||||
} else if (avg_sse64 > BASELINE_CYCLES) {
|
||||
const pct = (avg_sse64 - BASELINE_CYCLES) * 100 / BASELINE_CYCLES;
|
||||
print(" (+{d}% vs BASELINE", .{pct});
|
||||
} else {
|
||||
print(" (same as BASELINE", .{});
|
||||
}
|
||||
if (avg_sse64 > avg_sse) {
|
||||
const pct = (avg_sse64 - avg_sse) * 100 / avg_sse;
|
||||
print(", +{d}% vs SSE)\n", .{pct});
|
||||
} else if (avg_sse64 < avg_sse) {
|
||||
const pct = (avg_sse - avg_sse64) * 100 / avg_sse;
|
||||
print(", -{d}% vs SSE)\n", .{pct});
|
||||
} else {
|
||||
print(", same as SSE)\n", .{});
|
||||
}
|
||||
|
||||
// --- Output parity: run BASELINE then SSE with identical input, compare ALL outputs ---
|
||||
{
|
||||
@@ -1792,6 +1826,23 @@ pub fn main() void {
|
||||
} else {
|
||||
print(" parity: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs, total_len });
|
||||
}
|
||||
|
||||
// Run SSE64 with same input
|
||||
reset_and_run(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, &bone_rt, BONE_COUNT);
|
||||
|
||||
var diffs64: u32 = 0;
|
||||
off = 0;
|
||||
for (bufs) |b| {
|
||||
for (0..b.len) |i| {
|
||||
if (b.ptr[i] != snap[off + i]) diffs64 += 1;
|
||||
}
|
||||
off += b.len;
|
||||
}
|
||||
if (diffs64 == 0) {
|
||||
print(" parity64: PASS (SSE64 == BASELINE, {d} bytes checked)\n", .{total_len});
|
||||
} else {
|
||||
print(" parity64: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs64, total_len });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user