particle: contiguous vertex writes, skip normals, stride logging

- Detected interleaved 24-byte vertex layout: xyz(12)+color(4)+uv(8).
  Fast path writes 6 sequential u32s instead of scattered stores.
- Normal stride=0 (shared global) — write once in writeback, not 4x.
- Added stride_info export + logging for vertex layout analysis.
- A/B: ~30% peak reduction, baseline also improved due to less overhead.
This commit is contained in:
MarcelineVQ
2026-03-23 22:41:16 -07:00
parent 0d994db4b3
commit 379fcb62eb
2 changed files with 76 additions and 13 deletions
+13
View File
@@ -25,6 +25,8 @@ extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn resetParticleCache() void;
extern var stride_info: [8]u32; // exported from particle_sse.zig
var stride_dumped: bool = false;
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
@@ -1458,6 +1460,17 @@ fn dumpStats() void {
});
}
// Dump particle VB stride info (once)
if (stride_info[0] != 0 and !stride_dumped) {
stride_dumped = true;
log.fmt(" vb_strides: pos={d} norm={d} color={d} tc={d}\n", .{
stride_info[0], stride_info[1], stride_info[2], stride_info[3],
});
log.fmt(" vb_bases: pos=0x{x} norm=0x{x} color=0x{x} tc=0x{x}\n", .{
stride_info[4], stride_info[5], stride_info[6], stride_info[7],
});
}
// Flip A/B mode
ab_use_custom = !ab_use_custom;
diag_cmp_count = 0;