particle: contiguous vertex writes, skip normals, stride logging
- Detected interleaved 24-byte vertex layout: xyz(12)+color(4)+uv(8). Fast path writes 6 sequential u32s instead of scattered stores. - Normal stride=0 (shared global) — write once in writeback, not 4x. - Added stride_info export + logging for vertex layout analysis. - A/B: ~30% peak reduction, baseline also improved due to less overhead.
This commit is contained in:
@@ -25,6 +25,8 @@ extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
|
||||
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn resetParticleCache() void;
|
||||
extern var stride_info: [8]u32; // exported from particle_sse.zig
|
||||
var stride_dumped: bool = false;
|
||||
|
||||
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
|
||||
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
|
||||
@@ -1458,6 +1460,17 @@ fn dumpStats() void {
|
||||
});
|
||||
}
|
||||
|
||||
// Dump particle VB stride info (once)
|
||||
if (stride_info[0] != 0 and !stride_dumped) {
|
||||
stride_dumped = true;
|
||||
log.fmt(" vb_strides: pos={d} norm={d} color={d} tc={d}\n", .{
|
||||
stride_info[0], stride_info[1], stride_info[2], stride_info[3],
|
||||
});
|
||||
log.fmt(" vb_bases: pos=0x{x} norm=0x{x} color=0x{x} tc=0x{x}\n", .{
|
||||
stride_info[4], stride_info[5], stride_info[6], stride_info[7],
|
||||
});
|
||||
}
|
||||
|
||||
// Flip A/B mode
|
||||
ab_use_custom = !ab_use_custom;
|
||||
diag_cmp_count = 0;
|
||||
|
||||
Reference in New Issue
Block a user