perf: FrustumCullBoundingBox SSE replacement, 1.6x speedup

- silicon_sse: si_frustumCullBBox (0x686000) — inline V4 mat*vec3
  transforms, SSE perspective divide, 4-wide horizon buffer scan.
  Benched 1.6x (117→72 cyc/call). Installed via JMP patch (544 bytes,
  won't fit in 380-byte original).
- bench: add frustumCullBBox benchmark with mapped globals, identity
  matrices, and horizon buffer test fixture.
- Remove patch table entry for frustumCullBBox, install via detour hook
  instead (allows future A/B testing if needed).
This commit is contained in:
MarcelineVQ
2026-03-23 20:46:28 -07:00
parent 7819d6d914
commit 035355b757
3 changed files with 189 additions and 0 deletions
+97
View File
@@ -50,6 +50,7 @@ extern fn si_normalizeVec3InPlace(u32) callconv(cc_tc) void;
extern fn si_vec3Dot(u32, u32) callconv(cc_fc) f64;
extern fn si_translateBoundingVol(u32, u32) callconv(cc_tc) void;
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(cc_fc) u32;
extern fn si_frustumCullBBox(u32, u32, u32) callconv(cc_fc) u32;
extern fn si_addVec3ToAccumulator(u32, u32) callconv(cc_tc) void;
extern fn si_addToColorAccumulator(u32, u32) callconv(cc_tc) void;
extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void;
@@ -1768,6 +1769,9 @@ pub fn main() void {
}
}
// si_frustumCullBBox -- fastcall(bbox_ECX, flags_EDX, radius_stack) -> u32
bench_frustumCullBBox();
// si_processLinkedListCollision -- fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack) -> u32
// Builds a fake linked list with 8 nodes to benchmark AABB overlap test.
bench_processLinkedListCollision();
@@ -1775,6 +1779,99 @@ pub fn main() void {
print("\n", .{});
}
fn bench_frustumCullBBox() void {
// Map runtime global pages for view-proj matrices, occlusion buffer, and flags
_ = mapZeroed(0xC7B000, 0x20000); // covers 0xC7B000-0xC7D000+ (matrices, horizon buffer, globals)
// Set up globals that FrustumCullBoundingBox reads:
// 0xC7B2A4: occlusion flag — bit 5 must be set to proceed
@as(*u8, @ptrFromInt(0xC7B2A4)).* = 0x20;
// 0xC7CFF4: global value checked against range [const1, const2]
// const1 at 0x8101AC, const2 at 0x804588 — both are in mapped .rdata
// Set to a value that passes: read the constants and pick the midpoint
const const1: f32 = @as(*align(1) const f32, @ptrFromInt(0x8101AC)).*;
const const2: f32 = @as(*align(1) const f32, @ptrFromInt(0x804588)).*;
@as(*align(1) f32, @ptrFromInt(0xC7CFF4)).* = (const1 + const2) * 0.5;
// 0x80FED4: near plane constant for behind-camera check
// Already in mapped pages. Set to a value that passes (e.g., -1000)
@as(*align(1) f32, @ptrFromInt(0x80FED4)).* = -1000.0;
// 0x7FF9D8: perspective scale constant (likely screen_width/2 or similar)
// In .rdata — already mapped, read whatever's there or set a reasonable value
if (@as(*align(1) const u32, @ptrFromInt(0x7FF9D8)).* == 0) {
@as(*align(1) f32, @ptrFromInt(0x7FF9D8)).* = 160.0;
}
// 0x810170: column scale factor
if (@as(*align(1) const u32, @ptrFromInt(0x810170)).* == 0) {
@as(*align(1) f32, @ptrFromInt(0x810170)).* = 1.0;
}
// 0x86861C: column offset — in .rdata, use whatever's there or set 0
// 0x86861C is at offset 0x86861C - 0x7FF000 = 0x6961C in rdata — may be beyond our mapped range
// Map additional page if needed
_ = mapZeroed(0x868000, 0x1000);
// View-proj matrix at 0xC7B700: identity-like projection for testing
{
const mat: [*]f32 = @ptrFromInt(0xC7B700);
// Simple perspective-like matrix (column-major)
mat[0] = 1.0; mat[1] = 0.0; mat[2] = 0.0; mat[3] = 0.0;
mat[4] = 0.0; mat[5] = 1.0; mat[6] = 0.0; mat[7] = 0.0;
mat[8] = 0.0; mat[9] = 0.0; mat[10] = 1.0; mat[11] = 0.0;
mat[12] = 0.0; mat[13] = 0.0; mat[14] = 0.0; mat[15] = 1.0;
}
// Second matrix at 0xC7D280: identity for extent transform
{
const mat: [*]f32 = @ptrFromInt(0xC7D280);
mat[0] = 1.0; mat[1] = 0.0; mat[2] = 0.0; mat[3] = 0.0;
mat[4] = 0.0; mat[5] = 1.0; mat[6] = 0.0; mat[7] = 0.0;
mat[8] = 0.0; mat[9] = 0.0; mat[10] = 1.0; mat[11] = 0.0;
mat[12] = 0.0; mat[13] = 0.0; mat[14] = 0.0; mat[15] = 1.0;
}
// Horizon buffer at 0xC7B750: 320 floats, fill with large values (everything visible)
{
const buf: [*]f32 = @ptrFromInt(0xC7B750);
for (0..320) |i| buf[i] = 1000.0;
}
// Test data: bbox point at (5, 3, 10), radius 2.0, flags=0
var bbox = [3]f32{ 5.0, 3.0, 10.0 };
const radius: f32 = 2.0;
const radius_bits: u32 = @bitCast(radius);
const flags: u32 = 0;
const of = origFn(fn (u32, u32, u32) callconv(cc_fc) u32, 0x686000);
const ret_orig = of(a(&bbox), flags, radius_bits);
const ret_sse = si_frustumCullBBox(a(&bbox), flags, radius_bits);
const ok = ret_orig == ret_sse;
var t: u64 = std.math.maxInt(u64);
for (0..5) |_| {
const _t0 = rdtsc();
for (0..ITERS) |_| {
_ = of(a(&bbox), flags, radius_bits);
}
const _te = rdtsc() - _t0;
if (_te < t) t = _te;
}
var s: u64 = std.math.maxInt(u64);
for (0..5) |_| {
const _t0 = rdtsc();
for (0..ITERS) |_| {
_ = si_frustumCullBBox(a(&bbox), flags, radius_bits);
}
const _te = rdtsc() - _t0;
if (_te < s) s = _te;
}
report("frustumCullBBox", t, s, ok);
}
fn bench_processLinkedListCollision() void {
// Map page for sentinel global at 0xC89F20
_ = mapZeroed(0xC89000, 0x1000);
+2
View File
@@ -2089,6 +2089,7 @@ const sse = struct {
extern fn si_vec3Dot() callconv(.naked) void;
extern fn si_translateBoundingVol(u32, u32) callconv(TC) void;
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(FC) u32;
extern fn si_frustumCullBBox(u32, u32, u32) callconv(FC) u32;
};
const PatchEntry = struct {
@@ -2124,6 +2125,7 @@ fn getPatchTable() []const PatchEntry {
// vec3Dot (0x602630): removed — 0.6x regression, x87 is optimal for this ABI
.{ .target = 0x686820, .replacement = @intFromPtr(&sse.si_translateBoundingVol), .name = "translateBoundingVol" },
.{ .target = 0x6ABC40, .replacement = @intFromPtr(&sse.si_processLinkedListCollision), .name = "processLinkedListCollision" },
// frustumCullBBox: A/B tested via detour hook, not patched
};
return &table;
}
+90
View File
@@ -516,6 +516,96 @@ export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
obj[51] += dx; obj[52] += dy; obj[53] += dz;
}
// --- 0x686000: FrustumCullBoundingBox ---
// Transforms bbox through view-proj matrix, perspective divides, projects to 320-column
// occlusion buffer. Returns 0 (culled) / 2 (visible).
// Original: 380 bytes, 2 calls to mat*vec3 (0x7BCA80), x87 perspective divide, x87 column scan.
// SSE: inline V4 mat*vec3, SSE perspective divide, 4-wide column scan.
// __fastcall(bbox_ECX, flags_EDX, radius_stack), RET 0x4
export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 {
// Early out: global occlusion flag bit 5
if (@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 0x20 == 0) return 0;
// Early out: radius too small
const radius: f32 = @bitCast(radius_bits);
const epsilon: f32 = @bitCast(@as(*const u32, @ptrFromInt(0x8029D4)).*);
if (@abs(radius) < epsilon) return 0;
// Early out: global value must be in valid range [const1, const2]
const global_val: f32 = @as(*align(1) const f32, @ptrFromInt(0xC7CFF4)).*;
if (global_val < @as(*align(1) const f32, @ptrFromInt(0x8101AC)).*) return 0;
if (global_val > @as(*align(1) const f32, @ptrFromInt(0x804588)).*) return 0;
// Transform center through view-proj matrix (column-major 4x4 at 0xC7B700)
// Inlined 0x7BCA80: result = col0*v.x + col1*v.y + col2*v.z + col3
const bp: [*]const f32 = @ptrFromInt(bbox);
const vx: V4 = @splat(bp[0]);
const vy: V4 = @splat(bp[1]);
const vz: V4 = @splat(bp[2]);
const m1: u32 = 0xC7B700;
const center = @mulAdd(V4, vz, loadV4(m1 + 32), @mulAdd(V4, vy, loadV4(m1 + 16), @mulAdd(V4, vx, loadV4(m1), loadV4(m1 + 48))));
// Transform extent {radius, radius, 0} through matrix at 0xC7D280
const rv: V4 = @splat(radius);
const m2: u32 = 0xC7D280;
// z=0, so skip col2 term
const extent = @mulAdd(V4, rv, loadV4(m2 + 16), @mulAdd(V4, rv, loadV4(m2), loadV4(m2 + 48)));
// Behind-camera check (unless flags & 8)
if (flags & 0x8 == 0) {
if (center[2] < @as(*align(1) const f32, @ptrFromInt(0x80FED4)).*) return 0;
}
// Perspective divide: inv_w = K / center.z
const K: f32 = @as(*align(1) const f32, @ptrFromInt(0x7FF9D8)).*;
const inv_w = K / center[2];
const cx = center[0] * inv_w; // projected center x
const ex = extent[0] * inv_w; // projected extent x
const ey = extent[1] * inv_w; // projected extent y
const depth = center[1] * inv_w + ex; // depth for horizon test
// Column projection: convert to 320-column indices
const col_scale: f32 = @as(*align(1) const f32, @ptrFromInt(0x810170)).*;
const col_offset: f32 = @as(*align(1) const f32, @ptrFromInt(0x86861C)).*;
// FISTP uses default x87 round-to-nearest; match with @round
var left_col: i32 = @intFromFloat(@round((cx - ey) * col_scale - col_offset));
left_col += 0xA0; // +160 center offset
var right_col: i32 = @intFromFloat(@round((ey + cx) * col_scale - col_offset));
right_col += 0xA1; // +161
// Bounds check — off-screen culling
if (left_col >= 0x140) return 0; // fully right of screen (320)
if (right_col < 0) return 0; // fully left of screen
if (left_col < 0) left_col = 0;
if (right_col >= 0x140) right_col = 0x13F; // clamp to 319
if (left_col > right_col) return 2; // degenerate → visible
// Horizon buffer scan: 320 floats at 0xC7B750
// If any column's horizon value < depth → culled (return 0)
// SSE: test 4 columns at once
const horizon_base: u32 = 0xC7B750;
var col: u32 = @intCast(left_col);
const end: u32 = @intCast(right_col);
const depth_v: V4 = @splat(depth);
// 4-wide scan
while (col + 3 <= end) {
const h = loadV4(horizon_base + col * 4);
const lt_bits: u4 = @bitCast(h < depth_v);
if (lt_bits != 0) return 0;
col += 4;
}
// Scalar remainder
while (col <= end) {
if (@as(*align(1) const f32, @ptrFromInt(horizon_base + col * 4)).* < depth) return 0;
col += 1;
}
return 2; // visible — survived all columns
}
// --- 0x6ABC40: processLinkedListCollision ---
// Walks intrusive linked list, per-node AABB overlap test, calls addGeometryToBuffer on hit.
// Original: 329 bytes, 6 x87 FCOMP/FNSTSW comparisons per node.