From 10bd922cc3cccd1d74d2c9b0eb0cf38e32d03060 Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Tue, 17 Mar 2026 13:33:51 -0700 Subject: [PATCH] silicon: fix CC mismatches (normalizeVec3InPlace TC, packParticleColor TC, addVec3ToAccumulator remove phantom scale param, revert classifyPointFrustum/testOBBFrustum to TC); runtime JMP address resolution --- src/bench/main.zig | 6 ++-- src/silicon/silicon.zig | 71 ++++++++++++++++++++++++++----------- src/silicon/silicon_sse.zig | 9 ++--- 3 files changed, 59 insertions(+), 27 deletions(-) diff --git a/src/bench/main.zig b/src/bench/main.zig index 40dc13c..32292a5 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -46,12 +46,12 @@ extern fn si_calculateSinCos(u32, u32, u32) callconv(cc_sc) void; extern fn si_createZRotMat3x3(u32, u32) callconv(cc_tc) u32; extern fn si_transposeMat4x4(u32, u32) callconv(cc_tc) u32; extern fn si_mulMat3x4InPlace(u32, u32) callconv(cc_tc) u32; -extern fn si_normalizeVec3InPlace(u32) callconv(cc_fc) void; +extern fn si_normalizeVec3InPlace(u32) callconv(cc_tc) void; extern fn si_vec3Dot(u32, u32) callconv(cc_fc) f64; extern fn si_translateBoundingVol(u32, u32) callconv(cc_tc) void; -extern fn si_addVec3ToAccumulator(u32, u32, u32) callconv(cc_tc) void; +extern fn si_addVec3ToAccumulator(u32, u32) callconv(cc_tc) void; extern fn si_addToColorAccumulator(u32, u32) callconv(cc_tc) void; -extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_fc) void; +extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void; extern fn si_setParticleAlpha(u32, u32, u32) callconv(cc_fc) void; // fastcall(ECX=obj, EDX=unused, stack=alpha) extern fn si_ftol() callconv(.naked) void; diff --git a/src/silicon/silicon.zig b/src/silicon/silicon.zig index b9881dd..1edd4fe 100644 --- a/src/silicon/silicon.zig +++ b/src/silicon/silicon.zig @@ -2065,7 +2065,7 @@ pub fn installHooks() void { const sse = struct { // silicon_sse.zig exports (linked via object file) - extern fn si_normalizeVec3(u32, u32) callconv(TC) void; + extern fn si_normalizeVec3() callconv(.naked) void; extern fn si_mulMat3x4(u32, u32, u32) callconv(FC) u32; extern fn si_rotateMatByQuat(u32, u32) callconv(TC) u32; extern fn si_createRotMat3x4(u32, u32, u32, u32) callconv(FC) u32; @@ -2078,13 +2078,13 @@ const sse = struct { extern fn si_isPointInsideBounds() callconv(.naked) void; extern fn si_calculateSinCos(u32, u32, u32) callconv(SC) void; extern fn si_createZRotMat3x3(u32, u32) callconv(TC) u32; - extern fn si_transposeMat4x4(u32, u32) callconv(TC) u32; + extern fn si_transposeMat4x4() callconv(.naked) void; extern fn si_mulMat3x4InPlace(u32, u32) callconv(TC) u32; - extern fn si_normalizeVec3InPlace(u32) callconv(FC) void; - extern fn si_addVec3ToAccumulator(u32, u32, u32) callconv(TC) void; - extern fn si_addToColorAccumulator(u32, u32) callconv(TC) void; - extern fn si_packParticleColor(u32, u32, u32, u32) callconv(FC) void; - extern fn si_setParticleAlpha(u32, u32) callconv(FC) void; + extern fn si_normalizeVec3InPlace(u32) callconv(TC) void; + extern fn si_addVec3ToAccumulator(u32, u32) callconv(TC) void; + extern fn si_addToColorAccumulator() callconv(.naked) void; + extern fn si_packParticleColor(u32, u32, u32, u32) callconv(TC) void; + extern fn si_setParticleAlpha() callconv(.naked) void; extern fn si_ftol() callconv(.naked) void; extern fn si_vec3Dot() callconv(.naked) void; extern fn si_translateBoundingVol(u32, u32) callconv(TC) void; @@ -2126,22 +2126,53 @@ fn getPatchTable() []const PatchEntry { return &table; } +fn patchJmp(target: u32, replacement: u32) void { + const rel = @as(i32, @bitCast(replacement -% target -% 5)); + var patch = [5]u8{ 0xE9, 0, 0, 0, 0 }; + @as(*align(1) i32, @ptrCast(patch[1..5])).* = rel; + hook.writeProtected(target, &patch); +} + +fn patchDirect(target: u32, src: [*]const u8, size: usize) void { + hook.writeProtected(target, src[0..size]); +} + fn installPatches() u32 { var count: u32 = 0; - for (getPatchTable()) |entry| { - if (entry.direct_size > 0) { - // Direct byte copy: naked asm replacement fits in original - const src: [*]const u8 = @ptrFromInt(entry.replacement); - hook.writeProtected(entry.target, src[0..entry.direct_size]); - } else { - // JMP rel32: E9 XX XX XX XX - const rel = @as(i32, @bitCast(entry.replacement -% entry.target -% 5)); - var patch = [5]u8{ 0xE9, 0, 0, 0, 0 }; - @as(*align(1) i32, @ptrCast(patch[1..5])).* = rel; - hook.writeProtected(entry.target, &patch); + const patch = struct { + fn jmp(target: u32, comptime func: anytype) void { + patchJmp(target, @intFromPtr(func)); } - count += 1; - } + fn direct(target: u32, comptime func: anytype, size: usize) void { + patchDirect(target, @as([*]const u8, @ptrCast(func)), size); + } + }; + + patch.jmp(0x4549C0, &sse.si_normalizeVec3); + patch.jmp(0x7BAE60, &sse.si_mulMat3x4); + patch.jmp(0x7BDDB0, &sse.si_rotateMatByQuat); + patch.jmp(0x7BB860, &sse.si_createRotMat3x4); + patch.jmp(0x686C20, &sse.si_classifyPointFrustum); + patch.jmp(0x6DC5A0, &sse.si_checkBoxLineIntersect); + patch.jmp(0x6869C0, &sse.si_testOBBFrustum); + patch.jmp(0x686B80, &sse.si_testSphereFrustum); + patch.jmp(0x7C0570, &sse.si_quatSlerp); + patch.jmp(0x749280, &sse.si_calculateSinCos); + patch.jmp(0x7BE5B0, &sse.si_createZRotMat3x3); + patch.jmp(0x7BB420, &sse.si_mulMat3x4InPlace); + patch.jmp(0x6720F0, &sse.si_normalizeVec3InPlace); + patch.jmp(0x71BC70, &sse.si_addVec3ToAccumulator); + patch.jmp(0x71BF60, &sse.si_addToColorAccumulator); + patch.jmp(0x7B7A80, &sse.si_packParticleColor); + patch.jmp(0x7B7B10, &sse.si_setParticleAlpha); + patch.jmp(0x686820, &sse.si_translateBoundingVol); + count += 18; + + // Direct byte patches (naked asm that fits in original) + patch.direct(0x40A2B0, &sse.si_ftol, 9); + patch.direct(0x7BCEF0, &sse.si_transposeMat4x4, 64); + count += 2; + return count; } diff --git a/src/silicon/silicon_sse.zig b/src/silicon/silicon_sse.zig index 4ccaf8a..55b38b3 100644 --- a/src/silicon/silicon_sse.zig +++ b/src/silicon/silicon_sse.zig @@ -398,7 +398,7 @@ export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 { // --- 0x6720F0: normalizeVec3InPlace --- // sqrt + reciprocal. 14cy (2.2x). rsqrt+NR tested at 15cy — no gain, compiler's // vsqrtss+vdivss pipeline is already optimal for scalar inverse sqrt. -export fn si_normalizeVec3InPlace(vec: u32) callconv(FC) void { +export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void { const v: [*]f32 = @ptrFromInt(vec); const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); if (len > 1.0e-20) { @@ -410,10 +410,11 @@ export fn si_normalizeVec3InPlace(vec: u32) callconv(FC) void { } // --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) --- -export fn si_addVec3ToAccumulator(this: u32, vec: u32, scale_addr: u32) callconv(TC) void { +// thiscall(ECX=this, stack=vec). Scale is a global at 0x81207C, NOT a parameter. +export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void { const obj: [*]f32 = @ptrFromInt(this); const v: [*]const f32 = @ptrFromInt(vec); - const scale: f32 = @as(*const f32, @ptrFromInt(scale_addr)).*; + const scale: f32 = @as(*const f32, @ptrFromInt(0x81207C)).*; obj[21] += v[0]; obj[22] += v[1]; obj[23] += v[2]; @@ -443,7 +444,7 @@ export fn si_addToColorAccumulator() callconv(.naked) void { // --- 0x7B7A80: packParticleColor (2K/7.5s) --- // V4 multiply + clamp, then packed round+convert via @Vector(4, i32) for all channels at once. -export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(FC) void { +export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void { const base: [*]u8 = @ptrFromInt(obj); const out: *align(1) u32 = @ptrCast(base + 0x12C); const alpha = base[0x12F];