bone_sse: pure Zig SSE/FMA reimplementation, zero game function calls

Replace bone_sse.zig with a complete pure Zig implementation compiled
with SSE4.1 + FMA + AVX. All 18 game function calls replaced:

- findInterpIdx (0x713D50): temporal-coherence keyframe search
- interpAnimKF (0x713EA0): CompQuat lerp for rotation keyframes
- extractByte (0x71AE90): byte keyframe extraction
- getInterpolatedFloat (0x71AF20): float track with direct blend read
- callFtol (0x40A2B0): @intFromFloat replaces x87 __ftol
- callVec3SqMag (0x4549F0): inline FMA dot product
- callGetIndexOffset/callSetShortValue (0x71AFF0/0x71B010): direct ri16
- matMul (0x74A7C0): V4 FMA matmul (broadcast + 3 @mulAdd per row)
- buildRotFn (0x74B6B5): inline quat→matrix
- rotateQuat (0x7BDDB0): quat→matrix then FMA matmul
- scaleMat (0x7BDCA0): inline scale from vec3 ptr
- applyTrans (0x7BDC40): inline FMA dot product translation

Only 2 game calls remain:
- 0x409AEF: one-time atexit init (boneKeyframeLoop)
- 0x7B5F60: IsParticleBufferEmpty (reads game particle state)

Child recursion calls transformMatrix4x4_SSE directly instead of
going through the hook at 0x714260.

Detour cleaned up: REF is baseline, SSE activates via ab_use_custom
toggle. Diagnostic/bisect/FPU-comparison scaffolding removed.
build.zig: bone_sse gets dedicated target with sse4_1+fma+avx features.
This commit is contained in:
MarcelineVQ
2026-03-15 18:16:07 -07:00
parent 8fb1ece2b5
commit 3a031803ef
3 changed files with 762 additions and 1715 deletions
+7 -1
View File
@@ -63,11 +63,17 @@ pub fn build(b: *std.Build) void {
.optimize = .ReleaseFast,
}),
});
const bone_sse_target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .windows,
.abi = .msvc,
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
});
const bone_sse_obj = b.addObject(.{
.name = "bone_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/transform44/bone_sse.zig"),
.target = target,
.target = bone_sse_target,
.optimize = .ReleaseFast,
}),
});
+752 -1431
View File
File diff suppressed because it is too large Load Diff
+3 -283
View File
@@ -258,15 +258,7 @@ fn diagCompare(label: [*:0]const u8, snap: []const u8, live: u32, len: u32) void
}
}
// Test A: stripped detour — pure passthrough to REF, no profiling, no dispatch logic.
// If black → issue is REF code structure. If renders → detour overhead was the problem.
const DIRECT_REF_TEST = true;
fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(hook.cc.thiscall) void {
if (DIRECT_REF_TEST) {
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
return;
}
const start = rdtsc();
const model_data = hook.readMem(u32, this + 0x10);
@@ -295,282 +287,10 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
if (teardown_active) {
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
} else if (false and !is_early and diag_count < DIAG_MAX and t44_depth == 1) {
// --- DIAGNOSTIC (disabled): run original, snapshot, run REF, compare ---
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
// Gather region info from SceneObject
const model_ctr_d = hook.readMem(u32, this + 0x30);
const model_hdr_d = if (model_ctr_d != 0) hook.readMem(u32, model_ctr_d + 0x130) else 0;
const bc = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x34) else 0;
const bone_rt_base = hook.readMem(u32, this + 0x90);
const bone_out_base = hook.readMem(u32, this + 0x94);
const tex_out = hook.readMem(u32, this + 0xA0);
const col_out = hook.readMem(u32, this + 0xA8);
const scale2 = hook.readMem(u32, this + 0xB0);
const scale3 = hook.readMem(u32, this + 0xB4);
const gs_vals = hook.readMem(u32, this + 0x64);
const anim_ctx_d = hook.readMem(u32, this + 0x2C);
const emitter_d = hook.readMem(u32, this + 0x1CC);
const gs_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x14) else 0;
const tex_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x54) else 0;
const col_gate = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x64) else 0;
const col_count = if (model_hdr_d != 0 and col_gate != 0) hook.readMem(u32, model_hdr_d + 0x6C) else 0;
const bkf_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x74) else 0;
const rib_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x11C) else 0;
const p124_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x124) else 0;
const p134_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x134) else 0;
const p13c_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x13C) else 0;
// Define regions to compare (addr, len, label) — fit in 128KB buffer
const Region = struct { addr: u32, len: u32, label: [*:0]const u8 };
var regions: [20]Region = undefined;
var n_regions: u32 = 0;
// SceneObject: 0x000-0x3E0
regions[n_regions] = .{ .addr = this, .len = 0x3E0, .label = "SceneObject" };
n_regions += 1;
// Bone runtime: all bones
if (bc > 0 and bone_rt_base != 0) {
const brt_len = @min(bc * 0x118, 0x10000); // cap at 64KB
regions[n_regions] = .{ .addr = bone_rt_base, .len = brt_len, .label = "BoneRT" };
n_regions += 1;
}
// Bone output: all bones
if (bc > 0 and bone_out_base != 0) {
const bout_len = @min(bc * 0x40, 0x4000);
regions[n_regions] = .{ .addr = bone_out_base, .len = bout_len, .label = "BoneOut" };
n_regions += 1;
}
// Global sequence values
if (gs_count > 0 and gs_vals != 0) {
regions[n_regions] = .{ .addr = gs_vals, .len = gs_count * 4, .label = "GSValues" };
n_regions += 1;
}
// Texture animation output
if (tex_count > 0 and tex_out != 0) {
regions[n_regions] = .{ .addr = tex_out, .len = @min(tex_count * 0x50, 0x1000), .label = "TexAnim" };
n_regions += 1;
}
// Color animation output
if (col_count > 0 and col_out != 0) {
regions[n_regions] = .{ .addr = col_out, .len = @min(col_count * 0x20, 0x400), .label = "ColorAnim" };
n_regions += 1;
}
// Bone keyframe scale2/scale3 buffers
if (bkf_count > 0 and scale2 != 0) {
regions[n_regions] = .{ .addr = scale2, .len = @min(bkf_count * 0x98, 0x2000), .label = "BKF_Scale2" };
n_regions += 1;
}
if (bkf_count > 0 and scale3 != 0) {
regions[n_regions] = .{ .addr = scale3, .len = @min(bkf_count * 0x40, 0x1000), .label = "BKF_Scale3" };
n_regions += 1;
}
// Animation context (read-only but check)
if (anim_ctx_d != 0) {
regions[n_regions] = .{ .addr = anim_ctx_d, .len = 0x20, .label = "AnimCtx" };
n_regions += 1;
}
// Emitter context
if (emitter_d != 0) {
regions[n_regions] = .{ .addr = emitter_d, .len = 0x200, .label = "EmitterCtx" };
n_regions += 1;
}
// Ribbon emitter output (this+0x200)
if (rib_count > 0) {
const rib_out = hook.readMem(u32, this + 0x200);
if (rib_out != 0) {
regions[n_regions] = .{ .addr = rib_out, .len = @min(rib_count * 0x170, 0x4000), .label = "RibbonOut" };
n_regions += 1;
}
}
// Particle 0x124 output (this+0x3C4)
if (p124_count > 0) {
const p124_out = hook.readMem(u32, this + 0x3C4);
if (p124_out != 0) {
regions[n_regions] = .{ .addr = p124_out, .len = @min(p124_count * 0x84, 0x2000), .label = "Part124" };
n_regions += 1;
}
}
// Particle 0x134 output (this+0x3C8)
if (p134_count > 0) {
const p134_out = hook.readMem(u32, this + 0x3C8);
if (p134_out != 0) {
regions[n_regions] = .{ .addr = p134_out, .len = @min(p134_count * 0xD0, 0x4000), .label = "Part134" };
n_regions += 1;
}
}
// Particle 0x13C output (this+0x3D0)
if (p13c_count > 0) {
const p13c_out = hook.readMem(u32, this + 0x3D0);
if (p13c_out != 0) {
regions[n_regions] = .{ .addr = p13c_out, .len = @min(p13c_count * 0x16C, 0x8000), .label = "Part13C" };
n_regions += 1;
}
}
// Globals
regions[n_regions] = .{ .addr = 0xCF0400, .len = 0x100, .label = "Globals_CF04" };
n_regions += 1;
// Snapshot all regions after original ran
var buf_off: u32 = 0;
var region_starts: [20]u32 = undefined;
var ri: u32 = 0;
while (ri < n_regions) : (ri += 1) {
region_starts[ri] = buf_off;
const len = regions[ri].len;
if (buf_off + len <= diag_buf.len) {
diagSnapshot(diag_buf[buf_off .. buf_off + len], regions[ri].addr, len);
buf_off += len;
}
}
// Clear sync so REF doesn't early-exit
@as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0;
// Run REF
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
// Compare each region
log.fmt("=== DIAG COMPARE #{d} this=0x{x:0>8} bones={d} regions={d} buf_used={d}", .{ diag_count, this, bc, n_regions, buf_off });
// Log ALL game constants that might differ from static analysis
if (diag_count == 0) {
log.fmt(" CONST: s2f=0x{x:0>8} eps1=0x{x:0>8} eps2=0x{x:0>8} h3=0x{x:0>8} h5=0x{x:0>8} c74=0x{x:0>8} cd8=0x{x:0>8}", .{
hook.readMem(u32, 0x811610), // SHORT_TO_FLOAT
hook.readMem(u32, 0x8029d4), // epsilon 1
hook.readMem(u32, 0x80c5c8), // epsilon 2
hook.readMem(u32, 0x80297c), // hermite 3
hook.readMem(u32, 0x802990), // hermite 5/6
hook.readMem(u32, 0x7ffd74), // frequent FLD (particle sections)
hook.readMem(u32, 0x7ff9d8), // hermite FADD (particle sections)
});
}
ri = 0;
while (ri < n_regions) : (ri += 1) {
const len = regions[ri].len;
const snap_start = region_starts[ri];
if (snap_start + len <= diag_buf.len) {
diagCompare(regions[ri].label, diag_buf[snap_start .. snap_start + len], regions[ri].addr, len);
}
}
// TexAnim gate analysis: dump anim_frame_ctr and per-entry alpha kf_count
if (tex_count > 0) {
const tex_data_base = hook.readMem(u32, model_hdr_d + 0x58);
const afc = hook.readMem(u32, this + 0x8C);
log.fmt(" TexAnim gates: anim_frame_ctr={d} tex_count={d}", .{ afc, tex_count });
var ti: u32 = 0;
while (ti < tex_count and ti < 8) : (ti += 1) {
const td = tex_data_base + ti * 0x38;
const vec3_gate = hook.readMem(u32, td + 0x0C); // Vec3 kf_count
const alpha_gate = hook.readMem(u32, td + 0x28); // alpha kf_count
const alpha_mode = hook.readMem(u16, td + 0x1C); // alpha interp_mode
// Also read what's at the alpha output slot BEFORE REF wrote to it (from snapshot)
const alpha_out_off = ti * 0x50 + 0x3C; // offset within tex_anim_out buffer
// Find tex_anim snapshot
var snap_alpha_orig: u32 = 0xDEAD;
var live_alpha: u32 = 0xDEAD;
if (tex_out != 0 and alpha_out_off + 4 <= @min(tex_count * 0x50, 0x1000)) {
// Find the TexAnim snapshot in diag_buf
var si: u32 = 0;
while (si < n_regions) : (si += 1) {
if (regions[si].addr == tex_out) {
const soff = region_starts[si] + alpha_out_off;
if (soff + 4 <= diag_buf.len) {
snap_alpha_orig = @as(u32, diag_buf[soff]) | (@as(u32, diag_buf[soff + 1]) << 8) | (@as(u32, diag_buf[soff + 2]) << 16) | (@as(u32, diag_buf[soff + 3]) << 24);
}
break;
}
}
live_alpha = hook.readMem(u32, tex_out + alpha_out_off);
}
// Also read the raw short value and the constant at 0x811610
const alpha_data_base = hook.readMem(u32, td + 0x1C + 0x18); // AD.keyframe_base for alpha
const alpha_idx0 = hook.readMem(u32, tex_out + ti * 0x50 + 0x30); // idx0 from findInterpIdx
const raw_short: i16 = if (alpha_data_base != 0) @as(*align(1) const i16, @ptrFromInt(alpha_data_base + alpha_idx0 * 2)).* else 0;
const s2f_const = hook.readMem(u32, 0x811610); // SHORT_TO_FLOAT constant
log.fmt(" tex[{d}]: vec3_kf={d} alpha_kf={d} mode={d} orig=0x{x:0>8} ref=0x{x:0>8} short={d} s2f=0x{x:0>8}", .{ ti, vec3_gate, alpha_gate, alpha_mode, snap_alpha_orig, live_alpha, raw_short, s2f_const });
}
}
diag_count += 1;
} else if (ab_use_custom) {
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
} else {
// FPU state comparison: capture full x87 state after original vs REF
if (diag_count >= DIAG_MAX and diag_count < DIAG_MAX + 3 and t44_depth == 1 and !is_early) {
// Run original, capture FPU + MXCSR state
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
var fpu_orig: [108]u8 align(16) = undefined;
var mxcsr_orig: u32 = 0;
asm volatile ("fnsave (%[p])\n\tfrstor (%[p])\n\tstmxcsr (%[m])"
:: [p] "r" (@intFromPtr(&fpu_orig)), [m] "r" (@intFromPtr(&mxcsr_orig))
: "memory"
);
// Clear sync, run REF
@as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0;
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
var fpu_ref: [108]u8 align(16) = undefined;
var mxcsr_ref: u32 = 0;
asm volatile ("fnsave (%[p])\n\tfrstor (%[p])\n\tstmxcsr (%[m])"
:: [p] "r" (@intFromPtr(&fpu_ref)), [m] "r" (@intFromPtr(&mxcsr_ref))
: "memory"
);
// Compare and log FPU state
// FNSAVE layout (108 bytes): CW(4), SW(4), TW(4), IP(4), CS(4), DP(4), DS(4), ST0-ST7(8×10=80)
const cw_o = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + 0)).*;
const sw_o = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + 4)).*;
const tw_o = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + 8)).*;
const cw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 0)).*;
const sw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 4)).*;
const tw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 8)).*;
log.fmt(" FPU orig: CW=0x{x:0>4} SW=0x{x:0>4} TW=0x{x:0>4}", .{ cw_o & 0xFFFF, sw_o & 0xFFFF, tw_o & 0xFFFF });
log.fmt(" FPU ref: CW=0x{x:0>4} SW=0x{x:0>4} TW=0x{x:0>4}", .{ cw_r & 0xFFFF, sw_r & 0xFFFF, tw_r & 0xFFFF });
// MXCSR comparison — SSE control/status, never checked before
if (mxcsr_orig != mxcsr_ref) {
log.fmt(" *** MXCSR DIFF: orig=0x{x:0>8} ref=0x{x:0>8} (xor=0x{x:0>8})", .{ mxcsr_orig, mxcsr_ref, mxcsr_orig ^ mxcsr_ref });
} else {
log.fmt(" MXCSR match: 0x{x:0>8}", .{mxcsr_orig});
}
// Dump ST0-ST7 (10 bytes each, starting at offset 28)
var sti: u32 = 0;
while (sti < 8) : (sti += 1) {
const base = 28 + sti * 10;
const o0 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + base)).*;
const o1 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + base + 4)).*;
const o2 = @as(*align(1) const u16, @ptrFromInt(@intFromPtr(&fpu_orig) + base + 8)).*;
const r0 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + base)).*;
const r1 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + base + 4)).*;
const r2 = @as(*align(1) const u16, @ptrFromInt(@intFromPtr(&fpu_ref) + base + 8)).*;
if (o0 != r0 or o1 != r1 or o2 != r2) {
log.fmt(" ST{d} DIFF: orig={x:0>4}_{x:0>8}_{x:0>8} ref={x:0>4}_{x:0>8}_{x:0>8}", .{ sti, o2, o1, o0, r2, r1, r0 });
}
}
diag_count += 1;
} else {
// BISECT MODE: REF runs up to bisect_stop_section, then original provides rest
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
if (bisect_stop_section != 0) {
// REF returned early — clear sync so original doesn't early-exit, then run original
@as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0;
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
}
}
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
}
t44_depth -|= 1;