From 3a031803efa3327cdd198734ce4c089a3010d952 Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Sun, 15 Mar 2026 18:16:07 -0700 Subject: [PATCH] bone_sse: pure Zig SSE/FMA reimplementation, zero game function calls MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace bone_sse.zig with a complete pure Zig implementation compiled with SSE4.1 + FMA + AVX. All 18 game function calls replaced: - findInterpIdx (0x713D50): temporal-coherence keyframe search - interpAnimKF (0x713EA0): CompQuat lerp for rotation keyframes - extractByte (0x71AE90): byte keyframe extraction - getInterpolatedFloat (0x71AF20): float track with direct blend read - callFtol (0x40A2B0): @intFromFloat replaces x87 __ftol - callVec3SqMag (0x4549F0): inline FMA dot product - callGetIndexOffset/callSetShortValue (0x71AFF0/0x71B010): direct ri16 - matMul (0x74A7C0): V4 FMA matmul (broadcast + 3 @mulAdd per row) - buildRotFn (0x74B6B5): inline quat→matrix - rotateQuat (0x7BDDB0): quat→matrix then FMA matmul - scaleMat (0x7BDCA0): inline scale from vec3 ptr - applyTrans (0x7BDC40): inline FMA dot product translation Only 2 game calls remain: - 0x409AEF: one-time atexit init (boneKeyframeLoop) - 0x7B5F60: IsParticleBufferEmpty (reads game particle state) Child recursion calls transformMatrix4x4_SSE directly instead of going through the hook at 0x714260. Detour cleaned up: REF is baseline, SSE activates via ab_use_custom toggle. Diagnostic/bisect/FPU-comparison scaffolding removed. build.zig: bone_sse gets dedicated target with sse4_1+fma+avx features. --- build.zig | 8 +- src/transform44/bone_sse.zig | 2183 +++++++++++-------------------- src/transform44/transform44.zig | 286 +--- 3 files changed, 762 insertions(+), 1715 deletions(-) diff --git a/build.zig b/build.zig index 2605dca..c616c1a 100644 --- a/build.zig +++ b/build.zig @@ -63,11 +63,17 @@ pub fn build(b: *std.Build) void { .optimize = .ReleaseFast, }), }); + const bone_sse_target = b.resolveTargetQuery(.{ + .cpu_arch = .x86, + .os_tag = .windows, + .abi = .msvc, + .cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }), + }); const bone_sse_obj = b.addObject(.{ .name = "bone_sse", .root_module = b.createModule(.{ .root_source_file = b.path("src/transform44/bone_sse.zig"), - .target = target, + .target = bone_sse_target, .optimize = .ReleaseFast, }), }); diff --git a/src/transform44/bone_sse.zig b/src/transform44/bone_sse.zig index f5d1811..13e75ab 100644 --- a/src/transform44/bone_sse.zig +++ b/src/transform44/bone_sse.zig @@ -1,13 +1,12 @@ -//! SSE-optimized transformMatrix4x4 reimplementation. +//! Pure Zig SSE/FMA implementation of transformMatrix4x4 (0x714260). //! -//! Full standalone replacement for the 17703-byte bone transform engine at 0x714260. -//! Compiled ReleaseFast even in Debug builds (separate compilation unit pattern). -//! All helper functions (findInterpolationIndices, interpolateAnimationKeyframes, -//! scaleMatrix3x3ByVector, ApplyTranslationMatrix, rotateMatrixByQuaternion) are -//! reimplemented inline — no calls back to original game code. +//! Zero calls to game functions except: +//! - 0x409AEF: one-time global init (atexit registration) +//! - 0x7B5F60: IsParticleBufferEmpty (reads game state we can't replicate) +//! - 0x714260: recursive call for child SceneObjects (goes through hook) //! -//! Only external call: the original transformMatrix4x4 via the hook's callOriginal -//! for attachment recursion (the detour auto-dispatches to this SSE version). +//! Compiled with SSE4.1 + FMA + AVX for this compilation unit only. +//! Uses @mulAdd for FMA, @Vector(4, f32) for SIMD matrix ops. const V4 = @Vector(4, f32); @@ -17,24 +16,24 @@ const V4 = @Vector(4, f32); const SO = struct { const model_data_ptr: u32 = 0x010; - const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value - const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header + const anim_ctx_ptr: u32 = 0x02C; + const model_ctr_ptr: u32 = 0x030; const sync_value: u32 = 0x040; - const search_data_base: u32 = 0x04C; // prev timestamp for delta + const search_data_base: u32 = 0x04C; const emitter_flag: u32 = 0x050; - const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array - const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS + const gs_values_ptr: u32 = 0x064; + const gs_time_base: u32 = 0x068; const child_padding: u32 = 0x084; const anim_frame_ctr: u32 = 0x08C; - const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs - const bone_out_ptr: u32 = 0x094; // output bone matrices + const bone_rt_base: u32 = 0x090; + const bone_out_ptr: u32 = 0x094; const tex_anim_out: u32 = 0x0A0; const color_anim_out: u32 = 0x0A8; const scale1: u32 = 0x0AC; const scale2: u32 = 0x0B0; const scale3: u32 = 0x0B4; - const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward) - const world_xform: u32 = 0x10C; // float[16] world transform + const bb_row0: u32 = 0x0FC; + const world_xform: u32 = 0x10C; const field_17c: u32 = 0x17C; const field_180: u32 = 0x180; const field_184: u32 = 0x184; @@ -44,8 +43,8 @@ const SO = struct { const render_scale_x: u32 = 0x194; const render_scale_y: u32 = 0x198; const render_scale_z: u32 = 0x19C; - const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children) - const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children) + const world_pos: u32 = 0x1A0; + const render_pri: u32 = 0x1AC; const hierarchy_ptr: u32 = 0x1C8; const emitter_ctx: u32 = 0x1CC; const field_1d8: u32 = 0x1D8; @@ -58,23 +57,33 @@ const SO = struct { const add_remaining: u32 = 0x3D8; }; -// Bone runtime struct offsets (within 0x118-byte per-bone runtime) const BR = struct { - // Translation interpolation state - const trans_idx0: u32 = 0x00; // [0] lower keyframe index - const trans_idx1: u32 = 0x04; // [1] upper keyframe index - const trans_t: u32 = 0x08; // [2] interpolation factor (float bits) - const trans_x: u32 = 0x0C; // [3] interpolated translation X - const trans_y: u32 = 0x10; // [4] Y - const trans_z: u32 = 0x14; // [5] Z - // Secondary translation (crossfade) + const trans_idx0: u32 = 0x00; + const trans_idx1: u32 = 0x04; + const trans_t: u32 = 0x08; + const trans_x: u32 = 0x0C; + const trans_y: u32 = 0x10; + const trans_z: u32 = 0x14; const trans2_idx0: u32 = 0x18; const trans2_idx1: u32 = 0x1C; const trans2_t: u32 = 0x20; const trans2_x: u32 = 0x24; const trans2_y: u32 = 0x28; const trans2_z: u32 = 0x2C; - // Scale interpolation state (at puVar20 + 0x1a = offset 0x68) + const rot_idx0: u32 = 0x30; + const rot_idx1: u32 = 0x34; + const rot_t: u32 = 0x38; + const rot_x: u32 = 0x3C; + const rot_y: u32 = 0x40; + const rot_z: u32 = 0x44; + const rot_w: u32 = 0x48; + const rot2_idx0: u32 = 0x4C; + const rot2_idx1: u32 = 0x50; + const rot2_t: u32 = 0x54; + const rot2_x: u32 = 0x58; + const rot2_y: u32 = 0x5C; + const rot2_z: u32 = 0x60; + const rot2_w: u32 = 0x64; const scale_idx0: u32 = 0x68; const scale_idx1: u32 = 0x6C; const scale_t: u32 = 0x70; @@ -87,99 +96,61 @@ const BR = struct { const scale2_x: u32 = 0x8C; const scale2_y: u32 = 0x90; const scale2_z: u32 = 0x94; - // Primary animation time range - const prim_time: u32 = 0x98; // puVar20[0x26] - const prim_track: u32 = 0x9C; // puVar20[0x27] - const prim_anim: u32 = 0xA0; // puVar20[0x28] - const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index - // Secondary animation time range (crossfade) - const sec_start: u32 = 0xA8; // puVar20[0x2a] - const sec_end: u32 = 0xAC; // puVar20[0x2b] - const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion - const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e] - // Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30) - const rot_idx0: u32 = 0x30; - const rot_idx1: u32 = 0x34; - const rot_t: u32 = 0x38; - const rot_x: u32 = 0x3C; - const rot_y: u32 = 0x40; - const rot_z: u32 = 0x44; - const rot_w: u32 = 0x48; - // Secondary rotation - const rot2_idx0: u32 = 0x4C; - const rot2_idx1: u32 = 0x50; - const rot2_t: u32 = 0x54; - const rot2_x: u32 = 0x58; - const rot2_y: u32 = 0x5C; - const rot2_z: u32 = 0x60; - const rot2_w: u32 = 0x64; - // Secondary time range - const sec_time: u32 = 0xC4; // puVar20[0x31] - const sec_track: u32 = 0xC8; // puVar20[0x32] - const sec_slot: u32 = 0xD0; // puVar20[0x34] - const sec_start2: u32 = 0xD4; // puVar20[0x35] - const sec_end2: u32 = 0xD8; // puVar20[0x36] - const sec_offset2: u32 = 0xE4; // puVar20[0x39] - // Flags and weights - const flags2: u32 = 0xF4; // puVar20[0x3d] - const crossfade_end: u32 = 0x100; // puVar20[0x40] - const crossfade_inv: u32 = 0x104; // puVar20[0x41] - const crossfade_weight: u32 = 0x108; // puVar20[0x42] - const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade - const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c] + const prim_time: u32 = 0x98; + const prim_track: u32 = 0x9C; + const prim_anim: u32 = 0xA0; + const anim_slot: u32 = 0xA4; + const sec_start: u32 = 0xA8; + const sec_end: u32 = 0xAC; + const time_scale: u32 = 0xB0; + const sec_anim_offset: u32 = 0xB8; + const sec_time: u32 = 0xC4; + const sec_track: u32 = 0xC8; + const sec_slot: u32 = 0xD0; + const sec_start2: u32 = 0xD4; + const sec_end2: u32 = 0xD8; + const sec_offset2: u32 = 0xE4; + const flags2: u32 = 0xF4; + const crossfade_end: u32 = 0x100; + const crossfade_inv: u32 = 0x104; + const crossfade_weight: u32 = 0x108; + const blend_weight: u32 = 0x10C; + const bone_flag_cache: u32 = 0xF0; }; -// OldAnimationBlock struct offsets (28 bytes = 0x1C per track in v256 M2) -// Layout verified from M2 format + decompilation cross-reference: -// pMVar23->m31 (bone_def+0x34) = rot block+0x0C = nTimestamps (gates rotation) -// pMVar23->m12 (bone_def+0x18) = trans block+0x0C = nTimestamps (gates translation) -// pMVar23[1].m10 (bone_def+0x50) = scale block+0x0C = nTimestamps (gates scale) const AD = struct { - const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp) - const time_index: u32 = 0x02; // i16: global sequence index (-1 = none) - const track_count_flag: u32 = 0x04; // nRanges: 0 = single track - const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs - const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count - const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array - const nvalues: u32 = 0x14; // nValues: number of value entries - const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data + const interp_mode: u32 = 0x00; + const time_index: u32 = 0x02; + const track_count_flag: u32 = 0x04; + const keyframe_ranges: u32 = 0x08; + const keyframe_count: u32 = 0x0C; + const timestamps_ptr: u32 = 0x10; + const nvalues: u32 = 0x14; + const keyframe_base: u32 = 0x18; }; -// M2CompBone struct offsets (0x6C = 108 bytes per bone in v256 model) -// Layout: 12 bytes fixed header + 3x28 byte OldAnimationBlock tracks + 12 bytes pivot -// Track order: translation, rotation, scale (standard M2 order) const BD = struct { - const key_id: u32 = 0x00; // i32: key bone ID - const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.) - const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes - // Translation OldAnimationBlock (28 bytes, +0x0C to +0x27) + const key_id: u32 = 0x00; + const flags: u32 = 0x04; + const parent_bone: u32 = 0x08; const trans_anim: u32 = 0x0C; - const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation - // Rotation OldAnimationBlock (28 bytes, +0x28 to +0x43) + const trans_nts: u32 = 0x18; const rot_anim: u32 = 0x28; - const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation - // Scale OldAnimationBlock (28 bytes, +0x44 to +0x5F) + const rot_nts: u32 = 0x34; const scale_anim: u32 = 0x44; - const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation - // Pivot point (12 bytes, +0x60 to +0x6B) + const scale_nts: u32 = 0x50; const pivot_x: u32 = 0x60; const pivot_y: u32 = 0x64; const pivot_z: u32 = 0x68; }; -// Game constants -const ZERO_F: f32 = 0.0; -const ONE_F: f32 = 1.0; -const THREE_F: f32 = 3.0; -const BILLBOARD_EPSILON: f32 = @bitCast(@as(u32, 0x3727c5ac)); // ~1e-5, from DAT_008029d4 -const SHORT_TO_FLOAT: f32 = @bitCast(@as(u32, 0x38000000)); // 1/32768, DAT_00811610 (short→float conversion) -const HERMITE_3: f32 = 3.0; // DAT_0080297c -const HERMITE_5: f32 = 5.0; // DAT_00802990 (used as 3*5/3 in some bezier) - -// MSVC CRT sin/cos — linked from the WoW process -extern fn sinf(f32) f32; -extern fn cosf(f32) f32; - +// Runtime constants read from game memory (patched at startup) +fn getShortToFloat() f32 { + return rf32(0x00811610); +} +fn getBillboardEpsilon() f32 { + return rf32(0x008029d4); +} // ============================================================================= // Memory access helpers @@ -203,7 +174,6 @@ inline fn ri16(addr: u32) i16 { inline fn ru8(addr: u32) u8 { return @as(*const u8, @ptrFromInt(addr)).*; } - inline fn wu32(addr: u32, v: u32) void { @as(*align(1) u32, @ptrFromInt(addr)).* = v; } @@ -216,7 +186,6 @@ inline fn wu16(addr: u32, v: u16) void { inline fn wu8(addr: u32, v: u8) void { @as(*u8, @ptrFromInt(addr)).* = v; } - inline fn fbits(v: f32) u32 { return @bitCast(v); } @@ -225,68 +194,79 @@ inline fn ufloat(v: u32) f32 { } // ============================================================================= -// Math helpers — using @Vector(4, f32) for SSE +// Math helpers — pure Zig with @Vector and @mulAdd for FMA // ============================================================================= inline fn splat(v: f32) V4 { return @splat(v); } -/// 3-component lerp: a + (b - a) * t. Keyframes are 12 bytes (3 floats) apart. +/// Load 4 floats from memory as V4 +inline fn loadV4(addr: u32) V4 { + return .{ rf32(addr), rf32(addr + 4), rf32(addr + 8), rf32(addr + 12) }; +} + +/// Store V4 to memory +inline fn storeV4(addr: u32, v: V4) void { + wf32(addr, v[0]); + wf32(addr + 4, v[1]); + wf32(addr + 8, v[2]); + wf32(addr + 12, v[3]); +} + +/// 3-component lerp using FMA: a + (b - a) * t inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 { - const ax = rf32(a_addr); - const ay = rf32(a_addr + 4); - const az = rf32(a_addr + 8); - const bx = rf32(b_addr); - const by = rf32(b_addr + 4); - const bz = rf32(b_addr + 8); return .{ - (bx - ax) * t + ax, - (by - ay) * t + ay, - (bz - az) * t + az, + @mulAdd(f32, rf32(b_addr) - rf32(a_addr), t, rf32(a_addr)), + @mulAdd(f32, rf32(b_addr + 4) - rf32(a_addr + 4), t, rf32(a_addr + 4)), + @mulAdd(f32, rf32(b_addr + 8) - rf32(a_addr + 8), t, rf32(a_addr + 8)), }; } -/// Blend primary and secondary results: primary + (secondary - primary) * weight -inline fn blendVec3(primary: [3]f32, secondary: [3]f32, weight: f32) [3]f32 { - return .{ - (secondary[0] - primary[0]) * weight + primary[0], - (secondary[1] - primary[1]) * weight + primary[1], - (secondary[2] - primary[2]) * weight + primary[2], - }; +/// Vec3 squared magnitude (replaces game's 0x4549F0) +inline fn vec3SqMag(addr: u32) f32 { + const x = rf32(addr); + const y = rf32(addr + 4); + const z = rf32(addr + 8); + return @mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x)); } -/// Scale 3x3 rotation portion of a row-major 4x4 matrix by per-axis scale. -/// Row 0 *= scale.x, Row 1 *= scale.y, Row 2 *= scale.z +/// Vec3 squared magnitude from floats +inline fn vec3SqMagF(x: f32, y: f32, z: f32) f32 { + return @mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x)); +} + +/// Scale 3x3 rotation portion of row-major 4x4: Row N *= scale[N] inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void { - // Row 0 (offsets 0x00, 0x04, 0x08) wf32(mat + 0x00, rf32(mat + 0x00) * sx); wf32(mat + 0x04, rf32(mat + 0x04) * sx); wf32(mat + 0x08, rf32(mat + 0x08) * sx); - // Row 1 (offsets 0x10, 0x14, 0x18) wf32(mat + 0x10, rf32(mat + 0x10) * sy); wf32(mat + 0x14, rf32(mat + 0x14) * sy); wf32(mat + 0x18, rf32(mat + 0x18) * sy); - // Row 2 (offsets 0x20, 0x24, 0x28) wf32(mat + 0x20, rf32(mat + 0x20) * sz); wf32(mat + 0x24, rf32(mat + 0x24) * sz); wf32(mat + 0x28, rf32(mat + 0x28) * sz); } -/// Apply translation through rotation matrix: -/// mat[3][0] += dot(mat[0], t) -/// mat[3][1] += dot(mat[1], t) -/// mat[3][2] += dot(mat[2], t) -inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void { - wf32(mat + 0x30, tx * rf32(mat + 0x00) + ty * rf32(mat + 0x10) + tz * rf32(mat + 0x20) + rf32(mat + 0x30)); - wf32(mat + 0x34, tx * rf32(mat + 0x04) + ty * rf32(mat + 0x14) + tz * rf32(mat + 0x24) + rf32(mat + 0x34)); - wf32(mat + 0x38, tx * rf32(mat + 0x08) + ty * rf32(mat + 0x18) + tz * rf32(mat + 0x28) + rf32(mat + 0x38)); +/// Scale 3x3 rotation from a vec3 pointer in memory +inline fn scaleMatrix3x3FromPtr(mat: u32, vec3_ptr: u32) void { + scaleMatrix3x3(mat, rf32(vec3_ptr), rf32(vec3_ptr + 4), rf32(vec3_ptr + 8)); } -/// Quaternion → rotation matrix: OVERWRITES mat with the rotation matrix. -/// Matches the original game function at 0x74B6BB which writes directly -/// without multiplying by existing matrix contents. -/// Used in the bone loop where the matrix starts as identity. +/// Apply translation: mat[3] += mat * t (dot product per column) +inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void { + wf32(mat + 0x30, @mulAdd(f32, tz, rf32(mat + 0x20), @mulAdd(f32, ty, rf32(mat + 0x10), @mulAdd(f32, tx, rf32(mat + 0x00), rf32(mat + 0x30))))); + wf32(mat + 0x34, @mulAdd(f32, tz, rf32(mat + 0x24), @mulAdd(f32, ty, rf32(mat + 0x14), @mulAdd(f32, tx, rf32(mat + 0x04), rf32(mat + 0x34))))); + wf32(mat + 0x38, @mulAdd(f32, tz, rf32(mat + 0x28), @mulAdd(f32, ty, rf32(mat + 0x18), @mulAdd(f32, tx, rf32(mat + 0x08), rf32(mat + 0x38))))); +} + +/// Apply translation from vec3 pointer in memory +inline fn applyTranslationFromPtr(mat: u32, vec3_ptr: u32) void { + applyTranslation(mat, rf32(vec3_ptr), rf32(vec3_ptr + 4), rf32(vec3_ptr + 8)); +} + +/// Quaternion → rotation matrix: OVERWRITES mat (does not multiply) inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { const xx2 = qx * (qx + qx); const xy2 = qx * (qy + qy); @@ -297,360 +277,233 @@ inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void const wx2 = qw * (qx + qx); const wy2 = qw * (qy + qy); const wz2 = qw * (qz + qz); - - // Row 0 wf32(mat + 0x00, 1.0 - (yy2 + zz2)); wf32(mat + 0x04, xy2 + wz2); wf32(mat + 0x08, xz2 - wy2); wf32(mat + 0x0C, 0); - // Row 1 wf32(mat + 0x10, xy2 - wz2); wf32(mat + 0x14, 1.0 - (xx2 + zz2)); wf32(mat + 0x18, yz2 + wx2); wf32(mat + 0x1C, 0); - // Row 2 wf32(mat + 0x20, xz2 + wy2); wf32(mat + 0x24, yz2 - wx2); wf32(mat + 0x28, 1.0 - (xx2 + yy2)); wf32(mat + 0x2C, 0); - // Row 3 (translation = zero, w = 1) wf32(mat + 0x30, 0); wf32(mat + 0x34, 0); wf32(mat + 0x38, 0); wf32(mat + 0x3C, 1); } -/// Quaternion → rotation matrix, then multiply: mat = quat_rot * mat. -/// Standard quat→mat conversion + SSE 4x4 matrix multiply. -/// Used in bone keyframe processing where matrix already has content. +/// Build rotation matrix from quaternion at memory address +inline fn buildRotationMatrixFromPtr(mat: u32, quat_ptr: u32) void { + buildRotationMatrix(mat, rf32(quat_ptr), rf32(quat_ptr + 4), rf32(quat_ptr + 8), rf32(quat_ptr + 12)); +} + +/// 4x4 matrix multiply: dst = a * b (row-major). Safe for dst==a or dst==b. +/// Uses FMA: 4 broadcasts + 1 mul + 3 FMA per row = 16 ops total. +fn matMul4x4(dst: u32, a: u32, b: u32) void { + const b0 = loadV4(b); + const b1 = loadV4(b + 0x10); + const b2 = loadV4(b + 0x20); + const b3 = loadV4(b + 0x30); + // Pre-load all of A in case dst aliases a + var a_rows: [4]V4 = undefined; + inline for (0..4) |i| { + a_rows[i] = loadV4(a + @as(u32, @intCast(i)) * 0x10); + } + inline for (0..4) |i| { + const row = @mulAdd(V4, splat(a_rows[i][3]), b3, @mulAdd(V4, splat(a_rows[i][2]), b2, @mulAdd(V4, splat(a_rows[i][1]), b1, splat(a_rows[i][0]) * b0))); + storeV4(dst + @as(u32, @intCast(i)) * 0x10, row); + } +} + +/// Quaternion → rotation matrix, then multiply: mat = quat_rot * mat inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { - const xx2 = qx * (qx + qx); - const xy2 = qx * (qy + qy); - const xz2 = qx * (qz + qz); - const yy2 = qy * (qy + qy); - const yz2 = qy * (qz + qz); - const zz2 = qz * (qz + qz); - const wx2 = qw * (qx + qx); - const wy2 = qw * (qy + qy); - const wz2 = qw * (qz + qz); - - // Rotation matrix from quaternion (row-major) - const rot: [16]f32 = .{ - 1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0, - xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0, - xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0, - 0, 0, 0, 1, - }; - - // SSE matrix multiply: result = rot * mat - var tmp: [16]f32 = undefined; - const r0: V4 = .{ rf32(mat + 0x00), rf32(mat + 0x04), rf32(mat + 0x08), rf32(mat + 0x0C) }; - const r1: V4 = .{ rf32(mat + 0x10), rf32(mat + 0x14), rf32(mat + 0x18), rf32(mat + 0x1C) }; - const r2: V4 = .{ rf32(mat + 0x20), rf32(mat + 0x24), rf32(mat + 0x28), rf32(mat + 0x2C) }; - const r3: V4 = .{ rf32(mat + 0x30), rf32(mat + 0x34), rf32(mat + 0x38), rf32(mat + 0x3C) }; - - inline for (0..4) |i| { - const b = i * 4; - const out = splat(rot[b]) * r0 + splat(rot[b + 1]) * r1 + splat(rot[b + 2]) * r2 + splat(rot[b + 3]) * r3; - tmp[b] = out[0]; - tmp[b + 1] = out[1]; - tmp[b + 2] = out[2]; - tmp[b + 3] = out[3]; - } - - // Copy back - inline for (0..16) |i| { - wf32(mat + @as(u32, @intCast(i)) * 4, tmp[i]); - } + var tmp: [64]u8 align(16) = undefined; + const tmp_addr = @intFromPtr(&tmp); + buildRotationMatrix(tmp_addr, qx, qy, qz, qw); + matMul4x4(mat, tmp_addr, mat); } -/// 4x4 matrix multiply: dst = left × right (row-major) -/// Handles aliasing: dst may equal left or right. -inline fn matMul4x4(dst: u32, left: u32, right: u32) void { - const r0: V4 = .{ rf32(right + 0x00), rf32(right + 0x04), rf32(right + 0x08), rf32(right + 0x0C) }; - const r1: V4 = .{ rf32(right + 0x10), rf32(right + 0x14), rf32(right + 0x18), rf32(right + 0x1C) }; - const r2: V4 = .{ rf32(right + 0x20), rf32(right + 0x24), rf32(right + 0x28), rf32(right + 0x2C) }; - const r3: V4 = .{ rf32(right + 0x30), rf32(right + 0x34), rf32(right + 0x38), rf32(right + 0x3C) }; - - // Read all left rows before writing (handles dst==left aliasing) - var result: [16]f32 = undefined; - inline for (0..4) |i| { - const b = @as(u32, @intCast(i)) * 0x10; - const row = splat(rf32(left + b)) * r0 + splat(rf32(left + b + 4)) * r1 + splat(rf32(left + b + 8)) * r2 + splat(rf32(left + b + 12)) * r3; - result[i * 4 + 0] = row[0]; - result[i * 4 + 1] = row[1]; - result[i * 4 + 2] = row[2]; - result[i * 4 + 3] = row[3]; - } - inline for (0..16) |i| { - wf32(dst + @as(u32, @intCast(i)) * 4, result[i]); - } -} - -/// IsParticleBufferEmpty reimplemented from assembly at 0x7B5F60. -/// Returns true if buffer is NOT empty (has active particles). -/// Recursive: checks [this+0x64], then iterates children at [this+0x80]. -fn isParticleBufferNotEmpty(ptr: u32) bool { - if (ru32(ptr + 0x64) != 0) return true; - const count = ru32(ptr + 0x7C); - var i: u32 = 0; - while (i < count) : (i += 1) { - const child = ru32(ptr + 0x80 + i * 4); - if (isParticleBufferNotEmpty(child)) return true; - } - return false; +/// Rotate matrix by quaternion at memory address +inline fn rotateByQuaternionFromPtr(mat: u32, quat_ptr: u32) void { + rotateByQuaternion(mat, rf32(quat_ptr), rf32(quat_ptr + 4), rf32(quat_ptr + 8), rf32(quat_ptr + 12)); } /// Copy 16 floats (4x4 matrix) inline fn copyMat4(dst: u32, src: u32) void { - comptime var i: u32 = 0; - inline while (i < 64) : (i += 4) { - wu32(dst + i, ru32(src + i)); + inline for (0..4) |i| { + storeV4(dst + @as(u32, @intCast(i)) * 0x10, loadV4(src + @as(u32, @intCast(i)) * 0x10)); } } -/// Set identity matrix (16 floats) +/// Set identity matrix inline fn setIdentity(dst: u32) void { - inline for (0..16) |i| { - const val: f32 = if (i == 0 or i == 5 or i == 10 or i == 15) 1.0 else 0.0; - wf32(dst + @as(u32, @intCast(i)) * 4, val); - } + storeV4(dst, .{ 1, 0, 0, 0 }); + storeV4(dst + 0x10, .{ 0, 1, 0, 0 }); + storeV4(dst + 0x20, .{ 0, 0, 1, 0 }); + storeV4(dst + 0x30, .{ 0, 0, 0, 1 }); } -/// Normalize a 3-component vector in memory at addr. Uses squaredMagnitude + sqrt + divide. -/// Matches the original's pattern: call squaredMagnitude, sqrt, check epsilon, divide. +/// Normalize vec3 in memory. Returns unchanged if length < epsilon. inline fn normalizeVec3InPlace(addr: u32) void { - const x = rf32(addr); - const y = rf32(addr + 4); - const z = rf32(addr + 8); - const len = @sqrt(x * x + y * y + z * z); - if (@abs(len) >= BILLBOARD_EPSILON) { + const sq = vec3SqMag(addr); + const len = @sqrt(sq); + if (@abs(len) >= getBillboardEpsilon()) { const inv = 1.0 / len; - wf32(addr, x * inv); - wf32(addr + 4, y * inv); - wf32(addr + 8, z * inv); + wf32(addr, rf32(addr) * inv); + wf32(addr + 4, rf32(addr + 4) * inv); + wf32(addr + 8, rf32(addr + 8) * inv); } } -/// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small. -inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 { - const len_sq = x * x + y * y + z * z; - const len = @sqrt(len_sq); - if (len < BILLBOARD_EPSILON) return .{ x, y, z }; - const inv = 1.0 / len; - return .{ x * inv, y * inv, z * inv }; -} - -/// Cross product of two 3-component vectors -inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 { - return .{ - ay * bz - az * by, - az * bx - ax * bz, - ax * by - ay * bx, - }; +/// Float truncation (replaces game's __ftol at 0x40A2B0) +inline fn ftol(delta: i32, scale_addr: u32) i32 { + const f = @as(f32, @floatFromInt(delta)) * rf32(scale_addr); + // @intFromFloat truncates toward zero, matching MSVC __ftol + return @intFromFloat(f); } // ============================================================================= -// findInterpolationIndices — reimplemented from 0x713d50 (334 bytes) -// -// Three-tier search with temporal coherence: -// 1. Forward linear scan (hot path, 1-4 iterations typical) -// 2. Backward linear scan (negative delta) -// 3. Binary search (fallback) -// -// Output: indices[0] = lower index, [1] = upper index, [2] = interpolation t (float bits) +// Interpolation — pure Zig reimplementations // ============================================================================= -fn findInterpIdx( - this: u32, - search_value: u32, - track_index: u32, - anim_data: u32, - output: u32, -) void { - var min_idx: u32 = undefined; - var max_idx: u32 = undefined; +/// findInterpIdx: temporal-coherence keyframe search. +/// Reimplements game function at 0x713D50 (334 bytes). +/// Reads anim_data for timestamps/ranges, searches for bracket, computes t. +/// output[0] = lower idx (also cached position), [1] = upper idx, [2] = t bits. +fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) void { + const n_ranges = ru32(anim_data + AD.track_count_flag); + const n_timestamps = ru32(anim_data + AD.keyframe_count); + const ts_base = ru32(anim_data + AD.timestamps_ptr); - if (ru32(anim_data + AD.track_count_flag) == 0) { - min_idx = 0; - max_idx = ru32(anim_data + AD.keyframe_count) -% 1; - } else { - const ranges = ru32(anim_data + AD.keyframe_ranges); - max_idx = ru32(ranges + 4 + track_index * 8); - min_idx = ru32(ranges + track_index * 8); - } - - if (max_idx <= min_idx) { - wu32(output, min_idx); - wu32(output + 4, min_idx); - wu32(output + 8, 0); - return; - } - - // Check for global sequence override - var sv = search_value; + // Global sequence override const time_idx = ri16(anim_data + AD.time_index); - if (time_idx != -1) { - sv = ru32(ru32(this + SO.gs_values_ptr) + @as(u32, @bitCast(@as(i32, @intCast(time_idx)))) * 4); + const search: u32 = if (time_idx >= 0) blk: { + const gs_vals = ru32(this + SO.gs_values_ptr); + break :blk ru32(gs_vals + @as(u32, @intCast(time_idx)) * 4); + } else search_value; + + // Determine range for this track + var range_start: u32 = 0; + var range_count: u32 = n_timestamps; + if (n_ranges != 0 and n_ranges > track_index) { + const ranges = ru32(anim_data + AD.keyframe_ranges); + range_start = ru32(ranges + track_index * 8); + range_count = ru32(ranges + track_index * 8 + 4); } - const timestamps = ru32(anim_data + AD.timestamps_ptr); - var cur_idx = ru32(output); - const delta = sv -% ru32(timestamps + cur_idx * 4); - - if (delta < 500) { - // Forward linear scan (hot path) - if (cur_idx < max_idx) { - var tp = timestamps + 4 + cur_idx * 4; - while (cur_idx < max_idx) { - if (sv < ru32(tp)) break; - cur_idx += 1; - tp += 4; - } - } - } else if (delta < 0xFFFFFF0C) { - // Not within forward range and not backward — try forward from min or binary search - const delta_from_min = sv -% ru32(timestamps + min_idx * 4); - if (delta_from_min < 500) { - // Forward from min - var tp = timestamps + 4 + min_idx * 4; - cur_idx = min_idx; - while (min_idx < max_idx) { - cur_idx = min_idx; - if (sv < ru32(tp)) break; - min_idx += 1; - tp += 4; - cur_idx = min_idx; - } - } else { - // Binary search - var lo = min_idx; - var hi = max_idx; - while (lo < hi) { - cur_idx = (hi + lo) >> 1; - if (sv < ru32(timestamps + cur_idx * 4)) { - hi = cur_idx -% 1; - } else { - lo = cur_idx + 1; - if (sv < ru32(timestamps + 4 + cur_idx * 4)) break; - } - cur_idx = lo; - } - } - } else { - // Backward linear scan - if (min_idx < cur_idx) { - var tp = timestamps + cur_idx * 4; - while (min_idx < cur_idx) { - if (ru32(tp) <= sv) break; - cur_idx -= 1; - tp -= 4; - } - } - } - - const next_idx = cur_idx + 1; - if (ru32(anim_data + AD.keyframe_count) <= next_idx) { - wu32(output + 4, cur_idx); - wu32(output, cur_idx); + if (range_count == 0) return; + if (range_count == 1) { + wu32(output, range_start); + wu32(output + 4, range_start); wu32(output + 8, 0); return; } - wu32(output, cur_idx); - wu32(output + 4, next_idx); - const ts_cur = ri32(timestamps + cur_idx * 4); - const ts_next = ri32(timestamps + next_idx * 4); - const denom = ts_next - ts_cur; - if (denom != 0) { - const t: f32 = @as(f32, @floatFromInt(@as(i32, @bitCast(sv)) - ts_cur)) / @as(f32, @floatFromInt(denom)); - wu32(output + 8, fbits(t)); - } else { + const last = range_start + range_count - 1; + const first_ts = ru32(ts_base + range_start * 4); + const last_ts = ru32(ts_base + last * 4); + + if (search <= first_ts) { + wu32(output, range_start); + wu32(output + 4, range_start); wu32(output + 8, 0); + return; } + if (search >= last_ts) { + wu32(output, last); + wu32(output + 4, last); + wu32(output + 8, 0); + return; + } + + // Temporal coherence: start from cached index + var idx = ru32(output); + if (idx < range_start or idx >= last) idx = range_start; + + // Forward scan (hot path — animations advance forward) + if (ru32(ts_base + idx * 4) <= search) { + while (idx < last and ru32(ts_base + (idx + 1) * 4) <= search) { + idx += 1; + } + } else { + // Backward scan + while (idx > range_start and ru32(ts_base + idx * 4) > search) { + idx -= 1; + } + } + + // Compute interpolation factor + const ts_lo = ru32(ts_base + idx * 4); + const ts_hi = ru32(ts_base + (idx + 1) * 4); + const t: f32 = if (ts_hi > ts_lo) + @as(f32, @floatFromInt(search - ts_lo)) / @as(f32, @floatFromInt(ts_hi - ts_lo)) + else + 0.0; + + wu32(output, idx); + wu32(output + 4, idx + 1); + wu32(output + 8, @bitCast(t)); } -// ============================================================================= -// interpolateAnimationKeyframes — reimplemented from 0x713ea0 -// -// Calls findInterpIdx, does 4-component lerp (for quaternions). -// If crossfade active, does secondary lookup + blend. -// Output buffer layout: [idx0, idx1, t, x, y, z, w, sec_idx0, sec_idx1, sec_t, sx, sy, sz, sw] -// ============================================================================= - -inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) void { +/// Quaternion keyframe interpolation (replaces game's 0x713EA0). +/// Reads CompQuat (4×i16, 8 bytes per keyframe), converts to float, lerps. +/// Writes: output[0..2]=indices/t, output[3..6]=qx/qy/qz/qw, output[7..13]=secondary. +fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) void { findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); - const idx0 = ru32(output); - const interp_mode = ri16(anim_data + AD.interp_mode); + const mode = ri16(anim_data + AD.interp_mode); const kf_base = ru32(anim_data + AD.keyframe_base); + const s2f = getShortToFloat(); - if (interp_mode == 0) { - // No interpolation — copy directly (4 components, 16 bytes per keyframe) - const src = kf_base + idx0 * 0x10; - wu32(output + 0x0C, ru32(src)); - wu32(output + 0x10, ru32(src + 4)); - wu32(output + 0x14, ru32(src + 8)); - wu32(output + 0x18, ru32(src + 12)); + if (mode == 0) { + const src = kf_base + ru32(output) * 8; + inline for (0..4) |i| { + wf32(output + 0x0C + @as(u32, @intCast(i)) * 4, @as(f32, @floatFromInt(@as(i32, ri16(src + @as(u32, @intCast(i)) * 2)))) * s2f); + } return; } + // Lerp (mode 1+) const t = ufloat(ru32(output + 8)); - const a = kf_base + idx0 * 0x10; - const b = kf_base + ru32(output + 4) * 0x10; + const src0 = kf_base + ru32(output) * 8; + const src1 = kf_base + ru32(output + 4) * 8; + inline for (0..4) |i| { + const off: u32 = @intCast(i * 2); + const a = @as(f32, @floatFromInt(@as(i32, ri16(src0 + off)))) * s2f; + const b = @as(f32, @floatFromInt(@as(i32, ri16(src1 + off)))) * s2f; + wf32(output + 0x0C + @as(u32, @intCast(i)) * 4, @mulAdd(f32, b - a, t, a)); + } - // 4-component lerp - wf32(output + 0x0C, (rf32(b) - rf32(a)) * t + rf32(a)); - wf32(output + 0x10, (rf32(b + 4) - rf32(a + 4)) * t + rf32(a + 4)); - wf32(output + 0x14, (rf32(b + 8) - rf32(a + 8)) * t + rf32(a + 8)); - wf32(output + 0x18, (rf32(b + 12) - rf32(a + 12)) * t + rf32(a + 12)); - - // Crossfade blend - const blend = ufloat(ru32(bone_rt + BR.blend_weight)); - if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { + // Crossfade + const bw = ufloat(ru32(bone_rt + BR.blend_weight)); + if (bw != 0.0 and ri16(anim_data + AD.time_index) == -1) { findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x1C); - - const si0 = ru32(output + 0x1C); - const si1 = ru32(output + 0x20); const st = ufloat(ru32(output + 0x24)); - const sa = kf_base + si0 * 0x10; - const sb = kf_base + si1 * 0x10; - - // Secondary 4-component lerp - const sx = (rf32(sb) - rf32(sa)) * st + rf32(sa); - const sy = (rf32(sb + 4) - rf32(sa + 4)) * st + rf32(sa + 4); - const sz = (rf32(sb + 8) - rf32(sa + 8)) * st + rf32(sa + 8); - const sw = (rf32(sb + 12) - rf32(sa + 12)) * st + rf32(sa + 12); - wu32(output + 0x28, fbits(sx)); - wu32(output + 0x2C, fbits(sy)); - wu32(output + 0x30, fbits(sz)); - wu32(output + 0x34, fbits(sw)); - - // Blend: primary += (secondary - primary) * weight - wf32(output + 0x0C, (sx - rf32(output + 0x0C)) * blend + rf32(output + 0x0C)); - wf32(output + 0x10, (sy - rf32(output + 0x10)) * blend + rf32(output + 0x10)); - wf32(output + 0x14, (sz - rf32(output + 0x14)) * blend + rf32(output + 0x14)); - wf32(output + 0x18, (sw - rf32(output + 0x18)) * blend + rf32(output + 0x18)); + const ssrc0 = kf_base + ru32(output + 0x1C) * 8; + const ssrc1 = kf_base + ru32(output + 0x20) * 8; + inline for (0..4) |i| { + const off: u32 = @intCast(i * 2); + const a = @as(f32, @floatFromInt(@as(i32, ri16(ssrc0 + off)))) * s2f; + const b = @as(f32, @floatFromInt(@as(i32, ri16(ssrc1 + off)))) * s2f; + const sec = @mulAdd(f32, b - a, st, a); + wf32(output + 0x28 + @as(u32, @intCast(i)) * 4, sec); + const pri = rf32(output + 0x0C + @as(u32, @intCast(i)) * 4); + wf32(output + 0x0C + @as(u32, @intCast(i)) * 4, @mulAdd(f32, sec - pri, bw, pri)); + } } } -/// Interpolate a Vec3 track (12 bytes per keyframe) with crossfade support. -/// Writes result to output[3..5] (as u32 float bits). Uses output[0..2] for indices/t, -/// and output[6..11] for secondary crossfade state. -inline fn interpVec3Track( - this: u32, - bone_rt: u32, - anim_data: u32, - output: u32, - blend_weight: f32, -) void { +/// Vec3 track interpolation (12 bytes/kf) with crossfade. +inline fn interpVec3Track(this: u32, bone_rt: u32, anim_data: u32, output: u32, blend_weight: f32) void { findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); - const interp_mode = ri16(anim_data + AD.interp_mode); + const mode = ri16(anim_data + AD.interp_mode); const kf_base = ru32(anim_data + AD.keyframe_base); - if (interp_mode == 0) { - // No interpolation — copy keyframe directly + if (mode == 0) { const src = kf_base + ru32(output) * 0xC; wu32(output + 0x0C, ru32(src)); wu32(output + 0x10, ru32(src + 4)); @@ -666,7 +519,6 @@ inline fn interpVec3Track( wu32(output + 0x10, fbits(result[1])); wu32(output + 0x14, fbits(result[2])); - // Crossfade blend if (blend_weight != 0.0 and ri16(anim_data + AD.time_index) == -1) { findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x18); const st = ufloat(ru32(output + 0x20)); @@ -676,93 +528,70 @@ inline fn interpVec3Track( wu32(output + 0x24, fbits(sec[0])); wu32(output + 0x28, fbits(sec[1])); wu32(output + 0x2C, fbits(sec[2])); - - // Blend - const pri_x = ufloat(ru32(output + 0x0C)); - const pri_y = ufloat(ru32(output + 0x10)); - const pri_z = ufloat(ru32(output + 0x14)); - wu32(output + 0x0C, fbits((sec[0] - pri_x) * blend_weight + pri_x)); - wu32(output + 0x10, fbits((sec[1] - pri_y) * blend_weight + pri_y)); - wu32(output + 0x14, fbits((sec[2] - pri_z) * blend_weight + pri_z)); + inline for (0..3) |i| { + const off: u32 = @intCast(i * 4); + const pri = ufloat(ru32(output + 0x0C + off)); + wf32(output + 0x0C + off, @mulAdd(f32, sec[i] - pri, blend_weight, pri)); + } } } -/// Interpolate a single float track (4 bytes per keyframe) with crossfade. -/// Writes result to output[3] as float bits. -inline fn interpFloatTrack( - this: u32, - bone_rt: u32, - anim_data: u32, - output: u32, -) void { +/// Float track interpolation (4 bytes/kf) with crossfade. +inline fn interpFloatTrack(this: u32, bone_rt: u32, anim_data: u32, output: u32, blend_weight: f32) void { findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); - const interp_mode = ri16(anim_data + AD.interp_mode); - const kf_base = ru32(anim_data + AD.keyframe_base); - - if (interp_mode == 0) { - wu32(output + 0x0C, ru32(kf_base + ru32(output) * 4)); - return; - } - - const t = ufloat(ru32(output + 8)); - const a = rf32(kf_base + ru32(output) * 4); - const b = rf32(kf_base + ru32(output + 4) * 4); - wf32(output + 0x0C, (b - a) * t + a); - - // Crossfade - const blend = ufloat(ru32(bone_rt + BR.blend_weight)); - if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { - findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x10); - const st = ufloat(ru32(output + 0x18)); - const sa = rf32(kf_base + ru32(output + 0x10) * 4); - const sb = rf32(kf_base + ru32(output + 0x14) * 4); - const sec = (sb - sa) * st + sa; - wu32(output + 0x1C, fbits(sec)); - const pri = ufloat(ru32(output + 0x0C)); - wf32(output + 0x0C, (sec - pri) * blend + pri); - } -} - -// ============================================================================= -// Hermite basis functions — used by particle emitter tracks (modes 2, 3) -// h1 = 2t³ - 3t² + 1, h2 = t³ - 2t² + t, h3 = -2t³ + 3t², h4 = t³ - t² -// ============================================================================= - -inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { - const t2 = t * t; - const t3 = t2 * t; - return .{ - .h1 = 2 * t3 - 3 * t2 + 1, - .h2 = t3 - 2 * t2 + t, - .h3 = -2 * t3 + 3 * t2, - .h4 = t3 - t2, - }; -} - -inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { - const u = 1.0 - t; - const t2 = t * t; - const u_sq = u * u; - return .{ - .b0 = u_sq * u, - .b1 = 3 * u_sq * t, - .b2 = 3 * u * t2, - .b3 = t2 * t, - }; -} - -/// Vec3 interpolation with 36-byte keyframes and 4 modes (step/lerp/bezier/hermite). -/// Keyframe layout: [pos Vec3 (12), in_tangent Vec3 (12), out_tangent Vec3 (12)] = 36 bytes. -/// Used by 0x124 particle emitter tracks. Uses bone_rt_base (bone 0) for timing. -fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { - findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); - const mode = ri16(anim_data + AD.interp_mode); const kf_base = ru32(anim_data + AD.keyframe_base); if (mode == 0) { - // Step — copy Vec3 from keyframe at idx0*36 + wu32(output + 0x0C, ru32(kf_base + ru32(output) * 4)); + return; + } + + const t = ufloat(ru32(output + 8)); + const va = rf32(kf_base + ru32(output) * 4); + const vb = rf32(kf_base + ru32(output + 4) * 4); + wf32(output + 0x0C, @mulAdd(f32, vb - va, t, va)); + + if (blend_weight != 0.0 and ri16(anim_data + AD.time_index) == -1) { + findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x10); + const st = ufloat(ru32(output + 0x18)); + const sa = rf32(kf_base + ru32(output + 0x10) * 4); + const sb = rf32(kf_base + ru32(output + 0x14) * 4); + const sec = @mulAdd(f32, sb - sa, st, sa); + wu32(output + 0x1C, fbits(sec)); + const pri = ufloat(ru32(output + 0x0C)); + wf32(output + 0x0C, @mulAdd(f32, sec - pri, blend_weight, pri)); + } +} + +/// Hermite basis functions +inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { + const t2 = t * t; + const t3 = t2 * t; + return .{ + .h1 = @mulAdd(f32, 2, t3, @mulAdd(f32, -3, t2, 1)), + .h2 = @mulAdd(f32, t3, 1, @mulAdd(f32, -2, t2, t)), + .h3 = @mulAdd(f32, -2, t3, 3 * t2), + .h4 = t3 - t2, + }; +} + +/// Bezier (Bernstein) basis functions +inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { + const u = 1.0 - t; + const t2 = t * t; + const u_sq = u * u; + return .{ .b0 = u_sq * u, .b1 = 3 * u_sq * t, .b2 = 3 * u * t2, .b3 = t2 * t }; +} + +/// Vec3 track with 36-byte keyframes (pos+tangents), modes 0-3 + crossfade. +fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { + findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); + const mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + if (mode == 0) { const src = kf_base + ru32(output) * 36; wu32(output + 0x0C, ru32(src)); wu32(output + 0x10, ru32(src + 4)); @@ -775,47 +604,33 @@ fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) const kf_b = kf_base + ru32(output + 4) * 36; if (mode == 1) { - // Linear interpolation const result = lerpVec3(kf_a, kf_b, t); wu32(output + 0x0C, fbits(result[0])); wu32(output + 0x10, fbits(result[1])); wu32(output + 0x14, fbits(result[2])); } else if (mode == 3) { - // Hermite: h1*p0 + h2*m0_out + h3*p1 + h4*m1_in const h = hermiteBasis(t); var i: u32 = 0; while (i < 3) : (i += 1) { const off = i * 4; - const p0 = rf32(kf_a + off); - const m0 = rf32(kf_a + 0x18 + off); // out_tangent - const p1 = rf32(kf_b + off); - const m1 = rf32(kf_b + 0x0C + off); // in_tangent - wf32(output + 0x0C + off, h.h1 * p0 + h.h2 * m0 + h.h3 * p1 + h.h4 * m1); + wf32(output + 0x0C + off, @mulAdd(f32, h.h4, rf32(kf_b + 0x0C + off), @mulAdd(f32, h.h3, rf32(kf_b + off), @mulAdd(f32, h.h2, rf32(kf_a + 0x18 + off), h.h1 * rf32(kf_a + off))))); } } else if (mode == 2) { - // Bezier: b0*p0 + b1*m0_out + b2*m1_in + b3*p1 - const b = bezierBasis(t); + const bz = bezierBasis(t); var i: u32 = 0; while (i < 3) : (i += 1) { const off = i * 4; - const p0 = rf32(kf_a + off); - const m0 = rf32(kf_a + 0x18 + off); // out_tangent (control point) - const p1 = rf32(kf_b + off); - const m1 = rf32(kf_b + 0x0C + off); // in_tangent (control point) - wf32(output + 0x0C + off, b.b0 * p0 + b.b1 * m0 + b.b2 * m1 + b.b3 * p1); + wf32(output + 0x0C + off, @mulAdd(f32, bz.b3, rf32(kf_b + off), @mulAdd(f32, bz.b2, rf32(kf_b + 0x0C + off), @mulAdd(f32, bz.b1, rf32(kf_a + 0x18 + off), bz.b0 * rf32(kf_a + off))))); } - } else return; // mode 4+: no interp, leave output unchanged + } else {} // Unknown mode: skip primary, fall through to crossfade - // Crossfade blend const blend = rf32(bone_rt_base + BR.blend_weight); if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x18); - const st = ufloat(ru32(output + 0x20)); const skf_a = kf_base + ru32(output + 0x18) * 36; const skf_b = kf_base + ru32(output + 0x1C) * 36; const smode = ri16(anim_data + AD.interp_mode); - if (smode == 1) { const sec = lerpVec3(skf_a, skf_b, st); wu32(output + 0x24, fbits(sec[0])); @@ -826,44 +641,37 @@ fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) var i: u32 = 0; while (i < 3) : (i += 1) { const off = i * 4; - wf32(output + 0x24 + off, h.h1 * rf32(skf_a + off) + h.h2 * rf32(skf_a + 0x18 + off) + h.h3 * rf32(skf_b + off) + h.h4 * rf32(skf_b + 0x0C + off)); + wf32(output + 0x24 + off, @mulAdd(f32, h.h4, rf32(skf_b + 0x0C + off), @mulAdd(f32, h.h3, rf32(skf_b + off), @mulAdd(f32, h.h2, rf32(skf_a + 0x18 + off), h.h1 * rf32(skf_a + off))))); } } else if (smode == 2) { - const b = bezierBasis(st); + const bz = bezierBasis(st); var i: u32 = 0; while (i < 3) : (i += 1) { const off = i * 4; - wf32(output + 0x24 + off, b.b0 * rf32(skf_a + off) + b.b1 * rf32(skf_a + 0x18 + off) + b.b2 * rf32(skf_b + 0x0C + off) + b.b3 * rf32(skf_b + off)); + wf32(output + 0x24 + off, @mulAdd(f32, bz.b3, rf32(skf_b + off), @mulAdd(f32, bz.b2, rf32(skf_b + 0x0C + off), @mulAdd(f32, bz.b1, rf32(skf_a + 0x18 + off), bz.b0 * rf32(skf_a + off))))); } } else { - // Step for crossfade wu32(output + 0x24, ru32(skf_a)); wu32(output + 0x28, ru32(skf_a + 4)); wu32(output + 0x2C, ru32(skf_a + 8)); } - - // Blend: primary = primary + (secondary - primary) * blend var i: u32 = 0; while (i < 3) : (i += 1) { const off = i * 4; const pri = rf32(output + 0x0C + off); const sec = rf32(output + 0x24 + off); - wf32(output + 0x0C + off, (sec - pri) * blend + pri); + wf32(output + 0x0C + off, @mulAdd(f32, sec - pri, blend, pri)); } } } -/// Float interpolation with 12-byte keyframes and 4 modes (step/lerp/bezier/hermite). -/// Keyframe layout: [value (4), in_tangent (4), out_tangent (4)] = 12 bytes. -/// Used by 0x124 particle emitter Track 3. Uses bone_rt_base (bone 0) for timing. +/// Float track with 12-byte keyframes (value+tangents), modes 0-3 + crossfade. fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); - const mode = ri16(anim_data + AD.interp_mode); const kf_base = ru32(anim_data + AD.keyframe_base); if (mode == 0) { - // Step — copy float from keyframe at idx0*12 wu32(output + 0x0C, ru32(kf_base + ru32(output) * 12)); return; } @@ -873,152 +681,100 @@ fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) const kf_b = kf_base + ru32(output + 4) * 12; if (mode == 1) { - const a = rf32(kf_a); - const b = rf32(kf_b); - wf32(output + 0x0C, (b - a) * t + a); + wf32(output + 0x0C, @mulAdd(f32, rf32(kf_b) - rf32(kf_a), t, rf32(kf_a))); } else if (mode == 3) { - // Hermite: h1*p0 + h2*m0_out + h3*p1 + h4*m1_in const h = hermiteBasis(t); - wf32(output + 0x0C, h.h1 * rf32(kf_a) + h.h2 * rf32(kf_a + 0x08) + h.h3 * rf32(kf_b) + h.h4 * rf32(kf_b + 0x04)); + wf32(output + 0x0C, @mulAdd(f32, h.h4, rf32(kf_b + 0x04), @mulAdd(f32, h.h3, rf32(kf_b), @mulAdd(f32, h.h2, rf32(kf_a + 0x08), h.h1 * rf32(kf_a))))); } else if (mode == 2) { - // Bezier: b0*p0 + b1*m0_out + b2*m1_in + b3*p1 - const b = bezierBasis(t); - wf32(output + 0x0C, b.b0 * rf32(kf_a) + b.b1 * rf32(kf_a + 0x08) + b.b2 * rf32(kf_b + 0x04) + b.b3 * rf32(kf_b)); - } else return; + const bz = bezierBasis(t); + wf32(output + 0x0C, @mulAdd(f32, bz.b3, rf32(kf_b), @mulAdd(f32, bz.b2, rf32(kf_b + 0x04), @mulAdd(f32, bz.b1, rf32(kf_a + 0x08), bz.b0 * rf32(kf_a))))); + } else {} // Unknown mode: fall through to crossfade - // Crossfade const blend = rf32(bone_rt_base + BR.blend_weight); if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); - const st = ufloat(ru32(output + 0x18)); const skf_a = kf_base + ru32(output + 0x10) * 12; const skf_b = kf_base + ru32(output + 0x14) * 12; const smode = ri16(anim_data + AD.interp_mode); - var sec: f32 = undefined; if (smode == 1) { - sec = (rf32(skf_b) - rf32(skf_a)) * st + rf32(skf_a); + sec = @mulAdd(f32, rf32(skf_b) - rf32(skf_a), st, rf32(skf_a)); } else if (smode == 3) { const h = hermiteBasis(st); - sec = h.h1 * rf32(skf_a) + h.h2 * rf32(skf_a + 0x08) + h.h3 * rf32(skf_b) + h.h4 * rf32(skf_b + 0x04); + sec = @mulAdd(f32, h.h4, rf32(skf_b + 0x04), @mulAdd(f32, h.h3, rf32(skf_b), @mulAdd(f32, h.h2, rf32(skf_a + 0x08), h.h1 * rf32(skf_a)))); } else if (smode == 2) { const bz = bezierBasis(st); - sec = bz.b0 * rf32(skf_a) + bz.b1 * rf32(skf_a + 0x08) + bz.b2 * rf32(skf_b + 0x04) + bz.b3 * rf32(skf_b); + sec = @mulAdd(f32, bz.b3, rf32(skf_b), @mulAdd(f32, bz.b2, rf32(skf_b + 0x04), @mulAdd(f32, bz.b1, rf32(skf_a + 0x08), bz.b0 * rf32(skf_a)))); } else { sec = rf32(skf_a); } wf32(output + 0x1C, sec); const pri = rf32(output + 0x0C); - wf32(output + 0x0C, (sec - pri) * blend + pri); + wf32(output + 0x0C, @mulAdd(f32, sec - pri, blend, pri)); } } -// ============================================================================= -// getInterpolatedFloat — reimplemented from 0x71af20 -// Same as interpFloatTrack but uses the bone_rt directly (different register mapping) -// ============================================================================= +/// Float interpolation variant for getInterpolatedFloat (0x71AF20). +/// Identical to interpFloatTrack but reads blend from bone_rt+0x10C directly. +inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data: u32, output: u32) void { + findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data, output); + const mode = ri16(anim_data); + const kf_base = ru32(anim_data + 0x18); -inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void { - findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data_short_ptr, output); - - const interp_mode = ri16(anim_data_short_ptr); - const kf_base = ru32(anim_data_short_ptr + 0x18); - - if (interp_mode == 0) { + if (mode == 0) { wu32(output + 0x0C, ru32(kf_base + ru32(output) * 4)); return; } const t = ufloat(ru32(output + 8)); - const a = rf32(kf_base + ru32(output) * 4); - const b = rf32(kf_base + ru32(output + 4) * 4); - wf32(output + 0x0C, (b - a) * t + a); + const va = rf32(kf_base + ru32(output) * 4); + const vb = rf32(kf_base + ru32(output + 4) * 4); + wf32(output + 0x0C, @mulAdd(f32, vb - va, t, va)); const blend = rf32(bone_rt_addr + 0x10C); - if (blend != 0.0 and ri16(anim_data_short_ptr + 2) == -1) { - findInterpIdx(this, ru32(bone_rt_addr + 0xC4), ru32(bone_rt_addr + 0xC8), anim_data_short_ptr, output + 0x10); + if (blend != 0.0 and ri16(anim_data + 2) == -1) { + findInterpIdx(this, ru32(bone_rt_addr + 0xC4), ru32(bone_rt_addr + 0xC8), anim_data, output + 0x10); const st = ufloat(ru32(output + 0x18)); const sa = rf32(kf_base + ru32(output + 0x10) * 4); const sb = rf32(kf_base + ru32(output + 0x14) * 4); - const sec = (sb - sa) * st + sa; + const sec = @mulAdd(f32, sb - sa, st, sa); wu32(output + 0x1C, fbits(sec)); const pri = ufloat(ru32(output + 0x0C)); - wf32(output + 0x0C, (sec - pri) * blend + pri); + wf32(output + 0x0C, @mulAdd(f32, sec - pri, blend, pri)); } } -// ============================================================================= -// calculateScaledInverseMatrix — reimplemented from 0x7bd820 -// Used for billboarding. Transposes 3x3 rotation, scales by 1/scale^2, -// applies inverse translation. -// ============================================================================= +/// Byte keyframe extraction (replaces game's 0x71AE90). +/// findInterpIdx then reads a byte at keyframe_values[idx0]. +fn extractByte(this: u32, bone_rt: u32, anim_data: u32, output: u32) void { + findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), anim_data, output); + const kf_values = ru32(anim_data + AD.keyframe_base); + wu8(output + 0x0C, ru8(kf_values + ru32(output))); +} -fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void { - // Simple transpose for unit scale - if (@abs(scale - 1.0) < @as(f32, @bitCast(@as(u32, 0x35800000)))) { - // Transpose 3x3 - wf32(out + 0x00, rf32(this_mat + 0x00)); - wf32(out + 0x04, rf32(this_mat + 0x10)); - wf32(out + 0x08, rf32(this_mat + 0x20)); - wf32(out + 0x0C, 0); - wf32(out + 0x10, rf32(this_mat + 0x04)); - wf32(out + 0x14, rf32(this_mat + 0x14)); - wf32(out + 0x18, rf32(this_mat + 0x24)); - wf32(out + 0x1C, 0); - wf32(out + 0x20, rf32(this_mat + 0x08)); - wf32(out + 0x24, rf32(this_mat + 0x18)); - wf32(out + 0x28, rf32(this_mat + 0x28)); - wf32(out + 0x2C, 0); - wf32(out + 0x30, 0); - wf32(out + 0x34, 0); - wf32(out + 0x38, 0); - wf32(out + 0x3C, @as(f32, @bitCast(@as(u32, 0x3f800000)))); - // Apply inverse translation - applyTranslation(out, -rf32(this_mat + 0x30), -rf32(this_mat + 0x34), -rf32(this_mat + 0x38)); - return; +/// Short-value interpolation: reads i16 keyframes directly from memory. +fn shortInterpToFloat(anim_data: u32, output: u32) f32 { + const mode = ri16(anim_data); + const kf_base = ru32(anim_data + AD.keyframe_base); + const s2f = getShortToFloat(); + if (mode == 0) { + return @as(f32, @floatFromInt(@as(i32, ri16(kf_base + ru32(output) * 2)))) * s2f; + } else { + const t = ufloat(ru32(output + 8)); + const v0 = @as(f32, @floatFromInt(@as(i32, ri16(kf_base + ru32(output) * 2)))) * s2f; + const v1 = @as(f32, @floatFromInt(@as(i32, ri16(kf_base + ru32(output + 4) * 2)))) * s2f; + return @mulAdd(f32, v1 - v0, t, v0); } - - // Transpose 3x3 portion - wf32(out + 0x00, rf32(this_mat + 0x00)); - wf32(out + 0x04, rf32(this_mat + 0x10)); - wf32(out + 0x08, rf32(this_mat + 0x20)); - wf32(out + 0x0C, 0); - wf32(out + 0x10, rf32(this_mat + 0x04)); - wf32(out + 0x14, rf32(this_mat + 0x14)); - wf32(out + 0x18, rf32(this_mat + 0x24)); - wf32(out + 0x1C, 0); - wf32(out + 0x20, rf32(this_mat + 0x08)); - wf32(out + 0x24, rf32(this_mat + 0x18)); - wf32(out + 0x28, rf32(this_mat + 0x28)); - wf32(out + 0x2C, 0); - wf32(out + 0x30, 0); - wf32(out + 0x34, 0); - wf32(out + 0x38, 0); - wf32(out + 0x3C, @as(f32, @bitCast(@as(u32, 0x3f800000)))); - - // Scale by 1/(scale^2) - const inv_s2 = 1.0 / (scale * scale); - scaleMatrix3x3(out, inv_s2, inv_s2, inv_s2); - - // Apply inverse translation - applyTranslation(out, -rf32(this_mat + 0x30), -rf32(this_mat + 0x34), -rf32(this_mat + 0x38)); } // ============================================================================= // Main export: transformMatrix4x4_SSE -// -// Calling convention: C (all params on stack, since this is a separate -// compilation unit linked via addObject). The transform44.zig wrapper -// calls this with explicit params extracted from the fastcall detour. -// -// Params: this_ptr, mat1(parent_matrix*), mat2(position_vec3*), mat3(offset_vec3*), mat4(scale_float_bits) -// mat1 is the parent transform matrix — used for billboard matrix setup -// (initPPSG computes billboard_row0 = field_0xBC × mat1) // ============================================================================= -export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) void { +export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void { @setEvalBranchQuota(50000); + // ========================================================================= // Section 1: Entry checks // ========================================================================= @@ -1034,33 +790,23 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat const emitter_ctx = ru32(this + SO.emitter_ctx); if (emitter_ctx != 0) { - // Assembly 0x71429E-0x7142C1: emitter_ctx+0x50 != 0 AND this+0x1D8 != 0 const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0; - wu32(this + 0x50, has_emitter); // emitter_enable_flag + wu32(this + 0x50, has_emitter); wu32(this + 0x17C, ru32(emitter_ctx + 0x17C)); } // ========================================================================= // Section 3: World position/scale // ========================================================================= - const pos_ptr = mat2; // position input Vec3 - const ofs_ptr = mat3; // offset input Vec3 - const scale_f: f32 = @bitCast(mat4); // float scale + wf32(this + SO.world_pos + 0, rf32(mat2) * rf32(this + SO.field_184)); + wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(mat2 + 4)); + wf32(this + SO.world_pos + 8, rf32(this + SO.field_18c) * rf32(mat2 + 8)); - // world_pos = pos * per_axis_scale - wf32(this + SO.world_pos + 0, rf32(pos_ptr) * rf32(this + SO.field_184)); - wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(pos_ptr + 4)); - wf32(this + SO.world_pos + 8, @bitCast(fbits(rf32(this + SO.field_18c) * rf32(pos_ptr + 8)))); + wf32(this + SO.render_pri + 0, rf32(mat3) + rf32(this + SO.field_190)); + wf32(this + SO.render_pri + 4, rf32(this + SO.render_scale_x) + rf32(mat3 + 4)); + wf32(this + SO.render_pri + 8, rf32(this + SO.render_scale_y) + rf32(mat3 + 8)); - // render_pri = offset + existing fields - const rp0 = rf32(ofs_ptr) + rf32(this + SO.field_190); - const rp1 = rf32(this + SO.render_scale_x) + rf32(ofs_ptr + 4); - const rp2 = rf32(this + SO.render_scale_y) + rf32(ofs_ptr + 8); - wf32(this + SO.render_pri + 0, rp0); - wf32(this + SO.render_pri + 4, rp1); - wf32(this + SO.render_pri + 8, rp2); - - // render_scale_z = scale * field_180 + const scale_f: f32 = @bitCast(mat4); wf32(this + SO.render_scale_z, scale_f * rf32(this + SO.field_180)); // ========================================================================= @@ -1075,51 +821,30 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat var gi: u32 = 0; while (gi < gs_count) : (gi += 1) { const dur = ru32(gs_durations + gi * 4); - if (dur == 0) { - wu32(gs_values + gi * 4, 0); - } else { - wu32(gs_values + gi * 4, (timestamp -% time_base) % dur); - } + wu32(gs_values + gi * 4, if (dur == 0) 0 else (timestamp -% time_base) % dur); } } - // initPPSG: *(this+0xFC) = *(this+0xBC) × mat1 - // Assembly at 0x71438B: PUSH mat1, PUSH &0xBC, PUSH &0xFC, CALL 0x74A7C0 - // Reimplemented as inline SSE 4x4 matrix multiply. + // matMul: this+0xFC = this+0xBC × mat1 matMul4x4(this + 0xFC, this + 0xBC, mat1); // ========================================================================= - // Section 5: child_objects_padding (len_sq of world transform translation) + // Section 5: child_padding (sqmag of world transform translation row) // ========================================================================= - if (emitter_ctx == 0 or (ru8(emitter_ctx + 4) & 1) != 0) { - const wx = rf32(this + SO.world_xform + 8 * 4); // [8] - const wy = rf32(this + SO.world_xform + 9 * 4); // [9] - const wz = rf32(this + SO.world_xform + 10 * 4); // [10] - wu32(this + SO.child_padding, fbits(wx * wx + wy * wy + wz * wz)); + const emitter_ctx_5 = ru32(this + SO.emitter_ctx); + if (emitter_ctx_5 == 0 or (ru8(emitter_ctx_5 + 4) & 1) != 0) { + wu32(this + SO.child_padding, fbits(vec3SqMag(this + 0x12C))); } else { - wu32(this + SO.child_padding, ru32(emitter_ctx + 0x84)); + wu32(this + SO.child_padding, ru32(emitter_ctx_5 + 0x84)); } // ========================================================================= - // Section 6: Identity matrix init + timestamp delta + // Section 6: Identity matrices + timestamp delta // ========================================================================= - var local_mat: [16]f32 = .{ - 1, 0, 0, 0, - 0, 1, 0, 0, - 0, 0, 1, 0, - 0, 0, 0, 1, - }; + var local_mat: [16]f32 align(16) = .{ 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; const local_mat_addr = @intFromPtr(&local_mat); + var local_mat2: [16]f32 align(16) = .{ 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; - // Secondary identity (3x4 portion for the second matrix in decompilation) - var local_mat2: [16]f32 = .{ - 1, 0, 0, 0, - 0, 1, 0, 0, - 0, 0, 1, 0, - 0, 0, 0, 1, - }; - - // Timestamp delta tracking var time_delta_val: u32 = 0; const sdb = ru32(this + SO.search_data_base); if (sdb != 0) { @@ -1147,10 +872,8 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat const parent_idx_raw: i32 = @as(i32, @intCast(@as(i16, @bitCast(ru16(bdef + BD.parent_bone))))); // --- Animation time computation --- - // (Handle primary and secondary animation slot timing) const anim_slot_val = ri32(brt + BR.anim_slot); if (anim_slot_val == -1) { - // Inherit from parent bone if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118; wu32(brt + BR.prim_time, ru32(parent_rt + BR.prim_time)); @@ -1162,64 +885,45 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim)); } } else { - // Has own animation slot — compute time from animation lookup table. - // Assembly at 0x714561-0x71464E, verified line by line. - if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0 - // Add time delta to sec_start/sec_end - wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val); // [ESI+0xA8] - wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val); // [ESI+0xAC] + if (ru32(this + 0x4C) != 0) { + wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val); + wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val); } - - // anim_entry = anim_lookup_table + anim_slot * 0x44 - const anim_lookup = ru32(model_hdr + 0x20); // [EDX+0x20] + const anim_lookup = ru32(model_hdr + 0x20); const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44; - const cur_time = ru32(ru32(this + 0x2C) + 0xC); // [EBX+0x2C]+0xC = timestamp + const cur_time = ru32(ru32(this + 0x2C) + 0xC); - // Check looping flag: [anim_entry+0x10] & 1 if ((ru8(anim_entry + 0x10) & 1) == 0) { - // Looping: assembly at 0x7145F1-0x714631 const anim_end = ru32(anim_entry + 0x08); const anim_start = ru32(anim_entry + 0x04); if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { - // elapsed = (float)(cur_time - sec_start) * time_scale → __ftol const delta = cur_time -% ru32(brt + 0xA8); - const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xB0))); + const ftol_result = ftol(@as(i32, @bitCast(delta)), brt + 0xB0); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start); - wu32(brt + 0x98, anim_start +% frame); // prim_time + wu32(brt + 0x98, anim_start +% frame); + } else { + wu32(brt + 0x98, anim_start); } } else { - // Clamped: assembly at 0x71458E-0x7145E3 const sec_end_val = ru32(brt + 0xAC); const sec_start_val = ru32(brt + 0xA8); - - // Check if sec_end has passed (sec_end - cur_time <= 0 signed) if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) { - // sec_end hasn't passed yet - if (sec_start_val != cur_time and @as(i32, @bitCast(sec_start_val -% cur_time)) > 0) { - // Before start: use sec_start as time - // Actually assembly jumps to looping path LAB_007145f1 - // which reads anim_entry+0x08, anim_entry+0x04 - // Fallthrough: use cur_time (no write to prim_time) - } - // goto looping path + const effective_time = if (@as(i32, @bitCast(sec_start_val -% cur_time)) > 0) sec_start_val else cur_time; const anim_end = ru32(anim_entry + 0x08); const anim_start = ru32(anim_entry + 0x04); if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { - const delta = cur_time -% ru32(brt + 0xA8); - const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xB0))); + const delta = effective_time -% ru32(brt + 0xA8); + const ftol_result = ftol(@as(i32, @bitCast(delta)), brt + 0xB0); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start); wu32(brt + 0x98, anim_start +% frame); + } else { + wu32(brt + 0x98, anim_start); } } else { - // sec_end has passed — compute clamped position - // Assembly at 0x71458E-0x7145E3: - // delta = (sec_end - sec_start), scaled by [ESI+0xB0] const dur = sec_end_val -% sec_start_val; - const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(dur)))) * rf32(brt + 0xB0))); + const ftol_result = ftol(@as(i32, @bitCast(dur)), brt + 0xB0); const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8))); - if (offset < 0) { - // Clamp to anim_start wu32(brt + 0x98, ru32(anim_entry + 0x04)); } else { const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08))); @@ -1227,21 +931,16 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat if (offset <= anim_end_i - anim_start_i) { wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i))); } else { - // Clamp to anim_end wu32(brt + 0x98, ru32(anim_entry + 0x08)); } } } } - - // Store results: assembly at 0x714633-0x71464E - wu32(brt + 0x9C, ru32(brt + 0xA4)); // prim_track = anim_slot - // prim_time already set above - wu32(brt + 0xA0, bone_idx); // prim_anim = bone_idx + wu32(brt + 0x9C, ru32(brt + 0xA4)); + wu32(brt + 0xA0, bone_idx); } - // --- Secondary animation time (crossfade target) --- - // Similar pattern for the secondary/blend animation slot + // --- Secondary animation time --- const sec_slot_val = ri32(brt + BR.sec_slot); if (sec_slot_val == -1) { if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { @@ -1256,49 +955,44 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat wu32(brt + BR.sec_track, ru32(brt + BR.prim_track)); } } else { - // Secondary animation slot time computation. - // Assembly at 0x7146C1-0x7147C3, mirrors primary slot logic. - if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0 - wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val); // [ESI+0xD4] - wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val); // [ESI+0xD8] + if (ru32(this + 0x4C) != 0) { + wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val); + wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val); } - const sec_anim_lookup = ru32(model_hdr + 0x20); const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44; const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC); if ((ru8(sec_anim_entry + 0x10) & 1) == 0) { - // Looping const anim_end = ru32(sec_anim_entry + 0x08); const anim_start = ru32(sec_anim_entry + 0x04); if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { const delta = sec_cur_time -% ru32(brt + 0xD4); - const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xDC))); + const ftol_result = ftol(@as(i32, @bitCast(delta)), brt + 0xDC); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start); - wu32(brt + 0xC4, anim_start +% frame); // sec_time + wu32(brt + 0xC4, anim_start +% frame); + } else { + wu32(brt + 0xC4, anim_start); } } else { - // Clamped const sec_end_val = ru32(brt + 0xD8); const sec_start_val = ru32(brt + 0xD4); - if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) { - if (sec_start_val != sec_cur_time and @as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) { - // use sec_start - } + const effective_time = if (@as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) sec_start_val else sec_cur_time; const anim_end = ru32(sec_anim_entry + 0x08); const anim_start = ru32(sec_anim_entry + 0x04); if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { - const delta = sec_cur_time -% ru32(brt + 0xD4); - const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xDC))); + const delta = effective_time -% ru32(brt + 0xD4); + const ftol_result = ftol(@as(i32, @bitCast(delta)), brt + 0xDC); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start); wu32(brt + 0xC4, anim_start +% frame); + } else { + wu32(brt + 0xC4, anim_start); } } else { const dur = sec_end_val -% sec_start_val; - const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(dur)))) * rf32(brt + 0xDC))); + const ftol_result = ftol(@as(i32, @bitCast(dur)), brt + 0xDC); const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4))); - if (offset < 0) { wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04)); } else { @@ -1312,24 +1006,18 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat } } } - - // Store results: assembly at 0x714799-0x7147C3 - wu32(brt + 0xC8, ru32(brt + 0xD0)); // sec_track = sec_slot - // sec_time already set above - - // Check expiry: if (timestamp - crossfade_end >= 0) expire slot + wu32(brt + 0xC8, ru32(brt + 0xD0)); if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) { - wu32(brt + 0xD0, 0xFFFFFFFF); // expire secondary slot + wu32(brt + 0xD0, 0xFFFFFFFF); } } - // --- Blend weight (crossfade Hermite interpolation) --- + // --- Blend weight --- if (ri32(brt + BR.anim_slot) == -1 and ri32(brt + BR.sec_slot) == -1) { - // Inherit blend weight from parent if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { wu32(brt + BR.blend_weight, ru32(bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118 + BR.blend_weight)); } else if (bone_idx == 0) { - wu32(brt + BR.blend_weight, 0); // root bone, no blend + wu32(brt + BR.blend_weight, 0); } else { wu32(brt + BR.blend_weight, ru32(bone_rt_base + BR.blend_weight)); } @@ -1342,13 +1030,12 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat } else { const t_raw = @as(f32, @floatFromInt(cf_remaining)) * ufloat(ru32(brt + BR.crossfade_inv)); const t_clamped = if (t_raw < 0.0) @as(f32, 0.0) else if (t_raw > 1.0) @as(f32, 1.0) else t_raw; - // Hermite: (3 - 2t) * t^2 * weight - const h = (3.0 - 2.0 * t_clamped) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight)); + const h = @mulAdd(f32, -2.0, t_clamped, 3.0) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight)); wu32(brt + BR.blend_weight, fbits(h)); } } - // --- Parent bone transform inheritance --- + // --- Parent bone transform --- const combined_flags: u32 = ru32(brt + BR.flags2) | flags; var src_mat: u32 = undefined; @@ -1358,329 +1045,176 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40; src_mat = parent_out; - // Billboard pre-processing (flags & 7) if ((combined_flags & 7) != 0) { - // Copy parent matrix to local_mat and work from there for (0..16) |i| { local_mat[i] = rf32(parent_out + @as(u32, @intCast(i)) * 4); } - - // Apply pivot translation const pivot_x = rf32(bdef + BD.pivot_x); const pivot_y = rf32(bdef + BD.pivot_y); const pivot_z = rf32(bdef + BD.pivot_z); - - // Compute translated position - const tx = local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z + local_mat[12]; - const ty = local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z + local_mat[13]; - const tz = local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z + local_mat[14]; + const tx = @mulAdd(f32, local_mat[8], pivot_z, @mulAdd(f32, local_mat[4], pivot_y, @mulAdd(f32, local_mat[0], pivot_x, local_mat[12]))); + const ty = @mulAdd(f32, local_mat[9], pivot_z, @mulAdd(f32, local_mat[5], pivot_y, @mulAdd(f32, local_mat[1], pivot_x, local_mat[13]))); + const tz = @mulAdd(f32, local_mat[10], pivot_z, @mulAdd(f32, local_mat[6], pivot_y, @mulAdd(f32, local_mat[2], pivot_x, local_mat[14]))); const bb_type = combined_flags & 6; if (bb_type == 2) { - // Cylindrical billboard — normalize each column - const n0 = normalizeVec3(local_mat[0], local_mat[1], local_mat[2]); - local_mat[0] = n0[0]; - local_mat[1] = n0[1]; - local_mat[2] = n0[2]; - const n1 = normalizeVec3(local_mat[4], local_mat[5], local_mat[6]); - local_mat[4] = n1[0]; - local_mat[5] = n1[1]; - local_mat[6] = n1[2]; - const n2 = normalizeVec3(local_mat[8], local_mat[9], local_mat[10]); - local_mat[8] = n2[0]; - local_mat[9] = n2[1]; - local_mat[10] = n2[2]; + normalizeVec3InPlace(local_mat_addr); + normalizeVec3InPlace(local_mat_addr + 0x10); + normalizeVec3InPlace(local_mat_addr + 0x20); } else if (bb_type == 4) { - // Spherical billboard — inherit camera rotation with scale preservation - const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) }; - const cam_len_sq0 = cam0[0] * cam0[0] + cam0[1] * cam0[1] + cam0[2] * cam0[2]; + // Spherical billboard + const cam_sq = vec3SqMag(this + SO.bb_row0); var s0: f32 = 1.0; - if (cam_len_sq0 > @as(f32, @bitCast(@as(u32, 0x3727c5ac)))) { - const mat_len_sq0 = local_mat[0] * local_mat[0] + local_mat[1] * local_mat[1] + local_mat[2] * local_mat[2]; - s0 = @sqrt(mat_len_sq0 / cam_len_sq0); + if (cam_sq > rf32(0x0080c5c8)) { + s0 = @sqrt(vec3SqMagF(local_mat[0], local_mat[1], local_mat[2]) / cam_sq); } - local_mat[0] = s0 * cam0[0]; - local_mat[1] = s0 * cam0[1]; - local_mat[2] = s0 * cam0[2]; + local_mat[0] = s0 * rf32(this + SO.bb_row0); + local_mat[1] = s0 * rf32(this + SO.bb_row0 + 4); + local_mat[2] = s0 * rf32(this + SO.bb_row0 + 8); - const wt0 = rf32(this + SO.world_xform + 0 * 4); - const wt1 = rf32(this + SO.world_xform + 1 * 4); - const wt2 = rf32(this + SO.world_xform + 2 * 4); - const wt_len_sq = wt0 * wt0 + wt1 * wt1 + wt2 * wt2; + const wt_sq = vec3SqMag(this + SO.world_xform); var s1: f32 = 1.0; - if (wt_len_sq > @as(f32, @bitCast(@as(u32, 0x3727c5ac)))) { - const mat_len_sq1 = local_mat[4] * local_mat[4] + local_mat[5] * local_mat[5] + local_mat[6] * local_mat[6]; - s1 = @sqrt(mat_len_sq1 / wt_len_sq); + if (wt_sq > rf32(0x0080c5c8)) { + s1 = @sqrt(vec3SqMagF(local_mat[4], local_mat[5], local_mat[6]) / wt_sq); } - local_mat[4] = s1 * wt0; - local_mat[5] = s1 * wt1; - local_mat[6] = s1 * wt2; + local_mat[4] = s1 * rf32(this + SO.world_xform); + local_mat[5] = s1 * rf32(this + SO.world_xform + 4); + local_mat[6] = s1 * rf32(this + SO.world_xform + 8); - const wt4 = rf32(this + SO.world_xform + 4 * 4); - const wt5 = rf32(this + SO.world_xform + 5 * 4); - const wt6 = rf32(this + SO.world_xform + 6 * 4); - const wt_len_sq2 = wt4 * wt4 + wt5 * wt5 + wt6 * wt6; + const wt_sq2 = vec3SqMag(this + SO.world_xform + 16); var s2: f32 = 1.0; - if (wt_len_sq2 > @as(f32, @bitCast(@as(u32, 0x3727c5ac)))) { - const mat_len_sq2 = local_mat[8] * local_mat[8] + local_mat[9] * local_mat[9] + local_mat[10] * local_mat[10]; - s2 = @sqrt(mat_len_sq2 / wt_len_sq2); + if (wt_sq2 > rf32(0x0080c5c8)) { + s2 = @sqrt(vec3SqMagF(local_mat[8], local_mat[9], local_mat[10]) / wt_sq2); } - local_mat[8] = s2 * wt4; - local_mat[9] = s2 * wt5; - local_mat[10] = s2 * wt6; + local_mat[8] = s2 * rf32(this + SO.world_xform + 16); + local_mat[9] = s2 * rf32(this + SO.world_xform + 20); + local_mat[10] = s2 * rf32(this + SO.world_xform + 24); } else if (bb_type == 6) { - // Full billboard — copy camera rotation directly local_mat[0] = rf32(this + SO.bb_row0); local_mat[1] = rf32(this + SO.bb_row0 + 4); local_mat[2] = rf32(this + SO.bb_row0 + 8); - local_mat[4] = rf32(this + SO.world_xform + 0 * 4); - local_mat[5] = rf32(this + SO.world_xform + 1 * 4); - local_mat[6] = rf32(this + SO.world_xform + 2 * 4); - local_mat[8] = rf32(this + SO.world_xform + 4 * 4); - local_mat[9] = rf32(this + SO.world_xform + 5 * 4); - local_mat[10] = rf32(this + SO.world_xform + 6 * 4); + local_mat[4] = rf32(this + SO.world_xform); + local_mat[5] = rf32(this + SO.world_xform + 4); + local_mat[6] = rf32(this + SO.world_xform + 8); + local_mat[8] = rf32(this + SO.world_xform + 16); + local_mat[9] = rf32(this + SO.world_xform + 20); + local_mat[10] = rf32(this + SO.world_xform + 24); } - // Recompute translation: pos - rot * pivot if ((combined_flags & 1) == 0) { - local_mat[12] = tx - (local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z); - local_mat[13] = ty - (local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z); - local_mat[14] = tz - (local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z); + local_mat[12] = tx - @mulAdd(f32, local_mat[8], pivot_z, @mulAdd(f32, local_mat[4], pivot_y, local_mat[0] * pivot_x)); + local_mat[13] = ty - @mulAdd(f32, local_mat[9], pivot_z, @mulAdd(f32, local_mat[5], pivot_y, local_mat[1] * pivot_x)); + local_mat[14] = tz - @mulAdd(f32, local_mat[10], pivot_z, @mulAdd(f32, local_mat[6], pivot_y, local_mat[2] * pivot_x)); } else { - local_mat[12] = rf32(this + SO.world_xform + 8 * 4); - local_mat[13] = rf32(this + SO.world_xform + 9 * 4); - local_mat[14] = rf32(this + SO.world_xform + 10 * 4); + local_mat[12] = rf32(this + SO.world_xform + 32); + local_mat[13] = rf32(this + SO.world_xform + 36); + local_mat[14] = rf32(this + SO.world_xform + 40); } - src_mat = local_mat_addr; } } - // --- Rotation interpolation --- + // --- Rotation/Scale/Translation interpolation --- if ((combined_flags & 0x280) == 0) { - // No rotation animation — just copy parent - const dst = bone_out_base + bone_idx * 0x40; - copyMat4(dst, src_mat); + copyMat4(bone_out_base + bone_idx * 0x40, src_mat); } else { - // Reset to identity for composition local_mat2 = .{ 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; const lm2_addr = @intFromPtr(&local_mat2); - const rot_anim = bdef + BD.rot_anim; - const rot_kf_count = ru32(bdef + BD.rot_nts); - - // Step 1: Rotation — build rotation matrix from quaternion FIRST. - // The original at 0x74B6BB overwrites the bone-local matrix with the - // quaternion rotation matrix (it does NOT multiply — just writes directly). - // This runs BEFORE scale and translation so the translation offset - // (pivot - matrix * pivot) uses the correctly rotated matrix. - if (rot_kf_count != 0) { - if (ru32(this + SO.anim_frame_ctr) < rot_kf_count) { - interpAnimKF(this, brt, rot_anim, brt + BR.rot_idx0); + // Rotation + if (ru32(bdef + BD.rot_nts) != 0) { + if (ru32(this + SO.anim_frame_ctr) < ru32(bdef + BD.rot_nts)) { + interpAnimKF(this, brt, bdef + BD.rot_anim, brt + BR.rot_idx0); } - // Build rotation matrix from quaternion — overwrites local_mat2 - // exactly like the original at 0x74B6BB (no multiply, just write) - buildRotationMatrix(lm2_addr, ufloat(ru32(brt + BR.rot_x)), ufloat(ru32(brt + BR.rot_y)), ufloat(ru32(brt + BR.rot_z)), ufloat(ru32(brt + BR.rot_w))); + buildRotationMatrixFromPtr(lm2_addr, brt + BR.rot_x); } - // Step 2: Scale interpolation — applied after rotation - const scale_anim = bdef + BD.scale_anim; - const scale_kf_count = ru32(bdef + BD.scale_nts); - if (scale_kf_count != 0) { - if (ru32(this + SO.anim_frame_ctr) < scale_kf_count) { - interpVec3Track(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight))); + // Scale + if (ru32(bdef + BD.scale_nts) != 0) { + if (ru32(this + SO.anim_frame_ctr) < ru32(bdef + BD.scale_nts)) { + interpVec3Track(this, brt, bdef + BD.scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight))); } - scaleMatrix3x3(lm2_addr, ufloat(ru32(brt + BR.scale_x)), ufloat(ru32(brt + BR.scale_y)), ufloat(ru32(brt + BR.scale_z))); + scaleMatrix3x3FromPtr(lm2_addr, brt + BR.scale_x); } - // Conditional multiply: if flag bit 0x80 set AND bone_rt[0xF0] != 0, - // multiply bone_local by the matrix pointed to by bone_rt[0xF0]. - // Assembly at 0x714F7F-0x714F9C: - // TEST CL, CL / JNS skip - // MOV EAX, [ESI+0xF0] / TEST EAX, EAX / JZ skip - // PUSH EAX (right), PUSH &bone_local (left), PUSH &bone_local (output) - // CALL 0x74A7C0 (multiplyMatrix4x4: output = left * right) - // This is bone_local *= *(bone_rt+0xF0) + // Extra matrix multiply if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) { - const extra_mat = ru32(brt + BR.bone_flag_cache); - // In-place multiply: bone_local = bone_local * extra_mat - // Original copies to temp when output==left, then multiplies. - var tmp: [16]f32 = undefined; - for (0..16) |fi| { - tmp[fi] = local_mat2[fi]; - } - matMul4x4(lm2_addr, @intFromPtr(&tmp), extra_mat); + matMul4x4(lm2_addr, lm2_addr, ru32(brt + BR.bone_flag_cache)); } - // Step 3: Translation interpolation + // Translation var tx_val = rf32(bdef + BD.pivot_x); var ty_val = rf32(bdef + BD.pivot_y); var tz_val = rf32(bdef + BD.pivot_z); - - const trans_anim = bdef + BD.trans_anim; - const trans_kf_count = ru32(bdef + BD.trans_nts); - if (trans_kf_count != 0) { - if (ru32(this + SO.anim_frame_ctr) < trans_kf_count) { - interpVec3Track(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight))); + if (ru32(bdef + BD.trans_nts) != 0) { + if (ru32(this + SO.anim_frame_ctr) < ru32(bdef + BD.trans_nts)) { + interpVec3Track(this, brt, bdef + BD.trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight))); } tx_val += ufloat(ru32(brt + BR.trans_x)); ty_val += ufloat(ru32(brt + BR.trans_y)); tz_val += ufloat(ru32(brt + BR.trans_z)); } - // Step 4: Compute translation offset using the ROTATED+SCALED matrix. - // translation = (pivot + interp_trans) - bone_local_matrix * pivot const piv_x = rf32(bdef + BD.pivot_x); const piv_y = rf32(bdef + BD.pivot_y); const piv_z = rf32(bdef + BD.pivot_z); - local_mat2[12] = tx_val - (local_mat2[0] * piv_x + local_mat2[4] * piv_y + local_mat2[8] * piv_z); - local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z); - local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z); + local_mat2[12] = tx_val - @mulAdd(f32, local_mat2[8], piv_z, @mulAdd(f32, local_mat2[4], piv_y, local_mat2[0] * piv_x)); + local_mat2[13] = ty_val - @mulAdd(f32, local_mat2[9], piv_z, @mulAdd(f32, local_mat2[5], piv_y, local_mat2[1] * piv_x)); + local_mat2[14] = tz_val - @mulAdd(f32, local_mat2[10], piv_z, @mulAdd(f32, local_mat2[6], piv_y, local_mat2[2] * piv_x)); - // Write final composed matrix to output - const dst = bone_out_base + bone_idx * 0x40; - // Multiply: dst = local_mat2 * src_mat (parent) - const r0: V4 = .{ rf32(src_mat), rf32(src_mat + 4), rf32(src_mat + 8), rf32(src_mat + 12) }; - const r1: V4 = .{ rf32(src_mat + 16), rf32(src_mat + 20), rf32(src_mat + 24), rf32(src_mat + 28) }; - const r2: V4 = .{ rf32(src_mat + 32), rf32(src_mat + 36), rf32(src_mat + 40), rf32(src_mat + 44) }; - const r3: V4 = .{ rf32(src_mat + 48), rf32(src_mat + 52), rf32(src_mat + 56), rf32(src_mat + 60) }; - - inline for (0..4) |row| { - const b = row * 4; - const out = splat(local_mat2[b]) * r0 + splat(local_mat2[b + 1]) * r1 + splat(local_mat2[b + 2]) * r2 + splat(local_mat2[b + 3]) * r3; - wf32(dst + @as(u32, @intCast(b)) * 4, out[0]); - wf32(dst + @as(u32, @intCast(b)) * 4 + 4, out[1]); - wf32(dst + @as(u32, @intCast(b)) * 4 + 8, out[2]); - wf32(dst + @as(u32, @intCast(b)) * 4 + 12, out[3]); - } + matMul4x4(bone_out_base + bone_idx * 0x40, lm2_addr, src_mat); } - // --- Billboard post-processing (flags & 0x78) --- - // Assembly at 0x7151F9-0x71594E. Runs for BOTH animated and non-animated paths. - // Modifies the already-written bone output matrix in-place. + // --- Billboard post-processing --- if ((combined_flags & 0x78) != 0) { - // pMVar19 = bone_idx * 0x40 (byte offset for output) - // pfVar12 = bone_out_base + pMVar19 (output matrix ptr) - const out_off = bone_idx * 0x40; - const om = bone_out_base + out_off; // output matrix + const om = bone_out_base + bone_idx * 0x40; + const scale_len0 = @sqrt(vec3SqMag(om)); + const scale_len1 = @sqrt(vec3SqMag(om + 0x10)); + const scale_len2 = @sqrt(vec3SqMag(om + 0x20)); - // Compute scale lengths (sqrt of row length_sq for each row) - const scale_len0 = @sqrt(rf32(om + 0x08) * rf32(om + 0x08) + rf32(om + 0x04) * rf32(om + 0x04) + rf32(om) * rf32(om)); - const scale_len1 = @sqrt(rf32(om + 0x18) * rf32(om + 0x18) + rf32(om + 0x14) * rf32(om + 0x14) + rf32(om + 0x10) * rf32(om + 0x10)); - const scale_len2 = @sqrt(rf32(om + 0x28) * rf32(om + 0x28) + rf32(om + 0x24) * rf32(om + 0x24) + rf32(om + 0x20) * rf32(om + 0x20)); - - // Compute translated pivot position through the output matrix - // local_a8 = pivot * matrix + translation const bpx = rf32(bdef + BD.pivot_x); const bpy = rf32(bdef + BD.pivot_y); const bpz = rf32(bdef + BD.pivot_z); - const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30); - const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34); - const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38); + const pos_x = @mulAdd(f32, bpz, rf32(om + 0x20), @mulAdd(f32, bpy, rf32(om + 0x10), @mulAdd(f32, bpx, rf32(om), rf32(om + 0x30)))); + const pos_y = @mulAdd(f32, bpz, rf32(om + 0x24), @mulAdd(f32, bpy, rf32(om + 0x14), @mulAdd(f32, bpx, rf32(om + 0x04), rf32(om + 0x34)))); + const pos_z = @mulAdd(f32, bpz, rf32(om + 0x28), @mulAdd(f32, bpy, rf32(om + 0x18), @mulAdd(f32, bpx, rf32(om + 0x08), rf32(om + 0x38)))); - // Switch on billboard post-processing type const bb_post = combined_flags & 0x78; switch (bb_post) { 0x08 => { - // Type 8: decompilation lines 657-718 - // If no pre-billboard (local_1c == 0 i.e. flags & 0x280 was 0): - // set fixed rotation columns - // Else: use rotation matrix rows with negated first component, normalize - const had_anim = (combined_flags & 0x280) != 0; - if (!had_anim) { - // Fixed columns: row0={0,0,-1}, row1={1,0,0}, row2={0,1,0} - wf32(om, 0); - wf32(om + 0x04, 0); - wf32(om + 0x08, -1); - wf32(om + 0x10, 1); - wf32(om + 0x14, 0); - wf32(om + 0x18, 0); - wf32(om + 0x20, 0); - wf32(om + 0x24, 1); - wf32(om + 0x28, 0); + if ((combined_flags & 0x280) == 0) { + wf32(om, 0); wf32(om + 0x04, 0); wf32(om + 0x08, -1); + wf32(om + 0x10, 1); wf32(om + 0x14, 0); wf32(om + 0x18, 0); + wf32(om + 0x20, 0); wf32(om + 0x24, 1); wf32(om + 0x28, 0); } else { - // Row 0 = {local_e4, local_e0, -local_e8}, normalize - const r0x = local_mat2[1]; // local_e4 - const r0y = local_mat2[2]; // local_e0 - const r0z = -local_mat2[0]; // -local_e8 - wf32(om, r0x); - wf32(om + 0x04, r0y); - wf32(om + 0x08, r0z); - const n0 = normalizeVec3InPlace(om); - _ = n0; - // Row 1 = {local_d4, local_d0, -local_d8}, normalize - const r1x = local_mat2[5]; // local_d4 - const r1y = local_mat2[6]; // local_d0 - const r1z = -local_mat2[4]; // -local_d8 - wf32(om + 0x10, r1x); - wf32(om + 0x14, r1y); - wf32(om + 0x18, r1z); - const n1 = normalizeVec3InPlace(om + 0x10); - _ = n1; - // Row 2 = {local_c4, local_c0, -local_c8}, normalize - const r2x = local_mat2[9]; // local_c4 - const r2y = local_mat2[10]; // local_c0 - const r2z = -local_mat2[8]; // -local_c8 - wf32(om + 0x20, r2x); - wf32(om + 0x24, r2y); - wf32(om + 0x28, r2z); - const n2 = normalizeVec3InPlace(om + 0x20); - _ = n2; + wf32(om, local_mat2[1]); wf32(om + 0x04, local_mat2[2]); wf32(om + 0x08, -local_mat2[0]); + normalizeVec3InPlace(om); + wf32(om + 0x10, local_mat2[5]); wf32(om + 0x14, local_mat2[6]); wf32(om + 0x18, -local_mat2[4]); + normalizeVec3InPlace(om + 0x10); + wf32(om + 0x20, local_mat2[9]); wf32(om + 0x24, local_mat2[10]); wf32(om + 0x28, -local_mat2[8]); + normalizeVec3InPlace(om + 0x20); } }, 0x10 => { - // Type 16: normalize row0, set row1={row0.y, -row0.x, 0}, normalize, - // row2 = cross(row0, row1) - const n0 = normalizeVec3InPlace(om); - _ = n0; - const r0x = rf32(om); - const r0y = rf32(om + 0x04); - wf32(om + 0x10, r0y); - wf32(om + 0x14, -r0x); - wf32(om + 0x18, 0); - const n1 = normalizeVec3InPlace(om + 0x10); - _ = n1; - // row2 = cross(row0, row1) - wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14)); - wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18)); + normalizeVec3InPlace(om); + wf32(om + 0x10, rf32(om + 0x04)); wf32(om + 0x14, -rf32(om)); wf32(om + 0x18, 0); + normalizeVec3InPlace(om + 0x10); + wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18)); + wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10)); wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14)); }, 0x20 => { - // Type 32: normalize row1, set row0={-row1.y, row1.x, 0}, normalize, - // row2 = cross(row0, row1) - const n1 = normalizeVec3InPlace(om + 0x10); - _ = n1; - wf32(om, -rf32(om + 0x14)); - wf32(om + 0x04, rf32(om + 0x10)); - wf32(om + 0x08, 0); - const n0 = normalizeVec3InPlace(om); - _ = n0; - // row2 = cross(row0, row1) - wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14)); - wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18)); + normalizeVec3InPlace(om + 0x10); + wf32(om, -rf32(om + 0x14)); wf32(om + 0x04, rf32(om + 0x10)); wf32(om + 0x08, 0); + normalizeVec3InPlace(om); + wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18)); + wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10)); wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14)); }, 0x40 => { - // Type 64: normalize row2, set row1={row2.y, -row2.x, 0}, normalize, - // row0 = cross(row1, row2) - const r2_len = @sqrt(rf32(om + 0x20) * rf32(om + 0x20) + rf32(om + 0x24) * rf32(om + 0x24) + rf32(om + 0x28) * rf32(om + 0x28)); - if (@abs(r2_len) >= BILLBOARD_EPSILON) { - const inv = 1.0 / r2_len; - wf32(om + 0x20, rf32(om + 0x20) * inv); - wf32(om + 0x24, rf32(om + 0x24) * inv); - wf32(om + 0x28, rf32(om + 0x28) * inv); - } - wf32(om + 0x10, rf32(om + 0x24)); - wf32(om + 0x14, -rf32(om + 0x20)); - wf32(om + 0x18, 0); - const r1_len = @sqrt(rf32(om + 0x10) * rf32(om + 0x10) + rf32(om + 0x14) * rf32(om + 0x14) + rf32(om + 0x18) * rf32(om + 0x18)); - if (@abs(r1_len) >= BILLBOARD_EPSILON) { - const inv = 1.0 / r1_len; - wf32(om + 0x10, rf32(om + 0x10) * inv); - wf32(om + 0x14, rf32(om + 0x14) * inv); - wf32(om + 0x18, rf32(om + 0x18) * inv); - } - // row0 = cross(row2.y*row1.z - row2.z*row1.y, ...) + normalizeVec3InPlace(om + 0x20); + wf32(om + 0x10, rf32(om + 0x24)); wf32(om + 0x14, -rf32(om + 0x20)); wf32(om + 0x18, 0); + normalizeVec3InPlace(om + 0x10); wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14)); wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18)); wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10)); @@ -1688,154 +1222,132 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat else => {}, } - // Apply scale lengths back and recompute translation - // Assembly at 0x715868-0x71594B - wf32(om + 0x0C, 0); - wf32(om + 0x1C, 0); - wf32(om + 0x2C, 0); - // Scale each row by its original length - const r0x_s = rf32(om); - wf32(om, scale_len0 * r0x_s); - const r0y_s = rf32(om + 0x04); - wf32(om + 0x04, scale_len0 * r0y_s); - const r0z_s = rf32(om + 0x08); - wf32(om + 0x08, scale_len0 * r0z_s); - const r1x_s = rf32(om + 0x10); - wf32(om + 0x10, scale_len1 * r1x_s); - const r1y_s = rf32(om + 0x14); - wf32(om + 0x14, scale_len1 * r1y_s); - const r1z_s = rf32(om + 0x18); - wf32(om + 0x18, scale_len1 * r1z_s); - const r2x_s = rf32(om + 0x20); - wf32(om + 0x20, scale_len2 * r2x_s); - const r2y_s = rf32(om + 0x24); - wf32(om + 0x24, scale_len2 * r2y_s); - const r2z_s = rf32(om + 0x28); - wf32(om + 0x28, scale_len2 * r2z_s); + wf32(om + 0x0C, 0); wf32(om + 0x1C, 0); wf32(om + 0x2C, 0); + const r0x_s = rf32(om); const r0y_s = rf32(om + 0x04); const r0z_s = rf32(om + 0x08); + wf32(om, scale_len0 * r0x_s); wf32(om + 0x04, scale_len0 * r0y_s); wf32(om + 0x08, scale_len0 * r0z_s); + const r1x_s = rf32(om + 0x10); const r1y_s = rf32(om + 0x14); const r1z_s = rf32(om + 0x18); + wf32(om + 0x10, scale_len1 * r1x_s); wf32(om + 0x14, scale_len1 * r1y_s); wf32(om + 0x18, scale_len1 * r1z_s); + const r2x_s = rf32(om + 0x20); const r2y_s = rf32(om + 0x24); const r2z_s = rf32(om + 0x28); + wf32(om + 0x20, scale_len2 * r2x_s); wf32(om + 0x24, scale_len2 * r2y_s); wf32(om + 0x28, scale_len2 * r2z_s); - // Recompute translation: pos - scaled_matrix * pivot - wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz)); - wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz)); - wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz)); + wf32(om + 0x30, pos_x - @mulAdd(f32, scale_len2 * r2x_s, bpz, @mulAdd(f32, scale_len1 * r1x_s, bpy, scale_len0 * r0x_s * bpx))); + wf32(om + 0x34, pos_y - @mulAdd(f32, scale_len2 * r2y_s, bpz, @mulAdd(f32, scale_len1 * r1y_s, bpy, scale_len0 * r0y_s * bpx))); + wf32(om + 0x38, pos_z - @mulAdd(f32, scale_len2 * r2z_s, bpz, @mulAdd(f32, scale_len1 * r1z_s, bpy, scale_len0 * r0z_s * bpx))); wf32(om + 0x3C, 1.0); } } } // ========================================================================= - // Sections 8-11: Post-bone-loop animations - // These sections handle texture animation, color animation, bone keyframe - // post-processing, and particle emitters. They follow the same interpolation - // pattern as the bone loop but operate on different model data arrays. - // - // For the initial implementation, we delegate these to the patterns established - // above. Each section iterates over its respective model array and calls - // findInterpIdx + lerp + crossfade blend. + // Sections 8-12: Post-bone-loop animations // ========================================================================= - - // Section 8: Texture animation loop texAnimLoop(this, model_hdr); - - // Section 9: Color animation loop colorAnimLoop(this, model_hdr); - - // Section 10: Bone keyframe processing + wordAnimLoop(this, model_hdr); boneKeyframeLoop(this, model_hdr); - - // Section 11: Particle emitter loops particleLoops(this, model_hdr); - - // Section 12: Attachment recursion attachmentRecursion(this, model_hdr, bone_out_base); - // ========================================================================= // Section 13: Sync update - // ========================================================================= wu32(this + SO.sync_value, ru32(anim_ctx + 0x10)); } // ============================================================================= -// Post-bone-loop sections (extracted for readability) +// Post-bone-loop section functions // ============================================================================= fn texAnimLoop(this: u32, model_hdr: u32) void { const count = ru32(model_hdr + 0x54); if (count == 0) return; const data_base = ru32(model_hdr + 0x58); - const bone_rt_base = ru32(this + SO.bone_rt_base); + const brt_base = ru32(this + SO.bone_rt_base); const out_base = ru32(this + SO.tex_anim_out); - var i: u32 = 0; var data_off: u32 = 0; var out_off: u32 = 0; - while (i < count) : ({ - i += 1; - data_off += 0x38; - out_off += 0x14 * 4; - }) { - const anim_data = data_base + data_off; + while (i < count) : ({ i += 1; data_off += 0x38; out_off += 0x50; }) { + const ad = data_base + data_off; const output = out_base + out_off; - if (ru32(this + SO.anim_frame_ctr) < ru32(data_base + data_off + 0x0C)) { - interpVec3Track(this, bone_rt_base, anim_data, output, ufloat(ru32(bone_rt_base + BR.blend_weight))); + if (ru32(this + SO.anim_frame_ctr) < ru32(ad + 0x0C)) { + interpVec3Track(this, brt_base, ad, output, ufloat(ru32(brt_base + BR.blend_weight))); } - // Alpha/opacity track - if (ru32(this + SO.anim_frame_ctr) < ru32(anim_data + 0x28)) { - // Short value interpolation via getIndexOffset/setShortValue pattern - // This accesses short values at anim_data + 0x1C - const alpha_anim = anim_data + 0x1C; - const alpha_out = output + 0xC * 4; - findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), alpha_anim, alpha_out); - // Short value interpolation - const mode = ri16(alpha_anim); - const kf_base = ru32(alpha_anim + AD.keyframe_base); + if (ru32(this + SO.anim_frame_ctr) < ru32(ad + 0x28)) { + const alpha_ad = ad + 0x1C; + const alpha_out = output + 0x30; + findInterpIdx(this, ru32(brt_base + BR.prim_time), ru32(brt_base + BR.prim_track), alpha_ad, alpha_out); + const mode = ri16(alpha_ad); if (mode == 0) { - const sv = @as(f32, @floatFromInt(@as(i32, @intCast(@as(i16, @bitCast(ru16(kf_base + ru32(alpha_out) * 2))))))); - wf32(output + 0xF * 4, sv * SHORT_TO_FLOAT); + const kf = ru32(alpha_ad + 0x18); + wf32(alpha_out + 0x0C, @as(f32, @floatFromInt(@as(i32, ri16(kf + ru32(alpha_out) * 2)))) * getShortToFloat()); } else { - const t = ufloat(ru32(alpha_out + 8)); - const kf_data = alpha_anim + 0x08; // _padding field in AnimationData = keyframe_ranges offset - _ = kf_data; - // getIndexOffset: returns *(data+4) + idx * 2 = pointer to short - const short_base = ru32(alpha_anim + 0x18); // AD.keyframe_base = ofsValues (asm 0x715B33: [EAX+0x18]) - const v0 = @as(f32, @floatFromInt(@as(i32, @intCast(@as(i16, @bitCast(ru16(short_base + ru32(alpha_out) * 2))))))); - const v1 = @as(f32, @floatFromInt(@as(i32, @intCast(@as(i16, @bitCast(ru16(short_base + ru32(alpha_out + 4) * 2))))))); - wf32(output + 0xF * 4, (v1 * SHORT_TO_FLOAT - v0 * SHORT_TO_FLOAT) * t + v0 * SHORT_TO_FLOAT); + const primary = shortInterpToFloat(alpha_ad, alpha_out); + wf32(alpha_out + 0x0C, primary); + const bw = rf32(brt_base + BR.blend_weight); + if (bw != 0.0 and ri16(alpha_ad + 0x02) == -1) { + findInterpIdx(this, ru32(brt_base + BR.sec_time), ru32(brt_base + BR.sec_track), alpha_ad, alpha_out + 0x10); + const secondary = shortInterpToFloat(alpha_ad, alpha_out + 0x10); + wf32(alpha_out + 0x1C, secondary); + wf32(alpha_out + 0x0C, @mulAdd(f32, secondary - primary, bw, primary)); + } } } } } fn colorAnimLoop(this: u32, model_hdr: u32) void { - // Assembly: entry gate at model_hdr+0x64, loop bound at model_hdr+0x6C - if (ru32(model_hdr + 0x64) == 0) return; - const count = ru32(model_hdr + 0x6C); // loop bound from assembly 0x715F0A + const count = ru32(model_hdr + 0x64); + if (count == 0) return; const data_base = ru32(model_hdr + 0x68); - const bone_rt_base = ru32(this + SO.bone_rt_base); + const brt_base = ru32(this + SO.bone_rt_base); const out_base = ru32(this + SO.color_anim_out); - var i: u32 = 0; var data_off: u32 = 0; var out_off: u32 = 0; - while (i < count) : ({ - i += 1; - data_off += 0x1C; - out_off += 0x20; - }) { - const anim_data = data_base + data_off; + while (i < count) : ({ i += 1; data_off += 0x1C; out_off += 0x20; }) { + const ad = data_base + data_off; const output = out_base + out_off; - if (ru32(this + SO.anim_frame_ctr) < ru32(anim_data + 0x04)) { - findInterpIdx(this, ru32(bone_rt_base + BR.prim_time), ru32(bone_rt_base + BR.prim_track), anim_data, output); - const mode = ri16(anim_data); - const kf_base = ru32(anim_data + AD.keyframe_base); + if (ru32(this + SO.anim_frame_ctr) < ru32(ad + 0x0C)) { + findInterpIdx(this, ru32(brt_base + BR.prim_time), ru32(brt_base + BR.prim_track), ad, output); + const mode = ri16(ad); if (mode == 0) { - const sv = @as(f32, @floatFromInt(@as(i32, @intCast(@as(i16, @bitCast(ru16(kf_base + ru32(output) * 2))))))); - wf32(output + 0x0C, sv * SHORT_TO_FLOAT); + wf32(output + 0x0C, @as(f32, @floatFromInt(@as(i32, ri16(ru32(ad + 0x18) + ru32(output) * 2)))) * getShortToFloat()); } else { - const t = ufloat(ru32(output + 8)); - const short_base = ru32(anim_data + 0x18); // AD.keyframe_base (asm: [EDI+0x18] for short values) - const v0 = @as(f32, @floatFromInt(@as(i32, @intCast(@as(i16, @bitCast(ru16(short_base + ru32(output) * 2))))))); - const v1 = @as(f32, @floatFromInt(@as(i32, @intCast(@as(i16, @bitCast(ru16(short_base + ru32(output + 4) * 2))))))); - wf32(output + 0x0C, (v1 * SHORT_TO_FLOAT - v0 * SHORT_TO_FLOAT) * t + v0 * SHORT_TO_FLOAT); + const primary = shortInterpToFloat(ad, output); + wf32(output + 0x0C, primary); + const bw = rf32(brt_base + BR.blend_weight); + if (bw != 0.0 and ri16(ad + 0x02) == -1) { + findInterpIdx(this, ru32(brt_base + BR.sec_time), ru32(brt_base + BR.sec_track), ad, output + 0x10); + const secondary = shortInterpToFloat(ad, output + 0x10); + wf32(output + 0x1C, secondary); + wf32(output + 0x0C, @mulAdd(f32, secondary - primary, bw, primary)); + } + } + } + } +} + +fn wordAnimLoop(this: u32, model_hdr: u32) void { + const count = ru32(model_hdr + 0x6C); + if (count == 0) return; + const data_base = ru32(model_hdr + 0x70); + const brt_base = ru32(this + SO.bone_rt_base); + const out_base = ru32(this + SO.scale1); + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count) : ({ i += 1; data_off += 0x1C; out_off += 0x20; }) { + const ad = data_base + data_off; + const output = out_base + out_off; + if (ru32(this + SO.anim_frame_ctr) < ru32(ad + 0x0C)) { + findInterpIdx(this, ru32(brt_base + BR.prim_time), ru32(brt_base + BR.prim_track), ad, output); + const kf = ru32(ad + 0x18); + wu16(output + 0x0C, ru16(kf + ru32(output) * 2)); + if (ri16(ad) != 0) { + const bw = rf32(brt_base + BR.blend_weight); + if (bw != 0.0 and ri16(ad + 0x02) == -1) { + findInterpIdx(this, ru32(brt_base + BR.sec_time), ru32(brt_base + BR.sec_track), ad, output + 0x10); + wu16(output + 0x1C, ru16(kf + ru32(output + 0x10) * 2)); + } } } } @@ -1844,69 +1356,49 @@ fn colorAnimLoop(this: u32, model_hdr: u32) void { fn boneKeyframeLoop(this: u32, model_hdr: u32) void { const count = ru32(model_hdr + 0x74); if (count == 0) return; + if ((ru8(0xCF04C4) & 1) == 0) { + wu8(0xCF04C4, ru8(0xCF04C4) | 1); + wu32(0xCF043C, 0x3F000000); + wu32(0xCF0440, 0x3F000000); + wu32(0xCF0444, 0x00000000); + const initFn: *const fn (u32) callconv(.c) void = @ptrFromInt(0x409AEF); + initFn(0x7187E0); + } const data_base = ru32(model_hdr + 0x78); - const bone_rt_base = ru32(this + SO.bone_rt_base); + const brt_base = ru32(this + SO.bone_rt_base); const scale2_base = ru32(this + SO.scale2); const scale3_base = ru32(this + SO.scale3); - var i: u32 = 0; var data_off: u32 = 0; var out_off: u32 = 0; var mat_off: u32 = 0; - while (i < count) : ({ - i += 1; - data_off += 0x54; // assembly at 0x7163A2: ADD EDI, 0x54 - out_off += 0x98; // assembly at 0x7163A5: ADD ESI, 0x98 - mat_off += 0x40; // assembly at 0x715395: ADD EDX, 0x40 - }) { - const kf_data = data_base + data_off; - const output = @as(u32, @intCast(@as(i32, @bitCast(scale2_base)) + @as(i32, @bitCast(out_off)))); - const mat_out = @as(u32, @intCast(@as(i32, @bitCast(scale3_base)) + @as(i32, @bitCast(mat_off)))); - - // Init identity matrix for this keyframe entry + while (i < count) : ({ i += 1; data_off += 0x54; out_off += 0x98; mat_off += 0x40; }) { + const kf = data_base + data_off; + const output = scale2_base + out_off; + const mat_out = scale3_base + mat_off; setIdentity(mat_out); - - // Rotation: AnimData at kf_entry+0x1C, gate at kf_entry+0x28 - // Assembly at 0x715FDB: CMP [ECX+0x28], 0; AnimData at EDX+0x1C - if (ru32(kf_data + 0x28) != 0) { - interpAnimKF(this, bone_rt_base, kf_data + 0x1C, output + 0x30); - // ApplyTranslation(0.5, 0.5, 0.0), rotateByQuaternion, ApplyTranslation(-0.5, -0.5, 0.0) - applyTranslation(mat_out, 0.5, 0.5, 0.0); - rotateByQuaternion(mat_out, ufloat(ru32(output + 0x3C)), ufloat(ru32(output + 0x40)), ufloat(ru32(output + 0x44)), ufloat(ru32(output + 0x48))); - applyTranslation(mat_out, -0.5, -0.5, 0.0); + if (ru32(kf + 0x28) != 0) { + interpAnimKF(this, brt_base, kf + 0x1C, output + 0x30); + applyTranslationFromPtr(mat_out, 0xCF043C); + rotateByQuaternionFromPtr(mat_out, output + 0x3C); + applyTranslation(mat_out, -rf32(0xCF043C), -rf32(0xCF0440), -rf32(0xCF0444)); } - - // Scale: AnimData at kf_entry+0x38, gate at kf_entry+0x44 - // Assembly at 0x716052: CMP [ECX+0x44], 0; AnimData at EDX+0x38 - if (ru32(kf_data + 0x44) != 0) { - interpVec3Track(this, bone_rt_base, kf_data + 0x38, output + 0x68, ufloat(ru32(bone_rt_base + BR.blend_weight))); - applyTranslation(mat_out, 0.5, 0.5, 0.0); - scaleMatrix3x3(mat_out, ufloat(ru32(output + 0x74)), ufloat(ru32(output + 0x78)), ufloat(ru32(output + 0x7C))); - applyTranslation(mat_out, -0.5, -0.5, 0.0); + if (ru32(kf + 0x44) != 0) { + interpVec3Track(this, brt_base, kf + 0x38, output + 0x68, ufloat(ru32(brt_base + BR.blend_weight))); + applyTranslationFromPtr(mat_out, 0xCF043C); + scaleMatrix3x3FromPtr(mat_out, output + 0x74); + applyTranslation(mat_out, -rf32(0xCF043C), -rf32(0xCF0440), -rf32(0xCF0444)); } - - // Translation: AnimData at kf_entry+0x00, gate at kf_entry+0x0C - // Assembly at 0x716216: CMP [ECX+0x0C], 0; AnimData at kf_entry+0x00 - if (ru32(kf_data + 0x0C) != 0) { - interpVec3Track(this, bone_rt_base, kf_data, output, ufloat(ru32(bone_rt_base + BR.blend_weight))); - applyTranslation(mat_out, ufloat(ru32(output + 0x0C)), ufloat(ru32(output + 0x10)), ufloat(ru32(output + 0x14))); + if (ru32(kf + 0x0C) != 0) { + interpVec3Track(this, brt_base, kf, output, ufloat(ru32(brt_base + BR.blend_weight))); + applyTranslationFromPtr(mat_out, output + 0x0C); } } } fn particleLoops(this: u32, model_hdr: u32) void { - // Particle emitters are the largest section (~1000 lines of decompiled C). - // They follow the same interpolation patterns but with many sub-tracks per emitter. - // For the initial implementation, we handle the key tracks (position, speed, scale). - // The remaining tracks (color, alpha, emission rate, etc.) use identical patterns. - - // Ribbon emitters (model_hdr + 0x11C) ribbonEmitterLoop(this, model_hdr); - - // Particle emitters (model_hdr + 0x124) particleEmitterLoop(this, model_hdr); - - // Additional particle sections (model_hdr + 0x134, 0x13C) additionalParticleLoops(this, model_hdr); } @@ -1915,80 +1407,42 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32) void { if (count == 0) return; const data_base = ru32(model_hdr + 0x120); const out_base = ru32(this + SO.field_200); - const bone_rt_base = ru32(this + SO.bone_rt_base); + const brt_base = ru32(this + SO.bone_rt_base); const frame_ctr = ru32(this + SO.anim_frame_ctr); - var i: u32 = 0; while (i < count) : (i += 1) { - const entry = data_base + i * 0xD4; // asm 0x716ABC: ADD EDI, 0xD4 - const output = out_base + i * 0x170; // asm 0x716AC2: ADD ESI, 0x170 + const entry = data_base + i * 0xD4; + const output = out_base + i * 0x170; const bone_idx = @as(u32, ru16(entry + 2)); - const bone_rt = bone_rt_base + bone_idx * 0x118; + const bone_rt = brt_base + bone_idx * 0x118; - // ---- Visibility byte animation (asm 0x7163FC-0x7164F2) ---- - // First check: output+0x100 flag gates visibility animation if (ru32(output + 0x100) != 0) { - // Visibility animation gate: entry+0xC4 (asm 0x71640D) if (ru32(entry + 0xC4) != 0) { - // findInterpIdx for visibility byte: AD=entry+0xB8, output=output+0xE0 findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), entry + 0xB8, output + 0xE0); - const vis_idx0 = ru32(output + 0xE0); - const vis_values = ru32(entry + 0xD0); // entry+0xB8+0x18 = AD.keyframe_base - // Both mode 0 and mode != 0 store byte at ofsValues[idx0] - wu8(output + 0xEC, ru8(vis_values + vis_idx0)); - // Crossfade only in interp mode (asm 0x71649B-0x7164EC) + wu8(output + 0xEC, ru8(ru32(entry + 0xD0) + ru32(output + 0xE0))); if (ri16(entry + 0xB8) != 0) { if (rf32(bone_rt + BR.blend_weight) != 0.0 and ri16(entry + 0xBA) == -1) { findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), entry + 0xB8, output + 0xF0); - wu8(output + 0xFC, ru8(vis_values + ru32(output + 0xF0))); + wu8(output + 0xFC, ru8(ru32(entry + 0xD0) + ru32(output + 0xF0))); } } } } - // ---- Visibility gate (asm 0x7164F2-0x716514) ---- - // Process tracks if: (output+0x100 set AND visible) OR first frame - const should_process = blk: { - if (ru32(output + 0x100) != 0 and ru8(output + 0xEC) != 0) break :blk true; - if (frame_ctr == 0) break :blk true; - break :blk false; - }; + const should_process = (ru32(output + 0x100) != 0 and ru8(output + 0xEC) != 0) or frame_ctr == 0; if (!should_process) continue; - // ---- Track A (float): gate=entry+0x38, AD=entry+0x2C, output+0x30 ---- - // Assembly: 0x716514 CMP [EDX+0x38], [EBX+0x8C]; uses *4 addressing (float, not Vec3) - if (frame_ctr < ru32(entry + 0x38)) { - interpFloatTrack(this, bone_rt, entry + 0x2C, output + 0x30); - } - - // ---- Track B (Vec3): gate=entry+0x1C, AD=entry+0x10, output+0x00 ---- - // Assembly: 0x71660D CMP [EDX+0x1C]; uses *12 addressing (Vec3) + if (frame_ctr < ru32(entry + 0x38)) interpFloatTrack(this, bone_rt, entry + 0x2C, output + 0x30, ufloat(ru32(bone_rt + BR.blend_weight))); if (frame_ctr < ru32(entry + 0x1C)) { interpVec3Track(this, bone_rt, entry + 0x10, output, ufloat(ru32(bone_rt + BR.blend_weight))); - // Post-processing 1 (asm 0x71678A-0x7167CE): - // output+0x134 = Track_B_Vec3 * (Track_A_float * render_scale_z) - const scale1 = rf32(output + 0x3C) * rf32(this + SO.render_scale_z); - wf32(output + 0x134, rf32(output + 0x0C) * scale1); - wf32(output + 0x138, rf32(output + 0x10) * scale1); - wf32(output + 0x13C, rf32(output + 0x14) * scale1); + const s = rf32(output + 0x3C) * rf32(this + SO.render_scale_z); + wf32(output + 0x134, rf32(output + 0x0C) * s); wf32(output + 0x138, rf32(output + 0x10) * s); wf32(output + 0x13C, rf32(output + 0x14) * s); } - - // ---- Track C (float): gate=entry+0x70, AD=entry+0x64, output+0x80 ---- - // Assembly: 0x7167D4 CMP [EAX+0x70] - if (frame_ctr < ru32(entry + 0x70)) { - interpFloatTrack(this, bone_rt, entry + 0x64, output + 0x80); - } - - // ---- Track D (Vec3): gate=entry+0x54, AD=entry+0x48, output+0x50 ---- - // Assembly: 0x7168EE CMP [EDX+0x54]; uses *12 addressing (Vec3) + if (frame_ctr < ru32(entry + 0x70)) interpFloatTrack(this, bone_rt, entry + 0x64, output + 0x80, ufloat(ru32(bone_rt + BR.blend_weight))); if (frame_ctr < ru32(entry + 0x54)) { interpVec3Track(this, bone_rt, entry + 0x48, output + 0x50, ufloat(ru32(bone_rt + BR.blend_weight))); - // Post-processing 2 (asm 0x716A67-0x716AA6): - // output+0x140 = Track_D_Vec3 * (Track_C_float * render_scale_z) - const scale2 = rf32(output + 0x8C) * rf32(this + SO.render_scale_z); - wf32(output + 0x140, rf32(output + 0x5C) * scale2); - wf32(output + 0x144, rf32(output + 0x60) * scale2); - wf32(output + 0x148, rf32(output + 0x64) * scale2); + const s = rf32(output + 0x8C) * rf32(this + SO.render_scale_z); + wf32(output + 0x140, rf32(output + 0x5C) * s); wf32(output + 0x144, rf32(output + 0x60) * s); wf32(output + 0x148, rf32(output + 0x64) * s); } } } @@ -1998,257 +1452,150 @@ fn particleEmitterLoop(this: u32, model_hdr: u32) void { if (count == 0) return; const data_base = ru32(model_hdr + 0x128); const out_base = ru32(this + SO.particle1); - const bone_rt_base = ru32(this + SO.bone_rt_base); + const brt_base = ru32(this + SO.bone_rt_base); const frame_ctr = ru32(this + SO.anim_frame_ctr); - var i: u32 = 0; var data_off: u32 = 0; var out_off: u32 = 0; - while (i < count) : ({ - i += 1; - data_off += 0x7C; // asm 0x717624: ADD EDI, 0x7C - out_off += 0x84; // asm 0x717627: ADD ESI, 0x84 - }) { + while (i < count) : ({ i += 1; data_off += 0x7C; out_off += 0x84; }) { const entry = data_base + data_off; const output = out_base + out_off; - - // Assembly uses bone_rt_base directly (bone 0) for ALL tracks — NOT per-entry bone_idx. - // Verified: 0x716B2E MOV EAX,[EBX+0x90]; 0x716F58 same; 0x7173B1 same. - - // Track 1 (Vec3, 36-byte kf): gate=entry+0x1C, AD=entry+0x10, output=+0x00 - // Assembly: 0x716B19 CMP [EAX+0x1C]; 0x716B3B LEA ESI,[EDX+0x10] - if (frame_ctr < ru32(entry + 0x1C)) { - interpVec3Track36(this, bone_rt_base, entry + 0x10, output); - } - // Track 2 (Vec3, 36-byte kf): gate=entry+0x44, AD=entry+0x38, output=+0x30 - // Assembly: 0x716F44 MOV EDX,[ECX+0x44]; 0x716F55 LEA ECX,[EAX+0x38] - if (frame_ctr < ru32(entry + 0x44)) { - interpVec3Track36(this, bone_rt_base, entry + 0x38, output + 0x30); - } - // Track 3 (float, 12-byte kf): gate=entry+0x6C, AD=entry+0x60, output=+0x60 - // Assembly: 0x71739A MOV EDX,[ECX+0x6C]; 0x7173AE LEA EDI,[EAX+0x60] - if (frame_ctr < ru32(entry + 0x6C)) { - interpFloatTrack12(this, bone_rt_base, entry + 0x60, output + 0x60); - } + if (frame_ctr < ru32(entry + 0x1C)) interpVec3Track36(this, brt_base, entry + 0x10, output); + if (frame_ctr < ru32(entry + 0x44)) interpVec3Track36(this, brt_base, entry + 0x38, output + 0x30); + if (frame_ctr < ru32(entry + 0x6C)) interpFloatTrack12(this, brt_base, entry + 0x60, output + 0x60); } } fn additionalParticleLoops(this: u32, model_hdr: u32) void { - // Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A) - // Then additional_remaining reset at 0x717D6F - // Then model_hdr+0x13C section (asm 0x717D75-0x7185E3) - - // Section 12c: model_hdr+0x134 particle visibility/tracks - // count=+0x134, data=+0x138, output=this+0x3C8 - // Data stride 0xDC, output stride 0xD0 - // Each entry: bone_idx at +0x04, visibility at +0xCC - // Sub-tracks: visibility(+0xC0), position(+0x24), alpha(+0x40), - // speed(+0x5C), emission(+0x78), scale(+0xA4) + // Section 0x134 if (ru32(model_hdr + 0x134) != 0) { - const count0 = ru32(model_hdr + 0x134); - const data_base0 = ru32(model_hdr + 0x138); - const out_base0 = ru32(this + 0x3C8); // SO.particle2 - const bone_rt_base = ru32(this + SO.bone_rt_base); - + const cnt = ru32(model_hdr + 0x134); + const db = ru32(model_hdr + 0x138); + const ob = ru32(this + 0x3C8); + const brt_base = ru32(this + SO.bone_rt_base); var i: u32 = 0; - var data_off: u32 = 0; - var out_off: u32 = 0; - while (i < count0) : ({ - i += 1; - data_off += 0xDC; // asm 0x717D4D - out_off += 0xD0; // asm 0x717D53 - }) { - const entry = data_base0 + data_off; - const output = out_base0 + out_off; - - // Visibility check: entry+0xCC vs anim_frame_ctr + var doff: u32 = 0; + var ooff: u32 = 0; + while (i < cnt) : ({ i += 1; doff += 0xDC; ooff += 0xD0; }) { + const entry = db + doff; + const output = ob + ooff; if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) { - const bone_idx = @as(u32, ru16(entry + 0x04)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - // Visibility byte animation at entry+0xC0 - findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xC0, output + 0xB0); - const vis_mode = ri16(entry + 0xC0); - if (vis_mode == 0) { + const bi = @as(u32, ru16(entry + 0x04)); + const brt = brt_base + bi * 0x118; + findInterpIdx(this, ru32(brt + 0x98), ru32(brt + 0x9C), entry + 0xC0, output + 0xB0); + if (ri16(entry + 0xC0) == 0) { wu8(output + 0xBC, ru8(ru32(entry + 0xC0 + 0x18) + ru32(output + 0xB0))); } else { wu8(output + 0xBC, ru8(ru32(output + 0xB0) + ru32(entry + 0xD8))); - // Crossfade blend for visibility if needed - if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xC2) == -1) { - findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xC0, output + 0xC0); + if (rf32(brt + 0x10C) != 0.0 and ri16(entry + 0xC2) == -1) { + findInterpIdx(this, ru32(brt + 0xC4), ru32(brt + 0xC8), entry + 0xC0, output + 0xC0); wu8(output + 0xCC, ru8(ru32(output + 0xC0) + ru32(entry + 0xD8))); } } } - - // Position track: entry+0x24 vs entry+0x30 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x30)) { - const bone_idx = @as(u32, ru16(entry + 0x04)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - interpVec3Track(this, bone_rt, entry + 0x24, output, ufloat(ru32(bone_rt + BR.blend_weight))); + const bi = @as(u32, ru16(entry + 0x04)); + const brt = brt_base + bi * 0x118; + interpVec3Track(this, brt, entry + 0x24, output, ufloat(ru32(brt + BR.blend_weight))); } - - // Alpha track: entry+0x40 vs entry+0x4C if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x4C)) { - const bone_idx = @as(u32, ru16(entry + 0x04)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - // Short-value interpolation pattern - findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x40, output + 0x30); - const alpha_mode = ri16(entry + 0x40); - const alpha_base = ru32(entry + 0x40 + 0x18); - if (alpha_mode == 0) { - const sv = @as(f32, @floatFromInt(@as(i32, @intCast(ri16(alpha_base + ru32(output + 0x30) * 2))))); - wf32(output + 0x3C, sv * SHORT_TO_FLOAT); + const bi = @as(u32, ru16(entry + 0x04)); + const brt = brt_base + bi * 0x118; + findInterpIdx(this, ru32(brt + 0x98), ru32(brt + 0x9C), entry + 0x40, output + 0x30); + const kf_base = ru32(entry + 0x40 + AD.keyframe_base); + const s2f = getShortToFloat(); + if (ri16(entry + 0x40) == 0) { + wf32(output + 0x3C, @as(f32, @floatFromInt(@as(i32, ri16(kf_base + ru32(output + 0x30) * 2)))) * s2f); } else { const t = ufloat(ru32(output + 0x38)); - const short_ranges = ru32(entry + 0x40 + 0x18); // AD.keyframe_base for short values - const v0 = @as(f32, @floatFromInt(@as(i32, @intCast(ri16(short_ranges + ru32(output + 0x30) * 2))))); - const v1 = @as(f32, @floatFromInt(@as(i32, @intCast(ri16(short_ranges + ru32(output + 0x34) * 2))))); - wf32(output + 0x3C, (v1 * SHORT_TO_FLOAT - v0 * SHORT_TO_FLOAT) * t + v0 * SHORT_TO_FLOAT); + const v0 = @as(f32, @floatFromInt(@as(i32, ri16(kf_base + ru32(output + 0x30) * 2)))) * s2f; + const v1 = @as(f32, @floatFromInt(@as(i32, ri16(kf_base + ru32(output + 0x34) * 2)))) * s2f; + wf32(output + 0x3C, @mulAdd(f32, v1 - v0, t, v0)); } } - - // Speed track: entry+0x5C vs entry+0x68 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x68)) { - const bone_idx = @as(u32, ru16(entry + 0x04)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - interpFloatTrack(this, bone_rt, entry + 0x5C, output + 0x50); + const bi = @as(u32, ru16(entry + 0x04)); + const brt = brt_base + bi * 0x118; + interpFloatTrack(this, brt, entry + 0x5C, output + 0x50, ufloat(ru32(brt + BR.blend_weight))); } - - // Emission rate: entry+0x78 vs entry+0x84 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x84)) { - const bone_idx = @as(u32, ru16(entry + 0x04)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - interpFloatTrack(this, bone_rt, entry + 0x78, output + 0x70); + const bi = @as(u32, ru16(entry + 0x04)); + const brt = brt_base + bi * 0x118; + interpFloatTrack(this, brt, entry + 0x78, output + 0x70, ufloat(ru32(brt + BR.blend_weight))); } - - // Scale track: entry+0xA4 vs entry+0xB0 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) { - const bone_idx = @as(u32, ru16(entry + 0x04)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - // This uses getInterpolatedFloat pattern - findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xA4, output + 0x90); - const scale_mode = ri16(entry + 0xA4); - const scale_base = ru32(entry + 0xA4 + 0x18); - if (scale_mode == 0) { - wu16(output + 0x9C, ru16(scale_base + ru32(output + 0x90) * 2)); - } else { - wu16(output + 0x9C, ru16(scale_base + ru32(output + 0x90) * 2)); - // Crossfade - if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xA6) == -1) { - findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xA4, output + 0xA0); - wu16(output + 0xAC, ru16(scale_base + ru32(output + 0xA0) * 2)); + const bi = @as(u32, ru16(entry + 0x04)); + const brt = brt_base + bi * 0x118; + findInterpIdx(this, ru32(brt + 0x98), ru32(brt + 0x9C), entry + 0xA4, output + 0x90); + const kf_base = ru32(entry + 0xA4 + AD.keyframe_base); + wu16(output + 0x9C, ru16(kf_base + ru32(output + 0x90) * 2)); + if (ri16(entry + 0xA4) != 0) { + if (rf32(brt + 0x10C) != 0.0 and ri16(entry + 0xA6) == -1) { + findInterpIdx(this, ru32(brt + 0xC4), ru32(brt + 0xC8), entry + 0xA4, output + 0xA0); + wu16(output + 0xAC, ru16(kf_base + ru32(output + 0xA0) * 2)); } } } } } - // Additional remaining data reset — between 0x134 and 0x13C sections - // Assembly at 0x717D6F: MOV [EBX+0x3D8], 0 wu32(this + 0x3D8, 0); - // Section 12e: model_hdr+0x13C (largest particle section) - // count=+0x13C, data=+0x140 - // output1=this+0x3D0, output2=this+0x3D4 - // Data stride 0x1F8, output stride 0x16C + // Section 0x13C const count1 = ru32(model_hdr + 0x13C); if (count1 != 0) { - const data_base = ru32(model_hdr + 0x140); - const bone_rt_base = ru32(this + SO.bone_rt_base); - const particle_base = ru32(this + 0x3D0); // SO.particle3 - + const db = ru32(model_hdr + 0x140); + const brt_base = ru32(this + SO.bone_rt_base); + const pb = ru32(this + 0x3D0); var i: u32 = 0; - var data_off: u32 = 0; - var out_off: u32 = 0; - while (i < count1) : ({ - i += 1; - data_off += 0x1F8; // asm 0x7185CD - out_off += 0x16C; // asm 0x7185BA - }) { - const entry = data_base + data_off; - const output = particle_base + out_off; - const bone_idx = @as(u32, ru16(entry + 0x14)); - const bone_rt = bone_rt_base + bone_idx * 0x118; + var doff: u32 = 0; + var ooff: u32 = 0; + while (i < count1) : ({ i += 1; doff += 0x1F8; ooff += 0x16C; }) { + const entry = db + doff; + const output = pb + ooff; + const bi = @as(u32, ru16(entry + 0x14)); + const brt = brt_base + bi * 0x118; + const pp = ru32(this + 0x3D4); + const local_14 = ru32(pp + i * 4); - // All tracks from assembly 0x717D90-0x7185E3: - const particle_ptrs = ru32(this + 0x3D4); // [EBX+0x3D4] - const local_14 = ru32(particle_ptrs + i * 4); // per-emitter data ptr - - // Visibility: gate=entry+0x1E8, AnimData=entry+0x1DC, output=output+0x140 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x1E8)) { - findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x1DC, output + 0x140); + findInterpIdx(this, ru32(brt + 0x98), ru32(brt + 0x9C), entry + 0x1DC, output + 0x140); if (ri16(entry + 0x1DC) == 0) { wu8(output + 0x14C, ru8(ru32(entry + 0x1F4) + ru32(output + 0x140))); } else { wu8(output + 0x14C, ru8(ru32(output + 0x140) + ru32(entry + 0x1F4))); - if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0x1DE) == -1) { - findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0x1DC, output + 0x150); + if (rf32(brt + 0x10C) != 0.0 and ri16(entry + 0x1DE) == -1) { + findInterpIdx(this, ru32(brt + 0xC4), ru32(brt + 0xC8), entry + 0x1DC, output + 0x150); wu8(output + 0x15C, ru8(ru32(entry + 0x1F4) + ru32(output + 0x150))); } } } - // Emitter active flag: visibility && emitter_enable_flag - const vis_byte = ru8(output + 0x14C); - const emitter_active: u32 = if (vis_byte != 0 and ru32(this + 0x50) != 0) 1 else 0; - wu32(output + 0x160, emitter_active); - // IsParticleBufferEmpty check - var buf_active: u32 = 0; - if (emitter_active != 0) { - buf_active = 1; + const vis = ru8(output + 0x14C); + const ea: u32 = if (vis != 0 and ru32(this + 0x50) != 0) 1 else 0; + wu32(output + 0x160, ea); + var ba: u32 = 0; + if (ea != 0) { + ba = 1; } else { - // IsParticleBufferEmpty — reimplemented from assembly at 0x7B5F60 - if (isParticleBufferNotEmpty(local_14)) { - buf_active = 1; - } + const isEmptyFn: *const fn (u32) callconv(.{ .x86_fastcall = .{} }) u32 = @ptrFromInt(0x7B5F60); + if (isEmptyFn(local_14) != 0) ba = 1; } - wu32(output + 0x164, buf_active); - // OR into additional_remaining - wu32(this + 0x3D8, ru32(this + 0x3D8) | buf_active); + wu32(output + 0x164, ba); + wu32(this + 0x3D8, ru32(this + 0x3D8) | ba); - // Only process tracks if visible or first frame - if (vis_byte != 0 or ru32(this + SO.anim_frame_ctr) == 0) { - // Track 1: emission rate — gate=+0x40, AnimData=+0x34, output=+0x00 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x40)) { - interpFloatTrack(this, bone_rt, entry + 0x34, output); - } - // Track 2: speed — gate=+0x5C, AnimData=+0x50, output=+0x20 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x5C)) { - interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20); - } - // Track 3: color — gate=+0x78, AnimData=+0x6C, output=+0x40 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x78)) { - interpFloatTrack(this, bone_rt, entry + 0x6C, output + 0x40); - } - // Track 4 — gate=+0x94, AnimData=+0x88, output=+0x60 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x94)) { - interpFloatTrack(this, bone_rt, entry + 0x88, output + 0x60); - } - // Track 5 (Vec3 spline) — gate=+0xB0, AnimData=+0xA4, output=+0x80 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) { - interpFloatTrack(this, bone_rt, entry + 0xA4, output + 0x80); - } - // Track 6 — gate=+0xCC, AnimData=+0xC0, output=+0xA0 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) { - interpFloatTrack(this, bone_rt, entry + 0xC0, output + 0xA0); - } - // Track 7 — gate=+0xE8, AnimData=+0xDC, output=+0xC0 - // Uses getInterpolatedFloat (0x71AF20) - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xE8)) { - getInterpolatedFloat(this, bone_rt, entry + 0xDC, output + 0xC0); - } - // Track 8 — gate=+0x104, AnimData=+0xF8, output=+0xE0 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x104)) { - getInterpolatedFloat(this, bone_rt, entry + 0xF8, output + 0xE0); - } - // Track 9 — gate=+0x120, AnimData=+0x114, output=+0x100 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x120)) { - getInterpolatedFloat(this, bone_rt, entry + 0x114, output + 0x100); - } - // Track 10 — gate=+0x13C, AnimData=+0x130, output=+0x120 - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x13C)) { - getInterpolatedFloat(this, bone_rt, entry + 0x130, output + 0x120); - } + if (vis != 0 or ru32(this + SO.anim_frame_ctr) == 0) { + const bw = ufloat(ru32(brt + BR.blend_weight)); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x40)) interpFloatTrack(this, brt, entry + 0x34, output, bw); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x5C)) interpFloatTrack(this, brt, entry + 0x50, output + 0x20, bw); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x78)) interpFloatTrack(this, brt, entry + 0x6C, output + 0x40, bw); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x94)) interpFloatTrack(this, brt, entry + 0x88, output + 0x60, bw); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) interpFloatTrack(this, brt, entry + 0xA4, output + 0x80, bw); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) interpFloatTrack(this, brt, entry + 0xC0, output + 0xA0, bw); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xE8)) getInterpolatedFloat(this, brt, entry + 0xDC, output + 0xC0); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x104)) getInterpolatedFloat(this, brt, entry + 0xF8, output + 0xE0); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x120)) getInterpolatedFloat(this, brt, entry + 0x114, output + 0x100); + if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x13C)) getInterpolatedFloat(this, brt, entry + 0x130, output + 0x120); } } } @@ -2259,64 +1606,38 @@ fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32) void { if (hierarchy == 0) return; const attach_count = ru32(model_hdr + 0x104); - if (attach_count == 0) return; const attach_data = ru32(model_hdr + 0x108); - // Process attachment byte animations var att_i: u32 = 0; var att_off: u32 = 0; - while (att_i < attach_count) : ({ - att_i += 1; - att_off += 0x30; - }) { + while (att_i < attach_count) : ({ att_i += 1; att_off += 0x30; }) { const att_entry = attach_data + att_off; if (ru32(this + SO.anim_frame_ctr) < ru32(att_entry + 0x20)) { - const bone_idx = @as(u32, ru16(att_entry + 4)); - const bone_rt = ru32(this + SO.bone_rt_base) + bone_idx * 0x118; - // extractAnimationByteFromKeyframes — simplified - findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), att_entry + 0x14, hierarchy + att_i * 0x20); + const bi = @as(u32, ru16(att_entry + 4)); + const brt = ru32(this + SO.bone_rt_base) + bi * 0x118; + extractByte(this, brt, att_entry + 0x14, hierarchy + att_i * 0x20); } } - // Iterate child scene objects linked list var child = ru32(this + SO.hierarchy_idx); while (child != 0) { - // child->attach_idx at +0x1D4 (assembly-verified: MOV EAX,[ECX+0x1D4] at 0x718668) const attach_idx = ru32(child + 0x1D4); - - // Check if attachment is valid (0xFFFF = no attachment) if (attach_idx != 0xFFFF) { - const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C); - if (visible != 0) { + if (ru8(hierarchy + attach_idx * 0x20 + 0x0C) != 0) { const att_entry = attach_data + attach_idx * 0x30; - const bone_idx = @as(u32, ru16(att_entry + 4)); - const bone_mat = bone_out_base + bone_idx * 0x40; - - // Copy parent bone matrix to local - var local_1a0: [16]f32 = undefined; - for (0..16) |fi| { - local_1a0[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4); - } - - // Apply attachment offset translation + const bi = @as(u32, ru16(att_entry + 4)); + const bone_mat = bone_out_base + bi * 0x40; + var local: [16]f32 align(16) = undefined; + for (0..16) |fi| local[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4); const ox = rf32(att_entry + 8); const oy = rf32(att_entry + 0xC); const oz = rf32(att_entry + 0x10); - local_1a0[12] += local_1a0[0] * ox + local_1a0[4] * oy + local_1a0[8] * oz; - local_1a0[13] += local_1a0[1] * ox + local_1a0[5] * oy + local_1a0[9] * oz; - local_1a0[14] += local_1a0[2] * ox + local_1a0[6] * oy + local_1a0[10] * oz; - - // Recursive call for child attachment SceneObject. - // mat1 = attachment-adjusted parent bone matrix - // mat2 = parent's world position Vec3 - // mat3 = parent's render priority Vec3 (offset) - // mat4 = parent's render_scale_z (float as u32 bits) - transformMatrix4x4_SSE(child, @intFromPtr(&local_1a0), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z)); + local[12] += @mulAdd(f32, local[8], oz, @mulAdd(f32, local[4], oy, local[0] * ox)); + local[13] += @mulAdd(f32, local[9], oz, @mulAdd(f32, local[5], oy, local[1] * ox)); + local[14] += @mulAdd(f32, local[10], oz, @mulAdd(f32, local[6], oy, local[2] * ox)); + transformMatrix4x4_SSE(child, @intFromPtr(&local), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z)); } } - - // Next sibling in linked list - // Assembly-verified: MOV ECX,[ECX+0x1E4] at 0x718764 child = ru32(child + 0x1E4); } } diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index 6a64363..1aa7e32 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -258,15 +258,7 @@ fn diagCompare(label: [*:0]const u8, snap: []const u8, live: u32, len: u32) void } } -// Test A: stripped detour — pure passthrough to REF, no profiling, no dispatch logic. -// If black → issue is REF code structure. If renders → detour overhead was the problem. -const DIRECT_REF_TEST = true; - fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(hook.cc.thiscall) void { - if (DIRECT_REF_TEST) { - transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4); - return; - } const start = rdtsc(); const model_data = hook.readMem(u32, this + 0x10); @@ -295,282 +287,10 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco if (teardown_active) { transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - } else if (false and !is_early and diag_count < DIAG_MAX and t44_depth == 1) { - // --- DIAGNOSTIC (disabled): run original, snapshot, run REF, compare --- - transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - - // Gather region info from SceneObject - const model_ctr_d = hook.readMem(u32, this + 0x30); - const model_hdr_d = if (model_ctr_d != 0) hook.readMem(u32, model_ctr_d + 0x130) else 0; - const bc = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x34) else 0; - const bone_rt_base = hook.readMem(u32, this + 0x90); - const bone_out_base = hook.readMem(u32, this + 0x94); - const tex_out = hook.readMem(u32, this + 0xA0); - const col_out = hook.readMem(u32, this + 0xA8); - const scale2 = hook.readMem(u32, this + 0xB0); - const scale3 = hook.readMem(u32, this + 0xB4); - const gs_vals = hook.readMem(u32, this + 0x64); - const anim_ctx_d = hook.readMem(u32, this + 0x2C); - const emitter_d = hook.readMem(u32, this + 0x1CC); - const gs_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x14) else 0; - const tex_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x54) else 0; - const col_gate = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x64) else 0; - const col_count = if (model_hdr_d != 0 and col_gate != 0) hook.readMem(u32, model_hdr_d + 0x6C) else 0; - const bkf_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x74) else 0; - const rib_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x11C) else 0; - const p124_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x124) else 0; - const p134_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x134) else 0; - const p13c_count = if (model_hdr_d != 0) hook.readMem(u32, model_hdr_d + 0x13C) else 0; - - // Define regions to compare (addr, len, label) — fit in 128KB buffer - const Region = struct { addr: u32, len: u32, label: [*:0]const u8 }; - var regions: [20]Region = undefined; - var n_regions: u32 = 0; - - // SceneObject: 0x000-0x3E0 - regions[n_regions] = .{ .addr = this, .len = 0x3E0, .label = "SceneObject" }; - n_regions += 1; - - // Bone runtime: all bones - if (bc > 0 and bone_rt_base != 0) { - const brt_len = @min(bc * 0x118, 0x10000); // cap at 64KB - regions[n_regions] = .{ .addr = bone_rt_base, .len = brt_len, .label = "BoneRT" }; - n_regions += 1; - } - - // Bone output: all bones - if (bc > 0 and bone_out_base != 0) { - const bout_len = @min(bc * 0x40, 0x4000); - regions[n_regions] = .{ .addr = bone_out_base, .len = bout_len, .label = "BoneOut" }; - n_regions += 1; - } - - // Global sequence values - if (gs_count > 0 and gs_vals != 0) { - regions[n_regions] = .{ .addr = gs_vals, .len = gs_count * 4, .label = "GSValues" }; - n_regions += 1; - } - - // Texture animation output - if (tex_count > 0 and tex_out != 0) { - regions[n_regions] = .{ .addr = tex_out, .len = @min(tex_count * 0x50, 0x1000), .label = "TexAnim" }; - n_regions += 1; - } - - // Color animation output - if (col_count > 0 and col_out != 0) { - regions[n_regions] = .{ .addr = col_out, .len = @min(col_count * 0x20, 0x400), .label = "ColorAnim" }; - n_regions += 1; - } - - // Bone keyframe scale2/scale3 buffers - if (bkf_count > 0 and scale2 != 0) { - regions[n_regions] = .{ .addr = scale2, .len = @min(bkf_count * 0x98, 0x2000), .label = "BKF_Scale2" }; - n_regions += 1; - } - if (bkf_count > 0 and scale3 != 0) { - regions[n_regions] = .{ .addr = scale3, .len = @min(bkf_count * 0x40, 0x1000), .label = "BKF_Scale3" }; - n_regions += 1; - } - - // Animation context (read-only but check) - if (anim_ctx_d != 0) { - regions[n_regions] = .{ .addr = anim_ctx_d, .len = 0x20, .label = "AnimCtx" }; - n_regions += 1; - } - - // Emitter context - if (emitter_d != 0) { - regions[n_regions] = .{ .addr = emitter_d, .len = 0x200, .label = "EmitterCtx" }; - n_regions += 1; - } - - // Ribbon emitter output (this+0x200) - if (rib_count > 0) { - const rib_out = hook.readMem(u32, this + 0x200); - if (rib_out != 0) { - regions[n_regions] = .{ .addr = rib_out, .len = @min(rib_count * 0x170, 0x4000), .label = "RibbonOut" }; - n_regions += 1; - } - } - - // Particle 0x124 output (this+0x3C4) - if (p124_count > 0) { - const p124_out = hook.readMem(u32, this + 0x3C4); - if (p124_out != 0) { - regions[n_regions] = .{ .addr = p124_out, .len = @min(p124_count * 0x84, 0x2000), .label = "Part124" }; - n_regions += 1; - } - } - - // Particle 0x134 output (this+0x3C8) - if (p134_count > 0) { - const p134_out = hook.readMem(u32, this + 0x3C8); - if (p134_out != 0) { - regions[n_regions] = .{ .addr = p134_out, .len = @min(p134_count * 0xD0, 0x4000), .label = "Part134" }; - n_regions += 1; - } - } - - // Particle 0x13C output (this+0x3D0) - if (p13c_count > 0) { - const p13c_out = hook.readMem(u32, this + 0x3D0); - if (p13c_out != 0) { - regions[n_regions] = .{ .addr = p13c_out, .len = @min(p13c_count * 0x16C, 0x8000), .label = "Part13C" }; - n_regions += 1; - } - } - - // Globals - regions[n_regions] = .{ .addr = 0xCF0400, .len = 0x100, .label = "Globals_CF04" }; - n_regions += 1; - - // Snapshot all regions after original ran - var buf_off: u32 = 0; - var region_starts: [20]u32 = undefined; - var ri: u32 = 0; - while (ri < n_regions) : (ri += 1) { - region_starts[ri] = buf_off; - const len = regions[ri].len; - if (buf_off + len <= diag_buf.len) { - diagSnapshot(diag_buf[buf_off .. buf_off + len], regions[ri].addr, len); - buf_off += len; - } - } - - // Clear sync so REF doesn't early-exit - @as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0; - - // Run REF - transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4); - - // Compare each region - log.fmt("=== DIAG COMPARE #{d} this=0x{x:0>8} bones={d} regions={d} buf_used={d}", .{ diag_count, this, bc, n_regions, buf_off }); - - // Log ALL game constants that might differ from static analysis - if (diag_count == 0) { - log.fmt(" CONST: s2f=0x{x:0>8} eps1=0x{x:0>8} eps2=0x{x:0>8} h3=0x{x:0>8} h5=0x{x:0>8} c74=0x{x:0>8} cd8=0x{x:0>8}", .{ - hook.readMem(u32, 0x811610), // SHORT_TO_FLOAT - hook.readMem(u32, 0x8029d4), // epsilon 1 - hook.readMem(u32, 0x80c5c8), // epsilon 2 - hook.readMem(u32, 0x80297c), // hermite 3 - hook.readMem(u32, 0x802990), // hermite 5/6 - hook.readMem(u32, 0x7ffd74), // frequent FLD (particle sections) - hook.readMem(u32, 0x7ff9d8), // hermite FADD (particle sections) - }); - } - - ri = 0; - while (ri < n_regions) : (ri += 1) { - const len = regions[ri].len; - const snap_start = region_starts[ri]; - if (snap_start + len <= diag_buf.len) { - diagCompare(regions[ri].label, diag_buf[snap_start .. snap_start + len], regions[ri].addr, len); - } - } - - // TexAnim gate analysis: dump anim_frame_ctr and per-entry alpha kf_count - if (tex_count > 0) { - const tex_data_base = hook.readMem(u32, model_hdr_d + 0x58); - const afc = hook.readMem(u32, this + 0x8C); - log.fmt(" TexAnim gates: anim_frame_ctr={d} tex_count={d}", .{ afc, tex_count }); - var ti: u32 = 0; - while (ti < tex_count and ti < 8) : (ti += 1) { - const td = tex_data_base + ti * 0x38; - const vec3_gate = hook.readMem(u32, td + 0x0C); // Vec3 kf_count - const alpha_gate = hook.readMem(u32, td + 0x28); // alpha kf_count - const alpha_mode = hook.readMem(u16, td + 0x1C); // alpha interp_mode - // Also read what's at the alpha output slot BEFORE REF wrote to it (from snapshot) - const alpha_out_off = ti * 0x50 + 0x3C; // offset within tex_anim_out buffer - // Find tex_anim snapshot - var snap_alpha_orig: u32 = 0xDEAD; - var live_alpha: u32 = 0xDEAD; - if (tex_out != 0 and alpha_out_off + 4 <= @min(tex_count * 0x50, 0x1000)) { - // Find the TexAnim snapshot in diag_buf - var si: u32 = 0; - while (si < n_regions) : (si += 1) { - if (regions[si].addr == tex_out) { - const soff = region_starts[si] + alpha_out_off; - if (soff + 4 <= diag_buf.len) { - snap_alpha_orig = @as(u32, diag_buf[soff]) | (@as(u32, diag_buf[soff + 1]) << 8) | (@as(u32, diag_buf[soff + 2]) << 16) | (@as(u32, diag_buf[soff + 3]) << 24); - } - break; - } - } - live_alpha = hook.readMem(u32, tex_out + alpha_out_off); - } - // Also read the raw short value and the constant at 0x811610 - const alpha_data_base = hook.readMem(u32, td + 0x1C + 0x18); // AD.keyframe_base for alpha - const alpha_idx0 = hook.readMem(u32, tex_out + ti * 0x50 + 0x30); // idx0 from findInterpIdx - const raw_short: i16 = if (alpha_data_base != 0) @as(*align(1) const i16, @ptrFromInt(alpha_data_base + alpha_idx0 * 2)).* else 0; - const s2f_const = hook.readMem(u32, 0x811610); // SHORT_TO_FLOAT constant - log.fmt(" tex[{d}]: vec3_kf={d} alpha_kf={d} mode={d} orig=0x{x:0>8} ref=0x{x:0>8} short={d} s2f=0x{x:0>8}", .{ ti, vec3_gate, alpha_gate, alpha_mode, snap_alpha_orig, live_alpha, raw_short, s2f_const }); - } - } - - diag_count += 1; + } else if (ab_use_custom) { + transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); } else { - // FPU state comparison: capture full x87 state after original vs REF - if (diag_count >= DIAG_MAX and diag_count < DIAG_MAX + 3 and t44_depth == 1 and !is_early) { - // Run original, capture FPU + MXCSR state - transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - var fpu_orig: [108]u8 align(16) = undefined; - var mxcsr_orig: u32 = 0; - asm volatile ("fnsave (%[p])\n\tfrstor (%[p])\n\tstmxcsr (%[m])" - :: [p] "r" (@intFromPtr(&fpu_orig)), [m] "r" (@intFromPtr(&mxcsr_orig)) - : "memory" - ); - - // Clear sync, run REF - @as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0; - transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4); - var fpu_ref: [108]u8 align(16) = undefined; - var mxcsr_ref: u32 = 0; - asm volatile ("fnsave (%[p])\n\tfrstor (%[p])\n\tstmxcsr (%[m])" - :: [p] "r" (@intFromPtr(&fpu_ref)), [m] "r" (@intFromPtr(&mxcsr_ref)) - : "memory" - ); - - // Compare and log FPU state - // FNSAVE layout (108 bytes): CW(4), SW(4), TW(4), IP(4), CS(4), DP(4), DS(4), ST0-ST7(8×10=80) - const cw_o = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + 0)).*; - const sw_o = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + 4)).*; - const tw_o = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + 8)).*; - const cw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 0)).*; - const sw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 4)).*; - const tw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 8)).*; - log.fmt(" FPU orig: CW=0x{x:0>4} SW=0x{x:0>4} TW=0x{x:0>4}", .{ cw_o & 0xFFFF, sw_o & 0xFFFF, tw_o & 0xFFFF }); - log.fmt(" FPU ref: CW=0x{x:0>4} SW=0x{x:0>4} TW=0x{x:0>4}", .{ cw_r & 0xFFFF, sw_r & 0xFFFF, tw_r & 0xFFFF }); - // MXCSR comparison — SSE control/status, never checked before - if (mxcsr_orig != mxcsr_ref) { - log.fmt(" *** MXCSR DIFF: orig=0x{x:0>8} ref=0x{x:0>8} (xor=0x{x:0>8})", .{ mxcsr_orig, mxcsr_ref, mxcsr_orig ^ mxcsr_ref }); - } else { - log.fmt(" MXCSR match: 0x{x:0>8}", .{mxcsr_orig}); - } - // Dump ST0-ST7 (10 bytes each, starting at offset 28) - var sti: u32 = 0; - while (sti < 8) : (sti += 1) { - const base = 28 + sti * 10; - const o0 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + base)).*; - const o1 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_orig) + base + 4)).*; - const o2 = @as(*align(1) const u16, @ptrFromInt(@intFromPtr(&fpu_orig) + base + 8)).*; - const r0 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + base)).*; - const r1 = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + base + 4)).*; - const r2 = @as(*align(1) const u16, @ptrFromInt(@intFromPtr(&fpu_ref) + base + 8)).*; - if (o0 != r0 or o1 != r1 or o2 != r2) { - log.fmt(" ST{d} DIFF: orig={x:0>4}_{x:0>8}_{x:0>8} ref={x:0>4}_{x:0>8}_{x:0>8}", .{ sti, o2, o1, o0, r2, r1, r0 }); - } - } - diag_count += 1; - } else { - // BISECT MODE: REF runs up to bisect_stop_section, then original provides rest - transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4); - if (bisect_stop_section != 0) { - // REF returned early — clear sync so original doesn't early-exit, then run original - @as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0; - transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - } - } + transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4); } t44_depth -|= 1;