diff --git a/build.zig b/build.zig index bcb02ad..c842e1e 100644 --- a/build.zig +++ b/build.zig @@ -34,6 +34,7 @@ const module_list = [_]ModuleDesc{ .{ .name = "filecache", .desc = "Enable MPQ archive file cache" }, .{ .name = "ssemaths", .desc = "Enable UnitXP x87 math polyfill replacements (SSE)", .default = false }, .{ .name = "silicon", .desc = "Enable SSE2 math replacements (ported from libSiliconPatch)", .default = false }, + .{ .name = "performance", .desc = "Enable production SSE hooks (bone, particle, glyph cache)", .default = false }, }; pub fn build(b: *std.Build) void { @@ -55,11 +56,12 @@ pub fn build(b: *std.Build) void { }); const zhook_mod = zhook_dep.module("zhook"); - // Hot math — separate compilation units, always ReleaseFast + // Hot math — separate compilation units, always ReleaseFast. + // Source lives in src/performance/ — the production SSE module. const clip_sse_obj = b.addObject(.{ .name = "clip_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/transform44/clip_sse.zig"), + .root_source_file = b.path("src/performance/clip_sse.zig"), .target = target, .optimize = .ReleaseFast, }), @@ -73,7 +75,7 @@ pub fn build(b: *std.Build) void { const bone_sse_obj = b.addObject(.{ .name = "bone_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/transform44/bone_sse.zig"), + .root_source_file = b.path("src/performance/bone_sse.zig"), .target = bone_sse_target, .optimize = .ReleaseFast, }), @@ -106,7 +108,7 @@ pub fn build(b: *std.Build) void { const silicon_sse_obj = b.addObject(.{ .name = "silicon_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/silicon/silicon_sse.zig"), + .root_source_file = b.path("src/performance/silicon_sse.zig"), .target = bone_sse_target, // SSE4.1+FMA+AVX, same as bone_sse .optimize = .ReleaseFast, }), @@ -115,7 +117,7 @@ pub fn build(b: *std.Build) void { const particle_sse_obj = b.addObject(.{ .name = "particle_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/transform44/particle_sse.zig"), + .root_source_file = b.path("src/performance/particle_sse.zig"), .target = bone_sse_target, .optimize = .ReleaseFast, }), @@ -181,7 +183,7 @@ pub fn build(b: *std.Build) void { const bench_silicon_sse = b.addObject(.{ .name = "bench_silicon_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/silicon/silicon_sse.zig"), + .root_source_file = b.path("src/performance/silicon_sse.zig"), .target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .linux, @@ -193,7 +195,7 @@ pub fn build(b: *std.Build) void { const bench_bone_sse = b.addObject(.{ .name = "bench_bone_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/transform44/bone_sse.zig"), + .root_source_file = b.path("src/performance/bone_sse.zig"), .target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .linux, @@ -217,7 +219,7 @@ pub fn build(b: *std.Build) void { const bench_particle_sse = b.addObject(.{ .name = "bench_particle_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/transform44/particle_sse.zig"), + .root_source_file = b.path("src/performance/particle_sse.zig"), .target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .linux, diff --git a/src/main.zig b/src/main.zig index ef091d7..f944343 100644 --- a/src/main.zig +++ b/src/main.zig @@ -23,6 +23,7 @@ const build_opts = struct { const filecache = @import("build_options").enable_filecache; const ssemaths = @import("build_options").enable_ssemaths; const silicon = @import("build_options").enable_silicon; + const performance = @import("build_options").enable_performance; }; // Conditional module imports @@ -44,6 +45,7 @@ const addonperf = if (build_opts.addonperf) @import("addonperf/addonperf.zig") e const ssemaths = if (build_opts.ssemaths) @import("ssemaths/ssemaths.zig") else struct {}; const file_cache = if (build_opts.filecache) @import("filecache/filecache.zig") else struct {}; const silicon = if (build_opts.silicon) @import("silicon/silicon.zig") else struct {}; +const performance = if (build_opts.performance) @import("performance/performance.zig") else struct {}; const module_active = @import("module_active.zig"); @@ -112,6 +114,8 @@ fn registerLuaFunctions() void { if (build_opts.dpslog) { registerFunction("GetSpellInfo", @intFromPtr(&dpslog.luaGetSpellInfo)); registerFunction("CombatLogGetCurrentEventInfo", @intFromPtr(&dpslog.luaCombatLogGetCurrentEventInfo)); + registerFunction("UnitCastingInfo", @intFromPtr(&dpslog.luaUnitCastingInfo)); + registerFunction("UnitChannelInfo", @intFromPtr(&dpslog.luaUnitChannelInfo)); } if (build_opts.addonperf) { registerFunction("GetAddOnMemoryUsage", @intFromPtr(&addonperf.luaGetAddOnMemoryUsage)); @@ -853,6 +857,7 @@ const modules = [_]ModuleHooks{ if (build_opts.outline) .{ .name = outline.module_name, .remove = outline.cleanup, .is_active = outline.isActive } else .{}, if (build_opts.screenshot) .{ .name = screenshot.module_name, .remove = screenshot.removeHook, .is_active = screenshot.isActive } else .{}, if (build_opts.silicon) .{ .name = silicon.module_name, .install = silicon.installHooks, .remove = silicon.removeHooks, .is_active = silicon.isActive } else .{}, + if (build_opts.performance) .{ .name = performance.module_name, .install = performance.installHooks, .remove = performance.removeHooks, .is_active = performance.isActive } else .{}, }; fn shutdownDetour() callconv(hook.cc.stdcall) void { diff --git a/src/performance/bone_sse.zig b/src/performance/bone_sse.zig new file mode 100644 index 0000000..fccf4f8 --- /dev/null +++ b/src/performance/bone_sse.zig @@ -0,0 +1,2446 @@ +//! SSE-optimized transformMatrix4x4 reimplementation. +//! +//! Full standalone replacement for the 17703-byte bone transform engine at 0x714260. +//! Compiled ReleaseFast even in Debug builds (separate compilation unit pattern). +//! All helper functions (findInterpolationIndices, interpolateAnimationKeyframes, +//! scaleMatrix3x3ByVector, ApplyTranslationMatrix, rotateMatrixByQuaternion) are +//! reimplemented inline — no calls back to original game code. +//! +//! Only external call: the original transformMatrix4x4 via the hook's callOriginal +//! for attachment recursion (the detour auto-dispatches to this SSE version). + +const V4 = @Vector(4, f32); + + + + + + + + + + + +// ============================================================================= +// SceneObject field offsets — assembly-verified from [EBX+N] in transformMatrix4x4 +// ============================================================================= + +const SO = struct { + const model_data_ptr: u32 = 0x010; + const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value + const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header + const sync_value: u32 = 0x040; + const search_data_base: u32 = 0x04C; // prev timestamp for delta + const emitter_flag: u32 = 0x050; + const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array + const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS + const child_padding: u32 = 0x084; + const anim_frame_ctr: u32 = 0x08C; + const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs + const bone_out_ptr: u32 = 0x094; // output bone matrices + const tex_anim_out: u32 = 0x0A0; + const color_anim_out: u32 = 0x0A8; + const scale1: u32 = 0x0AC; + const scale2: u32 = 0x0B0; + const scale3: u32 = 0x0B4; + const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward) + const world_xform: u32 = 0x10C; // float[16] world transform + const field_17c: u32 = 0x17C; + const field_180: u32 = 0x180; + const field_184: u32 = 0x184; + const field_188: u32 = 0x188; + const field_18c: u32 = 0x18C; + const field_190: u32 = 0x190; + const render_scale_x: u32 = 0x194; + const render_scale_y: u32 = 0x198; + const render_scale_z: u32 = 0x19C; + const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children) + const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children) + const hierarchy_ptr: u32 = 0x1C8; + const emitter_ctx: u32 = 0x1CC; + const field_1d8: u32 = 0x1D8; + const hierarchy_idx: u32 = 0x1DC; + const field_200: u32 = 0x200; + const particle1: u32 = 0x3C4; + const particle2: u32 = 0x3C8; + const particle3: u32 = 0x3D0; + const particle4: u32 = 0x3D4; + const add_remaining: u32 = 0x3D8; +}; + +// Bone runtime struct offsets (within 0x118-byte per-bone runtime) +const BR = struct { + // Translation interpolation state + const trans_idx0: u32 = 0x00; // [0] lower keyframe index + const trans_idx1: u32 = 0x04; // [1] upper keyframe index + const trans_t: u32 = 0x08; // [2] interpolation factor (float bits) + const trans_x: u32 = 0x0C; // [3] interpolated translation X + const trans_y: u32 = 0x10; // [4] Y + const trans_z: u32 = 0x14; // [5] Z + // Secondary translation (crossfade) + const trans2_idx0: u32 = 0x18; + const trans2_idx1: u32 = 0x1C; + const trans2_t: u32 = 0x20; + const trans2_x: u32 = 0x24; + const trans2_y: u32 = 0x28; + const trans2_z: u32 = 0x2C; + // Scale interpolation state (at puVar20 + 0x1a = offset 0x68) + const scale_idx0: u32 = 0x68; + const scale_idx1: u32 = 0x6C; + const scale_t: u32 = 0x70; + const scale_x: u32 = 0x74; + const scale_y: u32 = 0x78; + const scale_z: u32 = 0x7C; + const scale2_idx0: u32 = 0x80; + const scale2_idx1: u32 = 0x84; + const scale2_t: u32 = 0x88; + const scale2_x: u32 = 0x8C; + const scale2_y: u32 = 0x90; + const scale2_z: u32 = 0x94; + // Primary animation time range + const prim_time: u32 = 0x98; // puVar20[0x26] + const prim_track: u32 = 0x9C; // puVar20[0x27] + const prim_anim: u32 = 0xA0; // puVar20[0x28] + const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index + // Secondary animation time range (crossfade) + const sec_start: u32 = 0xA8; // puVar20[0x2a] + const sec_end: u32 = 0xAC; // puVar20[0x2b] + const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion + const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e] + // Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30) + const rot_idx0: u32 = 0x30; + const rot_idx1: u32 = 0x34; + const rot_t: u32 = 0x38; + const rot_x: u32 = 0x3C; + const rot_y: u32 = 0x40; + const rot_z: u32 = 0x44; + const rot_w: u32 = 0x48; + // Secondary rotation + const rot2_idx0: u32 = 0x4C; + const rot2_idx1: u32 = 0x50; + const rot2_t: u32 = 0x54; + const rot2_x: u32 = 0x58; + const rot2_y: u32 = 0x5C; + const rot2_z: u32 = 0x60; + const rot2_w: u32 = 0x64; + // Secondary time range + const sec_time: u32 = 0xC4; // puVar20[0x31] + const sec_track: u32 = 0xC8; // puVar20[0x32] + const sec_slot: u32 = 0xD0; // puVar20[0x34] + const sec_start2: u32 = 0xD4; // puVar20[0x35] + const sec_end2: u32 = 0xD8; // puVar20[0x36] + const sec_offset2: u32 = 0xE4; // puVar20[0x39] + // Flags and weights + const flags2: u32 = 0xF4; // puVar20[0x3d] + const crossfade_end: u32 = 0x100; // puVar20[0x40] + const crossfade_inv: u32 = 0x104; // puVar20[0x41] + const crossfade_weight: u32 = 0x108; // puVar20[0x42] + const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade + const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c] +}; + +// OldAnimationBlock struct offsets (28 bytes = 0x1C per track in v256 M2) +// Layout verified from M2 format + decompilation cross-reference: +// pMVar23->m31 (bone_def+0x34) = rot block+0x0C = nTimestamps (gates rotation) +// pMVar23->m12 (bone_def+0x18) = trans block+0x0C = nTimestamps (gates translation) +// pMVar23[1].m10 (bone_def+0x50) = scale block+0x0C = nTimestamps (gates scale) +const AD = struct { + const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp) + const time_index: u32 = 0x02; // i16: global sequence index (-1 = none) + const track_count_flag: u32 = 0x04; // nRanges: 0 = single track + const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs + const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count + const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array + const nvalues: u32 = 0x14; // nValues: number of value entries + const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data +}; + +// M2CompBone struct offsets (0x6C = 108 bytes per bone in v256 model) +// Layout: 12 bytes fixed header + 3x28 byte OldAnimationBlock tracks + 12 bytes pivot +// Track order: translation, rotation, scale (standard M2 order) +const BD = struct { + const key_id: u32 = 0x00; // i32: key bone ID + const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.) + const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes + // Translation OldAnimationBlock (28 bytes, +0x0C to +0x27) + const trans_anim: u32 = 0x0C; + const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation + // Rotation OldAnimationBlock (28 bytes, +0x28 to +0x43) + const rot_anim: u32 = 0x28; + const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation + // Scale OldAnimationBlock (28 bytes, +0x44 to +0x5F) + const scale_anim: u32 = 0x44; + const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation + // Pivot point (12 bytes, +0x60 to +0x6B) + const pivot_x: u32 = 0x60; + const pivot_y: u32 = 0x64; + const pivot_z: u32 = 0x68; +}; + +// Game constants +const ZERO_F: f32 = 0.0; +const ONE_F: f32 = 1.0; +const THREE_F: f32 = 3.0; +// getBillboardEpsilon(): read from game memory (runtime 0x34800000, NOT static 0x3727c5ac from Ghidra) +fn getBillboardEpsilon() f32 { + return rf32(0x008029d4); +} +// getShortToFloat(): read from game memory at 0x00811610 (runtime value is 0x38000100 = 1/32767, +// NOT the static 0x38000000 = 1/32768 from Ghidra). The game patches this at startup. +fn getShortToFloat() f32 { + return rf32(0x00811610); +} +// MSVC CRT sin/cos — linked from the WoW process +extern fn sinf(f32) f32; +extern fn cosf(f32) f32; + +// Original transformMatrix4x4 for recursive attachment calls. +// The hook's detour will auto-dispatch to our SSE version. +const OrigTransformFn = *const fn (u32, u32, u32, u32, u32) callconv(.c) void; + +// ============================================================================= +// Memory access helpers +// ============================================================================= + +inline fn ru32(addr: u32) u32 { + return @as(*const u32, @ptrFromInt(addr)).*; +} +inline fn ri32(addr: u32) i32 { + return @as(*const i32, @ptrFromInt(addr)).*; +} +inline fn rf32(addr: u32) f32 { + return @as(*const f32, @ptrFromInt(addr)).*; +} +inline fn ru16(addr: u32) u16 { + return @as(*align(1) const u16, @ptrFromInt(addr)).*; +} +inline fn ri16(addr: u32) i16 { + return @as(*align(1) const i16, @ptrFromInt(addr)).*; +} +inline fn ru8(addr: u32) u8 { + return @as(*const u8, @ptrFromInt(addr)).*; +} + +inline fn wu32(addr: u32, v: u32) void { + @as(*u32, @ptrFromInt(addr)).* = v; +} +inline fn wf32(addr: u32, v: f32) void { + @as(*f32, @ptrFromInt(addr)).* = v; +} +inline fn wu16(addr: u32, v: u16) void { + @as(*align(1) u16, @ptrFromInt(addr)).* = v; +} +inline fn wu8(addr: u32, v: u8) void { + @as(*u8, @ptrFromInt(addr)).* = v; +} + +inline fn fbits(v: f32) u32 { + return @bitCast(v); +} +inline fn ufloat(v: u32) f32 { + return @bitCast(v); +} + +// ============================================================================= +// Math helpers — using @Vector(4, f32) for SSE +// ============================================================================= + +inline fn splat(v: f32) V4 { + return @splat(v); +} + +/// 3-component lerp: a + (b - a) * t. Uses @mulAdd → vfmadd. +inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 { + return .{ + @mulAdd(f32, rf32(b_addr) - rf32(a_addr), t, rf32(a_addr)), + @mulAdd(f32, rf32(b_addr + 4) - rf32(a_addr + 4), t, rf32(a_addr + 4)), + @mulAdd(f32, rf32(b_addr + 8) - rf32(a_addr + 8), t, rf32(a_addr + 8)), + }; +} + +/// Scale 3x3 rotation portion of a row-major 4x4 matrix by per-axis scale. +/// Row 0 *= scale.x, Row 1 *= scale.y, Row 2 *= scale.z +inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void { + // Row 0 (offsets 0x00, 0x04, 0x08) + wf32(mat + 0x00, rf32(mat + 0x00) * sx); + wf32(mat + 0x04, rf32(mat + 0x04) * sx); + wf32(mat + 0x08, rf32(mat + 0x08) * sx); + // Row 1 (offsets 0x10, 0x14, 0x18) + wf32(mat + 0x10, rf32(mat + 0x10) * sy); + wf32(mat + 0x14, rf32(mat + 0x14) * sy); + wf32(mat + 0x18, rf32(mat + 0x18) * sy); + // Row 2 (offsets 0x20, 0x24, 0x28) + wf32(mat + 0x20, rf32(mat + 0x20) * sz); + wf32(mat + 0x24, rf32(mat + 0x24) * sz); + wf32(mat + 0x28, rf32(mat + 0x28) * sz); +} + +/// Apply translation through rotation matrix: +/// mat[3][0] += dot(mat[0], t) +/// mat[3][1] += dot(mat[1], t) +/// mat[3][2] += dot(mat[2], t) +/// Uses @mulAdd chain → vfmadd for each dot product component. +inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void { + wf32(mat + 0x30, @mulAdd(f32, tz, rf32(mat + 0x20), @mulAdd(f32, ty, rf32(mat + 0x10), @mulAdd(f32, tx, rf32(mat + 0x00), rf32(mat + 0x30))))); + wf32(mat + 0x34, @mulAdd(f32, tz, rf32(mat + 0x24), @mulAdd(f32, ty, rf32(mat + 0x14), @mulAdd(f32, tx, rf32(mat + 0x04), rf32(mat + 0x34))))); + wf32(mat + 0x38, @mulAdd(f32, tz, rf32(mat + 0x28), @mulAdd(f32, ty, rf32(mat + 0x18), @mulAdd(f32, tx, rf32(mat + 0x08), rf32(mat + 0x38))))); +} + +/// Quaternion → rotation matrix as value. No memory writes. +inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 { + const xx2 = qx * (qx + qx); + const xy2 = qx * (qy + qy); + const xz2 = qx * (qz + qz); + const yy2 = qy * (qy + qy); + const yz2 = qy * (qz + qz); + const zz2 = qz * (qz + qz); + const wx2 = qw * (qx + qx); + const wy2 = qw * (qy + qy); + const wz2 = qw * (qz + qz); + + return .{ + 1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0, + xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0, + xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0, + 0, 0, 0, 1, + }; +} + +/// Quaternion → rotation matrix: writes to game memory via u32 address. +/// Used by boneKeyframeLoop where the matrix is in game memory. +inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { + const m = buildRotationMatrixVal(qx, qy, qz, qw); + inline for (0..16) |i| { + wf32(mat + @as(u32, @intCast(i * 4)), m[i]); + } +} + +/// Quaternion → rotation matrix × mat. Fused: builds quat rows as V4, multiplies in-register. +inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { + const xx2 = qx * (qx + qx); + const xy2 = qx * (qy + qy); + const xz2 = qx * (qz + qz); + const yy2 = qy * (qy + qy); + const yz2 = qy * (qz + qz); + const zz2 = qz * (qz + qz); + const wx2 = qw * (qx + qx); + const wy2 = qw * (qy + qy); + const wz2 = qw * (qz + qz); + + // Quat rotation rows as V4 — never touches memory + const q0 = V4{ 1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0 }; + const q1 = V4{ xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0 }; + const q2 = V4{ xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0 }; + + // Load mat rows + const m0 = V4{ rf32(mat), rf32(mat + 4), rf32(mat + 8), rf32(mat + 12) }; + const m1 = V4{ rf32(mat + 16), rf32(mat + 20), rf32(mat + 24), rf32(mat + 28) }; + const m2 = V4{ rf32(mat + 32), rf32(mat + 36), rf32(mat + 40), rf32(mat + 44) }; + const m3 = V4{ rf32(mat + 48), rf32(mat + 52), rf32(mat + 56), rf32(mat + 60) }; + + // result = quat_rot × mat + inline for ([_]struct { q: V4, off: u32 }{ .{ .q = q0, .off = 0 }, .{ .q = q1, .off = 16 }, .{ .q = q2, .off = 32 } }) |r| { + const row = @mulAdd(V4, @as(V4, @splat(r.q[2])), m2, @mulAdd(V4, @as(V4, @splat(r.q[1])), m1, @as(V4, @splat(r.q[0])) * m0)); + wf32(mat + r.off, row[0]); + wf32(mat + r.off + 4, row[1]); + wf32(mat + r.off + 8, row[2]); + wf32(mat + r.off + 12, row[3]); + } + // Row 3 = {0,0,0,1} × mat = m3 (unchanged) + wf32(mat + 48, m3[0]); + wf32(mat + 52, m3[1]); + wf32(mat + 56, m3[2]); + wf32(mat + 60, m3[3]); +} + +/// Copy 4x4 matrix (64 bytes) — 4 V4 loads/stores instead of 16 scalar copies. +inline fn copyMat4(dst: u32, src: u32) void { + inline for (0..4) |i| { + const off: u32 = @intCast(i * 16); + const row = V4{ rf32(src + off), rf32(src + off + 4), rf32(src + off + 8), rf32(src + off + 12) }; + wf32(dst + off, row[0]); + wf32(dst + off + 4, row[1]); + wf32(dst + off + 8, row[2]); + wf32(dst + off + 12, row[3]); + } +} + +/// 4x4 matrix multiply: dst = a * b (row-major). Safe for dst==a or dst==b. +/// Uses V4 + @mulAdd (FMA): 1 mul + 3 FMA per row = 16 SIMD ops total. +inline fn matMul4x4(dst: u32, a: u32, b: u32) void { + // Pre-load all rows of B + const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) }; + const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) }; + const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) }; + const b3 = V4{ rf32(b + 48), rf32(b + 52), rf32(b + 56), rf32(b + 60) }; + // Pre-load all rows of A (in case dst aliases a) + const a0 = V4{ rf32(a), rf32(a + 4), rf32(a + 8), rf32(a + 12) }; + const a1 = V4{ rf32(a + 16), rf32(a + 20), rf32(a + 24), rf32(a + 28) }; + const a2 = V4{ rf32(a + 32), rf32(a + 36), rf32(a + 40), rf32(a + 44) }; + const a3 = V4{ rf32(a + 48), rf32(a + 52), rf32(a + 56), rf32(a + 60) }; + const rows = [4]V4{ a0, a1, a2, a3 }; + // Compute: each output row = broadcast(a[row][col]) * b_row, accumulated with FMA + inline for (0..4) |i| { + const s0: V4 = @splat(rows[i][0]); + const s1: V4 = @splat(rows[i][1]); + const s2: V4 = @splat(rows[i][2]); + const s3: V4 = @splat(rows[i][3]); + const row = @mulAdd(V4, s3, b3, @mulAdd(V4, s2, b2, @mulAdd(V4, s1, b1, s0 * b0))); + const off: u32 = @intCast(i * 16); + wf32(dst + off, row[0]); + wf32(dst + off + 4, row[1]); + wf32(dst + off + 8, row[2]); + wf32(dst + off + 12, row[3]); + } +} + +/// 4x4 matrix multiply: dst = a * b. Left operand is a local array, right is game memory. +inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void { + const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) }; + const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) }; + const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) }; + const b3 = V4{ rf32(b + 48), rf32(b + 52), rf32(b + 56), rf32(b + 60) }; + const rows = [4]V4{ + V4{ a[0], a[1], a[2], a[3] }, + V4{ a[4], a[5], a[6], a[7] }, + V4{ a[8], a[9], a[10], a[11] }, + V4{ a[12], a[13], a[14], a[15] }, + }; + inline for (0..4) |i| { + const s0: V4 = @splat(rows[i][0]); + const s1: V4 = @splat(rows[i][1]); + const s2: V4 = @splat(rows[i][2]); + const s3: V4 = @splat(rows[i][3]); + const row = @mulAdd(V4, s3, b3, @mulAdd(V4, s2, b2, @mulAdd(V4, s1, b1, s0 * b0))); + const off: u32 = @intCast(i * 16); + wf32(dst + off, row[0]); + wf32(dst + off + 4, row[1]); + wf32(dst + off + 8, row[2]); + wf32(dst + off + 12, row[3]); + } +} + +/// In-place multiply: a = a * b (b from game memory). Returns new array. +inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 { + const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) }; + const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) }; + const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) }; + const b3 = V4{ rf32(b + 48), rf32(b + 52), rf32(b + 56), rf32(b + 60) }; + const rows = [4]V4{ + V4{ a[0], a[1], a[2], a[3] }, + V4{ a[4], a[5], a[6], a[7] }, + V4{ a[8], a[9], a[10], a[11] }, + V4{ a[12], a[13], a[14], a[15] }, + }; + var result: [16]f32 = undefined; + inline for (0..4) |i| { + const s0: V4 = @splat(rows[i][0]); + const s1: V4 = @splat(rows[i][1]); + const s2: V4 = @splat(rows[i][2]); + const s3: V4 = @splat(rows[i][3]); + const row = @mulAdd(V4, s3, b3, @mulAdd(V4, s2, b2, @mulAdd(V4, s1, b1, s0 * b0))); + result[i * 4] = row[0]; + result[i * 4 + 1] = row[1]; + result[i * 4 + 2] = row[2]; + result[i * 4 + 3] = row[3]; + } + return result; +} + +/// Set identity matrix (16 floats) +inline fn setIdentity(dst: u32) void { + inline for (0..16) |i| { + const val: f32 = if (i == 0 or i == 5 or i == 10 or i == 15) 1.0 else 0.0; + wf32(dst + @as(u32, @intCast(i)) * 4, val); + } +} + +/// Normalize a 3-component vector in memory at addr. +/// Calls game's vec3 squared magnitude (0x4549F0), then sqrt, epsilon check, divide. +/// Assembly pattern: CALL 0x4549F0 → FSQRT → FABS → FCOMP → FLD1 → FDIVRP → FMUL×3 +inline fn normalizeVec3InPlace(addr: u32) void { + const sq_mag = callVec3SqMag(addr); + const len = @sqrt(sq_mag); + if (@abs(len) >= getBillboardEpsilon()) { + const inv = 1.0 / len; + wf32(addr, rf32(addr) * inv); + wf32(addr + 4, rf32(addr + 4) * inv); + wf32(addr + 8, rf32(addr + 8) * inv); + } +} + +/// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small. +/// Writes vec3 to stack local and calls game's vec3 squared magnitude (0x4549F0). +inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 { + var v: [3]f32 = .{ x, y, z }; + const sq_mag = callVec3SqMag(@intFromPtr(&v)); + const len = @sqrt(sq_mag); + if (len < getBillboardEpsilon()) return .{ x, y, z }; + const inv = 1.0 / len; + return .{ x * inv, y * inv, z * inv }; +} + +/// Cross product of two 3-component vectors +inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 { + return .{ + ay * bz - az * by, + az * bx - ax * bz, + ax * by - ay * bx, + }; +} + +// ============================================================================= +// findInterpolationIndices — reimplemented from 0x713d50 (334 bytes) +// +// Three-tier search with temporal coherence: +// 1. Forward linear scan (hot path, 1-4 iterations typical) +// 2. Backward linear scan (negative delta) +// 3. Binary search (fallback) +// +// Output: indices[0] = lower index, [1] = upper index, [2] = interpolation t (float bits) +// ============================================================================= + +/// IsParticleBufferEmpty (0x7B5F60) — recursive tree check. +/// Returns true if any node in the tree has active particles (this->0x64 != 0). +fn isParticleBufferNotEmpty(ptr: u32) bool { + if (ru32(ptr + 0x64) != 0) return true; + const count = ru32(ptr + 0x7C); + if (count == 0) return false; + const children = ptr + 0x80; + var i: u32 = 0; + while (i < count) : (i += 1) { + if (isParticleBufferNotEmpty(ru32(children + i * 4))) return true; + } + return false; +} + +const InterpResult = struct { + idx0: u32, + idx1: u32, + t: f32, +}; + +/// Check if two AnimData tracks share the same temporal structure, +/// meaning findInterpIdx would produce identical (idx0, idx1, t) for both. +/// Both must use prim_time (time_index == -1) and have matching range/timestamp layout. +inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool { + if (ri16(ref_anim + AD.time_index) != -1) return false; + if (ri16(other_anim + AD.time_index) != -1) return false; + return ru32(ref_anim + AD.track_count_flag) == ru32(other_anim + AD.track_count_flag) and + ru32(ref_anim + AD.keyframe_ranges) == ru32(other_anim + AD.keyframe_ranges) and + ru32(ref_anim + AD.timestamps_ptr) == ru32(other_anim + AD.timestamps_ptr) and + ru32(ref_anim + AD.keyframe_count) == ru32(other_anim + AD.keyframe_count); +} + +/// Write temporal coherence cache for a reused result so next frame's forward scan starts right. +inline fn applyCachedResult(cached: InterpResult, output: u32) void { + wu32(output, cached.idx0); +} + +/// findInterpIdx: temporal-coherence keyframe search. +/// Reimplementation of game function at 0x713D50 (334 bytes). +/// Assembly-verified against t44_helpers_asm.txt. +/// Returns indices and t in registers; only writes output[0] for next-frame cache persistence. +inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult { + const n_ranges = ru32(anim_data + AD.track_count_flag); + + // Range selection: [start, last] not [start, count] + var range_start: u32 = undefined; + var range_last: u32 = undefined; + if (n_ranges != 0) { + const ranges = ru32(anim_data + AD.keyframe_ranges); + range_last = ru32(ranges + track_index * 8 + 4); + range_start = ru32(ranges + track_index * 8); + } else { + range_last = ru32(anim_data + AD.keyframe_count) -% 1; + range_start = 0; + } + + if (range_start >= range_last) { + wu32(output, range_start); + return .{ .idx0 = range_start, .idx1 = range_start, .t = 0.0 }; + } + + // Global sequence override: CMP AX,0xFFFF + const time_idx_raw = ri16(anim_data + AD.time_index); + const search: u32 = if (time_idx_raw != -1) blk: { + const gs_vals = ru32(this + SO.gs_values_ptr); + break :blk ru32(gs_vals + @as(u32, @intCast(@as(u16, @bitCast(time_idx_raw)))) * 4); + } else search_value; + + const ts_base = ru32(anim_data + AD.timestamps_ptr); + const cached = ru32(output); + const ts_cached = ru32(ts_base + cached * 4); + const delta: u32 = search -% ts_cached; + + var result: u32 = undefined; + + if (delta < 0x1F4) { + // Forward scan from cached + result = cached; + if (result < range_last) { + var ptr = ts_base + result * 4 + 4; + while (ru32(ptr) <= search) { + result += 1; + ptr += 4; + if (result >= range_last) break; + } + } + } else if (delta >= 0xFFFFFE0C) { + // Backward scan from cached + result = cached; + if (result > range_start) { + var ptr = ts_base + result * 4; + while (ru32(ptr) > search) { + result -= 1; + ptr -= 4; + if (result <= range_start) break; + } + } + } else { + // Check delta from range_start + const ts_first = ru32(ts_base + range_start * 4); + const delta_first: u32 = search -% ts_first; + if (delta_first < 0x1F4) { + result = range_start; + var ptr = ts_base + range_start * 4 + 4; + while (ru32(ptr) <= search) { + result += 1; + ptr += 4; + if (result >= range_last) break; + } + } else { + // Binary search + var lo = range_start; + var hi = range_last; + while (lo < hi) { + const mid = (hi +% lo) >> 1; + if (search < ru32(ts_base + mid * 4)) { + hi = mid -% 1; + } else { + if (search < ru32(ts_base + mid * 4 + 4)) { + lo = mid; + break; + } + lo = mid + 1; + } + } + result = lo; + } + } + + // Post-search: bounds check against total keyframe_count + const kf_count = ru32(anim_data + AD.keyframe_count); + const next = result + 1; + + if (next >= kf_count) { + wu32(output, result); + return .{ .idx0 = result, .idx1 = result, .t = 0.0 }; + } + + // Interpolation factor: FILD qword / FIDIV dword + const ts_lo = ru32(ts_base + result * 4); + const ts_hi = ru32(ts_base + next * 4); + const numer = search -% ts_lo; + const denom = ts_hi -% ts_lo; + const t: f32 = @as(f32, @floatFromInt(numer)) / @as(f32, @floatFromInt(@as(i32, @bitCast(denom)))); + + wu32(output, result); + return .{ .idx0 = result, .idx1 = next, .t = t }; +} + +// ============================================================================= +// interpolateAnimationKeyframes — reimplemented from 0x713ea0 +// +// Calls findInterpIdx, does 4-component lerp (for quaternions). +// If crossfade active, does secondary lookup + blend. +// Output buffer layout: [idx0, idx1, t, x, y, z, w, sec_idx0, sec_idx1, sec_t, sx, sy, sz, sw] +// ============================================================================= + +/// Quaternion keyframe interpolation — replaces game's 0x713EA0. +/// Assembly-verified: stride 16 (SHL EAX,4), values are 4×float, not CompQuat. +inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 { + return interpAnimKFCached(this, bone_rt, anim_data, output, null); +} + +/// Quaternion keyframe interpolation with optional cached primary InterpResult. +/// When cached_primary is non-null, skips findInterpIdx and uses the cached indices/t. +inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 { + const r = if (cached_primary) |c| blk: { + applyCachedResult(c, output); + break :blk c; + } else findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); + const mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + var result: [4]f32 = undefined; + + if (mode == 0) { + const src = kf_base + r.idx0 * 16; + inline for (0..4) |i| { + result[i] = rf32(src + @as(u32, @intCast(i * 4))); + } + } else { + const src0 = kf_base + r.idx0 * 16; + const src1 = kf_base + r.idx1 * 16; + inline for (0..4) |i| { + const off: u32 = @intCast(i * 4); + result[i] = @mulAdd(f32, rf32(src1 + off) - rf32(src0 + off), r.t, rf32(src0 + off)); + } + + // Crossfade — blend in registers, no re-read from output buffer + if (rf32(bone_rt + BR.blend_weight) != 0.0 and ri16(anim_data + AD.time_index) == -1) { + const sr = findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x1C); + const ssrc0 = kf_base + sr.idx0 * 16; + const ssrc1 = kf_base + sr.idx1 * 16; + const bw = rf32(bone_rt + BR.blend_weight); + inline for (0..4) |i| { + const off: u32 = @intCast(i * 4); + const sec = @mulAdd(f32, rf32(ssrc1 + off) - rf32(ssrc0 + off), sr.t, rf32(ssrc0 + off)); + wf32(output + 0x28 + off, sec); + result[i] = @mulAdd(f32, sec - result[i], bw, result[i]); + } + } + } + + // Write final result to memory for persistence (read on frames where gate is false) + inline for (0..4) |i| { + wf32(output + 0x0C + @as(u32, @intCast(i * 4)), result[i]); + } + return result; +} + +// ============================================================================= +// Game function call wrappers — replacing reimplementations with actual calls +// ============================================================================= + +/// Fast modulo for looping animations. The value is almost always < 2*length +/// (frame-to-frame delta is small), so a conditional subtract beats idiv. +inline fn fastMod(val: u32, len: u32) u32 { + var v = val; + if (v >= len) { + v -%= len; + if (v >= len) v = v % len; // fallback for large time skips + } + return v; +} + +/// Float truncation — replaces game's __ftol at 0x40A2B0. +/// Original: FILD i32 → FMUL f32 → __ftol, all in 80-bit x87 precision. +inline fn callFtol(delta: i32, scale_addr: u32) i32 { + return @intFromFloat(@as(f32, @floatFromInt(delta)) * rf32(scale_addr)); +} + +/// Vec3 squared magnitude — replaces game's 0x4549F0. Uses @mulAdd → vfmadd. +inline fn callVec3SqMag(vec3_ptr: u32) f32 { + const x = rf32(vec3_ptr); + const y = rf32(vec3_ptr + 4); + const z = rf32(vec3_ptr + 8); + return @mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x)); +} + +/// Read i16 at keyframe index. Replaces game's getIndexOffset (0x71AFF0) + setShortValue (0x71B010). +/// getIndexOffset returns table[4] + index*2, setShortValue copies a word. Direct read is equivalent. +inline fn readShortViaGame(table: u32, index: u32) i16 { + const values_ptr = ru32(table + 4); + return ri16(values_ptr + index * 2); +} + +/// Interpolate a Vec3 track (12 bytes per keyframe) with crossfade support. +/// Writes result to output[3..5] (as u32 float bits). Uses output[0..2] for indices/t, +/// and output[6..11] for secondary crossfade state. +inline fn interpVec3Track( + this: u32, + bone_rt: u32, + anim_data: u32, + output: u32, + blend_weight: f32, +) [3]f32 { + return interpVec3TrackCached(this, bone_rt, anim_data, output, blend_weight, null); +} + +/// Vec3 keyframe interpolation with optional cached primary InterpResult. +inline fn interpVec3TrackCached( + this: u32, + bone_rt: u32, + anim_data: u32, + output: u32, + blend_weight: f32, + cached_primary: ?InterpResult, +) [3]f32 { + const r = if (cached_primary) |c| blk: { + applyCachedResult(c, output); + break :blk c; + } else findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); + + const interp_mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + var result: [3]f32 = undefined; + + if (interp_mode == 0) { + const src = kf_base + r.idx0 * 0xC; + result = .{ rf32(src), rf32(src + 4), rf32(src + 8) }; + } else { + result = lerpVec3(kf_base + r.idx0 * 0xC, kf_base + r.idx1 * 0xC, r.t); + + // Crossfade — blend in registers + if (blend_weight != 0.0 and ri16(anim_data + AD.time_index) == -1) { + const sr = findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x18); + const sec = lerpVec3(kf_base + sr.idx0 * 0xC, kf_base + sr.idx1 * 0xC, sr.t); + wu32(output + 0x24, fbits(sec[0])); + wu32(output + 0x28, fbits(sec[1])); + wu32(output + 0x2C, fbits(sec[2])); + inline for (0..3) |i| { + result[i] = @mulAdd(f32, sec[i] - result[i], blend_weight, result[i]); + } + } + } + + // Write final result to memory for persistence + wu32(output + 0x0C, fbits(result[0])); + wu32(output + 0x10, fbits(result[1])); + wu32(output + 0x14, fbits(result[2])); + return result; +} + +/// Interpolate a single float track (4 bytes per keyframe) with crossfade. +/// Writes result to output[3] as float bits. +inline fn interpFloatTrack( + this: u32, + bone_rt: u32, + anim_data: u32, + output: u32, + blend_weight: f32, +) f32 { + const r = findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output); + + const interp_mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + var result: f32 = undefined; + + if (interp_mode == 0) { + result = rf32(kf_base + r.idx0 * 4); + } else { + const a = rf32(kf_base + r.idx0 * 4); + const b = rf32(kf_base + r.idx1 * 4); + result = @mulAdd(f32, b - a, r.t, a); + + // Crossfade — blend in registers + if (blend_weight != 0.0 and ri16(anim_data + AD.time_index) == -1) { + const sr = findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x10); + const sa = rf32(kf_base + sr.idx0 * 4); + const sb = rf32(kf_base + sr.idx1 * 4); + const sec = (sb - sa) * sr.t + sa; + wu32(output + 0x1C, fbits(sec)); + result = @mulAdd(f32, sec - result, blend_weight, result); + } + } + + wf32(output + 0x0C, result); + return result; +} + +// ============================================================================= +// Hermite/Bezier basis + particle emitter interp helpers +// ============================================================================= + +inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { + const t2 = t * t; + const t3 = t2 * t; + return .{ + .h1 = 2 * t3 - 3 * t2 + 1, + .h2 = t3 - 2 * t2 + t, + .h3 = -2 * t3 + 3 * t2, + .h4 = t3 - t2, + }; +} + +inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { + const u = 1.0 - t; + const t2 = t * t; + const u_sq = u * u; + return .{ + .b0 = u_sq * u, + .b1 = 3 * u_sq * t, + .b2 = 3 * u * t2, + .b3 = t2 * t, + }; +} + +inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { + const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); + + const mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + if (mode == 0) { + const src = kf_base + r.idx0 * 36; + wu32(output + 0x0C, ru32(src)); + wu32(output + 0x10, ru32(src + 4)); + wu32(output + 0x14, ru32(src + 8)); + return; + } + + const kf_a = kf_base + r.idx0 * 36; + const kf_b = kf_base + r.idx1 * 36; + + if (mode == 1) { + const result = lerpVec3(kf_a, kf_b, r.t); + wu32(output + 0x0C, fbits(result[0])); + wu32(output + 0x10, fbits(result[1])); + wu32(output + 0x14, fbits(result[2])); + } else if (mode == 3) { + const h = hermiteBasis(r.t); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + wf32(output + 0x0C + off, h.h1 * rf32(kf_a + off) + h.h2 * rf32(kf_a + 0x18 + off) + h.h3 * rf32(kf_b + off) + h.h4 * rf32(kf_b + 0x0C + off)); + } + } else if (mode == 2) { + const b = bezierBasis(r.t); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + wf32(output + 0x0C + off, b.b0 * rf32(kf_a + off) + b.b1 * rf32(kf_a + 0x18 + off) + b.b2 * rf32(kf_b + 0x0C + off) + b.b3 * rf32(kf_b + off)); + } + } else {} // Unknown mode: skip primary interp, fall through to crossfade (asm 0x716B98: JNZ crossfade_check) + + const blend = rf32(bone_rt_base + BR.blend_weight); + if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { + const sr = findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x18); + + const skf_a = kf_base + sr.idx0 * 36; + const skf_b = kf_base + sr.idx1 * 36; + const smode = ri16(anim_data + AD.interp_mode); + + if (smode == 1) { + const sec = lerpVec3(skf_a, skf_b, sr.t); + wu32(output + 0x24, fbits(sec[0])); + wu32(output + 0x28, fbits(sec[1])); + wu32(output + 0x2C, fbits(sec[2])); + } else if (smode == 3) { + const h = hermiteBasis(sr.t); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + wf32(output + 0x24 + off, h.h1 * rf32(skf_a + off) + h.h2 * rf32(skf_a + 0x18 + off) + h.h3 * rf32(skf_b + off) + h.h4 * rf32(skf_b + 0x0C + off)); + } + } else if (smode == 2) { + const b = bezierBasis(sr.t); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + wf32(output + 0x24 + off, b.b0 * rf32(skf_a + off) + b.b1 * rf32(skf_a + 0x18 + off) + b.b2 * rf32(skf_b + 0x0C + off) + b.b3 * rf32(skf_b + off)); + } + } else { + wu32(output + 0x24, ru32(skf_a)); + wu32(output + 0x28, ru32(skf_a + 4)); + wu32(output + 0x2C, ru32(skf_a + 8)); + } + + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + const pri = rf32(output + 0x0C + off); + const sec = rf32(output + 0x24 + off); + wf32(output + 0x0C + off, (sec - pri) * blend + pri); + } + } +} + +inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { + const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); + + const mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + if (mode == 0) { + wu32(output + 0x0C, ru32(kf_base + r.idx0 * 12)); + return; + } + + const kf_a = kf_base + r.idx0 * 12; + const kf_b = kf_base + r.idx1 * 12; + + if (mode == 1) { + const a = rf32(kf_a); + const b = rf32(kf_b); + wf32(output + 0x0C, @mulAdd(f32, b - a, r.t, a)); + } else if (mode == 3) { + const h = hermiteBasis(r.t); + wf32(output + 0x0C, h.h1 * rf32(kf_a) + h.h2 * rf32(kf_a + 0x08) + h.h3 * rf32(kf_b) + h.h4 * rf32(kf_b + 0x04)); + } else if (mode == 2) { + const b = bezierBasis(r.t); + wf32(output + 0x0C, b.b0 * rf32(kf_a) + b.b1 * rf32(kf_a + 0x08) + b.b2 * rf32(kf_b + 0x04) + b.b3 * rf32(kf_b)); + } else {} // Unknown mode: skip primary interp, fall through to crossfade + + const blend = rf32(bone_rt_base + BR.blend_weight); + if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { + const sr = findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); + + const skf_a = kf_base + sr.idx0 * 12; + const skf_b = kf_base + sr.idx1 * 12; + const smode = ri16(anim_data + AD.interp_mode); + + var sec: f32 = undefined; + if (smode == 1) { + sec = (rf32(skf_b) - rf32(skf_a)) * sr.t + rf32(skf_a); + } else if (smode == 3) { + const h = hermiteBasis(sr.t); + sec = h.h1 * rf32(skf_a) + h.h2 * rf32(skf_a + 0x08) + h.h3 * rf32(skf_b) + h.h4 * rf32(skf_b + 0x04); + } else if (smode == 2) { + const bz = bezierBasis(sr.t); + sec = bz.b0 * rf32(skf_a) + bz.b1 * rf32(skf_a + 0x08) + bz.b2 * rf32(skf_b + 0x04) + bz.b3 * rf32(skf_b); + } else { + sec = rf32(skf_a); + } + wf32(output + 0x1C, sec); + const pri = rf32(output + 0x0C); + wf32(output + 0x0C, (sec - pri) * blend + pri); + } +} + +// ============================================================================= +// getInterpolatedFloat — reimplemented from 0x71af20 +// Same as interpFloatTrack but uses the bone_rt directly (different register mapping) +// ============================================================================= + +inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void { + const r = findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data_short_ptr, output); + + const interp_mode = ri16(anim_data_short_ptr); + const kf_base = ru32(anim_data_short_ptr + 0x18); + + if (interp_mode == 0) { + wu32(output + 0x0C, ru32(kf_base + r.idx0 * 4)); + return; + } + + const a = rf32(kf_base + r.idx0 * 4); + const b = rf32(kf_base + r.idx1 * 4); + wf32(output + 0x0C, @mulAdd(f32, b - a, r.t, a)); + + const blend = rf32(bone_rt_addr + 0x10C); + if (blend != 0.0 and ri16(anim_data_short_ptr + 2) == -1) { + const sr = findInterpIdx(this, ru32(bone_rt_addr + 0xC4), ru32(bone_rt_addr + 0xC8), anim_data_short_ptr, output + 0x10); + const sa = rf32(kf_base + sr.idx0 * 4); + const sb = rf32(kf_base + sr.idx1 * 4); + const sec = (sb - sa) * sr.t + sa; + wu32(output + 0x1C, fbits(sec)); + const pri = ufloat(ru32(output + 0x0C)); + wf32(output + 0x0C, (sec - pri) * blend + pri); + } +} + +// ============================================================================= +// calculateScaledInverseMatrix — reimplemented from 0x7bd820 +// Used for billboarding. Transposes 3x3 rotation, scales by 1/scale^2, +// applies inverse translation. +// ============================================================================= + +fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void { + // Simple transpose for unit scale + if (@abs(scale - 1.0) < @as(f32, @bitCast(@as(u32, 0x35800000)))) { + // Transpose 3x3 + wf32(out + 0x00, rf32(this_mat + 0x00)); + wf32(out + 0x04, rf32(this_mat + 0x10)); + wf32(out + 0x08, rf32(this_mat + 0x20)); + wf32(out + 0x0C, 0); + wf32(out + 0x10, rf32(this_mat + 0x04)); + wf32(out + 0x14, rf32(this_mat + 0x14)); + wf32(out + 0x18, rf32(this_mat + 0x24)); + wf32(out + 0x1C, 0); + wf32(out + 0x20, rf32(this_mat + 0x08)); + wf32(out + 0x24, rf32(this_mat + 0x18)); + wf32(out + 0x28, rf32(this_mat + 0x28)); + wf32(out + 0x2C, 0); + wf32(out + 0x30, 0); + wf32(out + 0x34, 0); + wf32(out + 0x38, 0); + wf32(out + 0x3C, @as(f32, @bitCast(@as(u32, 0x3f800000)))); + // Apply inverse translation + applyTranslation(out, -rf32(this_mat + 0x30), -rf32(this_mat + 0x34), -rf32(this_mat + 0x38)); + return; + } + + // Transpose 3x3 portion + wf32(out + 0x00, rf32(this_mat + 0x00)); + wf32(out + 0x04, rf32(this_mat + 0x10)); + wf32(out + 0x08, rf32(this_mat + 0x20)); + wf32(out + 0x0C, 0); + wf32(out + 0x10, rf32(this_mat + 0x04)); + wf32(out + 0x14, rf32(this_mat + 0x14)); + wf32(out + 0x18, rf32(this_mat + 0x24)); + wf32(out + 0x1C, 0); + wf32(out + 0x20, rf32(this_mat + 0x08)); + wf32(out + 0x24, rf32(this_mat + 0x18)); + wf32(out + 0x28, rf32(this_mat + 0x28)); + wf32(out + 0x2C, 0); + wf32(out + 0x30, 0); + wf32(out + 0x34, 0); + wf32(out + 0x38, 0); + wf32(out + 0x3C, @as(f32, @bitCast(@as(u32, 0x3f800000)))); + + // Scale by 1/(scale^2) + const inv_s2 = 1.0 / (scale * scale); + scaleMatrix3x3(out, inv_s2, inv_s2, inv_s2); + + // Apply inverse translation + applyTranslation(out, -rf32(this_mat + 0x30), -rf32(this_mat + 0x34), -rf32(this_mat + 0x38)); +} + +// ============================================================================= +// Main export: transformMatrix4x4_REF +// +// Calling convention: x86_thiscall — matches the original at 0x714260 exactly. +// ECX=this, stack: mat1..mat4, callee cleans RET 0x10. +// +// Params: this_ptr(ECX), mat1(parent_matrix*), mat2(position_vec3*), +// mat3(offset_vec3*), mat4(scale_float_bits) +// ============================================================================= + +export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void { + + @setEvalBranchQuota(50000); + // ========================================================================= + // Section 1: Entry checks + // ========================================================================= + if (ru32(this + SO.model_data_ptr) == 0) return; + const anim_ctx = ru32(this + SO.anim_ctx_ptr); + if (ru32(this + SO.sync_value) == ru32(anim_ctx + 0x10)) return; + + + // ========================================================================= + // Section 2: Emitter setup + // ========================================================================= + const model_ctr = ru32(this + SO.model_ctr_ptr); + const model_hdr = ru32(model_ctr + 0x130); + const emitter_ctx = ru32(this + SO.emitter_ctx); + + if (emitter_ctx != 0) { + // Assembly 0x71429E-0x7142C1: emitter_ctx+0x50 != 0 AND this+0x1D8 != 0 + const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0; + wu32(this + 0x50, has_emitter); // emitter_enable_flag + wu32(this + 0x17C, ru32(emitter_ctx + 0x17C)); + } + + // ========================================================================= + // Section 3: World position/scale + // ========================================================================= + const pos_ptr = mat2; // position input Vec3 + const ofs_ptr = mat3; // offset input Vec3 + const scale_f: f32 = @bitCast(mat4); // float scale + + // world_pos = pos * per_axis_scale + wf32(this + SO.world_pos + 0, rf32(pos_ptr) * rf32(this + SO.field_184)); + wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(pos_ptr + 4)); + wf32(this + SO.world_pos + 8, @bitCast(fbits(rf32(this + SO.field_18c) * rf32(pos_ptr + 8)))); + + // render_pri = offset + existing fields + const rp0 = rf32(ofs_ptr) + rf32(this + SO.field_190); + const rp1 = rf32(this + SO.render_scale_x) + rf32(ofs_ptr + 4); + const rp2 = rf32(this + SO.render_scale_y) + rf32(ofs_ptr + 8); + wf32(this + SO.render_pri + 0, rp0); + wf32(this + SO.render_pri + 4, rp1); + wf32(this + SO.render_pri + 8, rp2); + + // render_scale_z = scale * field_180 + wf32(this + SO.render_scale_z, scale_f * rf32(this + SO.field_180)); + + // ========================================================================= + // Section 4: Global sequence processing + // ========================================================================= + const gs_count = ru32(model_hdr + 0x14); + if (gs_count != 0) { + const gs_durations = ru32(model_hdr + 0x18); + const gs_values = ru32(this + SO.gs_values_ptr); + const timestamp = ru32(anim_ctx + 0x0C); + const time_base = ru32(this + SO.gs_time_base); + var gi: u32 = 0; + while (gi < gs_count) : (gi += 1) { + const dur = ru32(gs_durations + gi * 4); + if (dur == 0) { + wu32(gs_values + gi * 4, 0); + } else { + wu32(gs_values + gi * 4, (timestamp -% time_base) % dur); + } + } + } + + // initParticlePixelShaderGeneration (0x74a7c0) — matrix multiply via JMP table. + // Computes: *(this+0xFC) = *(this+0xBC) × mat1 + // Assembly: PUSH mat1, PUSH &0xBC, PUSH &0xFC, CALL 0x74A7C0 + // 0x74A7C0 = JMP [0x876504] → runtime target (0x754A66 SSE version) + // Must call through 0x74A7C0, NOT 0x7507BB directly. + matMul4x4(this + 0xFC, this + 0xBC, mat1); + + // ========================================================================= + // Section 5: child_objects_padding (len_sq of world transform translation) + // Assembly re-reads emitter_ctx from this+0x1CC AFTER matMul (0x7143A0). + // ========================================================================= + const emitter_ctx_5 = ru32(this + SO.emitter_ctx); + if (emitter_ctx_5 == 0 or (ru8(emitter_ctx_5 + 4) & 1) != 0) { + const wx = rf32(this + SO.world_xform + 8 * 4); // [8] + const wy = rf32(this + SO.world_xform + 9 * 4); // [9] + const wz = rf32(this + SO.world_xform + 10 * 4); // [10] + wu32(this + SO.child_padding, fbits(wx * wx + wy * wy + wz * wz)); + } else { + wu32(this + SO.child_padding, ru32(emitter_ctx_5 + 0x84)); + } + + // ========================================================================= + // Section 6: Identity matrix init + timestamp delta + // ========================================================================= + var local_mat: [16]f32 = .{ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + }; + const local_mat_addr = @intFromPtr(&local_mat); + + // Secondary identity (3x4 portion for the second matrix in decompilation) + var local_mat2: [16]f32 = .{ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + }; + + // Timestamp delta tracking + // Assembly (0x7143EE-0x71451C): outer guard is this+0x4C != 0 (NOT anim_ctx). + // If stored value is 0, does NOTHING — never writes, never computes delta. + // Something else must initialize this+0x4C; we must NOT seed it ourselves. + var time_delta_val: u32 = 0; + const sdb = ru32(this + SO.search_data_base); + if (sdb != 0) { + const cur_ts = ru32(anim_ctx + 0x0C); + if (cur_ts != 0) { + time_delta_val = cur_ts -% sdb; + wu32(this + SO.search_data_base, cur_ts); + } + } + + // ========================================================================= + // Section 7: Main bone loop + // ========================================================================= + const bone_count = ru32(model_hdr + 0x34); + const bone_defs = ru32(model_hdr + 0x38); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const bone_out_base = ru32(this + SO.bone_out_ptr); + const frame_ctr = ru32(this + SO.anim_frame_ctr); + + if (bone_count != 0) { + var bone_idx: u32 = 0; + var bdef = bone_defs; + var brt = bone_rt_base; + while (bone_idx < bone_count) : ({ + bone_idx += 1; + bdef += 0x6C; + brt += 0x118; + }) { + const flags = ru32(bdef + BD.flags); + const parent_idx_raw: i32 = @as(i32, @intCast(@as(i16, @bitCast(ru16(bdef + BD.parent_bone))))); + + // --- Animation time computation --- + // (Handle primary and secondary animation slot timing) + const anim_slot_val = ri32(brt + BR.anim_slot); + if (anim_slot_val == -1) { + // Inherit from parent bone + if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { + const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118; + wu32(brt + BR.prim_time, ru32(parent_rt + BR.prim_time)); + wu32(brt + BR.prim_track, ru32(parent_rt + BR.prim_track)); + wu32(brt + BR.prim_anim, ru32(parent_rt + BR.prim_anim)); + } else if (bone_idx != 0) { + wu32(brt + BR.prim_time, ru32(bone_rt_base + BR.prim_time)); + wu32(brt + BR.prim_track, ru32(bone_rt_base + BR.prim_track)); + wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim)); + } + } else { + // Has own animation slot — compute time from animation lookup table. + // Assembly at 0x714561-0x71464E, verified line by line. + if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0 + // Add time delta to sec_start/sec_end + wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val); // [ESI+0xA8] + wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val); // [ESI+0xAC] + } + + // anim_entry = anim_lookup_table + anim_slot * 0x44 + const anim_lookup = ru32(model_hdr + 0x20); // [EDX+0x20] + const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44; + const cur_time = ru32(ru32(this + 0x2C) + 0xC); // [EBX+0x2C]+0xC = timestamp + + // Check looping flag: [anim_entry+0x10] & 1 + if ((ru8(anim_entry + 0x10) & 1) == 0) { + // Looping: assembly at 0x7145F1-0x714631 + const anim_end = ru32(anim_entry + 0x08); + const anim_start = ru32(anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + // elapsed = (float)(cur_time - sec_start) * time_scale → __ftol + const delta = cur_time -% ru32(brt + 0xA8); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start); + wu32(brt + 0x98, anim_start +% frame); // prim_time + } else { + // Assembly 0x714631: MOV EDX,EAX — fallback to anim_start + wu32(brt + 0x98, anim_start); + } + } else { + // Clamped: assembly at 0x71458E-0x7145E3 + const sec_end_val = ru32(brt + 0xAC); + const sec_start_val = ru32(brt + 0xA8); + + // Check if sec_end has passed (sec_end - cur_time <= 0 signed) + if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) { + // sec_end hasn't passed yet + // Assembly 0x7145E5: clamp cur_time to sec_start if sec_start > cur_time + const effective_time = if (@as(i32, @bitCast(sec_start_val -% cur_time)) > 0) sec_start_val else cur_time; + // goto looping path + const anim_end = ru32(anim_entry + 0x08); + const anim_start = ru32(anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = effective_time -% ru32(brt + 0xA8); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start); + wu32(brt + 0x98, anim_start +% frame); + } else { + wu32(brt + 0x98, anim_start); + } + } else { + // sec_end has passed — compute clamped position + // Assembly at 0x71458E-0x7145E3: + // delta = (sec_end - sec_start), scaled by [ESI+0xB0] + const dur = sec_end_val -% sec_start_val; + const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xB0); + const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8))); + + if (offset < 0) { + // Clamp to anim_start + wu32(brt + 0x98, ru32(anim_entry + 0x04)); + } else { + const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08))); + const anim_start_i = @as(i32, @bitCast(ru32(anim_entry + 0x04))); + if (offset <= anim_end_i - anim_start_i) { + wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i))); + } else { + // Clamp to anim_end + wu32(brt + 0x98, ru32(anim_entry + 0x08)); + } + } + } + } + + // Store results: assembly at 0x714633-0x71464E + wu32(brt + 0x9C, ru32(brt + 0xA4)); // prim_track = anim_slot + // prim_time already set above + wu32(brt + 0xA0, bone_idx); // prim_anim = bone_idx + } + + // --- Secondary animation time (crossfade target) --- + // Similar pattern for the secondary/blend animation slot + const sec_slot_val = ri32(brt + BR.sec_slot); + if (sec_slot_val == -1) { + if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { + const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118; + wu32(brt + BR.sec_time, ru32(parent_rt + BR.sec_time)); + wu32(brt + BR.sec_track, ru32(parent_rt + BR.sec_track)); + } else if (bone_idx != 0) { + wu32(brt + BR.sec_time, ru32(bone_rt_base + BR.sec_time)); + wu32(brt + BR.sec_track, ru32(bone_rt_base + BR.sec_track)); + } else { + wu32(brt + BR.sec_time, ru32(brt + BR.prim_time)); + wu32(brt + BR.sec_track, ru32(brt + BR.prim_track)); + } + } else { + // Secondary animation slot time computation. + // Assembly at 0x7146C1-0x7147C3, mirrors primary slot logic. + if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0 + wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val); // [ESI+0xD4] + wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val); // [ESI+0xD8] + } + + const sec_anim_lookup = ru32(model_hdr + 0x20); + const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44; + const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC); + + if ((ru8(sec_anim_entry + 0x10) & 1) == 0) { + // Looping + const anim_end = ru32(sec_anim_entry + 0x08); + const anim_start = ru32(sec_anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = sec_cur_time -% ru32(brt + 0xD4); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start); + wu32(brt + 0xC4, anim_start +% frame); // sec_time + } else { + wu32(brt + 0xC4, anim_start); + } + } else { + // Clamped + const sec_end_val = ru32(brt + 0xD8); + const sec_start_val = ru32(brt + 0xD4); + + if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) { + // Assembly 0x71474B: clamp sec_cur_time to sec_start if sec_start > sec_cur_time + const effective_time = if (@as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) sec_start_val else sec_cur_time; + const anim_end = ru32(sec_anim_entry + 0x08); + const anim_start = ru32(sec_anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = effective_time -% ru32(brt + 0xD4); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start); + wu32(brt + 0xC4, anim_start +% frame); + } else { + wu32(brt + 0xC4, anim_start); + } + } else { + const dur = sec_end_val -% sec_start_val; + const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xDC); + const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4))); + + if (offset < 0) { + wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04)); + } else { + const anim_end_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x08))); + const anim_start_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x04))); + if (offset <= anim_end_i - anim_start_i) { + wu32(brt + 0xC4, @as(u32, @bitCast(offset + anim_start_i))); + } else { + wu32(brt + 0xC4, ru32(sec_anim_entry + 0x08)); + } + } + } + } + + // Store results: assembly at 0x714799-0x7147C3 + wu32(brt + 0xC8, ru32(brt + 0xD0)); // sec_track = sec_slot + // sec_time already set above + + // Check expiry: if (timestamp - crossfade_end >= 0) expire slot + if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) { + wu32(brt + 0xD0, 0xFFFFFFFF); // expire secondary slot + } + } + + // --- Blend weight (crossfade Hermite interpolation) --- + if (ri32(brt + BR.anim_slot) == -1 and ri32(brt + BR.sec_slot) == -1) { + // Inherit blend weight from parent + if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { + wu32(brt + BR.blend_weight, ru32(bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118 + BR.blend_weight)); + } else if (bone_idx == 0) { + wu32(brt + BR.blend_weight, 0); // root bone, no blend + } else { + wu32(brt + BR.blend_weight, ru32(bone_rt_base + BR.blend_weight)); + } + } else { + const cf_remaining = ri32(brt + BR.crossfade_end) - ri32(anim_ctx + 0x0C); + if (cf_remaining < 1 or (ru32(brt + BR.prim_time) == ru32(brt + BR.sec_time) and + ru32(brt + BR.prim_track) == ru32(brt + BR.sec_track))) + { + wu32(brt + BR.blend_weight, 0); + } else { + const t_raw = @as(f32, @floatFromInt(cf_remaining)) * ufloat(ru32(brt + BR.crossfade_inv)); + const t_clamped = if (t_raw < 0.0) @as(f32, 0.0) else if (t_raw > 1.0) @as(f32, 1.0) else t_raw; + // Hermite: (3 - 2t) * t^2 * weight + const h = (3.0 - 2.0 * t_clamped) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight)); + wu32(brt + BR.blend_weight, fbits(h)); + } + } + + // --- Parent bone transform inheritance --- + const combined_flags: u32 = ru32(brt + BR.flags2) | flags; + var src_mat: u32 = undefined; + + if (ru16(bdef + BD.parent_bone) == 0xFFFF) { + src_mat = this + 0xFC; + } else { + const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40; + src_mat = parent_out; + + // Billboard pre-processing (flags & 7) + if ((combined_flags & 7) != 0) { + // Copy parent matrix to local_mat and work from there + for (0..16) |i| { + local_mat[i] = rf32(parent_out + @as(u32, @intCast(i)) * 4); + } + + // Apply pivot translation + const pivot_x = rf32(bdef + BD.pivot_x); + const pivot_y = rf32(bdef + BD.pivot_y); + const pivot_z = rf32(bdef + BD.pivot_z); + + // Compute translated position + const tx = local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z + local_mat[12]; + const ty = local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z + local_mat[13]; + const tz = local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z + local_mat[14]; + + const bb_type = combined_flags & 6; + if (bb_type == 2) { + // Cylindrical billboard — normalize each column + const n0 = normalizeVec3(local_mat[0], local_mat[1], local_mat[2]); + local_mat[0] = n0[0]; + local_mat[1] = n0[1]; + local_mat[2] = n0[2]; + const n1 = normalizeVec3(local_mat[4], local_mat[5], local_mat[6]); + local_mat[4] = n1[0]; + local_mat[5] = n1[1]; + local_mat[6] = n1[2]; + const n2 = normalizeVec3(local_mat[8], local_mat[9], local_mat[10]); + local_mat[8] = n2[0]; + local_mat[9] = n2[1]; + local_mat[10] = n2[2]; + } else if (bb_type == 4) { + // Spherical billboard — inherit camera rotation with scale preservation + // All sqmag computations MUST call game's vec3SqMag (0x4549F0) + const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) }; + const cam_len_sq0 = callVec3SqMag(this + SO.bb_row0); + var s0: f32 = 1.0; + if (cam_len_sq0 > rf32(0x0080c5c8)) { + var tmp0 = [3]f32{ local_mat[0], local_mat[1], local_mat[2] }; + const mat_len_sq0 = callVec3SqMag(@intFromPtr(&tmp0)); + s0 = @sqrt(mat_len_sq0 / cam_len_sq0); + } + local_mat[0] = s0 * cam0[0]; + local_mat[1] = s0 * cam0[1]; + local_mat[2] = s0 * cam0[2]; + + const wt0 = rf32(this + SO.world_xform + 0 * 4); + const wt1 = rf32(this + SO.world_xform + 1 * 4); + const wt2 = rf32(this + SO.world_xform + 2 * 4); + const wt_len_sq = callVec3SqMag(this + SO.world_xform); + var s1: f32 = 1.0; + if (wt_len_sq > rf32(0x0080c5c8)) { + var tmp1 = [3]f32{ local_mat[4], local_mat[5], local_mat[6] }; + const mat_len_sq1 = callVec3SqMag(@intFromPtr(&tmp1)); + s1 = @sqrt(mat_len_sq1 / wt_len_sq); + } + local_mat[4] = s1 * wt0; + local_mat[5] = s1 * wt1; + local_mat[6] = s1 * wt2; + + const wt4 = rf32(this + SO.world_xform + 4 * 4); + const wt5 = rf32(this + SO.world_xform + 5 * 4); + const wt6 = rf32(this + SO.world_xform + 6 * 4); + const wt_len_sq2 = callVec3SqMag(this + SO.world_xform + 16); + var s2: f32 = 1.0; + if (wt_len_sq2 > rf32(0x0080c5c8)) { + var tmp2 = [3]f32{ local_mat[8], local_mat[9], local_mat[10] }; + const mat_len_sq2 = callVec3SqMag(@intFromPtr(&tmp2)); + s2 = @sqrt(mat_len_sq2 / wt_len_sq2); + } + local_mat[8] = s2 * wt4; + local_mat[9] = s2 * wt5; + local_mat[10] = s2 * wt6; + } else if (bb_type == 6) { + // Full billboard — copy camera rotation directly + local_mat[0] = rf32(this + SO.bb_row0); + local_mat[1] = rf32(this + SO.bb_row0 + 4); + local_mat[2] = rf32(this + SO.bb_row0 + 8); + local_mat[4] = rf32(this + SO.world_xform + 0 * 4); + local_mat[5] = rf32(this + SO.world_xform + 1 * 4); + local_mat[6] = rf32(this + SO.world_xform + 2 * 4); + local_mat[8] = rf32(this + SO.world_xform + 4 * 4); + local_mat[9] = rf32(this + SO.world_xform + 5 * 4); + local_mat[10] = rf32(this + SO.world_xform + 6 * 4); + } + + // Recompute translation: pos - rot * pivot + if ((combined_flags & 1) == 0) { + local_mat[12] = tx - (local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z); + local_mat[13] = ty - (local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z); + local_mat[14] = tz - (local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z); + } else { + local_mat[12] = rf32(this + SO.world_xform + 8 * 4); + local_mat[13] = rf32(this + SO.world_xform + 9 * 4); + local_mat[14] = rf32(this + SO.world_xform + 10 * 4); + } + + src_mat = local_mat_addr; + } + } + + // --- Rotation interpolation --- + if ((combined_flags & 0x280) == 0) { + // No rotation animation — just copy parent + const dst = bone_out_base + bone_idx * 0x40; + copyMat4(dst, src_mat); + } else { + const rot_anim = bdef + BD.rot_anim; + const rot_kf_count = ru32(bdef + BD.rot_nts); + + // Capture primary InterpResult from rotation for reuse by scale/translation. + // findInterpIdx is the #1 leaf function in the engine; eliminating redundant + // calls saves ~70 cycles/bone (~23% of bone loop baseline). + var rot_primary_cache: ?InterpResult = null; + + // Rotation overwrites all 16 floats — skip identity init when present + if (rot_kf_count != 0) { + if (frame_ctr < rot_kf_count) { + // Call findInterpIdx via interpAnimKF — capture result for reuse + const rot_output = brt + BR.rot_idx0; + const r = findInterpIdx(this, ru32(brt + BR.prim_time), ru32(brt + BR.prim_track), rot_anim, rot_output); + rot_primary_cache = r; + const q = interpAnimKFCached(this, brt, rot_anim, rot_output, r); + local_mat2 = buildRotationMatrixVal(q[0], q[1], q[2], q[3]); + } else { + local_mat2 = buildRotationMatrixVal(rf32(brt + BR.rot_x), rf32(brt + BR.rot_y), rf32(brt + BR.rot_z), rf32(brt + BR.rot_w)); + } + } else { + local_mat2 = .{ 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 }; + } + + // Step 2: Scale interpolation — applied after rotation (inline array ops) + const scale_anim = bdef + BD.scale_anim; + const scale_kf_count = ru32(bdef + BD.scale_nts); + if (scale_kf_count != 0) { + var sx: f32 = undefined; + var sy: f32 = undefined; + var sz: f32 = undefined; + if (frame_ctr < scale_kf_count) { + // Reuse rotation's search result if temporal structure matches + const scale_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, scale_anim)) rot_primary_cache else null; + const s = interpVec3TrackCached(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)), scale_cache); + sx = s[0]; sy = s[1]; sz = s[2]; + } else { + sx = rf32(brt + BR.scale_x); sy = rf32(brt + BR.scale_y); sz = rf32(brt + BR.scale_z); + } + local_mat2[0] *= sx; local_mat2[1] *= sx; local_mat2[2] *= sx; + local_mat2[4] *= sy; local_mat2[5] *= sy; local_mat2[6] *= sy; + local_mat2[8] *= sz; local_mat2[9] *= sz; local_mat2[10] *= sz; + } + + // Conditional multiply: bone_local *= *(bone_rt+0xF0) + if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) { + local_mat2 = matMul4x4InPlace(local_mat2, ru32(brt + BR.bone_flag_cache)); + } + + // Step 3: Translation interpolation + var tx_val = rf32(bdef + BD.pivot_x); + var ty_val = rf32(bdef + BD.pivot_y); + var tz_val = rf32(bdef + BD.pivot_z); + + const trans_anim = bdef + BD.trans_anim; + const trans_kf_count = ru32(bdef + BD.trans_nts); + if (trans_kf_count != 0) { + if (frame_ctr < trans_kf_count) { + // Reuse rotation's search result if temporal structure matches + const trans_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, trans_anim)) rot_primary_cache else null; + const t = interpVec3TrackCached(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)), trans_cache); + tx_val += t[0]; + ty_val += t[1]; + tz_val += t[2]; + } else { + tx_val += rf32(brt + BR.trans_x); + ty_val += rf32(brt + BR.trans_y); + tz_val += rf32(brt + BR.trans_z); + } + } + + // Step 4: Compute translation offset using the ROTATED+SCALED matrix. + const piv_x = rf32(bdef + BD.pivot_x); + const piv_y = rf32(bdef + BD.pivot_y); + const piv_z = rf32(bdef + BD.pivot_z); + local_mat2[12] = tx_val - (local_mat2[0] * piv_x + local_mat2[4] * piv_y + local_mat2[8] * piv_z); + local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z); + local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z); + + // Write final composed matrix to output: dst = bone_local * parent + matMul4x4Local(bone_out_base + bone_idx * 0x40, local_mat2, src_mat); + } + + // --- Billboard post-processing (flags & 0x78) --- + // Assembly at 0x7151F9-0x71594E. Runs for BOTH animated and non-animated paths. + // Modifies the already-written bone output matrix in-place. + if ((combined_flags & 0x78) != 0) { + // pMVar19 = bone_idx * 0x40 (byte offset for output) + // pfVar12 = bone_out_base + pMVar19 (output matrix ptr) + const out_off = bone_idx * 0x40; + const om = bone_out_base + out_off; // output matrix + + // Compute scale lengths — MUST call game's vec3SqMag (0x4549F0), not inline + // Assembly: LEA ECX,[stack_vec3]; CALL 0x4549F0; FSQRT + const scale_len0 = @sqrt(callVec3SqMag(om)); + const scale_len1 = @sqrt(callVec3SqMag(om + 0x10)); + const scale_len2 = @sqrt(callVec3SqMag(om + 0x20)); + + // Compute translated pivot position through the output matrix + // local_a8 = pivot * matrix + translation + const bpx = rf32(bdef + BD.pivot_x); + const bpy = rf32(bdef + BD.pivot_y); + const bpz = rf32(bdef + BD.pivot_z); + const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30); + const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34); + const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38); + + // Switch on billboard post-processing type + const bb_post = combined_flags & 0x78; + switch (bb_post) { + 0x08 => { + // Type 8: decompilation lines 657-718 + // If no pre-billboard (local_1c == 0 i.e. flags & 0x280 was 0): + // set fixed rotation columns + // Else: use rotation matrix rows with negated first component, normalize + const had_anim = (combined_flags & 0x280) != 0; + if (!had_anim) { + // Fixed columns: row0={0,0,-1}, row1={1,0,0}, row2={0,1,0} + wf32(om, 0); + wf32(om + 0x04, 0); + wf32(om + 0x08, -1); + wf32(om + 0x10, 1); + wf32(om + 0x14, 0); + wf32(om + 0x18, 0); + wf32(om + 0x20, 0); + wf32(om + 0x24, 1); + wf32(om + 0x28, 0); + } else { + // Row 0 = {local_e4, local_e0, -local_e8}, normalize + const r0x = local_mat2[1]; // local_e4 + const r0y = local_mat2[2]; // local_e0 + const r0z = -local_mat2[0]; // -local_e8 + wf32(om, r0x); + wf32(om + 0x04, r0y); + wf32(om + 0x08, r0z); + const n0 = normalizeVec3InPlace(om); + _ = n0; + // Row 1 = {local_d4, local_d0, -local_d8}, normalize + const r1x = local_mat2[5]; // local_d4 + const r1y = local_mat2[6]; // local_d0 + const r1z = -local_mat2[4]; // -local_d8 + wf32(om + 0x10, r1x); + wf32(om + 0x14, r1y); + wf32(om + 0x18, r1z); + const n1 = normalizeVec3InPlace(om + 0x10); + _ = n1; + // Row 2 = {local_c4, local_c0, -local_c8}, normalize + const r2x = local_mat2[9]; // local_c4 + const r2y = local_mat2[10]; // local_c0 + const r2z = -local_mat2[8]; // -local_c8 + wf32(om + 0x20, r2x); + wf32(om + 0x24, r2y); + wf32(om + 0x28, r2z); + const n2 = normalizeVec3InPlace(om + 0x20); + _ = n2; + } + }, + 0x10 => { + // Type 16: normalize row0, set row1={row0.y, -row0.x, 0}, normalize, + // row2 = cross(row0, row1) + const n0 = normalizeVec3InPlace(om); + _ = n0; + const r0x = rf32(om); + const r0y = rf32(om + 0x04); + wf32(om + 0x10, r0y); + wf32(om + 0x14, -r0x); + wf32(om + 0x18, 0); + const n1 = normalizeVec3InPlace(om + 0x10); + _ = n1; + // row2 = -cross(row0, row1) — assembly uses negated cross product + wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18)); + wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10)); + wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14)); + }, + 0x20 => { + // Type 32: normalize row1, set row0={-row1.y, row1.x, 0}, normalize, + // row2 = -cross(row0, row1) + const n1 = normalizeVec3InPlace(om + 0x10); + _ = n1; + wf32(om, -rf32(om + 0x14)); + wf32(om + 0x04, rf32(om + 0x10)); + wf32(om + 0x08, 0); + const n0 = normalizeVec3InPlace(om); + _ = n0; + // row2 = -cross(row0, row1) — assembly uses negated cross product + wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18)); + wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10)); + wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14)); + }, + 0x40 => { + // Type 64: normalize row2, set row1={row2.y, -row2.x, 0}, normalize, + // row0 = cross(row1, row2) + normalizeVec3InPlace(om + 0x20); + wf32(om + 0x10, rf32(om + 0x24)); + wf32(om + 0x14, -rf32(om + 0x20)); + wf32(om + 0x18, 0); + normalizeVec3InPlace(om + 0x10); + // row0 = cross(row2.y*row1.z - row2.z*row1.y, ...) + wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14)); + wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18)); + wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10)); + }, + else => {}, + } + + // Apply scale lengths back and recompute translation + // Assembly at 0x715868-0x71594B + wf32(om + 0x0C, 0); + wf32(om + 0x1C, 0); + wf32(om + 0x2C, 0); + // Scale each row by its original length + const r0x_s = rf32(om); + wf32(om, scale_len0 * r0x_s); + const r0y_s = rf32(om + 0x04); + wf32(om + 0x04, scale_len0 * r0y_s); + const r0z_s = rf32(om + 0x08); + wf32(om + 0x08, scale_len0 * r0z_s); + const r1x_s = rf32(om + 0x10); + wf32(om + 0x10, scale_len1 * r1x_s); + const r1y_s = rf32(om + 0x14); + wf32(om + 0x14, scale_len1 * r1y_s); + const r1z_s = rf32(om + 0x18); + wf32(om + 0x18, scale_len1 * r1z_s); + const r2x_s = rf32(om + 0x20); + wf32(om + 0x20, scale_len2 * r2x_s); + const r2y_s = rf32(om + 0x24); + wf32(om + 0x24, scale_len2 * r2y_s); + const r2z_s = rf32(om + 0x28); + wf32(om + 0x28, scale_len2 * r2z_s); + + // Recompute translation: pos - scaled_matrix * pivot + wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz)); + wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz)); + wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz)); + wf32(om + 0x3C, 1.0); + } + } + } + + // ========================================================================= + // Sections 8-11: Post-bone-loop animations + // These sections handle texture animation, color animation, bone keyframe + // post-processing, and particle emitters. They follow the same interpolation + // pattern as the bone loop but operate on different model data arrays. + // + // For the initial implementation, we delegate these to the patterns established + // above. Each section iterates over its respective model array and calls + // findInterpIdx + lerp + crossfade blend. + // ========================================================================= + + // BISECT: stop after section 7 (bone loop) + + // Section 8: Texture animation loop + texAnimLoop(this, model_hdr, frame_ctr); + colorAnimLoop(this, model_hdr, frame_ctr); + + // model_hdr+0x6C = count, model_hdr+0x70 = data, output at this+0xAC (SO.scale1) + // Data stride 0x1C, output stride 0x20. Word copy with crossfade. + wordAnimLoop(this, model_hdr, frame_ctr); + boneKeyframeLoop(this, model_hdr); + particleLoops(this, model_hdr, frame_ctr); + attachmentRecursion(this, model_hdr, bone_out_base, frame_ctr); + + // ========================================================================= + // Section 13: Sync update + // ========================================================================= + wu32(this + SO.sync_value, ru32(anim_ctx + 0x10)); +} + +// ============================================================================= +// Post-bone-loop sections (extracted for readability) +// ============================================================================= + +fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + const count = ru32(model_hdr + 0x54); + if (count == 0) return; + const data_base = ru32(model_hdr + 0x58); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const out_base = ru32(this + SO.tex_anim_out); + const stf = getShortToFloat(); + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count) : ({ + i += 1; + data_off += 0x38; + out_off += 0x14 * 4; + }) { + const anim_data = data_base + data_off; + const output = out_base + out_off; + if (frame_ctr < ru32(data_base + data_off + 0x0C)) { + _ = interpVec3Track(this, bone_rt_base, anim_data, output, ufloat(ru32(bone_rt_base + BR.blend_weight))); + } + // Alpha/opacity track (assembly 0x715AF1-0x715C5E) + if (frame_ctr < ru32(anim_data + 0x28)) { + const alpha_anim = anim_data + 0x1C; + const alpha_out = output + 0x30; + const ar = findInterpIdx(this, ru32(bone_rt_base + BR.prim_time), ru32(bone_rt_base + BR.prim_track), alpha_anim, alpha_out); + const mode = ri16(alpha_anim); + if (mode == 0) { + const kf_data = ru32(alpha_anim + 0x18); + const sv = @as(f32, @floatFromInt(@as(i32, @as(*align(1) const i16, @ptrFromInt(kf_data + ar.idx0 * 2)).*))); + wf32(alpha_out + 0x0C, sv * stf); + } else { + const primary = shortInterpToFloat(alpha_anim, ar, stf); + wf32(alpha_out + 0x0C, primary); + + const bw = rf32(bone_rt_base + BR.blend_weight); + if (bw != 0.0 and ri16(alpha_anim + 0x02) == -1) { + const asr = findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), alpha_anim, alpha_out + 0x10); + const secondary = shortInterpToFloat(alpha_anim, asr, stf); + wf32(alpha_out + 0x1C, secondary); + wf32(alpha_out + 0x0C, @mulAdd(f32, secondary - primary, bw, primary)); + } + } + } + } +} + +/// Short-value interpolation: uses InterpResult indices, looks up short values, interpolates. +/// Shared by texAnimLoop alpha, colorAnimLoop, and word animation crossfade. +inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 { + const mode = ri16(anim_data); + const table = anim_data + AD.nvalues; + if (mode == 0) { + return @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, r.idx0)))) * stf; + } else { + const v1 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, r.idx1)))); + const v0 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, r.idx0)))); + return (v1 * stf - v0 * stf) * r.t + v0 * stf; + } +} + +fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + // Assembly: model_hdr+0x64 is both entry gate AND loop count + const count = ru32(model_hdr + 0x64); + if (count == 0) return; + const data_base = ru32(model_hdr + 0x68); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const out_base = ru32(this + SO.color_anim_out); + const stf = getShortToFloat(); + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count) : ({ + i += 1; + data_off += 0x1C; + out_off += 0x20; + }) { + const anim_data = data_base + data_off; + const output = out_base + out_off; + if (frame_ctr < ru32(anim_data + 0x0C)) { + const cr = findInterpIdx(this, ru32(bone_rt_base + BR.prim_time), ru32(bone_rt_base + BR.prim_track), anim_data, output); + const mode = ri16(anim_data); + if (mode == 0) { + const kf_data = ru32(anim_data + 0x18); + const sv = @as(f32, @floatFromInt(@as(i32, @as(*align(1) const i16, @ptrFromInt(kf_data + cr.idx0 * 2)).*))); + wf32(output + 0x0C, sv * stf); + } else { + const primary = shortInterpToFloat(anim_data, cr, stf); + wf32(output + 0x0C, primary); + + const bw = rf32(bone_rt_base + BR.blend_weight); + if (bw != 0.0 and ri16(anim_data + 0x02) == -1) { + const csr = findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); + const secondary = shortInterpToFloat(anim_data, csr, stf); + wf32(output + 0x1C, secondary); + wf32(output + 0x0C, @mulAdd(f32, secondary - primary, bw, primary)); + } + } + } + } +} + +fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + // Assembly 0x715E46-0x715F25: word/byte animation section + // model_hdr+0x6C = count, model_hdr+0x70 = data base + // Output at this+0xAC (SO.scale1), data stride 0x1C, output stride 0x20 + const count = ru32(model_hdr + 0x6C); + if (count == 0) return; + const data_base = ru32(model_hdr + 0x70); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const out_base = ru32(this + SO.scale1); + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count) : ({ + i += 1; + data_off += 0x1C; + out_off += 0x20; + }) { + const anim_data = data_base + data_off; + const output = out_base + out_off; + if (frame_ctr < ru32(anim_data + 0x0C)) { + const wr = findInterpIdx(this, ru32(bone_rt_base + BR.prim_time), ru32(bone_rt_base + BR.prim_track), anim_data, output); + // Word copy: read word from keyframe data via direct indexing + // Assembly (0x715EA3): MOV AX,[kf_data+idx*2]; MOV [output+0x0C],AX + const kf_data = ru32(anim_data + 0x18); + wu16(output + 0x0C, ru16(kf_data + wr.idx0 * 2)); + + // Crossfade (assembly 0x715EB4-0x715EFA) + // Original: JZ skip if mode==0, then check blend_weight > 0, then time_index == -1 + if (ri16(anim_data) == 0) { + // mode 0: no crossfade, skip + } else { + const bw = rf32(bone_rt_base + BR.blend_weight); + if (bw != 0.0 and ri16(anim_data + 0x02) == -1) { + const wsr = findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); + wu16(output + 0x1C, ru16(kf_data + wsr.idx0 * 2)); + } + } + } + } +} + +fn boneKeyframeLoop(this: u32, model_hdr: u32) void { + const count = ru32(model_hdr + 0x74); + if (count == 0) return; + + // One-time global init (assembly 0x715F45-0x715F81) + // Sets {0.5, 0.5, 0.0} constants at 0xCF043C and calls 0x409AEF + if ((ru8(0xCF04C4) & 1) == 0) { + wu8(0xCF04C4, ru8(0xCF04C4) | 1); + wu32(0xCF043C, 0x3F000000); // 0.5f + wu32(0xCF0440, 0x3F000000); // 0.5f + wu32(0xCF0444, 0x00000000); // 0.0f + // CALL 0x409AEF with arg 0x7187E0 (__cdecl, 1 stack param) + const initFn: *const fn (u32) callconv(.c) void = @ptrFromInt(0x409AEF); + initFn(0x7187E0); + } + + const data_base = ru32(model_hdr + 0x78); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const scale2_base = ru32(this + SO.scale2); + const scale3_base = ru32(this + SO.scale3); + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + var mat_off: u32 = 0; + while (i < count) : ({ + i += 1; + data_off += 0x54; // assembly at 0x7163A2: ADD EDI, 0x54 + out_off += 0x98; // assembly at 0x7163A5: ADD ESI, 0x98 + mat_off += 0x40; // assembly at 0x715395: ADD EDX, 0x40 + }) { + const kf_data = data_base + data_off; + const output = @as(u32, @intCast(@as(i32, @bitCast(scale2_base)) + @as(i32, @bitCast(out_off)))); + const mat_out = @as(u32, @intCast(@as(i32, @bitCast(scale3_base)) + @as(i32, @bitCast(mat_off)))); + + // Init identity matrix for this keyframe entry + setIdentity(mat_out); + + // Rotation: AnimData at kf_entry+0x1C, gate at kf_entry+0x28 + // Assembly at 0x715FDB: CMP [ECX+0x28], 0; AnimData at EDX+0x1C + if (ru32(kf_data + 0x28) != 0) { + const q = interpAnimKF(this, bone_rt_base, kf_data + 0x1C, output + 0x30); + applyTranslation(mat_out, rf32(0xCF043C), rf32(0xCF0440), rf32(0xCF0444)); + rotateByQuaternion(mat_out, q[0], q[1], q[2], q[3]); + applyTranslation(mat_out, -rf32(0xCF043C), -rf32(0xCF0440), -rf32(0xCF0444)); + } + + if (ru32(kf_data + 0x44) != 0) { + const s = interpVec3Track(this, bone_rt_base, kf_data + 0x38, output + 0x68, ufloat(ru32(bone_rt_base + BR.blend_weight))); + applyTranslation(mat_out, rf32(0xCF043C), rf32(0xCF0440), rf32(0xCF0444)); + scaleMatrix3x3(mat_out, s[0], s[1], s[2]); + applyTranslation(mat_out, -rf32(0xCF043C), -rf32(0xCF0440), -rf32(0xCF0444)); + } + + if (ru32(kf_data + 0x0C) != 0) { + const tv = interpVec3Track(this, bone_rt_base, kf_data, output, ufloat(ru32(bone_rt_base + BR.blend_weight))); + applyTranslation(mat_out, tv[0], tv[1], tv[2]); + } + } +} + +fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { + // Particle emitters are the largest section (~1000 lines of decompiled C). + // They follow the same interpolation patterns but with many sub-tracks per emitter. + // For the initial implementation, we handle the key tracks (position, speed, scale). + // The remaining tracks (color, alpha, emission rate, etc.) use identical patterns. + + // Ribbon emitters (model_hdr + 0x11C) + ribbonEmitterLoop(this, model_hdr, frame_ctr); + + // Particle emitters (model_hdr + 0x124) + particleEmitterLoop(this, model_hdr, frame_ctr); + + // Additional particle sections (model_hdr + 0x134, 0x13C) + additionalParticleLoops(this, model_hdr, frame_ctr); +} + +fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + const count = ru32(model_hdr + 0x11C); + if (count == 0) return; + const data_base = ru32(model_hdr + 0x120); + const out_base = ru32(this + SO.field_200); + const bone_rt_base = ru32(this + SO.bone_rt_base); + + var i: u32 = 0; + while (i < count) : (i += 1) { + const entry = data_base + i * 0xD4; // asm 0x716ABC: ADD EDI, 0xD4 + const output = out_base + i * 0x170; // asm 0x716AC2: ADD ESI, 0x170 + const bone_idx = @as(u32, ru16(entry + 2)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + + // ---- Visibility byte animation (asm 0x7163FC-0x7164F2) ---- + if (ru32(output + 0x100) != 0) { + if (ru32(entry + 0xC4) != 0) { + const vr = findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), entry + 0xB8, output + 0xE0); + const vis_values = ru32(entry + 0xD0); // entry+0xB8+0x18 = AD.keyframe_base + wu8(output + 0xEC, ru8(vis_values + vr.idx0)); + if (ri16(entry + 0xB8) != 0) { + if (rf32(bone_rt + BR.blend_weight) != 0.0 and ri16(entry + 0xBA) == -1) { + const vsr = findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), entry + 0xB8, output + 0xF0); + wu8(output + 0xFC, ru8(vis_values + vsr.idx0)); + } + } + } + } + + // ---- Visibility gate (asm 0x7164F2-0x716514) ---- + const should_process = blk: { + if (ru32(output + 0x100) != 0 and ru8(output + 0xEC) != 0) break :blk true; + if (frame_ctr == 0) break :blk true; + break :blk false; + }; + if (!should_process) continue; + + // ---- Track A (float): gate=entry+0x38, AD=entry+0x2C, output+0x30 ---- + if (frame_ctr < ru32(entry + 0x38)) { + _ = interpFloatTrack(this, bone_rt, entry + 0x2C, output + 0x30, ufloat(ru32(bone_rt + BR.blend_weight))); + } + + // ---- Track B (Vec3): gate=entry+0x1C, AD=entry+0x10, output+0x00 ---- + if (frame_ctr < ru32(entry + 0x1C)) { + const v = interpVec3Track(this, bone_rt, entry + 0x10, output, ufloat(ru32(bone_rt + BR.blend_weight))); + // Post-processing 1 (asm 0x71678A-0x7167CE) + const scale1 = rf32(output + 0x3C) * rf32(this + SO.render_scale_z); + wf32(output + 0x134, v[0] * scale1); + wf32(output + 0x138, v[1] * scale1); + wf32(output + 0x13C, v[2] * scale1); + } + + // ---- Track C (float): gate=entry+0x70, AD=entry+0x64, output+0x80 ---- + if (frame_ctr < ru32(entry + 0x70)) { + _ = interpFloatTrack(this, bone_rt, entry + 0x64, output + 0x80, ufloat(ru32(bone_rt + BR.blend_weight))); + } + + // ---- Track D (Vec3): gate=entry+0x54, AD=entry+0x48, output+0x50 ---- + if (frame_ctr < ru32(entry + 0x54)) { + const v2 = interpVec3Track(this, bone_rt, entry + 0x48, output + 0x50, ufloat(ru32(bone_rt + BR.blend_weight))); + // Post-processing 2 (asm 0x716A67-0x716AA6) + const scale2 = rf32(output + 0x8C) * rf32(this + SO.render_scale_z); + wf32(output + 0x140, v2[0] * scale2); + wf32(output + 0x144, v2[1] * scale2); + wf32(output + 0x148, v2[2] * scale2); + } + } +} + +fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + const count = ru32(model_hdr + 0x124); + if (count == 0) return; + const data_base = ru32(model_hdr + 0x128); + const out_base = ru32(this + SO.particle1); + const bone_rt_base = ru32(this + SO.bone_rt_base); + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count) : ({ + i += 1; + data_off += 0x7C; + out_off += 0x84; + }) { + const entry = data_base + data_off; + const output = out_base + out_off; + + // Assembly uses bone_rt_base directly (bone 0) — NOT per-entry bone_idx. + + if (frame_ctr < ru32(entry + 0x1C)) { + interpVec3Track36(this, bone_rt_base, entry + 0x10, output); + } + if (frame_ctr < ru32(entry + 0x44)) { + interpVec3Track36(this, bone_rt_base, entry + 0x38, output + 0x30); + } + if (frame_ctr < ru32(entry + 0x6C)) { + interpFloatTrack12(this, bone_rt_base, entry + 0x60, output + 0x60); + } + } +} + +fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + const stf = getShortToFloat(); + // Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A) + // Then additional_remaining reset at 0x717D6F + // Then model_hdr+0x13C section (asm 0x717D75-0x7185E3) + + // Section 12c: model_hdr+0x134 particle visibility/tracks + // count=+0x134, data=+0x138, output=this+0x3C8 + // Data stride 0xDC, output stride 0xD0 + // Each entry: bone_idx at +0x04, visibility at +0xCC + // Sub-tracks: visibility(+0xC0), position(+0x24), alpha(+0x40), + // speed(+0x5C), emission(+0x78), scale(+0xA4) + if (ru32(model_hdr + 0x134) != 0) { + const count0 = ru32(model_hdr + 0x134); + const data_base0 = ru32(model_hdr + 0x138); + const out_base0 = ru32(this + 0x3C8); // SO.particle2 + const bone_rt_base = ru32(this + SO.bone_rt_base); + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count0) : ({ + i += 1; + data_off += 0xDC; // asm 0x717D4D + out_off += 0xD0; // asm 0x717D53 + }) { + const entry = data_base0 + data_off; + const output = out_base0 + out_off; + + // Visibility check: entry+0xCC vs anim_frame_ctr + if (frame_ctr < ru32(entry + 0xCC)) { + const bone_idx = @as(u32, ru16(entry + 0x04)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + // Visibility byte animation at entry+0xC0 + const pvr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xC0, output + 0xB0); + const vis_mode = ri16(entry + 0xC0); + if (vis_mode == 0) { + wu8(output + 0xBC, ru8(ru32(entry + 0xC0 + 0x18) + pvr.idx0)); + } else { + wu8(output + 0xBC, ru8(pvr.idx0 + ru32(entry + 0xD8))); + // Crossfade blend for visibility if needed + if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xC2) == -1) { + const pvsr = findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xC0, output + 0xC0); + wu8(output + 0xCC, ru8(pvsr.idx0 + ru32(entry + 0xD8))); + } + } + } + + // Position track: entry+0x24 vs entry+0x30 + if (frame_ctr < ru32(entry + 0x30)) { + const bone_idx = @as(u32, ru16(entry + 0x04)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + _ = interpVec3Track(this, bone_rt, entry + 0x24, output, ufloat(ru32(bone_rt + BR.blend_weight))); + } + + // Alpha track: entry+0x40 vs entry+0x4C + // Short-value interpolation via game's getIndexOffset/setShortValue + if (frame_ctr < ru32(entry + 0x4C)) { + const bone_idx = @as(u32, ru16(entry + 0x04)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + const par = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x40, output + 0x30); + const alpha_mode = ri16(entry + 0x40); + const table = entry + 0x40 + AD.nvalues; + if (alpha_mode == 0) { + const sv = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, par.idx0)))); + wf32(output + 0x3C, sv * stf); + } else { + const v1 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, par.idx1)))); + const v0 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, par.idx0)))); + wf32(output + 0x3C, (v1 * stf - v0 * stf) * par.t + v0 * stf); + } + } + + // Speed track: entry+0x5C vs entry+0x68 + if (frame_ctr < ru32(entry + 0x68)) { + const bone_idx = @as(u32, ru16(entry + 0x04)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + _ = interpFloatTrack(this, bone_rt, entry + 0x5C, output + 0x50, ufloat(ru32(bone_rt + BR.blend_weight))); + } + + // Emission rate: entry+0x78 vs entry+0x84 + if (frame_ctr < ru32(entry + 0x84)) { + const bone_idx = @as(u32, ru16(entry + 0x04)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + _ = interpFloatTrack(this, bone_rt, entry + 0x78, output + 0x70, ufloat(ru32(bone_rt + BR.blend_weight))); + } + + // Scale track: entry+0xA4 vs entry+0xB0 + // Short value copy via game's getIndexOffset/setShortValue + if (frame_ctr < ru32(entry + 0xB0)) { + const bone_idx = @as(u32, ru16(entry + 0x04)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + const scr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xA4, output + 0x90); + const scale_values = ru32(entry + 0xA4 + AD.keyframe_base); + wu16(output + 0x9C, ru16(scale_values + scr.idx0 * 2)); + if (ri16(entry + 0xA4) != 0) { + if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xA6) == -1) { + const scsr = findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xA4, output + 0xA0); + wu16(output + 0xAC, ru16(scale_values + scsr.idx0 * 2)); + } + } + } + } + } + + // Additional remaining data reset — between 0x134 and 0x13C sections + // Assembly at 0x717D6F: MOV [EBX+0x3D8], 0 + wu32(this + 0x3D8, 0); + + // Section 12e: model_hdr+0x13C (largest particle section) + // count=+0x13C, data=+0x140 + // output1=this+0x3D0, output2=this+0x3D4 + // Data stride 0x1F8, output stride 0x16C + const count1 = ru32(model_hdr + 0x13C); + if (count1 != 0) { + const data_base = ru32(model_hdr + 0x140); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const particle_base = ru32(this + 0x3D0); // SO.particle3 + + var i: u32 = 0; + var data_off: u32 = 0; + var out_off: u32 = 0; + while (i < count1) : ({ + i += 1; + data_off += 0x1F8; // asm 0x7185CD + out_off += 0x16C; // asm 0x7185BA + }) { + const entry = data_base + data_off; + const output = particle_base + out_off; + const bone_idx = @as(u32, ru16(entry + 0x14)); + const bone_rt = bone_rt_base + bone_idx * 0x118; + + // All tracks from assembly 0x717D90-0x7185E3: + const particle_ptrs = ru32(this + 0x3D4); // [EBX+0x3D4] + const local_14 = ru32(particle_ptrs + i * 4); // per-emitter data ptr + + // Visibility: gate=entry+0x1E8, AnimData=entry+0x1DC, output=output+0x140 + if (frame_ctr < ru32(entry + 0x1E8)) { + const lvr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x1DC, output + 0x140); + if (ri16(entry + 0x1DC) == 0) { + wu8(output + 0x14C, ru8(ru32(entry + 0x1F4) + lvr.idx0)); + } else { + wu8(output + 0x14C, ru8(lvr.idx0 + ru32(entry + 0x1F4))); + if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0x1DE) == -1) { + const lvsr = findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0x1DC, output + 0x150); + wu8(output + 0x15C, ru8(ru32(entry + 0x1F4) + lvsr.idx0)); + } + } + } + + // Emitter active flag: visibility && emitter_enable_flag + const vis_byte = ru8(output + 0x14C); + const emitter_active: u32 = if (vis_byte != 0 and ru32(this + 0x50) != 0) 1 else 0; + wu32(output + 0x160, emitter_active); + // IsParticleBufferEmpty check + var buf_active: u32 = 0; + if (emitter_active != 0) { + buf_active = 1; + } else { + if (isParticleBufferNotEmpty(local_14)) { + buf_active = 1; + } + } + wu32(output + 0x164, buf_active); + // OR into additional_remaining + wu32(this + 0x3D8, ru32(this + 0x3D8) | buf_active); + + // Only process tracks if visible or first frame + if (vis_byte != 0 or frame_ctr == 0) { + // Track 1: emission rate — gate=+0x40, AnimData=+0x34, output=+0x00 + if (frame_ctr < ru32(entry + 0x40)) { + _ = interpFloatTrack(this, bone_rt, entry + 0x34, output, ufloat(ru32(bone_rt + BR.blend_weight))); + } + // Track 2: speed — gate=+0x5C, AnimData=+0x50, output=+0x20 + if (frame_ctr < ru32(entry + 0x5C)) { + _ = interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20, ufloat(ru32(bone_rt + BR.blend_weight))); + } + // Track 3: color — gate=+0x78, AnimData=+0x6C, output=+0x40 + if (frame_ctr < ru32(entry + 0x78)) { + _ = interpFloatTrack(this, bone_rt, entry + 0x6C, output + 0x40, ufloat(ru32(bone_rt + BR.blend_weight))); + } + // Track 4 — gate=+0x94, AnimData=+0x88, output=+0x60 + if (frame_ctr < ru32(entry + 0x94)) { + _ = interpFloatTrack(this, bone_rt, entry + 0x88, output + 0x60, ufloat(ru32(bone_rt + BR.blend_weight))); + } + // Track 5 (Vec3 spline) — gate=+0xB0, AnimData=+0xA4, output=+0x80 + if (frame_ctr < ru32(entry + 0xB0)) { + _ = interpFloatTrack(this, bone_rt, entry + 0xA4, output + 0x80, ufloat(ru32(bone_rt + BR.blend_weight))); + } + // Track 6 — gate=+0xCC, AnimData=+0xC0, output=+0xA0 + if (frame_ctr < ru32(entry + 0xCC)) { + _ = interpFloatTrack(this, bone_rt, entry + 0xC0, output + 0xA0, ufloat(ru32(bone_rt + BR.blend_weight))); + } + // Tracks 7-10: same as interpFloatTrack (0x71AF20 is identical logic) + if (frame_ctr < ru32(entry + 0xE8)) + _ = interpFloatTrack(this, bone_rt, entry + 0xDC, output + 0xC0, ufloat(ru32(bone_rt + BR.blend_weight))); + if (frame_ctr < ru32(entry + 0x104)) + _ = interpFloatTrack(this, bone_rt, entry + 0xF8, output + 0xE0, ufloat(ru32(bone_rt + BR.blend_weight))); + if (frame_ctr < ru32(entry + 0x120)) + _ = interpFloatTrack(this, bone_rt, entry + 0x114, output + 0x100, ufloat(ru32(bone_rt + BR.blend_weight))); + if (frame_ctr < ru32(entry + 0x13C)) + _ = interpFloatTrack(this, bone_rt, entry + 0x130, output + 0x120, ufloat(ru32(bone_rt + BR.blend_weight))); + } + } + } +} + +fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + const hierarchy = ru32(this + SO.hierarchy_ptr); + if (hierarchy == 0) return; + + // Attachment byte animation loop — skipped when attach_count==0 but + // child recursion below MUST still run. Original JBE 0x718657 jumps + // past this loop to the child section, NOT to the function exit. + const attach_count = ru32(model_hdr + 0x104); + const attach_data = ru32(model_hdr + 0x108); + + // Process attachment byte animations (only when attach_count > 0) + var att_i: u32 = 0; + var att_off: u32 = 0; + while (att_i < attach_count) : ({ + att_i += 1; + att_off += 0x30; + }) { + const att_entry = attach_data + att_off; + if (frame_ctr < ru32(att_entry + 0x20)) { + const bone_idx = @as(u32, ru16(att_entry + 4)); + const bone_rt = ru32(this + SO.bone_rt_base) + bone_idx * 0x118; + const anim_data = att_entry + 0x14; + const att_output = hierarchy + att_i * 0x20; + const atr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), anim_data, att_output); + wu8(att_output + 0x0C, ru8(ru32(anim_data + AD.keyframe_base) + atr.idx0)); + } + } + + // Iterate child scene objects linked list + var child = ru32(this + SO.hierarchy_idx); + while (child != 0) { + // child->attach_idx at +0x1D4 (assembly-verified: MOV EAX,[ECX+0x1D4] at 0x718668) + const attach_idx = ru32(child + 0x1D4); + + // Check if attachment is valid (0xFFFF = no attachment) + if (attach_idx != 0xFFFF) { + const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C); + if (visible != 0) { + const att_entry = attach_data + attach_idx * 0x30; + const bone_idx = @as(u32, ru16(att_entry + 4)); + const bone_mat = bone_out_base + bone_idx * 0x40; + + // Copy parent bone matrix to local + var local_1a0: [16]f32 = undefined; + for (0..16) |fi| { + local_1a0[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4); + } + + // Apply attachment offset translation + const ox = rf32(att_entry + 8); + const oy = rf32(att_entry + 0xC); + const oz = rf32(att_entry + 0x10); + local_1a0[12] += local_1a0[0] * ox + local_1a0[4] * oy + local_1a0[8] * oz; + local_1a0[13] += local_1a0[1] * ox + local_1a0[5] * oy + local_1a0[9] * oz; + local_1a0[14] += local_1a0[2] * ox + local_1a0[6] * oy + local_1a0[10] * oz; + + // Direct recursion — no hook overhead + transformImpl_SSE(child, @intFromPtr(&local_1a0), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z)); + } + } + + // Next sibling in linked list + // Assembly-verified: MOV ECX,[ECX+0x1E4] at 0x718764 + child = ru32(child + 0x1E4); + } +} diff --git a/src/performance/clip_sse.zig b/src/performance/clip_sse.zig new file mode 100644 index 0000000..45285f4 --- /dev/null +++ b/src/performance/clip_sse.zig @@ -0,0 +1,376 @@ +//! SSE-optimized collision math — compiled ReleaseFast even in Debug builds. +//! +//! Functions are exported and called via `extern fn` from the Debug transform44 module. +//! This separation is required because Zig module imports inherit the parent's optimization +//! level, so only a separate `addObject(.{ .optimize = .ReleaseFast })` gets real SSE codegen. + +const V4 = @Vector(4, f32); + +const CLIP_EPSILON: f32 = @bitCast(@as(u32, 0x3ab60b61)); // +0.00139, global at 0x80dfec +const MOVEMENT_EPSILON: f32 = @bitCast(@as(u32, 0x35800000)); // 9.54e-7, global at 0x8026bc + +export fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void { + const plane: [*]const f32 = @ptrFromInt(plane_addr); + const poly: [*]f32 = @ptrFromInt(poly_addr); + const new_attrib: f32 = @bitCast(attrib_bits); + + // Vertex count at poly+0xF0 (byte offset 240, float offset 60) + const count_ptr: *align(1) u32 = @ptrCast(poly + 60); + const n = count_ptr.*; + if (n == 0) return; + + // Load plane as 4-wide vector for SSE dot product + const pv: @Vector(4, f32) = .{ plane[0], plane[1], plane[2], plane[3] }; + + // --- Phase 1: Compute signed distances --- + var dists: [15]f32 = undefined; + var min_dist: f32 = @bitCast(@as(u32, 0x7f7fffff)); // FLT_MAX + var max_dist: f32 = @bitCast(@as(u32, 0xff7fffff)); // -FLT_MAX + + var i: u32 = 0; + while (i < n) : (i += 1) { + const b = i * 3; + const v: @Vector(4, f32) = .{ poly[b], poly[b + 1], poly[b + 2], 1.0 }; + const d = -@reduce(.Add, v * pv); + dists[i] = d; + if (d < min_dist) min_dist = d; + if (d > max_dist) max_dist = d; + } + + // --- Phase 2: Early exits (exact same thresholds as original) --- + if (min_dist > -CLIP_EPSILON) return; // all inside + if (max_dist < CLIP_EPSILON) { + count_ptr.* = 0; // all outside + return; + } + + // --- Phase 3: Copy polygon to local buffers --- + const attribs: [*]f32 = poly + 45; // +0xB4 + var tmp_v: [15 * 3]f32 = undefined; + var tmp_a: [15]f32 = undefined; + for (0..n * 3) |j| tmp_v[j] = poly[j]; + for (0..n) |j| tmp_a[j] = attribs[j]; + + // --- Phase 4: Sutherland-Hodgman edge walk --- + var out: u32 = 0; + var prev: u32 = n - 1; + var pd = dists[prev]; + + i = 0; + while (i < n) : (i += 1) { + const cd = dists[i]; + + if (pd >= 0.0) { + if (cd >= 0.0) { + emit(poly, attribs, &out, (tmp_v[i * 3 ..]).ptr, tmp_a[i]); + } else { + if (pd > CLIP_EPSILON) { + lerp(poly, attribs, &out, (tmp_v[prev * 3 ..]).ptr, (tmp_v[i * 3 ..]).ptr, pd / (cd - pd), new_attrib); + } + } + } else { + if (cd >= 0.0) { + if (cd > CLIP_EPSILON) { + lerp(poly, attribs, &out, (tmp_v[prev * 3 ..]).ptr, (tmp_v[i * 3 ..]).ptr, pd / (cd - pd), new_attrib); + } + emit(poly, attribs, &out, (tmp_v[i * 3 ..]).ptr, tmp_a[i]); + } + } + + prev = i; + pd = cd; + } + + count_ptr.* = if (out >= 3) out else 0; +} + +inline fn emit(poly: [*]f32, attribs: [*]f32, out: *u32, v: [*]const f32, a: f32) void { + const o = out.*; + const b = o * 3; + poly[b] = v[0]; + poly[b + 1] = v[1]; + poly[b + 2] = v[2]; + attribs[o] = a; + out.* = o + 1; +} + +inline fn lerp(poly: [*]f32, attribs: [*]f32, out: *u32, p: [*]const f32, c: [*]const f32, t: f32, a: f32) void { + const o = out.*; + const b = o * 3; + // intersection = prev - (curr - prev) * t + poly[b] = p[0] - (c[0] - p[0]) * t; + poly[b + 1] = p[1] - (c[1] - p[1]) * t; + poly[b + 2] = p[2] - (c[2] - p[2]) * t; + attribs[o] = a; + out.* = o + 1; +} + +// ============================================================================= +// BuildTrianglePlanes (0x632460) +// __fastcall(ECX=vertices, EDX=triangle_indices(byte*), +// stack: plane_normal*, offset_vector*, output_planes*) +// Returns: int (1=ok, 0=degenerate) +// +// Builds 4 clipping planes from a triangle + offset: +// planes[0..2]: edge planes (perpendicular to triangle, one per edge) +// planes[3]: cap plane (from plane_normal, offset by offset_vector) +// ============================================================================= + +export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 { + const verts: [*]const f32 = @ptrFromInt(verts_addr); + const indices: [*]const u8 = @ptrFromInt(indices_addr); + const plane_normal: [*]const f32 = @ptrFromInt(normal_addr); + const offset_vec: [*]const f32 = @ptrFromInt(offset_addr); + const output: [*]f32 = @ptrFromInt(out_addr); + + const ofs = loadV3(offset_vec); + + // Load 3 triangle vertices + var tv: [3]V4 = undefined; + for (0..3) |i| { + const idx = @as(u32, indices[i]) * 3; + tv[i] = loadV3(verts + idx); + } + + // Edge tables: for edge i, v0=i, v1=next, v2=opposite + const next = [3]u8{ 1, 2, 0 }; + const opp = [3]u8{ 2, 0, 1 }; + + // Build 3 edge planes + for (0..3) |i| { + const v0 = tv[i]; + const v1 = tv[next[i]]; + const v2 = tv[opp[i]]; + const offset_v = v0 + ofs; // offset vertex + + // Check for degenerate triangle: cross(offset_vec, v1 - v0) + const edge = v1 - v0; + const check_normal = cross3(ofs, edge); + if (dot3(check_normal, check_normal) < MOVEMENT_EPSILON) return 0; + + // Calculate plane from 3 points (v0, v1, offset_v) + const e1 = offset_v - v0; // = offset_vec + const e2 = v1 - v0; // = edge + var normal = cross3(e2, e1); + const len_sq = dot3(normal, normal); + const inv_len: V4 = @splat(1.0 / @sqrt(len_sq)); + normal = normal * inv_len; + const d = -dot3(normal, v0); + + // Check orientation: if opposite vertex is on positive side, negate + const to_v2 = v2 - v0; + const plane_idx = i * 4; + if (dot3(to_v2, normal) > 0.0) { + output[plane_idx + 0] = -normal[0]; + output[plane_idx + 1] = -normal[1]; + output[plane_idx + 2] = -normal[2]; + output[plane_idx + 3] = -d; + } else { + output[plane_idx + 0] = normal[0]; + output[plane_idx + 1] = normal[1]; + output[plane_idx + 2] = normal[2]; + output[plane_idx + 3] = d; + } + } + + // 4th plane: cap plane from plane_normal + const pn = loadV3(plane_normal); + const cap_point = tv[0] + ofs; + output[12] = plane_normal[0]; + output[13] = plane_normal[1]; + output[14] = plane_normal[2]; + output[15] = -dot3(pn, cap_point); + + return 1; +} + +// ============================================================================= +// rayTriangleIntersection (0x7c29f0) +// Möller-Trumbore ray-triangle intersection test +// __fastcall(ECX=ray[6], EDX=verts_base, stack: indices_u16[3], out_dist*, out_bary*, tolerance) +// Returns: 1 = hit, 0 = miss +// ============================================================================= + +export fn rayTriangleIntersection( + ray_addr: u32, + verts_addr: u32, + indices_addr: u32, + out_dist_addr: u32, + out_bary_addr: u32, + tolerance_bits: u32, +) u32 { + const ray: [*]const f32 = @ptrFromInt(ray_addr); + const verts: [*]const f32 = @ptrFromInt(verts_addr); + const indices: [*]const u16 = @ptrFromInt(indices_addr); + const tolerance: f32 = @bitCast(tolerance_bits); + + const neg_tol = -tolerance; + const one_plus_tol = 1.0 + tolerance; + + // Load vertex positions via indices (each vertex = 3 floats, stride = 12 bytes) + const idx0: u32 = @as(u32, indices[0]) * 3; + const idx1: u32 = @as(u32, indices[1]) * 3; + const idx2: u32 = @as(u32, indices[2]) * 3; + + const v0 = loadV3(verts + idx0); + const v1 = loadV3(verts + idx1); + const v2 = loadV3(verts + idx2); + + // Ray: origin at ray[0..3], direction at ray[3..6] + const origin = loadV3(ray); + const dir = loadV3(ray + 3); + + // Möller-Trumbore algorithm + const edge1 = v1 - v0; + const edge2 = v2 - v0; + const h = cross3(dir, edge2); + const det = dot3(edge1, h); + + // Determinant thresholds from WoW binary (both 0.0 at compile time = two-sided test) + const det_neg = @as(*const f32, @ptrFromInt(0x0081d9bc)).*; + const det_pos = @as(*const f32, @ptrFromInt(0x0080e2e4)).*; + + if (det > det_neg and det < det_pos) return 0; + + const inv_det = 1.0 / det; + + // Barycentric u + const s = origin - v0; + const u_val = dot3(s, h) * inv_det; + if (u_val < neg_tol or u_val > one_plus_tol) return 0; + + // Barycentric v + const q = cross3(s, edge1); + const v_val = dot3(dir, q) * inv_det; + if (v_val < neg_tol or (u_val + v_val) > one_plus_tol) return 0; + + // Intersection distance t + const t = dot3(edge2, q) * inv_det; + + if (out_dist_addr != 0) { + @as(*f32, @ptrFromInt(out_dist_addr)).* = t; + } + if (out_bary_addr != 0) { + const bary: [*]f32 = @ptrFromInt(out_bary_addr); + bary[0] = u_val; + bary[1] = v_val; + } + + return 1; +} + +// ============================================================================= +// multiplyMatrix4x4 (0x7bc6a0) +// SSE replacement for 542 bytes of x87 FPU matrix multiply. +// __fastcall(ECX=result, EDX=left, stack: right), RET 0x4 +// Row-major: result[i][j] = sum(left[i][k] * right[k][j], k=0..3) +// Returns result pointer (EAX = result_addr). +// ============================================================================= + +export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 { + const result: [*]f32 = @ptrFromInt(result_addr); + const left: [*]const f32 = @ptrFromInt(left_addr); + const right: [*]const f32 = @ptrFromInt(right_addr); + + // Load all 4 rows of right matrix + const r0: V4 = .{ right[0], right[1], right[2], right[3] }; + const r1: V4 = .{ right[4], right[5], right[6], right[7] }; + const r2: V4 = .{ right[8], right[9], right[10], right[11] }; + const r3: V4 = .{ right[12], right[13], right[14], right[15] }; + + // For each row of left, broadcast-multiply-add against right rows + inline for (0..4) |i| { + const b = i * 4; + const out = splat4(left[b]) * r0 + splat4(left[b + 1]) * r1 + splat4(left[b + 2]) * r2 + splat4(left[b + 3]) * r3; + storeV4(result + b, out); + } + + return result_addr; +} + +// ============================================================================= +// rotateMatrixByAxisAngle (0x7bdd60) +// Builds axis-angle rotation matrix (Rodrigues) then multiplies. +// Replaces 853 bytes of x87 FPU (createAxisAngleRotationMatrix 311B + +// multiplyMatrix4x4 542B) with SSE @Vector math. +// __thiscall(ECX=matrix, stack: angle, axis_ptr, is_unit_flag) +// ============================================================================= + +export fn rotateMatrixByAxisAngle( + matrix_addr: u32, + angle_bits: u32, + axis_addr: u32, + is_unit: u32, +) void { + const matrix: [*]f32 = @ptrFromInt(matrix_addr); + const angle: f32 = @bitCast(angle_bits); + const axis_ptr: [*]const f32 = @ptrFromInt(axis_addr); + + // Load and optionally normalize axis + var ax = axis_ptr[0]; + var ay = axis_ptr[1]; + var az = axis_ptr[2]; + + if (is_unit == 0) { + const inv_len = 1.0 / @sqrt(ax * ax + ay * ay + az * az); + ax *= inv_len; + ay *= inv_len; + az *= inv_len; + } + + // Rodrigues rotation matrix (row-major, matching WoW's layout) + const c = cosf(angle); + const s = sinf(angle); + const t = 1.0 - c; + + // Build rotation matrix on stack, then multiply: result = rot * input + // Row 3 of rotation is {0,0,0,1} so result row 3 = input row 3 (identity). + // We build the full 4x4 rot matrix and use multiplyMatrix4x4 for the multiply. + var rot: [16]f32 = .{ + ax * ax * t + c, ax * ay * t + az * s, ax * az * t - ay * s, 0, + ax * ay * t - az * s, ay * ay * t + c, ay * az * t + ax * s, 0, + ax * az * t + ay * s, ay * az * t - ax * s, az * az * t + c, 0, + 0, 0, 0, 1, + }; + + // result = rot * matrix, but result == matrix so use temp to avoid aliasing + var tmp: [16]f32 = undefined; + _ = multiplyMatrix4x4(@intFromPtr(&tmp), @intFromPtr(&rot), matrix_addr); + matrix[0..16].* = tmp; +} + +// MSVC CRT sin/cos — linked from the WoW process +extern fn sinf(f32) f32; +extern fn cosf(f32) f32; + +// ============================================================================= +// Vector helpers +// ============================================================================= + +inline fn loadV3(p: [*]const f32) V4 { + return .{ p[0], p[1], p[2], 0.0 }; +} + +inline fn dot3(a: V4, b: V4) f32 { + const p = a * b; + return p[0] + p[1] + p[2]; +} + +inline fn cross3(a: V4, b: V4) V4 { + const a_yzx = @shuffle(f32, a, undefined, [4]i32{ 1, 2, 0, 3 }); + const b_zxy = @shuffle(f32, b, undefined, [4]i32{ 2, 0, 1, 3 }); + const a_zxy = @shuffle(f32, a, undefined, [4]i32{ 2, 0, 1, 3 }); + const b_yzx = @shuffle(f32, b, undefined, [4]i32{ 1, 2, 0, 3 }); + return a_yzx * b_zxy - a_zxy * b_yzx; +} + +inline fn splat4(v: f32) V4 { + return @splat(v); +} + +inline fn storeV4(dst: [*]f32, v: V4) void { + dst[0] = v[0]; + dst[1] = v[1]; + dst[2] = v[2]; + dst[3] = v[3]; +} diff --git a/src/performance/particle_sse.zig b/src/performance/particle_sse.zig new file mode 100644 index 0000000..1d11b85 --- /dev/null +++ b/src/performance/particle_sse.zig @@ -0,0 +1,1205 @@ +//! particle_sse — SSE replacements for WoW 1.12.1 particle rendering pipeline. +//! +//! Compiled as a separate ReleaseFast unit (same pattern as bone_sse.zig / clip_sse.zig). +//! Functions are exported and called via `extern fn` from transform44.zig detour hooks. +//! +//! Assembly references: decompiled/asm_RenderParticleSprites.txt, +//! decomp_RenderParticleSprites.c, decomp_particle_helpers.c +//! +//! Faithful recreation of RenderParticleSprites (0x7B2A50, 2688 bytes). +//! Every section verified against assembly. Optimization comes later — +//! first priority is byte-identical output. + +const std = @import("std"); +const V4 = @Vector(4, f32); +const CC = std.builtin.CallingConvention; +const TC: CC = .{ .x86_thiscall = .{} }; +const FC: CC = .{ .x86_fastcall = .{} }; + +inline fn rf32(addr: u32) f32 { + return @as(*align(1) const f32, @ptrFromInt(addr)).*; +} +inline fn ri32(addr: u32) i32 { + return @as(*align(1) const i32, @ptrFromInt(addr)).*; +} +inline fn ru8(addr: u32) u8 { + return @as(*const u8, @ptrFromInt(addr)).*; +} +inline fn ru16(addr: u32) u16 { + return @as(*align(1) const u16, @ptrFromInt(addr)).*; +} +inline fn ru32(addr: u32) u32 { + return @as(*align(1) const u32, @ptrFromInt(addr)).*; +} +inline fn wf32(addr: u32, val: f32) void { + @as(*align(1) f32, @ptrFromInt(addr)).* = val; +} +inline fn wu32(addr: u32, val: u32) void { + @as(*align(1) u32, @ptrFromInt(addr)).* = val; +} +inline fn wu8(addr: u32, val: u8) void { + @as(*u8, @ptrFromInt(addr)).* = val; +} +inline fn loadV4(ptr: u32) V4 { + return @as(*align(1) const V4, @ptrFromInt(ptr)).*; +} + +// ============================================================================= +// Emitter struct offsets (this = ECX = ParticleSystemRenderer*) +// Assembly-derived from [edi+N] references in asm_RenderParticleSprites.txt +// ============================================================================= +const E = struct { + const uvCoordScale: u32 = 0x0C; // shift count for texture V index + const texScaleU: u32 = 0x10; // texture U scale factor + const texScaleV: u32 = 0x14; // texture V scale factor + const colorCtxBase: u32 = 0xBC; // base of color/orientation data array + const rotation_offset: u32 = 0x18C; // rotation angle scale + const particle_count_mask: u32 = 0x19C; // mask for particle index extraction + const orientation_base: u32 = 0x1A8; // orientation data ptr + const flags: u32 = 0x1AC; // rendering flags (u32) + const particle_size: u32 = 0x1B0; // base particle size + const visibility: u32 = 0x1B4; // visibility threshold + const alpha_scale: u32 = 0x1B8; // alpha scale offset + const alpha_value: u32 = 0x1C0; // alpha value + const extra_scale: u32 = 0x264; // additional scale factor + const rotation_axis: u32 = 0x284; // rotation axis vec3 (for 3D rotation path) + const tail_distance: u32 = 0xB4; // tail particle max distance +}; + +// ============================================================================= +// Global addresses +// ============================================================================= +const G = struct { + const float_1_0: u32 = 0x7FF9D8; // 1.0f + const zero_threshold: u32 = 0x7FFD74; // 0.0f (collision plane zero) + const max_particle_size: u32 = 0x7FFE58; // max clamp for particle size + const rounding_magic: u32 = 0x8029CC; // float-to-byte magic number + const depth_buffer: u32 = 0xCF58F0; // g_particleDepthBuffer (128 floats) + const world_matrix: u32 = 0xCF5B68; // g_worldMatrix (4x4) + const light_dir_x: u32 = 0xCF5878; // g_lightDirectionX + const light_dir_y: u32 = 0xCF587C; // g_lightDirectionY + const light_dir_z: u32 = 0xCF5880; // g_lightDirectionZ + // Billboard vertex offset lookup tables (4 vertices × {x,y} = 8 floats each table) + const billboard_offsets_x: u32 = 0x87D714; // g_billboardVertexOffsetsX (stride 8 per vertex) + const billboard_offsets_y: u32 = 0x87D718; // g_billboardVertexOffsetsY + // 3D billboard offset table (4 vertices × {x,y,z} = 12 floats) + const billboard_3d: u32 = 0x87D738; // g_transformedVertex table (stride 8 per vertex for 2D ref) + const billboard_3d_base: u32 = 0xCF5B30; // secondary 3D table base (-4/0/+4 indexed) + // Sprite texture offset lookup (4 vertices × {u,v}) + const sprite_tex_u: u32 = 0x87D72C; // texture U offsets (stride 8) + const sprite_tex_v: u32 = 0x87D730; // texture V offsets (stride 8) + // Tail particle texture data + const tail_tex_u0: u32 = 0x87D744; // tail tex offsets per vertex + const tail_tex_v0: u32 = 0x87D748; + const tail_tex_u1: u32 = 0x87D74C; + const tail_tex_v1: u32 = 0x87D750; + const tail_threshold: u32 = 0x80C744; // minimum velocity squared for tail rendering +}; + +// ============================================================================= +// Game function pointers (called from RenderParticleSprites) +// ============================================================================= + +/// calculateParticleColorAndScale (0x7B9B10) +/// __thiscall(ECX=colorCtx, stack: time, scale, outColor, outAlpha1, outAlpha2, outFloat) +const calcColorFn = *const fn (u32, u32, u32, u32, u32, u32, u32) callconv(TC) void; +const calcColor: calcColorFn = @ptrFromInt(0x7B9B10); + +/// UpdateLightingOffset / setupRenderState (0x58A230) +/// __cdecl() → returns ptr (used to check [ret+0x1C]) +const setupRenderFn = *const fn () callconv(.{ .x86_stdcall = .{} }) u32; +const setupRender: setupRenderFn = @ptrFromInt(0x58A230); + +/// transformVector3ByMatrix4x4 (0x7BCA80) +/// __fastcall(ECX=out, EDX=vec3, stack=mat4x4ptr), RET 0x4 +const transformVec3Fn = *const fn (u32, u32, u32) callconv(FC) u32; +const transformVec3: transformVec3Fn = @ptrFromInt(0x7BCA80); + +/// createAxisAngleRotationMatrix3x3 (0x7BE490) +/// __fastcall(ECX=outMat9, EDX=axisVec3, stack=angle_f32, isNormalized_char), RET 0x8 +/// Note: angle is passed as f32 bits on stack, isNormalized as u32 (char in low byte) +const createRotMatFn = *const fn (u32, u32, u32, u32) callconv(FC) u32; +const createRotMat: createRotMatFn = @ptrFromInt(0x7BE490); + +/// transformVector4ByMatrix4x4 (0x7BCB40) +/// __fastcall(ECX=out, EDX=vec3, stack=mat4x4ptr), RET 0x4 +const transformVec4Fn = *const fn (u32, u32, u32) callconv(FC) u32; +const transformVec4: transformVec4Fn = @ptrFromInt(0x7BCB40); + +// ============================================================================= +// VertexBuffers struct — the vertexBuffers parameter +// ============================================================================= +// vertexBuffers is a float** (array of pointers): +// [0] = vertexPos ptr (3 floats per vertex: x,y,z) +// [1] = normalPtr (3 floats: light direction) +// [2] = colorPtr (1 u32: packed BGRA color) +// [3] = texCoordPtr (2 floats: u,v) +// [4] = vertexStride (bytes to advance vertex ptr) +// [5] = normalStride (bytes to advance normal ptr) +// [6] = colorStride (bytes to advance color ptr) +// [7] = texCoordStride (bytes to advance texcoord ptr) +// [8] = vertexCount (incremented per vertex emitted) +const VB = struct { + const pos: u32 = 0; + const normal: u32 = 4; + const color: u32 = 8; + const texcoord: u32 = 12; + const pos_stride: u32 = 16; + const normal_stride: u32 = 20; + const color_stride: u32 = 24; + const texcoord_stride: u32 = 28; + const count: u32 = 32; +}; + +/// Cached vertex buffer state — avoids re-reading pointer array per vertex. +/// Load once at start, emit vertices via direct pointer math, write back at end. +const VBState = struct { + pos: u32, + normal: u32, + color_ptr: u32, + texcoord: u32, + pos_stride: u32, + normal_stride: u32, + color_stride: u32, + texcoord_stride: u32, + count: u32, + vb: u32, // base pointer for writeback + // Cached light direction (same for all vertices) + light: [3]u32, + + fn load(vb: u32) VBState { + logStrides(vb); + return .{ + .pos = ru32(vb + VB.pos), + .normal = ru32(vb + VB.normal), + .color_ptr = ru32(vb + VB.color), + .texcoord = ru32(vb + VB.texcoord), + .pos_stride = ru32(vb + VB.pos_stride), + .normal_stride = ru32(vb + VB.normal_stride), + .color_stride = ru32(vb + VB.color_stride), + .texcoord_stride = ru32(vb + VB.texcoord_stride), + .count = ru32(vb + VB.count), + .vb = vb, + .light = .{ ru32(G.light_dir_x), ru32(G.light_dir_y), ru32(G.light_dir_z) }, + }; + } + + fn emit(s: *VBState, px: f32, py: f32, pz: f32, color: u32, tu: f32, tv: f32) void { + // Vertex layout is interleaved 24 bytes: xyz(12) + color(4) + uv(8). + // All strides are 24 except normal (0, shared global). + // Write contiguously when stride == 24 and layout matches. + if (s.pos_stride == 24 and s.color_ptr == s.pos + 12 and s.texcoord == s.pos + 16) { + // Fast path: contiguous 24-byte vertex. V4 store for xyz+color (16 bytes), + // then 2 scalar stores for uv (8 bytes). Unaligned V4 store via vmovups. + const xyzc = V4{ px, py, pz, @bitCast(color) }; + @as(*align(1) V4, @ptrFromInt(s.pos)).* = xyzc; + wf32(s.pos + 16, tu); + wf32(s.pos + 20, tv); + s.pos += 24; + s.color_ptr += 24; + s.texcoord += 24; + } else { + // Fallback: scattered writes + wf32(s.pos, px); + wf32(s.pos + 4, py); + wf32(s.pos + 8, pz); + wu32(s.color_ptr, color); + wf32(s.texcoord, tu); + wf32(s.texcoord + 4, tv); + s.pos += s.pos_stride; + s.color_ptr += s.color_stride; + s.texcoord += s.texcoord_stride; + } + // Normal: stride=0 means shared global, write once (handled in writeback) + if (s.normal_stride != 0) { + wu32(s.normal, s.light[0]); + wu32(s.normal + 4, s.light[1]); + wu32(s.normal + 8, s.light[2]); + s.normal += s.normal_stride; + } + s.count += 1; + } + + fn writeback(s: *const VBState) void { + // Write normal once if stride==0 (shared global — same for all vertices) + if (s.normal_stride == 0) { + wu32(s.normal, s.light[0]); + wu32(s.normal + 4, s.light[1]); + wu32(s.normal + 8, s.light[2]); + } + wu32(s.vb + VB.pos, s.pos); + wu32(s.vb + VB.normal, s.normal); + wu32(s.vb + VB.color, s.color_ptr); + wu32(s.vb + VB.texcoord, s.texcoord); + wu32(s.vb + VB.count, s.count); + } +}; + +/// Emit one vertex using the old pointer-chasing path (for code paths not yet converted to VBState). +inline fn emitVertex(vb: u32, px: f32, py: f32, pz: f32, color: u32, tu: f32, tv: f32) void { + const pos_ptr = ru32(vb + VB.pos); + wf32(pos_ptr, px); + wf32(pos_ptr + 4, py); + wf32(pos_ptr + 8, pz); + const norm_ptr = ru32(vb + VB.normal); + wu32(norm_ptr, ru32(G.light_dir_x)); + wu32(norm_ptr + 4, ru32(G.light_dir_y)); + wu32(norm_ptr + 8, ru32(G.light_dir_z)); + wu32(ru32(vb + VB.color), color); + const tc_ptr = ru32(vb + VB.texcoord); + wf32(tc_ptr, tu); + wf32(tc_ptr + 4, tv); + wu32(vb + VB.count, ru32(vb + VB.count) + 1); + wu32(vb + VB.pos, ru32(vb + VB.pos) + ru32(vb + VB.pos_stride)); + wu32(vb + VB.normal, ru32(vb + VB.normal) + ru32(vb + VB.normal_stride)); + wu32(vb + VB.color, ru32(vb + VB.color) + ru32(vb + VB.color_stride)); + wu32(vb + VB.texcoord, ru32(vb + VB.texcoord) + ru32(vb + VB.texcoord_stride)); +} + +// Cached render state — setupRender() returns the same pointer all frame. +// Reset each frame via resetParticleCache() called from the frame hook. +var cached_render_state: u32 = 0; + +var stride_logged: bool = false; +var debug_logged: bool = false; +export var debug_vertex_count: u32 = 0; +export var debug_max_sprites: u32 = 0; +export var debug_fmt_index: u32 = 0; +export var debug_data_ptr: u32 = 0; + +/// Reset per-frame caches. Call from OnWorldUpdate or executeSceneRenderPass hook. +export fn resetParticleCache() void { + cached_render_state = 0; +} + +/// Log VB strides once for analysis. Called from first VBState.load. +fn logStrides(vb: u32) void { + if (stride_logged) return; + stride_logged = true; + // Write to a known memory location that the profiler can dump, or just use + // the debug console. For now, store in a global we can read. + stride_info = .{ + ru32(vb + VB.pos_stride), + ru32(vb + VB.normal_stride), + ru32(vb + VB.color_stride), + ru32(vb + VB.texcoord_stride), + ru32(vb + VB.pos), + ru32(vb + VB.normal), + ru32(vb + VB.color), + ru32(vb + VB.texcoord), + }; +} + +export var stride_info: [8]u32 = .{0} ** 8; + +// ============================================================================= +// RenderParticleSprites (0x7B2A50) +// __thiscall(ECX=emitter, stack=particleData, vertexBuffers), RET 0x8 +// Returns: 0 (culled) or 1 (rendered) +// +// Faithful recreation from assembly + Ghidra decompilation. +// ============================================================================= +export fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 { + const pd = particle_data; // particleData pointer (float*) + const vb = vertex_buffers; // vertexBuffers pointer (float**) + + // ========================================================================= + // Section 1: Early-out visibility checks (asm 0x7B2A5E-0x7B2B0B) + // ========================================================================= + + // Check visibility threshold: emitter+0x1B4 < 1.0 + var depth_index: u32 = 0; + + if (rf32(emitter + E.visibility) < rf32(G.float_1_0) or + rf32(emitter + E.alpha_value) != rf32(G.zero_threshold)) + { + // Compute clamped particle size + var clamped_size: f32 = rf32(emitter + E.particle_size) * rf32(pd + 0x1C); + if (clamped_size < rf32(G.zero_threshold)) { + clamped_size = rf32(G.zero_threshold); + } else if (clamped_size >= rf32(G.max_particle_size)) { + clamped_size = rf32(G.max_particle_size); + } + // Float-to-index conversion: add magic, extract bits, combine with particle data hash + const size_with_magic = clamped_size + rf32(G.rounding_magic); + depth_index = ((@as(u32, @bitCast(size_with_magic)) >> 14) + (particle_data >> 5)) & 0x7F; + } + + // Depth buffer cull check + if (rf32(emitter + E.visibility) < rf32(G.float_1_0) and + rf32(emitter + E.visibility) < rf32(G.depth_buffer + depth_index * 4)) + { + return 0; + } + + // ========================================================================= + // Section 2: Calculate color and scale (asm 0x7B2B0E-0x7B2B41) + // ========================================================================= + + // Compute colorCtx address: emitter + 0xBC + byte(particleData[0x0C]) * 96 + // Assembly: movzx eax,byte[ebx+0xC]; lea ecx,[eax+eax*2]; shl ecx,5; lea ecx,[ecx+edi+0xBC] + const color_ctx_offset: u32 = @as(u32, ru8(pd + 0x0C)) * 96; + const color_ctx = emitter + E.colorCtxBase + color_ctx_offset; + + // Inline calcColor: compute color, alpha, and sprite scale from colorCtx + // Original at 0x7B9B10, assembly-verified. Inlined to allow OoO overlap with cache misses. + const scale_param: f32 = @bitCast(ru32(emitter + E.orientation_base)); // arg2: float scale for alpha + const time_val: f32 = rf32(pd + 0x1C); + + // t = (time - ctx.timeBase) * ctx.timeScale * CONST1 + CONST2 + const t = (time_val - rf32(color_ctx + 0x2C)) * rf32(color_ctx + 0x30) * rf32(0x808AAC) + rf32(0x807A3C); + const magic: f32 = rf32(G.rounding_magic); + + // Color channels: (float)delta * t + (float)base [+ magic], extract byte via >>14 + // Alpha (byte 3): scaled by scale_param + const alpha_f = @mulAdd(f32, @as(f32, @floatFromInt(ri32(color_ctx + 0x04))), t, + @as(f32, @floatFromInt(@as(i32, ru8(color_ctx + 3))))) * scale_param + magic; + // Red (byte 2): no scale + const red_f = @mulAdd(f32, @as(f32, @floatFromInt(ri32(color_ctx + 0x08))), t, + @as(f32, @floatFromInt(@as(i32, ru8(color_ctx + 2))))) + magic; + // Green (byte 1): + const green_f = @mulAdd(f32, @as(f32, @floatFromInt(ri32(color_ctx + 0x0C))), t, + @as(f32, @floatFromInt(@as(i32, ru8(color_ctx + 1))))) + magic; + // Blue (byte 0): + const blue_f = @mulAdd(f32, @as(f32, @floatFromInt(ri32(color_ctx + 0x10))), t, + @as(f32, @floatFromInt(@as(i32, ru8(color_ctx + 0))))) + magic; + + var color_value: u32 = ((@as(u32, @bitCast(blue_f)) >> 14) & 0xFF) | + (((@as(u32, @bitCast(green_f)) >> 14) & 0xFF) << 8) | + (((@as(u32, @bitCast(red_f)) >> 14) & 0xFF) << 16) | + (((@as(u32, @bitCast(alpha_f)) >> 14) & 0xFF) << 24); + + // Sprite scale: t * ctx.scaleDelta + ctx.scaleBase + var sprite_scale: f32 = @mulAdd(f32, t, rf32(color_ctx + 0x28), rf32(color_ctx + 0x24)); + + // Alpha outputs (color_data1, color_data2) — used for texture index + var color_data1: u32 = undefined; + var color_data2: u32 = undefined; + const alpha_power = ru32(color_ctx + 0x50); + if (alpha_power == 0x3F800000) { + // Fast path: alphaPower == 1.0 (linear) + color_data1 = (@as(u32, @bitCast(@mulAdd(f32, @as(f32, @floatFromInt(ri32(color_ctx + 0x18))), t, + @as(f32, @floatFromInt(ri32(color_ctx + 0x14)))) + magic)) >> 14) & 0xFF; + color_data2 = (@as(u32, @bitCast(@mulAdd(f32, @as(f32, @floatFromInt(ri32(color_ctx + 0x20))), t, + @as(f32, @floatFromInt(ri32(color_ctx + 0x1C)))) + magic)) >> 14) & 0xFF; + } else { + // Slow path: pow scaling — fall back to game function call + calcColor(color_ctx, @bitCast(time_val), @bitCast(ru32(emitter + E.orientation_base)), + @intFromPtr(&color_value), @intFromPtr(&color_data1), @intFromPtr(&color_data2), @intFromPtr(&sprite_scale)); + } + + // ========================================================================= + // Section 3: Render state setup (asm 0x7B2B46) + // Cached: the render state pointer doesn't change within a frame. + // ========================================================================= + + const render_state = blk: { + if (cached_render_state != 0) break :blk cached_render_state; + const rs = setupRender(); + cached_render_state = rs; + break :blk rs; + }; + + // ========================================================================= + // Section 4: Color byte swizzle (asm 0x7B2B4B-0x7B2B6E) + // If render_state[0x1C] == 1, swizzle BGRA → RGBA + // ========================================================================= + + if (ru32(render_state + 0x1C) == 1) { + const b0: u8 = @truncate(color_value); + const b1: u8 = @truncate(color_value >> 8); + const b2: u8 = @truncate(color_value >> 16); + const b3: u8 = @truncate(color_value >> 24); + color_value = @as(u32, b2) | (@as(u32, b0) << 8) | (@as(u32, b3) << 16) | (@as(u32, b1) << 24); + } + + // ========================================================================= + // Section 5: Alpha/size scaling (asm 0x7B2B71-0x7B2BB1) + // ========================================================================= + + if (rf32(emitter + E.alpha_value) != rf32(G.zero_threshold)) { + sprite_scale = (rf32(G.depth_buffer + depth_index * 4) * rf32(emitter + E.alpha_value) + + rf32(emitter + E.alpha_scale)) * sprite_scale; + } + + // Read full flags as u32 for subsequent checks + const full_flags = ru32(emitter + E.flags); + + // Extra scale factor if flag 0x200 set + if ((full_flags & 0x200) != 0) { + sprite_scale = sprite_scale * rf32(emitter + E.extra_scale); + } + + // ========================================================================= + // Section 6: Position transform (asm 0x7B2BB4-0x7B2BC3) + // Inline V4 mat*vec3: result = col0*v.x + col1*v.y + col2*v.z + col3 + // ========================================================================= + + const pp: [*]const f32 = @ptrFromInt(pd); + const pvx: V4 = @splat(pp[0]); + const pvy: V4 = @splat(pp[1]); + const pvz: V4 = @splat(pp[2]); + const m: u32 = G.world_matrix; + const wp = @mulAdd(V4, pvz, loadV4(m + 32), @mulAdd(V4, pvy, loadV4(m + 16), @mulAdd(V4, pvx, loadV4(m), loadV4(m + 48)))); + const world_pos = [3]f32{ wp[0], wp[1], wp[2] }; + + // ========================================================================= + // Section 7: Branch on flag 0x4 — sprite vs tail rendering + // ========================================================================= + + if ((full_flags & 0x4) == 0) { + // No sprite rendering — jump to tail check at section 9 + } else { + // ===================================================================== + // Section 7a: Texture coordinate setup (asm 0x7B2BD5-0x7B2C05) + // ===================================================================== + + const count_mask = ru32(emitter + E.particle_count_mask) - 1; + const tex_index_raw = color_data1; + const tex_u_index: f32 = @floatFromInt(count_mask & tex_index_raw); + const shift_count: u5 = @truncate(ru32(emitter + E.uvCoordScale)); + const tex_v_raw: i32 = @as(i32, @bitCast(tex_index_raw)) >> shift_count; + const tex_v_index: f32 = @floatFromInt(tex_v_raw); + + const tex_u_base = tex_u_index * rf32(emitter + E.texScaleU); + const tex_v_base = tex_v_index * rf32(emitter + E.texScaleV); + const tex_scale_u = rf32(emitter + E.texScaleU); + const tex_scale_v = rf32(emitter + E.texScaleV); + + // Check rotation angle: if emitter+0x18C == 0.0, no rotation needed + const has_rotation = rf32(emitter + E.rotation_offset) != rf32(G.zero_threshold); + + if (!has_rotation) { + // ================================================================= + // Section 8a: No rotation — check 2D vs 3D billboard + // ================================================================= + + if ((full_flags & 0x2000) == 0) { + // --- 2D billboard (asm 0x7B2D10-0x7B2DD5) --- + // 4 vertices. Position uses [eax+0x87D714/718], but eax is incremented + // by 8 BEFORE the Y read and texcoord reads. So texcoords use eax+8. + // Assembly: eax starts at 0, adds 8 between X and Y reads. + // X: [eax+0x87D714], eax+=8, Y: [eax+0x87D710]=[eax_new+0x87D710] + // texU: [eax+0x87D72C], texV: [eax+0x87D730] (eax already incremented) + // Unrolled — inline for lets LLVM schedule stores across vertices. + { + var vs = VBState.load(vb); + const wpx = world_pos[0]; + const wpy = world_pos[1]; + const wpz = world_pos[2]; + inline for (0..4) |i| { + const off: u32 = @intCast(i * 8); + vs.emit( + @mulAdd(f32, sprite_scale, rf32(G.billboard_offsets_x + off), wpx), + @mulAdd(f32, sprite_scale, rf32(G.billboard_offsets_y + off), wpy), + wpz, + color_value, + @mulAdd(f32, rf32(G.sprite_tex_u + off + 8), tex_scale_u, tex_u_base), + @mulAdd(f32, rf32(G.sprite_tex_v + off + 8), tex_scale_v, tex_v_base), + ); + } + vs.writeback(); + } + } else { + // --- 3D billboard (asm 0x7B2C25-0x7B2D04) --- + { + var vs = VBState.load(vb); + const table_base: u32 = G.billboard_3d; + const ref_base: u32 = G.billboard_3d_base; + var vert: u32 = 0; + while (vert < 4) : (vert += 1) { + const tbl = ref_base + vert * 12; + vs.emit( + @mulAdd(f32, sprite_scale, rf32(tbl - 4), world_pos[0]), + @mulAdd(f32, sprite_scale, rf32(tbl), world_pos[1]), + @mulAdd(f32, sprite_scale, rf32(tbl + 4), world_pos[2]), + color_value, + @mulAdd(f32, rf32(table_base + vert * 8 - 4), tex_scale_u, tex_u_base), + @mulAdd(f32, rf32(table_base + vert * 8), tex_scale_v, tex_v_base), + ); + } + vs.writeback(); + } + } + } else { + // ================================================================= + // Section 8b: With rotation + // ================================================================= + + // Compute rotation angle: emitter+0x18C * particleData[7] + var rot_angle = rf32(emitter + E.rotation_offset) * rf32(pd + 0x1C); + + // Negate if flags indicate (asm 0x7B2DE8-0x7B2DF4) + const flag_byte: i8 = @bitCast(@as(u8, @truncate(full_flags >> 8))); + if (flag_byte < 0 and (particle_data & 0x20) != 0) { + rot_angle = -rot_angle; + } + + if ((full_flags & 0x2000) == 0) { + // --- 2D billboard with sin/cos rotation (asm 0x7B2F49-0x7B303B) --- + const cos_val = @cos(rot_angle); + const sin_val = @sin(rot_angle); + const scaled_sin = sin_val * sprite_scale; + const scaled_cos = cos_val * sprite_scale; + + { + var vs = VBState.load(vb); + const wpx = world_pos[0]; + const wpy = world_pos[1]; + const wpz = world_pos[2]; + inline for (0..4) |i| { + const off: u32 = @intCast(i * 8); + const ox = rf32(G.billboard_offsets_x + off); + const oy = rf32(G.billboard_offsets_y + off); + vs.emit( + @mulAdd(f32, ox, scaled_cos, wpx) - oy * scaled_sin, + @mulAdd(f32, oy, scaled_cos, @mulAdd(f32, ox, scaled_sin, wpy)), + wpz, + color_value, + @mulAdd(f32, rf32(G.sprite_tex_u + off + 8), tex_scale_u, tex_u_base), + @mulAdd(f32, rf32(G.sprite_tex_v + off + 8), tex_scale_v, tex_v_base), + ); + } + vs.writeback(); + } + } else { + // --- 3D billboard with rotation matrix (asm 0x7B2E00-0x7B2F41) --- + // Build rotation matrix from axis + angle, then transform each vertex + var rot_mat: [9]f32 = undefined; + _ = createRotMat(@intFromPtr(&rot_mat), emitter + E.rotation_axis, + @bitCast(rot_angle), 1); + + { + var vs = VBState.load(vb); + const ref_base: u32 = G.billboard_3d_base; + const tex_off_base: u32 = G.billboard_3d; + var vert: u32 = 0; + while (vert < 4) : (vert += 1) { + const tbl = ref_base + vert * 12; + const ix = rf32(tbl - 4); + const iy = rf32(tbl); + const iz = rf32(tbl + 4); + // mat3x3 * vec3, scaled, + worldPos + vs.emit( + @mulAdd(f32, rot_mat[2], iz, @mulAdd(f32, rot_mat[1], iy, rot_mat[0] * ix)) * sprite_scale + world_pos[0], + @mulAdd(f32, rot_mat[5], iz, @mulAdd(f32, rot_mat[4], iy, rot_mat[3] * ix)) * sprite_scale + world_pos[1], + @mulAdd(f32, rot_mat[8], iz, @mulAdd(f32, rot_mat[7], iy, rot_mat[6] * ix)) * sprite_scale + world_pos[2], + color_value, + @mulAdd(f32, rf32(tex_off_base + vert * 8 - 4), tex_scale_u, tex_u_base), + @mulAdd(f32, rf32(tex_off_base + vert * 8), tex_scale_v, tex_v_base), + ); + } + vs.writeback(); + } + } + } + } + + // ========================================================================= + // Section 9: Tail particle rendering (asm 0x7B3041-0x7B34C5) + // Flag 0x8 in emitter+0x1AC: velocity-based trail + // ========================================================================= + + if ((ru8(emitter + E.flags) & 0x8) != 0) { + // Tail particles: compute from velocity direction + const count_mask = ru32(emitter + E.particle_count_mask) - 1; + const tex_index_raw = color_data2; + const tex_u_index: f32 = @floatFromInt(count_mask & tex_index_raw); + const shift_count: u5 = @truncate(ru32(emitter + E.uvCoordScale)); + const tex_v_raw: i32 = @as(i32, @bitCast(tex_index_raw)) >> shift_count; + const tail_tex_u = tex_u_index * rf32(emitter + E.texScaleU); + const tail_tex_v: f32 = @as(f32, @floatFromInt(tex_v_raw)) * rf32(emitter + E.texScaleV); + + // Negate velocity vector + const neg_vel_x: f32 = -rf32(pd + 0x10); // particleData[4] + const neg_vel_y: f32 = -rf32(pd + 0x14); // particleData[5] + const neg_vel_z: f32 = -rf32(pd + 0x18); // particleData[6] + + // Get tail distance, clamp by particleData[7] if flag 0x1 set + var tail_dist: f32 = @bitCast(ru32(emitter + E.tail_distance)); + const tail_flag_byte = ru8(emitter + E.flags + 2); // byte at +0x1AE + if ((tail_flag_byte & 0x1) != 0 and rf32(pd + 0x1C) < tail_dist) { + tail_dist = rf32(pd + 0x1C); + } + + // Transform negated velocity through world matrix + var neg_vel = [3]f32{ neg_vel_x, neg_vel_y, neg_vel_z }; + var transformed_vel: [4]f32 = undefined; + _ = transformVec4(@intFromPtr(&transformed_vel), @intFromPtr(&neg_vel), G.world_matrix); + + const tx = tail_dist * transformed_vel[0]; + const ty = tail_dist * transformed_vel[1]; + const cos_sq = tx * tx + ty * ty; + + if (cos_sq >= rf32(G.tail_threshold)) { + // Velocity-based trail: 4 vertices forming a quad along velocity direction + const vel_z = tail_dist * transformed_vel[2] + world_pos[2]; + const inv_len = sprite_scale / @sqrt(cos_sq); + const perp_x = tx * inv_len; + const perp_y = inv_len * ty; + + const tex_su = rf32(emitter + E.texScaleU); + const tex_sv = rf32(emitter + E.texScaleV); + + var vs = VBState.load(vb); + vs.emit(world_pos[0] - perp_y, perp_x + world_pos[1], world_pos[2], color_value, + @mulAdd(f32, rf32(G.sprite_tex_u), tex_su, tail_tex_u), + @mulAdd(f32, rf32(G.sprite_tex_v), tex_sv, tail_tex_v)); + vs.emit(world_pos[0] + perp_y, world_pos[1] - perp_x, world_pos[2], color_value, + @mulAdd(f32, rf32(G.sprite_tex_u), tex_su, tail_tex_u), + @mulAdd(f32, rf32(G.sprite_tex_v), tex_sv, tail_tex_v)); + vs.emit(tx + world_pos[0] - perp_y, ty + world_pos[1] + perp_x, vel_z, color_value, + @mulAdd(f32, rf32(G.tail_tex_u0), tex_su, tail_tex_u), + @mulAdd(f32, rf32(G.tail_tex_v0), tex_sv, tail_tex_v)); + vs.emit(tx + world_pos[0] + perp_y, ty + world_pos[1] - perp_x, vel_z, color_value, + @mulAdd(f32, rf32(G.tail_tex_u1), tex_su, tail_tex_u), + @mulAdd(f32, rf32(G.tail_tex_v1), tex_sv, tail_tex_v)); + vs.writeback(); + return 1; + } + + // Fallback: velocity too small for trail, render as flat billboard + { + var vs = VBState.load(vb); + const tex_su = rf32(emitter + E.texScaleU); + const tex_sv = rf32(emitter + E.texScaleV); + var loop_off: u32 = 0; + while (loop_off < 0x20) : (loop_off += 8) { + vs.emit( + @mulAdd(f32, sprite_scale, rf32(G.billboard_offsets_x + loop_off), world_pos[0]), + @mulAdd(f32, sprite_scale, rf32(G.billboard_offsets_y + loop_off), world_pos[1]), + world_pos[2], + color_value, + @mulAdd(f32, rf32(G.sprite_tex_u + loop_off + 8), tex_su, tail_tex_u), + @mulAdd(f32, rf32(G.sprite_tex_v + loop_off + 8), tex_sv, tail_tex_v), + ); + } + vs.writeback(); + } + } + + return 1; +} + +// ============================================================================= +// Game function pointers for SetupParticleRendering +// ============================================================================= + +const SC = std.builtin.CallingConvention; +const StdCall: SC = .{ .x86_stdcall = .{} }; + +// 0x58B0B0: SetTransformMatrix — __thiscall(ECX=matrixPtr) +const gameSetTransformMatrix: *const fn (u32) callconv(TC) void = @ptrFromInt(0x58B0B0); +// 0x58B050: SetVertexShader — __thiscall(ECX=matrixPtr) +const gameSetVertexShader: *const fn (u32) callconv(TC) void = @ptrFromInt(0x58B050); +// 0x7BC6A0: multiplyMatrix4x4 — __fastcall(ECX=out, EDX=matA, stack=matB), RET 0x4, returns out +const gameMatMul: *const fn (u32, u32, u32) callconv(FC) u32 = @ptrFromInt(0x7BC6A0); +// 0x409AEF: validateMemoryOperation — __thiscall(ECX=ptr) +const gameValidateMem: *const fn (u32) callconv(TC) void = @ptrFromInt(0x409AEF); +// 0x4549F0: vec3SquaredMagnitude — __thiscall(ECX=vec3ptr), returns f64 in ST(0) +// Can't call directly from Zig due to FPU return. Use inline asm. +// All calling conventions verified from assembly at each CALL site. +// 0x589F40: BeginRender — no params visible before call +const gameBeginRender: *const fn () callconv(StdCall) void = @ptrFromInt(0x589F40); +// 0x44ACF0: GetTextureBuffer — __fastcall(ECX=texDataPtr, EDX=0, stack=0), returns ptr in EAX +const gameGetTexture: *const fn (u32, u32, u32) callconv(FC) u32 = @ptrFromInt(0x44ACF0); +// 0x589E80: SetTexture — __fastcall(ECX=slot, EDX=texturePtr) +const gameSetTexture: *const fn (u32, u32) callconv(FC) void = @ptrFromInt(0x589E80); +// 0x589A90: GetDataPointerByIndex — __thiscall(ECX=index), returns ptr +const gameGetDataPtr: *const fn (u32) callconv(TC) u32 = @ptrFromInt(0x589A90); +// 0x58A140: CreateVertexBuffer — __fastcall(ECX=0, EDX=dataPtr, stack=count), returns ptr +const gameCreateVB: *const fn (u32, u32, u32) callconv(FC) u32 = @ptrFromInt(0x58A140); +// 0x58A080: LockVertexBuffer — __thiscall(ECX=vbPtr), returns base offset +const gameLockVB: *const fn (u32) callconv(TC) u32 = @ptrFromInt(0x58A080); +// 0x589AB0: GetMatrixElementPointer — __fastcall(ECX=fmtIndex, EDX=elementIndex), returns ptr +const gameGetMatElem: *const fn (u32, u32) callconv(FC) u32 = @ptrFromInt(0x589AB0); +// 0x7B3A10: RenderParticleSystemSorted — __thiscall(ECX=emitter, stack=vbPtrs) +const gameRenderSorted: *const fn (u32, u32) callconv(TC) void = @ptrFromInt(0x7B3A10); +// 0x58A0A0: UnlockVertexBuffer — __fastcall(ECX=vbPtr, EDX=0) +const gameUnlockVB: *const fn (u32, u32) callconv(FC) void = @ptrFromInt(0x58A0A0); +// 0x58A7C0: DrawPrimitive — __fastcall(ECX=vbPtr, EDX=fmtIndex) +const gameDrawPrim: *const fn (u32, u32) callconv(FC) void = @ptrFromInt(0x58A7C0); +// 0x58A010: IsObjectActiveAndValid — __thiscall(ECX=objPtr), returns bool-like +const gameIsObjValid: *const fn (u32) callconv(TC) u32 = @ptrFromInt(0x58A010); +// 0x7B3C50: BuildIndexBuffer — __thiscall(ECX=emitter, stack=ibPtr, count) +// Actually: PUSH edx(count), PUSH ecx(ibPtr), mov ecx,ebx(emitter), CALL +const gameBuildIB: *const fn (u32, u32, u32) callconv(TC) void = @ptrFromInt(0x7B3C50); +// 0x58A800: SetStreamSource — __thiscall(ECX=ibPtr) +const gameSetStream: *const fn (u32) callconv(TC) void = @ptrFromInt(0x58A800); +// 0x58A830: CallGfxDeviceMethod_Wrapper — __fastcall(ECX=paramsPtr, EDX=param2) +const gameGfxCall: *const fn (u32, u32) callconv(FC) void = @ptrFromInt(0x58A830); +// 0x589F50: EndRender — no params +const gameEndRender: *const fn () callconv(StdCall) void = @ptrFromInt(0x589F50); + +// ============================================================================= +// Global addresses for SetupParticleRendering +// ============================================================================= +const SG = struct { + const world_matrix: u32 = 0xCF5B68; // g_worldMatrix (64 bytes, 4x4) + const light_dir_x: u32 = 0xCF5878; + const light_dir_y: u32 = 0xCF587C; + const light_dir_z: u32 = 0xCF5880; + const render_init_flags: u32 = 0xCF58EC; + const sprite_vertex_template: u32 = 0xCF5AF8; // 4 vertices × 3 floats = 48 bytes + const billboard_matrix: u32 = 0xCF5888; // 4x4 matrix (64 bytes, 0xCF5888-0xCF58C8) + const sprite_template_validator: u32 = 0xCF5B28; // for validateMemoryOperation + const billboard_validator: u32 = 0xCF58E8; // for validateMemoryOperation + const normal_validator: u32 = 0xCF586C; // for validateMemoryOperation + const default_normal: u32 = 0xCF5860; // 3 floats + const max_particle_sprites: u32 = 0xCF5B60; // u32 + const transformed_vertices: u32 = 0xCF5B30; // output of billboard transform (48 bytes) + const index_buffer_6: u32 = 0xCF5BAC; // ptr to index buffer for field_28==6 + const index_buffer_12: u32 = 0xCF5AF4; // ptr to index buffer for field_28==0xC + const billboard_epsilon: u32 = 0x8029D4; +}; + +// ============================================================================= +// SetupParticleRendering (0x7B3D20) +// __thiscall(ECX=emitter, stack=viewMatrix), RET 0x4 +// viewMatrix can be NULL. +// +// Faithful recreation from Ghidra decompilation + assembly. +// All game function calls preserved, matrix math inlined with V4. +// ============================================================================= +export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void { + // ========================================================================= + // Section 1: Identity matrices for render state + // Optimization: use static identity instead of rebuilding on stack each call. + // ========================================================================= + // Must be mutable — game functions may write to the matrix pointer + var identity_a = [16]u32{ + 0x3F800000, 0, 0, 0, + 0, 0x3F800000, 0, 0, + 0, 0, 0x3F800000, 0, + 0, 0, 0, 0x3F800000, + }; + var identity_b = [16]u32{ + 0x3F800000, 0, 0, 0, + 0, 0x3F800000, 0, 0, + 0, 0, 0x3F800000, 0, + 0, 0, 0, 0x3F800000, + }; + + gameSetTransformMatrix(@intFromPtr(&identity_a)); + gameSetVertexShader(@intFromPtr(&identity_b)); + + // ========================================================================= + // Section 2: Build translation matrix = identity with last row = (-x, -y, -z, 1) + // ========================================================================= + const neg_x = -rf32(emitter + 0x23C); + const neg_y = -rf32(emitter + 0x240); + const neg_z = -rf32(emitter + 0x244); + + var translation = [16]u32{ + 0x3F800000, 0, 0, 0, + 0, 0x3F800000, 0, 0, + 0, 0, 0x3F800000, 0, + @bitCast(neg_x), @bitCast(neg_y), @bitCast(neg_z), 0x3F800000, + }; + + const flags = ru32(emitter + 0x1AC); + + // ========================================================================= + // Section 3: Compute g_worldMatrix based on flags + // Three paths: flag 0x100 set, flag clear + viewMatrix != NULL, flag clear + NULL + // ========================================================================= + if ((flags & 0x100) != 0) { + // Path A: matmul(emitter_matrix × translation), then × identity (= just copy) + // emitter_matrix at emitter+0x1FC + var temp: [16]u32 = undefined; + _ = gameMatMul(@intFromPtr(&temp), emitter + 0x1FC, @intFromPtr(&translation)); + // Original does matmul(result, temp, identity) — identity is a no-op, just copy + copyMat4x4(SG.world_matrix, @intFromPtr(&temp)); + } else if (view_matrix != 0) { + // Path B: matmul(viewMatrix × translation), then × identity (= just copy) + var temp: [16]u32 = undefined; + _ = gameMatMul(@intFromPtr(&temp), view_matrix, @intFromPtr(&translation)); + copyMat4x4(SG.world_matrix, @intFromPtr(&temp)); + } else { + // Path C: matmul(translation × identity) = just copy translation + copyMat4x4(SG.world_matrix, @intFromPtr(&translation)); + } + + // ========================================================================= + // Section 4: Set light direction from identity row 2 = (0, 0, 1) + // (Original reads from identity matrix on stack; we know it's always (0,0,1)) + // ========================================================================= + // Actually, identity matrix row 2 in the stack layout: the identity at [ebp-0x54] + // has row 2 = {0, 0, 1, 0} stored at [ebp-0x34, -0x30, -0x2c, -0x28]. + // But this identity was passed to SetVertexShader which may have modified it? + // No — SetVertexShader just reads it. So light dir = identity[8,9,10] = (0, 0, 1). + // But wait: assembly shows mov eax,[ebp-0x34]; mov [0xCF5878],eax etc. + // [ebp-0x34] is identityMatrix.m20 = 0.0, [ebp-0x30] = m21 = 0.0, [ebp-0x2c] = m22 = 1.0 + wu32(SG.light_dir_x, 0); // 0.0 + wu32(SG.light_dir_y, 0); // 0.0 + wu32(SG.light_dir_z, 0x3F800000); // 1.0 + + // ========================================================================= + // Section 5: Flag 0x2000 — billboard/3D sprite setup + // ========================================================================= + if ((flags & 0x2000) != 0) { + // One-time sprite vertex template initialization + const init_flags = ru8(SG.render_init_flags); + if ((init_flags & 1) == 0) { + wu8(SG.render_init_flags, init_flags | 1); + // Write 4 sprite vertices: {x, y, z} × 4 + // Vertex 0: (-1, 1, 0), Vertex 1: (-1, -1, 0), Vertex 2: (1, 1, 0), Vertex 3: (1, -1, 0) + wu32(SG.sprite_vertex_template + 0, 0xBF800000); // -1.0 + wu32(SG.sprite_vertex_template + 4, 0x3F800000); // 1.0 + wu32(SG.sprite_vertex_template + 8, 0); // 0.0 + wu32(SG.sprite_vertex_template + 12, 0xBF800000); // -1.0 + wu32(SG.sprite_vertex_template + 16, 0xBF800000); // -1.0 + wu32(SG.sprite_vertex_template + 20, 0); // 0.0 + wu32(SG.sprite_vertex_template + 24, 0x3F800000); // 1.0 + wu32(SG.sprite_vertex_template + 28, 0x3F800000); // 1.0 + wu32(SG.sprite_vertex_template + 32, 0); // 0.0 + wu32(SG.sprite_vertex_template + 36, 0x3F800000); // 1.0 + wu32(SG.sprite_vertex_template + 40, 0xBF800000); // -1.0 + wu32(SG.sprite_vertex_template + 44, 0); // 0.0 + gameValidateMem(SG.sprite_template_validator); + } + + // One-time billboard identity matrix initialization + if ((init_flags & 2) == 0) { + wu8(SG.render_init_flags, ru8(SG.render_init_flags) | 2); + // Write identity 4x4 to billboard_matrix + const bm = SG.billboard_matrix; + inline for (0..16) |i| { + const is_diag = (i % 5 == 0 and i < 16); + wu32(bm + @as(u32, @intCast(i)) * 4, if (is_diag) @as(u32, 0x3F800000) else 0); + } + gameValidateMem(SG.billboard_validator); + } + + // Compute billboard matrix: depends on flag 0x100 + if ((flags & 0x100) == 0) { + // matmul(emitter+0x1FC, g_worldMatrix) → billboard_matrix + var temp2: [16]u32 = undefined; + _ = gameMatMul(@intFromPtr(&temp2), emitter + 0x1FC, SG.world_matrix); + copyMat4x4(SG.billboard_matrix, @intFromPtr(&temp2)); + } else { + // Just copy g_worldMatrix → billboard_matrix + copyMat4x4(SG.billboard_matrix, SG.world_matrix); + } + + // Transform 4 sprite vertices through billboard matrix + // 4 vertices × vec3, output to g_transformedVertices + { + const bm = SG.billboard_matrix; + const bm00 = rf32(bm); const bm01 = rf32(bm + 4); const bm02 = rf32(bm + 8); + const bm10 = rf32(bm + 16); const bm11 = rf32(bm + 20); const bm12 = rf32(bm + 24); + const bm20 = rf32(bm + 32); const bm21 = rf32(bm + 36); const bm22 = rf32(bm + 40); + + var vi: u32 = 0; + while (vi < 48) : (vi += 12) { + const sx = rf32(SG.sprite_vertex_template + vi); + const sy = rf32(SG.sprite_vertex_template + vi + 4); + const sz = rf32(SG.sprite_vertex_template + vi + 8); + wf32(SG.transformed_vertices + vi, @mulAdd(f32, bm20, sz, @mulAdd(f32, bm10, sy, bm00 * sx))); + wf32(SG.transformed_vertices + vi + 4, @mulAdd(f32, bm21, sz, @mulAdd(f32, bm11, sy, bm01 * sx))); + wf32(SG.transformed_vertices + vi + 8, @mulAdd(f32, bm22, sz, @mulAdd(f32, bm12, sy, bm02 * sx))); + } + } + + // Store billboard matrix row 2 as rotation axis in emitter+0x284 + wf32(emitter + 0x284, rf32(SG.billboard_matrix + 32)); + wf32(emitter + 0x288, rf32(SG.billboard_matrix + 36)); + wf32(emitter + 0x28C, rf32(SG.billboard_matrix + 40)); + + // Normalize the rotation axis + const ax = rf32(emitter + 0x284); + const ay = rf32(emitter + 0x288); + const az = rf32(emitter + 0x28C); + const sq_mag = @mulAdd(f32, az, az, @mulAdd(f32, ay, ay, ax * ax)); + const epsilon = rf32(SG.billboard_epsilon); + if (@sqrt(sq_mag) >= epsilon) { + const inv_len = 1.0 / @sqrt(sq_mag); + wf32(emitter + 0x284, ax * inv_len); + wf32(emitter + 0x288, ay * inv_len); + wf32(emitter + 0x28C, az * inv_len); + } + } + + // ========================================================================= + // Section 6: Begin render, texture, vertex buffer setup + // ========================================================================= + gameBeginRender(); + + const tex_id = ru32(emitter + 0x1A0); + const tex_ptr = gameGetTexture(tex_id, 0, 0); + if (tex_ptr == 0) { + // No texture — skip to end + gameEndRender(); + gameSetVertexShader(@intFromPtr(&identity_a)); + return; + } + + gameSetTexture(0x17, tex_ptr); + + // Compute max particle sprites: 0x4000 / emitter.vertexSize + const vert_size = ru32(emitter + 0x9C); + var max_sprites: u32 = 0x4000 / vert_size; + const emitter_max = ru32(emitter + 0x64); + if (emitter_max <= max_sprites) { + max_sprites = emitter_max; + } + wu32(SG.max_particle_sprites, max_sprites); + + // Determine vertex format index + const format_flag = ru32(emitter + 0x194); + const fmt_index: u32 = if ((format_flag & 1) != 0) 4 else 8; + + const data_ptr = gameGetDataPtr(fmt_index); + const vb_ptr = gameCreateVB(0, data_ptr, vert_size * max_sprites); + const vb_base = gameLockVB(vb_ptr); + + // Build vertex buffer pointer array (same layout as RenderParticleSprites expects) + var vb_ptrs: [9]u32 = undefined; + + // Position pointer + const pos_elem = gameGetMatElem(fmt_index, 0); + vb_ptrs[0] = pos_elem + vb_base; // pos ptr + vb_ptrs[4] = data_ptr; // pos stride + + // Normal pointer + if ((format_flag & 1) == 0) { + // No per-vertex normals — use shared default + const nflags = ru8(SG.render_init_flags); + if ((nflags & 4) == 0) { + wu8(SG.render_init_flags, nflags | 4); + wu32(SG.default_normal, 0); + wu32(SG.default_normal + 4, 0); + wu32(SG.default_normal + 8, 0); + gameValidateMem(SG.normal_validator); + } + vb_ptrs[1] = SG.default_normal; + vb_ptrs[5] = 0; // stride 0 = shared + } else { + const norm_elem = gameGetMatElem(fmt_index, 3); + vb_ptrs[1] = norm_elem + vb_base; + vb_ptrs[5] = data_ptr; + } + + // Color pointer + const color_elem = gameGetMatElem(fmt_index, 4); + vb_ptrs[2] = color_elem + vb_base; + vb_ptrs[6] = data_ptr; + + // Texcoord pointer + const tc_elem = gameGetMatElem(fmt_index, 5); + vb_ptrs[3] = tc_elem + vb_base; + vb_ptrs[7] = data_ptr; + + // Count + vb_ptrs[8] = 0; + + // ========================================================================= + // Section 7: Render particles + // ========================================================================= + gameRenderSorted(emitter, @intFromPtr(&vb_ptrs)); + + // DEBUG: log vertex count produced + if (!debug_logged and vb_ptrs[8] > 0) { + debug_logged = true; + debug_vertex_count = vb_ptrs[8]; + debug_max_sprites = max_sprites; + debug_fmt_index = fmt_index; + debug_data_ptr = data_ptr; + } + + gameUnlockVB(vb_ptr, 0); + gameDrawPrim(vb_ptr, fmt_index); + + // ========================================================================= + // Section 8: Index buffer setup + // ========================================================================= + const field_28 = ru32(emitter + 0x1C); + const renders_count = ru32(emitter + 0xA0); + if (field_28 == 6) { + var ib = ru32(SG.index_buffer_6); + if (gameIsObjValid(ib) == 0) { + gameBuildIB(emitter, ib, renders_count); + ib = ru32(SG.index_buffer_6); + } + gameSetStream(ib); + } else if (field_28 == 0xC) { + var ib = ru32(SG.index_buffer_12); + if (gameIsObjValid(ib) == 0) { + gameBuildIB(emitter, ib, renders_count); + ib = ru32(SG.index_buffer_12); + } + gameSetStream(ib); + } + + // ========================================================================= + // Section 9: Final setup + // ========================================================================= + const renders = ru32(emitter + 0xA0); + const calc_scale: f32 = @floatFromInt(renders * field_28); + wf32(emitter + 0x20, calc_scale); + + // CallGfxDeviceMethod_Wrapper — assembly-verified packed layout: + // [+0x00] u32 = 3 (primitive type) + // [+0x04] u32 = 0 (start index) + // [+0x08] u16 = (u16)(field_28 * renders) (verts per prim) + // [+0x0A] u16 = 0 + // [+0x0C] u16 = (u16)(vertex_count - 1) (prim count) + // fastcall(ECX=¶ms, EDX=1) + const calc_int: u16 = @truncate(renders_count * field_28); + const vertex_count: u32 = vb_ptrs[8]; + const prim_count: u16 = if (vertex_count > 0) @truncate(vertex_count - 1) else 0; + var gfx_bytes: [14]u8 align(4) = undefined; + @as(*u32, @ptrCast(gfx_bytes[0..4])).* = 3; + @as(*u32, @ptrCast(gfx_bytes[4..8])).* = 0; + @as(*u16, @ptrCast(gfx_bytes[8..10])).* = calc_int; + @as(*u16, @ptrCast(gfx_bytes[10..12])).* = 0; + @as(*u16, @ptrCast(gfx_bytes[12..14])).* = prim_count; + gameGfxCall(@intFromPtr(&gfx_bytes), 1); + + // End render and restore vertex shader + gameEndRender(); + gameSetVertexShader(@intFromPtr(&identity_a)); +} + +inline fn copyMat4x4(dst: u32, src: u32) void { + @as(*align(1) V4, @ptrFromInt(dst)).* = @as(*align(1) const V4, @ptrFromInt(src)).*; + @as(*align(1) V4, @ptrFromInt(dst + 16)).* = @as(*align(1) const V4, @ptrFromInt(src + 16)).*; + @as(*align(1) V4, @ptrFromInt(dst + 32)).* = @as(*align(1) const V4, @ptrFromInt(src + 32)).*; + @as(*align(1) V4, @ptrFromInt(dst + 48)).* = @as(*align(1) const V4, @ptrFromInt(src + 48)).*; +} + +// ============================================================================= +// RenderSpriteQuads (0x5A0F50) +// __thiscall(ECX=this, stack=spriteData, spriteCount, renderMode), RET 0xC +// +// Optimizations over original: +// 1. Hoisted invariant division out of inner loop (same result every iteration) +// 2. Inlined DisplayMode_CalculateOffset (trivial: table lookup + divide + subtract) +// 3. Cached texture validation bitmask check +// ============================================================================= + +// Game functions called by RenderSpriteQuads +const sqEmptyStub: *const fn (u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x590630); +const sqCalcMetrics: *const fn (u32, u32, u32, u32) callconv(TC) void = @ptrFromInt(0x592B00); +const sqGetAdapterInfo: *const fn (u32) callconv(TC) void = @ptrFromInt(0x5A1B20); + +// DisplayMode tables (from 0x592C10 disassembly) +const DISPLAY_MODE_DIVISOR_TABLE: u32 = 0x85ACF0; +const DISPLAY_MODE_OFFSET_TABLE: u32 = 0x85AD08; + +/// Inlined DisplayMode_CalculateOffset: table[type] divide + subtract +inline fn displayModeOffset(sprite_type: u32, count: u32) u32 { + const divisor = ru32(DISPLAY_MODE_DIVISOR_TABLE + sprite_type * 4); + const divided = if (divisor == 1) count else count / divisor; + return divided -% ru32(DISPLAY_MODE_OFFSET_TABLE + sprite_type * 4); +} + +export fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void { + // Early out: this+0xF2C == 0 + if (ru32(this + 0xF2C) == 0) return; + + // ========================================================================= + // Section 1: Texture validation (13 slots) + // ========================================================================= + const tex_bitmask = ru32(this + 0x27D8); + const tex_array_base = this + 0x27A4; + var all_valid: bool = true; + + var slot: u32 = 0; + while (slot < 13) : (slot += 1) { + if ((tex_bitmask & (@as(u32, 1) << @truncate(slot))) != 0) { + const tex_ptr = ru32(tex_array_base + slot * 4); + if (tex_ptr == 0 or !all_valid or ru8(tex_ptr + 0x1C) == 0 or ru8(tex_ptr + 0x1D) == 0) { + all_valid = false; + } + } + } + + // Render mode logic + var should_render: bool = undefined; + if (render_mode == 0) { + should_render = all_valid; // mode 0: render if NOT all valid → invert + // Wait: original does bVar8 = !bVar8 for mode 0, then checks if(bVar8) → early out + // So: if all_valid → !all_valid = false → don't early out → render + // if !all_valid → !all_valid = true → early out → don't render + // Simplified: render if all_valid + } else { + if (!all_valid) { + sqEmptyStub(0x85C7A8); + return; + } + const extra_ptr = ru32(this + 0x27EC); + if (ru8(extra_ptr + 0x1C) == 0) { + sqEmptyStub(0x85C7A8); + return; + } + should_render = ru8(extra_ptr + 0x1D) != 0; + } + + if (!should_render) { + sqEmptyStub(0x85C7A8); + return; + } + + // ========================================================================= + // Section 2: Setup calls + // ========================================================================= + sqCalcMetrics(this, sprite_data, sprite_count, render_mode); + sqGetAdapterInfo(this); + + if (sprite_count == 0) return; + + // ========================================================================= + // Section 3: Inner loop — hoisted invariant division + // ========================================================================= + + // The division this+0x27A4[0]+0x18 / this+0x27A4[0]+0xC is invariant across sprites. + // Original recomputes it per sprite. We hoist it. + var base_prim_count: u32 = 0; + if (ru32(this + 0x24C) == 0) { + const first_tex = ru32(this + 0x27A4); + if (first_tex != 0) { + const numerator = ru32(first_tex + 0x18); + const denominator = ru32(first_tex + 0x0C); + if (denominator != 0) { + base_prim_count = numerator / denominator; + } + } + } + + // D3D device vtable pointer + const device_ptr = ru32(this + 0x38A8); + const vtable = ru32(device_ptr); + + // Sprite data stride = 16 bytes, pointer starts at spriteData + 10 + var ptr = sprite_data + 10; + var remaining = sprite_count; + + while (remaining > 0) : (remaining -= 1) { + const count: u32 = @as(u32, ru16(ptr - 2)); // [esi-2] = sprite vertex count + if (count != 0) { + const sprite_type = ru32(ptr - 10); // [esi-0xA] = type/format index + const offset = displayModeOffset(sprite_type, count); + const lookup_val = ru32(0x80A14C + sprite_type * 4); + + if (render_mode == 0) { + // DrawPrimitive: vtable[0x144](device, lookup, basePrimCount, offset) + const draw_fn: *const fn (u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = + @ptrFromInt(ru32(vtable + 0x144)); + draw_fn(device_ptr, lookup_val, base_prim_count, offset); + } else { + const start_idx: u32 = @as(u32, ru16(ptr)); + const end_idx: u32 = @as(u32, ru16(ptr + 2)); + const extra_ptr = ru32(this + 0x27EC); + const extra_offset = (ru32(extra_ptr + 0x18) >> 1) + ru32(ptr - 6); + + // DrawIndexedPrimitive: vtable[0x148](device, lookup, basePrimCount, startIdx, count, extraOffset, offset) + const draw_fn: *const fn (u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = + @ptrFromInt(ru32(vtable + 0x148)); + draw_fn(device_ptr, lookup_val, base_prim_count, start_idx, end_idx - start_idx + 1, extra_offset, offset); + } + } + ptr += 16; + } +} + diff --git a/src/performance/performance.zig b/src/performance/performance.zig new file mode 100644 index 0000000..80a57c6 --- /dev/null +++ b/src/performance/performance.zig @@ -0,0 +1,186 @@ +//! performance — production SSE replacements with zero profiling overhead. +//! +//! Consolidates all verified permanent optimizations from transform44 and silicon +//! into a single clean module. No rdtsc, no A/B testing, no probe counters. +//! +//! Hooks: +//! - transformMatrix4x4 (bone SSE, 0x714260) +//! - RenderParticleSprites (particle SSE, 0x7B2A50) +//! - GetOrCreateCharacterGlyph (glyph cache, 0x5CA2D0) +//! - OnWorldUpdate (0x482EA0) — per-frame cache reset +//! +//! JMP patches (via silicon patch table): +//! - All silicon_sse functions (frustumCull, processLinkedListCollision, ftol, etc.) +//! are installed by the silicon module — not duplicated here. + +const hook = @import("zhook"); +const logging = @import("../logging.zig"); +const mod_mutex = @import("../mutex.zig"); + +pub const module_name: [*:0]const u8 = "performance"; + +var g_mutex: ?*anyopaque = null; +var g_is_hook_owner: bool = false; +var log: logging.Logger = .{}; + +pub fn isActive() bool { + return g_is_hook_owner; +} + +// ============================================================================= +// Extern SSE functions (from separate ReleaseFast compilation units) +// ============================================================================= + +extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void; +extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +extern fn resetParticleCache() void; + +// ============================================================================= +// transformMatrix4x4 hook (0x714260) +// ============================================================================= + +fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void { + transformImpl_SSE(this, mat1, mat2, mat3, mat4); +} + +const TransformFn = fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; +var transform_hook: hook.Detour(TransformFn) = .{}; +var teardown_active: bool = false; + +fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void { + if (teardown_active) { + transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); + } else { + transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); + } +} + +// ============================================================================= +// RenderParticleSprites hook (0x7B2A50) +// ============================================================================= + +const ParticleFn = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +var particle_hook: hook.Detour(ParticleFn) = .{}; + +fn particleDetour(a: u32, _: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + return @ptrFromInt(renderParticleSprites_SSE(a, c, d)); +} + +// ============================================================================= +// GetOrCreateCharacterGlyph cache (0x5CA2D0) +// ============================================================================= + +const GLYPH_CACHE_SHIFT = 12; +const GLYPH_CACHE_SIZE = 1 << GLYPH_CACHE_SHIFT; +const GLYPH_CACHE_MASK = GLYPH_CACHE_SIZE - 1; + +const GlyphCacheEntry = struct { + font_ptr: u32 = 0, + char_code: u32 = 0, + param2: u32 = 0, + width_bits: u32 = 0, +}; + +var glyph_cache: [GLYPH_CACHE_SIZE]GlyphCacheEntry = [_]GlyphCacheEntry{.{}} ** GLYPH_CACHE_SIZE; + +const GlyphFn = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +var glyph_hook: hook.Detour(GlyphFn) = .{}; + +fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + + const hash = ((a ^ c *% 0x9E3779B9) ^ d) & GLYPH_CACHE_MASK; + const entry = &glyph_cache[hash]; + + if (entry.font_ptr == a and entry.char_code == c and entry.param2 == d) { + asm volatile ("flds (%[p])" + :: [p] "r" (&entry.width_bits) + ); + return null; + } + + const ret = glyph_hook.callOriginal(.{ a, b, c, d }); + + var width_bits: u32 = undefined; + asm volatile ("fsts (%[p])" + :: [p] "r" (&width_bits) + ); + + entry.* = .{ + .font_ptr = a, + .char_code = c, + .param2 = d, + .width_bits = width_bits, + }; + + return ret; +} + +// ============================================================================= +// OnWorldUpdate hook (0x482EA0) — per-frame cache reset +// ============================================================================= + +const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void; +var world_update_hook: hook.Detour(WorldUpdateFn) = .{}; + +fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void { + resetParticleCache(); + world_update_hook.callOriginal(.{frame_count}); +} + +// ============================================================================= +// Teardown hook (0x491180) — protect bone SSE during logout cleanup +// ============================================================================= + +const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void; +var teardown_hook: hook.Detour(TeardownFn) = .{}; + +fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void { + teardown_active = true; + teardown_hook.callOriginal(.{}); + teardown_active = false; +} + +// ============================================================================= +// Install / Remove +// ============================================================================= + +pub fn installHooks() void { + const result = mod_mutex.acquire(module_name); + g_mutex = result.handle; + g_is_hook_owner = result.is_owner; + if (!g_is_hook_owner) return; + + log = logging.Logger.open(module_name, .both); + var installed: u32 = 0; + + // Bone transform SSE + if (transform_hook.attach(0x714260, &transformDetour) == .ok) installed += 1; + + // Particle rendering SSE + if (particle_hook.attach(0x7B2A50, &particleDetour) == .ok) installed += 1; + + // Glyph cache + if (glyph_hook.attach(0x5CA2D0, &glyphDetour) == .ok) installed += 1; + + // Per-frame cache reset + if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1; + + // Teardown guard + if (teardown_hook.attach(0x491180, &teardownDetour) == .ok) installed += 1; + + log.fmt("performance: {d} hooks installed\n", .{installed}); +} + +pub fn removeHooks() void { + if (g_is_hook_owner) { + transform_hook.detach(); + particle_hook.detach(); + glyph_hook.detach(); + world_update_hook.detach(); + teardown_hook.detach(); + log.close(); + mod_mutex.release(&g_mutex); + } + g_is_hook_owner = false; +} diff --git a/src/performance/silicon_sse.zig b/src/performance/silicon_sse.zig new file mode 100644 index 0000000..89bc7ca --- /dev/null +++ b/src/performance/silicon_sse.zig @@ -0,0 +1,709 @@ +//! SSE math implementations for silicon module. +//! +//! Pure math functions extracted from silicon.zig for standalone compilation. +//! Used by both the DLL (via silicon.zig wrappers) and the bench harness. +//! All functions use C ABI (export fn) — callers handle CC translation. +//! +//! Compiled with SSE4.1+FMA+AVX. Uses @Vector(4, f32) and @mulAdd throughout. + +const std = @import("std"); +const V4 = @Vector(4, f32); +const CC = std.builtin.CallingConvention; +const TC: CC = .{ .x86_thiscall = .{} }; +const FC: CC = .{ .x86_fastcall = .{} }; +const SC: CC = .{ .x86_stdcall = .{} }; + +inline fn loadV4(ptr: u32) V4 { + return @as(*align(1) const V4, @ptrFromInt(ptr)).*; +} + +inline fn loadV3_1(ptr: u32) V4 { + const p: [*]const f32 = @ptrFromInt(ptr); + return V4{ p[0], p[1], p[2], 1.0 }; +} + +inline fn dot3v(a: V4, b: V4) f32 { + return @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0])); +} + +inline fn dot4v(a: V4, b: V4) f32 { + return @mulAdd(f32, a[3], b[3], @mulAdd(f32, a[2], b[2], @mulAdd(f32, a[1], b[1], a[0] * b[0]))); +} + +/// Fast reciprocal via vrcpss + Newton-Raphson. ~5 cycles, ~23-bit accuracy (vs vdivss ~14 cycles). +inline fn fastRecip(x: f32) f32 { + var rcp: f32 = undefined; + // vrcpss: 12-bit approximation of 1/x + rcp = asm ("vrcpss %[x], %[x], %[rcp]" + : [rcp] "=x" (-> f32), + : [x] "x" (x), + ); + // Newton-Raphson: rcp = rcp * (2 - x * rcp) + return rcp * @mulAdd(f32, -x, rcp, 2.0); +} + +/// Round float to nearest integer via cvtss2si (uses MXCSR rounding mode, default = round-to-nearest). +/// Single instruction, replaces @round + @intFromFloat (~6 instructions). +inline fn cvtss2si(x: f32) i32 { + return asm ("vcvtss2si %[x], %[out]" + : [out] "=r" (-> i32), + : [x] "x" (x), + ); +} + +// --- 0x4549C0: normalizeVec3 (137K/7.5s) --- +// Naked thiscall: ECX=vec, [ESP+4]=length_bits. RET 4. Original: 38 bytes. +// rcpss + NR for fast reciprocal, then 3 multiplies. +export fn si_normalizeVec3() callconv(.naked) void { + asm volatile ( + // xmm0 = 1.0 / length (via rcpss + Newton-Raphson) + \\vmovss 4(%%esp), %%xmm0 + \\vrcpss %%xmm0, %%xmm0, %%xmm1 + \\vmulss %%xmm1, %%xmm0, %%xmm2 + \\vaddss %%xmm1, %%xmm1, %%xmm0 + \\vfnmadd231ss %%xmm1, %%xmm2, %%xmm0 + // xmm0 = refined 1/length. Multiply 3 components. + \\vmulss (%%ecx), %%xmm0, %%xmm1 + \\vmovss %%xmm1, (%%ecx) + \\vmulss 4(%%ecx), %%xmm0, %%xmm1 + \\vmovss %%xmm1, 4(%%ecx) + \\vmulss 8(%%ecx), %%xmm0, %%xmm1 + \\vmovss %%xmm1, 8(%%ecx) + \\ret $4 + ); +} + +// --- 0x7BAE60: mulMat3x4 --- +// out = A * B (3x4 layout: 3x3 rotation + 3 translation) +// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz] +// V4 per row: broadcast b[row*3+k], multiply with a's columns, accumulate. +export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 { + const dst: [*]f32 = @ptrFromInt(out); + const aa: [*]const f32 = @ptrFromInt(a_ptr); + const b: [*]const f32 = @ptrFromInt(b_ptr); + + // a is row-major 3x3. a[0..2] = row 0, a[3..5] = row 1, a[6..8] = row 2. + // The formula: dst[row*3+col] = a[col]*b[row*3] + a[col+3]*b[row*3+1] + a[col+6]*b[row*3+2] + // This means: for each output row, we broadcast b's row elements and dot with a's "columns" + // a's "column k" is {a[k], a[k+3], a[k+6]} — that's picking one element from each of a's rows. + + // Load a's columns (stride 3) + const ac0 = V4{ aa[0], aa[3], aa[6], aa[9] }; // col 0 of a (+ translation x) + const ac1 = V4{ aa[1], aa[4], aa[7], aa[10] }; // col 1 of a (+ translation y) + const ac2 = V4{ aa[2], aa[5], aa[8], aa[11] }; // col 2 of a (+ translation z) + + // Rotation: 3 output rows + inline for (0..3) |row| { + const br0: V4 = @splat(b[row * 3]); + const br1: V4 = @splat(b[row * 3 + 1]); + const br2: V4 = @splat(b[row * 3 + 2]); + const result = @mulAdd(V4, ac2, br2, @mulAdd(V4, ac1, br1, ac0 * br0)); + dst[row * 3] = result[0]; + dst[row * 3 + 1] = result[1]; + dst[row * 3 + 2] = result[2]; + } + + // Translation: dst[9+i] = a_trans dot b_col_i + b_trans[i] + // = ac0[3]*b[i] + ac1[3]*b[i+3] + ac2[3]*b[i+6] + b[9+i] + // Using the V4 approach: broadcast b elements, same ac columns, take lane 3 + add b_trans + // Translation: scalar @mulAdd (b's columns don't align for V4) + inline for (0..3) |col| { + dst[9 + col] = @mulAdd(f32, aa[11], b[col + 6], @mulAdd(f32, aa[10], b[col + 3], @mulAdd(f32, aa[9], b[col], b[9 + col]))); + } + + return out; +} + +// --- 0x7BDDB0: rotateMatByQuat --- +// builds rotation matrix from quaternion, multiplies with existing 4x4 matrix +// Uses V4 for the matrix multiply (same pattern as bone_sse) +export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 { + const q: [*]const f32 = @ptrFromInt(quat); + const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3]; + const x2 = x + x; const y2 = y + y; const z2 = z + z; + const xx = x * x2; const xy = x * y2; const xz = x * z2; + const yy = y * y2; const yz = y * z2; const zz = z * z2; + const wx = w * x2; const wy = w * y2; const wz = w * z2; + + const q0 = V4{ 1.0 - (yy + zz), xy + wz, xz - wy, 0 }; + const q1 = V4{ xy - wz, 1.0 - (xx + zz), yz + wx, 0 }; + const q2 = V4{ xz + wy, yz - wx, 1.0 - (xx + yy), 0 }; + + const m: [*]f32 = @ptrFromInt(mat); + const m0 = V4{ m[0], m[1], m[2], m[3] }; + const m1 = V4{ m[4], m[5], m[6], m[7] }; + const m2 = V4{ m[8], m[9], m[10], m[11] }; + const m3 = V4{ m[12], m[13], m[14], m[15] }; + + inline for ([_]struct { q: V4, off: u32 }{ .{ .q = q0, .off = 0 }, .{ .q = q1, .off = 4 }, .{ .q = q2, .off = 8 } }) |r| { + const row = @mulAdd(V4, @as(V4, @splat(r.q[2])), m2, @mulAdd(V4, @as(V4, @splat(r.q[1])), m1, @as(V4, @splat(r.q[0])) * m0)); + m[r.off] = row[0]; m[r.off + 1] = row[1]; m[r.off + 2] = row[2]; m[r.off + 3] = row[3]; + } + // Row 3 unchanged (identity row) + m[12] = m3[0]; m[13] = m3[1]; m[14] = m3[2]; m[15] = m3[3]; + return mat; +} + +// --- 0x7BB860: createRotMat3x4 --- +// Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries. +export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 { + const m: [*]f32 = @ptrFromInt(out); + const ax: [*]const f32 = @ptrFromInt(axis_ptr); + var x = ax[0]; var y = ax[1]; var z = ax[2]; + if (is_normalized == 0) { + const len = @sqrt(@mulAdd(f32, z, z, @mulAdd(f32, y, y, x * x))); + if (len > 1.0e-20) { const inv = 1.0 / len; x *= inv; y *= inv; z *= inv; } + } + const angle: f32 = @bitCast(angle_bits); + // x87 FSINCOS: single instruction computes both sin and cos + // FSINCOS tested at 148cy — x87 microcode is slow on modern CPUs. + // Library sinf+cosf (~35cy each = 70cy total) is 2x faster. + const c = @cos(angle); const s = @sin(angle); const t = 1.0 - c; + m[0] = @mulAdd(f32, t * x, x, c); m[1] = @mulAdd(f32, s, z, t * x * y); m[2] = @mulAdd(f32, -s, y, t * x * z); + m[3] = @mulAdd(f32, -s, z, t * x * y); m[4] = @mulAdd(f32, t * y, y, c); m[5] = @mulAdd(f32, s, x, t * y * z); + m[6] = @mulAdd(f32, s, y, t * x * z); m[7] = @mulAdd(f32, -s, x, t * y * z); m[8] = @mulAdd(f32, t * z, z, c); + m[9] = 0; m[10] = 0; m[11] = 0; + return out; +} + +// --- 0x6329E0: distanceToPlane (525K/7.5s) --- +// __fastcall(ECX=point, EDX=plane, stack=direction), returns f64 via ST(0), RET 0x4. +export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 { + const p: [*]const f32 = @ptrFromInt(point); + const pl: [*]const f32 = @ptrFromInt(plane); + const dir: [*]const f32 = @ptrFromInt(direction); + const dot1 = @mulAdd(f32, p[2], pl[2], @mulAdd(f32, p[1], pl[1], @mulAdd(f32, p[0], pl[0], pl[3]))); + const dot2 = @mulAdd(f32, dir[2], pl[2], @mulAdd(f32, dir[1], pl[1], dir[0] * pl[0])); + if (@abs(dot2) < 1.0e-20) return 0.0; + return @as(f64, dot1) / @as(f64, dot2); +} + + +// --- 0x686C20: classifyPointFrustum (3.2M/7.5s) --- +// Tests point against 6 frustum planes, produces 6-bit bitmask. +// Scalar @mulAdd dot4 per plane — the FMA chain has best throughput for this pattern. +// Tried: V4 batch 4 planes (gather kills it), V4 hsum (shuffle overhead kills it). +export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 { + const mask: *u32 = @ptrFromInt(out_mask); + const pt = loadV3_1(point); + var bits: u32 = 0; + inline for (0..6) |i| { + const pl = loadV4(planes_ptr + i * 16); + const dist = dot4v(pt, pl); + // Branchless: extract sign bit via bit cast + bits |= (@as(u32, @bitCast(dist)) >> 31) << i; + } + mask.* = bits; + return planes_ptr; +} + +// --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) --- +// Slab AABB test. Branchless min/max for t0/t1 swap and tmin/tmax accumulation. +export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 { + const bmin: [*]const f32 = @ptrFromInt(box_ptr); + const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC); + const start: [*]const f32 = @ptrFromInt(line_start); + const end_pt: [*]const f32 = @ptrFromInt(line_end); + var tmin: f32 = 0.0; + var tmax: f32 = 1.0; + inline for (0..3) |i| { + const dir = end_pt[i] - start[i]; + if (@abs(dir) < 1.0e-20) { + if (start[i] < bmin[i] or start[i] > bmax[i]) return 0; + } else { + const inv_dir = 1.0 / dir; + const ta = (bmin[i] - start[i]) * inv_dir; + const tb = (bmax[i] - start[i]) * inv_dir; + // Branchless swap: vminss/vmaxss instead of compare+branch + tmin = @max(tmin, @min(ta, tb)); + tmax = @min(tmax, @max(ta, tb)); + if (tmin > tmax) return 0; + } + } + return 1; +} + +// --- 0x6869C0: testOBBFrustum --- +// Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test. +export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 { + const aabb: [*]const f32 = @ptrFromInt(aabb_ptr); + const rot: [*]const f32 = @ptrFromInt(rot_ptr); + const t: [*]const f32 = @ptrFromInt(trans_ptr); + + // Rotation columns as V4 (xyz + 0 for w) + const rc0 = V4{ rot[0], rot[1], rot[2], 0 }; + const rc1 = V4{ rot[3], rot[4], rot[5], 0 }; + const rc2 = V4{ rot[6], rot[7], rot[8], 0 }; + const tv = V4{ t[0], t[1], t[2], 1.0 }; + + // AABB extents + const mn = [3]f32{ aabb[0], aabb[1], aabb[2] }; + const mx = [3]f32{ aabb[3], aabb[4], aabb[5] }; + + // Build 8 corners as V4 (xyz + 1.0 for plane dot w/ d term) + var corners: [8]V4 = undefined; + inline for (0..8) |i| { + const lx: f32 = if (i & 1 != 0) mx[0] else mn[0]; + const ly: f32 = if (i & 2 != 0) mx[1] else mn[1]; + const lz: f32 = if (i & 4 != 0) mx[2] else mn[2]; + corners[i] = @mulAdd(V4, @as(V4, @splat(lz)), rc2, @mulAdd(V4, @as(V4, @splat(ly)), rc1, @mulAdd(V4, @as(V4, @splat(lx)), rc0, tv))); + } + + // Extract x/y/z/w components across corners for batched plane tests + // Group A: corners 0-3, Group B: corners 4-7 + const cx_a = V4{ corners[0][0], corners[1][0], corners[2][0], corners[3][0] }; + const cy_a = V4{ corners[0][1], corners[1][1], corners[2][1], corners[3][1] }; + const cz_a = V4{ corners[0][2], corners[1][2], corners[2][2], corners[3][2] }; + + const cx_b = V4{ corners[4][0], corners[5][0], corners[6][0], corners[7][0] }; + const cy_b = V4{ corners[4][1], corners[5][1], corners[6][1], corners[7][1] }; + const cz_b = V4{ corners[4][2], corners[5][2], corners[6][2], corners[7][2] }; + + // Test each plane: compute 4 dots at a time, branchless sign check + inline for (0..6) |p| { + const pl = loadV4(planes_ptr + p * 16); + const pnx: V4 = @splat(pl[0]); + const pny: V4 = @splat(pl[1]); + const pnz: V4 = @splat(pl[2]); + const pd: V4 = @splat(pl[3]); + + // 4 dots for corners 0-3 + const da = @mulAdd(V4, cz_a, pnz, @mulAdd(V4, cy_a, pny, @mulAdd(V4, cx_a, pnx, pd))); + // 4 dots for corners 4-7 + const db = @mulAdd(V4, cz_b, pnz, @mulAdd(V4, cy_b, pny, @mulAdd(V4, cx_b, pnx, pd))); + + // All outside if all 8 distances are negative + // Branchless: extract sign bits via comparison + const neg_a = da < @as(V4, @splat(@as(f32, 0))); + const neg_b = db < @as(V4, @splat(@as(f32, 0))); + const mask_a: u4 = @bitCast(neg_a); + const mask_b: u4 = @bitCast(neg_b); + if (mask_a == 0xF and mask_b == 0xF) return 0; + } + return 3; +} + +// --- 0x686B80: testSphereFrustum (375K/7.5s) --- +// Zig thiscall: naked asm tested at 10cy (vhaddps slow), Zig dot4v at 8cy. +export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 { + const s: [*]const f32 = @ptrFromInt(sphere); + const center = V4{ s[0], s[1], s[2], 1.0 }; + const r = s[3]; + inline for (0..6) |i| { + const pl = loadV4(planes_ptr + i * 16); + if (dot4v(center, pl) < -r) return 0; + } + return 3; +} + + +// --- 0x7C0570: quatSlerp --- +// V4 for final blend, @mulAdd for dot product +export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 { + const dst: [*]f32 = @ptrFromInt(out); + const av = loadV4(a_ptr); + const bv = loadV4(b_ptr); + const tt: f32 = @bitCast(t_bits); + var dot_val = dot4v(av, bv); + var sign: f32 = 1.0; + if (dot_val < 0) { dot_val = -dot_val; sign = -1.0; } + var s0: f32 = undefined; + var s1: f32 = undefined; + if (dot_val > 0.9995) { + s0 = 1.0 - tt; + s1 = tt * sign; + } else { + const theta = std.math.acos(dot_val); + const sin_theta = @sin(theta); + const inv_sin = 1.0 / sin_theta; + s0 = @sin((1.0 - tt) * theta) * inv_sin; + s1 = @sin(tt * theta) * inv_sin * sign; + } + const result = @mulAdd(V4, @as(V4, @splat(s1)), bv, @as(V4, @splat(s0)) * av); + dst[0] = result[0]; dst[1] = result[1]; dst[2] = result[2]; dst[3] = result[3]; + return out; +} + +// --- 0x699330: isPointInsideBounds (1.7M/7.5s) --- +// __fastcall(ECX=a, EDX=b), returns u32. +export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 { + const va: [*]const f32 = @ptrFromInt(a); + const vb: [*]const f32 = @ptrFromInt(b); + if (vb[0] <= va[0] and vb[1] <= va[1] and vb[2] <= va[2]) return 1; + return 0; +} + +// --- 0x749280: calculateSinCos --- +export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void { + const angle: f32 = @bitCast(angle_bits); + const sp: *f32 = @ptrFromInt(out_sin); + const cp: *f32 = @ptrFromInt(out_cos); + sp.* = @sin(angle); + cp.* = @cos(angle); +} + +// --- 0x7BE5B0: createZRotMat3x3 --- +export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 { + const m: [*]f32 = @ptrFromInt(out); + const angle: f32 = @bitCast(angle_bits); + const c = @cos(angle); const s = @sin(angle); + m[0] = c; m[1] = s; m[2] = 0; + m[3] = -s; m[4] = c; m[5] = 0; + m[6] = 0; m[7] = 0; m[8] = 1; + return out; +} + +// --- 0x7BCEF0: transposeMat4x4 --- +// Naked thiscall: ECX=src, [ESP+4]=dst. RET 4. Original: 156 bytes. +// SSE unpacklo/unpackhi transpose: 4 loads + 4 shuffles + 4 stores. +export fn si_transposeMat4x4() callconv(.naked) void { + asm volatile ( + \\mov 4(%%esp), %%eax + // Load 4 rows from src (ECX) + \\vmovups (%%ecx), %%xmm0 + \\vmovups 16(%%ecx), %%xmm1 + \\vmovups 32(%%ecx), %%xmm2 + \\vmovups 48(%%ecx), %%xmm3 + // Transpose via unpacklo/unpackhi + \\vunpcklps %%xmm1, %%xmm0, %%xmm4 + \\vunpckhps %%xmm1, %%xmm0, %%xmm5 + \\vunpcklps %%xmm3, %%xmm2, %%xmm6 + \\vunpckhps %%xmm3, %%xmm2, %%xmm7 + // Combine into final columns + \\vmovlhps %%xmm6, %%xmm4, %%xmm0 + \\vmovhlps %%xmm4, %%xmm6, %%xmm1 + \\vmovlhps %%xmm7, %%xmm5, %%xmm2 + \\vmovhlps %%xmm5, %%xmm7, %%xmm3 + // Store to dst (EAX) + \\vmovups %%xmm0, (%%eax) + \\vmovups %%xmm1, 16(%%eax) + \\vmovups %%xmm2, 32(%%eax) + \\vmovups %%xmm3, 48(%%eax) + // Return src in EAX + \\mov %%ecx, %%eax + \\ret $4 + ); +} + +// --- 0x7BB420: mulMat3x4InPlace --- +// this = this * matB. V4 columns loaded upfront, write directly back (no tmp needed). +export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 { + const a: [*]f32 = @ptrFromInt(mat_a); + const b: [*]const f32 = @ptrFromInt(mat_b); + + // Load everything into locals to eliminate aliasing + const ac0 = V4{ a[0], a[3], a[6], a[9] }; + const ac1 = V4{ a[1], a[4], a[7], a[10] }; + const ac2 = V4{ a[2], a[5], a[8], a[11] }; + const b_local: [12]f32 = .{ b[0], b[1], b[2], b[3], b[4], b[5], b[6], b[7], b[8], b[9], b[10], b[11] }; + + // Rotation: write directly back to a + inline for (0..3) |row| { + const br0: V4 = @splat(b_local[row * 3]); + const br1: V4 = @splat(b_local[row * 3 + 1]); + const br2: V4 = @splat(b_local[row * 3 + 2]); + const result = @mulAdd(V4, ac2, br2, @mulAdd(V4, ac1, br1, ac0 * br0)); + a[row * 3] = result[0]; + a[row * 3 + 1] = result[1]; + a[row * 3 + 2] = result[2]; + } + + // Translation + inline for (0..3) |col| { + a[9 + col] = @mulAdd(f32, ac2[3], b_local[col + 6], @mulAdd(f32, ac1[3], b_local[col + 3], @mulAdd(f32, ac0[3], b_local[col], b_local[9 + col]))); + } + + return mat_a; +} + +// --- 0x6720F0: normalizeVec3InPlace --- +// sqrt + reciprocal. 14cy (2.2x). rsqrt+NR tested at 15cy — no gain, compiler's +// vsqrtss+vdivss pipeline is already optimal for scalar inverse sqrt. +export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void { + const v: [*]f32 = @ptrFromInt(vec); + const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); + if (len > 1.0e-20) { + const inv = 1.0 / len; + v[0] *= inv; + v[1] *= inv; + v[2] *= inv; + } +} + +// --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) --- +// thiscall(ECX=this, stack=vec). Scale is a global at 0x81207C, NOT a parameter. +export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void { + const obj: [*]f32 = @ptrFromInt(this); + const v: [*]const f32 = @ptrFromInt(vec); + const scale: f32 = @as(*const f32, @ptrFromInt(0x81207C)).*; + obj[21] += v[0]; + obj[22] += v[1]; + obj[23] += v[2]; + obj[33] = @mulAdd(f32, v[0], scale, obj[33]); + obj[42] = @mulAdd(f32, v[1], scale, obj[42]); + obj[51] = @mulAdd(f32, v[2], scale, obj[51]); +} + +// --- 0x71BF60: addToColorAccumulator (10K/7.5s) --- +// Naked thiscall: ECX=this, [ESP+4]=color_ptr. RET 4. Original: 34 bytes. +// 3 SSE adds at this+0x6C from color[0..2]. +export fn si_addToColorAccumulator() callconv(.naked) void { + asm volatile ( + \\mov 4(%%esp), %%eax + \\vmovss (%%eax), %%xmm0 + \\vaddss 0x6C(%%ecx), %%xmm0, %%xmm0 + \\vmovss %%xmm0, 0x6C(%%ecx) + \\vmovss 4(%%eax), %%xmm0 + \\vaddss 0x70(%%ecx), %%xmm0, %%xmm0 + \\vmovss %%xmm0, 0x70(%%ecx) + \\vmovss 8(%%eax), %%xmm0 + \\vaddss 0x74(%%ecx), %%xmm0, %%xmm0 + \\vmovss %%xmm0, 0x74(%%ecx) + \\ret $4 + ); +} + +// --- 0x7B7A80: packParticleColor (2K/7.5s) --- +// V4 multiply + clamp, then packed round+convert via @Vector(4, i32) for all channels at once. +export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void { + const base: [*]u8 = @ptrFromInt(obj); + const out: *align(1) u32 = @ptrCast(base + 0x12C); + const alpha = base[0x12F]; + const rgb = V4{ @bitCast(r_bits), @bitCast(g_bits), @bitCast(b_bits), 0 } * @as(V4, @splat(@as(f32, 255.0))); + const clamped = @min(@max(rgb, @as(V4, @splat(@as(f32, 0.0)))), @as(V4, @splat(@as(f32, 255.0)))); + // Packed round + convert: single roundps + cvtps2dq (SSE4.1) + const rounded = @round(clamped); + const ints: @Vector(4, i32) = @intFromFloat(rounded); + out.* = @as(u32, alpha) << 24 | @as(u32, @intCast(ints[0])) << 16 | @as(u32, @intCast(ints[1])) << 8 | @as(u32, @intCast(ints[2])); +} + +// --- 0x7B7B10: setParticleAlpha (2K/7.5s) --- +// Naked fastcall: ECX=obj, [ESP+4]=alpha_bits. RET 4. +// Clamp alpha*255 to [0,255], write byte to obj+0x12F. +export fn si_setParticleAlpha() callconv(.naked) void { + asm volatile ( + \\vmovss 4(%%esp), %%xmm0 + \\mov $0x437F0000, %%eax + \\vmovd %%eax, %%xmm1 + \\vmulss %%xmm1, %%xmm0, %%xmm0 + \\vxorps %%xmm2, %%xmm2, %%xmm2 + \\vmaxss %%xmm2, %%xmm0, %%xmm0 + \\vminss %%xmm1, %%xmm0, %%xmm0 + \\vcvtss2si %%xmm0, %%eax + \\mov %%al, 0x12F(%%ecx) + \\ret $4 + ); +} + + +// --- 0x40A2B0: __ftol --- +// Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64). +// SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original) +export fn si_ftol() callconv(.naked) void { + asm volatile ( + \\sub $8, %%esp + \\fisttpll (%%esp) + \\pop %%eax + \\pop %%edx + \\ret + ); +} + +// --- 0x602630: vec3Dot (31K/7.5s) --- +// __fastcall(ECX=a, EDX=b), returns f64 via ST(0). +export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 { + const va: [*]const f32 = @ptrFromInt(a); + const vb: [*]const f32 = @ptrFromInt(b); + return @floatCast(@mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0]))); +} + +// --- 0x686820: translateBoundingVol --- +// @mulAdd for plane distances. Scalar corner adds (stride 3 — V4 unaligned tested, slower). +export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void { + const obj: [*]f32 = @ptrFromInt(this); + const off: [*]const f32 = @ptrFromInt(offset); + const dx = off[0]; const dy = off[1]; const dz = off[2]; + inline for (0..8) |i| { + const base = 24 + i * 3; + obj[base] += dx; + obj[base + 1] += dy; + obj[base + 2] += dz; + } + inline for (0..6) |i| { + const base = i * 4; + obj[base + 3] -= @mulAdd(f32, obj[base + 2], dz, @mulAdd(f32, obj[base + 1], dy, obj[base] * dx)); + } + obj[48] += dx; obj[49] += dy; obj[50] += dz; + obj[51] += dx; obj[52] += dy; obj[53] += dz; +} + +// --- 0x686000: FrustumCullBoundingBox --- +// Transforms bbox through view-proj matrix, perspective divides, projects to 320-column +// occlusion buffer. Returns 0 (culled) / 2 (visible). +// Original: 380 bytes, 2 calls to mat*vec3 (0x7BCA80), x87 perspective divide, x87 column scan. +// SSE: inline V4 mat*vec3, SSE perspective divide, 4-wide column scan. +// __fastcall(bbox_ECX, flags_EDX, radius_stack), RET 0x4 +export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 { + // Early out: global occlusion flag bit 5 + if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 0x20) == 0) return 0; + + // Early out: radius too small + const radius: f32 = @bitCast(radius_bits); + const epsilon: f32 = @bitCast(@as(*const u32, @ptrFromInt(0x8029D4)).*); + if (@abs(radius) < epsilon) return 0; + + // Early out: global value must be in valid range [const1, const2] + const global_val: f32 = @as(*align(1) const f32, @ptrFromInt(0xC7CFF4)).*; + if (global_val < @as(*align(1) const f32, @ptrFromInt(0x8101AC)).*) return 0; + if (global_val > @as(*align(1) const f32, @ptrFromInt(0x804588)).*) return 0; + + // Transform center through view-proj matrix (column-major 4x4 at 0xC7B700) + // Inlined 0x7BCA80: result = col0*v.x + col1*v.y + col2*v.z + col3 + const bp: [*]const f32 = @ptrFromInt(bbox); + const vx: V4 = @splat(bp[0]); + const vy: V4 = @splat(bp[1]); + const vz: V4 = @splat(bp[2]); + + const m1: u32 = 0xC7B700; + const center = @mulAdd(V4, vz, loadV4(m1 + 32), @mulAdd(V4, vy, loadV4(m1 + 16), @mulAdd(V4, vx, loadV4(m1), loadV4(m1 + 48)))); + + // Transform extent {radius, radius, 0} through matrix at 0xC7D280 + const rv: V4 = @splat(radius); + const m2: u32 = 0xC7D280; + // z=0, so skip col2 term + const extent = @mulAdd(V4, rv, loadV4(m2 + 16), @mulAdd(V4, rv, loadV4(m2), loadV4(m2 + 48))); + + // Behind-camera check (unless flags & 8) + if ((flags & 0x8) == 0) { + if (center[2] < @as(*align(1) const f32, @ptrFromInt(0x80FED4)).*) return 0; + } + + // Perspective divide: inv_w = K * rcpss(center.z) with Newton-Raphson refinement. + // vrcpss + NR: ~5 cycles vs vdivss ~14 cycles. Column indices only need ~1px accuracy. + const K: f32 = @as(*align(1) const f32, @ptrFromInt(0x7FF9D8)).*; + const cz = center[2]; + const inv_w = K * fastRecip(cz); + const cx = center[0] * inv_w; + const ex = extent[0] * inv_w; + const ey = extent[1] * inv_w; + const depth = @mulAdd(f32, center[1], inv_w, ex); + + // Column projection: convert to 320-column indices + const col_scale: f32 = @as(*align(1) const f32, @ptrFromInt(0x810170)).*; + const col_offset: f32 = @as(*align(1) const f32, @ptrFromInt(0x86861C)).*; + + // cvtss2si: round-to-nearest in one instruction (MXCSR default mode). + // Replaces @round + @intFromFloat which generates vroundss + sign handling (~6 insns). + var left_col: i32 = cvtss2si(@mulAdd(f32, cx - ey, col_scale, -col_offset)); + left_col += 0xA0; + var right_col: i32 = cvtss2si(@mulAdd(f32, ey + cx, col_scale, -col_offset)); + right_col += 0xA1; + + // Bounds check — off-screen culling + if (left_col >= 0x140) return 0; // fully right of screen (320) + if (right_col < 0) return 0; // fully left of screen + if (left_col < 0) left_col = 0; + if (right_col >= 0x140) right_col = 0x13F; // clamp to 319 + if (left_col > right_col) return 2; // degenerate → visible + + // Horizon buffer scan: 320 floats at 0xC7B750 + // If any column's horizon value < depth → culled (return 0) + // SSE: test 4 columns at once + const horizon_base: u32 = 0xC7B750; + var col: u32 = @intCast(left_col); + const end: u32 = @intCast(right_col); + const depth_v: V4 = @splat(depth); + + // 4-wide scan + while (col + 3 <= end) { + const h = loadV4(horizon_base + col * 4); + const lt_bits: u4 = @bitCast(h < depth_v); + if (lt_bits != 0) return 0; + col += 4; + } + // Scalar remainder + while (col <= end) { + if (@as(*align(1) const f32, @ptrFromInt(horizon_base + col * 4)).* < depth) return 0; + col += 1; + } + + return 2; // visible — survived all columns +} + +// --- 0x6ABC40: processLinkedListCollision --- +// Walks intrusive linked list, per-node AABB overlap test, calls addGeometryToBuffer on hit. +// Original: 329 bytes, 6 x87 FCOMP/FNSTSW comparisons per node. +// SSE: 2 V4 loads + 2 CMPPS + AND + MOVMSK replaces the 6 scalar comparisons. +// +// __fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack), RET 0x8 +// addGeometryToBuffer at 0x6ABD90: __fastcall(queryBox_ECX, nodeData_EDX, resultBuf_stack), RET 0x4 +// Visited sentinel: *(u32*)0xC89F20 +export fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 { + if ((flags & 0xF0000F) == 0) return 1; + + const addGeometryToBuffer: *const fn (u32, u32, u32) callconv(FC) void = @ptrFromInt(0x6ABD90); + + // Load query box min/max as V4 for SSE AABB test + // queryBox layout: min(+0,+4,+8), max(+0xC,+0x10,+0x14) + const q_min = loadV4(query_box); // {qmin.x, qmin.y, qmin.z, } + const q_max = loadV4(query_box + 0x0C); // {qmax.x, qmax.y, qmax.z, } + + const sentinel = @as(*const u32, @ptrFromInt(0xC89F20)).*; + const link_offset = @as(*const u32, @ptrFromInt(list_head)).*; + + // First node: listHead[2] (offset +8) + var node: u32 = @as(*const u32, @ptrFromInt(list_head + 8)).*; + + // Linked list tag bit: bit 0 set = end sentinel + if (node & 1 != 0 or node == 0) return 1; + + while (node & 1 == 0 and node != 0) { + const node_data = @as(*const u32, @ptrFromInt(node + 4)).*; + const prev_node = node; + + // Skip: flags bit 0x100 set + const node_flags = @as(*const u16, @ptrFromInt(node_data + 0x0C)).*; + if ((node_flags & 0x100) == 0) { + // Skip: already visited or null + const visited = @as(*const u32, @ptrFromInt(node_data + 0x8C)).*; + const active = @as(*const u32, @ptrFromInt(node_data + 0x88)).*; + if (visited != sentinel and active != 0) { + // Type discriminator: pick flag mask + const type_a = @as(*const u32, @ptrFromInt(node_data + 0x180)).*; + const type_b = @as(*const u32, @ptrFromInt(node_data + 0x184)).*; + const mask = if ((type_a | type_b) != 0) flags & 0xF00000 else flags & 0xF; + + if (mask != 0) { + // Bit 7 of flags byte: if clear, abort with 0 + if (@as(i8, @bitCast(@as(u8, @truncate(node_flags)))) >= 0) return 0; + + // --- SSE AABB overlap test --- + // node AABB at node_data+0x14C: min(3 floats), max(3 floats) + const n_min = loadV4(node_data + 0x14C); // {nmin.x, nmin.y, nmin.z, } + const n_max = loadV4(node_data + 0x158); // {nmax.x, nmax.y, nmax.z, } + + // Overlap: nodeMin < queryMax AND queryMin <= nodeMax + // Compare lane-wise, check low 3 bits of mask + const lt_mask = n_min < q_max; + const le_mask = q_min <= n_max; + const lt_bits: u4 = @bitCast(lt_mask); + const le_bits: u4 = @bitCast(le_mask); + const bits = lt_bits & le_bits; + + if ((bits & 0x7) == 0x7) { + addGeometryToBuffer(query_box, node_data, result_buf); + } + + // Mark visited + @as(*u32, @ptrFromInt(node_data + 0x8C)).* = sentinel; + } + } + } + + // Advance: next = *(node + link_offset + 4) + // Original: MOV ECX,[EAX + EDX*1 + 4] where EAX=*listHead, EDX=node + node = @as(*const u32, @ptrFromInt(link_offset + prev_node + 4)).*; + } + + return 1; +}