diff --git a/build.zig b/build.zig index 0c58dc8..ffcbf74 100644 --- a/build.zig +++ b/build.zig @@ -19,23 +19,23 @@ const ModuleDesc = struct { /// affect the module's DLL, minor for feature changes, major for breaking. const module_list = [_]ModuleDesc{ .{ .name = "pngscreenshots", .version = "1.0.1", .desc = "Enable screenshot module", .src_dir = "screenshot" }, - .{ .name = "interact", .version = "1.0", .desc = "Enable interact module", .addon_name = "Interact" }, + .{ .name = "interact", .version = "1.1.0", .desc = "Enable interact module", .addon_name = "Interact" }, .{ .name = "outline", .version = "1.0", .desc = "Enable outline module", .default = false, .addon_name = "Outline" }, - .{ .name = "worldmarkers", .version = "1.0", .desc = "Enable world markers module", .addon_name = "WorldMarkers", .addon_hidden = true }, + .{ .name = "worldmarkers", .version = "1.1", .desc = "Enable world markers module", .addon_name = "WorldMarkers", .addon_hidden = true }, .{ .name = "framecrash", .version = "1.0", .desc = "Enable framecrash fix", .default = false }, - .{ .name = "logsessions", .version = "1.0.1", .desc = "Enable log session rotation", .addon_name = "LogSessions" }, + .{ .name = "logsessions", .version = "1.1.0", .desc = "Enable log session rotation", .addon_name = "LogSessions", .addon_hidden = true }, .{ .name = "minimapicons", .version = "1.0.1", .desc = "Enable custom minimap icons", .addon_name = "MinimapIcons" }, .{ .name = "transmogfix", .version = "1.0.1", .desc = "Enable transmog update coalescing" }, .{ .name = "customassets", .version = "1.0.1", .desc = "Enable loose file loading & permissive patch glob" }, .{ .name = "healtextfix", .version = "1.0.1", .desc = "Enable SuperWoW heal text fix" }, .{ .name = "bigcursor", .version = "1.0.1", .desc = "Enable big cursor module" }, - .{ .name = "clickthrough", .version = "1.0.2", .desc = "Enable GO click-through" }, + .{ .name = "clickthrough", .version = "1.0.3", .desc = "Enable GO click-through" }, .{ .name = "dpslog", .version = "0.1", .desc = "Enable structured combat log events for addons", .default = false }, .{ .name = "transform44", .version = "1.0", .desc = "Enable transform44 profiling/A/B testing (dev only)", .default = false }, .{ .name = "addonperf", .version = "1.0", .desc = "Enable addon memory/CPU profiling API", .default = false }, .{ .name = "ssemaths", .version = "1.0", .desc = "Enable UnitXP x87 math polyfill replacements (SSE)", .default = false }, .{ .name = "silicon", .version = "1.0", .desc = "Enable SSE2 math replacements (ported from libSiliconPatch)", .default = false }, - .{ .name = "weirdperformance", .version = "1.1.1", .desc = "Enable production performance optimizations (SSE, inflate, filecache, timer, luastr, luavm)", .default = true }, + .{ .name = "weirdperformance", .version = "1.2.1", .desc = "Enable production performance optimizations (SSE, inflate, filecache, timer, luastr, luavm)", .default = true }, .{ .name = "superweirdo", .version = "0.1", .desc = "Enable GO loot sparkle on interactable objects", .default = false }, .{ .name = "luagc", .version = "0.1", .desc = "Enable incremental Lua GC (replaces stop-the-world mark+sweep)", .default = false }, }; @@ -235,6 +235,18 @@ pub fn build(b: *std.Build) void { .optimize = .ReleaseFast, }), }); + const bench_bone_sse64 = b.addObject(.{ + .name = "bench_bone_sse64", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/weirdperformance/bone_sse64.zig"), + .target = b.resolveTargetQuery(.{ + .cpu_arch = .x86, + .os_tag = .linux, + .cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }), + }), + .optimize = .ReleaseFast, + }), + }); const bench_bone_baseline = b.addObject(.{ .name = "bench_bone_baseline", .root_module = b.createModule(.{ @@ -270,6 +282,7 @@ pub fn build(b: *std.Build) void { bench.root_module.addObject(bench_math_sse); bench.root_module.addObject(bench_silicon_sse); bench.root_module.addObject(bench_bone_sse); + bench.root_module.addObject(bench_bone_sse64); bench.root_module.addObject(bench_bone_baseline); bench.root_module.addObject(bench_particle_sse); bench.root_module.addObject(bench_cull_sse); diff --git a/src/bench/main.zig b/src/bench/main.zig index 140e08f..d1b0c07 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -1662,6 +1662,7 @@ pub fn main() void { const ofs = [3]f32{ 0, 0, 0 }; const sb: u32 = @bitCast(@as(f32, 1.0)); const transformImpl_SSE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE" }); + const transformImpl_SSE64 = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE64" }); const transformImpl_BASELINE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.c) void, .{ .name = "transformImpl_BASELINE" }); // Pre-set boneKeyframe init flag so we skip the atexit call (Windows CRT, can't run on Linux) @@ -1716,6 +1717,20 @@ pub fn main() void { const best_sse = run_bench_fn(transformImpl_SSE, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS); const avg_sse = best_sse / T44_ITERS; + // Warmup + bench bone_sse64 (f64-intermediate variant) + for (0..500) |iter| { + wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(iter * 2)), .little); + wu(u32, scene_obj[0x40..0x44], 0, .little); + transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb); + } + for (0..500) |iter| { + wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(999 - iter * 2)), .little); + wu(u32, scene_obj[0x40..0x44], 0, .little); + transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb); + } + const best_sse64 = run_bench_fn(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS); + const avg_sse64 = best_sse64 / T44_ITERS; + print(" BASELINE: {d} cycles/call (frozen)\n", .{BASELINE_CYCLES}); print(" SSE: {d} cycles/call", .{avg_sse}); if (avg_sse < BASELINE_CYCLES) { @@ -1727,6 +1742,25 @@ pub fn main() void { } else { print(" (same)\n", .{}); } + print(" SSE64: {d} cycles/call", .{avg_sse64}); + if (avg_sse64 < BASELINE_CYCLES) { + const pct = (BASELINE_CYCLES - avg_sse64) * 100 / BASELINE_CYCLES; + print(" (-{d}% vs BASELINE", .{pct}); + } else if (avg_sse64 > BASELINE_CYCLES) { + const pct = (avg_sse64 - BASELINE_CYCLES) * 100 / BASELINE_CYCLES; + print(" (+{d}% vs BASELINE", .{pct}); + } else { + print(" (same as BASELINE", .{}); + } + if (avg_sse64 > avg_sse) { + const pct = (avg_sse64 - avg_sse) * 100 / avg_sse; + print(", +{d}% vs SSE)\n", .{pct}); + } else if (avg_sse64 < avg_sse) { + const pct = (avg_sse - avg_sse64) * 100 / avg_sse; + print(", -{d}% vs SSE)\n", .{pct}); + } else { + print(", same as SSE)\n", .{}); + } // --- Output parity: run BASELINE then SSE with identical input, compare ALL outputs --- { @@ -1792,6 +1826,23 @@ pub fn main() void { } else { print(" parity: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs, total_len }); } + + // Run SSE64 with same input + reset_and_run(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, &bone_rt, BONE_COUNT); + + var diffs64: u32 = 0; + off = 0; + for (bufs) |b| { + for (0..b.len) |i| { + if (b.ptr[i] != snap[off + i]) diffs64 += 1; + } + off += b.len; + } + if (diffs64 == 0) { + print(" parity64: PASS (SSE64 == BASELINE, {d} bytes checked)\n", .{total_len}); + } else { + print(" parity64: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs64, total_len }); + } } } diff --git a/src/weirdperformance/bone_sse.zig b/src/weirdperformance/bone_sse.zig index 0ecc37a..06a2cde 100644 --- a/src/weirdperformance/bone_sse.zig +++ b/src/weirdperformance/bone_sse.zig @@ -25,118 +25,118 @@ const V4 = @Vector(4, f32); // SceneObject field offsets — assembly-verified from [EBX+N] in transformMatrix4x4 // ============================================================================= -const SO = struct { - const model_data_ptr: u32 = 0x010; - const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value - const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header - const sync_value: u32 = 0x040; - const search_data_base: u32 = 0x04C; // prev timestamp for delta - const emitter_flag: u32 = 0x050; - const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array - const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS - const child_padding: u32 = 0x084; - const anim_frame_ctr: u32 = 0x08C; - const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs - const bone_out_ptr: u32 = 0x094; // output bone matrices - const tex_anim_out: u32 = 0x0A0; - const color_anim_out: u32 = 0x0A8; - const scale1: u32 = 0x0AC; - const scale2: u32 = 0x0B0; - const scale3: u32 = 0x0B4; - const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward) - const world_xform: u32 = 0x10C; // float[16] world transform - const field_17c: u32 = 0x17C; - const field_180: u32 = 0x180; - const field_184: u32 = 0x184; - const field_188: u32 = 0x188; - const field_18c: u32 = 0x18C; - const field_190: u32 = 0x190; - const render_scale_x: u32 = 0x194; - const render_scale_y: u32 = 0x198; - const render_scale_z: u32 = 0x19C; - const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children) - const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children) - const hierarchy_ptr: u32 = 0x1C8; - const emitter_ctx: u32 = 0x1CC; - const field_1d8: u32 = 0x1D8; - const hierarchy_idx: u32 = 0x1DC; - const field_200: u32 = 0x200; - const particle1: u32 = 0x3C4; - const particle2: u32 = 0x3C8; - const particle3: u32 = 0x3D0; - const particle4: u32 = 0x3D4; - const add_remaining: u32 = 0x3D8; +pub const SO = struct { + pub const model_data_ptr: u32 = 0x010; + pub const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value + pub const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header + pub const sync_value: u32 = 0x040; + pub const search_data_base: u32 = 0x04C; // prev timestamp for delta + pub const emitter_flag: u32 = 0x050; + pub const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array + pub const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS + pub const child_padding: u32 = 0x084; + pub const anim_frame_ctr: u32 = 0x08C; + pub const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs + pub const bone_out_ptr: u32 = 0x094; // output bone matrices + pub const tex_anim_out: u32 = 0x0A0; + pub const color_anim_out: u32 = 0x0A8; + pub const scale1: u32 = 0x0AC; + pub const scale2: u32 = 0x0B0; + pub const scale3: u32 = 0x0B4; + pub const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward) + pub const world_xform: u32 = 0x10C; // float[16] world transform + pub const field_17c: u32 = 0x17C; + pub const field_180: u32 = 0x180; + pub const field_184: u32 = 0x184; + pub const field_188: u32 = 0x188; + pub const field_18c: u32 = 0x18C; + pub const field_190: u32 = 0x190; + pub const render_scale_x: u32 = 0x194; + pub const render_scale_y: u32 = 0x198; + pub const render_scale_z: u32 = 0x19C; + pub const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children) + pub const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children) + pub const hierarchy_ptr: u32 = 0x1C8; + pub const emitter_ctx: u32 = 0x1CC; + pub const field_1d8: u32 = 0x1D8; + pub const hierarchy_idx: u32 = 0x1DC; + pub const field_200: u32 = 0x200; + pub const particle1: u32 = 0x3C4; + pub const particle2: u32 = 0x3C8; + pub const particle3: u32 = 0x3D0; + pub const particle4: u32 = 0x3D4; + pub const add_remaining: u32 = 0x3D8; }; // Bone runtime struct offsets (within 0x118-byte per-bone runtime) -const BR = struct { +pub const BR = struct { // Translation interpolation state - const trans_idx0: u32 = 0x00; // [0] lower keyframe index - const trans_idx1: u32 = 0x04; // [1] upper keyframe index - const trans_t: u32 = 0x08; // [2] interpolation factor (float bits) - const trans_x: u32 = 0x0C; // [3] interpolated translation X - const trans_y: u32 = 0x10; // [4] Y - const trans_z: u32 = 0x14; // [5] Z + pub const trans_idx0: u32 = 0x00; // [0] lower keyframe index + pub const trans_idx1: u32 = 0x04; // [1] upper keyframe index + pub const trans_t: u32 = 0x08; // [2] interpolation factor (float bits) + pub const trans_x: u32 = 0x0C; // [3] interpolated translation X + pub const trans_y: u32 = 0x10; // [4] Y + pub const trans_z: u32 = 0x14; // [5] Z // Secondary translation (crossfade) - const trans2_idx0: u32 = 0x18; - const trans2_idx1: u32 = 0x1C; - const trans2_t: u32 = 0x20; - const trans2_x: u32 = 0x24; - const trans2_y: u32 = 0x28; - const trans2_z: u32 = 0x2C; + pub const trans2_idx0: u32 = 0x18; + pub const trans2_idx1: u32 = 0x1C; + pub const trans2_t: u32 = 0x20; + pub const trans2_x: u32 = 0x24; + pub const trans2_y: u32 = 0x28; + pub const trans2_z: u32 = 0x2C; // Scale interpolation state (at puVar20 + 0x1a = offset 0x68) - const scale_idx0: u32 = 0x68; - const scale_idx1: u32 = 0x6C; - const scale_t: u32 = 0x70; - const scale_x: u32 = 0x74; - const scale_y: u32 = 0x78; - const scale_z: u32 = 0x7C; - const scale2_idx0: u32 = 0x80; - const scale2_idx1: u32 = 0x84; - const scale2_t: u32 = 0x88; - const scale2_x: u32 = 0x8C; - const scale2_y: u32 = 0x90; - const scale2_z: u32 = 0x94; + pub const scale_idx0: u32 = 0x68; + pub const scale_idx1: u32 = 0x6C; + pub const scale_t: u32 = 0x70; + pub const scale_x: u32 = 0x74; + pub const scale_y: u32 = 0x78; + pub const scale_z: u32 = 0x7C; + pub const scale2_idx0: u32 = 0x80; + pub const scale2_idx1: u32 = 0x84; + pub const scale2_t: u32 = 0x88; + pub const scale2_x: u32 = 0x8C; + pub const scale2_y: u32 = 0x90; + pub const scale2_z: u32 = 0x94; // Primary animation time range - const prim_time: u32 = 0x98; // puVar20[0x26] - const prim_track: u32 = 0x9C; // puVar20[0x27] - const prim_anim: u32 = 0xA0; // puVar20[0x28] - const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index + pub const prim_time: u32 = 0x98; // puVar20[0x26] + pub const prim_track: u32 = 0x9C; // puVar20[0x27] + pub const prim_anim: u32 = 0xA0; // puVar20[0x28] + pub const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index // Secondary animation time range (crossfade) - const sec_start: u32 = 0xA8; // puVar20[0x2a] - const sec_end: u32 = 0xAC; // puVar20[0x2b] - const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion - const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e] + pub const sec_start: u32 = 0xA8; // puVar20[0x2a] + pub const sec_end: u32 = 0xAC; // puVar20[0x2b] + pub const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion + pub const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e] // Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30) - const rot_idx0: u32 = 0x30; - const rot_idx1: u32 = 0x34; - const rot_t: u32 = 0x38; - const rot_x: u32 = 0x3C; - const rot_y: u32 = 0x40; - const rot_z: u32 = 0x44; - const rot_w: u32 = 0x48; + pub const rot_idx0: u32 = 0x30; + pub const rot_idx1: u32 = 0x34; + pub const rot_t: u32 = 0x38; + pub const rot_x: u32 = 0x3C; + pub const rot_y: u32 = 0x40; + pub const rot_z: u32 = 0x44; + pub const rot_w: u32 = 0x48; // Secondary rotation - const rot2_idx0: u32 = 0x4C; - const rot2_idx1: u32 = 0x50; - const rot2_t: u32 = 0x54; - const rot2_x: u32 = 0x58; - const rot2_y: u32 = 0x5C; - const rot2_z: u32 = 0x60; - const rot2_w: u32 = 0x64; + pub const rot2_idx0: u32 = 0x4C; + pub const rot2_idx1: u32 = 0x50; + pub const rot2_t: u32 = 0x54; + pub const rot2_x: u32 = 0x58; + pub const rot2_y: u32 = 0x5C; + pub const rot2_z: u32 = 0x60; + pub const rot2_w: u32 = 0x64; // Secondary time range - const sec_time: u32 = 0xC4; // puVar20[0x31] - const sec_track: u32 = 0xC8; // puVar20[0x32] - const sec_slot: u32 = 0xD0; // puVar20[0x34] - const sec_start2: u32 = 0xD4; // puVar20[0x35] - const sec_end2: u32 = 0xD8; // puVar20[0x36] - const sec_offset2: u32 = 0xE4; // puVar20[0x39] + pub const sec_time: u32 = 0xC4; // puVar20[0x31] + pub const sec_track: u32 = 0xC8; // puVar20[0x32] + pub const sec_slot: u32 = 0xD0; // puVar20[0x34] + pub const sec_start2: u32 = 0xD4; // puVar20[0x35] + pub const sec_end2: u32 = 0xD8; // puVar20[0x36] + pub const sec_offset2: u32 = 0xE4; // puVar20[0x39] // Flags and weights - const flags2: u32 = 0xF4; // puVar20[0x3d] - const crossfade_end: u32 = 0x100; // puVar20[0x40] - const crossfade_inv: u32 = 0x104; // puVar20[0x41] - const crossfade_weight: u32 = 0x108; // puVar20[0x42] - const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade - const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c] + pub const flags2: u32 = 0xF4; // puVar20[0x3d] + pub const crossfade_end: u32 = 0x100; // puVar20[0x40] + pub const crossfade_inv: u32 = 0x104; // puVar20[0x41] + pub const crossfade_weight: u32 = 0x108; // puVar20[0x42] + pub const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade + pub const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c] }; // OldAnimationBlock struct offsets (28 bytes = 0x1C per track in v256 M2) @@ -144,37 +144,37 @@ const BR = struct { // pMVar23->m31 (bone_def+0x34) = rot block+0x0C = nTimestamps (gates rotation) // pMVar23->m12 (bone_def+0x18) = trans block+0x0C = nTimestamps (gates translation) // pMVar23[1].m10 (bone_def+0x50) = scale block+0x0C = nTimestamps (gates scale) -const AD = struct { - const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp) - const time_index: u32 = 0x02; // i16: global sequence index (-1 = none) - const track_count_flag: u32 = 0x04; // nRanges: 0 = single track - const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs - const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count - const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array - const nvalues: u32 = 0x14; // nValues: number of value entries - const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data +pub const AD = struct { + pub const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp) + pub const time_index: u32 = 0x02; // i16: global sequence index (-1 = none) + pub const track_count_flag: u32 = 0x04; // nRanges: 0 = single track + pub const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs + pub const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count + pub const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array + pub const nvalues: u32 = 0x14; // nValues: number of value entries + pub const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data }; // M2CompBone struct offsets (0x6C = 108 bytes per bone in v256 model) // Layout: 12 bytes fixed header + 3x28 byte OldAnimationBlock tracks + 12 bytes pivot // Track order: translation, rotation, scale (standard M2 order) -const BD = struct { - const key_id: u32 = 0x00; // i32: key bone ID - const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.) - const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes +pub const BD = struct { + pub const key_id: u32 = 0x00; // i32: key bone ID + pub const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.) + pub const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes // Translation OldAnimationBlock (28 bytes, +0x0C to +0x27) - const trans_anim: u32 = 0x0C; - const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation + pub const trans_anim: u32 = 0x0C; + pub const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation // Rotation OldAnimationBlock (28 bytes, +0x28 to +0x43) - const rot_anim: u32 = 0x28; - const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation + pub const rot_anim: u32 = 0x28; + pub const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation // Scale OldAnimationBlock (28 bytes, +0x44 to +0x5F) - const scale_anim: u32 = 0x44; - const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation + pub const scale_anim: u32 = 0x44; + pub const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation // Pivot point (12 bytes, +0x60 to +0x6B) - const pivot_x: u32 = 0x60; - const pivot_y: u32 = 0x64; - const pivot_z: u32 = 0x68; + pub const pivot_x: u32 = 0x60; + pub const pivot_y: u32 = 0x64; + pub const pivot_z: u32 = 0x68; }; // Game constants @@ -182,12 +182,12 @@ const ZERO_F: f32 = 0.0; const ONE_F: f32 = 1.0; const THREE_F: f32 = 3.0; // getBillboardEpsilon(): read from game memory (runtime 0x34800000, NOT static 0x3727c5ac from Ghidra) -fn getBillboardEpsilon() f32 { +pub fn getBillboardEpsilon() f32 { return rf32(0x008029d4); } // getShortToFloat(): read from game memory at 0x00811610 (runtime value is 0x38000100 = 1/32767, // NOT the static 0x38000000 = 1/32768 from Ghidra). The game patches this at startup. -fn getShortToFloat() f32 { +pub fn getShortToFloat() f32 { return rf32(0x00811610); } // MSVC CRT sin/cos — linked from the WoW process @@ -202,42 +202,42 @@ const OrigTransformFn = *const fn (u32, u32, u32, u32, u32) callconv(.c) void; // Memory access helpers // ============================================================================= -inline fn ru32(addr: u32) u32 { +pub inline fn ru32(addr: u32) u32 { return @as(*const u32, @ptrFromInt(addr)).*; } -inline fn ri32(addr: u32) i32 { +pub inline fn ri32(addr: u32) i32 { return @as(*const i32, @ptrFromInt(addr)).*; } -inline fn rf32(addr: u32) f32 { +pub inline fn rf32(addr: u32) f32 { return @as(*const f32, @ptrFromInt(addr)).*; } -inline fn ru16(addr: u32) u16 { +pub inline fn ru16(addr: u32) u16 { return @as(*align(1) const u16, @ptrFromInt(addr)).*; } -inline fn ri16(addr: u32) i16 { +pub inline fn ri16(addr: u32) i16 { return @as(*align(1) const i16, @ptrFromInt(addr)).*; } -inline fn ru8(addr: u32) u8 { +pub inline fn ru8(addr: u32) u8 { return @as(*const u8, @ptrFromInt(addr)).*; } -inline fn wu32(addr: u32, v: u32) void { +pub inline fn wu32(addr: u32, v: u32) void { @as(*u32, @ptrFromInt(addr)).* = v; } -inline fn wf32(addr: u32, v: f32) void { +pub inline fn wf32(addr: u32, v: f32) void { @as(*f32, @ptrFromInt(addr)).* = v; } -inline fn wu16(addr: u32, v: u16) void { +pub inline fn wu16(addr: u32, v: u16) void { @as(*align(1) u16, @ptrFromInt(addr)).* = v; } -inline fn wu8(addr: u32, v: u8) void { +pub inline fn wu8(addr: u32, v: u8) void { @as(*u8, @ptrFromInt(addr)).* = v; } -inline fn fbits(v: f32) u32 { +pub inline fn fbits(v: f32) u32 { return @bitCast(v); } -inline fn ufloat(v: u32) f32 { +pub inline fn ufloat(v: u32) f32 { return @bitCast(v); } @@ -245,12 +245,12 @@ inline fn ufloat(v: u32) f32 { // Math helpers — using @Vector(4, f32) for SSE // ============================================================================= -inline fn splat(v: f32) V4 { +pub inline fn splat(v: f32) V4 { return @splat(v); } /// 3-component lerp: a + (b - a) * t. Uses @mulAdd → vfmadd. -inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 { +pub inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 { return .{ @mulAdd(f32, rf32(b_addr) - rf32(a_addr), t, rf32(a_addr)), @mulAdd(f32, rf32(b_addr + 4) - rf32(a_addr + 4), t, rf32(a_addr + 4)), @@ -260,7 +260,7 @@ inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 { /// Scale 3x3 rotation portion of a row-major 4x4 matrix by per-axis scale. /// Row 0 *= scale.x, Row 1 *= scale.y, Row 2 *= scale.z -inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void { +pub inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void { // Row 0 (offsets 0x00, 0x04, 0x08) wf32(mat + 0x00, rf32(mat + 0x00) * sx); wf32(mat + 0x04, rf32(mat + 0x04) * sx); @@ -279,15 +279,15 @@ inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void { /// mat[3][0] += dot(mat[0], t) /// mat[3][1] += dot(mat[1], t) /// mat[3][2] += dot(mat[2], t) -/// Uses @mulAdd chain → vfmadd for each dot product component. -inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void { +/// Uses @mulAdd chain for each dot product component. +pub inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void { wf32(mat + 0x30, @mulAdd(f32, tz, rf32(mat + 0x20), @mulAdd(f32, ty, rf32(mat + 0x10), @mulAdd(f32, tx, rf32(mat + 0x00), rf32(mat + 0x30))))); wf32(mat + 0x34, @mulAdd(f32, tz, rf32(mat + 0x24), @mulAdd(f32, ty, rf32(mat + 0x14), @mulAdd(f32, tx, rf32(mat + 0x04), rf32(mat + 0x34))))); wf32(mat + 0x38, @mulAdd(f32, tz, rf32(mat + 0x28), @mulAdd(f32, ty, rf32(mat + 0x18), @mulAdd(f32, tx, rf32(mat + 0x08), rf32(mat + 0x38))))); } /// Quaternion → rotation matrix as value. No memory writes. -inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 { +pub inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 { const xx2 = qx * (qx + qx); const xy2 = qx * (qy + qy); const xz2 = qx * (qz + qz); @@ -308,7 +308,7 @@ inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 { /// Quaternion → rotation matrix: writes to game memory via u32 address. /// Used by boneKeyframeLoop where the matrix is in game memory. -inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { +pub inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { const m = buildRotationMatrixVal(qx, qy, qz, qw); inline for (0..16) |i| { wf32(mat + @as(u32, @intCast(i * 4)), m[i]); @@ -316,7 +316,7 @@ inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void } /// Quaternion → rotation matrix × mat. Fused: builds quat rows as V4, multiplies in-register. -inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { +pub inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void { const xx2 = qx * (qx + qx); const xy2 = qx * (qy + qy); const xz2 = qx * (qz + qz); @@ -354,7 +354,7 @@ inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void } /// Copy 4x4 matrix (64 bytes) — 4 V4 loads/stores instead of 16 scalar copies. -inline fn copyMat4(dst: u32, src: u32) void { +pub inline fn copyMat4(dst: u32, src: u32) void { inline for (0..4) |i| { const off: u32 = @intCast(i * 16); const row = V4{ rf32(src + off), rf32(src + off + 4), rf32(src + off + 8), rf32(src + off + 12) }; @@ -367,19 +367,16 @@ inline fn copyMat4(dst: u32, src: u32) void { /// 4x4 matrix multiply: dst = a * b (row-major). Safe for dst==a or dst==b. /// Uses V4 + @mulAdd (FMA): 1 mul + 3 FMA per row = 16 SIMD ops total. -inline fn matMul4x4(dst: u32, a: u32, b: u32) void { - // Pre-load all rows of B +pub inline fn matMul4x4(dst: u32, a: u32, b: u32) void { const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) }; const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) }; const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) }; const b3 = V4{ rf32(b + 48), rf32(b + 52), rf32(b + 56), rf32(b + 60) }; - // Pre-load all rows of A (in case dst aliases a) const a0 = V4{ rf32(a), rf32(a + 4), rf32(a + 8), rf32(a + 12) }; const a1 = V4{ rf32(a + 16), rf32(a + 20), rf32(a + 24), rf32(a + 28) }; const a2 = V4{ rf32(a + 32), rf32(a + 36), rf32(a + 40), rf32(a + 44) }; const a3 = V4{ rf32(a + 48), rf32(a + 52), rf32(a + 56), rf32(a + 60) }; const rows = [4]V4{ a0, a1, a2, a3 }; - // Compute: each output row = broadcast(a[row][col]) * b_row, accumulated with FMA inline for (0..4) |i| { const s0: V4 = @splat(rows[i][0]); const s1: V4 = @splat(rows[i][1]); @@ -395,7 +392,7 @@ inline fn matMul4x4(dst: u32, a: u32, b: u32) void { } /// 4x4 matrix multiply: dst = a * b. Left operand is a local array, right is game memory. -inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void { +pub inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void { const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) }; const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) }; const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) }; @@ -421,7 +418,7 @@ inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void { } /// In-place multiply: a = a * b (b from game memory). Returns new array. -inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 { +pub inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 { const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) }; const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) }; const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) }; @@ -448,7 +445,7 @@ inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 { } /// Set identity matrix (16 floats) -inline fn setIdentity(dst: u32) void { +pub inline fn setIdentity(dst: u32) void { inline for (0..16) |i| { const val: f32 = if (i == 0 or i == 5 or i == 10 or i == 15) 1.0 else 0.0; wf32(dst + @as(u32, @intCast(i)) * 4, val); @@ -458,7 +455,7 @@ inline fn setIdentity(dst: u32) void { /// Normalize a 3-component vector in memory at addr. /// Calls game's vec3 squared magnitude (0x4549F0), then sqrt, epsilon check, divide. /// Assembly pattern: CALL 0x4549F0 → FSQRT → FABS → FCOMP → FLD1 → FDIVRP → FMUL×3 -inline fn normalizeVec3InPlace(addr: u32) void { +pub inline fn normalizeVec3InPlace(addr: u32) void { const sq_mag = callVec3SqMag(addr); const len = @sqrt(sq_mag); if (@abs(len) >= getBillboardEpsilon()) { @@ -471,7 +468,7 @@ inline fn normalizeVec3InPlace(addr: u32) void { /// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small. /// Writes vec3 to stack local and calls game's vec3 squared magnitude (0x4549F0). -inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 { +pub inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 { var v: [3]f32 = .{ x, y, z }; const sq_mag = callVec3SqMag(@intFromPtr(&v)); const len = @sqrt(sq_mag); @@ -481,7 +478,7 @@ inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 { } /// Cross product of two 3-component vectors -inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 { +pub inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 { return .{ ay * bz - az * by, az * bx - ax * bz, @@ -502,7 +499,7 @@ inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 /// IsParticleBufferEmpty (0x7B5F60) — recursive tree check. /// Returns true if any node in the tree has active particles (this->0x64 != 0). -fn isParticleBufferNotEmpty(ptr: u32) bool { +pub fn isParticleBufferNotEmpty(ptr: u32) bool { if (ru32(ptr + 0x64) != 0) return true; const count = ru32(ptr + 0x7C); if (count == 0) return false; @@ -514,7 +511,7 @@ fn isParticleBufferNotEmpty(ptr: u32) bool { return false; } -const InterpResult = struct { +pub const InterpResult = struct { idx0: u32, idx1: u32, t: f32, @@ -523,7 +520,7 @@ const InterpResult = struct { /// Check if two AnimData tracks share the same temporal structure, /// meaning findInterpIdx would produce identical (idx0, idx1, t) for both. /// Both must use prim_time (time_index == -1) and have matching range/timestamp layout. -inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool { +pub inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool { if (ri16(ref_anim + AD.time_index) != -1) return false; if (ri16(other_anim + AD.time_index) != -1) return false; return ru32(ref_anim + AD.track_count_flag) == ru32(other_anim + AD.track_count_flag) and @@ -533,7 +530,7 @@ inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool { } /// Write temporal coherence cache for a reused result so next frame's forward scan starts right. -inline fn applyCachedResult(cached: InterpResult, output: u32) void { +pub inline fn applyCachedResult(cached: InterpResult, output: u32) void { wu32(output, cached.idx0); } @@ -541,7 +538,7 @@ inline fn applyCachedResult(cached: InterpResult, output: u32) void { /// Reimplementation of game function at 0x713D50 (334 bytes). /// Assembly-verified against t44_helpers_asm.txt. /// Returns indices and t in registers; only writes output[0] for next-frame cache persistence. -inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult { +pub inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult { const n_ranges = ru32(anim_data + AD.track_count_flag); // Range selection: [start, last] not [start, count] @@ -659,13 +656,13 @@ inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_dat /// Quaternion keyframe interpolation — replaces game's 0x713EA0. /// Assembly-verified: stride 16 (SHL EAX,4), values are 4×float, not CompQuat. -inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 { +pub inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 { return interpAnimKFCached(this, bone_rt, anim_data, output, null); } /// Quaternion keyframe interpolation with optional cached primary InterpResult. /// When cached_primary is non-null, skips findInterpIdx and uses the cached indices/t. -inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 { +pub inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 { const r = if (cached_primary) |c| blk: { applyCachedResult(c, output); break :blk c; @@ -716,7 +713,7 @@ inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u3 /// Fast modulo for looping animations. The value is almost always < 2*length /// (frame-to-frame delta is small), so a conditional subtract beats idiv. -inline fn fastMod(val: u32, len: u32) u32 { +pub inline fn fastMod(val: u32, len: u32) u32 { var v = val; if (v >= len) { v -%= len; @@ -727,12 +724,12 @@ inline fn fastMod(val: u32, len: u32) u32 { /// Float truncation — replaces game's __ftol at 0x40A2B0. /// Original: FILD i32 → FMUL f32 → __ftol, all in 80-bit x87 precision. -inline fn callFtol(delta: i32, scale_addr: u32) i32 { +pub inline fn callFtol(delta: i32, scale_addr: u32) i32 { return @intFromFloat(@as(f32, @floatFromInt(delta)) * rf32(scale_addr)); } /// Vec3 squared magnitude — replaces game's 0x4549F0. Uses @mulAdd → vfmadd. -inline fn callVec3SqMag(vec3_ptr: u32) f32 { +pub inline fn callVec3SqMag(vec3_ptr: u32) f32 { const x = rf32(vec3_ptr); const y = rf32(vec3_ptr + 4); const z = rf32(vec3_ptr + 8); @@ -741,7 +738,7 @@ inline fn callVec3SqMag(vec3_ptr: u32) f32 { /// Read i16 at keyframe index. Replaces game's getIndexOffset (0x71AFF0) + setShortValue (0x71B010). /// getIndexOffset returns table[4] + index*2, setShortValue copies a word. Direct read is equivalent. -inline fn readShortViaGame(table: u32, index: u32) i16 { +pub inline fn readShortViaGame(table: u32, index: u32) i16 { const values_ptr = ru32(table + 4); return ri16(values_ptr + index * 2); } @@ -749,7 +746,7 @@ inline fn readShortViaGame(table: u32, index: u32) i16 { /// Interpolate a Vec3 track (12 bytes per keyframe) with crossfade support. /// Writes result to output[3..5] (as u32 float bits). Uses output[0..2] for indices/t, /// and output[6..11] for secondary crossfade state. -inline fn interpVec3Track( +pub inline fn interpVec3Track( this: u32, bone_rt: u32, anim_data: u32, @@ -760,7 +757,7 @@ inline fn interpVec3Track( } /// Vec3 keyframe interpolation with optional cached primary InterpResult. -inline fn interpVec3TrackCached( +pub inline fn interpVec3TrackCached( this: u32, bone_rt: u32, anim_data: u32, @@ -806,7 +803,7 @@ inline fn interpVec3TrackCached( /// Interpolate a single float track (4 bytes per keyframe) with crossfade. /// Writes result to output[3] as float bits. -inline fn interpFloatTrack( +pub inline fn interpFloatTrack( this: u32, bone_rt: u32, anim_data: u32, @@ -846,7 +843,7 @@ inline fn interpFloatTrack( // Hermite/Bezier basis + particle emitter interp helpers // ============================================================================= -inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { +pub inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { const t2 = t * t; const t3 = t2 * t; return .{ @@ -857,7 +854,7 @@ inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { }; } -inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { +pub inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { const u = 1.0 - t; const t2 = t * t; const u_sq = u * u; @@ -869,7 +866,7 @@ inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { }; } -inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { +pub inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); const mode = ri16(anim_data + AD.interp_mode); @@ -950,7 +947,7 @@ inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output } } -inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { +pub inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); const mode = ri16(anim_data + AD.interp_mode); @@ -1007,7 +1004,7 @@ inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, outpu // Same as interpFloatTrack but uses the bone_rt directly (different register mapping) // ============================================================================= -inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void { +pub inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void { const r = findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data_short_ptr, output); const interp_mode = ri16(anim_data_short_ptr); @@ -1040,7 +1037,7 @@ inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr // applies inverse translation. // ============================================================================= -fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void { +pub fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void { // Simple transpose for unit scale if (@abs(scale - 1.0) < @as(f32, @bitCast(@as(u32, 0x35800000)))) { // Transpose 3x3 @@ -1472,68 +1469,60 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) const pivot_y = rf32(bdef + BD.pivot_y); const pivot_z = rf32(bdef + BD.pivot_z); - // Compute translated position - const tx = local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z + local_mat[12]; - const ty = local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z + local_mat[13]; - const tz = local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z + local_mat[14]; + // Compute translated position — accumulation order must match + // original x87. Row 0 uses (pz + px + py), rows 1/2 use (pz + py + px). + const tx = local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[12]; + const ty = local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x + local_mat[13]; + const tz = local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x + local_mat[14]; const bb_type = combined_flags & 6; + const billboard_eps_f64: f64 = @floatCast(rf32(0x008029d4)); + const cull_eps_f64: f64 = @floatCast(rf32(0x0080c5c8)); if (bb_type == 2) { - // Cylindrical billboard — normalize each column - const n0 = normalizeVec3(local_mat[0], local_mat[1], local_mat[2]); - local_mat[0] = n0[0]; - local_mat[1] = n0[1]; - local_mat[2] = n0[2]; - const n1 = normalizeVec3(local_mat[4], local_mat[5], local_mat[6]); - local_mat[4] = n1[0]; - local_mat[5] = n1[1]; - local_mat[6] = n1[2]; - const n2 = normalizeVec3(local_mat[8], local_mat[9], local_mat[10]); - local_mat[8] = n2[0]; - local_mat[9] = n2[1]; - local_mat[10] = n2[2]; + // Cylindrical billboard — normalize each column in f64 to + // match x87's extended-precision 1/sqrt. Previous f32 impl + // drifted from x87 by a ULP per axis, causing particle + // emitter orientation to jitter on camera motion and + // flicker against ground effects. + inline for ([_]u32{ 0, 4, 8 }) |row_off| { + const cx: f64 = @floatCast(local_mat[row_off]); + const cy: f64 = @floatCast(local_mat[row_off + 1]); + const cz: f64 = @floatCast(local_mat[row_off + 2]); + const len_sq = cx * cx + cy * cy + cz * cz; + const len = @sqrt(len_sq); + if (len >= billboard_eps_f64) { + const inv = 1.0 / len; + local_mat[row_off] = @floatCast(cx * inv); + local_mat[row_off + 1] = @floatCast(cy * inv); + local_mat[row_off + 2] = @floatCast(cz * inv); + } + } } else if (bb_type == 4) { - // Spherical billboard — inherit camera rotation with scale preservation - // All sqmag computations MUST call game's vec3SqMag (0x4549F0) - const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) }; - const cam_len_sq0 = callVec3SqMag(this + SO.bb_row0); - var s0: f32 = 1.0; - if (cam_len_sq0 > rf32(0x0080c5c8)) { - var tmp0 = [3]f32{ local_mat[0], local_mat[1], local_mat[2] }; - const mat_len_sq0 = callVec3SqMag(@intFromPtr(&tmp0)); - s0 = @sqrt(mat_len_sq0 / cam_len_sq0); + // Spherical billboard — inherit camera basis, rescale to + // preserve each column's length. All intermediates in f64 + // to match x87's 80-bit temporaries. + inline for ([_]struct { row_off: u32, src_off: u32 }{ + .{ .row_off = 0, .src_off = SO.bb_row0 }, + .{ .row_off = 4, .src_off = SO.world_xform }, + .{ .row_off = 8, .src_off = SO.world_xform + 16 }, + }) |p| { + const src_addr = this + p.src_off; + const cam_x: f64 = @floatCast(rf32(src_addr)); + const cam_y: f64 = @floatCast(rf32(src_addr + 4)); + const cam_z: f64 = @floatCast(rf32(src_addr + 8)); + const cam_len_sq = cam_x * cam_x + cam_y * cam_y + cam_z * cam_z; + var s: f64 = 1.0; + if (cam_len_sq > cull_eps_f64) { + const mx: f64 = @floatCast(local_mat[p.row_off]); + const my: f64 = @floatCast(local_mat[p.row_off + 1]); + const mz: f64 = @floatCast(local_mat[p.row_off + 2]); + const mat_len_sq = mx * mx + my * my + mz * mz; + s = @sqrt(mat_len_sq / cam_len_sq); + } + local_mat[p.row_off] = @floatCast(s * cam_x); + local_mat[p.row_off + 1] = @floatCast(s * cam_y); + local_mat[p.row_off + 2] = @floatCast(s * cam_z); } - local_mat[0] = s0 * cam0[0]; - local_mat[1] = s0 * cam0[1]; - local_mat[2] = s0 * cam0[2]; - - const wt0 = rf32(this + SO.world_xform + 0 * 4); - const wt1 = rf32(this + SO.world_xform + 1 * 4); - const wt2 = rf32(this + SO.world_xform + 2 * 4); - const wt_len_sq = callVec3SqMag(this + SO.world_xform); - var s1: f32 = 1.0; - if (wt_len_sq > rf32(0x0080c5c8)) { - var tmp1 = [3]f32{ local_mat[4], local_mat[5], local_mat[6] }; - const mat_len_sq1 = callVec3SqMag(@intFromPtr(&tmp1)); - s1 = @sqrt(mat_len_sq1 / wt_len_sq); - } - local_mat[4] = s1 * wt0; - local_mat[5] = s1 * wt1; - local_mat[6] = s1 * wt2; - - const wt4 = rf32(this + SO.world_xform + 4 * 4); - const wt5 = rf32(this + SO.world_xform + 5 * 4); - const wt6 = rf32(this + SO.world_xform + 6 * 4); - const wt_len_sq2 = callVec3SqMag(this + SO.world_xform + 16); - var s2: f32 = 1.0; - if (wt_len_sq2 > rf32(0x0080c5c8)) { - var tmp2 = [3]f32{ local_mat[8], local_mat[9], local_mat[10] }; - const mat_len_sq2 = callVec3SqMag(@intFromPtr(&tmp2)); - s2 = @sqrt(mat_len_sq2 / wt_len_sq2); - } - local_mat[8] = s2 * wt4; - local_mat[9] = s2 * wt5; - local_mat[10] = s2 * wt6; } else if (bb_type == 6) { // Full billboard — copy camera rotation directly local_mat[0] = rf32(this + SO.bb_row0); @@ -1547,11 +1536,12 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) local_mat[10] = rf32(this + SO.world_xform + 6 * 4); } - // Recompute translation: pos - rot * pivot + // Recompute translation: pos - rot * pivot. + // Accumulation order mirrors the tx/ty/tz computation above. if ((combined_flags & 1) == 0) { - local_mat[12] = tx - (local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z); - local_mat[13] = ty - (local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z); - local_mat[14] = tz - (local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z); + local_mat[12] = tx - (local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y); + local_mat[13] = ty - (local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x); + local_mat[14] = tz - (local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x); } else { local_mat[12] = rf32(this + SO.world_xform + 8 * 4); local_mat[13] = rf32(this + SO.world_xform + 9 * 4); @@ -1671,9 +1661,13 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) const bpx = rf32(bdef + BD.pivot_x); const bpy = rf32(bdef + BD.pivot_y); const bpz = rf32(bdef + BD.pivot_z); + // Accumulation order mirrors original x87: + // pos_x: px + py + pz + const + // pos_y: py + pz + px + const + // pos_z: py + pz + px + const const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30); - const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34); - const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38); + const pos_y = bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + bpx * rf32(om + 0x04) + rf32(om + 0x34); + const pos_z = bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + bpx * rf32(om + 0x08) + rf32(om + 0x38); // Switch on billboard post-processing type const bb_post = combined_flags & 0x78; @@ -1798,10 +1792,14 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) const r2z_s = rf32(om + 0x28); wf32(om + 0x28, scale_len2 * r2z_s); - // Recompute translation: pos - scaled_matrix * pivot - wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz)); - wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz)); - wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz)); + // Recompute translation: pos - scaled_matrix * pivot. + // Accumulation order must match original x87: row0 + row2 + row1 + // (pivot_x, then pivot_z, then pivot_y). f32 addition isn't associative — + // this ordering matters for matching terrain-pipeline precision and + // avoiding z-fighting on ground-aligned billboard spell effects. + wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len2 * r2x_s * bpz + scale_len1 * r1x_s * bpy)); + wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len2 * r2y_s * bpz + scale_len1 * r1y_s * bpy)); + wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len2 * r2z_s * bpz + scale_len1 * r1z_s * bpy)); wf32(om + 0x3C, 1.0); } } @@ -1818,8 +1816,6 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) // findInterpIdx + lerp + crossfade blend. // ========================================================================= - // BISECT: stop after section 7 (bone loop) - // Section 8: Texture animation loop texAnimLoop(this, model_hdr, frame_ctr); colorAnimLoop(this, model_hdr, frame_ctr); @@ -1841,7 +1837,7 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) // Post-bone-loop sections (extracted for readability) // ============================================================================= -fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); const count = ru32(model_hdr + 0x54); if (count == 0) return; @@ -1891,7 +1887,7 @@ fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { /// Short-value interpolation: uses InterpResult indices, looks up short values, interpolates. /// Shared by texAnimLoop alpha, colorAnimLoop, and word animation crossfade. -inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 { +pub inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 { const mode = ri16(anim_data); const table = anim_data + AD.nvalues; if (mode == 0) { @@ -1903,7 +1899,7 @@ inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 { } } -fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); // Assembly: model_hdr+0x64 is both entry gate AND loop count const count = ru32(model_hdr + 0x64); @@ -1946,7 +1942,7 @@ fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { } } -fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); // Assembly 0x715E46-0x715F25: word/byte animation section // model_hdr+0x6C = count, model_hdr+0x70 = data base @@ -1989,7 +1985,7 @@ fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { } } -fn boneKeyframeLoop(this: u32, model_hdr: u32) void { +pub fn boneKeyframeLoop(this: u32, model_hdr: u32) void { const count = ru32(model_hdr + 0x74); if (count == 0) return; @@ -2050,7 +2046,7 @@ fn boneKeyframeLoop(this: u32, model_hdr: u32) void { } } -fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { // Particle emitters are the largest section (~1000 lines of decompiled C). // They follow the same interpolation patterns but with many sub-tracks per emitter. // For the initial implementation, we handle the key tracks (position, speed, scale). @@ -2066,7 +2062,7 @@ fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { additionalParticleLoops(this, model_hdr, frame_ctr); } -fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); const count = ru32(model_hdr + 0x11C); if (count == 0) return; @@ -2136,7 +2132,7 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { } } -fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); const count = ru32(model_hdr + 0x124); if (count == 0) return; @@ -2169,7 +2165,7 @@ fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void { } } -fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { +pub fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); const stf = getShortToFloat(); // Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A) @@ -2377,7 +2373,7 @@ fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void { } } -fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void { +pub fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void { @setEvalBranchQuota(50000); const hierarchy = ru32(this + SO.hierarchy_ptr); if (hierarchy == 0) return; diff --git a/src/weirdperformance/bone_sse64.zig b/src/weirdperformance/bone_sse64.zig new file mode 100644 index 0000000..52531bd --- /dev/null +++ b/src/weirdperformance/bone_sse64.zig @@ -0,0 +1,913 @@ +//! f64-intermediate port of transformMatrix4x4 (0x714260). +//! +//! Mirrors bone_sse.zig but holds intermediate bone matrices as [16]f64 and +//! runs all matrix-math operations in double precision, only narrowing to f32 +//! when writing into the bone output buffer in game memory. +//! +//! Rounding profile matches the original x87 implementation: load f32, compute +//! in extended precision (here 53-bit mantissa f64 vs x87 64-bit mantissa — +//! indistinguishable once truncated to f32), store f32 once at the end. +//! +//! Interpolation, keyframe search, billboard math, ftol, and game-callback +//! helpers are imported from bone_sse since they produce f32 scalars that +//! widen implicitly when multiplied with f64 matrices. +//! +//! Compiled ReleaseFast with AVX enabled. Uses @Vector(4, f64) for SIMD +//! matrix multiplies on ymm registers. + +const bone_sse = @import("bone_sse.zig"); + +const V4d = @Vector(4, f64); + +// ============================================================================= +// Imports from bone_sse — constants, memory helpers, interp/billboard helpers +// ============================================================================= + +const SO = bone_sse.SO; +const BR = bone_sse.BR; +const BD = bone_sse.BD; +const AD = bone_sse.AD; +const InterpResult = bone_sse.InterpResult; + +const ru32 = bone_sse.ru32; +const ri32 = bone_sse.ri32; +const rf32 = bone_sse.rf32; +const ru16 = bone_sse.ru16; +const ri16 = bone_sse.ri16; +const ru8 = bone_sse.ru8; +const wu32 = bone_sse.wu32; +const wf32 = bone_sse.wf32; +const wu16 = bone_sse.wu16; +const wu8 = bone_sse.wu8; +const fbits = bone_sse.fbits; +const ufloat = bone_sse.ufloat; + +const normalizeVec3 = bone_sse.normalizeVec3; +const normalizeVec3InPlace = bone_sse.normalizeVec3InPlace; +const crossVec3 = bone_sse.crossVec3; + +const canReuseInterp = bone_sse.canReuseInterp; +const findInterpIdx = bone_sse.findInterpIdx; +const interpAnimKFCached = bone_sse.interpAnimKFCached; +const interpVec3TrackCached = bone_sse.interpVec3TrackCached; +const callFtol = bone_sse.callFtol; +const callVec3SqMag = bone_sse.callVec3SqMag; +const fastMod = bone_sse.fastMod; + +const texAnimLoop = bone_sse.texAnimLoop; +const colorAnimLoop = bone_sse.colorAnimLoop; +const wordAnimLoop = bone_sse.wordAnimLoop; +const boneKeyframeLoop = bone_sse.boneKeyframeLoop; +const particleLoops = bone_sse.particleLoops; +// NOTE: we do NOT import bone_sse.attachmentRecursion — that version recurses +// into bone_sse.transformImpl_SSE (f32) directly, which would force attached +// child models onto the f32 path while the parent is f64. We reimplement it +// below so attachment recursion stays inside bone_sse64's f64 pipeline. + +// ============================================================================= +// f64 memory helpers +// ============================================================================= + +inline fn rf64(addr: u32) f64 { + return @floatCast(rf32(addr)); +} + +inline fn wf32_narrow(addr: u32, v: f64) void { + wf32(addr, @floatCast(v)); +} + +inline fn loadV4d(addr: u32) V4d { + return V4d{ rf64(addr), rf64(addr + 4), rf64(addr + 8), rf64(addr + 12) }; +} + +inline fn storeV4d(addr: u32, v: V4d) void { + wf32(addr + 0, @floatCast(v[0])); + wf32(addr + 4, @floatCast(v[1])); + wf32(addr + 8, @floatCast(v[2])); + wf32(addr + 12, @floatCast(v[3])); +} + +// ============================================================================= +// f64 matrix operations +// ============================================================================= + +/// 4x4 matrix multiply: dst = a * b. Memory-to-memory with f64 intermediates. +inline fn matMul4x4_64(dst: u32, a: u32, b: u32) void { + const b0 = loadV4d(b); + const b1 = loadV4d(b + 16); + const b2 = loadV4d(b + 32); + const b3 = loadV4d(b + 48); + const a0 = loadV4d(a); + const a1 = loadV4d(a + 16); + const a2 = loadV4d(a + 32); + const a3 = loadV4d(a + 48); + const rows = [4]V4d{ a0, a1, a2, a3 }; + + inline for (0..4) |i| { + const s0: V4d = @splat(rows[i][0]); + const s1: V4d = @splat(rows[i][1]); + const s2: V4d = @splat(rows[i][2]); + const s3: V4d = @splat(rows[i][3]); + const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3; + const off: u32 = @intCast(i * 16); + storeV4d(dst + off, row); + } +} + +/// 4x4 multiply: dst_mem = a_local * b_mem. Left operand is [16]f64 local. +inline fn matMul4x4Local_64(dst: u32, a: [16]f64, b: u32) void { + const b0 = loadV4d(b); + const b1 = loadV4d(b + 16); + const b2 = loadV4d(b + 32); + const b3 = loadV4d(b + 48); + + const rows = [4]V4d{ + V4d{ a[0], a[1], a[2], a[3] }, + V4d{ a[4], a[5], a[6], a[7] }, + V4d{ a[8], a[9], a[10], a[11] }, + V4d{ a[12], a[13], a[14], a[15] }, + }; + + inline for (0..4) |i| { + const s0: V4d = @splat(rows[i][0]); + const s1: V4d = @splat(rows[i][1]); + const s2: V4d = @splat(rows[i][2]); + const s3: V4d = @splat(rows[i][3]); + const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3; + const off: u32 = @intCast(i * 16); + storeV4d(dst + off, row); + } +} + +/// In-place f64 multiply: a = a * b_mem. Returns new [16]f64. +inline fn matMul4x4InPlace_64(a: [16]f64, b: u32) [16]f64 { + const b0 = loadV4d(b); + const b1 = loadV4d(b + 16); + const b2 = loadV4d(b + 32); + const b3 = loadV4d(b + 48); + + const rows = [4]V4d{ + V4d{ a[0], a[1], a[2], a[3] }, + V4d{ a[4], a[5], a[6], a[7] }, + V4d{ a[8], a[9], a[10], a[11] }, + V4d{ a[12], a[13], a[14], a[15] }, + }; + + var result: [16]f64 = undefined; + inline for (0..4) |i| { + const s0: V4d = @splat(rows[i][0]); + const s1: V4d = @splat(rows[i][1]); + const s2: V4d = @splat(rows[i][2]); + const s3: V4d = @splat(rows[i][3]); + const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3; + result[i * 4 + 0] = row[0]; + result[i * 4 + 1] = row[1]; + result[i * 4 + 2] = row[2]; + result[i * 4 + 3] = row[3]; + } + return result; +} + +/// Quaternion → 4x4 rotation matrix as f64 local. Identity last row/col. +inline fn buildRotationMatrix_64(qx: f64, qy: f64, qz: f64, qw: f64) [16]f64 { + const xx2 = qx * (qx + qx); + const xy2 = qx * (qy + qy); + const xz2 = qx * (qz + qz); + const yy2 = qy * (qy + qy); + const yz2 = qy * (qz + qz); + const zz2 = qz * (qz + qz); + const wx2 = qw * (qx + qx); + const wy2 = qw * (qy + qy); + const wz2 = qw * (qz + qz); + + return [16]f64{ + 1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0.0, + xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0.0, + xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0.0, + 0.0, 0.0, 0.0, 1.0, + }; +} + +/// Copy a 4x4 matrix in game memory (bit-exact, no precision change). +inline fn copyMat4(dst: u32, src: u32) void { + inline for (0..8) |i| { + const off = @as(u32, @intCast(i)) * 8; + @as(*u64, @ptrFromInt(dst + off)).* = @as(*const u64, @ptrFromInt(src + off)).*; + } +} + +// ============================================================================= +// Attachment recursion — f64 version that re-enters transformImpl_SSE64 for +// child scene objects. Mirrors bone_sse.attachmentRecursion exactly except +// it calls our f64 implementation for child transforms. +// ============================================================================= + +fn attachmentRecursion64(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void { + @setEvalBranchQuota(50000); + const hierarchy = ru32(this + SO.hierarchy_ptr); + if (hierarchy == 0) return; + + const attach_count = ru32(model_hdr + 0x104); + const attach_data = ru32(model_hdr + 0x108); + + var att_i: u32 = 0; + var att_off: u32 = 0; + while (att_i < attach_count) : ({ + att_i += 1; + att_off += 0x30; + }) { + const att_entry = attach_data + att_off; + if (frame_ctr < ru32(att_entry + 0x20)) { + const bone_idx = @as(u32, ru16(att_entry + 4)); + const bone_rt = ru32(this + SO.bone_rt_base) + bone_idx * 0x118; + const anim_data = att_entry + 0x14; + const att_output = hierarchy + att_i * 0x20; + const atr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), anim_data, att_output); + wu8(att_output + 0x0C, ru8(ru32(anim_data + AD.keyframe_base) + atr.idx0)); + } + } + + var child = ru32(this + SO.hierarchy_idx); + while (child != 0) { + const attach_idx = ru32(child + 0x1D4); + + if (attach_idx != 0xFFFF) { + const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C); + if (visible != 0) { + const att_entry = attach_data + attach_idx * 0x30; + const bone_idx = @as(u32, ru16(att_entry + 4)); + const bone_mat = bone_out_base + bone_idx * 0x40; + + // Copy parent bone matrix to local (f32, matches original). + // The matrix itself is stored f32 in game memory; we preserve + // that on the wire but widen to f64 for the offset math below. + var local_1a0: [16]f32 = undefined; + for (0..16) |fi| { + local_1a0[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4); + } + + // Apply attachment offset translation in f64, narrow on store. + const ox: f64 = @floatCast(rf32(att_entry + 8)); + const oy: f64 = @floatCast(rf32(att_entry + 0xC)); + const oz: f64 = @floatCast(rf32(att_entry + 0x10)); + const m0x: f64 = @floatCast(local_1a0[0]); + const m4x: f64 = @floatCast(local_1a0[4]); + const m8x: f64 = @floatCast(local_1a0[8]); + const m1x: f64 = @floatCast(local_1a0[1]); + const m5x: f64 = @floatCast(local_1a0[5]); + const m9x: f64 = @floatCast(local_1a0[9]); + const m2x: f64 = @floatCast(local_1a0[2]); + const m6x: f64 = @floatCast(local_1a0[6]); + const m10x: f64 = @floatCast(local_1a0[10]); + local_1a0[12] = @floatCast(@as(f64, @floatCast(local_1a0[12])) + m0x * ox + m4x * oy + m8x * oz); + local_1a0[13] = @floatCast(@as(f64, @floatCast(local_1a0[13])) + m1x * ox + m5x * oy + m9x * oz); + local_1a0[14] = @floatCast(@as(f64, @floatCast(local_1a0[14])) + m2x * ox + m6x * oy + m10x * oz); + + // Recurse into the f64 implementation so attachment children stay + // on the same precision path as the parent. + transformImpl_SSE64(child, @intFromPtr(&local_1a0), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z)); + } + } + + child = ru32(child + 0x1E4); + } +} + +// ============================================================================= +// Main entry point — f64-intermediate reimplementation. +// +// Follows the same structure as bone_sse.transformImpl_SSE. Per-bone work +// builds local_mat / local_mat2 as [16]f64 and uses the f64 matrix helpers +// above. Non-matrix helpers (interp, billboard, loops) are imported from +// bone_sse — their f32 outputs widen implicitly when used in f64 math. +// Attachment recursion is handled by a local f64 version (above) so child +// scene objects stay on the f64 pipeline. +// ============================================================================= + +pub fn transformImpl_SSE64(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void { + @setEvalBranchQuota(50000); + + // Section 1: Entry checks + if (ru32(this + SO.model_data_ptr) == 0) return; + const anim_ctx = ru32(this + SO.anim_ctx_ptr); + if (ru32(this + SO.sync_value) == ru32(anim_ctx + 0x10)) return; + + // Section 2: Emitter setup + const model_ctr = ru32(this + SO.model_ctr_ptr); + const model_hdr = ru32(model_ctr + 0x130); + const emitter_ctx = ru32(this + SO.emitter_ctx); + + if (emitter_ctx != 0) { + const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0; + wu32(this + 0x50, has_emitter); + wu32(this + 0x17C, ru32(emitter_ctx + 0x17C)); + } + + // Section 3: World position/scale + const pos_ptr = mat2; + const ofs_ptr = mat3; + const scale_f: f32 = @bitCast(mat4); + + wf32(this + SO.world_pos + 0, rf32(pos_ptr) * rf32(this + SO.field_184)); + wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(pos_ptr + 4)); + wf32(this + SO.world_pos + 8, @bitCast(fbits(rf32(this + SO.field_18c) * rf32(pos_ptr + 8)))); + + const rp0 = rf32(ofs_ptr) + rf32(this + SO.field_190); + const rp1 = rf32(this + SO.render_scale_x) + rf32(ofs_ptr + 4); + const rp2 = rf32(this + SO.render_scale_y) + rf32(ofs_ptr + 8); + wf32(this + SO.render_pri + 0, rp0); + wf32(this + SO.render_pri + 4, rp1); + wf32(this + SO.render_pri + 8, rp2); + + wf32(this + SO.render_scale_z, scale_f * rf32(this + SO.field_180)); + + // Section 4: Global sequence processing + const gs_count = ru32(model_hdr + 0x14); + if (gs_count != 0) { + const gs_durations = ru32(model_hdr + 0x18); + const gs_values = ru32(this + SO.gs_values_ptr); + const timestamp = ru32(anim_ctx + 0x0C); + const time_base = ru32(this + SO.gs_time_base); + var gi: u32 = 0; + while (gi < gs_count) : (gi += 1) { + const dur = ru32(gs_durations + gi * 4); + if (dur == 0) { + wu32(gs_values + gi * 4, 0); + } else { + wu32(gs_values + gi * 4, (timestamp -% time_base) % dur); + } + } + } + + // Root matrix multiply (f64 intermediates). + // Replaces the original initParticlePixelShaderGeneration(0x74a7c0) dispatch. + matMul4x4_64(this + 0xFC, this + 0xBC, mat1); + + // Section 5: child_objects_padding — compute in f64, narrow on store. + // May be used as a culling threshold; matching precision keeps boundary + // conditions stable across frames. + const emitter_ctx_5 = ru32(this + SO.emitter_ctx); + if (emitter_ctx_5 == 0 or (ru8(emitter_ctx_5 + 4) & 1) != 0) { + const wx: f64 = rf64(this + SO.world_xform + 8 * 4); + const wy: f64 = rf64(this + SO.world_xform + 9 * 4); + const wz: f64 = rf64(this + SO.world_xform + 10 * 4); + const len_sq: f32 = @floatCast(wx * wx + wy * wy + wz * wz); + wu32(this + SO.child_padding, fbits(len_sq)); + } else { + wu32(this + SO.child_padding, ru32(emitter_ctx_5 + 0x84)); + } + + // Section 6: Identity matrices as f64 locals + timestamp delta + var local_mat: [16]f64 = .{ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + }; + + var local_mat2: [16]f64 = .{ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + }; + + var time_delta_val: u32 = 0; + const sdb = ru32(this + SO.search_data_base); + if (sdb != 0) { + const cur_ts = ru32(anim_ctx + 0x0C); + if (cur_ts != 0) { + time_delta_val = cur_ts -% sdb; + wu32(this + SO.search_data_base, cur_ts); + } + } + + // Section 7: Main bone loop + const bone_count = ru32(model_hdr + 0x34); + const bone_defs = ru32(model_hdr + 0x38); + const bone_rt_base = ru32(this + SO.bone_rt_base); + const bone_out_base = ru32(this + SO.bone_out_ptr); + const frame_ctr = ru32(this + SO.anim_frame_ctr); + + if (bone_count != 0) { + var bone_idx: u32 = 0; + var bdef = bone_defs; + var brt = bone_rt_base; + while (bone_idx < bone_count) : ({ + bone_idx += 1; + bdef += 0x6C; + brt += 0x118; + }) { + const flags = ru32(bdef + BD.flags); + const parent_idx_raw: i32 = @as(i32, @intCast(@as(i16, @bitCast(ru16(bdef + BD.parent_bone))))); + + // --- Animation time computation (primary slot) --- + const anim_slot_val = ri32(brt + BR.anim_slot); + if (anim_slot_val == -1) { + if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { + const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118; + wu32(brt + BR.prim_time, ru32(parent_rt + BR.prim_time)); + wu32(brt + BR.prim_track, ru32(parent_rt + BR.prim_track)); + wu32(brt + BR.prim_anim, ru32(parent_rt + BR.prim_anim)); + } else if (bone_idx != 0) { + wu32(brt + BR.prim_time, ru32(bone_rt_base + BR.prim_time)); + wu32(brt + BR.prim_track, ru32(bone_rt_base + BR.prim_track)); + wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim)); + } + } else { + if (ru32(this + 0x4C) != 0) { + wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val); + wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val); + } + + const anim_lookup = ru32(model_hdr + 0x20); + const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44; + const cur_time = ru32(ru32(this + 0x2C) + 0xC); + + if ((ru8(anim_entry + 0x10) & 1) == 0) { + const anim_end = ru32(anim_entry + 0x08); + const anim_start = ru32(anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = cur_time -% ru32(brt + 0xA8); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start); + wu32(brt + 0x98, anim_start +% frame); + } else { + wu32(brt + 0x98, anim_start); + } + } else { + const sec_end_val = ru32(brt + 0xAC); + const sec_start_val = ru32(brt + 0xA8); + + if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) { + const effective_time = if (@as(i32, @bitCast(sec_start_val -% cur_time)) > 0) sec_start_val else cur_time; + const anim_end = ru32(anim_entry + 0x08); + const anim_start = ru32(anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = effective_time -% ru32(brt + 0xA8); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start); + wu32(brt + 0x98, anim_start +% frame); + } else { + wu32(brt + 0x98, anim_start); + } + } else { + const dur = sec_end_val -% sec_start_val; + const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xB0); + const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8))); + + if (offset < 0) { + wu32(brt + 0x98, ru32(anim_entry + 0x04)); + } else { + const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08))); + const anim_start_i = @as(i32, @bitCast(ru32(anim_entry + 0x04))); + if (offset <= anim_end_i - anim_start_i) { + wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i))); + } else { + wu32(brt + 0x98, ru32(anim_entry + 0x08)); + } + } + } + } + + wu32(brt + 0x9C, ru32(brt + 0xA4)); + wu32(brt + 0xA0, bone_idx); + } + + // --- Secondary slot --- + const sec_slot_val = ri32(brt + BR.sec_slot); + if (sec_slot_val == -1) { + if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { + const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118; + wu32(brt + BR.sec_time, ru32(parent_rt + BR.sec_time)); + wu32(brt + BR.sec_track, ru32(parent_rt + BR.sec_track)); + } else if (bone_idx != 0) { + wu32(brt + BR.sec_time, ru32(bone_rt_base + BR.sec_time)); + wu32(brt + BR.sec_track, ru32(bone_rt_base + BR.sec_track)); + } else { + wu32(brt + BR.sec_time, ru32(brt + BR.prim_time)); + wu32(brt + BR.sec_track, ru32(brt + BR.prim_track)); + } + } else { + if (ru32(this + 0x4C) != 0) { + wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val); + wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val); + } + + const sec_anim_lookup = ru32(model_hdr + 0x20); + const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44; + const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC); + + if ((ru8(sec_anim_entry + 0x10) & 1) == 0) { + const anim_end = ru32(sec_anim_entry + 0x08); + const anim_start = ru32(sec_anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = sec_cur_time -% ru32(brt + 0xD4); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start); + wu32(brt + 0xC4, anim_start +% frame); + } else { + wu32(brt + 0xC4, anim_start); + } + } else { + const sec_end_val = ru32(brt + 0xD8); + const sec_start_val = ru32(brt + 0xD4); + + if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) { + const effective_time = if (@as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) sec_start_val else sec_cur_time; + const anim_end = ru32(sec_anim_entry + 0x08); + const anim_start = ru32(sec_anim_entry + 0x04); + if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + const delta = effective_time -% ru32(brt + 0xD4); + const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC); + const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start); + wu32(brt + 0xC4, anim_start +% frame); + } else { + wu32(brt + 0xC4, anim_start); + } + } else { + const dur = sec_end_val -% sec_start_val; + const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xDC); + const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4))); + + if (offset < 0) { + wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04)); + } else { + const anim_end_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x08))); + const anim_start_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x04))); + if (offset <= anim_end_i - anim_start_i) { + wu32(brt + 0xC4, @as(u32, @bitCast(offset + anim_start_i))); + } else { + wu32(brt + 0xC4, ru32(sec_anim_entry + 0x08)); + } + } + } + } + + wu32(brt + 0xC8, ru32(brt + 0xD0)); + + if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) { + wu32(brt + 0xD0, 0xFFFFFFFF); + } + } + + // --- Blend weight --- + if (ri32(brt + BR.anim_slot) == -1 and ri32(brt + BR.sec_slot) == -1) { + if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) { + wu32(brt + BR.blend_weight, ru32(bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118 + BR.blend_weight)); + } else if (bone_idx == 0) { + wu32(brt + BR.blend_weight, 0); + } else { + wu32(brt + BR.blend_weight, ru32(bone_rt_base + BR.blend_weight)); + } + } else { + const cf_remaining = ri32(brt + BR.crossfade_end) - ri32(anim_ctx + 0x0C); + if (cf_remaining < 1 or (ru32(brt + BR.prim_time) == ru32(brt + BR.sec_time) and + ru32(brt + BR.prim_track) == ru32(brt + BR.sec_track))) + { + wu32(brt + BR.blend_weight, 0); + } else { + const t_raw = @as(f32, @floatFromInt(cf_remaining)) * ufloat(ru32(brt + BR.crossfade_inv)); + const t_clamped = if (t_raw < 0.0) @as(f32, 0.0) else if (t_raw > 1.0) @as(f32, 1.0) else t_raw; + const h = (3.0 - 2.0 * t_clamped) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight)); + wu32(brt + BR.blend_weight, fbits(h)); + } + } + + // --- Parent bone transform inheritance --- + const combined_flags: u32 = ru32(brt + BR.flags2) | flags; + var src_mat: u32 = undefined; + + // Address of local_mat (as if it were a game-memory matrix — since the + // matMul helpers read from memory, we write local_mat out to a scratch + // f32 buffer when src_mat aliases it. The billboard path below stores + // f32-narrowed values into local_mat's address via a scratch buffer.) + var local_mat_f32: [16]f32 = undefined; + const local_mat_addr = @intFromPtr(&local_mat_f32); + + if (ru16(bdef + BD.parent_bone) == 0xFFFF) { + src_mat = this + 0xFC; + } else { + const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40; + src_mat = parent_out; + + if ((combined_flags & 7) != 0) { + // Copy parent matrix to local_mat (f64 widen) + for (0..16) |i| { + local_mat[i] = rf64(parent_out + @as(u32, @intCast(i)) * 4); + } + + const pivot_x: f64 = rf64(bdef + BD.pivot_x); + const pivot_y: f64 = rf64(bdef + BD.pivot_y); + const pivot_z: f64 = rf64(bdef + BD.pivot_z); + + // Compute translated position — accumulation order must match + // original x87. Row 0 uses (pz + px + py), rows 1/2 use (pz + py + px). + const tx = local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[12]; + const ty = local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x + local_mat[13]; + const tz = local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x + local_mat[14]; + + const bb_type = combined_flags & 6; + const billboard_eps: f64 = @floatCast(rf32(0x008029d4)); + const cull_eps: f64 = @floatCast(rf32(0x0080c5c8)); + if (bb_type == 2) { + // Cylindrical billboard — normalize columns in f64 to match + // x87's extended-precision 1/sqrt pattern. Previous impl went + // f32→f32 through normalizeVec3; that diverged from x87 by + // up to a ULP per axis, which shifted particle emitter + // orientation each camera frame and caused transparency + // flicker against ground effects. + inline for ([_]u32{ 0, 4, 8 }) |row_off| { + const cx = local_mat[row_off]; + const cy = local_mat[row_off + 1]; + const cz = local_mat[row_off + 2]; + const len_sq = cx * cx + cy * cy + cz * cz; + const len = @sqrt(len_sq); + if (len >= billboard_eps) { + const inv = 1.0 / len; + local_mat[row_off] = cx * inv; + local_mat[row_off + 1] = cy * inv; + local_mat[row_off + 2] = cz * inv; + } + } + } else if (bb_type == 4) { + // Spherical billboard — inherit camera basis, rescale to + // preserve original column length. Keep all intermediates + // in f64 matching x87's 80-bit temporaries. + inline for ([_]struct { row_off: u32, src_off: u32 }{ + .{ .row_off = 0, .src_off = SO.bb_row0 }, + .{ .row_off = 4, .src_off = SO.world_xform }, + .{ .row_off = 8, .src_off = SO.world_xform + 16 }, + }) |p| { + const src_addr = this + p.src_off; + const cam_x: f64 = rf64(src_addr); + const cam_y: f64 = rf64(src_addr + 4); + const cam_z: f64 = rf64(src_addr + 8); + const cam_len_sq = cam_x * cam_x + cam_y * cam_y + cam_z * cam_z; + var s: f64 = 1.0; + if (cam_len_sq > cull_eps) { + const mx = local_mat[p.row_off]; + const my = local_mat[p.row_off + 1]; + const mz = local_mat[p.row_off + 2]; + const mat_len_sq = mx * mx + my * my + mz * mz; + s = @sqrt(mat_len_sq / cam_len_sq); + } + local_mat[p.row_off] = s * cam_x; + local_mat[p.row_off + 1] = s * cam_y; + local_mat[p.row_off + 2] = s * cam_z; + } + } else if (bb_type == 6) { + local_mat[0] = rf64(this + SO.bb_row0); + local_mat[1] = rf64(this + SO.bb_row0 + 4); + local_mat[2] = rf64(this + SO.bb_row0 + 8); + local_mat[4] = rf64(this + SO.world_xform + 0 * 4); + local_mat[5] = rf64(this + SO.world_xform + 1 * 4); + local_mat[6] = rf64(this + SO.world_xform + 2 * 4); + local_mat[8] = rf64(this + SO.world_xform + 4 * 4); + local_mat[9] = rf64(this + SO.world_xform + 5 * 4); + local_mat[10] = rf64(this + SO.world_xform + 6 * 4); + } + + // Recompute translation — mirror accumulation order of tx/ty/tz above. + if ((combined_flags & 1) == 0) { + local_mat[12] = tx - (local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y); + local_mat[13] = ty - (local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x); + local_mat[14] = tz - (local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x); + } else { + local_mat[12] = rf64(this + SO.world_xform + 8 * 4); + local_mat[13] = rf64(this + SO.world_xform + 9 * 4); + local_mat[14] = rf64(this + SO.world_xform + 10 * 4); + } + + // Narrow local_mat to f32 scratch, set src_mat to its address + for (0..16) |i| { + local_mat_f32[i] = @floatCast(local_mat[i]); + } + src_mat = local_mat_addr; + } + } + + // --- Rotation / Scale / Translation / Final multiply --- + if ((combined_flags & 0x280) == 0) { + const dst = bone_out_base + bone_idx * 0x40; + copyMat4(dst, src_mat); + } else { + const rot_anim = bdef + BD.rot_anim; + const rot_kf_count = ru32(bdef + BD.rot_nts); + + var rot_primary_cache: ?InterpResult = null; + + if (rot_kf_count != 0) { + if (frame_ctr < rot_kf_count) { + const rot_output = brt + BR.rot_idx0; + const r = findInterpIdx(this, ru32(brt + BR.prim_time), ru32(brt + BR.prim_track), rot_anim, rot_output); + rot_primary_cache = r; + const q = interpAnimKFCached(this, brt, rot_anim, rot_output, r); + local_mat2 = buildRotationMatrix_64(q[0], q[1], q[2], q[3]); + } else { + local_mat2 = buildRotationMatrix_64(rf64(brt + BR.rot_x), rf64(brt + BR.rot_y), rf64(brt + BR.rot_z), rf64(brt + BR.rot_w)); + } + } else { + local_mat2 = .{ + 1, 0, 0, 0, + 0, 1, 0, 0, + 0, 0, 1, 0, + 0, 0, 0, 1, + }; + } + + // Step 2: Scale interpolation — f64 arithmetic + const scale_anim = bdef + BD.scale_anim; + const scale_kf_count = ru32(bdef + BD.scale_nts); + if (scale_kf_count != 0) { + var sx: f64 = undefined; + var sy: f64 = undefined; + var sz: f64 = undefined; + if (frame_ctr < scale_kf_count) { + const scale_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, scale_anim)) rot_primary_cache else null; + const s = interpVec3TrackCached(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)), scale_cache); + sx = s[0]; sy = s[1]; sz = s[2]; + } else { + sx = rf64(brt + BR.scale_x); sy = rf64(brt + BR.scale_y); sz = rf64(brt + BR.scale_z); + } + local_mat2[0] *= sx; local_mat2[1] *= sx; local_mat2[2] *= sx; + local_mat2[4] *= sy; local_mat2[5] *= sy; local_mat2[6] *= sy; + local_mat2[8] *= sz; local_mat2[9] *= sz; local_mat2[10] *= sz; + } + + // Conditional bone-flag matrix multiply (f64 in-place) + if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) { + local_mat2 = matMul4x4InPlace_64(local_mat2, ru32(brt + BR.bone_flag_cache)); + } + + // Step 3: Translation interpolation (f64) + var tx_val: f64 = rf64(bdef + BD.pivot_x); + var ty_val: f64 = rf64(bdef + BD.pivot_y); + var tz_val: f64 = rf64(bdef + BD.pivot_z); + + const trans_anim = bdef + BD.trans_anim; + const trans_kf_count = ru32(bdef + BD.trans_nts); + if (trans_kf_count != 0) { + if (frame_ctr < trans_kf_count) { + const trans_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, trans_anim)) rot_primary_cache else null; + const t = interpVec3TrackCached(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)), trans_cache); + tx_val += @as(f64, t[0]); + ty_val += @as(f64, t[1]); + tz_val += @as(f64, t[2]); + } else { + tx_val += rf64(brt + BR.trans_x); + ty_val += rf64(brt + BR.trans_y); + tz_val += rf64(brt + BR.trans_z); + } + } + + // Step 4: Translation offset using ROTATED+SCALED matrix (f64) + const piv_x: f64 = rf64(bdef + BD.pivot_x); + const piv_y: f64 = rf64(bdef + BD.pivot_y); + const piv_z: f64 = rf64(bdef + BD.pivot_z); + local_mat2[12] = tx_val - (local_mat2[0] * piv_x + local_mat2[4] * piv_y + local_mat2[8] * piv_z); + local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z); + local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z); + + // Final: dst = bone_local * parent (f64 mul, narrow on store) + matMul4x4Local_64(bone_out_base + bone_idx * 0x40, local_mat2, src_mat); + } + + // --- Billboard post-processing (flags & 0x78) --- + if ((combined_flags & 0x78) != 0) { + const out_off = bone_idx * 0x40; + const om = bone_out_base + out_off; + + const scale_len0 = @sqrt(callVec3SqMag(om)); + const scale_len1 = @sqrt(callVec3SqMag(om + 0x10)); + const scale_len2 = @sqrt(callVec3SqMag(om + 0x20)); + + const bpx = rf32(bdef + BD.pivot_x); + const bpy = rf32(bdef + BD.pivot_y); + const bpz = rf32(bdef + BD.pivot_z); + // Accumulation order mirrors original x87: + // pos_x: px + py + pz + const + // pos_y: py + pz + px + const + // pos_z: py + pz + px + const + const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30); + const pos_y = bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + bpx * rf32(om + 0x04) + rf32(om + 0x34); + const pos_z = bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + bpx * rf32(om + 0x08) + rf32(om + 0x38); + + const bb_post = combined_flags & 0x78; + switch (bb_post) { + 0x08 => { + const had_anim = (combined_flags & 0x280) != 0; + if (!had_anim) { + wf32(om, 0); + wf32(om + 0x04, 0); + wf32(om + 0x08, -1); + wf32(om + 0x10, 1); + wf32(om + 0x14, 0); + wf32(om + 0x18, 0); + wf32(om + 0x20, 0); + wf32(om + 0x24, 1); + wf32(om + 0x28, 0); + } else { + // Row 0 = {local_e4, local_e0, -local_e8} + const r0x: f32 = @floatCast(local_mat2[1]); + const r0y: f32 = @floatCast(local_mat2[2]); + const r0z: f32 = @floatCast(-local_mat2[0]); + wf32(om, r0x); + wf32(om + 0x04, r0y); + wf32(om + 0x08, r0z); + normalizeVec3InPlace(om); + const r1x: f32 = @floatCast(local_mat2[5]); + const r1y: f32 = @floatCast(local_mat2[6]); + const r1z: f32 = @floatCast(-local_mat2[4]); + wf32(om + 0x10, r1x); + wf32(om + 0x14, r1y); + wf32(om + 0x18, r1z); + normalizeVec3InPlace(om + 0x10); + const r2x: f32 = @floatCast(local_mat2[9]); + const r2y: f32 = @floatCast(local_mat2[10]); + const r2z: f32 = @floatCast(-local_mat2[8]); + wf32(om + 0x20, r2x); + wf32(om + 0x24, r2y); + wf32(om + 0x28, r2z); + normalizeVec3InPlace(om + 0x20); + } + }, + 0x10 => { + normalizeVec3InPlace(om); + const r0x = rf32(om); + const r0y = rf32(om + 0x04); + wf32(om + 0x10, r0y); + wf32(om + 0x14, -r0x); + wf32(om + 0x18, 0); + normalizeVec3InPlace(om + 0x10); + wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18)); + wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10)); + wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14)); + }, + 0x20 => { + normalizeVec3InPlace(om + 0x10); + wf32(om, -rf32(om + 0x14)); + wf32(om + 0x04, rf32(om + 0x10)); + wf32(om + 0x08, 0); + normalizeVec3InPlace(om); + wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18)); + wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10)); + wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14)); + }, + 0x40 => { + normalizeVec3InPlace(om + 0x20); + wf32(om + 0x10, rf32(om + 0x24)); + wf32(om + 0x14, -rf32(om + 0x20)); + wf32(om + 0x18, 0); + normalizeVec3InPlace(om + 0x10); + wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14)); + wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18)); + wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10)); + }, + else => {}, + } + + // Apply scale lengths back and recompute translation + wf32(om + 0x0C, 0); + wf32(om + 0x1C, 0); + wf32(om + 0x2C, 0); + const r0x_s = rf32(om); + wf32(om, scale_len0 * r0x_s); + const r0y_s = rf32(om + 0x04); + wf32(om + 0x04, scale_len0 * r0y_s); + const r0z_s = rf32(om + 0x08); + wf32(om + 0x08, scale_len0 * r0z_s); + const r1x_s = rf32(om + 0x10); + wf32(om + 0x10, scale_len1 * r1x_s); + const r1y_s = rf32(om + 0x14); + wf32(om + 0x14, scale_len1 * r1y_s); + const r1z_s = rf32(om + 0x18); + wf32(om + 0x18, scale_len1 * r1z_s); + const r2x_s = rf32(om + 0x20); + wf32(om + 0x20, scale_len2 * r2x_s); + const r2y_s = rf32(om + 0x24); + wf32(om + 0x24, scale_len2 * r2y_s); + const r2z_s = rf32(om + 0x28); + wf32(om + 0x28, scale_len2 * r2z_s); + + // Accumulation order must match original x87: row0 + row2 + row1 + // (pivot_x, then pivot_z, then pivot_y). f32 addition isn't associative — + // this ordering is load-bearing for spell-effect z-fighting avoidance. + wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len2 * r2x_s * bpz + scale_len1 * r1x_s * bpy)); + wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len2 * r2y_s * bpz + scale_len1 * r1y_s * bpy)); + wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len2 * r2z_s * bpz + scale_len1 * r1z_s * bpy)); + wf32(om + 0x3C, 1.0); + } + } + } + + // Sections 8-13: delegated to bone_sse's f32 implementations. + texAnimLoop(this, model_hdr, frame_ctr); + colorAnimLoop(this, model_hdr, frame_ctr); + wordAnimLoop(this, model_hdr, frame_ctr); + boneKeyframeLoop(this, model_hdr); + particleLoops(this, model_hdr, frame_ctr); + attachmentRecursion64(this, model_hdr, bone_out_base, frame_ctr); + + wu32(this + SO.sync_value, ru32(anim_ctx + 0x10)); +} diff --git a/src/weirdperformance/weirdperformance.zig b/src/weirdperformance/weirdperformance.zig index a70cd3b..8f39d18 100644 --- a/src/weirdperformance/weirdperformance.zig +++ b/src/weirdperformance/weirdperformance.zig @@ -38,6 +38,7 @@ pub fn isActive() bool { // ============================================================================= const bone_sse = @import("bone_sse.zig"); +const bone_sse64 = @import("bone_sse64.zig"); const particle_sse = @import("particle_sse.zig"); const clip_sse = @import("clip_sse.zig"); const cull_sse = @import("cull_sse.zig"); @@ -195,6 +196,72 @@ fn destroyObjMgrDetour() callconv(hook.cc.stdcall) void { destroy_objmgr_hook.callOriginal(.{}); } +// ============================================================================= +// Spell ground effect diagnostic — safe cross-reference approach +// createModelAttachment saves model ptrs, ManageRenderListNode checks matches. +// No deferred pointer reads — only value comparisons. +// ============================================================================= + +const ModelAttachFn = fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; +var model_attach_hook: hook.Detour(ModelAttachFn) = .{}; + +const MAX_TRACKED = 16; +var tracked_models: [MAX_TRACKED]u32 = [_]u32{0} ** MAX_TRACKED; +var tracked_next: u32 = 0; + +fn modelAttachDetour(parent: u32, path_ptr: u32, flags: u32) callconv(.{ .x86_thiscall = .{} }) u32 { + const result = model_attach_hook.callOriginal(.{ parent, path_ptr, flags }); + + if (path_ptr != 0 and result != 0) { + const path: [*]const u8 = @ptrFromInt(path_ptr); + if (path[0] == 'S' and path[1] == 'p' and path[2] == 'e' and path[3] == 'l' and path[4] == 'l' and path[5] == 's') { + const path_z: [*:0]const u8 = @ptrFromInt(path_ptr); + log.fmt("[spell] created 0x{x}: {s}\n", .{ result, path_z }); + tracked_models[tracked_next % MAX_TRACKED] = result; + tracked_next +%= 1; + } + } + return result; +} + +// CM2Model_ManageRenderListNode (0x710B90) +// __thiscall(ECX=model, add_remove), RET 0x4 +const ManageRLFn = fn (u32, u32) callconv(.{ .x86_thiscall = .{} }) void; +var manage_rl_hook: hook.Detour(ManageRLFn) = .{}; + +fn manageRLDetour(model: u32, add_remove: u32) callconv(.{ .x86_thiscall = .{} }) void { + for (&tracked_models) |tp| { + if (tp != 0 and tp == model) { + if (add_remove != 0) { + log.fmt("[spell] 0x{x} ADDED to render list\n", .{model}); + } else { + log.fmt("[spell] 0x{x} REMOVED from render list\n", .{model}); + } + break; + } + } + manage_rl_hook.callOriginal(.{ model, add_remove }); +} + +// ============================================================================= +// ProcessProjectileMovementWithCollisionAndTargetValidation fix (0x61e1d0) +// __thiscall(ECX=missile, target_ptr, param_3), RET 0x8 +// +// Vanilla bug: when target_ptr != 0 (a unit/dynobj exists at the AoE location), +// the code takes an alternate path that skips ProcessMissileSpellEffects entirely. +// This means the area effect ground model (e.g. InfectedSecretion_Marked.m2) is +// never created. Fix: force target_ptr=0 so the area effect path always runs. +// Target-unit visuals fire separately through ProcessSpellVisualKit. +// ============================================================================= + +const ProjMoveFn = fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; +var proj_move_hook: hook.Detour(ProjMoveFn) = .{}; + +fn projMoveDetour(missile: u32, target_ptr: u32, param_3: u32) callconv(.{ .x86_thiscall = .{} }) void { + _ = target_ptr; + proj_move_hook.callOriginal(.{ missile, 0, param_3 }); +} + // ============================================================================= // OnWorldUpdate hook (0x482EA0) — per-frame cache reset // ============================================================================= @@ -298,7 +365,7 @@ pub fn installHooks() void { var installed: u32 = 0; // Bone transform SSE - if (transform_hook.attach(0x714260, &bone_sse.transformImpl_SSE) == .ok) installed += 1; + if (transform_hook.attach(0x714260, &bone_sse64.transformImpl_SSE64) == .ok) installed += 1; // Frustum clip SSE (1.9x speedup) if (clip_hook.attach(0x6318C0, &clip_sse.clipPolygonToSinglePlane) == .ok) installed += 1; @@ -306,19 +373,16 @@ pub fn installHooks() void { // Particle rendering SSE if (particle_hook.attach(0x7B2A50, &particleDetour) == .ok) installed += 1; - // Glyph cache - // Glyph cache removed -- game has internal glyph cache, our hook only sees misses (~30/frame) - - // GUID lookup cache -- A/B testing via transform44 - // if (findguid_hook.attach(0x464890, &findguidDetour) == .ok) installed += 1; - // if (obj_delete_hook.attach(0x464920, &objDeleteDetour) == .ok) installed += 1; - - // GUID cache disabled for now - // if (destroy_objmgr_hook.attach(0x467700, &destroyObjMgrDetour) == .ok) installed += 1; - // Per-frame cache reset if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1; + // Spell ground effect diagnostics + if (model_attach_hook.attach(0x707350, &modelAttachDetour) == .ok) installed += 1; + if (manage_rl_hook.attach(0x710B90, &manageRLDetour) == .ok) installed += 1; + + // Area effect ground model fix + if (proj_move_hook.attach(0x61e1d0, &projMoveDetour) == .ok) installed += 1; + // Silicon SSE binary patches _ = installPatches(); @@ -358,6 +422,9 @@ pub fn removeHooks() void { obj_delete_hook.detach(); destroy_objmgr_hook.detach(); world_update_hook.detach(); + model_attach_hook.detach(); + manage_rl_hook.detach(); + proj_move_hook.detach(); log.close(); mod_mutex.release(&g_mutex); }