weirdperformance: f64-intermediate bone transform to eliminate spell/doodad z-fighting

Adds bone_sse64.zig as an f64-intermediate port of transformMatrix4x4, used as
the active hook. M2 bone matrices are built and multiplied as [16]f64 and only
narrow to f32 on final store into the bone output buffer -- matching the x87
original's rounding profile (wide intermediates, single f32 store) and keeping
M2 vertex positions aligned with the terrain/projected-texture pipeline.

Also fixes, in both bone_sse (f32) and bone_sse64:

- Pre-billboard tx/ty/tz accumulation order (row 0 = pz+px+py; rows 1/2 = pz+py+px)
- Post-billboard pos_y/pos_z accumulation order (py+pz+px)
- Post-billboard scale-recompute accumulation order (row0 + row2 + row1)
- Billboard types 2/4 normalize using f64 intermediates (load-bearing for camera
  basis vectors -- pure f32 drifted from x87 by a ULP per axis and caused
  particle emitters to jitter on camera motion)

Additional bone_sse64-specific changes:

- Local attachmentRecursion64 that recurses into transformImpl_SSE64 instead of
  bone_sse.transformImpl_SSE, so attached child models stay on the f64 path
- child_padding (this+0x84) computed with f64 intermediates

bone_sse remains the reference f32 implementation; its struct fields, inline
helpers, and section-loop fns are now `pub` so bone_sse64 can share them
(types/interpolation helpers/post-loop loops). Artifact size is unchanged.

build.zig adds bench_bone_sse64 object; src/bench/main.zig runs the new variant
through the same warmup/timing harness and prints SSE vs SSE64 vs BASELINE
cycles plus a parity check.
This commit is contained in:
MarcelineVQ
2026-04-16 22:58:09 -07:00
parent 37c96eb577
commit f500147fc7
5 changed files with 1310 additions and 270 deletions
+18 -5
View File
@@ -19,23 +19,23 @@ const ModuleDesc = struct {
/// affect the module's DLL, minor for feature changes, major for breaking.
const module_list = [_]ModuleDesc{
.{ .name = "pngscreenshots", .version = "1.0.1", .desc = "Enable screenshot module", .src_dir = "screenshot" },
.{ .name = "interact", .version = "1.0", .desc = "Enable interact module", .addon_name = "Interact" },
.{ .name = "interact", .version = "1.1.0", .desc = "Enable interact module", .addon_name = "Interact" },
.{ .name = "outline", .version = "1.0", .desc = "Enable outline module", .default = false, .addon_name = "Outline" },
.{ .name = "worldmarkers", .version = "1.0", .desc = "Enable world markers module", .addon_name = "WorldMarkers", .addon_hidden = true },
.{ .name = "worldmarkers", .version = "1.1", .desc = "Enable world markers module", .addon_name = "WorldMarkers", .addon_hidden = true },
.{ .name = "framecrash", .version = "1.0", .desc = "Enable framecrash fix", .default = false },
.{ .name = "logsessions", .version = "1.0.1", .desc = "Enable log session rotation", .addon_name = "LogSessions" },
.{ .name = "logsessions", .version = "1.1.0", .desc = "Enable log session rotation", .addon_name = "LogSessions", .addon_hidden = true },
.{ .name = "minimapicons", .version = "1.0.1", .desc = "Enable custom minimap icons", .addon_name = "MinimapIcons" },
.{ .name = "transmogfix", .version = "1.0.1", .desc = "Enable transmog update coalescing" },
.{ .name = "customassets", .version = "1.0.1", .desc = "Enable loose file loading & permissive patch glob" },
.{ .name = "healtextfix", .version = "1.0.1", .desc = "Enable SuperWoW heal text fix" },
.{ .name = "bigcursor", .version = "1.0.1", .desc = "Enable big cursor module" },
.{ .name = "clickthrough", .version = "1.0.2", .desc = "Enable GO click-through" },
.{ .name = "clickthrough", .version = "1.0.3", .desc = "Enable GO click-through" },
.{ .name = "dpslog", .version = "0.1", .desc = "Enable structured combat log events for addons", .default = false },
.{ .name = "transform44", .version = "1.0", .desc = "Enable transform44 profiling/A/B testing (dev only)", .default = false },
.{ .name = "addonperf", .version = "1.0", .desc = "Enable addon memory/CPU profiling API", .default = false },
.{ .name = "ssemaths", .version = "1.0", .desc = "Enable UnitXP x87 math polyfill replacements (SSE)", .default = false },
.{ .name = "silicon", .version = "1.0", .desc = "Enable SSE2 math replacements (ported from libSiliconPatch)", .default = false },
.{ .name = "weirdperformance", .version = "1.1.1", .desc = "Enable production performance optimizations (SSE, inflate, filecache, timer, luastr, luavm)", .default = true },
.{ .name = "weirdperformance", .version = "1.2.1", .desc = "Enable production performance optimizations (SSE, inflate, filecache, timer, luastr, luavm)", .default = true },
.{ .name = "superweirdo", .version = "0.1", .desc = "Enable GO loot sparkle on interactable objects", .default = false },
.{ .name = "luagc", .version = "0.1", .desc = "Enable incremental Lua GC (replaces stop-the-world mark+sweep)", .default = false },
};
@@ -235,6 +235,18 @@ pub fn build(b: *std.Build) void {
.optimize = .ReleaseFast,
}),
});
const bench_bone_sse64 = b.addObject(.{
.name = "bench_bone_sse64",
.root_module = b.createModule(.{
.root_source_file = b.path("src/weirdperformance/bone_sse64.zig"),
.target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .linux,
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
}),
.optimize = .ReleaseFast,
}),
});
const bench_bone_baseline = b.addObject(.{
.name = "bench_bone_baseline",
.root_module = b.createModule(.{
@@ -270,6 +282,7 @@ pub fn build(b: *std.Build) void {
bench.root_module.addObject(bench_math_sse);
bench.root_module.addObject(bench_silicon_sse);
bench.root_module.addObject(bench_bone_sse);
bench.root_module.addObject(bench_bone_sse64);
bench.root_module.addObject(bench_bone_baseline);
bench.root_module.addObject(bench_particle_sse);
bench.root_module.addObject(bench_cull_sse);
+51
View File
@@ -1662,6 +1662,7 @@ pub fn main() void {
const ofs = [3]f32{ 0, 0, 0 };
const sb: u32 = @bitCast(@as(f32, 1.0));
const transformImpl_SSE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE" });
const transformImpl_SSE64 = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE64" });
const transformImpl_BASELINE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.c) void, .{ .name = "transformImpl_BASELINE" });
// Pre-set boneKeyframe init flag so we skip the atexit call (Windows CRT, can't run on Linux)
@@ -1716,6 +1717,20 @@ pub fn main() void {
const best_sse = run_bench_fn(transformImpl_SSE, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
const avg_sse = best_sse / T44_ITERS;
// Warmup + bench bone_sse64 (f64-intermediate variant)
for (0..500) |iter| {
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(iter * 2)), .little);
wu(u32, scene_obj[0x40..0x44], 0, .little);
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
}
for (0..500) |iter| {
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(999 - iter * 2)), .little);
wu(u32, scene_obj[0x40..0x44], 0, .little);
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
}
const best_sse64 = run_bench_fn(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
const avg_sse64 = best_sse64 / T44_ITERS;
print(" BASELINE: {d} cycles/call (frozen)\n", .{BASELINE_CYCLES});
print(" SSE: {d} cycles/call", .{avg_sse});
if (avg_sse < BASELINE_CYCLES) {
@@ -1727,6 +1742,25 @@ pub fn main() void {
} else {
print(" (same)\n", .{});
}
print(" SSE64: {d} cycles/call", .{avg_sse64});
if (avg_sse64 < BASELINE_CYCLES) {
const pct = (BASELINE_CYCLES - avg_sse64) * 100 / BASELINE_CYCLES;
print(" (-{d}% vs BASELINE", .{pct});
} else if (avg_sse64 > BASELINE_CYCLES) {
const pct = (avg_sse64 - BASELINE_CYCLES) * 100 / BASELINE_CYCLES;
print(" (+{d}% vs BASELINE", .{pct});
} else {
print(" (same as BASELINE", .{});
}
if (avg_sse64 > avg_sse) {
const pct = (avg_sse64 - avg_sse) * 100 / avg_sse;
print(", +{d}% vs SSE)\n", .{pct});
} else if (avg_sse64 < avg_sse) {
const pct = (avg_sse - avg_sse64) * 100 / avg_sse;
print(", -{d}% vs SSE)\n", .{pct});
} else {
print(", same as SSE)\n", .{});
}
// --- Output parity: run BASELINE then SSE with identical input, compare ALL outputs ---
{
@@ -1792,6 +1826,23 @@ pub fn main() void {
} else {
print(" parity: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs, total_len });
}
// Run SSE64 with same input
reset_and_run(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, &bone_rt, BONE_COUNT);
var diffs64: u32 = 0;
off = 0;
for (bufs) |b| {
for (0..b.len) |i| {
if (b.ptr[i] != snap[off + i]) diffs64 += 1;
}
off += b.len;
}
if (diffs64 == 0) {
print(" parity64: PASS (SSE64 == BASELINE, {d} bytes checked)\n", .{total_len});
} else {
print(" parity64: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs64, total_len });
}
}
}
+250 -254
View File
@@ -25,118 +25,118 @@ const V4 = @Vector(4, f32);
// SceneObject field offsets — assembly-verified from [EBX+N] in transformMatrix4x4
// =============================================================================
const SO = struct {
const model_data_ptr: u32 = 0x010;
const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value
const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header
const sync_value: u32 = 0x040;
const search_data_base: u32 = 0x04C; // prev timestamp for delta
const emitter_flag: u32 = 0x050;
const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array
const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS
const child_padding: u32 = 0x084;
const anim_frame_ctr: u32 = 0x08C;
const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs
const bone_out_ptr: u32 = 0x094; // output bone matrices
const tex_anim_out: u32 = 0x0A0;
const color_anim_out: u32 = 0x0A8;
const scale1: u32 = 0x0AC;
const scale2: u32 = 0x0B0;
const scale3: u32 = 0x0B4;
const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward)
const world_xform: u32 = 0x10C; // float[16] world transform
const field_17c: u32 = 0x17C;
const field_180: u32 = 0x180;
const field_184: u32 = 0x184;
const field_188: u32 = 0x188;
const field_18c: u32 = 0x18C;
const field_190: u32 = 0x190;
const render_scale_x: u32 = 0x194;
const render_scale_y: u32 = 0x198;
const render_scale_z: u32 = 0x19C;
const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children)
const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children)
const hierarchy_ptr: u32 = 0x1C8;
const emitter_ctx: u32 = 0x1CC;
const field_1d8: u32 = 0x1D8;
const hierarchy_idx: u32 = 0x1DC;
const field_200: u32 = 0x200;
const particle1: u32 = 0x3C4;
const particle2: u32 = 0x3C8;
const particle3: u32 = 0x3D0;
const particle4: u32 = 0x3D4;
const add_remaining: u32 = 0x3D8;
pub const SO = struct {
pub const model_data_ptr: u32 = 0x010;
pub const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value
pub const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header
pub const sync_value: u32 = 0x040;
pub const search_data_base: u32 = 0x04C; // prev timestamp for delta
pub const emitter_flag: u32 = 0x050;
pub const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array
pub const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS
pub const child_padding: u32 = 0x084;
pub const anim_frame_ctr: u32 = 0x08C;
pub const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs
pub const bone_out_ptr: u32 = 0x094; // output bone matrices
pub const tex_anim_out: u32 = 0x0A0;
pub const color_anim_out: u32 = 0x0A8;
pub const scale1: u32 = 0x0AC;
pub const scale2: u32 = 0x0B0;
pub const scale3: u32 = 0x0B4;
pub const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward)
pub const world_xform: u32 = 0x10C; // float[16] world transform
pub const field_17c: u32 = 0x17C;
pub const field_180: u32 = 0x180;
pub const field_184: u32 = 0x184;
pub const field_188: u32 = 0x188;
pub const field_18c: u32 = 0x18C;
pub const field_190: u32 = 0x190;
pub const render_scale_x: u32 = 0x194;
pub const render_scale_y: u32 = 0x198;
pub const render_scale_z: u32 = 0x19C;
pub const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children)
pub const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children)
pub const hierarchy_ptr: u32 = 0x1C8;
pub const emitter_ctx: u32 = 0x1CC;
pub const field_1d8: u32 = 0x1D8;
pub const hierarchy_idx: u32 = 0x1DC;
pub const field_200: u32 = 0x200;
pub const particle1: u32 = 0x3C4;
pub const particle2: u32 = 0x3C8;
pub const particle3: u32 = 0x3D0;
pub const particle4: u32 = 0x3D4;
pub const add_remaining: u32 = 0x3D8;
};
// Bone runtime struct offsets (within 0x118-byte per-bone runtime)
const BR = struct {
pub const BR = struct {
// Translation interpolation state
const trans_idx0: u32 = 0x00; // [0] lower keyframe index
const trans_idx1: u32 = 0x04; // [1] upper keyframe index
const trans_t: u32 = 0x08; // [2] interpolation factor (float bits)
const trans_x: u32 = 0x0C; // [3] interpolated translation X
const trans_y: u32 = 0x10; // [4] Y
const trans_z: u32 = 0x14; // [5] Z
pub const trans_idx0: u32 = 0x00; // [0] lower keyframe index
pub const trans_idx1: u32 = 0x04; // [1] upper keyframe index
pub const trans_t: u32 = 0x08; // [2] interpolation factor (float bits)
pub const trans_x: u32 = 0x0C; // [3] interpolated translation X
pub const trans_y: u32 = 0x10; // [4] Y
pub const trans_z: u32 = 0x14; // [5] Z
// Secondary translation (crossfade)
const trans2_idx0: u32 = 0x18;
const trans2_idx1: u32 = 0x1C;
const trans2_t: u32 = 0x20;
const trans2_x: u32 = 0x24;
const trans2_y: u32 = 0x28;
const trans2_z: u32 = 0x2C;
pub const trans2_idx0: u32 = 0x18;
pub const trans2_idx1: u32 = 0x1C;
pub const trans2_t: u32 = 0x20;
pub const trans2_x: u32 = 0x24;
pub const trans2_y: u32 = 0x28;
pub const trans2_z: u32 = 0x2C;
// Scale interpolation state (at puVar20 + 0x1a = offset 0x68)
const scale_idx0: u32 = 0x68;
const scale_idx1: u32 = 0x6C;
const scale_t: u32 = 0x70;
const scale_x: u32 = 0x74;
const scale_y: u32 = 0x78;
const scale_z: u32 = 0x7C;
const scale2_idx0: u32 = 0x80;
const scale2_idx1: u32 = 0x84;
const scale2_t: u32 = 0x88;
const scale2_x: u32 = 0x8C;
const scale2_y: u32 = 0x90;
const scale2_z: u32 = 0x94;
pub const scale_idx0: u32 = 0x68;
pub const scale_idx1: u32 = 0x6C;
pub const scale_t: u32 = 0x70;
pub const scale_x: u32 = 0x74;
pub const scale_y: u32 = 0x78;
pub const scale_z: u32 = 0x7C;
pub const scale2_idx0: u32 = 0x80;
pub const scale2_idx1: u32 = 0x84;
pub const scale2_t: u32 = 0x88;
pub const scale2_x: u32 = 0x8C;
pub const scale2_y: u32 = 0x90;
pub const scale2_z: u32 = 0x94;
// Primary animation time range
const prim_time: u32 = 0x98; // puVar20[0x26]
const prim_track: u32 = 0x9C; // puVar20[0x27]
const prim_anim: u32 = 0xA0; // puVar20[0x28]
const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index
pub const prim_time: u32 = 0x98; // puVar20[0x26]
pub const prim_track: u32 = 0x9C; // puVar20[0x27]
pub const prim_anim: u32 = 0xA0; // puVar20[0x28]
pub const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index
// Secondary animation time range (crossfade)
const sec_start: u32 = 0xA8; // puVar20[0x2a]
const sec_end: u32 = 0xAC; // puVar20[0x2b]
const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion
const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e]
pub const sec_start: u32 = 0xA8; // puVar20[0x2a]
pub const sec_end: u32 = 0xAC; // puVar20[0x2b]
pub const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion
pub const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e]
// Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30)
const rot_idx0: u32 = 0x30;
const rot_idx1: u32 = 0x34;
const rot_t: u32 = 0x38;
const rot_x: u32 = 0x3C;
const rot_y: u32 = 0x40;
const rot_z: u32 = 0x44;
const rot_w: u32 = 0x48;
pub const rot_idx0: u32 = 0x30;
pub const rot_idx1: u32 = 0x34;
pub const rot_t: u32 = 0x38;
pub const rot_x: u32 = 0x3C;
pub const rot_y: u32 = 0x40;
pub const rot_z: u32 = 0x44;
pub const rot_w: u32 = 0x48;
// Secondary rotation
const rot2_idx0: u32 = 0x4C;
const rot2_idx1: u32 = 0x50;
const rot2_t: u32 = 0x54;
const rot2_x: u32 = 0x58;
const rot2_y: u32 = 0x5C;
const rot2_z: u32 = 0x60;
const rot2_w: u32 = 0x64;
pub const rot2_idx0: u32 = 0x4C;
pub const rot2_idx1: u32 = 0x50;
pub const rot2_t: u32 = 0x54;
pub const rot2_x: u32 = 0x58;
pub const rot2_y: u32 = 0x5C;
pub const rot2_z: u32 = 0x60;
pub const rot2_w: u32 = 0x64;
// Secondary time range
const sec_time: u32 = 0xC4; // puVar20[0x31]
const sec_track: u32 = 0xC8; // puVar20[0x32]
const sec_slot: u32 = 0xD0; // puVar20[0x34]
const sec_start2: u32 = 0xD4; // puVar20[0x35]
const sec_end2: u32 = 0xD8; // puVar20[0x36]
const sec_offset2: u32 = 0xE4; // puVar20[0x39]
pub const sec_time: u32 = 0xC4; // puVar20[0x31]
pub const sec_track: u32 = 0xC8; // puVar20[0x32]
pub const sec_slot: u32 = 0xD0; // puVar20[0x34]
pub const sec_start2: u32 = 0xD4; // puVar20[0x35]
pub const sec_end2: u32 = 0xD8; // puVar20[0x36]
pub const sec_offset2: u32 = 0xE4; // puVar20[0x39]
// Flags and weights
const flags2: u32 = 0xF4; // puVar20[0x3d]
const crossfade_end: u32 = 0x100; // puVar20[0x40]
const crossfade_inv: u32 = 0x104; // puVar20[0x41]
const crossfade_weight: u32 = 0x108; // puVar20[0x42]
const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade
const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c]
pub const flags2: u32 = 0xF4; // puVar20[0x3d]
pub const crossfade_end: u32 = 0x100; // puVar20[0x40]
pub const crossfade_inv: u32 = 0x104; // puVar20[0x41]
pub const crossfade_weight: u32 = 0x108; // puVar20[0x42]
pub const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade
pub const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c]
};
// OldAnimationBlock struct offsets (28 bytes = 0x1C per track in v256 M2)
@@ -144,37 +144,37 @@ const BR = struct {
// pMVar23->m31 (bone_def+0x34) = rot block+0x0C = nTimestamps (gates rotation)
// pMVar23->m12 (bone_def+0x18) = trans block+0x0C = nTimestamps (gates translation)
// pMVar23[1].m10 (bone_def+0x50) = scale block+0x0C = nTimestamps (gates scale)
const AD = struct {
const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp)
const time_index: u32 = 0x02; // i16: global sequence index (-1 = none)
const track_count_flag: u32 = 0x04; // nRanges: 0 = single track
const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs
const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count
const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array
const nvalues: u32 = 0x14; // nValues: number of value entries
const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data
pub const AD = struct {
pub const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp)
pub const time_index: u32 = 0x02; // i16: global sequence index (-1 = none)
pub const track_count_flag: u32 = 0x04; // nRanges: 0 = single track
pub const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs
pub const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count
pub const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array
pub const nvalues: u32 = 0x14; // nValues: number of value entries
pub const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data
};
// M2CompBone struct offsets (0x6C = 108 bytes per bone in v256 model)
// Layout: 12 bytes fixed header + 3x28 byte OldAnimationBlock tracks + 12 bytes pivot
// Track order: translation, rotation, scale (standard M2 order)
const BD = struct {
const key_id: u32 = 0x00; // i32: key bone ID
const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.)
const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes
pub const BD = struct {
pub const key_id: u32 = 0x00; // i32: key bone ID
pub const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.)
pub const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes
// Translation OldAnimationBlock (28 bytes, +0x0C to +0x27)
const trans_anim: u32 = 0x0C;
const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation
pub const trans_anim: u32 = 0x0C;
pub const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation
// Rotation OldAnimationBlock (28 bytes, +0x28 to +0x43)
const rot_anim: u32 = 0x28;
const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation
pub const rot_anim: u32 = 0x28;
pub const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation
// Scale OldAnimationBlock (28 bytes, +0x44 to +0x5F)
const scale_anim: u32 = 0x44;
const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation
pub const scale_anim: u32 = 0x44;
pub const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation
// Pivot point (12 bytes, +0x60 to +0x6B)
const pivot_x: u32 = 0x60;
const pivot_y: u32 = 0x64;
const pivot_z: u32 = 0x68;
pub const pivot_x: u32 = 0x60;
pub const pivot_y: u32 = 0x64;
pub const pivot_z: u32 = 0x68;
};
// Game constants
@@ -182,12 +182,12 @@ const ZERO_F: f32 = 0.0;
const ONE_F: f32 = 1.0;
const THREE_F: f32 = 3.0;
// getBillboardEpsilon(): read from game memory (runtime 0x34800000, NOT static 0x3727c5ac from Ghidra)
fn getBillboardEpsilon() f32 {
pub fn getBillboardEpsilon() f32 {
return rf32(0x008029d4);
}
// getShortToFloat(): read from game memory at 0x00811610 (runtime value is 0x38000100 = 1/32767,
// NOT the static 0x38000000 = 1/32768 from Ghidra). The game patches this at startup.
fn getShortToFloat() f32 {
pub fn getShortToFloat() f32 {
return rf32(0x00811610);
}
// MSVC CRT sin/cos — linked from the WoW process
@@ -202,42 +202,42 @@ const OrigTransformFn = *const fn (u32, u32, u32, u32, u32) callconv(.c) void;
// Memory access helpers
// =============================================================================
inline fn ru32(addr: u32) u32 {
pub inline fn ru32(addr: u32) u32 {
return @as(*const u32, @ptrFromInt(addr)).*;
}
inline fn ri32(addr: u32) i32 {
pub inline fn ri32(addr: u32) i32 {
return @as(*const i32, @ptrFromInt(addr)).*;
}
inline fn rf32(addr: u32) f32 {
pub inline fn rf32(addr: u32) f32 {
return @as(*const f32, @ptrFromInt(addr)).*;
}
inline fn ru16(addr: u32) u16 {
pub inline fn ru16(addr: u32) u16 {
return @as(*align(1) const u16, @ptrFromInt(addr)).*;
}
inline fn ri16(addr: u32) i16 {
pub inline fn ri16(addr: u32) i16 {
return @as(*align(1) const i16, @ptrFromInt(addr)).*;
}
inline fn ru8(addr: u32) u8 {
pub inline fn ru8(addr: u32) u8 {
return @as(*const u8, @ptrFromInt(addr)).*;
}
inline fn wu32(addr: u32, v: u32) void {
pub inline fn wu32(addr: u32, v: u32) void {
@as(*u32, @ptrFromInt(addr)).* = v;
}
inline fn wf32(addr: u32, v: f32) void {
pub inline fn wf32(addr: u32, v: f32) void {
@as(*f32, @ptrFromInt(addr)).* = v;
}
inline fn wu16(addr: u32, v: u16) void {
pub inline fn wu16(addr: u32, v: u16) void {
@as(*align(1) u16, @ptrFromInt(addr)).* = v;
}
inline fn wu8(addr: u32, v: u8) void {
pub inline fn wu8(addr: u32, v: u8) void {
@as(*u8, @ptrFromInt(addr)).* = v;
}
inline fn fbits(v: f32) u32 {
pub inline fn fbits(v: f32) u32 {
return @bitCast(v);
}
inline fn ufloat(v: u32) f32 {
pub inline fn ufloat(v: u32) f32 {
return @bitCast(v);
}
@@ -245,12 +245,12 @@ inline fn ufloat(v: u32) f32 {
// Math helpers — using @Vector(4, f32) for SSE
// =============================================================================
inline fn splat(v: f32) V4 {
pub inline fn splat(v: f32) V4 {
return @splat(v);
}
/// 3-component lerp: a + (b - a) * t. Uses @mulAdd → vfmadd.
inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
pub inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
return .{
@mulAdd(f32, rf32(b_addr) - rf32(a_addr), t, rf32(a_addr)),
@mulAdd(f32, rf32(b_addr + 4) - rf32(a_addr + 4), t, rf32(a_addr + 4)),
@@ -260,7 +260,7 @@ inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
/// Scale 3x3 rotation portion of a row-major 4x4 matrix by per-axis scale.
/// Row 0 *= scale.x, Row 1 *= scale.y, Row 2 *= scale.z
inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
pub inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
// Row 0 (offsets 0x00, 0x04, 0x08)
wf32(mat + 0x00, rf32(mat + 0x00) * sx);
wf32(mat + 0x04, rf32(mat + 0x04) * sx);
@@ -279,15 +279,15 @@ inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
/// mat[3][0] += dot(mat[0], t)
/// mat[3][1] += dot(mat[1], t)
/// mat[3][2] += dot(mat[2], t)
/// Uses @mulAdd chain → vfmadd for each dot product component.
inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void {
/// Uses @mulAdd chain for each dot product component.
pub inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void {
wf32(mat + 0x30, @mulAdd(f32, tz, rf32(mat + 0x20), @mulAdd(f32, ty, rf32(mat + 0x10), @mulAdd(f32, tx, rf32(mat + 0x00), rf32(mat + 0x30)))));
wf32(mat + 0x34, @mulAdd(f32, tz, rf32(mat + 0x24), @mulAdd(f32, ty, rf32(mat + 0x14), @mulAdd(f32, tx, rf32(mat + 0x04), rf32(mat + 0x34)))));
wf32(mat + 0x38, @mulAdd(f32, tz, rf32(mat + 0x28), @mulAdd(f32, ty, rf32(mat + 0x18), @mulAdd(f32, tx, rf32(mat + 0x08), rf32(mat + 0x38)))));
}
/// Quaternion → rotation matrix as value. No memory writes.
inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 {
pub inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 {
const xx2 = qx * (qx + qx);
const xy2 = qx * (qy + qy);
const xz2 = qx * (qz + qz);
@@ -308,7 +308,7 @@ inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 {
/// Quaternion → rotation matrix: writes to game memory via u32 address.
/// Used by boneKeyframeLoop where the matrix is in game memory.
inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
pub inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
const m = buildRotationMatrixVal(qx, qy, qz, qw);
inline for (0..16) |i| {
wf32(mat + @as(u32, @intCast(i * 4)), m[i]);
@@ -316,7 +316,7 @@ inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void
}
/// Quaternion → rotation matrix × mat. Fused: builds quat rows as V4, multiplies in-register.
inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
pub inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
const xx2 = qx * (qx + qx);
const xy2 = qx * (qy + qy);
const xz2 = qx * (qz + qz);
@@ -354,7 +354,7 @@ inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void
}
/// Copy 4x4 matrix (64 bytes) — 4 V4 loads/stores instead of 16 scalar copies.
inline fn copyMat4(dst: u32, src: u32) void {
pub inline fn copyMat4(dst: u32, src: u32) void {
inline for (0..4) |i| {
const off: u32 = @intCast(i * 16);
const row = V4{ rf32(src + off), rf32(src + off + 4), rf32(src + off + 8), rf32(src + off + 12) };
@@ -367,19 +367,16 @@ inline fn copyMat4(dst: u32, src: u32) void {
/// 4x4 matrix multiply: dst = a * b (row-major). Safe for dst==a or dst==b.
/// Uses V4 + @mulAdd (FMA): 1 mul + 3 FMA per row = 16 SIMD ops total.
inline fn matMul4x4(dst: u32, a: u32, b: u32) void {
// Pre-load all rows of B
pub inline fn matMul4x4(dst: u32, a: u32, b: u32) void {
const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) };
const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) };
const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) };
const b3 = V4{ rf32(b + 48), rf32(b + 52), rf32(b + 56), rf32(b + 60) };
// Pre-load all rows of A (in case dst aliases a)
const a0 = V4{ rf32(a), rf32(a + 4), rf32(a + 8), rf32(a + 12) };
const a1 = V4{ rf32(a + 16), rf32(a + 20), rf32(a + 24), rf32(a + 28) };
const a2 = V4{ rf32(a + 32), rf32(a + 36), rf32(a + 40), rf32(a + 44) };
const a3 = V4{ rf32(a + 48), rf32(a + 52), rf32(a + 56), rf32(a + 60) };
const rows = [4]V4{ a0, a1, a2, a3 };
// Compute: each output row = broadcast(a[row][col]) * b_row, accumulated with FMA
inline for (0..4) |i| {
const s0: V4 = @splat(rows[i][0]);
const s1: V4 = @splat(rows[i][1]);
@@ -395,7 +392,7 @@ inline fn matMul4x4(dst: u32, a: u32, b: u32) void {
}
/// 4x4 matrix multiply: dst = a * b. Left operand is a local array, right is game memory.
inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void {
pub inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void {
const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) };
const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) };
const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) };
@@ -421,7 +418,7 @@ inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void {
}
/// In-place multiply: a = a * b (b from game memory). Returns new array.
inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 {
pub inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 {
const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) };
const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) };
const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) };
@@ -448,7 +445,7 @@ inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 {
}
/// Set identity matrix (16 floats)
inline fn setIdentity(dst: u32) void {
pub inline fn setIdentity(dst: u32) void {
inline for (0..16) |i| {
const val: f32 = if (i == 0 or i == 5 or i == 10 or i == 15) 1.0 else 0.0;
wf32(dst + @as(u32, @intCast(i)) * 4, val);
@@ -458,7 +455,7 @@ inline fn setIdentity(dst: u32) void {
/// Normalize a 3-component vector in memory at addr.
/// Calls game's vec3 squared magnitude (0x4549F0), then sqrt, epsilon check, divide.
/// Assembly pattern: CALL 0x4549F0 → FSQRT → FABS → FCOMP → FLD1 → FDIVRP → FMUL×3
inline fn normalizeVec3InPlace(addr: u32) void {
pub inline fn normalizeVec3InPlace(addr: u32) void {
const sq_mag = callVec3SqMag(addr);
const len = @sqrt(sq_mag);
if (@abs(len) >= getBillboardEpsilon()) {
@@ -471,7 +468,7 @@ inline fn normalizeVec3InPlace(addr: u32) void {
/// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small.
/// Writes vec3 to stack local and calls game's vec3 squared magnitude (0x4549F0).
inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
pub inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
var v: [3]f32 = .{ x, y, z };
const sq_mag = callVec3SqMag(@intFromPtr(&v));
const len = @sqrt(sq_mag);
@@ -481,7 +478,7 @@ inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
}
/// Cross product of two 3-component vectors
inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 {
pub inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 {
return .{
ay * bz - az * by,
az * bx - ax * bz,
@@ -502,7 +499,7 @@ inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32
/// IsParticleBufferEmpty (0x7B5F60) — recursive tree check.
/// Returns true if any node in the tree has active particles (this->0x64 != 0).
fn isParticleBufferNotEmpty(ptr: u32) bool {
pub fn isParticleBufferNotEmpty(ptr: u32) bool {
if (ru32(ptr + 0x64) != 0) return true;
const count = ru32(ptr + 0x7C);
if (count == 0) return false;
@@ -514,7 +511,7 @@ fn isParticleBufferNotEmpty(ptr: u32) bool {
return false;
}
const InterpResult = struct {
pub const InterpResult = struct {
idx0: u32,
idx1: u32,
t: f32,
@@ -523,7 +520,7 @@ const InterpResult = struct {
/// Check if two AnimData tracks share the same temporal structure,
/// meaning findInterpIdx would produce identical (idx0, idx1, t) for both.
/// Both must use prim_time (time_index == -1) and have matching range/timestamp layout.
inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
pub inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
if (ri16(ref_anim + AD.time_index) != -1) return false;
if (ri16(other_anim + AD.time_index) != -1) return false;
return ru32(ref_anim + AD.track_count_flag) == ru32(other_anim + AD.track_count_flag) and
@@ -533,7 +530,7 @@ inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
}
/// Write temporal coherence cache for a reused result so next frame's forward scan starts right.
inline fn applyCachedResult(cached: InterpResult, output: u32) void {
pub inline fn applyCachedResult(cached: InterpResult, output: u32) void {
wu32(output, cached.idx0);
}
@@ -541,7 +538,7 @@ inline fn applyCachedResult(cached: InterpResult, output: u32) void {
/// Reimplementation of game function at 0x713D50 (334 bytes).
/// Assembly-verified against t44_helpers_asm.txt.
/// Returns indices and t in registers; only writes output[0] for next-frame cache persistence.
inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult {
pub inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult {
const n_ranges = ru32(anim_data + AD.track_count_flag);
// Range selection: [start, last] not [start, count]
@@ -659,13 +656,13 @@ inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_dat
/// Quaternion keyframe interpolation — replaces game's 0x713EA0.
/// Assembly-verified: stride 16 (SHL EAX,4), values are 4×float, not CompQuat.
inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 {
pub inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 {
return interpAnimKFCached(this, bone_rt, anim_data, output, null);
}
/// Quaternion keyframe interpolation with optional cached primary InterpResult.
/// When cached_primary is non-null, skips findInterpIdx and uses the cached indices/t.
inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 {
pub inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 {
const r = if (cached_primary) |c| blk: {
applyCachedResult(c, output);
break :blk c;
@@ -716,7 +713,7 @@ inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u3
/// Fast modulo for looping animations. The value is almost always < 2*length
/// (frame-to-frame delta is small), so a conditional subtract beats idiv.
inline fn fastMod(val: u32, len: u32) u32 {
pub inline fn fastMod(val: u32, len: u32) u32 {
var v = val;
if (v >= len) {
v -%= len;
@@ -727,12 +724,12 @@ inline fn fastMod(val: u32, len: u32) u32 {
/// Float truncation — replaces game's __ftol at 0x40A2B0.
/// Original: FILD i32 → FMUL f32 → __ftol, all in 80-bit x87 precision.
inline fn callFtol(delta: i32, scale_addr: u32) i32 {
pub inline fn callFtol(delta: i32, scale_addr: u32) i32 {
return @intFromFloat(@as(f32, @floatFromInt(delta)) * rf32(scale_addr));
}
/// Vec3 squared magnitude — replaces game's 0x4549F0. Uses @mulAdd → vfmadd.
inline fn callVec3SqMag(vec3_ptr: u32) f32 {
pub inline fn callVec3SqMag(vec3_ptr: u32) f32 {
const x = rf32(vec3_ptr);
const y = rf32(vec3_ptr + 4);
const z = rf32(vec3_ptr + 8);
@@ -741,7 +738,7 @@ inline fn callVec3SqMag(vec3_ptr: u32) f32 {
/// Read i16 at keyframe index. Replaces game's getIndexOffset (0x71AFF0) + setShortValue (0x71B010).
/// getIndexOffset returns table[4] + index*2, setShortValue copies a word. Direct read is equivalent.
inline fn readShortViaGame(table: u32, index: u32) i16 {
pub inline fn readShortViaGame(table: u32, index: u32) i16 {
const values_ptr = ru32(table + 4);
return ri16(values_ptr + index * 2);
}
@@ -749,7 +746,7 @@ inline fn readShortViaGame(table: u32, index: u32) i16 {
/// Interpolate a Vec3 track (12 bytes per keyframe) with crossfade support.
/// Writes result to output[3..5] (as u32 float bits). Uses output[0..2] for indices/t,
/// and output[6..11] for secondary crossfade state.
inline fn interpVec3Track(
pub inline fn interpVec3Track(
this: u32,
bone_rt: u32,
anim_data: u32,
@@ -760,7 +757,7 @@ inline fn interpVec3Track(
}
/// Vec3 keyframe interpolation with optional cached primary InterpResult.
inline fn interpVec3TrackCached(
pub inline fn interpVec3TrackCached(
this: u32,
bone_rt: u32,
anim_data: u32,
@@ -806,7 +803,7 @@ inline fn interpVec3TrackCached(
/// Interpolate a single float track (4 bytes per keyframe) with crossfade.
/// Writes result to output[3] as float bits.
inline fn interpFloatTrack(
pub inline fn interpFloatTrack(
this: u32,
bone_rt: u32,
anim_data: u32,
@@ -846,7 +843,7 @@ inline fn interpFloatTrack(
// Hermite/Bezier basis + particle emitter interp helpers
// =============================================================================
inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
pub inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
const t2 = t * t;
const t3 = t2 * t;
return .{
@@ -857,7 +854,7 @@ inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
};
}
inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
pub inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
const u = 1.0 - t;
const t2 = t * t;
const u_sq = u * u;
@@ -869,7 +866,7 @@ inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
};
}
inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
pub inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output);
const mode = ri16(anim_data + AD.interp_mode);
@@ -950,7 +947,7 @@ inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output
}
}
inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
pub inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output);
const mode = ri16(anim_data + AD.interp_mode);
@@ -1007,7 +1004,7 @@ inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, outpu
// Same as interpFloatTrack but uses the bone_rt directly (different register mapping)
// =============================================================================
inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void {
pub inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void {
const r = findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data_short_ptr, output);
const interp_mode = ri16(anim_data_short_ptr);
@@ -1040,7 +1037,7 @@ inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr
// applies inverse translation.
// =============================================================================
fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
pub fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
// Simple transpose for unit scale
if (@abs(scale - 1.0) < @as(f32, @bitCast(@as(u32, 0x35800000)))) {
// Transpose 3x3
@@ -1472,68 +1469,60 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
const pivot_y = rf32(bdef + BD.pivot_y);
const pivot_z = rf32(bdef + BD.pivot_z);
// Compute translated position
const tx = local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z + local_mat[12];
const ty = local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z + local_mat[13];
const tz = local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z + local_mat[14];
// Compute translated position — accumulation order must match
// original x87. Row 0 uses (pz + px + py), rows 1/2 use (pz + py + px).
const tx = local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[12];
const ty = local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x + local_mat[13];
const tz = local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x + local_mat[14];
const bb_type = combined_flags & 6;
const billboard_eps_f64: f64 = @floatCast(rf32(0x008029d4));
const cull_eps_f64: f64 = @floatCast(rf32(0x0080c5c8));
if (bb_type == 2) {
// Cylindrical billboard — normalize each column
const n0 = normalizeVec3(local_mat[0], local_mat[1], local_mat[2]);
local_mat[0] = n0[0];
local_mat[1] = n0[1];
local_mat[2] = n0[2];
const n1 = normalizeVec3(local_mat[4], local_mat[5], local_mat[6]);
local_mat[4] = n1[0];
local_mat[5] = n1[1];
local_mat[6] = n1[2];
const n2 = normalizeVec3(local_mat[8], local_mat[9], local_mat[10]);
local_mat[8] = n2[0];
local_mat[9] = n2[1];
local_mat[10] = n2[2];
// Cylindrical billboard — normalize each column in f64 to
// match x87's extended-precision 1/sqrt. Previous f32 impl
// drifted from x87 by a ULP per axis, causing particle
// emitter orientation to jitter on camera motion and
// flicker against ground effects.
inline for ([_]u32{ 0, 4, 8 }) |row_off| {
const cx: f64 = @floatCast(local_mat[row_off]);
const cy: f64 = @floatCast(local_mat[row_off + 1]);
const cz: f64 = @floatCast(local_mat[row_off + 2]);
const len_sq = cx * cx + cy * cy + cz * cz;
const len = @sqrt(len_sq);
if (len >= billboard_eps_f64) {
const inv = 1.0 / len;
local_mat[row_off] = @floatCast(cx * inv);
local_mat[row_off + 1] = @floatCast(cy * inv);
local_mat[row_off + 2] = @floatCast(cz * inv);
}
}
} else if (bb_type == 4) {
// Spherical billboard — inherit camera rotation with scale preservation
// All sqmag computations MUST call game's vec3SqMag (0x4549F0)
const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) };
const cam_len_sq0 = callVec3SqMag(this + SO.bb_row0);
var s0: f32 = 1.0;
if (cam_len_sq0 > rf32(0x0080c5c8)) {
var tmp0 = [3]f32{ local_mat[0], local_mat[1], local_mat[2] };
const mat_len_sq0 = callVec3SqMag(@intFromPtr(&tmp0));
s0 = @sqrt(mat_len_sq0 / cam_len_sq0);
// Spherical billboard — inherit camera basis, rescale to
// preserve each column's length. All intermediates in f64
// to match x87's 80-bit temporaries.
inline for ([_]struct { row_off: u32, src_off: u32 }{
.{ .row_off = 0, .src_off = SO.bb_row0 },
.{ .row_off = 4, .src_off = SO.world_xform },
.{ .row_off = 8, .src_off = SO.world_xform + 16 },
}) |p| {
const src_addr = this + p.src_off;
const cam_x: f64 = @floatCast(rf32(src_addr));
const cam_y: f64 = @floatCast(rf32(src_addr + 4));
const cam_z: f64 = @floatCast(rf32(src_addr + 8));
const cam_len_sq = cam_x * cam_x + cam_y * cam_y + cam_z * cam_z;
var s: f64 = 1.0;
if (cam_len_sq > cull_eps_f64) {
const mx: f64 = @floatCast(local_mat[p.row_off]);
const my: f64 = @floatCast(local_mat[p.row_off + 1]);
const mz: f64 = @floatCast(local_mat[p.row_off + 2]);
const mat_len_sq = mx * mx + my * my + mz * mz;
s = @sqrt(mat_len_sq / cam_len_sq);
}
local_mat[p.row_off] = @floatCast(s * cam_x);
local_mat[p.row_off + 1] = @floatCast(s * cam_y);
local_mat[p.row_off + 2] = @floatCast(s * cam_z);
}
local_mat[0] = s0 * cam0[0];
local_mat[1] = s0 * cam0[1];
local_mat[2] = s0 * cam0[2];
const wt0 = rf32(this + SO.world_xform + 0 * 4);
const wt1 = rf32(this + SO.world_xform + 1 * 4);
const wt2 = rf32(this + SO.world_xform + 2 * 4);
const wt_len_sq = callVec3SqMag(this + SO.world_xform);
var s1: f32 = 1.0;
if (wt_len_sq > rf32(0x0080c5c8)) {
var tmp1 = [3]f32{ local_mat[4], local_mat[5], local_mat[6] };
const mat_len_sq1 = callVec3SqMag(@intFromPtr(&tmp1));
s1 = @sqrt(mat_len_sq1 / wt_len_sq);
}
local_mat[4] = s1 * wt0;
local_mat[5] = s1 * wt1;
local_mat[6] = s1 * wt2;
const wt4 = rf32(this + SO.world_xform + 4 * 4);
const wt5 = rf32(this + SO.world_xform + 5 * 4);
const wt6 = rf32(this + SO.world_xform + 6 * 4);
const wt_len_sq2 = callVec3SqMag(this + SO.world_xform + 16);
var s2: f32 = 1.0;
if (wt_len_sq2 > rf32(0x0080c5c8)) {
var tmp2 = [3]f32{ local_mat[8], local_mat[9], local_mat[10] };
const mat_len_sq2 = callVec3SqMag(@intFromPtr(&tmp2));
s2 = @sqrt(mat_len_sq2 / wt_len_sq2);
}
local_mat[8] = s2 * wt4;
local_mat[9] = s2 * wt5;
local_mat[10] = s2 * wt6;
} else if (bb_type == 6) {
// Full billboard — copy camera rotation directly
local_mat[0] = rf32(this + SO.bb_row0);
@@ -1547,11 +1536,12 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
local_mat[10] = rf32(this + SO.world_xform + 6 * 4);
}
// Recompute translation: pos - rot * pivot
// Recompute translation: pos - rot * pivot.
// Accumulation order mirrors the tx/ty/tz computation above.
if ((combined_flags & 1) == 0) {
local_mat[12] = tx - (local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z);
local_mat[13] = ty - (local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z);
local_mat[14] = tz - (local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z);
local_mat[12] = tx - (local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y);
local_mat[13] = ty - (local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x);
local_mat[14] = tz - (local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x);
} else {
local_mat[12] = rf32(this + SO.world_xform + 8 * 4);
local_mat[13] = rf32(this + SO.world_xform + 9 * 4);
@@ -1671,9 +1661,13 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
const bpx = rf32(bdef + BD.pivot_x);
const bpy = rf32(bdef + BD.pivot_y);
const bpz = rf32(bdef + BD.pivot_z);
// Accumulation order mirrors original x87:
// pos_x: px + py + pz + const
// pos_y: py + pz + px + const
// pos_z: py + pz + px + const
const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30);
const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34);
const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38);
const pos_y = bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + bpx * rf32(om + 0x04) + rf32(om + 0x34);
const pos_z = bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + bpx * rf32(om + 0x08) + rf32(om + 0x38);
// Switch on billboard post-processing type
const bb_post = combined_flags & 0x78;
@@ -1798,10 +1792,14 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
const r2z_s = rf32(om + 0x28);
wf32(om + 0x28, scale_len2 * r2z_s);
// Recompute translation: pos - scaled_matrix * pivot
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz));
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz));
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz));
// Recompute translation: pos - scaled_matrix * pivot.
// Accumulation order must match original x87: row0 + row2 + row1
// (pivot_x, then pivot_z, then pivot_y). f32 addition isn't associative —
// this ordering matters for matching terrain-pipeline precision and
// avoiding z-fighting on ground-aligned billboard spell effects.
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len2 * r2x_s * bpz + scale_len1 * r1x_s * bpy));
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len2 * r2y_s * bpz + scale_len1 * r1y_s * bpy));
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len2 * r2z_s * bpz + scale_len1 * r1z_s * bpy));
wf32(om + 0x3C, 1.0);
}
}
@@ -1818,8 +1816,6 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
// findInterpIdx + lerp + crossfade blend.
// =========================================================================
// BISECT: stop after section 7 (bone loop)
// Section 8: Texture animation loop
texAnimLoop(this, model_hdr, frame_ctr);
colorAnimLoop(this, model_hdr, frame_ctr);
@@ -1841,7 +1837,7 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
// Post-bone-loop sections (extracted for readability)
// =============================================================================
fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
const count = ru32(model_hdr + 0x54);
if (count == 0) return;
@@ -1891,7 +1887,7 @@ fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
/// Short-value interpolation: uses InterpResult indices, looks up short values, interpolates.
/// Shared by texAnimLoop alpha, colorAnimLoop, and word animation crossfade.
inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 {
pub inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 {
const mode = ri16(anim_data);
const table = anim_data + AD.nvalues;
if (mode == 0) {
@@ -1903,7 +1899,7 @@ inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 {
}
}
fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
// Assembly: model_hdr+0x64 is both entry gate AND loop count
const count = ru32(model_hdr + 0x64);
@@ -1946,7 +1942,7 @@ fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
}
}
fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
// Assembly 0x715E46-0x715F25: word/byte animation section
// model_hdr+0x6C = count, model_hdr+0x70 = data base
@@ -1989,7 +1985,7 @@ fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
}
}
fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
pub fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
const count = ru32(model_hdr + 0x74);
if (count == 0) return;
@@ -2050,7 +2046,7 @@ fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
}
}
fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
// Particle emitters are the largest section (~1000 lines of decompiled C).
// They follow the same interpolation patterns but with many sub-tracks per emitter.
// For the initial implementation, we handle the key tracks (position, speed, scale).
@@ -2066,7 +2062,7 @@ fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
additionalParticleLoops(this, model_hdr, frame_ctr);
}
fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
const count = ru32(model_hdr + 0x11C);
if (count == 0) return;
@@ -2136,7 +2132,7 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
}
}
fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
const count = ru32(model_hdr + 0x124);
if (count == 0) return;
@@ -2169,7 +2165,7 @@ fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
}
}
fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
pub fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
const stf = getShortToFloat();
// Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A)
@@ -2377,7 +2373,7 @@ fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
}
}
fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void {
pub fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
const hierarchy = ru32(this + SO.hierarchy_ptr);
if (hierarchy == 0) return;
+913
View File
@@ -0,0 +1,913 @@
//! f64-intermediate port of transformMatrix4x4 (0x714260).
//!
//! Mirrors bone_sse.zig but holds intermediate bone matrices as [16]f64 and
//! runs all matrix-math operations in double precision, only narrowing to f32
//! when writing into the bone output buffer in game memory.
//!
//! Rounding profile matches the original x87 implementation: load f32, compute
//! in extended precision (here 53-bit mantissa f64 vs x87 64-bit mantissa —
//! indistinguishable once truncated to f32), store f32 once at the end.
//!
//! Interpolation, keyframe search, billboard math, ftol, and game-callback
//! helpers are imported from bone_sse since they produce f32 scalars that
//! widen implicitly when multiplied with f64 matrices.
//!
//! Compiled ReleaseFast with AVX enabled. Uses @Vector(4, f64) for SIMD
//! matrix multiplies on ymm registers.
const bone_sse = @import("bone_sse.zig");
const V4d = @Vector(4, f64);
// =============================================================================
// Imports from bone_sse — constants, memory helpers, interp/billboard helpers
// =============================================================================
const SO = bone_sse.SO;
const BR = bone_sse.BR;
const BD = bone_sse.BD;
const AD = bone_sse.AD;
const InterpResult = bone_sse.InterpResult;
const ru32 = bone_sse.ru32;
const ri32 = bone_sse.ri32;
const rf32 = bone_sse.rf32;
const ru16 = bone_sse.ru16;
const ri16 = bone_sse.ri16;
const ru8 = bone_sse.ru8;
const wu32 = bone_sse.wu32;
const wf32 = bone_sse.wf32;
const wu16 = bone_sse.wu16;
const wu8 = bone_sse.wu8;
const fbits = bone_sse.fbits;
const ufloat = bone_sse.ufloat;
const normalizeVec3 = bone_sse.normalizeVec3;
const normalizeVec3InPlace = bone_sse.normalizeVec3InPlace;
const crossVec3 = bone_sse.crossVec3;
const canReuseInterp = bone_sse.canReuseInterp;
const findInterpIdx = bone_sse.findInterpIdx;
const interpAnimKFCached = bone_sse.interpAnimKFCached;
const interpVec3TrackCached = bone_sse.interpVec3TrackCached;
const callFtol = bone_sse.callFtol;
const callVec3SqMag = bone_sse.callVec3SqMag;
const fastMod = bone_sse.fastMod;
const texAnimLoop = bone_sse.texAnimLoop;
const colorAnimLoop = bone_sse.colorAnimLoop;
const wordAnimLoop = bone_sse.wordAnimLoop;
const boneKeyframeLoop = bone_sse.boneKeyframeLoop;
const particleLoops = bone_sse.particleLoops;
// NOTE: we do NOT import bone_sse.attachmentRecursion — that version recurses
// into bone_sse.transformImpl_SSE (f32) directly, which would force attached
// child models onto the f32 path while the parent is f64. We reimplement it
// below so attachment recursion stays inside bone_sse64's f64 pipeline.
// =============================================================================
// f64 memory helpers
// =============================================================================
inline fn rf64(addr: u32) f64 {
return @floatCast(rf32(addr));
}
inline fn wf32_narrow(addr: u32, v: f64) void {
wf32(addr, @floatCast(v));
}
inline fn loadV4d(addr: u32) V4d {
return V4d{ rf64(addr), rf64(addr + 4), rf64(addr + 8), rf64(addr + 12) };
}
inline fn storeV4d(addr: u32, v: V4d) void {
wf32(addr + 0, @floatCast(v[0]));
wf32(addr + 4, @floatCast(v[1]));
wf32(addr + 8, @floatCast(v[2]));
wf32(addr + 12, @floatCast(v[3]));
}
// =============================================================================
// f64 matrix operations
// =============================================================================
/// 4x4 matrix multiply: dst = a * b. Memory-to-memory with f64 intermediates.
inline fn matMul4x4_64(dst: u32, a: u32, b: u32) void {
const b0 = loadV4d(b);
const b1 = loadV4d(b + 16);
const b2 = loadV4d(b + 32);
const b3 = loadV4d(b + 48);
const a0 = loadV4d(a);
const a1 = loadV4d(a + 16);
const a2 = loadV4d(a + 32);
const a3 = loadV4d(a + 48);
const rows = [4]V4d{ a0, a1, a2, a3 };
inline for (0..4) |i| {
const s0: V4d = @splat(rows[i][0]);
const s1: V4d = @splat(rows[i][1]);
const s2: V4d = @splat(rows[i][2]);
const s3: V4d = @splat(rows[i][3]);
const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3;
const off: u32 = @intCast(i * 16);
storeV4d(dst + off, row);
}
}
/// 4x4 multiply: dst_mem = a_local * b_mem. Left operand is [16]f64 local.
inline fn matMul4x4Local_64(dst: u32, a: [16]f64, b: u32) void {
const b0 = loadV4d(b);
const b1 = loadV4d(b + 16);
const b2 = loadV4d(b + 32);
const b3 = loadV4d(b + 48);
const rows = [4]V4d{
V4d{ a[0], a[1], a[2], a[3] },
V4d{ a[4], a[5], a[6], a[7] },
V4d{ a[8], a[9], a[10], a[11] },
V4d{ a[12], a[13], a[14], a[15] },
};
inline for (0..4) |i| {
const s0: V4d = @splat(rows[i][0]);
const s1: V4d = @splat(rows[i][1]);
const s2: V4d = @splat(rows[i][2]);
const s3: V4d = @splat(rows[i][3]);
const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3;
const off: u32 = @intCast(i * 16);
storeV4d(dst + off, row);
}
}
/// In-place f64 multiply: a = a * b_mem. Returns new [16]f64.
inline fn matMul4x4InPlace_64(a: [16]f64, b: u32) [16]f64 {
const b0 = loadV4d(b);
const b1 = loadV4d(b + 16);
const b2 = loadV4d(b + 32);
const b3 = loadV4d(b + 48);
const rows = [4]V4d{
V4d{ a[0], a[1], a[2], a[3] },
V4d{ a[4], a[5], a[6], a[7] },
V4d{ a[8], a[9], a[10], a[11] },
V4d{ a[12], a[13], a[14], a[15] },
};
var result: [16]f64 = undefined;
inline for (0..4) |i| {
const s0: V4d = @splat(rows[i][0]);
const s1: V4d = @splat(rows[i][1]);
const s2: V4d = @splat(rows[i][2]);
const s3: V4d = @splat(rows[i][3]);
const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3;
result[i * 4 + 0] = row[0];
result[i * 4 + 1] = row[1];
result[i * 4 + 2] = row[2];
result[i * 4 + 3] = row[3];
}
return result;
}
/// Quaternion → 4x4 rotation matrix as f64 local. Identity last row/col.
inline fn buildRotationMatrix_64(qx: f64, qy: f64, qz: f64, qw: f64) [16]f64 {
const xx2 = qx * (qx + qx);
const xy2 = qx * (qy + qy);
const xz2 = qx * (qz + qz);
const yy2 = qy * (qy + qy);
const yz2 = qy * (qz + qz);
const zz2 = qz * (qz + qz);
const wx2 = qw * (qx + qx);
const wy2 = qw * (qy + qy);
const wz2 = qw * (qz + qz);
return [16]f64{
1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0.0,
xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0.0,
xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0.0,
0.0, 0.0, 0.0, 1.0,
};
}
/// Copy a 4x4 matrix in game memory (bit-exact, no precision change).
inline fn copyMat4(dst: u32, src: u32) void {
inline for (0..8) |i| {
const off = @as(u32, @intCast(i)) * 8;
@as(*u64, @ptrFromInt(dst + off)).* = @as(*const u64, @ptrFromInt(src + off)).*;
}
}
// =============================================================================
// Attachment recursion — f64 version that re-enters transformImpl_SSE64 for
// child scene objects. Mirrors bone_sse.attachmentRecursion exactly except
// it calls our f64 implementation for child transforms.
// =============================================================================
fn attachmentRecursion64(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void {
@setEvalBranchQuota(50000);
const hierarchy = ru32(this + SO.hierarchy_ptr);
if (hierarchy == 0) return;
const attach_count = ru32(model_hdr + 0x104);
const attach_data = ru32(model_hdr + 0x108);
var att_i: u32 = 0;
var att_off: u32 = 0;
while (att_i < attach_count) : ({
att_i += 1;
att_off += 0x30;
}) {
const att_entry = attach_data + att_off;
if (frame_ctr < ru32(att_entry + 0x20)) {
const bone_idx = @as(u32, ru16(att_entry + 4));
const bone_rt = ru32(this + SO.bone_rt_base) + bone_idx * 0x118;
const anim_data = att_entry + 0x14;
const att_output = hierarchy + att_i * 0x20;
const atr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), anim_data, att_output);
wu8(att_output + 0x0C, ru8(ru32(anim_data + AD.keyframe_base) + atr.idx0));
}
}
var child = ru32(this + SO.hierarchy_idx);
while (child != 0) {
const attach_idx = ru32(child + 0x1D4);
if (attach_idx != 0xFFFF) {
const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C);
if (visible != 0) {
const att_entry = attach_data + attach_idx * 0x30;
const bone_idx = @as(u32, ru16(att_entry + 4));
const bone_mat = bone_out_base + bone_idx * 0x40;
// Copy parent bone matrix to local (f32, matches original).
// The matrix itself is stored f32 in game memory; we preserve
// that on the wire but widen to f64 for the offset math below.
var local_1a0: [16]f32 = undefined;
for (0..16) |fi| {
local_1a0[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4);
}
// Apply attachment offset translation in f64, narrow on store.
const ox: f64 = @floatCast(rf32(att_entry + 8));
const oy: f64 = @floatCast(rf32(att_entry + 0xC));
const oz: f64 = @floatCast(rf32(att_entry + 0x10));
const m0x: f64 = @floatCast(local_1a0[0]);
const m4x: f64 = @floatCast(local_1a0[4]);
const m8x: f64 = @floatCast(local_1a0[8]);
const m1x: f64 = @floatCast(local_1a0[1]);
const m5x: f64 = @floatCast(local_1a0[5]);
const m9x: f64 = @floatCast(local_1a0[9]);
const m2x: f64 = @floatCast(local_1a0[2]);
const m6x: f64 = @floatCast(local_1a0[6]);
const m10x: f64 = @floatCast(local_1a0[10]);
local_1a0[12] = @floatCast(@as(f64, @floatCast(local_1a0[12])) + m0x * ox + m4x * oy + m8x * oz);
local_1a0[13] = @floatCast(@as(f64, @floatCast(local_1a0[13])) + m1x * ox + m5x * oy + m9x * oz);
local_1a0[14] = @floatCast(@as(f64, @floatCast(local_1a0[14])) + m2x * ox + m6x * oy + m10x * oz);
// Recurse into the f64 implementation so attachment children stay
// on the same precision path as the parent.
transformImpl_SSE64(child, @intFromPtr(&local_1a0), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z));
}
}
child = ru32(child + 0x1E4);
}
}
// =============================================================================
// Main entry point — f64-intermediate reimplementation.
//
// Follows the same structure as bone_sse.transformImpl_SSE. Per-bone work
// builds local_mat / local_mat2 as [16]f64 and uses the f64 matrix helpers
// above. Non-matrix helpers (interp, billboard, loops) are imported from
// bone_sse — their f32 outputs widen implicitly when used in f64 math.
// Attachment recursion is handled by a local f64 version (above) so child
// scene objects stay on the f64 pipeline.
// =============================================================================
pub fn transformImpl_SSE64(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
@setEvalBranchQuota(50000);
// Section 1: Entry checks
if (ru32(this + SO.model_data_ptr) == 0) return;
const anim_ctx = ru32(this + SO.anim_ctx_ptr);
if (ru32(this + SO.sync_value) == ru32(anim_ctx + 0x10)) return;
// Section 2: Emitter setup
const model_ctr = ru32(this + SO.model_ctr_ptr);
const model_hdr = ru32(model_ctr + 0x130);
const emitter_ctx = ru32(this + SO.emitter_ctx);
if (emitter_ctx != 0) {
const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0;
wu32(this + 0x50, has_emitter);
wu32(this + 0x17C, ru32(emitter_ctx + 0x17C));
}
// Section 3: World position/scale
const pos_ptr = mat2;
const ofs_ptr = mat3;
const scale_f: f32 = @bitCast(mat4);
wf32(this + SO.world_pos + 0, rf32(pos_ptr) * rf32(this + SO.field_184));
wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(pos_ptr + 4));
wf32(this + SO.world_pos + 8, @bitCast(fbits(rf32(this + SO.field_18c) * rf32(pos_ptr + 8))));
const rp0 = rf32(ofs_ptr) + rf32(this + SO.field_190);
const rp1 = rf32(this + SO.render_scale_x) + rf32(ofs_ptr + 4);
const rp2 = rf32(this + SO.render_scale_y) + rf32(ofs_ptr + 8);
wf32(this + SO.render_pri + 0, rp0);
wf32(this + SO.render_pri + 4, rp1);
wf32(this + SO.render_pri + 8, rp2);
wf32(this + SO.render_scale_z, scale_f * rf32(this + SO.field_180));
// Section 4: Global sequence processing
const gs_count = ru32(model_hdr + 0x14);
if (gs_count != 0) {
const gs_durations = ru32(model_hdr + 0x18);
const gs_values = ru32(this + SO.gs_values_ptr);
const timestamp = ru32(anim_ctx + 0x0C);
const time_base = ru32(this + SO.gs_time_base);
var gi: u32 = 0;
while (gi < gs_count) : (gi += 1) {
const dur = ru32(gs_durations + gi * 4);
if (dur == 0) {
wu32(gs_values + gi * 4, 0);
} else {
wu32(gs_values + gi * 4, (timestamp -% time_base) % dur);
}
}
}
// Root matrix multiply (f64 intermediates).
// Replaces the original initParticlePixelShaderGeneration(0x74a7c0) dispatch.
matMul4x4_64(this + 0xFC, this + 0xBC, mat1);
// Section 5: child_objects_padding — compute in f64, narrow on store.
// May be used as a culling threshold; matching precision keeps boundary
// conditions stable across frames.
const emitter_ctx_5 = ru32(this + SO.emitter_ctx);
if (emitter_ctx_5 == 0 or (ru8(emitter_ctx_5 + 4) & 1) != 0) {
const wx: f64 = rf64(this + SO.world_xform + 8 * 4);
const wy: f64 = rf64(this + SO.world_xform + 9 * 4);
const wz: f64 = rf64(this + SO.world_xform + 10 * 4);
const len_sq: f32 = @floatCast(wx * wx + wy * wy + wz * wz);
wu32(this + SO.child_padding, fbits(len_sq));
} else {
wu32(this + SO.child_padding, ru32(emitter_ctx_5 + 0x84));
}
// Section 6: Identity matrices as f64 locals + timestamp delta
var local_mat: [16]f64 = .{
1, 0, 0, 0,
0, 1, 0, 0,
0, 0, 1, 0,
0, 0, 0, 1,
};
var local_mat2: [16]f64 = .{
1, 0, 0, 0,
0, 1, 0, 0,
0, 0, 1, 0,
0, 0, 0, 1,
};
var time_delta_val: u32 = 0;
const sdb = ru32(this + SO.search_data_base);
if (sdb != 0) {
const cur_ts = ru32(anim_ctx + 0x0C);
if (cur_ts != 0) {
time_delta_val = cur_ts -% sdb;
wu32(this + SO.search_data_base, cur_ts);
}
}
// Section 7: Main bone loop
const bone_count = ru32(model_hdr + 0x34);
const bone_defs = ru32(model_hdr + 0x38);
const bone_rt_base = ru32(this + SO.bone_rt_base);
const bone_out_base = ru32(this + SO.bone_out_ptr);
const frame_ctr = ru32(this + SO.anim_frame_ctr);
if (bone_count != 0) {
var bone_idx: u32 = 0;
var bdef = bone_defs;
var brt = bone_rt_base;
while (bone_idx < bone_count) : ({
bone_idx += 1;
bdef += 0x6C;
brt += 0x118;
}) {
const flags = ru32(bdef + BD.flags);
const parent_idx_raw: i32 = @as(i32, @intCast(@as(i16, @bitCast(ru16(bdef + BD.parent_bone)))));
// --- Animation time computation (primary slot) ---
const anim_slot_val = ri32(brt + BR.anim_slot);
if (anim_slot_val == -1) {
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118;
wu32(brt + BR.prim_time, ru32(parent_rt + BR.prim_time));
wu32(brt + BR.prim_track, ru32(parent_rt + BR.prim_track));
wu32(brt + BR.prim_anim, ru32(parent_rt + BR.prim_anim));
} else if (bone_idx != 0) {
wu32(brt + BR.prim_time, ru32(bone_rt_base + BR.prim_time));
wu32(brt + BR.prim_track, ru32(bone_rt_base + BR.prim_track));
wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim));
}
} else {
if (ru32(this + 0x4C) != 0) {
wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val);
wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val);
}
const anim_lookup = ru32(model_hdr + 0x20);
const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44;
const cur_time = ru32(ru32(this + 0x2C) + 0xC);
if ((ru8(anim_entry + 0x10) & 1) == 0) {
const anim_end = ru32(anim_entry + 0x08);
const anim_start = ru32(anim_entry + 0x04);
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
const delta = cur_time -% ru32(brt + 0xA8);
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0);
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start);
wu32(brt + 0x98, anim_start +% frame);
} else {
wu32(brt + 0x98, anim_start);
}
} else {
const sec_end_val = ru32(brt + 0xAC);
const sec_start_val = ru32(brt + 0xA8);
if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) {
const effective_time = if (@as(i32, @bitCast(sec_start_val -% cur_time)) > 0) sec_start_val else cur_time;
const anim_end = ru32(anim_entry + 0x08);
const anim_start = ru32(anim_entry + 0x04);
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
const delta = effective_time -% ru32(brt + 0xA8);
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0);
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start);
wu32(brt + 0x98, anim_start +% frame);
} else {
wu32(brt + 0x98, anim_start);
}
} else {
const dur = sec_end_val -% sec_start_val;
const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xB0);
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8)));
if (offset < 0) {
wu32(brt + 0x98, ru32(anim_entry + 0x04));
} else {
const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08)));
const anim_start_i = @as(i32, @bitCast(ru32(anim_entry + 0x04)));
if (offset <= anim_end_i - anim_start_i) {
wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i)));
} else {
wu32(brt + 0x98, ru32(anim_entry + 0x08));
}
}
}
}
wu32(brt + 0x9C, ru32(brt + 0xA4));
wu32(brt + 0xA0, bone_idx);
}
// --- Secondary slot ---
const sec_slot_val = ri32(brt + BR.sec_slot);
if (sec_slot_val == -1) {
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118;
wu32(brt + BR.sec_time, ru32(parent_rt + BR.sec_time));
wu32(brt + BR.sec_track, ru32(parent_rt + BR.sec_track));
} else if (bone_idx != 0) {
wu32(brt + BR.sec_time, ru32(bone_rt_base + BR.sec_time));
wu32(brt + BR.sec_track, ru32(bone_rt_base + BR.sec_track));
} else {
wu32(brt + BR.sec_time, ru32(brt + BR.prim_time));
wu32(brt + BR.sec_track, ru32(brt + BR.prim_track));
}
} else {
if (ru32(this + 0x4C) != 0) {
wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val);
wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val);
}
const sec_anim_lookup = ru32(model_hdr + 0x20);
const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44;
const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC);
if ((ru8(sec_anim_entry + 0x10) & 1) == 0) {
const anim_end = ru32(sec_anim_entry + 0x08);
const anim_start = ru32(sec_anim_entry + 0x04);
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
const delta = sec_cur_time -% ru32(brt + 0xD4);
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC);
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start);
wu32(brt + 0xC4, anim_start +% frame);
} else {
wu32(brt + 0xC4, anim_start);
}
} else {
const sec_end_val = ru32(brt + 0xD8);
const sec_start_val = ru32(brt + 0xD4);
if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) {
const effective_time = if (@as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) sec_start_val else sec_cur_time;
const anim_end = ru32(sec_anim_entry + 0x08);
const anim_start = ru32(sec_anim_entry + 0x04);
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
const delta = effective_time -% ru32(brt + 0xD4);
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC);
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start);
wu32(brt + 0xC4, anim_start +% frame);
} else {
wu32(brt + 0xC4, anim_start);
}
} else {
const dur = sec_end_val -% sec_start_val;
const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xDC);
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4)));
if (offset < 0) {
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04));
} else {
const anim_end_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x08)));
const anim_start_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x04)));
if (offset <= anim_end_i - anim_start_i) {
wu32(brt + 0xC4, @as(u32, @bitCast(offset + anim_start_i)));
} else {
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x08));
}
}
}
}
wu32(brt + 0xC8, ru32(brt + 0xD0));
if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) {
wu32(brt + 0xD0, 0xFFFFFFFF);
}
}
// --- Blend weight ---
if (ri32(brt + BR.anim_slot) == -1 and ri32(brt + BR.sec_slot) == -1) {
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
wu32(brt + BR.blend_weight, ru32(bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118 + BR.blend_weight));
} else if (bone_idx == 0) {
wu32(brt + BR.blend_weight, 0);
} else {
wu32(brt + BR.blend_weight, ru32(bone_rt_base + BR.blend_weight));
}
} else {
const cf_remaining = ri32(brt + BR.crossfade_end) - ri32(anim_ctx + 0x0C);
if (cf_remaining < 1 or (ru32(brt + BR.prim_time) == ru32(brt + BR.sec_time) and
ru32(brt + BR.prim_track) == ru32(brt + BR.sec_track)))
{
wu32(brt + BR.blend_weight, 0);
} else {
const t_raw = @as(f32, @floatFromInt(cf_remaining)) * ufloat(ru32(brt + BR.crossfade_inv));
const t_clamped = if (t_raw < 0.0) @as(f32, 0.0) else if (t_raw > 1.0) @as(f32, 1.0) else t_raw;
const h = (3.0 - 2.0 * t_clamped) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight));
wu32(brt + BR.blend_weight, fbits(h));
}
}
// --- Parent bone transform inheritance ---
const combined_flags: u32 = ru32(brt + BR.flags2) | flags;
var src_mat: u32 = undefined;
// Address of local_mat (as if it were a game-memory matrix — since the
// matMul helpers read from memory, we write local_mat out to a scratch
// f32 buffer when src_mat aliases it. The billboard path below stores
// f32-narrowed values into local_mat's address via a scratch buffer.)
var local_mat_f32: [16]f32 = undefined;
const local_mat_addr = @intFromPtr(&local_mat_f32);
if (ru16(bdef + BD.parent_bone) == 0xFFFF) {
src_mat = this + 0xFC;
} else {
const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40;
src_mat = parent_out;
if ((combined_flags & 7) != 0) {
// Copy parent matrix to local_mat (f64 widen)
for (0..16) |i| {
local_mat[i] = rf64(parent_out + @as(u32, @intCast(i)) * 4);
}
const pivot_x: f64 = rf64(bdef + BD.pivot_x);
const pivot_y: f64 = rf64(bdef + BD.pivot_y);
const pivot_z: f64 = rf64(bdef + BD.pivot_z);
// Compute translated position — accumulation order must match
// original x87. Row 0 uses (pz + px + py), rows 1/2 use (pz + py + px).
const tx = local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[12];
const ty = local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x + local_mat[13];
const tz = local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x + local_mat[14];
const bb_type = combined_flags & 6;
const billboard_eps: f64 = @floatCast(rf32(0x008029d4));
const cull_eps: f64 = @floatCast(rf32(0x0080c5c8));
if (bb_type == 2) {
// Cylindrical billboard — normalize columns in f64 to match
// x87's extended-precision 1/sqrt pattern. Previous impl went
// f32→f32 through normalizeVec3; that diverged from x87 by
// up to a ULP per axis, which shifted particle emitter
// orientation each camera frame and caused transparency
// flicker against ground effects.
inline for ([_]u32{ 0, 4, 8 }) |row_off| {
const cx = local_mat[row_off];
const cy = local_mat[row_off + 1];
const cz = local_mat[row_off + 2];
const len_sq = cx * cx + cy * cy + cz * cz;
const len = @sqrt(len_sq);
if (len >= billboard_eps) {
const inv = 1.0 / len;
local_mat[row_off] = cx * inv;
local_mat[row_off + 1] = cy * inv;
local_mat[row_off + 2] = cz * inv;
}
}
} else if (bb_type == 4) {
// Spherical billboard — inherit camera basis, rescale to
// preserve original column length. Keep all intermediates
// in f64 matching x87's 80-bit temporaries.
inline for ([_]struct { row_off: u32, src_off: u32 }{
.{ .row_off = 0, .src_off = SO.bb_row0 },
.{ .row_off = 4, .src_off = SO.world_xform },
.{ .row_off = 8, .src_off = SO.world_xform + 16 },
}) |p| {
const src_addr = this + p.src_off;
const cam_x: f64 = rf64(src_addr);
const cam_y: f64 = rf64(src_addr + 4);
const cam_z: f64 = rf64(src_addr + 8);
const cam_len_sq = cam_x * cam_x + cam_y * cam_y + cam_z * cam_z;
var s: f64 = 1.0;
if (cam_len_sq > cull_eps) {
const mx = local_mat[p.row_off];
const my = local_mat[p.row_off + 1];
const mz = local_mat[p.row_off + 2];
const mat_len_sq = mx * mx + my * my + mz * mz;
s = @sqrt(mat_len_sq / cam_len_sq);
}
local_mat[p.row_off] = s * cam_x;
local_mat[p.row_off + 1] = s * cam_y;
local_mat[p.row_off + 2] = s * cam_z;
}
} else if (bb_type == 6) {
local_mat[0] = rf64(this + SO.bb_row0);
local_mat[1] = rf64(this + SO.bb_row0 + 4);
local_mat[2] = rf64(this + SO.bb_row0 + 8);
local_mat[4] = rf64(this + SO.world_xform + 0 * 4);
local_mat[5] = rf64(this + SO.world_xform + 1 * 4);
local_mat[6] = rf64(this + SO.world_xform + 2 * 4);
local_mat[8] = rf64(this + SO.world_xform + 4 * 4);
local_mat[9] = rf64(this + SO.world_xform + 5 * 4);
local_mat[10] = rf64(this + SO.world_xform + 6 * 4);
}
// Recompute translation — mirror accumulation order of tx/ty/tz above.
if ((combined_flags & 1) == 0) {
local_mat[12] = tx - (local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y);
local_mat[13] = ty - (local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x);
local_mat[14] = tz - (local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x);
} else {
local_mat[12] = rf64(this + SO.world_xform + 8 * 4);
local_mat[13] = rf64(this + SO.world_xform + 9 * 4);
local_mat[14] = rf64(this + SO.world_xform + 10 * 4);
}
// Narrow local_mat to f32 scratch, set src_mat to its address
for (0..16) |i| {
local_mat_f32[i] = @floatCast(local_mat[i]);
}
src_mat = local_mat_addr;
}
}
// --- Rotation / Scale / Translation / Final multiply ---
if ((combined_flags & 0x280) == 0) {
const dst = bone_out_base + bone_idx * 0x40;
copyMat4(dst, src_mat);
} else {
const rot_anim = bdef + BD.rot_anim;
const rot_kf_count = ru32(bdef + BD.rot_nts);
var rot_primary_cache: ?InterpResult = null;
if (rot_kf_count != 0) {
if (frame_ctr < rot_kf_count) {
const rot_output = brt + BR.rot_idx0;
const r = findInterpIdx(this, ru32(brt + BR.prim_time), ru32(brt + BR.prim_track), rot_anim, rot_output);
rot_primary_cache = r;
const q = interpAnimKFCached(this, brt, rot_anim, rot_output, r);
local_mat2 = buildRotationMatrix_64(q[0], q[1], q[2], q[3]);
} else {
local_mat2 = buildRotationMatrix_64(rf64(brt + BR.rot_x), rf64(brt + BR.rot_y), rf64(brt + BR.rot_z), rf64(brt + BR.rot_w));
}
} else {
local_mat2 = .{
1, 0, 0, 0,
0, 1, 0, 0,
0, 0, 1, 0,
0, 0, 0, 1,
};
}
// Step 2: Scale interpolation — f64 arithmetic
const scale_anim = bdef + BD.scale_anim;
const scale_kf_count = ru32(bdef + BD.scale_nts);
if (scale_kf_count != 0) {
var sx: f64 = undefined;
var sy: f64 = undefined;
var sz: f64 = undefined;
if (frame_ctr < scale_kf_count) {
const scale_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, scale_anim)) rot_primary_cache else null;
const s = interpVec3TrackCached(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)), scale_cache);
sx = s[0]; sy = s[1]; sz = s[2];
} else {
sx = rf64(brt + BR.scale_x); sy = rf64(brt + BR.scale_y); sz = rf64(brt + BR.scale_z);
}
local_mat2[0] *= sx; local_mat2[1] *= sx; local_mat2[2] *= sx;
local_mat2[4] *= sy; local_mat2[5] *= sy; local_mat2[6] *= sy;
local_mat2[8] *= sz; local_mat2[9] *= sz; local_mat2[10] *= sz;
}
// Conditional bone-flag matrix multiply (f64 in-place)
if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) {
local_mat2 = matMul4x4InPlace_64(local_mat2, ru32(brt + BR.bone_flag_cache));
}
// Step 3: Translation interpolation (f64)
var tx_val: f64 = rf64(bdef + BD.pivot_x);
var ty_val: f64 = rf64(bdef + BD.pivot_y);
var tz_val: f64 = rf64(bdef + BD.pivot_z);
const trans_anim = bdef + BD.trans_anim;
const trans_kf_count = ru32(bdef + BD.trans_nts);
if (trans_kf_count != 0) {
if (frame_ctr < trans_kf_count) {
const trans_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, trans_anim)) rot_primary_cache else null;
const t = interpVec3TrackCached(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)), trans_cache);
tx_val += @as(f64, t[0]);
ty_val += @as(f64, t[1]);
tz_val += @as(f64, t[2]);
} else {
tx_val += rf64(brt + BR.trans_x);
ty_val += rf64(brt + BR.trans_y);
tz_val += rf64(brt + BR.trans_z);
}
}
// Step 4: Translation offset using ROTATED+SCALED matrix (f64)
const piv_x: f64 = rf64(bdef + BD.pivot_x);
const piv_y: f64 = rf64(bdef + BD.pivot_y);
const piv_z: f64 = rf64(bdef + BD.pivot_z);
local_mat2[12] = tx_val - (local_mat2[0] * piv_x + local_mat2[4] * piv_y + local_mat2[8] * piv_z);
local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z);
local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z);
// Final: dst = bone_local * parent (f64 mul, narrow on store)
matMul4x4Local_64(bone_out_base + bone_idx * 0x40, local_mat2, src_mat);
}
// --- Billboard post-processing (flags & 0x78) ---
if ((combined_flags & 0x78) != 0) {
const out_off = bone_idx * 0x40;
const om = bone_out_base + out_off;
const scale_len0 = @sqrt(callVec3SqMag(om));
const scale_len1 = @sqrt(callVec3SqMag(om + 0x10));
const scale_len2 = @sqrt(callVec3SqMag(om + 0x20));
const bpx = rf32(bdef + BD.pivot_x);
const bpy = rf32(bdef + BD.pivot_y);
const bpz = rf32(bdef + BD.pivot_z);
// Accumulation order mirrors original x87:
// pos_x: px + py + pz + const
// pos_y: py + pz + px + const
// pos_z: py + pz + px + const
const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30);
const pos_y = bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + bpx * rf32(om + 0x04) + rf32(om + 0x34);
const pos_z = bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + bpx * rf32(om + 0x08) + rf32(om + 0x38);
const bb_post = combined_flags & 0x78;
switch (bb_post) {
0x08 => {
const had_anim = (combined_flags & 0x280) != 0;
if (!had_anim) {
wf32(om, 0);
wf32(om + 0x04, 0);
wf32(om + 0x08, -1);
wf32(om + 0x10, 1);
wf32(om + 0x14, 0);
wf32(om + 0x18, 0);
wf32(om + 0x20, 0);
wf32(om + 0x24, 1);
wf32(om + 0x28, 0);
} else {
// Row 0 = {local_e4, local_e0, -local_e8}
const r0x: f32 = @floatCast(local_mat2[1]);
const r0y: f32 = @floatCast(local_mat2[2]);
const r0z: f32 = @floatCast(-local_mat2[0]);
wf32(om, r0x);
wf32(om + 0x04, r0y);
wf32(om + 0x08, r0z);
normalizeVec3InPlace(om);
const r1x: f32 = @floatCast(local_mat2[5]);
const r1y: f32 = @floatCast(local_mat2[6]);
const r1z: f32 = @floatCast(-local_mat2[4]);
wf32(om + 0x10, r1x);
wf32(om + 0x14, r1y);
wf32(om + 0x18, r1z);
normalizeVec3InPlace(om + 0x10);
const r2x: f32 = @floatCast(local_mat2[9]);
const r2y: f32 = @floatCast(local_mat2[10]);
const r2z: f32 = @floatCast(-local_mat2[8]);
wf32(om + 0x20, r2x);
wf32(om + 0x24, r2y);
wf32(om + 0x28, r2z);
normalizeVec3InPlace(om + 0x20);
}
},
0x10 => {
normalizeVec3InPlace(om);
const r0x = rf32(om);
const r0y = rf32(om + 0x04);
wf32(om + 0x10, r0y);
wf32(om + 0x14, -r0x);
wf32(om + 0x18, 0);
normalizeVec3InPlace(om + 0x10);
wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18));
wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10));
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
},
0x20 => {
normalizeVec3InPlace(om + 0x10);
wf32(om, -rf32(om + 0x14));
wf32(om + 0x04, rf32(om + 0x10));
wf32(om + 0x08, 0);
normalizeVec3InPlace(om);
wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18));
wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10));
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
},
0x40 => {
normalizeVec3InPlace(om + 0x20);
wf32(om + 0x10, rf32(om + 0x24));
wf32(om + 0x14, -rf32(om + 0x20));
wf32(om + 0x18, 0);
normalizeVec3InPlace(om + 0x10);
wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14));
wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18));
wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10));
},
else => {},
}
// Apply scale lengths back and recompute translation
wf32(om + 0x0C, 0);
wf32(om + 0x1C, 0);
wf32(om + 0x2C, 0);
const r0x_s = rf32(om);
wf32(om, scale_len0 * r0x_s);
const r0y_s = rf32(om + 0x04);
wf32(om + 0x04, scale_len0 * r0y_s);
const r0z_s = rf32(om + 0x08);
wf32(om + 0x08, scale_len0 * r0z_s);
const r1x_s = rf32(om + 0x10);
wf32(om + 0x10, scale_len1 * r1x_s);
const r1y_s = rf32(om + 0x14);
wf32(om + 0x14, scale_len1 * r1y_s);
const r1z_s = rf32(om + 0x18);
wf32(om + 0x18, scale_len1 * r1z_s);
const r2x_s = rf32(om + 0x20);
wf32(om + 0x20, scale_len2 * r2x_s);
const r2y_s = rf32(om + 0x24);
wf32(om + 0x24, scale_len2 * r2y_s);
const r2z_s = rf32(om + 0x28);
wf32(om + 0x28, scale_len2 * r2z_s);
// Accumulation order must match original x87: row0 + row2 + row1
// (pivot_x, then pivot_z, then pivot_y). f32 addition isn't associative —
// this ordering is load-bearing for spell-effect z-fighting avoidance.
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len2 * r2x_s * bpz + scale_len1 * r1x_s * bpy));
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len2 * r2y_s * bpz + scale_len1 * r1y_s * bpy));
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len2 * r2z_s * bpz + scale_len1 * r1z_s * bpy));
wf32(om + 0x3C, 1.0);
}
}
}
// Sections 8-13: delegated to bone_sse's f32 implementations.
texAnimLoop(this, model_hdr, frame_ctr);
colorAnimLoop(this, model_hdr, frame_ctr);
wordAnimLoop(this, model_hdr, frame_ctr);
boneKeyframeLoop(this, model_hdr);
particleLoops(this, model_hdr, frame_ctr);
attachmentRecursion64(this, model_hdr, bone_out_base, frame_ctr);
wu32(this + SO.sync_value, ru32(anim_ctx + 0x10));
}
+78 -11
View File
@@ -38,6 +38,7 @@ pub fn isActive() bool {
// =============================================================================
const bone_sse = @import("bone_sse.zig");
const bone_sse64 = @import("bone_sse64.zig");
const particle_sse = @import("particle_sse.zig");
const clip_sse = @import("clip_sse.zig");
const cull_sse = @import("cull_sse.zig");
@@ -195,6 +196,72 @@ fn destroyObjMgrDetour() callconv(hook.cc.stdcall) void {
destroy_objmgr_hook.callOriginal(.{});
}
// =============================================================================
// Spell ground effect diagnostic — safe cross-reference approach
// createModelAttachment saves model ptrs, ManageRenderListNode checks matches.
// No deferred pointer reads — only value comparisons.
// =============================================================================
const ModelAttachFn = fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
var model_attach_hook: hook.Detour(ModelAttachFn) = .{};
const MAX_TRACKED = 16;
var tracked_models: [MAX_TRACKED]u32 = [_]u32{0} ** MAX_TRACKED;
var tracked_next: u32 = 0;
fn modelAttachDetour(parent: u32, path_ptr: u32, flags: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
const result = model_attach_hook.callOriginal(.{ parent, path_ptr, flags });
if (path_ptr != 0 and result != 0) {
const path: [*]const u8 = @ptrFromInt(path_ptr);
if (path[0] == 'S' and path[1] == 'p' and path[2] == 'e' and path[3] == 'l' and path[4] == 'l' and path[5] == 's') {
const path_z: [*:0]const u8 = @ptrFromInt(path_ptr);
log.fmt("[spell] created 0x{x}: {s}\n", .{ result, path_z });
tracked_models[tracked_next % MAX_TRACKED] = result;
tracked_next +%= 1;
}
}
return result;
}
// CM2Model_ManageRenderListNode (0x710B90)
// __thiscall(ECX=model, add_remove), RET 0x4
const ManageRLFn = fn (u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
var manage_rl_hook: hook.Detour(ManageRLFn) = .{};
fn manageRLDetour(model: u32, add_remove: u32) callconv(.{ .x86_thiscall = .{} }) void {
for (&tracked_models) |tp| {
if (tp != 0 and tp == model) {
if (add_remove != 0) {
log.fmt("[spell] 0x{x} ADDED to render list\n", .{model});
} else {
log.fmt("[spell] 0x{x} REMOVED from render list\n", .{model});
}
break;
}
}
manage_rl_hook.callOriginal(.{ model, add_remove });
}
// =============================================================================
// ProcessProjectileMovementWithCollisionAndTargetValidation fix (0x61e1d0)
// __thiscall(ECX=missile, target_ptr, param_3), RET 0x8
//
// Vanilla bug: when target_ptr != 0 (a unit/dynobj exists at the AoE location),
// the code takes an alternate path that skips ProcessMissileSpellEffects entirely.
// This means the area effect ground model (e.g. InfectedSecretion_Marked.m2) is
// never created. Fix: force target_ptr=0 so the area effect path always runs.
// Target-unit visuals fire separately through ProcessSpellVisualKit.
// =============================================================================
const ProjMoveFn = fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
var proj_move_hook: hook.Detour(ProjMoveFn) = .{};
fn projMoveDetour(missile: u32, target_ptr: u32, param_3: u32) callconv(.{ .x86_thiscall = .{} }) void {
_ = target_ptr;
proj_move_hook.callOriginal(.{ missile, 0, param_3 });
}
// =============================================================================
// OnWorldUpdate hook (0x482EA0) — per-frame cache reset
// =============================================================================
@@ -298,7 +365,7 @@ pub fn installHooks() void {
var installed: u32 = 0;
// Bone transform SSE
if (transform_hook.attach(0x714260, &bone_sse.transformImpl_SSE) == .ok) installed += 1;
if (transform_hook.attach(0x714260, &bone_sse64.transformImpl_SSE64) == .ok) installed += 1;
// Frustum clip SSE (1.9x speedup)
if (clip_hook.attach(0x6318C0, &clip_sse.clipPolygonToSinglePlane) == .ok) installed += 1;
@@ -306,19 +373,16 @@ pub fn installHooks() void {
// Particle rendering SSE
if (particle_hook.attach(0x7B2A50, &particleDetour) == .ok) installed += 1;
// Glyph cache
// Glyph cache removed -- game has internal glyph cache, our hook only sees misses (~30/frame)
// GUID lookup cache -- A/B testing via transform44
// if (findguid_hook.attach(0x464890, &findguidDetour) == .ok) installed += 1;
// if (obj_delete_hook.attach(0x464920, &objDeleteDetour) == .ok) installed += 1;
// GUID cache disabled for now
// if (destroy_objmgr_hook.attach(0x467700, &destroyObjMgrDetour) == .ok) installed += 1;
// Per-frame cache reset
if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1;
// Spell ground effect diagnostics
if (model_attach_hook.attach(0x707350, &modelAttachDetour) == .ok) installed += 1;
if (manage_rl_hook.attach(0x710B90, &manageRLDetour) == .ok) installed += 1;
// Area effect ground model fix
if (proj_move_hook.attach(0x61e1d0, &projMoveDetour) == .ok) installed += 1;
// Silicon SSE binary patches
_ = installPatches();
@@ -358,6 +422,9 @@ pub fn removeHooks() void {
obj_delete_hook.detach();
destroy_objmgr_hook.detach();
world_update_hook.detach();
model_attach_hook.detach();
manage_rl_hook.detach();
proj_move_hook.detach();
log.close();
mod_mutex.release(&g_mutex);
}