weirdperformance: f64-intermediate bone transform to eliminate spell/doodad z-fighting
Adds bone_sse64.zig as an f64-intermediate port of transformMatrix4x4, used as the active hook. M2 bone matrices are built and multiplied as [16]f64 and only narrow to f32 on final store into the bone output buffer -- matching the x87 original's rounding profile (wide intermediates, single f32 store) and keeping M2 vertex positions aligned with the terrain/projected-texture pipeline. Also fixes, in both bone_sse (f32) and bone_sse64: - Pre-billboard tx/ty/tz accumulation order (row 0 = pz+px+py; rows 1/2 = pz+py+px) - Post-billboard pos_y/pos_z accumulation order (py+pz+px) - Post-billboard scale-recompute accumulation order (row0 + row2 + row1) - Billboard types 2/4 normalize using f64 intermediates (load-bearing for camera basis vectors -- pure f32 drifted from x87 by a ULP per axis and caused particle emitters to jitter on camera motion) Additional bone_sse64-specific changes: - Local attachmentRecursion64 that recurses into transformImpl_SSE64 instead of bone_sse.transformImpl_SSE, so attached child models stay on the f64 path - child_padding (this+0x84) computed with f64 intermediates bone_sse remains the reference f32 implementation; its struct fields, inline helpers, and section-loop fns are now `pub` so bone_sse64 can share them (types/interpolation helpers/post-loop loops). Artifact size is unchanged. build.zig adds bench_bone_sse64 object; src/bench/main.zig runs the new variant through the same warmup/timing harness and prints SSE vs SSE64 vs BASELINE cycles plus a parity check.
This commit is contained in:
@@ -19,23 +19,23 @@ const ModuleDesc = struct {
|
||||
/// affect the module's DLL, minor for feature changes, major for breaking.
|
||||
const module_list = [_]ModuleDesc{
|
||||
.{ .name = "pngscreenshots", .version = "1.0.1", .desc = "Enable screenshot module", .src_dir = "screenshot" },
|
||||
.{ .name = "interact", .version = "1.0", .desc = "Enable interact module", .addon_name = "Interact" },
|
||||
.{ .name = "interact", .version = "1.1.0", .desc = "Enable interact module", .addon_name = "Interact" },
|
||||
.{ .name = "outline", .version = "1.0", .desc = "Enable outline module", .default = false, .addon_name = "Outline" },
|
||||
.{ .name = "worldmarkers", .version = "1.0", .desc = "Enable world markers module", .addon_name = "WorldMarkers", .addon_hidden = true },
|
||||
.{ .name = "worldmarkers", .version = "1.1", .desc = "Enable world markers module", .addon_name = "WorldMarkers", .addon_hidden = true },
|
||||
.{ .name = "framecrash", .version = "1.0", .desc = "Enable framecrash fix", .default = false },
|
||||
.{ .name = "logsessions", .version = "1.0.1", .desc = "Enable log session rotation", .addon_name = "LogSessions" },
|
||||
.{ .name = "logsessions", .version = "1.1.0", .desc = "Enable log session rotation", .addon_name = "LogSessions", .addon_hidden = true },
|
||||
.{ .name = "minimapicons", .version = "1.0.1", .desc = "Enable custom minimap icons", .addon_name = "MinimapIcons" },
|
||||
.{ .name = "transmogfix", .version = "1.0.1", .desc = "Enable transmog update coalescing" },
|
||||
.{ .name = "customassets", .version = "1.0.1", .desc = "Enable loose file loading & permissive patch glob" },
|
||||
.{ .name = "healtextfix", .version = "1.0.1", .desc = "Enable SuperWoW heal text fix" },
|
||||
.{ .name = "bigcursor", .version = "1.0.1", .desc = "Enable big cursor module" },
|
||||
.{ .name = "clickthrough", .version = "1.0.2", .desc = "Enable GO click-through" },
|
||||
.{ .name = "clickthrough", .version = "1.0.3", .desc = "Enable GO click-through" },
|
||||
.{ .name = "dpslog", .version = "0.1", .desc = "Enable structured combat log events for addons", .default = false },
|
||||
.{ .name = "transform44", .version = "1.0", .desc = "Enable transform44 profiling/A/B testing (dev only)", .default = false },
|
||||
.{ .name = "addonperf", .version = "1.0", .desc = "Enable addon memory/CPU profiling API", .default = false },
|
||||
.{ .name = "ssemaths", .version = "1.0", .desc = "Enable UnitXP x87 math polyfill replacements (SSE)", .default = false },
|
||||
.{ .name = "silicon", .version = "1.0", .desc = "Enable SSE2 math replacements (ported from libSiliconPatch)", .default = false },
|
||||
.{ .name = "weirdperformance", .version = "1.1.1", .desc = "Enable production performance optimizations (SSE, inflate, filecache, timer, luastr, luavm)", .default = true },
|
||||
.{ .name = "weirdperformance", .version = "1.2.1", .desc = "Enable production performance optimizations (SSE, inflate, filecache, timer, luastr, luavm)", .default = true },
|
||||
.{ .name = "superweirdo", .version = "0.1", .desc = "Enable GO loot sparkle on interactable objects", .default = false },
|
||||
.{ .name = "luagc", .version = "0.1", .desc = "Enable incremental Lua GC (replaces stop-the-world mark+sweep)", .default = false },
|
||||
};
|
||||
@@ -235,6 +235,18 @@ pub fn build(b: *std.Build) void {
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bench_bone_sse64 = b.addObject(.{
|
||||
.name = "bench_bone_sse64",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/weirdperformance/bone_sse64.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .linux,
|
||||
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
|
||||
}),
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bench_bone_baseline = b.addObject(.{
|
||||
.name = "bench_bone_baseline",
|
||||
.root_module = b.createModule(.{
|
||||
@@ -270,6 +282,7 @@ pub fn build(b: *std.Build) void {
|
||||
bench.root_module.addObject(bench_math_sse);
|
||||
bench.root_module.addObject(bench_silicon_sse);
|
||||
bench.root_module.addObject(bench_bone_sse);
|
||||
bench.root_module.addObject(bench_bone_sse64);
|
||||
bench.root_module.addObject(bench_bone_baseline);
|
||||
bench.root_module.addObject(bench_particle_sse);
|
||||
bench.root_module.addObject(bench_cull_sse);
|
||||
|
||||
@@ -1662,6 +1662,7 @@ pub fn main() void {
|
||||
const ofs = [3]f32{ 0, 0, 0 };
|
||||
const sb: u32 = @bitCast(@as(f32, 1.0));
|
||||
const transformImpl_SSE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE" });
|
||||
const transformImpl_SSE64 = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void, .{ .name = "transformImpl_SSE64" });
|
||||
const transformImpl_BASELINE = @extern(*const fn (u32, u32, u32, u32, u32) callconv(.c) void, .{ .name = "transformImpl_BASELINE" });
|
||||
|
||||
// Pre-set boneKeyframe init flag so we skip the atexit call (Windows CRT, can't run on Linux)
|
||||
@@ -1716,6 +1717,20 @@ pub fn main() void {
|
||||
const best_sse = run_bench_fn(transformImpl_SSE, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
|
||||
const avg_sse = best_sse / T44_ITERS;
|
||||
|
||||
// Warmup + bench bone_sse64 (f64-intermediate variant)
|
||||
for (0..500) |iter| {
|
||||
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(iter * 2)), .little);
|
||||
wu(u32, scene_obj[0x40..0x44], 0, .little);
|
||||
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
|
||||
}
|
||||
for (0..500) |iter| {
|
||||
wu(u32, anim_ctx_mem[0x0C..0x10], @as(u32, @intCast(999 - iter * 2)), .little);
|
||||
wu(u32, scene_obj[0x40..0x44], 0, .little);
|
||||
transformImpl_SSE64(so, @intFromPtr(&parent_mat), @intFromPtr(&pos), @intFromPtr(&ofs), sb);
|
||||
}
|
||||
const best_sse64 = run_bench_fn(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, T44_ITERS);
|
||||
const avg_sse64 = best_sse64 / T44_ITERS;
|
||||
|
||||
print(" BASELINE: {d} cycles/call (frozen)\n", .{BASELINE_CYCLES});
|
||||
print(" SSE: {d} cycles/call", .{avg_sse});
|
||||
if (avg_sse < BASELINE_CYCLES) {
|
||||
@@ -1727,6 +1742,25 @@ pub fn main() void {
|
||||
} else {
|
||||
print(" (same)\n", .{});
|
||||
}
|
||||
print(" SSE64: {d} cycles/call", .{avg_sse64});
|
||||
if (avg_sse64 < BASELINE_CYCLES) {
|
||||
const pct = (BASELINE_CYCLES - avg_sse64) * 100 / BASELINE_CYCLES;
|
||||
print(" (-{d}% vs BASELINE", .{pct});
|
||||
} else if (avg_sse64 > BASELINE_CYCLES) {
|
||||
const pct = (avg_sse64 - BASELINE_CYCLES) * 100 / BASELINE_CYCLES;
|
||||
print(" (+{d}% vs BASELINE", .{pct});
|
||||
} else {
|
||||
print(" (same as BASELINE", .{});
|
||||
}
|
||||
if (avg_sse64 > avg_sse) {
|
||||
const pct = (avg_sse64 - avg_sse) * 100 / avg_sse;
|
||||
print(", +{d}% vs SSE)\n", .{pct});
|
||||
} else if (avg_sse64 < avg_sse) {
|
||||
const pct = (avg_sse - avg_sse64) * 100 / avg_sse;
|
||||
print(", -{d}% vs SSE)\n", .{pct});
|
||||
} else {
|
||||
print(", same as SSE)\n", .{});
|
||||
}
|
||||
|
||||
// --- Output parity: run BASELINE then SSE with identical input, compare ALL outputs ---
|
||||
{
|
||||
@@ -1792,6 +1826,23 @@ pub fn main() void {
|
||||
} else {
|
||||
print(" parity: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs, total_len });
|
||||
}
|
||||
|
||||
// Run SSE64 with same input
|
||||
reset_and_run(transformImpl_SSE64, so, pm, pp, po, sb, &scene_obj, &anim_ctx_mem, &bone_rt, BONE_COUNT);
|
||||
|
||||
var diffs64: u32 = 0;
|
||||
off = 0;
|
||||
for (bufs) |b| {
|
||||
for (0..b.len) |i| {
|
||||
if (b.ptr[i] != snap[off + i]) diffs64 += 1;
|
||||
}
|
||||
off += b.len;
|
||||
}
|
||||
if (diffs64 == 0) {
|
||||
print(" parity64: PASS (SSE64 == BASELINE, {d} bytes checked)\n", .{total_len});
|
||||
} else {
|
||||
print(" parity64: FAIL ({d} byte diffs across {d} bytes)\n", .{ diffs64, total_len });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+250
-254
@@ -25,118 +25,118 @@ const V4 = @Vector(4, f32);
|
||||
// SceneObject field offsets — assembly-verified from [EBX+N] in transformMatrix4x4
|
||||
// =============================================================================
|
||||
|
||||
const SO = struct {
|
||||
const model_data_ptr: u32 = 0x010;
|
||||
const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value
|
||||
const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header
|
||||
const sync_value: u32 = 0x040;
|
||||
const search_data_base: u32 = 0x04C; // prev timestamp for delta
|
||||
const emitter_flag: u32 = 0x050;
|
||||
const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array
|
||||
const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS
|
||||
const child_padding: u32 = 0x084;
|
||||
const anim_frame_ctr: u32 = 0x08C;
|
||||
const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs
|
||||
const bone_out_ptr: u32 = 0x094; // output bone matrices
|
||||
const tex_anim_out: u32 = 0x0A0;
|
||||
const color_anim_out: u32 = 0x0A8;
|
||||
const scale1: u32 = 0x0AC;
|
||||
const scale2: u32 = 0x0B0;
|
||||
const scale3: u32 = 0x0B4;
|
||||
const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward)
|
||||
const world_xform: u32 = 0x10C; // float[16] world transform
|
||||
const field_17c: u32 = 0x17C;
|
||||
const field_180: u32 = 0x180;
|
||||
const field_184: u32 = 0x184;
|
||||
const field_188: u32 = 0x188;
|
||||
const field_18c: u32 = 0x18C;
|
||||
const field_190: u32 = 0x190;
|
||||
const render_scale_x: u32 = 0x194;
|
||||
const render_scale_y: u32 = 0x198;
|
||||
const render_scale_z: u32 = 0x19C;
|
||||
const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children)
|
||||
const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children)
|
||||
const hierarchy_ptr: u32 = 0x1C8;
|
||||
const emitter_ctx: u32 = 0x1CC;
|
||||
const field_1d8: u32 = 0x1D8;
|
||||
const hierarchy_idx: u32 = 0x1DC;
|
||||
const field_200: u32 = 0x200;
|
||||
const particle1: u32 = 0x3C4;
|
||||
const particle2: u32 = 0x3C8;
|
||||
const particle3: u32 = 0x3D0;
|
||||
const particle4: u32 = 0x3D4;
|
||||
const add_remaining: u32 = 0x3D8;
|
||||
pub const SO = struct {
|
||||
pub const model_data_ptr: u32 = 0x010;
|
||||
pub const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value
|
||||
pub const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header
|
||||
pub const sync_value: u32 = 0x040;
|
||||
pub const search_data_base: u32 = 0x04C; // prev timestamp for delta
|
||||
pub const emitter_flag: u32 = 0x050;
|
||||
pub const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array
|
||||
pub const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS
|
||||
pub const child_padding: u32 = 0x084;
|
||||
pub const anim_frame_ctr: u32 = 0x08C;
|
||||
pub const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs
|
||||
pub const bone_out_ptr: u32 = 0x094; // output bone matrices
|
||||
pub const tex_anim_out: u32 = 0x0A0;
|
||||
pub const color_anim_out: u32 = 0x0A8;
|
||||
pub const scale1: u32 = 0x0AC;
|
||||
pub const scale2: u32 = 0x0B0;
|
||||
pub const scale3: u32 = 0x0B4;
|
||||
pub const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward)
|
||||
pub const world_xform: u32 = 0x10C; // float[16] world transform
|
||||
pub const field_17c: u32 = 0x17C;
|
||||
pub const field_180: u32 = 0x180;
|
||||
pub const field_184: u32 = 0x184;
|
||||
pub const field_188: u32 = 0x188;
|
||||
pub const field_18c: u32 = 0x18C;
|
||||
pub const field_190: u32 = 0x190;
|
||||
pub const render_scale_x: u32 = 0x194;
|
||||
pub const render_scale_y: u32 = 0x198;
|
||||
pub const render_scale_z: u32 = 0x19C;
|
||||
pub const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children)
|
||||
pub const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children)
|
||||
pub const hierarchy_ptr: u32 = 0x1C8;
|
||||
pub const emitter_ctx: u32 = 0x1CC;
|
||||
pub const field_1d8: u32 = 0x1D8;
|
||||
pub const hierarchy_idx: u32 = 0x1DC;
|
||||
pub const field_200: u32 = 0x200;
|
||||
pub const particle1: u32 = 0x3C4;
|
||||
pub const particle2: u32 = 0x3C8;
|
||||
pub const particle3: u32 = 0x3D0;
|
||||
pub const particle4: u32 = 0x3D4;
|
||||
pub const add_remaining: u32 = 0x3D8;
|
||||
};
|
||||
|
||||
// Bone runtime struct offsets (within 0x118-byte per-bone runtime)
|
||||
const BR = struct {
|
||||
pub const BR = struct {
|
||||
// Translation interpolation state
|
||||
const trans_idx0: u32 = 0x00; // [0] lower keyframe index
|
||||
const trans_idx1: u32 = 0x04; // [1] upper keyframe index
|
||||
const trans_t: u32 = 0x08; // [2] interpolation factor (float bits)
|
||||
const trans_x: u32 = 0x0C; // [3] interpolated translation X
|
||||
const trans_y: u32 = 0x10; // [4] Y
|
||||
const trans_z: u32 = 0x14; // [5] Z
|
||||
pub const trans_idx0: u32 = 0x00; // [0] lower keyframe index
|
||||
pub const trans_idx1: u32 = 0x04; // [1] upper keyframe index
|
||||
pub const trans_t: u32 = 0x08; // [2] interpolation factor (float bits)
|
||||
pub const trans_x: u32 = 0x0C; // [3] interpolated translation X
|
||||
pub const trans_y: u32 = 0x10; // [4] Y
|
||||
pub const trans_z: u32 = 0x14; // [5] Z
|
||||
// Secondary translation (crossfade)
|
||||
const trans2_idx0: u32 = 0x18;
|
||||
const trans2_idx1: u32 = 0x1C;
|
||||
const trans2_t: u32 = 0x20;
|
||||
const trans2_x: u32 = 0x24;
|
||||
const trans2_y: u32 = 0x28;
|
||||
const trans2_z: u32 = 0x2C;
|
||||
pub const trans2_idx0: u32 = 0x18;
|
||||
pub const trans2_idx1: u32 = 0x1C;
|
||||
pub const trans2_t: u32 = 0x20;
|
||||
pub const trans2_x: u32 = 0x24;
|
||||
pub const trans2_y: u32 = 0x28;
|
||||
pub const trans2_z: u32 = 0x2C;
|
||||
// Scale interpolation state (at puVar20 + 0x1a = offset 0x68)
|
||||
const scale_idx0: u32 = 0x68;
|
||||
const scale_idx1: u32 = 0x6C;
|
||||
const scale_t: u32 = 0x70;
|
||||
const scale_x: u32 = 0x74;
|
||||
const scale_y: u32 = 0x78;
|
||||
const scale_z: u32 = 0x7C;
|
||||
const scale2_idx0: u32 = 0x80;
|
||||
const scale2_idx1: u32 = 0x84;
|
||||
const scale2_t: u32 = 0x88;
|
||||
const scale2_x: u32 = 0x8C;
|
||||
const scale2_y: u32 = 0x90;
|
||||
const scale2_z: u32 = 0x94;
|
||||
pub const scale_idx0: u32 = 0x68;
|
||||
pub const scale_idx1: u32 = 0x6C;
|
||||
pub const scale_t: u32 = 0x70;
|
||||
pub const scale_x: u32 = 0x74;
|
||||
pub const scale_y: u32 = 0x78;
|
||||
pub const scale_z: u32 = 0x7C;
|
||||
pub const scale2_idx0: u32 = 0x80;
|
||||
pub const scale2_idx1: u32 = 0x84;
|
||||
pub const scale2_t: u32 = 0x88;
|
||||
pub const scale2_x: u32 = 0x8C;
|
||||
pub const scale2_y: u32 = 0x90;
|
||||
pub const scale2_z: u32 = 0x94;
|
||||
// Primary animation time range
|
||||
const prim_time: u32 = 0x98; // puVar20[0x26]
|
||||
const prim_track: u32 = 0x9C; // puVar20[0x27]
|
||||
const prim_anim: u32 = 0xA0; // puVar20[0x28]
|
||||
const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index
|
||||
pub const prim_time: u32 = 0x98; // puVar20[0x26]
|
||||
pub const prim_track: u32 = 0x9C; // puVar20[0x27]
|
||||
pub const prim_anim: u32 = 0xA0; // puVar20[0x28]
|
||||
pub const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index
|
||||
// Secondary animation time range (crossfade)
|
||||
const sec_start: u32 = 0xA8; // puVar20[0x2a]
|
||||
const sec_end: u32 = 0xAC; // puVar20[0x2b]
|
||||
const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion
|
||||
const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e]
|
||||
pub const sec_start: u32 = 0xA8; // puVar20[0x2a]
|
||||
pub const sec_end: u32 = 0xAC; // puVar20[0x2b]
|
||||
pub const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion
|
||||
pub const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e]
|
||||
// Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30)
|
||||
const rot_idx0: u32 = 0x30;
|
||||
const rot_idx1: u32 = 0x34;
|
||||
const rot_t: u32 = 0x38;
|
||||
const rot_x: u32 = 0x3C;
|
||||
const rot_y: u32 = 0x40;
|
||||
const rot_z: u32 = 0x44;
|
||||
const rot_w: u32 = 0x48;
|
||||
pub const rot_idx0: u32 = 0x30;
|
||||
pub const rot_idx1: u32 = 0x34;
|
||||
pub const rot_t: u32 = 0x38;
|
||||
pub const rot_x: u32 = 0x3C;
|
||||
pub const rot_y: u32 = 0x40;
|
||||
pub const rot_z: u32 = 0x44;
|
||||
pub const rot_w: u32 = 0x48;
|
||||
// Secondary rotation
|
||||
const rot2_idx0: u32 = 0x4C;
|
||||
const rot2_idx1: u32 = 0x50;
|
||||
const rot2_t: u32 = 0x54;
|
||||
const rot2_x: u32 = 0x58;
|
||||
const rot2_y: u32 = 0x5C;
|
||||
const rot2_z: u32 = 0x60;
|
||||
const rot2_w: u32 = 0x64;
|
||||
pub const rot2_idx0: u32 = 0x4C;
|
||||
pub const rot2_idx1: u32 = 0x50;
|
||||
pub const rot2_t: u32 = 0x54;
|
||||
pub const rot2_x: u32 = 0x58;
|
||||
pub const rot2_y: u32 = 0x5C;
|
||||
pub const rot2_z: u32 = 0x60;
|
||||
pub const rot2_w: u32 = 0x64;
|
||||
// Secondary time range
|
||||
const sec_time: u32 = 0xC4; // puVar20[0x31]
|
||||
const sec_track: u32 = 0xC8; // puVar20[0x32]
|
||||
const sec_slot: u32 = 0xD0; // puVar20[0x34]
|
||||
const sec_start2: u32 = 0xD4; // puVar20[0x35]
|
||||
const sec_end2: u32 = 0xD8; // puVar20[0x36]
|
||||
const sec_offset2: u32 = 0xE4; // puVar20[0x39]
|
||||
pub const sec_time: u32 = 0xC4; // puVar20[0x31]
|
||||
pub const sec_track: u32 = 0xC8; // puVar20[0x32]
|
||||
pub const sec_slot: u32 = 0xD0; // puVar20[0x34]
|
||||
pub const sec_start2: u32 = 0xD4; // puVar20[0x35]
|
||||
pub const sec_end2: u32 = 0xD8; // puVar20[0x36]
|
||||
pub const sec_offset2: u32 = 0xE4; // puVar20[0x39]
|
||||
// Flags and weights
|
||||
const flags2: u32 = 0xF4; // puVar20[0x3d]
|
||||
const crossfade_end: u32 = 0x100; // puVar20[0x40]
|
||||
const crossfade_inv: u32 = 0x104; // puVar20[0x41]
|
||||
const crossfade_weight: u32 = 0x108; // puVar20[0x42]
|
||||
const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade
|
||||
const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c]
|
||||
pub const flags2: u32 = 0xF4; // puVar20[0x3d]
|
||||
pub const crossfade_end: u32 = 0x100; // puVar20[0x40]
|
||||
pub const crossfade_inv: u32 = 0x104; // puVar20[0x41]
|
||||
pub const crossfade_weight: u32 = 0x108; // puVar20[0x42]
|
||||
pub const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade
|
||||
pub const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c]
|
||||
};
|
||||
|
||||
// OldAnimationBlock struct offsets (28 bytes = 0x1C per track in v256 M2)
|
||||
@@ -144,37 +144,37 @@ const BR = struct {
|
||||
// pMVar23->m31 (bone_def+0x34) = rot block+0x0C = nTimestamps (gates rotation)
|
||||
// pMVar23->m12 (bone_def+0x18) = trans block+0x0C = nTimestamps (gates translation)
|
||||
// pMVar23[1].m10 (bone_def+0x50) = scale block+0x0C = nTimestamps (gates scale)
|
||||
const AD = struct {
|
||||
const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp)
|
||||
const time_index: u32 = 0x02; // i16: global sequence index (-1 = none)
|
||||
const track_count_flag: u32 = 0x04; // nRanges: 0 = single track
|
||||
const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs
|
||||
const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count
|
||||
const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array
|
||||
const nvalues: u32 = 0x14; // nValues: number of value entries
|
||||
const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data
|
||||
pub const AD = struct {
|
||||
pub const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp)
|
||||
pub const time_index: u32 = 0x02; // i16: global sequence index (-1 = none)
|
||||
pub const track_count_flag: u32 = 0x04; // nRanges: 0 = single track
|
||||
pub const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs
|
||||
pub const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count
|
||||
pub const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array
|
||||
pub const nvalues: u32 = 0x14; // nValues: number of value entries
|
||||
pub const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data
|
||||
};
|
||||
|
||||
// M2CompBone struct offsets (0x6C = 108 bytes per bone in v256 model)
|
||||
// Layout: 12 bytes fixed header + 3x28 byte OldAnimationBlock tracks + 12 bytes pivot
|
||||
// Track order: translation, rotation, scale (standard M2 order)
|
||||
const BD = struct {
|
||||
const key_id: u32 = 0x00; // i32: key bone ID
|
||||
const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.)
|
||||
const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes
|
||||
pub const BD = struct {
|
||||
pub const key_id: u32 = 0x00; // i32: key bone ID
|
||||
pub const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.)
|
||||
pub const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes
|
||||
// Translation OldAnimationBlock (28 bytes, +0x0C to +0x27)
|
||||
const trans_anim: u32 = 0x0C;
|
||||
const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation
|
||||
pub const trans_anim: u32 = 0x0C;
|
||||
pub const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation
|
||||
// Rotation OldAnimationBlock (28 bytes, +0x28 to +0x43)
|
||||
const rot_anim: u32 = 0x28;
|
||||
const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation
|
||||
pub const rot_anim: u32 = 0x28;
|
||||
pub const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation
|
||||
// Scale OldAnimationBlock (28 bytes, +0x44 to +0x5F)
|
||||
const scale_anim: u32 = 0x44;
|
||||
const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation
|
||||
pub const scale_anim: u32 = 0x44;
|
||||
pub const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation
|
||||
// Pivot point (12 bytes, +0x60 to +0x6B)
|
||||
const pivot_x: u32 = 0x60;
|
||||
const pivot_y: u32 = 0x64;
|
||||
const pivot_z: u32 = 0x68;
|
||||
pub const pivot_x: u32 = 0x60;
|
||||
pub const pivot_y: u32 = 0x64;
|
||||
pub const pivot_z: u32 = 0x68;
|
||||
};
|
||||
|
||||
// Game constants
|
||||
@@ -182,12 +182,12 @@ const ZERO_F: f32 = 0.0;
|
||||
const ONE_F: f32 = 1.0;
|
||||
const THREE_F: f32 = 3.0;
|
||||
// getBillboardEpsilon(): read from game memory (runtime 0x34800000, NOT static 0x3727c5ac from Ghidra)
|
||||
fn getBillboardEpsilon() f32 {
|
||||
pub fn getBillboardEpsilon() f32 {
|
||||
return rf32(0x008029d4);
|
||||
}
|
||||
// getShortToFloat(): read from game memory at 0x00811610 (runtime value is 0x38000100 = 1/32767,
|
||||
// NOT the static 0x38000000 = 1/32768 from Ghidra). The game patches this at startup.
|
||||
fn getShortToFloat() f32 {
|
||||
pub fn getShortToFloat() f32 {
|
||||
return rf32(0x00811610);
|
||||
}
|
||||
// MSVC CRT sin/cos — linked from the WoW process
|
||||
@@ -202,42 +202,42 @@ const OrigTransformFn = *const fn (u32, u32, u32, u32, u32) callconv(.c) void;
|
||||
// Memory access helpers
|
||||
// =============================================================================
|
||||
|
||||
inline fn ru32(addr: u32) u32 {
|
||||
pub inline fn ru32(addr: u32) u32 {
|
||||
return @as(*const u32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
inline fn ri32(addr: u32) i32 {
|
||||
pub inline fn ri32(addr: u32) i32 {
|
||||
return @as(*const i32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
inline fn rf32(addr: u32) f32 {
|
||||
pub inline fn rf32(addr: u32) f32 {
|
||||
return @as(*const f32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
inline fn ru16(addr: u32) u16 {
|
||||
pub inline fn ru16(addr: u32) u16 {
|
||||
return @as(*align(1) const u16, @ptrFromInt(addr)).*;
|
||||
}
|
||||
inline fn ri16(addr: u32) i16 {
|
||||
pub inline fn ri16(addr: u32) i16 {
|
||||
return @as(*align(1) const i16, @ptrFromInt(addr)).*;
|
||||
}
|
||||
inline fn ru8(addr: u32) u8 {
|
||||
pub inline fn ru8(addr: u32) u8 {
|
||||
return @as(*const u8, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
inline fn wu32(addr: u32, v: u32) void {
|
||||
pub inline fn wu32(addr: u32, v: u32) void {
|
||||
@as(*u32, @ptrFromInt(addr)).* = v;
|
||||
}
|
||||
inline fn wf32(addr: u32, v: f32) void {
|
||||
pub inline fn wf32(addr: u32, v: f32) void {
|
||||
@as(*f32, @ptrFromInt(addr)).* = v;
|
||||
}
|
||||
inline fn wu16(addr: u32, v: u16) void {
|
||||
pub inline fn wu16(addr: u32, v: u16) void {
|
||||
@as(*align(1) u16, @ptrFromInt(addr)).* = v;
|
||||
}
|
||||
inline fn wu8(addr: u32, v: u8) void {
|
||||
pub inline fn wu8(addr: u32, v: u8) void {
|
||||
@as(*u8, @ptrFromInt(addr)).* = v;
|
||||
}
|
||||
|
||||
inline fn fbits(v: f32) u32 {
|
||||
pub inline fn fbits(v: f32) u32 {
|
||||
return @bitCast(v);
|
||||
}
|
||||
inline fn ufloat(v: u32) f32 {
|
||||
pub inline fn ufloat(v: u32) f32 {
|
||||
return @bitCast(v);
|
||||
}
|
||||
|
||||
@@ -245,12 +245,12 @@ inline fn ufloat(v: u32) f32 {
|
||||
// Math helpers — using @Vector(4, f32) for SSE
|
||||
// =============================================================================
|
||||
|
||||
inline fn splat(v: f32) V4 {
|
||||
pub inline fn splat(v: f32) V4 {
|
||||
return @splat(v);
|
||||
}
|
||||
|
||||
/// 3-component lerp: a + (b - a) * t. Uses @mulAdd → vfmadd.
|
||||
inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
|
||||
pub inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
|
||||
return .{
|
||||
@mulAdd(f32, rf32(b_addr) - rf32(a_addr), t, rf32(a_addr)),
|
||||
@mulAdd(f32, rf32(b_addr + 4) - rf32(a_addr + 4), t, rf32(a_addr + 4)),
|
||||
@@ -260,7 +260,7 @@ inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
|
||||
|
||||
/// Scale 3x3 rotation portion of a row-major 4x4 matrix by per-axis scale.
|
||||
/// Row 0 *= scale.x, Row 1 *= scale.y, Row 2 *= scale.z
|
||||
inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
|
||||
pub inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
|
||||
// Row 0 (offsets 0x00, 0x04, 0x08)
|
||||
wf32(mat + 0x00, rf32(mat + 0x00) * sx);
|
||||
wf32(mat + 0x04, rf32(mat + 0x04) * sx);
|
||||
@@ -279,15 +279,15 @@ inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
|
||||
/// mat[3][0] += dot(mat[0], t)
|
||||
/// mat[3][1] += dot(mat[1], t)
|
||||
/// mat[3][2] += dot(mat[2], t)
|
||||
/// Uses @mulAdd chain → vfmadd for each dot product component.
|
||||
inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void {
|
||||
/// Uses @mulAdd chain for each dot product component.
|
||||
pub inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void {
|
||||
wf32(mat + 0x30, @mulAdd(f32, tz, rf32(mat + 0x20), @mulAdd(f32, ty, rf32(mat + 0x10), @mulAdd(f32, tx, rf32(mat + 0x00), rf32(mat + 0x30)))));
|
||||
wf32(mat + 0x34, @mulAdd(f32, tz, rf32(mat + 0x24), @mulAdd(f32, ty, rf32(mat + 0x14), @mulAdd(f32, tx, rf32(mat + 0x04), rf32(mat + 0x34)))));
|
||||
wf32(mat + 0x38, @mulAdd(f32, tz, rf32(mat + 0x28), @mulAdd(f32, ty, rf32(mat + 0x18), @mulAdd(f32, tx, rf32(mat + 0x08), rf32(mat + 0x38)))));
|
||||
}
|
||||
|
||||
/// Quaternion → rotation matrix as value. No memory writes.
|
||||
inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 {
|
||||
pub inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 {
|
||||
const xx2 = qx * (qx + qx);
|
||||
const xy2 = qx * (qy + qy);
|
||||
const xz2 = qx * (qz + qz);
|
||||
@@ -308,7 +308,7 @@ inline fn buildRotationMatrixVal(qx: f32, qy: f32, qz: f32, qw: f32) [16]f32 {
|
||||
|
||||
/// Quaternion → rotation matrix: writes to game memory via u32 address.
|
||||
/// Used by boneKeyframeLoop where the matrix is in game memory.
|
||||
inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||||
pub inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||||
const m = buildRotationMatrixVal(qx, qy, qz, qw);
|
||||
inline for (0..16) |i| {
|
||||
wf32(mat + @as(u32, @intCast(i * 4)), m[i]);
|
||||
@@ -316,7 +316,7 @@ inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void
|
||||
}
|
||||
|
||||
/// Quaternion → rotation matrix × mat. Fused: builds quat rows as V4, multiplies in-register.
|
||||
inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||||
pub inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||||
const xx2 = qx * (qx + qx);
|
||||
const xy2 = qx * (qy + qy);
|
||||
const xz2 = qx * (qz + qz);
|
||||
@@ -354,7 +354,7 @@ inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void
|
||||
}
|
||||
|
||||
/// Copy 4x4 matrix (64 bytes) — 4 V4 loads/stores instead of 16 scalar copies.
|
||||
inline fn copyMat4(dst: u32, src: u32) void {
|
||||
pub inline fn copyMat4(dst: u32, src: u32) void {
|
||||
inline for (0..4) |i| {
|
||||
const off: u32 = @intCast(i * 16);
|
||||
const row = V4{ rf32(src + off), rf32(src + off + 4), rf32(src + off + 8), rf32(src + off + 12) };
|
||||
@@ -367,19 +367,16 @@ inline fn copyMat4(dst: u32, src: u32) void {
|
||||
|
||||
/// 4x4 matrix multiply: dst = a * b (row-major). Safe for dst==a or dst==b.
|
||||
/// Uses V4 + @mulAdd (FMA): 1 mul + 3 FMA per row = 16 SIMD ops total.
|
||||
inline fn matMul4x4(dst: u32, a: u32, b: u32) void {
|
||||
// Pre-load all rows of B
|
||||
pub inline fn matMul4x4(dst: u32, a: u32, b: u32) void {
|
||||
const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) };
|
||||
const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) };
|
||||
const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) };
|
||||
const b3 = V4{ rf32(b + 48), rf32(b + 52), rf32(b + 56), rf32(b + 60) };
|
||||
// Pre-load all rows of A (in case dst aliases a)
|
||||
const a0 = V4{ rf32(a), rf32(a + 4), rf32(a + 8), rf32(a + 12) };
|
||||
const a1 = V4{ rf32(a + 16), rf32(a + 20), rf32(a + 24), rf32(a + 28) };
|
||||
const a2 = V4{ rf32(a + 32), rf32(a + 36), rf32(a + 40), rf32(a + 44) };
|
||||
const a3 = V4{ rf32(a + 48), rf32(a + 52), rf32(a + 56), rf32(a + 60) };
|
||||
const rows = [4]V4{ a0, a1, a2, a3 };
|
||||
// Compute: each output row = broadcast(a[row][col]) * b_row, accumulated with FMA
|
||||
inline for (0..4) |i| {
|
||||
const s0: V4 = @splat(rows[i][0]);
|
||||
const s1: V4 = @splat(rows[i][1]);
|
||||
@@ -395,7 +392,7 @@ inline fn matMul4x4(dst: u32, a: u32, b: u32) void {
|
||||
}
|
||||
|
||||
/// 4x4 matrix multiply: dst = a * b. Left operand is a local array, right is game memory.
|
||||
inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void {
|
||||
pub inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void {
|
||||
const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) };
|
||||
const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) };
|
||||
const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) };
|
||||
@@ -421,7 +418,7 @@ inline fn matMul4x4Local(dst: u32, a: [16]f32, b: u32) void {
|
||||
}
|
||||
|
||||
/// In-place multiply: a = a * b (b from game memory). Returns new array.
|
||||
inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 {
|
||||
pub inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 {
|
||||
const b0 = V4{ rf32(b), rf32(b + 4), rf32(b + 8), rf32(b + 12) };
|
||||
const b1 = V4{ rf32(b + 16), rf32(b + 20), rf32(b + 24), rf32(b + 28) };
|
||||
const b2 = V4{ rf32(b + 32), rf32(b + 36), rf32(b + 40), rf32(b + 44) };
|
||||
@@ -448,7 +445,7 @@ inline fn matMul4x4InPlace(a: [16]f32, b: u32) [16]f32 {
|
||||
}
|
||||
|
||||
/// Set identity matrix (16 floats)
|
||||
inline fn setIdentity(dst: u32) void {
|
||||
pub inline fn setIdentity(dst: u32) void {
|
||||
inline for (0..16) |i| {
|
||||
const val: f32 = if (i == 0 or i == 5 or i == 10 or i == 15) 1.0 else 0.0;
|
||||
wf32(dst + @as(u32, @intCast(i)) * 4, val);
|
||||
@@ -458,7 +455,7 @@ inline fn setIdentity(dst: u32) void {
|
||||
/// Normalize a 3-component vector in memory at addr.
|
||||
/// Calls game's vec3 squared magnitude (0x4549F0), then sqrt, epsilon check, divide.
|
||||
/// Assembly pattern: CALL 0x4549F0 → FSQRT → FABS → FCOMP → FLD1 → FDIVRP → FMUL×3
|
||||
inline fn normalizeVec3InPlace(addr: u32) void {
|
||||
pub inline fn normalizeVec3InPlace(addr: u32) void {
|
||||
const sq_mag = callVec3SqMag(addr);
|
||||
const len = @sqrt(sq_mag);
|
||||
if (@abs(len) >= getBillboardEpsilon()) {
|
||||
@@ -471,7 +468,7 @@ inline fn normalizeVec3InPlace(addr: u32) void {
|
||||
|
||||
/// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small.
|
||||
/// Writes vec3 to stack local and calls game's vec3 squared magnitude (0x4549F0).
|
||||
inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
|
||||
pub inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
|
||||
var v: [3]f32 = .{ x, y, z };
|
||||
const sq_mag = callVec3SqMag(@intFromPtr(&v));
|
||||
const len = @sqrt(sq_mag);
|
||||
@@ -481,7 +478,7 @@ inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
|
||||
}
|
||||
|
||||
/// Cross product of two 3-component vectors
|
||||
inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 {
|
||||
pub inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 {
|
||||
return .{
|
||||
ay * bz - az * by,
|
||||
az * bx - ax * bz,
|
||||
@@ -502,7 +499,7 @@ inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32
|
||||
|
||||
/// IsParticleBufferEmpty (0x7B5F60) — recursive tree check.
|
||||
/// Returns true if any node in the tree has active particles (this->0x64 != 0).
|
||||
fn isParticleBufferNotEmpty(ptr: u32) bool {
|
||||
pub fn isParticleBufferNotEmpty(ptr: u32) bool {
|
||||
if (ru32(ptr + 0x64) != 0) return true;
|
||||
const count = ru32(ptr + 0x7C);
|
||||
if (count == 0) return false;
|
||||
@@ -514,7 +511,7 @@ fn isParticleBufferNotEmpty(ptr: u32) bool {
|
||||
return false;
|
||||
}
|
||||
|
||||
const InterpResult = struct {
|
||||
pub const InterpResult = struct {
|
||||
idx0: u32,
|
||||
idx1: u32,
|
||||
t: f32,
|
||||
@@ -523,7 +520,7 @@ const InterpResult = struct {
|
||||
/// Check if two AnimData tracks share the same temporal structure,
|
||||
/// meaning findInterpIdx would produce identical (idx0, idx1, t) for both.
|
||||
/// Both must use prim_time (time_index == -1) and have matching range/timestamp layout.
|
||||
inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
|
||||
pub inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
|
||||
if (ri16(ref_anim + AD.time_index) != -1) return false;
|
||||
if (ri16(other_anim + AD.time_index) != -1) return false;
|
||||
return ru32(ref_anim + AD.track_count_flag) == ru32(other_anim + AD.track_count_flag) and
|
||||
@@ -533,7 +530,7 @@ inline fn canReuseInterp(ref_anim: u32, other_anim: u32) bool {
|
||||
}
|
||||
|
||||
/// Write temporal coherence cache for a reused result so next frame's forward scan starts right.
|
||||
inline fn applyCachedResult(cached: InterpResult, output: u32) void {
|
||||
pub inline fn applyCachedResult(cached: InterpResult, output: u32) void {
|
||||
wu32(output, cached.idx0);
|
||||
}
|
||||
|
||||
@@ -541,7 +538,7 @@ inline fn applyCachedResult(cached: InterpResult, output: u32) void {
|
||||
/// Reimplementation of game function at 0x713D50 (334 bytes).
|
||||
/// Assembly-verified against t44_helpers_asm.txt.
|
||||
/// Returns indices and t in registers; only writes output[0] for next-frame cache persistence.
|
||||
inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult {
|
||||
pub inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_data: u32, output: u32) InterpResult {
|
||||
const n_ranges = ru32(anim_data + AD.track_count_flag);
|
||||
|
||||
// Range selection: [start, last] not [start, count]
|
||||
@@ -659,13 +656,13 @@ inline fn findInterpIdx(this: u32, search_value: u32, track_index: u32, anim_dat
|
||||
|
||||
/// Quaternion keyframe interpolation — replaces game's 0x713EA0.
|
||||
/// Assembly-verified: stride 16 (SHL EAX,4), values are 4×float, not CompQuat.
|
||||
inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 {
|
||||
pub inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) [4]f32 {
|
||||
return interpAnimKFCached(this, bone_rt, anim_data, output, null);
|
||||
}
|
||||
|
||||
/// Quaternion keyframe interpolation with optional cached primary InterpResult.
|
||||
/// When cached_primary is non-null, skips findInterpIdx and uses the cached indices/t.
|
||||
inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 {
|
||||
pub inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u32, cached_primary: ?InterpResult) [4]f32 {
|
||||
const r = if (cached_primary) |c| blk: {
|
||||
applyCachedResult(c, output);
|
||||
break :blk c;
|
||||
@@ -716,7 +713,7 @@ inline fn interpAnimKFCached(this: u32, bone_rt: u32, anim_data: u32, output: u3
|
||||
|
||||
/// Fast modulo for looping animations. The value is almost always < 2*length
|
||||
/// (frame-to-frame delta is small), so a conditional subtract beats idiv.
|
||||
inline fn fastMod(val: u32, len: u32) u32 {
|
||||
pub inline fn fastMod(val: u32, len: u32) u32 {
|
||||
var v = val;
|
||||
if (v >= len) {
|
||||
v -%= len;
|
||||
@@ -727,12 +724,12 @@ inline fn fastMod(val: u32, len: u32) u32 {
|
||||
|
||||
/// Float truncation — replaces game's __ftol at 0x40A2B0.
|
||||
/// Original: FILD i32 → FMUL f32 → __ftol, all in 80-bit x87 precision.
|
||||
inline fn callFtol(delta: i32, scale_addr: u32) i32 {
|
||||
pub inline fn callFtol(delta: i32, scale_addr: u32) i32 {
|
||||
return @intFromFloat(@as(f32, @floatFromInt(delta)) * rf32(scale_addr));
|
||||
}
|
||||
|
||||
/// Vec3 squared magnitude — replaces game's 0x4549F0. Uses @mulAdd → vfmadd.
|
||||
inline fn callVec3SqMag(vec3_ptr: u32) f32 {
|
||||
pub inline fn callVec3SqMag(vec3_ptr: u32) f32 {
|
||||
const x = rf32(vec3_ptr);
|
||||
const y = rf32(vec3_ptr + 4);
|
||||
const z = rf32(vec3_ptr + 8);
|
||||
@@ -741,7 +738,7 @@ inline fn callVec3SqMag(vec3_ptr: u32) f32 {
|
||||
|
||||
/// Read i16 at keyframe index. Replaces game's getIndexOffset (0x71AFF0) + setShortValue (0x71B010).
|
||||
/// getIndexOffset returns table[4] + index*2, setShortValue copies a word. Direct read is equivalent.
|
||||
inline fn readShortViaGame(table: u32, index: u32) i16 {
|
||||
pub inline fn readShortViaGame(table: u32, index: u32) i16 {
|
||||
const values_ptr = ru32(table + 4);
|
||||
return ri16(values_ptr + index * 2);
|
||||
}
|
||||
@@ -749,7 +746,7 @@ inline fn readShortViaGame(table: u32, index: u32) i16 {
|
||||
/// Interpolate a Vec3 track (12 bytes per keyframe) with crossfade support.
|
||||
/// Writes result to output[3..5] (as u32 float bits). Uses output[0..2] for indices/t,
|
||||
/// and output[6..11] for secondary crossfade state.
|
||||
inline fn interpVec3Track(
|
||||
pub inline fn interpVec3Track(
|
||||
this: u32,
|
||||
bone_rt: u32,
|
||||
anim_data: u32,
|
||||
@@ -760,7 +757,7 @@ inline fn interpVec3Track(
|
||||
}
|
||||
|
||||
/// Vec3 keyframe interpolation with optional cached primary InterpResult.
|
||||
inline fn interpVec3TrackCached(
|
||||
pub inline fn interpVec3TrackCached(
|
||||
this: u32,
|
||||
bone_rt: u32,
|
||||
anim_data: u32,
|
||||
@@ -806,7 +803,7 @@ inline fn interpVec3TrackCached(
|
||||
|
||||
/// Interpolate a single float track (4 bytes per keyframe) with crossfade.
|
||||
/// Writes result to output[3] as float bits.
|
||||
inline fn interpFloatTrack(
|
||||
pub inline fn interpFloatTrack(
|
||||
this: u32,
|
||||
bone_rt: u32,
|
||||
anim_data: u32,
|
||||
@@ -846,7 +843,7 @@ inline fn interpFloatTrack(
|
||||
// Hermite/Bezier basis + particle emitter interp helpers
|
||||
// =============================================================================
|
||||
|
||||
inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
|
||||
pub inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
|
||||
const t2 = t * t;
|
||||
const t3 = t2 * t;
|
||||
return .{
|
||||
@@ -857,7 +854,7 @@ inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
|
||||
};
|
||||
}
|
||||
|
||||
inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
|
||||
pub inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
|
||||
const u = 1.0 - t;
|
||||
const t2 = t * t;
|
||||
const u_sq = u * u;
|
||||
@@ -869,7 +866,7 @@ inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
|
||||
};
|
||||
}
|
||||
|
||||
inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
|
||||
pub inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
|
||||
const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output);
|
||||
|
||||
const mode = ri16(anim_data + AD.interp_mode);
|
||||
@@ -950,7 +947,7 @@ inline fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output
|
||||
}
|
||||
}
|
||||
|
||||
inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
|
||||
pub inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
|
||||
const r = findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output);
|
||||
|
||||
const mode = ri16(anim_data + AD.interp_mode);
|
||||
@@ -1007,7 +1004,7 @@ inline fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, outpu
|
||||
// Same as interpFloatTrack but uses the bone_rt directly (different register mapping)
|
||||
// =============================================================================
|
||||
|
||||
inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void {
|
||||
pub inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void {
|
||||
const r = findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data_short_ptr, output);
|
||||
|
||||
const interp_mode = ri16(anim_data_short_ptr);
|
||||
@@ -1040,7 +1037,7 @@ inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr
|
||||
// applies inverse translation.
|
||||
// =============================================================================
|
||||
|
||||
fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
|
||||
pub fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
|
||||
// Simple transpose for unit scale
|
||||
if (@abs(scale - 1.0) < @as(f32, @bitCast(@as(u32, 0x35800000)))) {
|
||||
// Transpose 3x3
|
||||
@@ -1472,68 +1469,60 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
const pivot_y = rf32(bdef + BD.pivot_y);
|
||||
const pivot_z = rf32(bdef + BD.pivot_z);
|
||||
|
||||
// Compute translated position
|
||||
const tx = local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z + local_mat[12];
|
||||
const ty = local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z + local_mat[13];
|
||||
const tz = local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z + local_mat[14];
|
||||
// Compute translated position — accumulation order must match
|
||||
// original x87. Row 0 uses (pz + px + py), rows 1/2 use (pz + py + px).
|
||||
const tx = local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[12];
|
||||
const ty = local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x + local_mat[13];
|
||||
const tz = local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x + local_mat[14];
|
||||
|
||||
const bb_type = combined_flags & 6;
|
||||
const billboard_eps_f64: f64 = @floatCast(rf32(0x008029d4));
|
||||
const cull_eps_f64: f64 = @floatCast(rf32(0x0080c5c8));
|
||||
if (bb_type == 2) {
|
||||
// Cylindrical billboard — normalize each column
|
||||
const n0 = normalizeVec3(local_mat[0], local_mat[1], local_mat[2]);
|
||||
local_mat[0] = n0[0];
|
||||
local_mat[1] = n0[1];
|
||||
local_mat[2] = n0[2];
|
||||
const n1 = normalizeVec3(local_mat[4], local_mat[5], local_mat[6]);
|
||||
local_mat[4] = n1[0];
|
||||
local_mat[5] = n1[1];
|
||||
local_mat[6] = n1[2];
|
||||
const n2 = normalizeVec3(local_mat[8], local_mat[9], local_mat[10]);
|
||||
local_mat[8] = n2[0];
|
||||
local_mat[9] = n2[1];
|
||||
local_mat[10] = n2[2];
|
||||
// Cylindrical billboard — normalize each column in f64 to
|
||||
// match x87's extended-precision 1/sqrt. Previous f32 impl
|
||||
// drifted from x87 by a ULP per axis, causing particle
|
||||
// emitter orientation to jitter on camera motion and
|
||||
// flicker against ground effects.
|
||||
inline for ([_]u32{ 0, 4, 8 }) |row_off| {
|
||||
const cx: f64 = @floatCast(local_mat[row_off]);
|
||||
const cy: f64 = @floatCast(local_mat[row_off + 1]);
|
||||
const cz: f64 = @floatCast(local_mat[row_off + 2]);
|
||||
const len_sq = cx * cx + cy * cy + cz * cz;
|
||||
const len = @sqrt(len_sq);
|
||||
if (len >= billboard_eps_f64) {
|
||||
const inv = 1.0 / len;
|
||||
local_mat[row_off] = @floatCast(cx * inv);
|
||||
local_mat[row_off + 1] = @floatCast(cy * inv);
|
||||
local_mat[row_off + 2] = @floatCast(cz * inv);
|
||||
}
|
||||
}
|
||||
} else if (bb_type == 4) {
|
||||
// Spherical billboard — inherit camera rotation with scale preservation
|
||||
// All sqmag computations MUST call game's vec3SqMag (0x4549F0)
|
||||
const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) };
|
||||
const cam_len_sq0 = callVec3SqMag(this + SO.bb_row0);
|
||||
var s0: f32 = 1.0;
|
||||
if (cam_len_sq0 > rf32(0x0080c5c8)) {
|
||||
var tmp0 = [3]f32{ local_mat[0], local_mat[1], local_mat[2] };
|
||||
const mat_len_sq0 = callVec3SqMag(@intFromPtr(&tmp0));
|
||||
s0 = @sqrt(mat_len_sq0 / cam_len_sq0);
|
||||
// Spherical billboard — inherit camera basis, rescale to
|
||||
// preserve each column's length. All intermediates in f64
|
||||
// to match x87's 80-bit temporaries.
|
||||
inline for ([_]struct { row_off: u32, src_off: u32 }{
|
||||
.{ .row_off = 0, .src_off = SO.bb_row0 },
|
||||
.{ .row_off = 4, .src_off = SO.world_xform },
|
||||
.{ .row_off = 8, .src_off = SO.world_xform + 16 },
|
||||
}) |p| {
|
||||
const src_addr = this + p.src_off;
|
||||
const cam_x: f64 = @floatCast(rf32(src_addr));
|
||||
const cam_y: f64 = @floatCast(rf32(src_addr + 4));
|
||||
const cam_z: f64 = @floatCast(rf32(src_addr + 8));
|
||||
const cam_len_sq = cam_x * cam_x + cam_y * cam_y + cam_z * cam_z;
|
||||
var s: f64 = 1.0;
|
||||
if (cam_len_sq > cull_eps_f64) {
|
||||
const mx: f64 = @floatCast(local_mat[p.row_off]);
|
||||
const my: f64 = @floatCast(local_mat[p.row_off + 1]);
|
||||
const mz: f64 = @floatCast(local_mat[p.row_off + 2]);
|
||||
const mat_len_sq = mx * mx + my * my + mz * mz;
|
||||
s = @sqrt(mat_len_sq / cam_len_sq);
|
||||
}
|
||||
local_mat[p.row_off] = @floatCast(s * cam_x);
|
||||
local_mat[p.row_off + 1] = @floatCast(s * cam_y);
|
||||
local_mat[p.row_off + 2] = @floatCast(s * cam_z);
|
||||
}
|
||||
local_mat[0] = s0 * cam0[0];
|
||||
local_mat[1] = s0 * cam0[1];
|
||||
local_mat[2] = s0 * cam0[2];
|
||||
|
||||
const wt0 = rf32(this + SO.world_xform + 0 * 4);
|
||||
const wt1 = rf32(this + SO.world_xform + 1 * 4);
|
||||
const wt2 = rf32(this + SO.world_xform + 2 * 4);
|
||||
const wt_len_sq = callVec3SqMag(this + SO.world_xform);
|
||||
var s1: f32 = 1.0;
|
||||
if (wt_len_sq > rf32(0x0080c5c8)) {
|
||||
var tmp1 = [3]f32{ local_mat[4], local_mat[5], local_mat[6] };
|
||||
const mat_len_sq1 = callVec3SqMag(@intFromPtr(&tmp1));
|
||||
s1 = @sqrt(mat_len_sq1 / wt_len_sq);
|
||||
}
|
||||
local_mat[4] = s1 * wt0;
|
||||
local_mat[5] = s1 * wt1;
|
||||
local_mat[6] = s1 * wt2;
|
||||
|
||||
const wt4 = rf32(this + SO.world_xform + 4 * 4);
|
||||
const wt5 = rf32(this + SO.world_xform + 5 * 4);
|
||||
const wt6 = rf32(this + SO.world_xform + 6 * 4);
|
||||
const wt_len_sq2 = callVec3SqMag(this + SO.world_xform + 16);
|
||||
var s2: f32 = 1.0;
|
||||
if (wt_len_sq2 > rf32(0x0080c5c8)) {
|
||||
var tmp2 = [3]f32{ local_mat[8], local_mat[9], local_mat[10] };
|
||||
const mat_len_sq2 = callVec3SqMag(@intFromPtr(&tmp2));
|
||||
s2 = @sqrt(mat_len_sq2 / wt_len_sq2);
|
||||
}
|
||||
local_mat[8] = s2 * wt4;
|
||||
local_mat[9] = s2 * wt5;
|
||||
local_mat[10] = s2 * wt6;
|
||||
} else if (bb_type == 6) {
|
||||
// Full billboard — copy camera rotation directly
|
||||
local_mat[0] = rf32(this + SO.bb_row0);
|
||||
@@ -1547,11 +1536,12 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
local_mat[10] = rf32(this + SO.world_xform + 6 * 4);
|
||||
}
|
||||
|
||||
// Recompute translation: pos - rot * pivot
|
||||
// Recompute translation: pos - rot * pivot.
|
||||
// Accumulation order mirrors the tx/ty/tz computation above.
|
||||
if ((combined_flags & 1) == 0) {
|
||||
local_mat[12] = tx - (local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z);
|
||||
local_mat[13] = ty - (local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z);
|
||||
local_mat[14] = tz - (local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z);
|
||||
local_mat[12] = tx - (local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y);
|
||||
local_mat[13] = ty - (local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x);
|
||||
local_mat[14] = tz - (local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x);
|
||||
} else {
|
||||
local_mat[12] = rf32(this + SO.world_xform + 8 * 4);
|
||||
local_mat[13] = rf32(this + SO.world_xform + 9 * 4);
|
||||
@@ -1671,9 +1661,13 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
const bpx = rf32(bdef + BD.pivot_x);
|
||||
const bpy = rf32(bdef + BD.pivot_y);
|
||||
const bpz = rf32(bdef + BD.pivot_z);
|
||||
// Accumulation order mirrors original x87:
|
||||
// pos_x: px + py + pz + const
|
||||
// pos_y: py + pz + px + const
|
||||
// pos_z: py + pz + px + const
|
||||
const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30);
|
||||
const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34);
|
||||
const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38);
|
||||
const pos_y = bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + bpx * rf32(om + 0x04) + rf32(om + 0x34);
|
||||
const pos_z = bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + bpx * rf32(om + 0x08) + rf32(om + 0x38);
|
||||
|
||||
// Switch on billboard post-processing type
|
||||
const bb_post = combined_flags & 0x78;
|
||||
@@ -1798,10 +1792,14 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
const r2z_s = rf32(om + 0x28);
|
||||
wf32(om + 0x28, scale_len2 * r2z_s);
|
||||
|
||||
// Recompute translation: pos - scaled_matrix * pivot
|
||||
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz));
|
||||
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz));
|
||||
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz));
|
||||
// Recompute translation: pos - scaled_matrix * pivot.
|
||||
// Accumulation order must match original x87: row0 + row2 + row1
|
||||
// (pivot_x, then pivot_z, then pivot_y). f32 addition isn't associative —
|
||||
// this ordering matters for matching terrain-pipeline precision and
|
||||
// avoiding z-fighting on ground-aligned billboard spell effects.
|
||||
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len2 * r2x_s * bpz + scale_len1 * r1x_s * bpy));
|
||||
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len2 * r2y_s * bpz + scale_len1 * r1y_s * bpy));
|
||||
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len2 * r2z_s * bpz + scale_len1 * r1z_s * bpy));
|
||||
wf32(om + 0x3C, 1.0);
|
||||
}
|
||||
}
|
||||
@@ -1818,8 +1816,6 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
// findInterpIdx + lerp + crossfade blend.
|
||||
// =========================================================================
|
||||
|
||||
// BISECT: stop after section 7 (bone loop)
|
||||
|
||||
// Section 8: Texture animation loop
|
||||
texAnimLoop(this, model_hdr, frame_ctr);
|
||||
colorAnimLoop(this, model_hdr, frame_ctr);
|
||||
@@ -1841,7 +1837,7 @@ pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32)
|
||||
// Post-bone-loop sections (extracted for readability)
|
||||
// =============================================================================
|
||||
|
||||
fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
const count = ru32(model_hdr + 0x54);
|
||||
if (count == 0) return;
|
||||
@@ -1891,7 +1887,7 @@ fn texAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
|
||||
/// Short-value interpolation: uses InterpResult indices, looks up short values, interpolates.
|
||||
/// Shared by texAnimLoop alpha, colorAnimLoop, and word animation crossfade.
|
||||
inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 {
|
||||
pub inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 {
|
||||
const mode = ri16(anim_data);
|
||||
const table = anim_data + AD.nvalues;
|
||||
if (mode == 0) {
|
||||
@@ -1903,7 +1899,7 @@ inline fn shortInterpToFloat(anim_data: u32, r: InterpResult, stf: f32) f32 {
|
||||
}
|
||||
}
|
||||
|
||||
fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
// Assembly: model_hdr+0x64 is both entry gate AND loop count
|
||||
const count = ru32(model_hdr + 0x64);
|
||||
@@ -1946,7 +1942,7 @@ fn colorAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
// Assembly 0x715E46-0x715F25: word/byte animation section
|
||||
// model_hdr+0x6C = count, model_hdr+0x70 = data base
|
||||
@@ -1989,7 +1985,7 @@ fn wordAnimLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
|
||||
pub fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
|
||||
const count = ru32(model_hdr + 0x74);
|
||||
if (count == 0) return;
|
||||
|
||||
@@ -2050,7 +2046,7 @@ fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
// Particle emitters are the largest section (~1000 lines of decompiled C).
|
||||
// They follow the same interpolation patterns but with many sub-tracks per emitter.
|
||||
// For the initial implementation, we handle the key tracks (position, speed, scale).
|
||||
@@ -2066,7 +2062,7 @@ fn particleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
additionalParticleLoops(this, model_hdr, frame_ctr);
|
||||
}
|
||||
|
||||
fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
const count = ru32(model_hdr + 0x11C);
|
||||
if (count == 0) return;
|
||||
@@ -2136,7 +2132,7 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
const count = ru32(model_hdr + 0x124);
|
||||
if (count == 0) return;
|
||||
@@ -2169,7 +2165,7 @@ fn particleEmitterLoop(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
pub fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
const stf = getShortToFloat();
|
||||
// Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A)
|
||||
@@ -2377,7 +2373,7 @@ fn additionalParticleLoops(this: u32, model_hdr: u32, frame_ctr: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void {
|
||||
pub fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
const hierarchy = ru32(this + SO.hierarchy_ptr);
|
||||
if (hierarchy == 0) return;
|
||||
|
||||
@@ -0,0 +1,913 @@
|
||||
//! f64-intermediate port of transformMatrix4x4 (0x714260).
|
||||
//!
|
||||
//! Mirrors bone_sse.zig but holds intermediate bone matrices as [16]f64 and
|
||||
//! runs all matrix-math operations in double precision, only narrowing to f32
|
||||
//! when writing into the bone output buffer in game memory.
|
||||
//!
|
||||
//! Rounding profile matches the original x87 implementation: load f32, compute
|
||||
//! in extended precision (here 53-bit mantissa f64 vs x87 64-bit mantissa —
|
||||
//! indistinguishable once truncated to f32), store f32 once at the end.
|
||||
//!
|
||||
//! Interpolation, keyframe search, billboard math, ftol, and game-callback
|
||||
//! helpers are imported from bone_sse since they produce f32 scalars that
|
||||
//! widen implicitly when multiplied with f64 matrices.
|
||||
//!
|
||||
//! Compiled ReleaseFast with AVX enabled. Uses @Vector(4, f64) for SIMD
|
||||
//! matrix multiplies on ymm registers.
|
||||
|
||||
const bone_sse = @import("bone_sse.zig");
|
||||
|
||||
const V4d = @Vector(4, f64);
|
||||
|
||||
// =============================================================================
|
||||
// Imports from bone_sse — constants, memory helpers, interp/billboard helpers
|
||||
// =============================================================================
|
||||
|
||||
const SO = bone_sse.SO;
|
||||
const BR = bone_sse.BR;
|
||||
const BD = bone_sse.BD;
|
||||
const AD = bone_sse.AD;
|
||||
const InterpResult = bone_sse.InterpResult;
|
||||
|
||||
const ru32 = bone_sse.ru32;
|
||||
const ri32 = bone_sse.ri32;
|
||||
const rf32 = bone_sse.rf32;
|
||||
const ru16 = bone_sse.ru16;
|
||||
const ri16 = bone_sse.ri16;
|
||||
const ru8 = bone_sse.ru8;
|
||||
const wu32 = bone_sse.wu32;
|
||||
const wf32 = bone_sse.wf32;
|
||||
const wu16 = bone_sse.wu16;
|
||||
const wu8 = bone_sse.wu8;
|
||||
const fbits = bone_sse.fbits;
|
||||
const ufloat = bone_sse.ufloat;
|
||||
|
||||
const normalizeVec3 = bone_sse.normalizeVec3;
|
||||
const normalizeVec3InPlace = bone_sse.normalizeVec3InPlace;
|
||||
const crossVec3 = bone_sse.crossVec3;
|
||||
|
||||
const canReuseInterp = bone_sse.canReuseInterp;
|
||||
const findInterpIdx = bone_sse.findInterpIdx;
|
||||
const interpAnimKFCached = bone_sse.interpAnimKFCached;
|
||||
const interpVec3TrackCached = bone_sse.interpVec3TrackCached;
|
||||
const callFtol = bone_sse.callFtol;
|
||||
const callVec3SqMag = bone_sse.callVec3SqMag;
|
||||
const fastMod = bone_sse.fastMod;
|
||||
|
||||
const texAnimLoop = bone_sse.texAnimLoop;
|
||||
const colorAnimLoop = bone_sse.colorAnimLoop;
|
||||
const wordAnimLoop = bone_sse.wordAnimLoop;
|
||||
const boneKeyframeLoop = bone_sse.boneKeyframeLoop;
|
||||
const particleLoops = bone_sse.particleLoops;
|
||||
// NOTE: we do NOT import bone_sse.attachmentRecursion — that version recurses
|
||||
// into bone_sse.transformImpl_SSE (f32) directly, which would force attached
|
||||
// child models onto the f32 path while the parent is f64. We reimplement it
|
||||
// below so attachment recursion stays inside bone_sse64's f64 pipeline.
|
||||
|
||||
// =============================================================================
|
||||
// f64 memory helpers
|
||||
// =============================================================================
|
||||
|
||||
inline fn rf64(addr: u32) f64 {
|
||||
return @floatCast(rf32(addr));
|
||||
}
|
||||
|
||||
inline fn wf32_narrow(addr: u32, v: f64) void {
|
||||
wf32(addr, @floatCast(v));
|
||||
}
|
||||
|
||||
inline fn loadV4d(addr: u32) V4d {
|
||||
return V4d{ rf64(addr), rf64(addr + 4), rf64(addr + 8), rf64(addr + 12) };
|
||||
}
|
||||
|
||||
inline fn storeV4d(addr: u32, v: V4d) void {
|
||||
wf32(addr + 0, @floatCast(v[0]));
|
||||
wf32(addr + 4, @floatCast(v[1]));
|
||||
wf32(addr + 8, @floatCast(v[2]));
|
||||
wf32(addr + 12, @floatCast(v[3]));
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// f64 matrix operations
|
||||
// =============================================================================
|
||||
|
||||
/// 4x4 matrix multiply: dst = a * b. Memory-to-memory with f64 intermediates.
|
||||
inline fn matMul4x4_64(dst: u32, a: u32, b: u32) void {
|
||||
const b0 = loadV4d(b);
|
||||
const b1 = loadV4d(b + 16);
|
||||
const b2 = loadV4d(b + 32);
|
||||
const b3 = loadV4d(b + 48);
|
||||
const a0 = loadV4d(a);
|
||||
const a1 = loadV4d(a + 16);
|
||||
const a2 = loadV4d(a + 32);
|
||||
const a3 = loadV4d(a + 48);
|
||||
const rows = [4]V4d{ a0, a1, a2, a3 };
|
||||
|
||||
inline for (0..4) |i| {
|
||||
const s0: V4d = @splat(rows[i][0]);
|
||||
const s1: V4d = @splat(rows[i][1]);
|
||||
const s2: V4d = @splat(rows[i][2]);
|
||||
const s3: V4d = @splat(rows[i][3]);
|
||||
const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3;
|
||||
const off: u32 = @intCast(i * 16);
|
||||
storeV4d(dst + off, row);
|
||||
}
|
||||
}
|
||||
|
||||
/// 4x4 multiply: dst_mem = a_local * b_mem. Left operand is [16]f64 local.
|
||||
inline fn matMul4x4Local_64(dst: u32, a: [16]f64, b: u32) void {
|
||||
const b0 = loadV4d(b);
|
||||
const b1 = loadV4d(b + 16);
|
||||
const b2 = loadV4d(b + 32);
|
||||
const b3 = loadV4d(b + 48);
|
||||
|
||||
const rows = [4]V4d{
|
||||
V4d{ a[0], a[1], a[2], a[3] },
|
||||
V4d{ a[4], a[5], a[6], a[7] },
|
||||
V4d{ a[8], a[9], a[10], a[11] },
|
||||
V4d{ a[12], a[13], a[14], a[15] },
|
||||
};
|
||||
|
||||
inline for (0..4) |i| {
|
||||
const s0: V4d = @splat(rows[i][0]);
|
||||
const s1: V4d = @splat(rows[i][1]);
|
||||
const s2: V4d = @splat(rows[i][2]);
|
||||
const s3: V4d = @splat(rows[i][3]);
|
||||
const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3;
|
||||
const off: u32 = @intCast(i * 16);
|
||||
storeV4d(dst + off, row);
|
||||
}
|
||||
}
|
||||
|
||||
/// In-place f64 multiply: a = a * b_mem. Returns new [16]f64.
|
||||
inline fn matMul4x4InPlace_64(a: [16]f64, b: u32) [16]f64 {
|
||||
const b0 = loadV4d(b);
|
||||
const b1 = loadV4d(b + 16);
|
||||
const b2 = loadV4d(b + 32);
|
||||
const b3 = loadV4d(b + 48);
|
||||
|
||||
const rows = [4]V4d{
|
||||
V4d{ a[0], a[1], a[2], a[3] },
|
||||
V4d{ a[4], a[5], a[6], a[7] },
|
||||
V4d{ a[8], a[9], a[10], a[11] },
|
||||
V4d{ a[12], a[13], a[14], a[15] },
|
||||
};
|
||||
|
||||
var result: [16]f64 = undefined;
|
||||
inline for (0..4) |i| {
|
||||
const s0: V4d = @splat(rows[i][0]);
|
||||
const s1: V4d = @splat(rows[i][1]);
|
||||
const s2: V4d = @splat(rows[i][2]);
|
||||
const s3: V4d = @splat(rows[i][3]);
|
||||
const row = s0 * b0 + s1 * b1 + s2 * b2 + s3 * b3;
|
||||
result[i * 4 + 0] = row[0];
|
||||
result[i * 4 + 1] = row[1];
|
||||
result[i * 4 + 2] = row[2];
|
||||
result[i * 4 + 3] = row[3];
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// Quaternion → 4x4 rotation matrix as f64 local. Identity last row/col.
|
||||
inline fn buildRotationMatrix_64(qx: f64, qy: f64, qz: f64, qw: f64) [16]f64 {
|
||||
const xx2 = qx * (qx + qx);
|
||||
const xy2 = qx * (qy + qy);
|
||||
const xz2 = qx * (qz + qz);
|
||||
const yy2 = qy * (qy + qy);
|
||||
const yz2 = qy * (qz + qz);
|
||||
const zz2 = qz * (qz + qz);
|
||||
const wx2 = qw * (qx + qx);
|
||||
const wy2 = qw * (qy + qy);
|
||||
const wz2 = qw * (qz + qz);
|
||||
|
||||
return [16]f64{
|
||||
1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0.0,
|
||||
xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0.0,
|
||||
xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0.0,
|
||||
0.0, 0.0, 0.0, 1.0,
|
||||
};
|
||||
}
|
||||
|
||||
/// Copy a 4x4 matrix in game memory (bit-exact, no precision change).
|
||||
inline fn copyMat4(dst: u32, src: u32) void {
|
||||
inline for (0..8) |i| {
|
||||
const off = @as(u32, @intCast(i)) * 8;
|
||||
@as(*u64, @ptrFromInt(dst + off)).* = @as(*const u64, @ptrFromInt(src + off)).*;
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Attachment recursion — f64 version that re-enters transformImpl_SSE64 for
|
||||
// child scene objects. Mirrors bone_sse.attachmentRecursion exactly except
|
||||
// it calls our f64 implementation for child transforms.
|
||||
// =============================================================================
|
||||
|
||||
fn attachmentRecursion64(this: u32, model_hdr: u32, bone_out_base: u32, frame_ctr: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
const hierarchy = ru32(this + SO.hierarchy_ptr);
|
||||
if (hierarchy == 0) return;
|
||||
|
||||
const attach_count = ru32(model_hdr + 0x104);
|
||||
const attach_data = ru32(model_hdr + 0x108);
|
||||
|
||||
var att_i: u32 = 0;
|
||||
var att_off: u32 = 0;
|
||||
while (att_i < attach_count) : ({
|
||||
att_i += 1;
|
||||
att_off += 0x30;
|
||||
}) {
|
||||
const att_entry = attach_data + att_off;
|
||||
if (frame_ctr < ru32(att_entry + 0x20)) {
|
||||
const bone_idx = @as(u32, ru16(att_entry + 4));
|
||||
const bone_rt = ru32(this + SO.bone_rt_base) + bone_idx * 0x118;
|
||||
const anim_data = att_entry + 0x14;
|
||||
const att_output = hierarchy + att_i * 0x20;
|
||||
const atr = findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), anim_data, att_output);
|
||||
wu8(att_output + 0x0C, ru8(ru32(anim_data + AD.keyframe_base) + atr.idx0));
|
||||
}
|
||||
}
|
||||
|
||||
var child = ru32(this + SO.hierarchy_idx);
|
||||
while (child != 0) {
|
||||
const attach_idx = ru32(child + 0x1D4);
|
||||
|
||||
if (attach_idx != 0xFFFF) {
|
||||
const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C);
|
||||
if (visible != 0) {
|
||||
const att_entry = attach_data + attach_idx * 0x30;
|
||||
const bone_idx = @as(u32, ru16(att_entry + 4));
|
||||
const bone_mat = bone_out_base + bone_idx * 0x40;
|
||||
|
||||
// Copy parent bone matrix to local (f32, matches original).
|
||||
// The matrix itself is stored f32 in game memory; we preserve
|
||||
// that on the wire but widen to f64 for the offset math below.
|
||||
var local_1a0: [16]f32 = undefined;
|
||||
for (0..16) |fi| {
|
||||
local_1a0[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4);
|
||||
}
|
||||
|
||||
// Apply attachment offset translation in f64, narrow on store.
|
||||
const ox: f64 = @floatCast(rf32(att_entry + 8));
|
||||
const oy: f64 = @floatCast(rf32(att_entry + 0xC));
|
||||
const oz: f64 = @floatCast(rf32(att_entry + 0x10));
|
||||
const m0x: f64 = @floatCast(local_1a0[0]);
|
||||
const m4x: f64 = @floatCast(local_1a0[4]);
|
||||
const m8x: f64 = @floatCast(local_1a0[8]);
|
||||
const m1x: f64 = @floatCast(local_1a0[1]);
|
||||
const m5x: f64 = @floatCast(local_1a0[5]);
|
||||
const m9x: f64 = @floatCast(local_1a0[9]);
|
||||
const m2x: f64 = @floatCast(local_1a0[2]);
|
||||
const m6x: f64 = @floatCast(local_1a0[6]);
|
||||
const m10x: f64 = @floatCast(local_1a0[10]);
|
||||
local_1a0[12] = @floatCast(@as(f64, @floatCast(local_1a0[12])) + m0x * ox + m4x * oy + m8x * oz);
|
||||
local_1a0[13] = @floatCast(@as(f64, @floatCast(local_1a0[13])) + m1x * ox + m5x * oy + m9x * oz);
|
||||
local_1a0[14] = @floatCast(@as(f64, @floatCast(local_1a0[14])) + m2x * ox + m6x * oy + m10x * oz);
|
||||
|
||||
// Recurse into the f64 implementation so attachment children stay
|
||||
// on the same precision path as the parent.
|
||||
transformImpl_SSE64(child, @intFromPtr(&local_1a0), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z));
|
||||
}
|
||||
}
|
||||
|
||||
child = ru32(child + 0x1E4);
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Main entry point — f64-intermediate reimplementation.
|
||||
//
|
||||
// Follows the same structure as bone_sse.transformImpl_SSE. Per-bone work
|
||||
// builds local_mat / local_mat2 as [16]f64 and uses the f64 matrix helpers
|
||||
// above. Non-matrix helpers (interp, billboard, loops) are imported from
|
||||
// bone_sse — their f32 outputs widen implicitly when used in f64 math.
|
||||
// Attachment recursion is handled by a local f64 version (above) so child
|
||||
// scene objects stay on the f64 pipeline.
|
||||
// =============================================================================
|
||||
|
||||
pub fn transformImpl_SSE64(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
|
||||
// Section 1: Entry checks
|
||||
if (ru32(this + SO.model_data_ptr) == 0) return;
|
||||
const anim_ctx = ru32(this + SO.anim_ctx_ptr);
|
||||
if (ru32(this + SO.sync_value) == ru32(anim_ctx + 0x10)) return;
|
||||
|
||||
// Section 2: Emitter setup
|
||||
const model_ctr = ru32(this + SO.model_ctr_ptr);
|
||||
const model_hdr = ru32(model_ctr + 0x130);
|
||||
const emitter_ctx = ru32(this + SO.emitter_ctx);
|
||||
|
||||
if (emitter_ctx != 0) {
|
||||
const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0;
|
||||
wu32(this + 0x50, has_emitter);
|
||||
wu32(this + 0x17C, ru32(emitter_ctx + 0x17C));
|
||||
}
|
||||
|
||||
// Section 3: World position/scale
|
||||
const pos_ptr = mat2;
|
||||
const ofs_ptr = mat3;
|
||||
const scale_f: f32 = @bitCast(mat4);
|
||||
|
||||
wf32(this + SO.world_pos + 0, rf32(pos_ptr) * rf32(this + SO.field_184));
|
||||
wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(pos_ptr + 4));
|
||||
wf32(this + SO.world_pos + 8, @bitCast(fbits(rf32(this + SO.field_18c) * rf32(pos_ptr + 8))));
|
||||
|
||||
const rp0 = rf32(ofs_ptr) + rf32(this + SO.field_190);
|
||||
const rp1 = rf32(this + SO.render_scale_x) + rf32(ofs_ptr + 4);
|
||||
const rp2 = rf32(this + SO.render_scale_y) + rf32(ofs_ptr + 8);
|
||||
wf32(this + SO.render_pri + 0, rp0);
|
||||
wf32(this + SO.render_pri + 4, rp1);
|
||||
wf32(this + SO.render_pri + 8, rp2);
|
||||
|
||||
wf32(this + SO.render_scale_z, scale_f * rf32(this + SO.field_180));
|
||||
|
||||
// Section 4: Global sequence processing
|
||||
const gs_count = ru32(model_hdr + 0x14);
|
||||
if (gs_count != 0) {
|
||||
const gs_durations = ru32(model_hdr + 0x18);
|
||||
const gs_values = ru32(this + SO.gs_values_ptr);
|
||||
const timestamp = ru32(anim_ctx + 0x0C);
|
||||
const time_base = ru32(this + SO.gs_time_base);
|
||||
var gi: u32 = 0;
|
||||
while (gi < gs_count) : (gi += 1) {
|
||||
const dur = ru32(gs_durations + gi * 4);
|
||||
if (dur == 0) {
|
||||
wu32(gs_values + gi * 4, 0);
|
||||
} else {
|
||||
wu32(gs_values + gi * 4, (timestamp -% time_base) % dur);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Root matrix multiply (f64 intermediates).
|
||||
// Replaces the original initParticlePixelShaderGeneration(0x74a7c0) dispatch.
|
||||
matMul4x4_64(this + 0xFC, this + 0xBC, mat1);
|
||||
|
||||
// Section 5: child_objects_padding — compute in f64, narrow on store.
|
||||
// May be used as a culling threshold; matching precision keeps boundary
|
||||
// conditions stable across frames.
|
||||
const emitter_ctx_5 = ru32(this + SO.emitter_ctx);
|
||||
if (emitter_ctx_5 == 0 or (ru8(emitter_ctx_5 + 4) & 1) != 0) {
|
||||
const wx: f64 = rf64(this + SO.world_xform + 8 * 4);
|
||||
const wy: f64 = rf64(this + SO.world_xform + 9 * 4);
|
||||
const wz: f64 = rf64(this + SO.world_xform + 10 * 4);
|
||||
const len_sq: f32 = @floatCast(wx * wx + wy * wy + wz * wz);
|
||||
wu32(this + SO.child_padding, fbits(len_sq));
|
||||
} else {
|
||||
wu32(this + SO.child_padding, ru32(emitter_ctx_5 + 0x84));
|
||||
}
|
||||
|
||||
// Section 6: Identity matrices as f64 locals + timestamp delta
|
||||
var local_mat: [16]f64 = .{
|
||||
1, 0, 0, 0,
|
||||
0, 1, 0, 0,
|
||||
0, 0, 1, 0,
|
||||
0, 0, 0, 1,
|
||||
};
|
||||
|
||||
var local_mat2: [16]f64 = .{
|
||||
1, 0, 0, 0,
|
||||
0, 1, 0, 0,
|
||||
0, 0, 1, 0,
|
||||
0, 0, 0, 1,
|
||||
};
|
||||
|
||||
var time_delta_val: u32 = 0;
|
||||
const sdb = ru32(this + SO.search_data_base);
|
||||
if (sdb != 0) {
|
||||
const cur_ts = ru32(anim_ctx + 0x0C);
|
||||
if (cur_ts != 0) {
|
||||
time_delta_val = cur_ts -% sdb;
|
||||
wu32(this + SO.search_data_base, cur_ts);
|
||||
}
|
||||
}
|
||||
|
||||
// Section 7: Main bone loop
|
||||
const bone_count = ru32(model_hdr + 0x34);
|
||||
const bone_defs = ru32(model_hdr + 0x38);
|
||||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||||
const bone_out_base = ru32(this + SO.bone_out_ptr);
|
||||
const frame_ctr = ru32(this + SO.anim_frame_ctr);
|
||||
|
||||
if (bone_count != 0) {
|
||||
var bone_idx: u32 = 0;
|
||||
var bdef = bone_defs;
|
||||
var brt = bone_rt_base;
|
||||
while (bone_idx < bone_count) : ({
|
||||
bone_idx += 1;
|
||||
bdef += 0x6C;
|
||||
brt += 0x118;
|
||||
}) {
|
||||
const flags = ru32(bdef + BD.flags);
|
||||
const parent_idx_raw: i32 = @as(i32, @intCast(@as(i16, @bitCast(ru16(bdef + BD.parent_bone)))));
|
||||
|
||||
// --- Animation time computation (primary slot) ---
|
||||
const anim_slot_val = ri32(brt + BR.anim_slot);
|
||||
if (anim_slot_val == -1) {
|
||||
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
|
||||
const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118;
|
||||
wu32(brt + BR.prim_time, ru32(parent_rt + BR.prim_time));
|
||||
wu32(brt + BR.prim_track, ru32(parent_rt + BR.prim_track));
|
||||
wu32(brt + BR.prim_anim, ru32(parent_rt + BR.prim_anim));
|
||||
} else if (bone_idx != 0) {
|
||||
wu32(brt + BR.prim_time, ru32(bone_rt_base + BR.prim_time));
|
||||
wu32(brt + BR.prim_track, ru32(bone_rt_base + BR.prim_track));
|
||||
wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim));
|
||||
}
|
||||
} else {
|
||||
if (ru32(this + 0x4C) != 0) {
|
||||
wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val);
|
||||
wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val);
|
||||
}
|
||||
|
||||
const anim_lookup = ru32(model_hdr + 0x20);
|
||||
const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44;
|
||||
const cur_time = ru32(ru32(this + 0x2C) + 0xC);
|
||||
|
||||
if ((ru8(anim_entry + 0x10) & 1) == 0) {
|
||||
const anim_end = ru32(anim_entry + 0x08);
|
||||
const anim_start = ru32(anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = cur_time -% ru32(brt + 0xA8);
|
||||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0);
|
||||
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start);
|
||||
wu32(brt + 0x98, anim_start +% frame);
|
||||
} else {
|
||||
wu32(brt + 0x98, anim_start);
|
||||
}
|
||||
} else {
|
||||
const sec_end_val = ru32(brt + 0xAC);
|
||||
const sec_start_val = ru32(brt + 0xA8);
|
||||
|
||||
if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) {
|
||||
const effective_time = if (@as(i32, @bitCast(sec_start_val -% cur_time)) > 0) sec_start_val else cur_time;
|
||||
const anim_end = ru32(anim_entry + 0x08);
|
||||
const anim_start = ru32(anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = effective_time -% ru32(brt + 0xA8);
|
||||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0);
|
||||
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8), anim_end -% anim_start);
|
||||
wu32(brt + 0x98, anim_start +% frame);
|
||||
} else {
|
||||
wu32(brt + 0x98, anim_start);
|
||||
}
|
||||
} else {
|
||||
const dur = sec_end_val -% sec_start_val;
|
||||
const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xB0);
|
||||
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8)));
|
||||
|
||||
if (offset < 0) {
|
||||
wu32(brt + 0x98, ru32(anim_entry + 0x04));
|
||||
} else {
|
||||
const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08)));
|
||||
const anim_start_i = @as(i32, @bitCast(ru32(anim_entry + 0x04)));
|
||||
if (offset <= anim_end_i - anim_start_i) {
|
||||
wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i)));
|
||||
} else {
|
||||
wu32(brt + 0x98, ru32(anim_entry + 0x08));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
wu32(brt + 0x9C, ru32(brt + 0xA4));
|
||||
wu32(brt + 0xA0, bone_idx);
|
||||
}
|
||||
|
||||
// --- Secondary slot ---
|
||||
const sec_slot_val = ri32(brt + BR.sec_slot);
|
||||
if (sec_slot_val == -1) {
|
||||
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
|
||||
const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118;
|
||||
wu32(brt + BR.sec_time, ru32(parent_rt + BR.sec_time));
|
||||
wu32(brt + BR.sec_track, ru32(parent_rt + BR.sec_track));
|
||||
} else if (bone_idx != 0) {
|
||||
wu32(brt + BR.sec_time, ru32(bone_rt_base + BR.sec_time));
|
||||
wu32(brt + BR.sec_track, ru32(bone_rt_base + BR.sec_track));
|
||||
} else {
|
||||
wu32(brt + BR.sec_time, ru32(brt + BR.prim_time));
|
||||
wu32(brt + BR.sec_track, ru32(brt + BR.prim_track));
|
||||
}
|
||||
} else {
|
||||
if (ru32(this + 0x4C) != 0) {
|
||||
wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val);
|
||||
wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val);
|
||||
}
|
||||
|
||||
const sec_anim_lookup = ru32(model_hdr + 0x20);
|
||||
const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44;
|
||||
const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC);
|
||||
|
||||
if ((ru8(sec_anim_entry + 0x10) & 1) == 0) {
|
||||
const anim_end = ru32(sec_anim_entry + 0x08);
|
||||
const anim_start = ru32(sec_anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = sec_cur_time -% ru32(brt + 0xD4);
|
||||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC);
|
||||
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start);
|
||||
wu32(brt + 0xC4, anim_start +% frame);
|
||||
} else {
|
||||
wu32(brt + 0xC4, anim_start);
|
||||
}
|
||||
} else {
|
||||
const sec_end_val = ru32(brt + 0xD8);
|
||||
const sec_start_val = ru32(brt + 0xD4);
|
||||
|
||||
if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) {
|
||||
const effective_time = if (@as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) sec_start_val else sec_cur_time;
|
||||
const anim_end = ru32(sec_anim_entry + 0x08);
|
||||
const anim_start = ru32(sec_anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = effective_time -% ru32(brt + 0xD4);
|
||||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC);
|
||||
const frame = fastMod(@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4), anim_end -% anim_start);
|
||||
wu32(brt + 0xC4, anim_start +% frame);
|
||||
} else {
|
||||
wu32(brt + 0xC4, anim_start);
|
||||
}
|
||||
} else {
|
||||
const dur = sec_end_val -% sec_start_val;
|
||||
const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xDC);
|
||||
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4)));
|
||||
|
||||
if (offset < 0) {
|
||||
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04));
|
||||
} else {
|
||||
const anim_end_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x08)));
|
||||
const anim_start_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x04)));
|
||||
if (offset <= anim_end_i - anim_start_i) {
|
||||
wu32(brt + 0xC4, @as(u32, @bitCast(offset + anim_start_i)));
|
||||
} else {
|
||||
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x08));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
wu32(brt + 0xC8, ru32(brt + 0xD0));
|
||||
|
||||
if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) {
|
||||
wu32(brt + 0xD0, 0xFFFFFFFF);
|
||||
}
|
||||
}
|
||||
|
||||
// --- Blend weight ---
|
||||
if (ri32(brt + BR.anim_slot) == -1 and ri32(brt + BR.sec_slot) == -1) {
|
||||
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
|
||||
wu32(brt + BR.blend_weight, ru32(bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118 + BR.blend_weight));
|
||||
} else if (bone_idx == 0) {
|
||||
wu32(brt + BR.blend_weight, 0);
|
||||
} else {
|
||||
wu32(brt + BR.blend_weight, ru32(bone_rt_base + BR.blend_weight));
|
||||
}
|
||||
} else {
|
||||
const cf_remaining = ri32(brt + BR.crossfade_end) - ri32(anim_ctx + 0x0C);
|
||||
if (cf_remaining < 1 or (ru32(brt + BR.prim_time) == ru32(brt + BR.sec_time) and
|
||||
ru32(brt + BR.prim_track) == ru32(brt + BR.sec_track)))
|
||||
{
|
||||
wu32(brt + BR.blend_weight, 0);
|
||||
} else {
|
||||
const t_raw = @as(f32, @floatFromInt(cf_remaining)) * ufloat(ru32(brt + BR.crossfade_inv));
|
||||
const t_clamped = if (t_raw < 0.0) @as(f32, 0.0) else if (t_raw > 1.0) @as(f32, 1.0) else t_raw;
|
||||
const h = (3.0 - 2.0 * t_clamped) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight));
|
||||
wu32(brt + BR.blend_weight, fbits(h));
|
||||
}
|
||||
}
|
||||
|
||||
// --- Parent bone transform inheritance ---
|
||||
const combined_flags: u32 = ru32(brt + BR.flags2) | flags;
|
||||
var src_mat: u32 = undefined;
|
||||
|
||||
// Address of local_mat (as if it were a game-memory matrix — since the
|
||||
// matMul helpers read from memory, we write local_mat out to a scratch
|
||||
// f32 buffer when src_mat aliases it. The billboard path below stores
|
||||
// f32-narrowed values into local_mat's address via a scratch buffer.)
|
||||
var local_mat_f32: [16]f32 = undefined;
|
||||
const local_mat_addr = @intFromPtr(&local_mat_f32);
|
||||
|
||||
if (ru16(bdef + BD.parent_bone) == 0xFFFF) {
|
||||
src_mat = this + 0xFC;
|
||||
} else {
|
||||
const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40;
|
||||
src_mat = parent_out;
|
||||
|
||||
if ((combined_flags & 7) != 0) {
|
||||
// Copy parent matrix to local_mat (f64 widen)
|
||||
for (0..16) |i| {
|
||||
local_mat[i] = rf64(parent_out + @as(u32, @intCast(i)) * 4);
|
||||
}
|
||||
|
||||
const pivot_x: f64 = rf64(bdef + BD.pivot_x);
|
||||
const pivot_y: f64 = rf64(bdef + BD.pivot_y);
|
||||
const pivot_z: f64 = rf64(bdef + BD.pivot_z);
|
||||
|
||||
// Compute translated position — accumulation order must match
|
||||
// original x87. Row 0 uses (pz + px + py), rows 1/2 use (pz + py + px).
|
||||
const tx = local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[12];
|
||||
const ty = local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x + local_mat[13];
|
||||
const tz = local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x + local_mat[14];
|
||||
|
||||
const bb_type = combined_flags & 6;
|
||||
const billboard_eps: f64 = @floatCast(rf32(0x008029d4));
|
||||
const cull_eps: f64 = @floatCast(rf32(0x0080c5c8));
|
||||
if (bb_type == 2) {
|
||||
// Cylindrical billboard — normalize columns in f64 to match
|
||||
// x87's extended-precision 1/sqrt pattern. Previous impl went
|
||||
// f32→f32 through normalizeVec3; that diverged from x87 by
|
||||
// up to a ULP per axis, which shifted particle emitter
|
||||
// orientation each camera frame and caused transparency
|
||||
// flicker against ground effects.
|
||||
inline for ([_]u32{ 0, 4, 8 }) |row_off| {
|
||||
const cx = local_mat[row_off];
|
||||
const cy = local_mat[row_off + 1];
|
||||
const cz = local_mat[row_off + 2];
|
||||
const len_sq = cx * cx + cy * cy + cz * cz;
|
||||
const len = @sqrt(len_sq);
|
||||
if (len >= billboard_eps) {
|
||||
const inv = 1.0 / len;
|
||||
local_mat[row_off] = cx * inv;
|
||||
local_mat[row_off + 1] = cy * inv;
|
||||
local_mat[row_off + 2] = cz * inv;
|
||||
}
|
||||
}
|
||||
} else if (bb_type == 4) {
|
||||
// Spherical billboard — inherit camera basis, rescale to
|
||||
// preserve original column length. Keep all intermediates
|
||||
// in f64 matching x87's 80-bit temporaries.
|
||||
inline for ([_]struct { row_off: u32, src_off: u32 }{
|
||||
.{ .row_off = 0, .src_off = SO.bb_row0 },
|
||||
.{ .row_off = 4, .src_off = SO.world_xform },
|
||||
.{ .row_off = 8, .src_off = SO.world_xform + 16 },
|
||||
}) |p| {
|
||||
const src_addr = this + p.src_off;
|
||||
const cam_x: f64 = rf64(src_addr);
|
||||
const cam_y: f64 = rf64(src_addr + 4);
|
||||
const cam_z: f64 = rf64(src_addr + 8);
|
||||
const cam_len_sq = cam_x * cam_x + cam_y * cam_y + cam_z * cam_z;
|
||||
var s: f64 = 1.0;
|
||||
if (cam_len_sq > cull_eps) {
|
||||
const mx = local_mat[p.row_off];
|
||||
const my = local_mat[p.row_off + 1];
|
||||
const mz = local_mat[p.row_off + 2];
|
||||
const mat_len_sq = mx * mx + my * my + mz * mz;
|
||||
s = @sqrt(mat_len_sq / cam_len_sq);
|
||||
}
|
||||
local_mat[p.row_off] = s * cam_x;
|
||||
local_mat[p.row_off + 1] = s * cam_y;
|
||||
local_mat[p.row_off + 2] = s * cam_z;
|
||||
}
|
||||
} else if (bb_type == 6) {
|
||||
local_mat[0] = rf64(this + SO.bb_row0);
|
||||
local_mat[1] = rf64(this + SO.bb_row0 + 4);
|
||||
local_mat[2] = rf64(this + SO.bb_row0 + 8);
|
||||
local_mat[4] = rf64(this + SO.world_xform + 0 * 4);
|
||||
local_mat[5] = rf64(this + SO.world_xform + 1 * 4);
|
||||
local_mat[6] = rf64(this + SO.world_xform + 2 * 4);
|
||||
local_mat[8] = rf64(this + SO.world_xform + 4 * 4);
|
||||
local_mat[9] = rf64(this + SO.world_xform + 5 * 4);
|
||||
local_mat[10] = rf64(this + SO.world_xform + 6 * 4);
|
||||
}
|
||||
|
||||
// Recompute translation — mirror accumulation order of tx/ty/tz above.
|
||||
if ((combined_flags & 1) == 0) {
|
||||
local_mat[12] = tx - (local_mat[8] * pivot_z + local_mat[0] * pivot_x + local_mat[4] * pivot_y);
|
||||
local_mat[13] = ty - (local_mat[9] * pivot_z + local_mat[5] * pivot_y + local_mat[1] * pivot_x);
|
||||
local_mat[14] = tz - (local_mat[10] * pivot_z + local_mat[6] * pivot_y + local_mat[2] * pivot_x);
|
||||
} else {
|
||||
local_mat[12] = rf64(this + SO.world_xform + 8 * 4);
|
||||
local_mat[13] = rf64(this + SO.world_xform + 9 * 4);
|
||||
local_mat[14] = rf64(this + SO.world_xform + 10 * 4);
|
||||
}
|
||||
|
||||
// Narrow local_mat to f32 scratch, set src_mat to its address
|
||||
for (0..16) |i| {
|
||||
local_mat_f32[i] = @floatCast(local_mat[i]);
|
||||
}
|
||||
src_mat = local_mat_addr;
|
||||
}
|
||||
}
|
||||
|
||||
// --- Rotation / Scale / Translation / Final multiply ---
|
||||
if ((combined_flags & 0x280) == 0) {
|
||||
const dst = bone_out_base + bone_idx * 0x40;
|
||||
copyMat4(dst, src_mat);
|
||||
} else {
|
||||
const rot_anim = bdef + BD.rot_anim;
|
||||
const rot_kf_count = ru32(bdef + BD.rot_nts);
|
||||
|
||||
var rot_primary_cache: ?InterpResult = null;
|
||||
|
||||
if (rot_kf_count != 0) {
|
||||
if (frame_ctr < rot_kf_count) {
|
||||
const rot_output = brt + BR.rot_idx0;
|
||||
const r = findInterpIdx(this, ru32(brt + BR.prim_time), ru32(brt + BR.prim_track), rot_anim, rot_output);
|
||||
rot_primary_cache = r;
|
||||
const q = interpAnimKFCached(this, brt, rot_anim, rot_output, r);
|
||||
local_mat2 = buildRotationMatrix_64(q[0], q[1], q[2], q[3]);
|
||||
} else {
|
||||
local_mat2 = buildRotationMatrix_64(rf64(brt + BR.rot_x), rf64(brt + BR.rot_y), rf64(brt + BR.rot_z), rf64(brt + BR.rot_w));
|
||||
}
|
||||
} else {
|
||||
local_mat2 = .{
|
||||
1, 0, 0, 0,
|
||||
0, 1, 0, 0,
|
||||
0, 0, 1, 0,
|
||||
0, 0, 0, 1,
|
||||
};
|
||||
}
|
||||
|
||||
// Step 2: Scale interpolation — f64 arithmetic
|
||||
const scale_anim = bdef + BD.scale_anim;
|
||||
const scale_kf_count = ru32(bdef + BD.scale_nts);
|
||||
if (scale_kf_count != 0) {
|
||||
var sx: f64 = undefined;
|
||||
var sy: f64 = undefined;
|
||||
var sz: f64 = undefined;
|
||||
if (frame_ctr < scale_kf_count) {
|
||||
const scale_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, scale_anim)) rot_primary_cache else null;
|
||||
const s = interpVec3TrackCached(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)), scale_cache);
|
||||
sx = s[0]; sy = s[1]; sz = s[2];
|
||||
} else {
|
||||
sx = rf64(brt + BR.scale_x); sy = rf64(brt + BR.scale_y); sz = rf64(brt + BR.scale_z);
|
||||
}
|
||||
local_mat2[0] *= sx; local_mat2[1] *= sx; local_mat2[2] *= sx;
|
||||
local_mat2[4] *= sy; local_mat2[5] *= sy; local_mat2[6] *= sy;
|
||||
local_mat2[8] *= sz; local_mat2[9] *= sz; local_mat2[10] *= sz;
|
||||
}
|
||||
|
||||
// Conditional bone-flag matrix multiply (f64 in-place)
|
||||
if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) {
|
||||
local_mat2 = matMul4x4InPlace_64(local_mat2, ru32(brt + BR.bone_flag_cache));
|
||||
}
|
||||
|
||||
// Step 3: Translation interpolation (f64)
|
||||
var tx_val: f64 = rf64(bdef + BD.pivot_x);
|
||||
var ty_val: f64 = rf64(bdef + BD.pivot_y);
|
||||
var tz_val: f64 = rf64(bdef + BD.pivot_z);
|
||||
|
||||
const trans_anim = bdef + BD.trans_anim;
|
||||
const trans_kf_count = ru32(bdef + BD.trans_nts);
|
||||
if (trans_kf_count != 0) {
|
||||
if (frame_ctr < trans_kf_count) {
|
||||
const trans_cache = if (rot_primary_cache != null and canReuseInterp(rot_anim, trans_anim)) rot_primary_cache else null;
|
||||
const t = interpVec3TrackCached(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)), trans_cache);
|
||||
tx_val += @as(f64, t[0]);
|
||||
ty_val += @as(f64, t[1]);
|
||||
tz_val += @as(f64, t[2]);
|
||||
} else {
|
||||
tx_val += rf64(brt + BR.trans_x);
|
||||
ty_val += rf64(brt + BR.trans_y);
|
||||
tz_val += rf64(brt + BR.trans_z);
|
||||
}
|
||||
}
|
||||
|
||||
// Step 4: Translation offset using ROTATED+SCALED matrix (f64)
|
||||
const piv_x: f64 = rf64(bdef + BD.pivot_x);
|
||||
const piv_y: f64 = rf64(bdef + BD.pivot_y);
|
||||
const piv_z: f64 = rf64(bdef + BD.pivot_z);
|
||||
local_mat2[12] = tx_val - (local_mat2[0] * piv_x + local_mat2[4] * piv_y + local_mat2[8] * piv_z);
|
||||
local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z);
|
||||
local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z);
|
||||
|
||||
// Final: dst = bone_local * parent (f64 mul, narrow on store)
|
||||
matMul4x4Local_64(bone_out_base + bone_idx * 0x40, local_mat2, src_mat);
|
||||
}
|
||||
|
||||
// --- Billboard post-processing (flags & 0x78) ---
|
||||
if ((combined_flags & 0x78) != 0) {
|
||||
const out_off = bone_idx * 0x40;
|
||||
const om = bone_out_base + out_off;
|
||||
|
||||
const scale_len0 = @sqrt(callVec3SqMag(om));
|
||||
const scale_len1 = @sqrt(callVec3SqMag(om + 0x10));
|
||||
const scale_len2 = @sqrt(callVec3SqMag(om + 0x20));
|
||||
|
||||
const bpx = rf32(bdef + BD.pivot_x);
|
||||
const bpy = rf32(bdef + BD.pivot_y);
|
||||
const bpz = rf32(bdef + BD.pivot_z);
|
||||
// Accumulation order mirrors original x87:
|
||||
// pos_x: px + py + pz + const
|
||||
// pos_y: py + pz + px + const
|
||||
// pos_z: py + pz + px + const
|
||||
const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30);
|
||||
const pos_y = bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + bpx * rf32(om + 0x04) + rf32(om + 0x34);
|
||||
const pos_z = bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + bpx * rf32(om + 0x08) + rf32(om + 0x38);
|
||||
|
||||
const bb_post = combined_flags & 0x78;
|
||||
switch (bb_post) {
|
||||
0x08 => {
|
||||
const had_anim = (combined_flags & 0x280) != 0;
|
||||
if (!had_anim) {
|
||||
wf32(om, 0);
|
||||
wf32(om + 0x04, 0);
|
||||
wf32(om + 0x08, -1);
|
||||
wf32(om + 0x10, 1);
|
||||
wf32(om + 0x14, 0);
|
||||
wf32(om + 0x18, 0);
|
||||
wf32(om + 0x20, 0);
|
||||
wf32(om + 0x24, 1);
|
||||
wf32(om + 0x28, 0);
|
||||
} else {
|
||||
// Row 0 = {local_e4, local_e0, -local_e8}
|
||||
const r0x: f32 = @floatCast(local_mat2[1]);
|
||||
const r0y: f32 = @floatCast(local_mat2[2]);
|
||||
const r0z: f32 = @floatCast(-local_mat2[0]);
|
||||
wf32(om, r0x);
|
||||
wf32(om + 0x04, r0y);
|
||||
wf32(om + 0x08, r0z);
|
||||
normalizeVec3InPlace(om);
|
||||
const r1x: f32 = @floatCast(local_mat2[5]);
|
||||
const r1y: f32 = @floatCast(local_mat2[6]);
|
||||
const r1z: f32 = @floatCast(-local_mat2[4]);
|
||||
wf32(om + 0x10, r1x);
|
||||
wf32(om + 0x14, r1y);
|
||||
wf32(om + 0x18, r1z);
|
||||
normalizeVec3InPlace(om + 0x10);
|
||||
const r2x: f32 = @floatCast(local_mat2[9]);
|
||||
const r2y: f32 = @floatCast(local_mat2[10]);
|
||||
const r2z: f32 = @floatCast(-local_mat2[8]);
|
||||
wf32(om + 0x20, r2x);
|
||||
wf32(om + 0x24, r2y);
|
||||
wf32(om + 0x28, r2z);
|
||||
normalizeVec3InPlace(om + 0x20);
|
||||
}
|
||||
},
|
||||
0x10 => {
|
||||
normalizeVec3InPlace(om);
|
||||
const r0x = rf32(om);
|
||||
const r0y = rf32(om + 0x04);
|
||||
wf32(om + 0x10, r0y);
|
||||
wf32(om + 0x14, -r0x);
|
||||
wf32(om + 0x18, 0);
|
||||
normalizeVec3InPlace(om + 0x10);
|
||||
wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18));
|
||||
wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10));
|
||||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||||
},
|
||||
0x20 => {
|
||||
normalizeVec3InPlace(om + 0x10);
|
||||
wf32(om, -rf32(om + 0x14));
|
||||
wf32(om + 0x04, rf32(om + 0x10));
|
||||
wf32(om + 0x08, 0);
|
||||
normalizeVec3InPlace(om);
|
||||
wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18));
|
||||
wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10));
|
||||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||||
},
|
||||
0x40 => {
|
||||
normalizeVec3InPlace(om + 0x20);
|
||||
wf32(om + 0x10, rf32(om + 0x24));
|
||||
wf32(om + 0x14, -rf32(om + 0x20));
|
||||
wf32(om + 0x18, 0);
|
||||
normalizeVec3InPlace(om + 0x10);
|
||||
wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14));
|
||||
wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18));
|
||||
wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10));
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
|
||||
// Apply scale lengths back and recompute translation
|
||||
wf32(om + 0x0C, 0);
|
||||
wf32(om + 0x1C, 0);
|
||||
wf32(om + 0x2C, 0);
|
||||
const r0x_s = rf32(om);
|
||||
wf32(om, scale_len0 * r0x_s);
|
||||
const r0y_s = rf32(om + 0x04);
|
||||
wf32(om + 0x04, scale_len0 * r0y_s);
|
||||
const r0z_s = rf32(om + 0x08);
|
||||
wf32(om + 0x08, scale_len0 * r0z_s);
|
||||
const r1x_s = rf32(om + 0x10);
|
||||
wf32(om + 0x10, scale_len1 * r1x_s);
|
||||
const r1y_s = rf32(om + 0x14);
|
||||
wf32(om + 0x14, scale_len1 * r1y_s);
|
||||
const r1z_s = rf32(om + 0x18);
|
||||
wf32(om + 0x18, scale_len1 * r1z_s);
|
||||
const r2x_s = rf32(om + 0x20);
|
||||
wf32(om + 0x20, scale_len2 * r2x_s);
|
||||
const r2y_s = rf32(om + 0x24);
|
||||
wf32(om + 0x24, scale_len2 * r2y_s);
|
||||
const r2z_s = rf32(om + 0x28);
|
||||
wf32(om + 0x28, scale_len2 * r2z_s);
|
||||
|
||||
// Accumulation order must match original x87: row0 + row2 + row1
|
||||
// (pivot_x, then pivot_z, then pivot_y). f32 addition isn't associative —
|
||||
// this ordering is load-bearing for spell-effect z-fighting avoidance.
|
||||
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len2 * r2x_s * bpz + scale_len1 * r1x_s * bpy));
|
||||
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len2 * r2y_s * bpz + scale_len1 * r1y_s * bpy));
|
||||
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len2 * r2z_s * bpz + scale_len1 * r1z_s * bpy));
|
||||
wf32(om + 0x3C, 1.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sections 8-13: delegated to bone_sse's f32 implementations.
|
||||
texAnimLoop(this, model_hdr, frame_ctr);
|
||||
colorAnimLoop(this, model_hdr, frame_ctr);
|
||||
wordAnimLoop(this, model_hdr, frame_ctr);
|
||||
boneKeyframeLoop(this, model_hdr);
|
||||
particleLoops(this, model_hdr, frame_ctr);
|
||||
attachmentRecursion64(this, model_hdr, bone_out_base, frame_ctr);
|
||||
|
||||
wu32(this + SO.sync_value, ru32(anim_ctx + 0x10));
|
||||
}
|
||||
@@ -38,6 +38,7 @@ pub fn isActive() bool {
|
||||
// =============================================================================
|
||||
|
||||
const bone_sse = @import("bone_sse.zig");
|
||||
const bone_sse64 = @import("bone_sse64.zig");
|
||||
const particle_sse = @import("particle_sse.zig");
|
||||
const clip_sse = @import("clip_sse.zig");
|
||||
const cull_sse = @import("cull_sse.zig");
|
||||
@@ -195,6 +196,72 @@ fn destroyObjMgrDetour() callconv(hook.cc.stdcall) void {
|
||||
destroy_objmgr_hook.callOriginal(.{});
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Spell ground effect diagnostic — safe cross-reference approach
|
||||
// createModelAttachment saves model ptrs, ManageRenderListNode checks matches.
|
||||
// No deferred pointer reads — only value comparisons.
|
||||
// =============================================================================
|
||||
|
||||
const ModelAttachFn = fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
var model_attach_hook: hook.Detour(ModelAttachFn) = .{};
|
||||
|
||||
const MAX_TRACKED = 16;
|
||||
var tracked_models: [MAX_TRACKED]u32 = [_]u32{0} ** MAX_TRACKED;
|
||||
var tracked_next: u32 = 0;
|
||||
|
||||
fn modelAttachDetour(parent: u32, path_ptr: u32, flags: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
const result = model_attach_hook.callOriginal(.{ parent, path_ptr, flags });
|
||||
|
||||
if (path_ptr != 0 and result != 0) {
|
||||
const path: [*]const u8 = @ptrFromInt(path_ptr);
|
||||
if (path[0] == 'S' and path[1] == 'p' and path[2] == 'e' and path[3] == 'l' and path[4] == 'l' and path[5] == 's') {
|
||||
const path_z: [*:0]const u8 = @ptrFromInt(path_ptr);
|
||||
log.fmt("[spell] created 0x{x}: {s}\n", .{ result, path_z });
|
||||
tracked_models[tracked_next % MAX_TRACKED] = result;
|
||||
tracked_next +%= 1;
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// CM2Model_ManageRenderListNode (0x710B90)
|
||||
// __thiscall(ECX=model, add_remove), RET 0x4
|
||||
const ManageRLFn = fn (u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
var manage_rl_hook: hook.Detour(ManageRLFn) = .{};
|
||||
|
||||
fn manageRLDetour(model: u32, add_remove: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||||
for (&tracked_models) |tp| {
|
||||
if (tp != 0 and tp == model) {
|
||||
if (add_remove != 0) {
|
||||
log.fmt("[spell] 0x{x} ADDED to render list\n", .{model});
|
||||
} else {
|
||||
log.fmt("[spell] 0x{x} REMOVED from render list\n", .{model});
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
manage_rl_hook.callOriginal(.{ model, add_remove });
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// ProcessProjectileMovementWithCollisionAndTargetValidation fix (0x61e1d0)
|
||||
// __thiscall(ECX=missile, target_ptr, param_3), RET 0x8
|
||||
//
|
||||
// Vanilla bug: when target_ptr != 0 (a unit/dynobj exists at the AoE location),
|
||||
// the code takes an alternate path that skips ProcessMissileSpellEffects entirely.
|
||||
// This means the area effect ground model (e.g. InfectedSecretion_Marked.m2) is
|
||||
// never created. Fix: force target_ptr=0 so the area effect path always runs.
|
||||
// Target-unit visuals fire separately through ProcessSpellVisualKit.
|
||||
// =============================================================================
|
||||
|
||||
const ProjMoveFn = fn (u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
var proj_move_hook: hook.Detour(ProjMoveFn) = .{};
|
||||
|
||||
fn projMoveDetour(missile: u32, target_ptr: u32, param_3: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||||
_ = target_ptr;
|
||||
proj_move_hook.callOriginal(.{ missile, 0, param_3 });
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// OnWorldUpdate hook (0x482EA0) — per-frame cache reset
|
||||
// =============================================================================
|
||||
@@ -298,7 +365,7 @@ pub fn installHooks() void {
|
||||
var installed: u32 = 0;
|
||||
|
||||
// Bone transform SSE
|
||||
if (transform_hook.attach(0x714260, &bone_sse.transformImpl_SSE) == .ok) installed += 1;
|
||||
if (transform_hook.attach(0x714260, &bone_sse64.transformImpl_SSE64) == .ok) installed += 1;
|
||||
|
||||
// Frustum clip SSE (1.9x speedup)
|
||||
if (clip_hook.attach(0x6318C0, &clip_sse.clipPolygonToSinglePlane) == .ok) installed += 1;
|
||||
@@ -306,19 +373,16 @@ pub fn installHooks() void {
|
||||
// Particle rendering SSE
|
||||
if (particle_hook.attach(0x7B2A50, &particleDetour) == .ok) installed += 1;
|
||||
|
||||
// Glyph cache
|
||||
// Glyph cache removed -- game has internal glyph cache, our hook only sees misses (~30/frame)
|
||||
|
||||
// GUID lookup cache -- A/B testing via transform44
|
||||
// if (findguid_hook.attach(0x464890, &findguidDetour) == .ok) installed += 1;
|
||||
// if (obj_delete_hook.attach(0x464920, &objDeleteDetour) == .ok) installed += 1;
|
||||
|
||||
// GUID cache disabled for now
|
||||
// if (destroy_objmgr_hook.attach(0x467700, &destroyObjMgrDetour) == .ok) installed += 1;
|
||||
|
||||
// Per-frame cache reset
|
||||
if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) installed += 1;
|
||||
|
||||
// Spell ground effect diagnostics
|
||||
if (model_attach_hook.attach(0x707350, &modelAttachDetour) == .ok) installed += 1;
|
||||
if (manage_rl_hook.attach(0x710B90, &manageRLDetour) == .ok) installed += 1;
|
||||
|
||||
// Area effect ground model fix
|
||||
if (proj_move_hook.attach(0x61e1d0, &projMoveDetour) == .ok) installed += 1;
|
||||
|
||||
// Silicon SSE binary patches
|
||||
_ = installPatches();
|
||||
|
||||
@@ -358,6 +422,9 @@ pub fn removeHooks() void {
|
||||
obj_delete_hook.detach();
|
||||
destroy_objmgr_hook.detach();
|
||||
world_update_hook.detach();
|
||||
model_attach_hook.detach();
|
||||
manage_rl_hook.detach();
|
||||
proj_move_hook.detach();
|
||||
log.close();
|
||||
mod_mutex.release(&g_mutex);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user