- Replace all reimplemented game functions with actual game calls: vec3_sqmag (0x4549F0), __ftol (0x40A2B0), getIndexOffset (0x71AFF0), setShortValue (0x71B010) — matching assembly exactly - Fix 3 wrong hardcoded constants that differ at runtime from Ghidra static values: SHORT_TO_FLOAT: 0x38000000→0x38000100 (1/32767 not 1/32768) BILLBOARD_EPSILON: 0x3727c5ac→0x34800000 HERMITE_5: 5.0→6.0 All now read from game memory at runtime - Fix timestamp delta guard (this+0x4C): was guarding on stored value, assembly guards on anim_ctx pointer — prevents first-frame initialization - Change REF calling convention to thiscall matching original - Add comprehensive memory comparison diagnostic (original vs REF) - Disable interpKfDetour hook (was pure passthrough)
2194 lines
103 KiB
Zig
2194 lines
103 KiB
Zig
//! SSE-optimized transformMatrix4x4 reimplementation.
|
||
//!
|
||
//! Full standalone replacement for the 17703-byte bone transform engine at 0x714260.
|
||
//! Compiled ReleaseFast even in Debug builds (separate compilation unit pattern).
|
||
//! All helper functions (findInterpolationIndices, interpolateAnimationKeyframes,
|
||
//! scaleMatrix3x3ByVector, ApplyTranslationMatrix, rotateMatrixByQuaternion) are
|
||
//! reimplemented inline — no calls back to original game code.
|
||
//!
|
||
//! Only external call: the original transformMatrix4x4 via the hook's callOriginal
|
||
//! for attachment recursion (the detour auto-dispatches to this SSE version).
|
||
|
||
const V4 = @Vector(4, f32);
|
||
|
||
// DEBUG: FPU state logging
|
||
export var dbg_fpu_logged: u32 = 0;
|
||
export var dbg_fpu_value: u16 = 0;
|
||
|
||
|
||
// =============================================================================
|
||
// SceneObject field offsets — assembly-verified from [EBX+N] in transformMatrix4x4
|
||
// =============================================================================
|
||
|
||
const SO = struct {
|
||
const model_data_ptr: u32 = 0x010;
|
||
const anim_ctx_ptr: u32 = 0x02C; // +0xC=timestamp, +0x10=sync_value
|
||
const model_ctr_ptr: u32 = 0x030; // +0x130=M2 header
|
||
const sync_value: u32 = 0x040;
|
||
const search_data_base: u32 = 0x04C; // prev timestamp for delta
|
||
const emitter_flag: u32 = 0x050;
|
||
const gs_values_ptr: u32 = 0x064; // pointer to global sequence value array
|
||
const gs_time_base: u32 = 0x068; // subtracted from timestamp for GS
|
||
const child_padding: u32 = 0x084;
|
||
const anim_frame_ctr: u32 = 0x08C;
|
||
const bone_rt_base: u32 = 0x090; // array of 0x118-byte bone runtime structs
|
||
const bone_out_ptr: u32 = 0x094; // output bone matrices
|
||
const tex_anim_out: u32 = 0x0A0;
|
||
const color_anim_out: u32 = 0x0A8;
|
||
const scale1: u32 = 0x0AC;
|
||
const scale2: u32 = 0x0B0;
|
||
const scale3: u32 = 0x0B4;
|
||
const bb_row0: u32 = 0x0FC; // billboard matrix row 0 (camera forward)
|
||
const world_xform: u32 = 0x10C; // float[16] world transform
|
||
const field_17c: u32 = 0x17C;
|
||
const field_180: u32 = 0x180;
|
||
const field_184: u32 = 0x184;
|
||
const field_188: u32 = 0x188;
|
||
const field_18c: u32 = 0x18C;
|
||
const field_190: u32 = 0x190;
|
||
const render_scale_x: u32 = 0x194;
|
||
const render_scale_y: u32 = 0x198;
|
||
const render_scale_z: u32 = 0x19C;
|
||
const world_pos: u32 = 0x1A0; // Vec3 (passed as param_3 to children)
|
||
const render_pri: u32 = 0x1AC; // Vec3 (passed as param_4 to children)
|
||
const hierarchy_ptr: u32 = 0x1C8;
|
||
const emitter_ctx: u32 = 0x1CC;
|
||
const field_1d8: u32 = 0x1D8;
|
||
const hierarchy_idx: u32 = 0x1DC;
|
||
const field_200: u32 = 0x200;
|
||
const particle1: u32 = 0x3C4;
|
||
const particle2: u32 = 0x3C8;
|
||
const particle3: u32 = 0x3D0;
|
||
const particle4: u32 = 0x3D4;
|
||
const add_remaining: u32 = 0x3D8;
|
||
};
|
||
|
||
// Bone runtime struct offsets (within 0x118-byte per-bone runtime)
|
||
const BR = struct {
|
||
// Translation interpolation state
|
||
const trans_idx0: u32 = 0x00; // [0] lower keyframe index
|
||
const trans_idx1: u32 = 0x04; // [1] upper keyframe index
|
||
const trans_t: u32 = 0x08; // [2] interpolation factor (float bits)
|
||
const trans_x: u32 = 0x0C; // [3] interpolated translation X
|
||
const trans_y: u32 = 0x10; // [4] Y
|
||
const trans_z: u32 = 0x14; // [5] Z
|
||
// Secondary translation (crossfade)
|
||
const trans2_idx0: u32 = 0x18;
|
||
const trans2_idx1: u32 = 0x1C;
|
||
const trans2_t: u32 = 0x20;
|
||
const trans2_x: u32 = 0x24;
|
||
const trans2_y: u32 = 0x28;
|
||
const trans2_z: u32 = 0x2C;
|
||
// Scale interpolation state (at puVar20 + 0x1a = offset 0x68)
|
||
const scale_idx0: u32 = 0x68;
|
||
const scale_idx1: u32 = 0x6C;
|
||
const scale_t: u32 = 0x70;
|
||
const scale_x: u32 = 0x74;
|
||
const scale_y: u32 = 0x78;
|
||
const scale_z: u32 = 0x7C;
|
||
const scale2_idx0: u32 = 0x80;
|
||
const scale2_idx1: u32 = 0x84;
|
||
const scale2_t: u32 = 0x88;
|
||
const scale2_x: u32 = 0x8C;
|
||
const scale2_y: u32 = 0x90;
|
||
const scale2_z: u32 = 0x94;
|
||
// Primary animation time range
|
||
const prim_time: u32 = 0x98; // puVar20[0x26]
|
||
const prim_track: u32 = 0x9C; // puVar20[0x27]
|
||
const prim_anim: u32 = 0xA0; // puVar20[0x28]
|
||
const anim_slot: u32 = 0xA4; // puVar20[0x29] - animation slot index
|
||
// Secondary animation time range (crossfade)
|
||
const sec_start: u32 = 0xA8; // puVar20[0x2a]
|
||
const sec_end: u32 = 0xAC; // puVar20[0x2b]
|
||
const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion
|
||
const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e]
|
||
// Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30)
|
||
const rot_idx0: u32 = 0x30;
|
||
const rot_idx1: u32 = 0x34;
|
||
const rot_t: u32 = 0x38;
|
||
const rot_x: u32 = 0x3C;
|
||
const rot_y: u32 = 0x40;
|
||
const rot_z: u32 = 0x44;
|
||
const rot_w: u32 = 0x48;
|
||
// Secondary rotation
|
||
const rot2_idx0: u32 = 0x4C;
|
||
const rot2_idx1: u32 = 0x50;
|
||
const rot2_t: u32 = 0x54;
|
||
const rot2_x: u32 = 0x58;
|
||
const rot2_y: u32 = 0x5C;
|
||
const rot2_z: u32 = 0x60;
|
||
const rot2_w: u32 = 0x64;
|
||
// Secondary time range
|
||
const sec_time: u32 = 0xC4; // puVar20[0x31]
|
||
const sec_track: u32 = 0xC8; // puVar20[0x32]
|
||
const sec_slot: u32 = 0xD0; // puVar20[0x34]
|
||
const sec_start2: u32 = 0xD4; // puVar20[0x35]
|
||
const sec_end2: u32 = 0xD8; // puVar20[0x36]
|
||
const sec_offset2: u32 = 0xE4; // puVar20[0x39]
|
||
// Flags and weights
|
||
const flags2: u32 = 0xF4; // puVar20[0x3d]
|
||
const crossfade_end: u32 = 0x100; // puVar20[0x40]
|
||
const crossfade_inv: u32 = 0x104; // puVar20[0x41]
|
||
const crossfade_weight: u32 = 0x108; // puVar20[0x42]
|
||
const blend_weight: u32 = 0x10C; // puVar20[0x43] - blend weight for crossfade
|
||
const bone_flag_cache: u32 = 0xF0; // puVar20[0x3c]
|
||
};
|
||
|
||
// OldAnimationBlock struct offsets (28 bytes = 0x1C per track in v256 M2)
|
||
// Layout verified from M2 format + decompilation cross-reference:
|
||
// pMVar23->m31 (bone_def+0x34) = rot block+0x0C = nTimestamps (gates rotation)
|
||
// pMVar23->m12 (bone_def+0x18) = trans block+0x0C = nTimestamps (gates translation)
|
||
// pMVar23[1].m10 (bone_def+0x50) = scale block+0x0C = nTimestamps (gates scale)
|
||
const AD = struct {
|
||
const interp_mode: u32 = 0x00; // u16: interpolation mode (0=none, 1=lerp)
|
||
const time_index: u32 = 0x02; // i16: global sequence index (-1 = none)
|
||
const track_count_flag: u32 = 0x04; // nRanges: 0 = single track
|
||
const keyframe_ranges: u32 = 0x08; // ofsRanges: ptr to per-track range pairs
|
||
const keyframe_count: u32 = 0x0C; // nTimestamps: total keyframe count
|
||
const timestamps_ptr: u32 = 0x10; // ofsTimestamps: ptr to timestamp array
|
||
const nvalues: u32 = 0x14; // nValues: number of value entries
|
||
const keyframe_base: u32 = 0x18; // ofsValues: ptr to keyframe data
|
||
};
|
||
|
||
// M2CompBone struct offsets (0x6C = 108 bytes per bone in v256 model)
|
||
// Layout: 12 bytes fixed header + 3x28 byte OldAnimationBlock tracks + 12 bytes pivot
|
||
// Track order: translation, rotation, scale (standard M2 order)
|
||
const BD = struct {
|
||
const key_id: u32 = 0x00; // i32: key bone ID
|
||
const flags: u32 = 0x04; // u32: bone flags (billboard type in bits 0-6, etc.)
|
||
const parent_bone: u32 = 0x08; // i16 at low bytes, submesh_id u16 at high bytes
|
||
// Translation OldAnimationBlock (28 bytes, +0x0C to +0x27)
|
||
const trans_anim: u32 = 0x0C;
|
||
const trans_nts: u32 = 0x18; // nTimestamps — gates translation interpolation
|
||
// Rotation OldAnimationBlock (28 bytes, +0x28 to +0x43)
|
||
const rot_anim: u32 = 0x28;
|
||
const rot_nts: u32 = 0x34; // nTimestamps — gates rotation interpolation
|
||
// Scale OldAnimationBlock (28 bytes, +0x44 to +0x5F)
|
||
const scale_anim: u32 = 0x44;
|
||
const scale_nts: u32 = 0x50; // nTimestamps — gates scale interpolation
|
||
// Pivot point (12 bytes, +0x60 to +0x6B)
|
||
const pivot_x: u32 = 0x60;
|
||
const pivot_y: u32 = 0x64;
|
||
const pivot_z: u32 = 0x68;
|
||
};
|
||
|
||
// Game constants
|
||
const ZERO_F: f32 = 0.0;
|
||
const ONE_F: f32 = 1.0;
|
||
const THREE_F: f32 = 3.0;
|
||
// getBillboardEpsilon(): read from game memory (runtime 0x34800000, NOT static 0x3727c5ac from Ghidra)
|
||
fn getBillboardEpsilon() f32 {
|
||
return rf32(0x008029d4);
|
||
}
|
||
// getShortToFloat(): read from game memory at 0x00811610 (runtime value is 0x38000100 = 1/32767,
|
||
// NOT the static 0x38000000 = 1/32768 from Ghidra). The game patches this at startup.
|
||
fn getShortToFloat() f32 {
|
||
return rf32(0x00811610);
|
||
}
|
||
const HERMITE_3: f32 = 3.0; // DAT_0080297c
|
||
// getHermite5(): runtime value is 0x40c00000 (6.0), NOT static 0x40a00000 (5.0) from Ghidra
|
||
fn getHermite5() f32 {
|
||
return rf32(0x00802990);
|
||
}
|
||
|
||
// MSVC CRT sin/cos — linked from the WoW process
|
||
extern fn sinf(f32) f32;
|
||
extern fn cosf(f32) f32;
|
||
|
||
// Original transformMatrix4x4 for recursive attachment calls.
|
||
// The hook's detour will auto-dispatch to our SSE version.
|
||
const OrigTransformFn = *const fn (u32, u32, u32, u32, u32) callconv(.c) void;
|
||
|
||
// =============================================================================
|
||
// Memory access helpers
|
||
// =============================================================================
|
||
|
||
inline fn ru32(addr: u32) u32 {
|
||
return @as(*align(1) const u32, @ptrFromInt(addr)).*;
|
||
}
|
||
inline fn ri32(addr: u32) i32 {
|
||
return @as(*align(1) const i32, @ptrFromInt(addr)).*;
|
||
}
|
||
inline fn rf32(addr: u32) f32 {
|
||
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
|
||
}
|
||
inline fn ru16(addr: u32) u16 {
|
||
return @as(*align(1) const u16, @ptrFromInt(addr)).*;
|
||
}
|
||
inline fn ri16(addr: u32) i16 {
|
||
return @as(*align(1) const i16, @ptrFromInt(addr)).*;
|
||
}
|
||
inline fn ru8(addr: u32) u8 {
|
||
return @as(*const u8, @ptrFromInt(addr)).*;
|
||
}
|
||
|
||
inline fn wu32(addr: u32, v: u32) void {
|
||
@as(*align(1) u32, @ptrFromInt(addr)).* = v;
|
||
}
|
||
inline fn wf32(addr: u32, v: f32) void {
|
||
@as(*align(1) f32, @ptrFromInt(addr)).* = v;
|
||
}
|
||
inline fn wu16(addr: u32, v: u16) void {
|
||
@as(*align(1) u16, @ptrFromInt(addr)).* = v;
|
||
}
|
||
inline fn wu8(addr: u32, v: u8) void {
|
||
@as(*u8, @ptrFromInt(addr)).* = v;
|
||
}
|
||
|
||
inline fn fbits(v: f32) u32 {
|
||
return @bitCast(v);
|
||
}
|
||
inline fn ufloat(v: u32) f32 {
|
||
return @bitCast(v);
|
||
}
|
||
|
||
// =============================================================================
|
||
// Math helpers — using @Vector(4, f32) for SSE
|
||
// =============================================================================
|
||
|
||
inline fn splat(v: f32) V4 {
|
||
return @splat(v);
|
||
}
|
||
|
||
/// 3-component lerp: a + (b - a) * t. Keyframes are 12 bytes (3 floats) apart.
|
||
inline fn lerpVec3(a_addr: u32, b_addr: u32, t: f32) [3]f32 {
|
||
const ax = rf32(a_addr);
|
||
const ay = rf32(a_addr + 4);
|
||
const az = rf32(a_addr + 8);
|
||
const bx = rf32(b_addr);
|
||
const by = rf32(b_addr + 4);
|
||
const bz = rf32(b_addr + 8);
|
||
return .{
|
||
(bx - ax) * t + ax,
|
||
(by - ay) * t + ay,
|
||
(bz - az) * t + az,
|
||
};
|
||
}
|
||
|
||
/// Blend primary and secondary results: primary + (secondary - primary) * weight
|
||
inline fn blendVec3(primary: [3]f32, secondary: [3]f32, weight: f32) [3]f32 {
|
||
return .{
|
||
(secondary[0] - primary[0]) * weight + primary[0],
|
||
(secondary[1] - primary[1]) * weight + primary[1],
|
||
(secondary[2] - primary[2]) * weight + primary[2],
|
||
};
|
||
}
|
||
|
||
/// Scale 3x3 rotation portion of a row-major 4x4 matrix by per-axis scale.
|
||
/// Row 0 *= scale.x, Row 1 *= scale.y, Row 2 *= scale.z
|
||
inline fn scaleMatrix3x3(mat: u32, sx: f32, sy: f32, sz: f32) void {
|
||
// Row 0 (offsets 0x00, 0x04, 0x08)
|
||
wf32(mat + 0x00, rf32(mat + 0x00) * sx);
|
||
wf32(mat + 0x04, rf32(mat + 0x04) * sx);
|
||
wf32(mat + 0x08, rf32(mat + 0x08) * sx);
|
||
// Row 1 (offsets 0x10, 0x14, 0x18)
|
||
wf32(mat + 0x10, rf32(mat + 0x10) * sy);
|
||
wf32(mat + 0x14, rf32(mat + 0x14) * sy);
|
||
wf32(mat + 0x18, rf32(mat + 0x18) * sy);
|
||
// Row 2 (offsets 0x20, 0x24, 0x28)
|
||
wf32(mat + 0x20, rf32(mat + 0x20) * sz);
|
||
wf32(mat + 0x24, rf32(mat + 0x24) * sz);
|
||
wf32(mat + 0x28, rf32(mat + 0x28) * sz);
|
||
}
|
||
|
||
/// Apply translation through rotation matrix:
|
||
/// mat[3][0] += dot(mat[0], t)
|
||
/// mat[3][1] += dot(mat[1], t)
|
||
/// mat[3][2] += dot(mat[2], t)
|
||
inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void {
|
||
wf32(mat + 0x30, tx * rf32(mat + 0x00) + ty * rf32(mat + 0x10) + tz * rf32(mat + 0x20) + rf32(mat + 0x30));
|
||
wf32(mat + 0x34, tx * rf32(mat + 0x04) + ty * rf32(mat + 0x14) + tz * rf32(mat + 0x24) + rf32(mat + 0x34));
|
||
wf32(mat + 0x38, tx * rf32(mat + 0x08) + ty * rf32(mat + 0x18) + tz * rf32(mat + 0x28) + rf32(mat + 0x38));
|
||
}
|
||
|
||
/// Quaternion → rotation matrix: OVERWRITES mat with the rotation matrix.
|
||
/// Matches the original game function at 0x74B6BB which writes directly
|
||
/// without multiplying by existing matrix contents.
|
||
/// Used in the bone loop where the matrix starts as identity.
|
||
inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||
const xx2 = qx * (qx + qx);
|
||
const xy2 = qx * (qy + qy);
|
||
const xz2 = qx * (qz + qz);
|
||
const yy2 = qy * (qy + qy);
|
||
const yz2 = qy * (qz + qz);
|
||
const zz2 = qz * (qz + qz);
|
||
const wx2 = qw * (qx + qx);
|
||
const wy2 = qw * (qy + qy);
|
||
const wz2 = qw * (qz + qz);
|
||
|
||
// Row 0
|
||
wf32(mat + 0x00, 1.0 - (yy2 + zz2));
|
||
wf32(mat + 0x04, xy2 + wz2);
|
||
wf32(mat + 0x08, xz2 - wy2);
|
||
wf32(mat + 0x0C, 0);
|
||
// Row 1
|
||
wf32(mat + 0x10, xy2 - wz2);
|
||
wf32(mat + 0x14, 1.0 - (xx2 + zz2));
|
||
wf32(mat + 0x18, yz2 + wx2);
|
||
wf32(mat + 0x1C, 0);
|
||
// Row 2
|
||
wf32(mat + 0x20, xz2 + wy2);
|
||
wf32(mat + 0x24, yz2 - wx2);
|
||
wf32(mat + 0x28, 1.0 - (xx2 + yy2));
|
||
wf32(mat + 0x2C, 0);
|
||
// Row 3 (translation = zero, w = 1)
|
||
wf32(mat + 0x30, 0);
|
||
wf32(mat + 0x34, 0);
|
||
wf32(mat + 0x38, 0);
|
||
wf32(mat + 0x3C, 1);
|
||
}
|
||
|
||
/// Quaternion → rotation matrix, then multiply: mat = quat_rot * mat.
|
||
/// Standard quat→mat conversion + SSE 4x4 matrix multiply.
|
||
/// Used in bone keyframe processing where matrix already has content.
|
||
inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||
const xx2 = qx * (qx + qx);
|
||
const xy2 = qx * (qy + qy);
|
||
const xz2 = qx * (qz + qz);
|
||
const yy2 = qy * (qy + qy);
|
||
const yz2 = qy * (qz + qz);
|
||
const zz2 = qz * (qz + qz);
|
||
const wx2 = qw * (qx + qx);
|
||
const wy2 = qw * (qy + qy);
|
||
const wz2 = qw * (qz + qz);
|
||
|
||
// Rotation matrix from quaternion (row-major)
|
||
const rot: [16]f32 = .{
|
||
1.0 - (yy2 + zz2), xy2 + wz2, xz2 - wy2, 0,
|
||
xy2 - wz2, 1.0 - (xx2 + zz2), yz2 + wx2, 0,
|
||
xz2 + wy2, yz2 - wx2, 1.0 - (xx2 + yy2), 0,
|
||
0, 0, 0, 1,
|
||
};
|
||
|
||
// SSE matrix multiply: result = rot * mat
|
||
var tmp: [16]f32 = undefined;
|
||
const r0: V4 = .{ rf32(mat + 0x00), rf32(mat + 0x04), rf32(mat + 0x08), rf32(mat + 0x0C) };
|
||
const r1: V4 = .{ rf32(mat + 0x10), rf32(mat + 0x14), rf32(mat + 0x18), rf32(mat + 0x1C) };
|
||
const r2: V4 = .{ rf32(mat + 0x20), rf32(mat + 0x24), rf32(mat + 0x28), rf32(mat + 0x2C) };
|
||
const r3: V4 = .{ rf32(mat + 0x30), rf32(mat + 0x34), rf32(mat + 0x38), rf32(mat + 0x3C) };
|
||
|
||
inline for (0..4) |i| {
|
||
const b = i * 4;
|
||
const out = splat(rot[b]) * r0 + splat(rot[b + 1]) * r1 + splat(rot[b + 2]) * r2 + splat(rot[b + 3]) * r3;
|
||
tmp[b] = out[0];
|
||
tmp[b + 1] = out[1];
|
||
tmp[b + 2] = out[2];
|
||
tmp[b + 3] = out[3];
|
||
}
|
||
|
||
// Copy back
|
||
inline for (0..16) |i| {
|
||
wf32(mat + @as(u32, @intCast(i)) * 4, tmp[i]);
|
||
}
|
||
}
|
||
|
||
/// Copy 16 floats (4x4 matrix)
|
||
inline fn copyMat4(dst: u32, src: u32) void {
|
||
comptime var i: u32 = 0;
|
||
inline while (i < 64) : (i += 4) {
|
||
wu32(dst + i, ru32(src + i));
|
||
}
|
||
}
|
||
|
||
/// Set identity matrix (16 floats)
|
||
inline fn setIdentity(dst: u32) void {
|
||
inline for (0..16) |i| {
|
||
const val: f32 = if (i == 0 or i == 5 or i == 10 or i == 15) 1.0 else 0.0;
|
||
wf32(dst + @as(u32, @intCast(i)) * 4, val);
|
||
}
|
||
}
|
||
|
||
/// Normalize a 3-component vector in memory at addr.
|
||
/// Calls game's vec3 squared magnitude (0x4549F0), then sqrt, epsilon check, divide.
|
||
/// Assembly pattern: CALL 0x4549F0 → FSQRT → FABS → FCOMP → FLD1 → FDIVRP → FMUL×3
|
||
inline fn normalizeVec3InPlace(addr: u32) void {
|
||
const sq_mag = callVec3SqMag(addr);
|
||
const len = @sqrt(sq_mag);
|
||
if (@abs(len) >= getBillboardEpsilon()) {
|
||
const inv = 1.0 / len;
|
||
wf32(addr, rf32(addr) * inv);
|
||
wf32(addr + 4, rf32(addr + 4) * inv);
|
||
wf32(addr + 8, rf32(addr + 8) * inv);
|
||
}
|
||
}
|
||
|
||
/// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small.
|
||
/// Writes vec3 to stack local and calls game's vec3 squared magnitude (0x4549F0).
|
||
inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
|
||
var v: [3]f32 = .{ x, y, z };
|
||
const sq_mag = callVec3SqMag(@intFromPtr(&v));
|
||
const len = @sqrt(sq_mag);
|
||
if (len < getBillboardEpsilon()) return .{ x, y, z };
|
||
const inv = 1.0 / len;
|
||
return .{ x * inv, y * inv, z * inv };
|
||
}
|
||
|
||
/// Cross product of two 3-component vectors
|
||
inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32 {
|
||
return .{
|
||
ay * bz - az * by,
|
||
az * bx - ax * bz,
|
||
ax * by - ay * bx,
|
||
};
|
||
}
|
||
|
||
// =============================================================================
|
||
// findInterpolationIndices — reimplemented from 0x713d50 (334 bytes)
|
||
//
|
||
// Three-tier search with temporal coherence:
|
||
// 1. Forward linear scan (hot path, 1-4 iterations typical)
|
||
// 2. Backward linear scan (negative delta)
|
||
// 3. Binary search (fallback)
|
||
//
|
||
// Output: indices[0] = lower index, [1] = upper index, [2] = interpolation t (float bits)
|
||
// =============================================================================
|
||
|
||
/// Calls game's findInterpolationIndices at 0x713D50.
|
||
/// __thiscall(ECX=this, stack: search_value, track_index, anim_data, output)
|
||
fn findInterpIdx(
|
||
this: u32,
|
||
search_value: u32,
|
||
track_index: u32,
|
||
anim_data: u32,
|
||
output: u32,
|
||
) void {
|
||
const gameFn: *const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x713D50);
|
||
gameFn(this, 0, search_value, track_index, anim_data, output);
|
||
}
|
||
|
||
// =============================================================================
|
||
// interpolateAnimationKeyframes — reimplemented from 0x713ea0
|
||
//
|
||
// Calls findInterpIdx, does 4-component lerp (for quaternions).
|
||
// If crossfade active, does secondary lookup + blend.
|
||
// Output buffer layout: [idx0, idx1, t, x, y, z, w, sec_idx0, sec_idx1, sec_t, sx, sy, sz, sw]
|
||
// =============================================================================
|
||
|
||
/// Calls game's interpolateAnimationKeyframes at 0x713EA0.
|
||
/// __fastcall(ECX=this, EDX=bone_rt, stack: anim_data, output)
|
||
inline fn interpAnimKF(this: u32, bone_rt: u32, anim_data: u32, output: u32) void {
|
||
const gameFn: *const fn (u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x713EA0);
|
||
gameFn(this, bone_rt, anim_data, output);
|
||
}
|
||
|
||
// =============================================================================
|
||
// Game function call wrappers — replacing reimplementations with actual calls
|
||
// =============================================================================
|
||
|
||
/// Call game's __ftol at 0x40A2B0 with the exact assembly pattern:
|
||
/// FILD [delta_ptr]; FMUL [scale_addr]; CALL __ftol
|
||
/// __ftol reads ST0, returns truncated i32 in EAX, pops ST0.
|
||
/// scale_addr is a u32 address pointing to a f32 in memory (e.g., brt + 0xB0).
|
||
inline fn callFtol(delta: i32, scale_addr: u32) i32 {
|
||
var delta_copy = delta;
|
||
var result: i32 = undefined;
|
||
asm volatile ("fildl (%[delta_ptr])\n\tfmuls (%[scale_ptr])\n\tcall *%[fn_ptr]"
|
||
: [result] "={eax}" (result),
|
||
: [delta_ptr] "r" (@intFromPtr(&delta_copy)),
|
||
[scale_ptr] "r" (scale_addr),
|
||
[fn_ptr] "r" (@as(u32, 0x40A2B0)),
|
||
: .{ .edx = true }
|
||
);
|
||
return result;
|
||
}
|
||
|
||
/// Call game's vec3 squared magnitude at 0x4549F0.
|
||
/// __thiscall(ECX=vec3_ptr) → f32 in ST0 (squared magnitude, NOT length)
|
||
/// Uses inline asm to guarantee correct ST0 capture — Zig's f32 return handling
|
||
/// for x86_fastcall with SSE enabled may not emit FSTP, leaking the x87 stack.
|
||
inline fn callVec3SqMag(vec3_ptr: u32) f32 {
|
||
var result: f32 = undefined;
|
||
asm volatile ("call *%[fn_ptr]\n\tfstps (%[out])"
|
||
:
|
||
: [fn_ptr] "r" (@as(u32, 0x4549F0)),
|
||
[out] "r" (@intFromPtr(&result)),
|
||
[ecx] "{ecx}" (vec3_ptr),
|
||
: .{ .eax = true, .edx = true, .ecx = true }
|
||
);
|
||
return result;
|
||
}
|
||
|
||
/// Call game's getIndexOffset at 0x71AFF0.
|
||
/// __thiscall(ECX=table_ptr, stack: index) → u32 pointer to value
|
||
/// table_ptr = anim_data + AD.nvalues (0x14), pointing to {nValues, ofsValues}
|
||
inline fn callGetIndexOffset(table: u32, index: u32) u32 {
|
||
const func: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32 = @ptrFromInt(0x71AFF0);
|
||
return func(table, 0, index);
|
||
}
|
||
|
||
/// Call game's setShortValue at 0x71B010.
|
||
/// __thiscall(ECX=output_ptr, stack: source_ptr) → void
|
||
/// Copies a short value from source to output.
|
||
inline fn callSetShortValue(output: u32, source: u32) void {
|
||
const func: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x71B010);
|
||
func(output, 0, source);
|
||
}
|
||
|
||
/// Read a short value at keyframe index via game functions.
|
||
/// Matches assembly pattern: getIndexOffset → setShortValue → MOVSX.
|
||
inline fn readShortViaGame(table: u32, index: u32) i16 {
|
||
var result: i16 align(2) = undefined;
|
||
const ptr = callGetIndexOffset(table, index);
|
||
callSetShortValue(@intFromPtr(&result), ptr);
|
||
return result;
|
||
}
|
||
|
||
/// Interpolate a Vec3 track (12 bytes per keyframe) with crossfade support.
|
||
/// Writes result to output[3..5] (as u32 float bits). Uses output[0..2] for indices/t,
|
||
/// and output[6..11] for secondary crossfade state.
|
||
inline fn interpVec3Track(
|
||
this: u32,
|
||
bone_rt: u32,
|
||
anim_data: u32,
|
||
output: u32,
|
||
blend_weight: f32,
|
||
) void {
|
||
findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output);
|
||
|
||
const interp_mode = ri16(anim_data + AD.interp_mode);
|
||
const kf_base = ru32(anim_data + AD.keyframe_base);
|
||
|
||
if (interp_mode == 0) {
|
||
// No interpolation — copy keyframe directly
|
||
const src = kf_base + ru32(output) * 0xC;
|
||
wu32(output + 0x0C, ru32(src));
|
||
wu32(output + 0x10, ru32(src + 4));
|
||
wu32(output + 0x14, ru32(src + 8));
|
||
return;
|
||
}
|
||
|
||
const t = ufloat(ru32(output + 8));
|
||
const a = kf_base + ru32(output) * 0xC;
|
||
const b = kf_base + ru32(output + 4) * 0xC;
|
||
const result = lerpVec3(a, b, t);
|
||
wu32(output + 0x0C, fbits(result[0]));
|
||
wu32(output + 0x10, fbits(result[1]));
|
||
wu32(output + 0x14, fbits(result[2]));
|
||
|
||
// Crossfade blend
|
||
if (blend_weight != 0.0 and ri16(anim_data + AD.time_index) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x18);
|
||
const st = ufloat(ru32(output + 0x20));
|
||
const sa = kf_base + ru32(output + 0x18) * 0xC;
|
||
const sb = kf_base + ru32(output + 0x1C) * 0xC;
|
||
const sec = lerpVec3(sa, sb, st);
|
||
wu32(output + 0x24, fbits(sec[0]));
|
||
wu32(output + 0x28, fbits(sec[1]));
|
||
wu32(output + 0x2C, fbits(sec[2]));
|
||
|
||
// Blend
|
||
const pri_x = ufloat(ru32(output + 0x0C));
|
||
const pri_y = ufloat(ru32(output + 0x10));
|
||
const pri_z = ufloat(ru32(output + 0x14));
|
||
wu32(output + 0x0C, fbits((sec[0] - pri_x) * blend_weight + pri_x));
|
||
wu32(output + 0x10, fbits((sec[1] - pri_y) * blend_weight + pri_y));
|
||
wu32(output + 0x14, fbits((sec[2] - pri_z) * blend_weight + pri_z));
|
||
}
|
||
}
|
||
|
||
/// Interpolate a single float track (4 bytes per keyframe) with crossfade.
|
||
/// Writes result to output[3] as float bits.
|
||
inline fn interpFloatTrack(
|
||
this: u32,
|
||
bone_rt: u32,
|
||
anim_data: u32,
|
||
output: u32,
|
||
) void {
|
||
findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), anim_data, output);
|
||
|
||
const interp_mode = ri16(anim_data + AD.interp_mode);
|
||
const kf_base = ru32(anim_data + AD.keyframe_base);
|
||
|
||
if (interp_mode == 0) {
|
||
wu32(output + 0x0C, ru32(kf_base + ru32(output) * 4));
|
||
return;
|
||
}
|
||
|
||
const t = ufloat(ru32(output + 8));
|
||
const a = rf32(kf_base + ru32(output) * 4);
|
||
const b = rf32(kf_base + ru32(output + 4) * 4);
|
||
wf32(output + 0x0C, (b - a) * t + a);
|
||
|
||
// Crossfade
|
||
const blend = ufloat(ru32(bone_rt + BR.blend_weight));
|
||
if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), anim_data, output + 0x10);
|
||
const st = ufloat(ru32(output + 0x18));
|
||
const sa = rf32(kf_base + ru32(output + 0x10) * 4);
|
||
const sb = rf32(kf_base + ru32(output + 0x14) * 4);
|
||
const sec = (sb - sa) * st + sa;
|
||
wu32(output + 0x1C, fbits(sec));
|
||
const pri = ufloat(ru32(output + 0x0C));
|
||
wf32(output + 0x0C, (sec - pri) * blend + pri);
|
||
}
|
||
}
|
||
|
||
// =============================================================================
|
||
// Hermite/Bezier basis + particle emitter interp helpers
|
||
// =============================================================================
|
||
|
||
inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } {
|
||
const t2 = t * t;
|
||
const t3 = t2 * t;
|
||
return .{
|
||
.h1 = 2 * t3 - 3 * t2 + 1,
|
||
.h2 = t3 - 2 * t2 + t,
|
||
.h3 = -2 * t3 + 3 * t2,
|
||
.h4 = t3 - t2,
|
||
};
|
||
}
|
||
|
||
inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } {
|
||
const u = 1.0 - t;
|
||
const t2 = t * t;
|
||
const u_sq = u * u;
|
||
return .{
|
||
.b0 = u_sq * u,
|
||
.b1 = 3 * u_sq * t,
|
||
.b2 = 3 * u * t2,
|
||
.b3 = t2 * t,
|
||
};
|
||
}
|
||
|
||
fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
|
||
findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output);
|
||
|
||
const mode = ri16(anim_data + AD.interp_mode);
|
||
const kf_base = ru32(anim_data + AD.keyframe_base);
|
||
|
||
if (mode == 0) {
|
||
const src = kf_base + ru32(output) * 36;
|
||
wu32(output + 0x0C, ru32(src));
|
||
wu32(output + 0x10, ru32(src + 4));
|
||
wu32(output + 0x14, ru32(src + 8));
|
||
return;
|
||
}
|
||
|
||
const t = ufloat(ru32(output + 8));
|
||
const kf_a = kf_base + ru32(output) * 36;
|
||
const kf_b = kf_base + ru32(output + 4) * 36;
|
||
|
||
if (mode == 1) {
|
||
const result = lerpVec3(kf_a, kf_b, t);
|
||
wu32(output + 0x0C, fbits(result[0]));
|
||
wu32(output + 0x10, fbits(result[1]));
|
||
wu32(output + 0x14, fbits(result[2]));
|
||
} else if (mode == 3) {
|
||
const h = hermiteBasis(t);
|
||
var i: u32 = 0;
|
||
while (i < 3) : (i += 1) {
|
||
const off = i * 4;
|
||
wf32(output + 0x0C + off, h.h1 * rf32(kf_a + off) + h.h2 * rf32(kf_a + 0x18 + off) + h.h3 * rf32(kf_b + off) + h.h4 * rf32(kf_b + 0x0C + off));
|
||
}
|
||
} else if (mode == 2) {
|
||
const b = bezierBasis(t);
|
||
var i: u32 = 0;
|
||
while (i < 3) : (i += 1) {
|
||
const off = i * 4;
|
||
wf32(output + 0x0C + off, b.b0 * rf32(kf_a + off) + b.b1 * rf32(kf_a + 0x18 + off) + b.b2 * rf32(kf_b + 0x0C + off) + b.b3 * rf32(kf_b + off));
|
||
}
|
||
} else return;
|
||
|
||
const blend = rf32(bone_rt_base + BR.blend_weight);
|
||
if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x18);
|
||
|
||
const st = ufloat(ru32(output + 0x20));
|
||
const skf_a = kf_base + ru32(output + 0x18) * 36;
|
||
const skf_b = kf_base + ru32(output + 0x1C) * 36;
|
||
const smode = ri16(anim_data + AD.interp_mode);
|
||
|
||
if (smode == 1) {
|
||
const sec = lerpVec3(skf_a, skf_b, st);
|
||
wu32(output + 0x24, fbits(sec[0]));
|
||
wu32(output + 0x28, fbits(sec[1]));
|
||
wu32(output + 0x2C, fbits(sec[2]));
|
||
} else if (smode == 3) {
|
||
const h = hermiteBasis(st);
|
||
var i: u32 = 0;
|
||
while (i < 3) : (i += 1) {
|
||
const off = i * 4;
|
||
wf32(output + 0x24 + off, h.h1 * rf32(skf_a + off) + h.h2 * rf32(skf_a + 0x18 + off) + h.h3 * rf32(skf_b + off) + h.h4 * rf32(skf_b + 0x0C + off));
|
||
}
|
||
} else if (smode == 2) {
|
||
const b = bezierBasis(st);
|
||
var i: u32 = 0;
|
||
while (i < 3) : (i += 1) {
|
||
const off = i * 4;
|
||
wf32(output + 0x24 + off, b.b0 * rf32(skf_a + off) + b.b1 * rf32(skf_a + 0x18 + off) + b.b2 * rf32(skf_b + 0x0C + off) + b.b3 * rf32(skf_b + off));
|
||
}
|
||
} else {
|
||
wu32(output + 0x24, ru32(skf_a));
|
||
wu32(output + 0x28, ru32(skf_a + 4));
|
||
wu32(output + 0x2C, ru32(skf_a + 8));
|
||
}
|
||
|
||
var i: u32 = 0;
|
||
while (i < 3) : (i += 1) {
|
||
const off = i * 4;
|
||
const pri = rf32(output + 0x0C + off);
|
||
const sec = rf32(output + 0x24 + off);
|
||
wf32(output + 0x0C + off, (sec - pri) * blend + pri);
|
||
}
|
||
}
|
||
}
|
||
|
||
fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void {
|
||
findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output);
|
||
|
||
const mode = ri16(anim_data + AD.interp_mode);
|
||
const kf_base = ru32(anim_data + AD.keyframe_base);
|
||
|
||
if (mode == 0) {
|
||
wu32(output + 0x0C, ru32(kf_base + ru32(output) * 12));
|
||
return;
|
||
}
|
||
|
||
const t = ufloat(ru32(output + 8));
|
||
const kf_a = kf_base + ru32(output) * 12;
|
||
const kf_b = kf_base + ru32(output + 4) * 12;
|
||
|
||
if (mode == 1) {
|
||
const a = rf32(kf_a);
|
||
const b = rf32(kf_b);
|
||
wf32(output + 0x0C, (b - a) * t + a);
|
||
} else if (mode == 3) {
|
||
const h = hermiteBasis(t);
|
||
wf32(output + 0x0C, h.h1 * rf32(kf_a) + h.h2 * rf32(kf_a + 0x08) + h.h3 * rf32(kf_b) + h.h4 * rf32(kf_b + 0x04));
|
||
} else if (mode == 2) {
|
||
const b = bezierBasis(t);
|
||
wf32(output + 0x0C, b.b0 * rf32(kf_a) + b.b1 * rf32(kf_a + 0x08) + b.b2 * rf32(kf_b + 0x04) + b.b3 * rf32(kf_b));
|
||
} else return;
|
||
|
||
const blend = rf32(bone_rt_base + BR.blend_weight);
|
||
if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10);
|
||
|
||
const st = ufloat(ru32(output + 0x18));
|
||
const skf_a = kf_base + ru32(output + 0x10) * 12;
|
||
const skf_b = kf_base + ru32(output + 0x14) * 12;
|
||
const smode = ri16(anim_data + AD.interp_mode);
|
||
|
||
var sec: f32 = undefined;
|
||
if (smode == 1) {
|
||
sec = (rf32(skf_b) - rf32(skf_a)) * st + rf32(skf_a);
|
||
} else if (smode == 3) {
|
||
const h = hermiteBasis(st);
|
||
sec = h.h1 * rf32(skf_a) + h.h2 * rf32(skf_a + 0x08) + h.h3 * rf32(skf_b) + h.h4 * rf32(skf_b + 0x04);
|
||
} else if (smode == 2) {
|
||
const bz = bezierBasis(st);
|
||
sec = bz.b0 * rf32(skf_a) + bz.b1 * rf32(skf_a + 0x08) + bz.b2 * rf32(skf_b + 0x04) + bz.b3 * rf32(skf_b);
|
||
} else {
|
||
sec = rf32(skf_a);
|
||
}
|
||
wf32(output + 0x1C, sec);
|
||
const pri = rf32(output + 0x0C);
|
||
wf32(output + 0x0C, (sec - pri) * blend + pri);
|
||
}
|
||
}
|
||
|
||
// =============================================================================
|
||
// getInterpolatedFloat — reimplemented from 0x71af20
|
||
// Same as interpFloatTrack but uses the bone_rt directly (different register mapping)
|
||
// =============================================================================
|
||
|
||
inline fn getInterpolatedFloat(this: u32, bone_rt_addr: u32, anim_data_short_ptr: u32, output: u32) void {
|
||
findInterpIdx(this, ru32(bone_rt_addr + 0x98), ru32(bone_rt_addr + 0x9C), anim_data_short_ptr, output);
|
||
|
||
const interp_mode = ri16(anim_data_short_ptr);
|
||
const kf_base = ru32(anim_data_short_ptr + 0x18);
|
||
|
||
if (interp_mode == 0) {
|
||
wu32(output + 0x0C, ru32(kf_base + ru32(output) * 4));
|
||
return;
|
||
}
|
||
|
||
const t = ufloat(ru32(output + 8));
|
||
const a = rf32(kf_base + ru32(output) * 4);
|
||
const b = rf32(kf_base + ru32(output + 4) * 4);
|
||
wf32(output + 0x0C, (b - a) * t + a);
|
||
|
||
const blend = rf32(bone_rt_addr + 0x10C);
|
||
if (blend != 0.0 and ri16(anim_data_short_ptr + 2) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt_addr + 0xC4), ru32(bone_rt_addr + 0xC8), anim_data_short_ptr, output + 0x10);
|
||
const st = ufloat(ru32(output + 0x18));
|
||
const sa = rf32(kf_base + ru32(output + 0x10) * 4);
|
||
const sb = rf32(kf_base + ru32(output + 0x14) * 4);
|
||
const sec = (sb - sa) * st + sa;
|
||
wu32(output + 0x1C, fbits(sec));
|
||
const pri = ufloat(ru32(output + 0x0C));
|
||
wf32(output + 0x0C, (sec - pri) * blend + pri);
|
||
}
|
||
}
|
||
|
||
// =============================================================================
|
||
// calculateScaledInverseMatrix — reimplemented from 0x7bd820
|
||
// Used for billboarding. Transposes 3x3 rotation, scales by 1/scale^2,
|
||
// applies inverse translation.
|
||
// =============================================================================
|
||
|
||
fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
|
||
// Simple transpose for unit scale
|
||
if (@abs(scale - 1.0) < @as(f32, @bitCast(@as(u32, 0x35800000)))) {
|
||
// Transpose 3x3
|
||
wf32(out + 0x00, rf32(this_mat + 0x00));
|
||
wf32(out + 0x04, rf32(this_mat + 0x10));
|
||
wf32(out + 0x08, rf32(this_mat + 0x20));
|
||
wf32(out + 0x0C, 0);
|
||
wf32(out + 0x10, rf32(this_mat + 0x04));
|
||
wf32(out + 0x14, rf32(this_mat + 0x14));
|
||
wf32(out + 0x18, rf32(this_mat + 0x24));
|
||
wf32(out + 0x1C, 0);
|
||
wf32(out + 0x20, rf32(this_mat + 0x08));
|
||
wf32(out + 0x24, rf32(this_mat + 0x18));
|
||
wf32(out + 0x28, rf32(this_mat + 0x28));
|
||
wf32(out + 0x2C, 0);
|
||
wf32(out + 0x30, 0);
|
||
wf32(out + 0x34, 0);
|
||
wf32(out + 0x38, 0);
|
||
wf32(out + 0x3C, @as(f32, @bitCast(@as(u32, 0x3f800000))));
|
||
// Apply inverse translation
|
||
applyTranslation(out, -rf32(this_mat + 0x30), -rf32(this_mat + 0x34), -rf32(this_mat + 0x38));
|
||
return;
|
||
}
|
||
|
||
// Transpose 3x3 portion
|
||
wf32(out + 0x00, rf32(this_mat + 0x00));
|
||
wf32(out + 0x04, rf32(this_mat + 0x10));
|
||
wf32(out + 0x08, rf32(this_mat + 0x20));
|
||
wf32(out + 0x0C, 0);
|
||
wf32(out + 0x10, rf32(this_mat + 0x04));
|
||
wf32(out + 0x14, rf32(this_mat + 0x14));
|
||
wf32(out + 0x18, rf32(this_mat + 0x24));
|
||
wf32(out + 0x1C, 0);
|
||
wf32(out + 0x20, rf32(this_mat + 0x08));
|
||
wf32(out + 0x24, rf32(this_mat + 0x18));
|
||
wf32(out + 0x28, rf32(this_mat + 0x28));
|
||
wf32(out + 0x2C, 0);
|
||
wf32(out + 0x30, 0);
|
||
wf32(out + 0x34, 0);
|
||
wf32(out + 0x38, 0);
|
||
wf32(out + 0x3C, @as(f32, @bitCast(@as(u32, 0x3f800000))));
|
||
|
||
// Scale by 1/(scale^2)
|
||
const inv_s2 = 1.0 / (scale * scale);
|
||
scaleMatrix3x3(out, inv_s2, inv_s2, inv_s2);
|
||
|
||
// Apply inverse translation
|
||
applyTranslation(out, -rf32(this_mat + 0x30), -rf32(this_mat + 0x34), -rf32(this_mat + 0x38));
|
||
}
|
||
|
||
// =============================================================================
|
||
// Main export: transformMatrix4x4_REF
|
||
//
|
||
// Calling convention: x86_thiscall — matches the original at 0x714260 exactly.
|
||
// ECX=this, stack: mat1..mat4, callee cleans RET 0x10.
|
||
//
|
||
// Params: this_ptr(ECX), mat1(parent_matrix*), mat2(position_vec3*),
|
||
// mat3(offset_vec3*), mat4(scale_float_bits)
|
||
// =============================================================================
|
||
|
||
export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||
|
||
@setEvalBranchQuota(50000);
|
||
// =========================================================================
|
||
// Section 1: Entry checks
|
||
// =========================================================================
|
||
if (ru32(this + SO.model_data_ptr) == 0) return;
|
||
const anim_ctx = ru32(this + SO.anim_ctx_ptr);
|
||
if (ru32(this + SO.sync_value) == ru32(anim_ctx + 0x10)) return;
|
||
|
||
|
||
// =========================================================================
|
||
// Section 2: Emitter setup
|
||
// =========================================================================
|
||
const model_ctr = ru32(this + SO.model_ctr_ptr);
|
||
const model_hdr = ru32(model_ctr + 0x130);
|
||
const emitter_ctx = ru32(this + SO.emitter_ctx);
|
||
|
||
if (emitter_ctx != 0) {
|
||
// Assembly 0x71429E-0x7142C1: emitter_ctx+0x50 != 0 AND this+0x1D8 != 0
|
||
const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0;
|
||
wu32(this + 0x50, has_emitter); // emitter_enable_flag
|
||
wu32(this + 0x17C, ru32(emitter_ctx + 0x17C));
|
||
}
|
||
|
||
// =========================================================================
|
||
// Section 3: World position/scale
|
||
// =========================================================================
|
||
const pos_ptr = mat2; // position input Vec3
|
||
const ofs_ptr = mat3; // offset input Vec3
|
||
const scale_f: f32 = @bitCast(mat4); // float scale
|
||
|
||
// world_pos = pos * per_axis_scale
|
||
wf32(this + SO.world_pos + 0, rf32(pos_ptr) * rf32(this + SO.field_184));
|
||
wf32(this + SO.world_pos + 4, rf32(this + SO.field_188) * rf32(pos_ptr + 4));
|
||
wf32(this + SO.world_pos + 8, @bitCast(fbits(rf32(this + SO.field_18c) * rf32(pos_ptr + 8))));
|
||
|
||
// render_pri = offset + existing fields
|
||
const rp0 = rf32(ofs_ptr) + rf32(this + SO.field_190);
|
||
const rp1 = rf32(this + SO.render_scale_x) + rf32(ofs_ptr + 4);
|
||
const rp2 = rf32(this + SO.render_scale_y) + rf32(ofs_ptr + 8);
|
||
wf32(this + SO.render_pri + 0, rp0);
|
||
wf32(this + SO.render_pri + 4, rp1);
|
||
wf32(this + SO.render_pri + 8, rp2);
|
||
|
||
// render_scale_z = scale * field_180
|
||
wf32(this + SO.render_scale_z, scale_f * rf32(this + SO.field_180));
|
||
|
||
// =========================================================================
|
||
// Section 4: Global sequence processing
|
||
// =========================================================================
|
||
const gs_count = ru32(model_hdr + 0x14);
|
||
if (gs_count != 0) {
|
||
const gs_durations = ru32(model_hdr + 0x18);
|
||
const gs_values = ru32(this + SO.gs_values_ptr);
|
||
const timestamp = ru32(anim_ctx + 0x0C);
|
||
const time_base = ru32(this + SO.gs_time_base);
|
||
var gi: u32 = 0;
|
||
while (gi < gs_count) : (gi += 1) {
|
||
const dur = ru32(gs_durations + gi * 4);
|
||
if (dur == 0) {
|
||
wu32(gs_values + gi * 4, 0);
|
||
} else {
|
||
wu32(gs_values + gi * 4, (timestamp -% time_base) % dur);
|
||
}
|
||
}
|
||
}
|
||
|
||
// initParticlePixelShaderGeneration (0x74a7c0) — matrix multiply via JMP table.
|
||
// Computes: *(this+0xFC) = *(this+0xBC) × mat1
|
||
// Assembly: PUSH mat1, PUSH &0xBC, PUSH &0xFC, CALL 0x74A7C0
|
||
// 0x74A7C0 = JMP [0x876504] → runtime target (0x754A66 SSE version)
|
||
// Must call through 0x74A7C0, NOT 0x7507BB directly.
|
||
{
|
||
const matMul: *const fn (u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x74A7C0);
|
||
matMul(this + 0xFC, this + 0xBC, mat1);
|
||
}
|
||
|
||
// =========================================================================
|
||
// Section 5: child_objects_padding (len_sq of world transform translation)
|
||
// =========================================================================
|
||
if (emitter_ctx == 0 or (ru8(emitter_ctx + 4) & 1) != 0) {
|
||
const wx = rf32(this + SO.world_xform + 8 * 4); // [8]
|
||
const wy = rf32(this + SO.world_xform + 9 * 4); // [9]
|
||
const wz = rf32(this + SO.world_xform + 10 * 4); // [10]
|
||
wu32(this + SO.child_padding, fbits(wx * wx + wy * wy + wz * wz));
|
||
} else {
|
||
wu32(this + SO.child_padding, ru32(emitter_ctx + 0x84));
|
||
}
|
||
|
||
// =========================================================================
|
||
// Section 6: Identity matrix init + timestamp delta
|
||
// =========================================================================
|
||
var local_mat: [16]f32 = .{
|
||
1, 0, 0, 0,
|
||
0, 1, 0, 0,
|
||
0, 0, 1, 0,
|
||
0, 0, 0, 1,
|
||
};
|
||
const local_mat_addr = @intFromPtr(&local_mat);
|
||
|
||
// Secondary identity (3x4 portion for the second matrix in decompilation)
|
||
var local_mat2: [16]f32 = .{
|
||
1, 0, 0, 0,
|
||
0, 1, 0, 0,
|
||
0, 0, 1, 0,
|
||
0, 0, 0, 1,
|
||
};
|
||
|
||
// Timestamp delta tracking
|
||
// Assembly guard: if (anim_ctx != 0 AND anim_ctx->timestamp != 0)
|
||
// NOT guarded on the stored value at this+0x4C — must always write on first frame
|
||
var time_delta_val: u32 = 0;
|
||
const cur_ts = ru32(anim_ctx + 0x0C);
|
||
if (cur_ts != 0) {
|
||
const sdb = ru32(this + SO.search_data_base);
|
||
if (sdb != 0) {
|
||
time_delta_val = cur_ts -% sdb;
|
||
}
|
||
wu32(this + SO.search_data_base, cur_ts);
|
||
}
|
||
|
||
// =========================================================================
|
||
// Section 7: Main bone loop
|
||
// =========================================================================
|
||
const bone_count = ru32(model_hdr + 0x34);
|
||
const bone_defs = ru32(model_hdr + 0x38);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const bone_out_base = ru32(this + SO.bone_out_ptr);
|
||
|
||
if (bone_count != 0) {
|
||
var bone_idx: u32 = 0;
|
||
while (bone_idx < bone_count) : (bone_idx += 1) {
|
||
const bdef = bone_defs + bone_idx * 0x6C;
|
||
const brt = bone_rt_base + bone_idx * 0x118;
|
||
const flags = ru32(bdef + BD.flags);
|
||
const parent_idx_raw: i32 = @as(i32, @intCast(@as(i16, @bitCast(ru16(bdef + BD.parent_bone)))));
|
||
|
||
// --- Animation time computation ---
|
||
// (Handle primary and secondary animation slot timing)
|
||
const anim_slot_val = ri32(brt + BR.anim_slot);
|
||
if (anim_slot_val == -1) {
|
||
// Inherit from parent bone
|
||
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
|
||
const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118;
|
||
wu32(brt + BR.prim_time, ru32(parent_rt + BR.prim_time));
|
||
wu32(brt + BR.prim_track, ru32(parent_rt + BR.prim_track));
|
||
wu32(brt + BR.prim_anim, ru32(parent_rt + BR.prim_anim));
|
||
} else if (bone_idx != 0) {
|
||
wu32(brt + BR.prim_time, ru32(bone_rt_base + BR.prim_time));
|
||
wu32(brt + BR.prim_track, ru32(bone_rt_base + BR.prim_track));
|
||
wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim));
|
||
}
|
||
} else {
|
||
// Has own animation slot — compute time from animation lookup table.
|
||
// Assembly at 0x714561-0x71464E, verified line by line.
|
||
if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0
|
||
// Add time delta to sec_start/sec_end
|
||
wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val); // [ESI+0xA8]
|
||
wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val); // [ESI+0xAC]
|
||
}
|
||
|
||
// anim_entry = anim_lookup_table + anim_slot * 0x44
|
||
const anim_lookup = ru32(model_hdr + 0x20); // [EDX+0x20]
|
||
const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44;
|
||
const cur_time = ru32(ru32(this + 0x2C) + 0xC); // [EBX+0x2C]+0xC = timestamp
|
||
|
||
// Check looping flag: [anim_entry+0x10] & 1
|
||
if ((ru8(anim_entry + 0x10) & 1) == 0) {
|
||
// Looping: assembly at 0x7145F1-0x714631
|
||
const anim_end = ru32(anim_entry + 0x08);
|
||
const anim_start = ru32(anim_entry + 0x04);
|
||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||
// elapsed = (float)(cur_time - sec_start) * time_scale → __ftol
|
||
const delta = cur_time -% ru32(brt + 0xA8);
|
||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0);
|
||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start);
|
||
wu32(brt + 0x98, anim_start +% frame); // prim_time
|
||
}
|
||
} else {
|
||
// Clamped: assembly at 0x71458E-0x7145E3
|
||
const sec_end_val = ru32(brt + 0xAC);
|
||
const sec_start_val = ru32(brt + 0xA8);
|
||
|
||
// Check if sec_end has passed (sec_end - cur_time <= 0 signed)
|
||
if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) {
|
||
// sec_end hasn't passed yet
|
||
if (sec_start_val != cur_time and @as(i32, @bitCast(sec_start_val -% cur_time)) > 0) {
|
||
// Before start: use sec_start as time
|
||
// Actually assembly jumps to looping path LAB_007145f1
|
||
// which reads anim_entry+0x08, anim_entry+0x04
|
||
// Fallthrough: use cur_time (no write to prim_time)
|
||
}
|
||
// goto looping path
|
||
const anim_end = ru32(anim_entry + 0x08);
|
||
const anim_start = ru32(anim_entry + 0x04);
|
||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||
const delta = cur_time -% ru32(brt + 0xA8);
|
||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0);
|
||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start);
|
||
wu32(brt + 0x98, anim_start +% frame);
|
||
}
|
||
} else {
|
||
// sec_end has passed — compute clamped position
|
||
// Assembly at 0x71458E-0x7145E3:
|
||
// delta = (sec_end - sec_start), scaled by [ESI+0xB0]
|
||
const dur = sec_end_val -% sec_start_val;
|
||
const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xB0);
|
||
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8)));
|
||
|
||
if (offset < 0) {
|
||
// Clamp to anim_start
|
||
wu32(brt + 0x98, ru32(anim_entry + 0x04));
|
||
} else {
|
||
const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08)));
|
||
const anim_start_i = @as(i32, @bitCast(ru32(anim_entry + 0x04)));
|
||
if (offset <= anim_end_i - anim_start_i) {
|
||
wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i)));
|
||
} else {
|
||
// Clamp to anim_end
|
||
wu32(brt + 0x98, ru32(anim_entry + 0x08));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Store results: assembly at 0x714633-0x71464E
|
||
wu32(brt + 0x9C, ru32(brt + 0xA4)); // prim_track = anim_slot
|
||
// prim_time already set above
|
||
wu32(brt + 0xA0, bone_idx); // prim_anim = bone_idx
|
||
}
|
||
|
||
// --- Secondary animation time (crossfade target) ---
|
||
// Similar pattern for the secondary/blend animation slot
|
||
const sec_slot_val = ri32(brt + BR.sec_slot);
|
||
if (sec_slot_val == -1) {
|
||
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
|
||
const parent_rt = bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118;
|
||
wu32(brt + BR.sec_time, ru32(parent_rt + BR.sec_time));
|
||
wu32(brt + BR.sec_track, ru32(parent_rt + BR.sec_track));
|
||
} else if (bone_idx != 0) {
|
||
wu32(brt + BR.sec_time, ru32(bone_rt_base + BR.sec_time));
|
||
wu32(brt + BR.sec_track, ru32(bone_rt_base + BR.sec_track));
|
||
} else {
|
||
wu32(brt + BR.sec_time, ru32(brt + BR.prim_time));
|
||
wu32(brt + BR.sec_track, ru32(brt + BR.prim_track));
|
||
}
|
||
} else {
|
||
// Secondary animation slot time computation.
|
||
// Assembly at 0x7146C1-0x7147C3, mirrors primary slot logic.
|
||
if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0
|
||
wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val); // [ESI+0xD4]
|
||
wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val); // [ESI+0xD8]
|
||
}
|
||
|
||
const sec_anim_lookup = ru32(model_hdr + 0x20);
|
||
const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44;
|
||
const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC);
|
||
|
||
if ((ru8(sec_anim_entry + 0x10) & 1) == 0) {
|
||
// Looping
|
||
const anim_end = ru32(sec_anim_entry + 0x08);
|
||
const anim_start = ru32(sec_anim_entry + 0x04);
|
||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||
const delta = sec_cur_time -% ru32(brt + 0xD4);
|
||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC);
|
||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start);
|
||
wu32(brt + 0xC4, anim_start +% frame); // sec_time
|
||
}
|
||
} else {
|
||
// Clamped
|
||
const sec_end_val = ru32(brt + 0xD8);
|
||
const sec_start_val = ru32(brt + 0xD4);
|
||
|
||
if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) {
|
||
if (sec_start_val != sec_cur_time and @as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) {
|
||
// use sec_start
|
||
}
|
||
const anim_end = ru32(sec_anim_entry + 0x08);
|
||
const anim_start = ru32(sec_anim_entry + 0x04);
|
||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||
const delta = sec_cur_time -% ru32(brt + 0xD4);
|
||
const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC);
|
||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start);
|
||
wu32(brt + 0xC4, anim_start +% frame);
|
||
}
|
||
} else {
|
||
const dur = sec_end_val -% sec_start_val;
|
||
const ftol_result = callFtol(@as(i32, @bitCast(dur)), brt + 0xDC);
|
||
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4)));
|
||
|
||
if (offset < 0) {
|
||
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04));
|
||
} else {
|
||
const anim_end_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x08)));
|
||
const anim_start_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x04)));
|
||
if (offset <= anim_end_i - anim_start_i) {
|
||
wu32(brt + 0xC4, @as(u32, @bitCast(offset + anim_start_i)));
|
||
} else {
|
||
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x08));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Store results: assembly at 0x714799-0x7147C3
|
||
wu32(brt + 0xC8, ru32(brt + 0xD0)); // sec_track = sec_slot
|
||
// sec_time already set above
|
||
|
||
// Check expiry: if (timestamp - crossfade_end >= 0) expire slot
|
||
if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) {
|
||
wu32(brt + 0xD0, 0xFFFFFFFF); // expire secondary slot
|
||
}
|
||
}
|
||
|
||
// --- Blend weight (crossfade Hermite interpolation) ---
|
||
if (ri32(brt + BR.anim_slot) == -1 and ri32(brt + BR.sec_slot) == -1) {
|
||
// Inherit blend weight from parent
|
||
if (parent_idx_raw >= 0 and @as(u32, @intCast(parent_idx_raw)) < bone_count) {
|
||
wu32(brt + BR.blend_weight, ru32(bone_rt_base + @as(u32, @intCast(parent_idx_raw)) * 0x118 + BR.blend_weight));
|
||
} else if (bone_idx == 0) {
|
||
wu32(brt + BR.blend_weight, 0); // root bone, no blend
|
||
} else {
|
||
wu32(brt + BR.blend_weight, ru32(bone_rt_base + BR.blend_weight));
|
||
}
|
||
} else {
|
||
const cf_remaining = ri32(brt + BR.crossfade_end) - ri32(anim_ctx + 0x0C);
|
||
if (cf_remaining < 1 or (ru32(brt + BR.prim_time) == ru32(brt + BR.sec_time) and
|
||
ru32(brt + BR.prim_track) == ru32(brt + BR.sec_track)))
|
||
{
|
||
wu32(brt + BR.blend_weight, 0);
|
||
} else {
|
||
const t_raw = @as(f32, @floatFromInt(cf_remaining)) * ufloat(ru32(brt + BR.crossfade_inv));
|
||
const t_clamped = if (t_raw < 0.0) @as(f32, 0.0) else if (t_raw > 1.0) @as(f32, 1.0) else t_raw;
|
||
// Hermite: (3 - 2t) * t^2 * weight
|
||
const h = (3.0 - 2.0 * t_clamped) * t_clamped * t_clamped * ufloat(ru32(brt + BR.crossfade_weight));
|
||
wu32(brt + BR.blend_weight, fbits(h));
|
||
}
|
||
}
|
||
|
||
// --- Parent bone transform inheritance ---
|
||
const combined_flags: u32 = ru32(brt + BR.flags2) | flags;
|
||
var src_mat: u32 = undefined;
|
||
|
||
if (ru16(bdef + BD.parent_bone) == 0xFFFF) {
|
||
src_mat = this + 0xFC;
|
||
} else {
|
||
const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40;
|
||
src_mat = parent_out;
|
||
|
||
// Billboard pre-processing (flags & 7)
|
||
if ((combined_flags & 7) != 0) {
|
||
// Copy parent matrix to local_mat and work from there
|
||
for (0..16) |i| {
|
||
local_mat[i] = rf32(parent_out + @as(u32, @intCast(i)) * 4);
|
||
}
|
||
|
||
// Apply pivot translation
|
||
const pivot_x = rf32(bdef + BD.pivot_x);
|
||
const pivot_y = rf32(bdef + BD.pivot_y);
|
||
const pivot_z = rf32(bdef + BD.pivot_z);
|
||
|
||
// Compute translated position
|
||
const tx = local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z + local_mat[12];
|
||
const ty = local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z + local_mat[13];
|
||
const tz = local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z + local_mat[14];
|
||
|
||
const bb_type = combined_flags & 6;
|
||
if (bb_type == 2) {
|
||
// Cylindrical billboard — normalize each column
|
||
const n0 = normalizeVec3(local_mat[0], local_mat[1], local_mat[2]);
|
||
local_mat[0] = n0[0];
|
||
local_mat[1] = n0[1];
|
||
local_mat[2] = n0[2];
|
||
const n1 = normalizeVec3(local_mat[4], local_mat[5], local_mat[6]);
|
||
local_mat[4] = n1[0];
|
||
local_mat[5] = n1[1];
|
||
local_mat[6] = n1[2];
|
||
const n2 = normalizeVec3(local_mat[8], local_mat[9], local_mat[10]);
|
||
local_mat[8] = n2[0];
|
||
local_mat[9] = n2[1];
|
||
local_mat[10] = n2[2];
|
||
} else if (bb_type == 4) {
|
||
// Spherical billboard — inherit camera rotation with scale preservation
|
||
const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) };
|
||
const cam_len_sq0 = cam0[0] * cam0[0] + cam0[1] * cam0[1] + cam0[2] * cam0[2];
|
||
var s0: f32 = 1.0;
|
||
if (cam_len_sq0 > rf32(0x0080c5c8)) {
|
||
const mat_len_sq0 = local_mat[0] * local_mat[0] + local_mat[1] * local_mat[1] + local_mat[2] * local_mat[2];
|
||
s0 = @sqrt(mat_len_sq0 / cam_len_sq0);
|
||
}
|
||
local_mat[0] = s0 * cam0[0];
|
||
local_mat[1] = s0 * cam0[1];
|
||
local_mat[2] = s0 * cam0[2];
|
||
|
||
const wt0 = rf32(this + SO.world_xform + 0 * 4);
|
||
const wt1 = rf32(this + SO.world_xform + 1 * 4);
|
||
const wt2 = rf32(this + SO.world_xform + 2 * 4);
|
||
const wt_len_sq = wt0 * wt0 + wt1 * wt1 + wt2 * wt2;
|
||
var s1: f32 = 1.0;
|
||
if (wt_len_sq > rf32(0x0080c5c8)) {
|
||
const mat_len_sq1 = local_mat[4] * local_mat[4] + local_mat[5] * local_mat[5] + local_mat[6] * local_mat[6];
|
||
s1 = @sqrt(mat_len_sq1 / wt_len_sq);
|
||
}
|
||
local_mat[4] = s1 * wt0;
|
||
local_mat[5] = s1 * wt1;
|
||
local_mat[6] = s1 * wt2;
|
||
|
||
const wt4 = rf32(this + SO.world_xform + 4 * 4);
|
||
const wt5 = rf32(this + SO.world_xform + 5 * 4);
|
||
const wt6 = rf32(this + SO.world_xform + 6 * 4);
|
||
const wt_len_sq2 = wt4 * wt4 + wt5 * wt5 + wt6 * wt6;
|
||
var s2: f32 = 1.0;
|
||
if (wt_len_sq2 > rf32(0x0080c5c8)) {
|
||
const mat_len_sq2 = local_mat[8] * local_mat[8] + local_mat[9] * local_mat[9] + local_mat[10] * local_mat[10];
|
||
s2 = @sqrt(mat_len_sq2 / wt_len_sq2);
|
||
}
|
||
local_mat[8] = s2 * wt4;
|
||
local_mat[9] = s2 * wt5;
|
||
local_mat[10] = s2 * wt6;
|
||
} else if (bb_type == 6) {
|
||
// Full billboard — copy camera rotation directly
|
||
local_mat[0] = rf32(this + SO.bb_row0);
|
||
local_mat[1] = rf32(this + SO.bb_row0 + 4);
|
||
local_mat[2] = rf32(this + SO.bb_row0 + 8);
|
||
local_mat[4] = rf32(this + SO.world_xform + 0 * 4);
|
||
local_mat[5] = rf32(this + SO.world_xform + 1 * 4);
|
||
local_mat[6] = rf32(this + SO.world_xform + 2 * 4);
|
||
local_mat[8] = rf32(this + SO.world_xform + 4 * 4);
|
||
local_mat[9] = rf32(this + SO.world_xform + 5 * 4);
|
||
local_mat[10] = rf32(this + SO.world_xform + 6 * 4);
|
||
}
|
||
|
||
// Recompute translation: pos - rot * pivot
|
||
if ((combined_flags & 1) == 0) {
|
||
local_mat[12] = tx - (local_mat[0] * pivot_x + local_mat[4] * pivot_y + local_mat[8] * pivot_z);
|
||
local_mat[13] = ty - (local_mat[1] * pivot_x + local_mat[5] * pivot_y + local_mat[9] * pivot_z);
|
||
local_mat[14] = tz - (local_mat[2] * pivot_x + local_mat[6] * pivot_y + local_mat[10] * pivot_z);
|
||
} else {
|
||
local_mat[12] = rf32(this + SO.world_xform + 8 * 4);
|
||
local_mat[13] = rf32(this + SO.world_xform + 9 * 4);
|
||
local_mat[14] = rf32(this + SO.world_xform + 10 * 4);
|
||
}
|
||
|
||
src_mat = local_mat_addr;
|
||
}
|
||
}
|
||
|
||
// --- Rotation interpolation ---
|
||
if ((combined_flags & 0x280) == 0) {
|
||
// No rotation animation — just copy parent
|
||
const dst = bone_out_base + bone_idx * 0x40;
|
||
copyMat4(dst, src_mat);
|
||
} else {
|
||
// Reset to identity for composition
|
||
local_mat2 = .{ 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 0, 1 };
|
||
const lm2_addr = @intFromPtr(&local_mat2);
|
||
|
||
const rot_anim = bdef + BD.rot_anim;
|
||
const rot_kf_count = ru32(bdef + BD.rot_nts);
|
||
|
||
// Step 1: Rotation — build rotation matrix from quaternion FIRST.
|
||
// The original at 0x74B6BB overwrites the bone-local matrix with the
|
||
// quaternion rotation matrix (it does NOT multiply — just writes directly).
|
||
// This runs BEFORE scale and translation so the translation offset
|
||
// (pivot - matrix * pivot) uses the correctly rotated matrix.
|
||
if (rot_kf_count != 0) {
|
||
if (ru32(this + SO.anim_frame_ctr) < rot_kf_count) {
|
||
// Assembly: CALL 0x713EA0 — __fastcall(ECX=this, EDX=bone_rt, stack: anim_data, output)
|
||
const interpAnimKFFn: *const fn (u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x713EA0);
|
||
interpAnimKFFn(this, brt, rot_anim, brt + BR.rot_idx0);
|
||
}
|
||
// Assembly: CALL 0x74B6B5 — JMP table, __stdcall(mat_ptr, quat_ptr), RET 0x8
|
||
const buildRotFn: *const fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x74B6B5);
|
||
buildRotFn(lm2_addr, brt + BR.rot_x);
|
||
}
|
||
|
||
// Step 2: Scale interpolation — applied after rotation
|
||
const scale_anim = bdef + BD.scale_anim;
|
||
const scale_kf_count = ru32(bdef + BD.scale_nts);
|
||
if (scale_kf_count != 0) {
|
||
if (ru32(this + SO.anim_frame_ctr) < scale_kf_count) {
|
||
interpVec3Track(this, brt, scale_anim, brt + BR.scale_idx0, ufloat(ru32(brt + BR.blend_weight)));
|
||
}
|
||
// Assembly: CALL 0x7BDCA0 — scaleMatrix3x3ByVector
|
||
// __thiscall(ECX=mat, stack=vec3_ptr)
|
||
const scaleMat: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x7BDCA0);
|
||
scaleMat(lm2_addr, 0, brt + BR.scale_x);
|
||
}
|
||
|
||
// Conditional multiply: if flag bit 0x80 set AND bone_rt[0xF0] != 0,
|
||
// multiply bone_local by the matrix pointed to by bone_rt[0xF0].
|
||
// Assembly at 0x714F7F-0x714F9C:
|
||
// TEST CL, CL / JNS skip
|
||
// MOV EAX, [ESI+0xF0] / TEST EAX, EAX / JZ skip
|
||
// PUSH EAX (right), PUSH &bone_local (left), PUSH &bone_local (output)
|
||
// CALL 0x74A7C0 (multiplyMatrix4x4: output = left * right)
|
||
// This is bone_local *= *(bone_rt+0xF0)
|
||
if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) {
|
||
const extra_mat = ru32(brt + BR.bone_flag_cache); // pointer to additional matrix
|
||
// In-place multiply: bone_local = bone_local * extra_mat
|
||
// Assembly: CALL 0x74A7C0 (JMP table → SSE matmul)
|
||
const matMul: *const fn (u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x74A7C0);
|
||
matMul(lm2_addr, lm2_addr, extra_mat);
|
||
}
|
||
|
||
// Step 3: Translation interpolation
|
||
var tx_val = rf32(bdef + BD.pivot_x);
|
||
var ty_val = rf32(bdef + BD.pivot_y);
|
||
var tz_val = rf32(bdef + BD.pivot_z);
|
||
|
||
const trans_anim = bdef + BD.trans_anim;
|
||
const trans_kf_count = ru32(bdef + BD.trans_nts);
|
||
if (trans_kf_count != 0) {
|
||
if (ru32(this + SO.anim_frame_ctr) < trans_kf_count) {
|
||
interpVec3Track(this, brt, trans_anim, brt + BR.trans_idx0, ufloat(ru32(brt + BR.blend_weight)));
|
||
}
|
||
tx_val += ufloat(ru32(brt + BR.trans_x));
|
||
ty_val += ufloat(ru32(brt + BR.trans_y));
|
||
tz_val += ufloat(ru32(brt + BR.trans_z));
|
||
}
|
||
|
||
// Step 4: Compute translation offset using the ROTATED+SCALED matrix.
|
||
// translation = (pivot + interp_trans) - bone_local_matrix * pivot
|
||
const piv_x = rf32(bdef + BD.pivot_x);
|
||
const piv_y = rf32(bdef + BD.pivot_y);
|
||
const piv_z = rf32(bdef + BD.pivot_z);
|
||
local_mat2[12] = tx_val - (local_mat2[0] * piv_x + local_mat2[4] * piv_y + local_mat2[8] * piv_z);
|
||
local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z);
|
||
local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z);
|
||
|
||
// Write final composed matrix to output: dst = bone_local * parent
|
||
// Assembly: CALL 0x74A7C0 (JMP table → SSE matmul) at 0x7151BA
|
||
const dst = bone_out_base + bone_idx * 0x40;
|
||
{
|
||
const matMul: *const fn (u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x74A7C0);
|
||
matMul(dst, lm2_addr, src_mat);
|
||
}
|
||
}
|
||
|
||
// --- Billboard post-processing (flags & 0x78) ---
|
||
// Assembly at 0x7151F9-0x71594E. Runs for BOTH animated and non-animated paths.
|
||
// Modifies the already-written bone output matrix in-place.
|
||
if ((combined_flags & 0x78) != 0) {
|
||
// pMVar19 = bone_idx * 0x40 (byte offset for output)
|
||
// pfVar12 = bone_out_base + pMVar19 (output matrix ptr)
|
||
const out_off = bone_idx * 0x40;
|
||
const om = bone_out_base + out_off; // output matrix
|
||
|
||
// Compute scale lengths (sqrt of row length_sq for each row)
|
||
const scale_len0 = @sqrt(rf32(om + 0x08) * rf32(om + 0x08) + rf32(om + 0x04) * rf32(om + 0x04) + rf32(om) * rf32(om));
|
||
const scale_len1 = @sqrt(rf32(om + 0x18) * rf32(om + 0x18) + rf32(om + 0x14) * rf32(om + 0x14) + rf32(om + 0x10) * rf32(om + 0x10));
|
||
const scale_len2 = @sqrt(rf32(om + 0x28) * rf32(om + 0x28) + rf32(om + 0x24) * rf32(om + 0x24) + rf32(om + 0x20) * rf32(om + 0x20));
|
||
|
||
// Compute translated pivot position through the output matrix
|
||
// local_a8 = pivot * matrix + translation
|
||
const bpx = rf32(bdef + BD.pivot_x);
|
||
const bpy = rf32(bdef + BD.pivot_y);
|
||
const bpz = rf32(bdef + BD.pivot_z);
|
||
const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30);
|
||
const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34);
|
||
const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38);
|
||
|
||
// Switch on billboard post-processing type
|
||
const bb_post = combined_flags & 0x78;
|
||
switch (bb_post) {
|
||
0x08 => {
|
||
// Type 8: decompilation lines 657-718
|
||
// If no pre-billboard (local_1c == 0 i.e. flags & 0x280 was 0):
|
||
// set fixed rotation columns
|
||
// Else: use rotation matrix rows with negated first component, normalize
|
||
const had_anim = (combined_flags & 0x280) != 0;
|
||
if (!had_anim) {
|
||
// Fixed columns: row0={0,0,-1}, row1={1,0,0}, row2={0,1,0}
|
||
wf32(om, 0);
|
||
wf32(om + 0x04, 0);
|
||
wf32(om + 0x08, -1);
|
||
wf32(om + 0x10, 1);
|
||
wf32(om + 0x14, 0);
|
||
wf32(om + 0x18, 0);
|
||
wf32(om + 0x20, 0);
|
||
wf32(om + 0x24, 1);
|
||
wf32(om + 0x28, 0);
|
||
} else {
|
||
// Row 0 = {local_e4, local_e0, -local_e8}, normalize
|
||
const r0x = local_mat2[1]; // local_e4
|
||
const r0y = local_mat2[2]; // local_e0
|
||
const r0z = -local_mat2[0]; // -local_e8
|
||
wf32(om, r0x);
|
||
wf32(om + 0x04, r0y);
|
||
wf32(om + 0x08, r0z);
|
||
const n0 = normalizeVec3InPlace(om);
|
||
_ = n0;
|
||
// Row 1 = {local_d4, local_d0, -local_d8}, normalize
|
||
const r1x = local_mat2[5]; // local_d4
|
||
const r1y = local_mat2[6]; // local_d0
|
||
const r1z = -local_mat2[4]; // -local_d8
|
||
wf32(om + 0x10, r1x);
|
||
wf32(om + 0x14, r1y);
|
||
wf32(om + 0x18, r1z);
|
||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||
_ = n1;
|
||
// Row 2 = {local_c4, local_c0, -local_c8}, normalize
|
||
const r2x = local_mat2[9]; // local_c4
|
||
const r2y = local_mat2[10]; // local_c0
|
||
const r2z = -local_mat2[8]; // -local_c8
|
||
wf32(om + 0x20, r2x);
|
||
wf32(om + 0x24, r2y);
|
||
wf32(om + 0x28, r2z);
|
||
const n2 = normalizeVec3InPlace(om + 0x20);
|
||
_ = n2;
|
||
}
|
||
},
|
||
0x10 => {
|
||
// Type 16: normalize row0, set row1={row0.y, -row0.x, 0}, normalize,
|
||
// row2 = cross(row0, row1)
|
||
const n0 = normalizeVec3InPlace(om);
|
||
_ = n0;
|
||
const r0x = rf32(om);
|
||
const r0y = rf32(om + 0x04);
|
||
wf32(om + 0x10, r0y);
|
||
wf32(om + 0x14, -r0x);
|
||
wf32(om + 0x18, 0);
|
||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||
_ = n1;
|
||
// row2 = cross(row0, row1)
|
||
wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14));
|
||
wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18));
|
||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||
},
|
||
0x20 => {
|
||
// Type 32: normalize row1, set row0={-row1.y, row1.x, 0}, normalize,
|
||
// row2 = cross(row0, row1)
|
||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||
_ = n1;
|
||
wf32(om, -rf32(om + 0x14));
|
||
wf32(om + 0x04, rf32(om + 0x10));
|
||
wf32(om + 0x08, 0);
|
||
const n0 = normalizeVec3InPlace(om);
|
||
_ = n0;
|
||
// row2 = cross(row0, row1)
|
||
wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14));
|
||
wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18));
|
||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||
},
|
||
0x40 => {
|
||
// Type 64: normalize row2, set row1={row2.y, -row2.x, 0}, normalize,
|
||
// row0 = cross(row1, row2)
|
||
normalizeVec3InPlace(om + 0x20);
|
||
wf32(om + 0x10, rf32(om + 0x24));
|
||
wf32(om + 0x14, -rf32(om + 0x20));
|
||
wf32(om + 0x18, 0);
|
||
normalizeVec3InPlace(om + 0x10);
|
||
// row0 = cross(row2.y*row1.z - row2.z*row1.y, ...)
|
||
wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14));
|
||
wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18));
|
||
wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10));
|
||
},
|
||
else => {},
|
||
}
|
||
|
||
// Apply scale lengths back and recompute translation
|
||
// Assembly at 0x715868-0x71594B
|
||
wf32(om + 0x0C, 0);
|
||
wf32(om + 0x1C, 0);
|
||
wf32(om + 0x2C, 0);
|
||
// Scale each row by its original length
|
||
const r0x_s = rf32(om);
|
||
wf32(om, scale_len0 * r0x_s);
|
||
const r0y_s = rf32(om + 0x04);
|
||
wf32(om + 0x04, scale_len0 * r0y_s);
|
||
const r0z_s = rf32(om + 0x08);
|
||
wf32(om + 0x08, scale_len0 * r0z_s);
|
||
const r1x_s = rf32(om + 0x10);
|
||
wf32(om + 0x10, scale_len1 * r1x_s);
|
||
const r1y_s = rf32(om + 0x14);
|
||
wf32(om + 0x14, scale_len1 * r1y_s);
|
||
const r1z_s = rf32(om + 0x18);
|
||
wf32(om + 0x18, scale_len1 * r1z_s);
|
||
const r2x_s = rf32(om + 0x20);
|
||
wf32(om + 0x20, scale_len2 * r2x_s);
|
||
const r2y_s = rf32(om + 0x24);
|
||
wf32(om + 0x24, scale_len2 * r2y_s);
|
||
const r2z_s = rf32(om + 0x28);
|
||
wf32(om + 0x28, scale_len2 * r2z_s);
|
||
|
||
// Recompute translation: pos - scaled_matrix * pivot
|
||
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz));
|
||
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz));
|
||
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz));
|
||
wf32(om + 0x3C, 1.0);
|
||
}
|
||
}
|
||
}
|
||
|
||
// =========================================================================
|
||
// Sections 8-11: Post-bone-loop animations
|
||
// These sections handle texture animation, color animation, bone keyframe
|
||
// post-processing, and particle emitters. They follow the same interpolation
|
||
// pattern as the bone loop but operate on different model data arrays.
|
||
//
|
||
// For the initial implementation, we delegate these to the patterns established
|
||
// above. Each section iterates over its respective model array and calls
|
||
// findInterpIdx + lerp + crossfade blend.
|
||
// =========================================================================
|
||
|
||
// Section 8: Texture animation loop
|
||
texAnimLoop(this, model_hdr);
|
||
|
||
// Section 9: Color animation loop
|
||
colorAnimLoop(this, model_hdr);
|
||
|
||
// Section 10: Bone keyframe processing
|
||
boneKeyframeLoop(this, model_hdr);
|
||
|
||
// Section 11: Particle emitter loops
|
||
particleLoops(this, model_hdr);
|
||
|
||
// Section 12: Attachment recursion
|
||
attachmentRecursion(this, model_hdr, bone_out_base);
|
||
|
||
// =========================================================================
|
||
// Section 13: Sync update
|
||
// =========================================================================
|
||
wu32(this + SO.sync_value, ru32(anim_ctx + 0x10));
|
||
}
|
||
|
||
// =============================================================================
|
||
// Post-bone-loop sections (extracted for readability)
|
||
// =============================================================================
|
||
|
||
fn texAnimLoop(this: u32, model_hdr: u32) void {
|
||
const count = ru32(model_hdr + 0x54);
|
||
if (count == 0) return;
|
||
const data_base = ru32(model_hdr + 0x58);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const out_base = ru32(this + SO.tex_anim_out);
|
||
|
||
var i: u32 = 0;
|
||
var data_off: u32 = 0;
|
||
var out_off: u32 = 0;
|
||
while (i < count) : ({
|
||
i += 1;
|
||
data_off += 0x38;
|
||
out_off += 0x14 * 4;
|
||
}) {
|
||
const anim_data = data_base + data_off;
|
||
const output = out_base + out_off;
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(data_base + data_off + 0x0C)) {
|
||
interpVec3Track(this, bone_rt_base, anim_data, output, ufloat(ru32(bone_rt_base + BR.blend_weight)));
|
||
}
|
||
// Alpha/opacity track
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(anim_data + 0x28)) {
|
||
// Short value interpolation via getIndexOffset/setShortValue pattern
|
||
// This accesses short values at anim_data + 0x1C
|
||
const alpha_anim = anim_data + 0x1C;
|
||
const alpha_out = output + 0xC * 4;
|
||
findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), alpha_anim, alpha_out);
|
||
// Short value interpolation via game's getIndexOffset/setShortValue
|
||
// Assembly: CALL 0x71AFF0 (getIndexOffset) + CALL 0x71B010 (setShortValue)
|
||
const mode = ri16(alpha_anim);
|
||
const table = alpha_anim + AD.nvalues; // ECX = anim_data + 0x14
|
||
if (mode == 0) {
|
||
const sv = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(alpha_out)))));
|
||
wf32(output + 0xF * 4, sv * getShortToFloat());
|
||
} else {
|
||
const t = ufloat(ru32(alpha_out + 8));
|
||
// Assembly reads idx1 first, then idx0 (pairs A1/A2)
|
||
const v1 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(alpha_out + 4)))));
|
||
const v0 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(alpha_out)))));
|
||
wf32(output + 0xF * 4, (v1 * getShortToFloat() - v0 * getShortToFloat()) * t + v0 * getShortToFloat());
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
fn colorAnimLoop(this: u32, model_hdr: u32) void {
|
||
// Assembly: entry gate at model_hdr+0x64, loop bound at model_hdr+0x6C
|
||
if (ru32(model_hdr + 0x64) == 0) return;
|
||
const count = ru32(model_hdr + 0x6C); // loop bound from assembly 0x715F0A
|
||
const data_base = ru32(model_hdr + 0x68);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const out_base = ru32(this + SO.color_anim_out);
|
||
|
||
var i: u32 = 0;
|
||
var data_off: u32 = 0;
|
||
var out_off: u32 = 0;
|
||
while (i < count) : ({
|
||
i += 1;
|
||
data_off += 0x1C;
|
||
out_off += 0x20;
|
||
}) {
|
||
const anim_data = data_base + data_off;
|
||
const output = out_base + out_off;
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(anim_data + 0x04)) {
|
||
findInterpIdx(this, ru32(bone_rt_base + BR.prim_time), ru32(bone_rt_base + BR.prim_track), anim_data, output);
|
||
// Short value interpolation via game's getIndexOffset/setShortValue
|
||
const mode = ri16(anim_data);
|
||
const table = anim_data + AD.nvalues; // ECX = anim_data + 0x14
|
||
if (mode == 0) {
|
||
const sv = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(output)))));
|
||
wf32(output + 0x0C, sv * getShortToFloat());
|
||
} else {
|
||
const t = ufloat(ru32(output + 8));
|
||
const v1 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(output + 4)))));
|
||
const v0 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(output)))));
|
||
wf32(output + 0x0C, (v1 * getShortToFloat() - v0 * getShortToFloat()) * t + v0 * getShortToFloat());
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
|
||
const count = ru32(model_hdr + 0x74);
|
||
if (count == 0) return;
|
||
|
||
// One-time global init (assembly 0x715F45-0x715F81)
|
||
// Sets {0.5, 0.5, 0.0} constants at 0xCF043C and calls 0x409AEF
|
||
if ((ru8(0xCF04C4) & 1) == 0) {
|
||
wu8(0xCF04C4, ru8(0xCF04C4) | 1);
|
||
wu32(0xCF043C, 0x3F000000); // 0.5f
|
||
wu32(0xCF0440, 0x3F000000); // 0.5f
|
||
wu32(0xCF0444, 0x00000000); // 0.0f
|
||
// CALL 0x409AEF with arg 0x7187E0 (__cdecl, 1 stack param)
|
||
const initFn: *const fn (u32) callconv(.c) void = @ptrFromInt(0x409AEF);
|
||
initFn(0x7187E0);
|
||
}
|
||
|
||
const data_base = ru32(model_hdr + 0x78);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const scale2_base = ru32(this + SO.scale2);
|
||
const scale3_base = ru32(this + SO.scale3);
|
||
|
||
var i: u32 = 0;
|
||
var data_off: u32 = 0;
|
||
var out_off: u32 = 0;
|
||
var mat_off: u32 = 0;
|
||
while (i < count) : ({
|
||
i += 1;
|
||
data_off += 0x54; // assembly at 0x7163A2: ADD EDI, 0x54
|
||
out_off += 0x98; // assembly at 0x7163A5: ADD ESI, 0x98
|
||
mat_off += 0x40; // assembly at 0x715395: ADD EDX, 0x40
|
||
}) {
|
||
const kf_data = data_base + data_off;
|
||
const output = @as(u32, @intCast(@as(i32, @bitCast(scale2_base)) + @as(i32, @bitCast(out_off))));
|
||
const mat_out = @as(u32, @intCast(@as(i32, @bitCast(scale3_base)) + @as(i32, @bitCast(mat_off))));
|
||
|
||
// Init identity matrix for this keyframe entry
|
||
setIdentity(mat_out);
|
||
|
||
// Rotation: AnimData at kf_entry+0x1C, gate at kf_entry+0x28
|
||
// Assembly at 0x715FDB: CMP [ECX+0x28], 0; AnimData at EDX+0x1C
|
||
if (ru32(kf_data + 0x28) != 0) {
|
||
// Assembly: CALL 0x713EA0 — interpAnimKF
|
||
const interpKF: *const fn (u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x713EA0);
|
||
interpKF(this, bone_rt_base, kf_data + 0x1C, output + 0x30);
|
||
// Assembly: PUSH 0xCF043C, MOV ECX=mat, CALL 0x7BDC40 — applyTranslation
|
||
const applyTrans: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x7BDC40);
|
||
applyTrans(mat_out, 0, 0xCF043C);
|
||
// Assembly: PUSH quat_ptr, MOV ECX=mat, CALL 0x7BDDB0 — rotateByQuaternion
|
||
const rotateQuat: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x7BDDB0);
|
||
rotateQuat(mat_out, 0, output + 0x3C);
|
||
// Assembly: negate 0xCF043C values to stack, PUSH, CALL 0x7BDC40
|
||
var neg_trans: [3]f32 = .{ -rf32(0xCF043C), -rf32(0xCF0440), -rf32(0xCF0444) };
|
||
applyTrans(mat_out, 0, @intFromPtr(&neg_trans));
|
||
}
|
||
|
||
// Scale: AnimData at kf_entry+0x38, gate at kf_entry+0x44
|
||
// Assembly at 0x716052: CMP [ECX+0x44], 0; AnimData at EDX+0x38
|
||
if (ru32(kf_data + 0x44) != 0) {
|
||
interpVec3Track(this, bone_rt_base, kf_data + 0x38, output + 0x68, ufloat(ru32(bone_rt_base + BR.blend_weight)));
|
||
// Assembly: PUSH 0xCF043C, MOV ECX=mat, CALL 0x7BDC40
|
||
const applyTrans2: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x7BDC40);
|
||
applyTrans2(mat_out, 0, 0xCF043C);
|
||
// Assembly: PUSH scale_vec, MOV ECX=mat, CALL 0x7BDCA0
|
||
const scaleMat2: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x7BDCA0);
|
||
scaleMat2(mat_out, 0, output + 0x74);
|
||
// Assembly: negate, CALL 0x7BDC40
|
||
var neg_trans2: [3]f32 = .{ -rf32(0xCF043C), -rf32(0xCF0440), -rf32(0xCF0444) };
|
||
applyTrans2(mat_out, 0, @intFromPtr(&neg_trans2));
|
||
}
|
||
|
||
// Translation: AnimData at kf_entry+0x00, gate at kf_entry+0x0C
|
||
// Assembly at 0x716216: CMP [ECX+0x0C], 0; AnimData at kf_entry+0x00
|
||
if (ru32(kf_data + 0x0C) != 0) {
|
||
interpVec3Track(this, bone_rt_base, kf_data, output, ufloat(ru32(bone_rt_base + BR.blend_weight)));
|
||
// Assembly: PUSH trans_vec, MOV ECX=mat, CALL 0x7BDC40
|
||
const applyTrans3: *const fn (u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x7BDC40);
|
||
applyTrans3(mat_out, 0, output + 0x0C);
|
||
}
|
||
}
|
||
}
|
||
|
||
fn particleLoops(this: u32, model_hdr: u32) void {
|
||
// Particle emitters are the largest section (~1000 lines of decompiled C).
|
||
// They follow the same interpolation patterns but with many sub-tracks per emitter.
|
||
// For the initial implementation, we handle the key tracks (position, speed, scale).
|
||
// The remaining tracks (color, alpha, emission rate, etc.) use identical patterns.
|
||
|
||
// Ribbon emitters (model_hdr + 0x11C)
|
||
ribbonEmitterLoop(this, model_hdr);
|
||
|
||
// Particle emitters (model_hdr + 0x124)
|
||
particleEmitterLoop(this, model_hdr);
|
||
|
||
// Additional particle sections (model_hdr + 0x134, 0x13C)
|
||
additionalParticleLoops(this, model_hdr);
|
||
}
|
||
|
||
fn ribbonEmitterLoop(this: u32, model_hdr: u32) void {
|
||
const count = ru32(model_hdr + 0x11C);
|
||
if (count == 0) return;
|
||
const data_base = ru32(model_hdr + 0x120);
|
||
const out_base = ru32(this + SO.field_200);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const frame_ctr = ru32(this + SO.anim_frame_ctr);
|
||
|
||
var i: u32 = 0;
|
||
while (i < count) : (i += 1) {
|
||
const entry = data_base + i * 0xD4; // asm 0x716ABC: ADD EDI, 0xD4
|
||
const output = out_base + i * 0x170; // asm 0x716AC2: ADD ESI, 0x170
|
||
const bone_idx = @as(u32, ru16(entry + 2));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
|
||
// ---- Visibility byte animation (asm 0x7163FC-0x7164F2) ----
|
||
if (ru32(output + 0x100) != 0) {
|
||
if (ru32(entry + 0xC4) != 0) {
|
||
findInterpIdx(this, ru32(bone_rt + BR.prim_time), ru32(bone_rt + BR.prim_track), entry + 0xB8, output + 0xE0);
|
||
const vis_idx0 = ru32(output + 0xE0);
|
||
const vis_values = ru32(entry + 0xD0); // entry+0xB8+0x18 = AD.keyframe_base
|
||
wu8(output + 0xEC, ru8(vis_values + vis_idx0));
|
||
if (ri16(entry + 0xB8) != 0) {
|
||
if (rf32(bone_rt + BR.blend_weight) != 0.0 and ri16(entry + 0xBA) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt + BR.sec_time), ru32(bone_rt + BR.sec_track), entry + 0xB8, output + 0xF0);
|
||
wu8(output + 0xFC, ru8(vis_values + ru32(output + 0xF0)));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// ---- Visibility gate (asm 0x7164F2-0x716514) ----
|
||
const should_process = blk: {
|
||
if (ru32(output + 0x100) != 0 and ru8(output + 0xEC) != 0) break :blk true;
|
||
if (frame_ctr == 0) break :blk true;
|
||
break :blk false;
|
||
};
|
||
if (!should_process) continue;
|
||
|
||
// ---- Track A (float): gate=entry+0x38, AD=entry+0x2C, output+0x30 ----
|
||
if (frame_ctr < ru32(entry + 0x38)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0x2C, output + 0x30);
|
||
}
|
||
|
||
// ---- Track B (Vec3): gate=entry+0x1C, AD=entry+0x10, output+0x00 ----
|
||
if (frame_ctr < ru32(entry + 0x1C)) {
|
||
interpVec3Track(this, bone_rt, entry + 0x10, output, ufloat(ru32(bone_rt + BR.blend_weight)));
|
||
// Post-processing 1 (asm 0x71678A-0x7167CE)
|
||
const scale1 = rf32(output + 0x3C) * rf32(this + SO.render_scale_z);
|
||
wf32(output + 0x134, rf32(output + 0x0C) * scale1);
|
||
wf32(output + 0x138, rf32(output + 0x10) * scale1);
|
||
wf32(output + 0x13C, rf32(output + 0x14) * scale1);
|
||
}
|
||
|
||
// ---- Track C (float): gate=entry+0x70, AD=entry+0x64, output+0x80 ----
|
||
if (frame_ctr < ru32(entry + 0x70)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0x64, output + 0x80);
|
||
}
|
||
|
||
// ---- Track D (Vec3): gate=entry+0x54, AD=entry+0x48, output+0x50 ----
|
||
if (frame_ctr < ru32(entry + 0x54)) {
|
||
interpVec3Track(this, bone_rt, entry + 0x48, output + 0x50, ufloat(ru32(bone_rt + BR.blend_weight)));
|
||
// Post-processing 2 (asm 0x716A67-0x716AA6)
|
||
const scale2 = rf32(output + 0x8C) * rf32(this + SO.render_scale_z);
|
||
wf32(output + 0x140, rf32(output + 0x5C) * scale2);
|
||
wf32(output + 0x144, rf32(output + 0x60) * scale2);
|
||
wf32(output + 0x148, rf32(output + 0x64) * scale2);
|
||
}
|
||
}
|
||
}
|
||
|
||
fn particleEmitterLoop(this: u32, model_hdr: u32) void {
|
||
const count = ru32(model_hdr + 0x124);
|
||
if (count == 0) return;
|
||
const data_base = ru32(model_hdr + 0x128);
|
||
const out_base = ru32(this + SO.particle1);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const frame_ctr = ru32(this + SO.anim_frame_ctr);
|
||
|
||
var i: u32 = 0;
|
||
var data_off: u32 = 0;
|
||
var out_off: u32 = 0;
|
||
while (i < count) : ({
|
||
i += 1;
|
||
data_off += 0x7C;
|
||
out_off += 0x84;
|
||
}) {
|
||
const entry = data_base + data_off;
|
||
const output = out_base + out_off;
|
||
|
||
// Assembly uses bone_rt_base directly (bone 0) — NOT per-entry bone_idx.
|
||
|
||
if (frame_ctr < ru32(entry + 0x1C)) {
|
||
interpVec3Track36(this, bone_rt_base, entry + 0x10, output);
|
||
}
|
||
if (frame_ctr < ru32(entry + 0x44)) {
|
||
interpVec3Track36(this, bone_rt_base, entry + 0x38, output + 0x30);
|
||
}
|
||
if (frame_ctr < ru32(entry + 0x6C)) {
|
||
interpFloatTrack12(this, bone_rt_base, entry + 0x60, output + 0x60);
|
||
}
|
||
}
|
||
}
|
||
|
||
fn additionalParticleLoops(this: u32, model_hdr: u32) void {
|
||
// Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A)
|
||
// Then additional_remaining reset at 0x717D6F
|
||
// Then model_hdr+0x13C section (asm 0x717D75-0x7185E3)
|
||
|
||
// Section 12c: model_hdr+0x134 particle visibility/tracks
|
||
// count=+0x134, data=+0x138, output=this+0x3C8
|
||
// Data stride 0xDC, output stride 0xD0
|
||
// Each entry: bone_idx at +0x04, visibility at +0xCC
|
||
// Sub-tracks: visibility(+0xC0), position(+0x24), alpha(+0x40),
|
||
// speed(+0x5C), emission(+0x78), scale(+0xA4)
|
||
if (ru32(model_hdr + 0x134) != 0) {
|
||
const count0 = ru32(model_hdr + 0x134);
|
||
const data_base0 = ru32(model_hdr + 0x138);
|
||
const out_base0 = ru32(this + 0x3C8); // SO.particle2
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
|
||
var i: u32 = 0;
|
||
var data_off: u32 = 0;
|
||
var out_off: u32 = 0;
|
||
while (i < count0) : ({
|
||
i += 1;
|
||
data_off += 0xDC; // asm 0x717D4D
|
||
out_off += 0xD0; // asm 0x717D53
|
||
}) {
|
||
const entry = data_base0 + data_off;
|
||
const output = out_base0 + out_off;
|
||
|
||
// Visibility check: entry+0xCC vs anim_frame_ctr
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) {
|
||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
// Visibility byte animation at entry+0xC0
|
||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xC0, output + 0xB0);
|
||
const vis_mode = ri16(entry + 0xC0);
|
||
if (vis_mode == 0) {
|
||
wu8(output + 0xBC, ru8(ru32(entry + 0xC0 + 0x18) + ru32(output + 0xB0)));
|
||
} else {
|
||
wu8(output + 0xBC, ru8(ru32(output + 0xB0) + ru32(entry + 0xD8)));
|
||
// Crossfade blend for visibility if needed
|
||
if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xC2) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xC0, output + 0xC0);
|
||
wu8(output + 0xCC, ru8(ru32(output + 0xC0) + ru32(entry + 0xD8)));
|
||
}
|
||
}
|
||
}
|
||
|
||
// Position track: entry+0x24 vs entry+0x30
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x30)) {
|
||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
interpVec3Track(this, bone_rt, entry + 0x24, output, ufloat(ru32(bone_rt + BR.blend_weight)));
|
||
}
|
||
|
||
// Alpha track: entry+0x40 vs entry+0x4C
|
||
// Short-value interpolation via game's getIndexOffset/setShortValue
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x4C)) {
|
||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x40, output + 0x30);
|
||
const alpha_mode = ri16(entry + 0x40);
|
||
const table = entry + 0x40 + AD.nvalues;
|
||
if (alpha_mode == 0) {
|
||
const sv = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(output + 0x30)))));
|
||
wf32(output + 0x3C, sv * getShortToFloat());
|
||
} else {
|
||
const t = ufloat(ru32(output + 0x38));
|
||
const v1 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(output + 0x34)))));
|
||
const v0 = @as(f32, @floatFromInt(@as(i32, readShortViaGame(table, ru32(output + 0x30)))));
|
||
wf32(output + 0x3C, (v1 * getShortToFloat() - v0 * getShortToFloat()) * t + v0 * getShortToFloat());
|
||
}
|
||
}
|
||
|
||
// Speed track: entry+0x5C vs entry+0x68
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x68)) {
|
||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
interpFloatTrack(this, bone_rt, entry + 0x5C, output + 0x50);
|
||
}
|
||
|
||
// Emission rate: entry+0x78 vs entry+0x84
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x84)) {
|
||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
interpFloatTrack(this, bone_rt, entry + 0x78, output + 0x70);
|
||
}
|
||
|
||
// Scale track: entry+0xA4 vs entry+0xB0
|
||
// Short value copy via game's getIndexOffset/setShortValue
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) {
|
||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xA4, output + 0x90);
|
||
const scale_table = entry + 0xA4 + AD.nvalues;
|
||
// Mode 0: copy short value at idx0
|
||
// Mode != 0: also copy idx0 short (this track uses raw short output, not float lerp)
|
||
const ptr0 = callGetIndexOffset(scale_table, ru32(output + 0x90));
|
||
callSetShortValue(output + 0x9C, ptr0);
|
||
if (ri16(entry + 0xA4) != 0) {
|
||
// Crossfade
|
||
if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xA6) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xA4, output + 0xA0);
|
||
const ptr_sec = callGetIndexOffset(scale_table, ru32(output + 0xA0));
|
||
callSetShortValue(output + 0xAC, ptr_sec);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Additional remaining data reset — between 0x134 and 0x13C sections
|
||
// Assembly at 0x717D6F: MOV [EBX+0x3D8], 0
|
||
wu32(this + 0x3D8, 0);
|
||
|
||
// Section 12e: model_hdr+0x13C (largest particle section)
|
||
// count=+0x13C, data=+0x140
|
||
// output1=this+0x3D0, output2=this+0x3D4
|
||
// Data stride 0x1F8, output stride 0x16C
|
||
const count1 = ru32(model_hdr + 0x13C);
|
||
if (count1 != 0) {
|
||
const data_base = ru32(model_hdr + 0x140);
|
||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||
const particle_base = ru32(this + 0x3D0); // SO.particle3
|
||
|
||
var i: u32 = 0;
|
||
var data_off: u32 = 0;
|
||
var out_off: u32 = 0;
|
||
while (i < count1) : ({
|
||
i += 1;
|
||
data_off += 0x1F8; // asm 0x7185CD
|
||
out_off += 0x16C; // asm 0x7185BA
|
||
}) {
|
||
const entry = data_base + data_off;
|
||
const output = particle_base + out_off;
|
||
const bone_idx = @as(u32, ru16(entry + 0x14));
|
||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||
|
||
// All tracks from assembly 0x717D90-0x7185E3:
|
||
const particle_ptrs = ru32(this + 0x3D4); // [EBX+0x3D4]
|
||
const local_14 = ru32(particle_ptrs + i * 4); // per-emitter data ptr
|
||
|
||
// Visibility: gate=entry+0x1E8, AnimData=entry+0x1DC, output=output+0x140
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x1E8)) {
|
||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x1DC, output + 0x140);
|
||
if (ri16(entry + 0x1DC) == 0) {
|
||
wu8(output + 0x14C, ru8(ru32(entry + 0x1F4) + ru32(output + 0x140)));
|
||
} else {
|
||
wu8(output + 0x14C, ru8(ru32(output + 0x140) + ru32(entry + 0x1F4)));
|
||
if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0x1DE) == -1) {
|
||
findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0x1DC, output + 0x150);
|
||
wu8(output + 0x15C, ru8(ru32(entry + 0x1F4) + ru32(output + 0x150)));
|
||
}
|
||
}
|
||
}
|
||
|
||
// Emitter active flag: visibility && emitter_enable_flag
|
||
const vis_byte = ru8(output + 0x14C);
|
||
const emitter_active: u32 = if (vis_byte != 0 and ru32(this + 0x50) != 0) 1 else 0;
|
||
wu32(output + 0x160, emitter_active);
|
||
// IsParticleBufferEmpty check
|
||
var buf_active: u32 = 0;
|
||
if (emitter_active != 0) {
|
||
buf_active = 1;
|
||
} else {
|
||
// Call IsParticleBufferEmpty (0x7B5F60)
|
||
// Assembly: MOV ECX,[EBP-0x10]; CALL 0x7B5F60
|
||
// __thiscall(ECX=ptr), plain RET, returns 0 or 1 in EAX
|
||
const isEmptyFn: *const fn (u32) callconv(.{ .x86_fastcall = .{} }) u32 = @ptrFromInt(0x7B5F60);
|
||
if (isEmptyFn(local_14) != 0) {
|
||
buf_active = 1;
|
||
}
|
||
}
|
||
wu32(output + 0x164, buf_active);
|
||
// OR into additional_remaining
|
||
wu32(this + 0x3D8, ru32(this + 0x3D8) | buf_active);
|
||
|
||
// Only process tracks if visible or first frame
|
||
if (vis_byte != 0 or ru32(this + SO.anim_frame_ctr) == 0) {
|
||
// Track 1: emission rate — gate=+0x40, AnimData=+0x34, output=+0x00
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x40)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0x34, output);
|
||
}
|
||
// Track 2: speed — gate=+0x5C, AnimData=+0x50, output=+0x20
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x5C)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20);
|
||
}
|
||
// Track 3: color — gate=+0x78, AnimData=+0x6C, output=+0x40
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x78)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0x6C, output + 0x40);
|
||
}
|
||
// Track 4 — gate=+0x94, AnimData=+0x88, output=+0x60
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x94)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0x88, output + 0x60);
|
||
}
|
||
// Track 5 (Vec3 spline) — gate=+0xB0, AnimData=+0xA4, output=+0x80
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0xA4, output + 0x80);
|
||
}
|
||
// Track 6 — gate=+0xCC, AnimData=+0xC0, output=+0xA0
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) {
|
||
interpFloatTrack(this, bone_rt, entry + 0xC0, output + 0xA0);
|
||
}
|
||
// Track 7 — gate=+0xE8, AnimData=+0xDC, output=+0xC0
|
||
// Uses getInterpolatedFloat (0x71AF20)
|
||
// Tracks 7-10: CALL 0x71AF20 — getInterpolatedFloat
|
||
// __fastcall(ECX=this, EDX=bone_rt, stack: anim_data, output)
|
||
const getInterpFloat: *const fn (u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x71AF20);
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xE8)) {
|
||
getInterpFloat(this, bone_rt, entry + 0xDC, output + 0xC0);
|
||
}
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x104)) {
|
||
getInterpFloat(this, bone_rt, entry + 0xF8, output + 0xE0);
|
||
}
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x120)) {
|
||
getInterpFloat(this, bone_rt, entry + 0x114, output + 0x100);
|
||
}
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x13C)) {
|
||
getInterpFloat(this, bone_rt, entry + 0x130, output + 0x120);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32) void {
|
||
const hierarchy = ru32(this + SO.hierarchy_ptr);
|
||
if (hierarchy == 0) return;
|
||
|
||
const attach_count = ru32(model_hdr + 0x104);
|
||
if (attach_count == 0) return;
|
||
const attach_data = ru32(model_hdr + 0x108);
|
||
|
||
// Process attachment byte animations
|
||
var att_i: u32 = 0;
|
||
var att_off: u32 = 0;
|
||
while (att_i < attach_count) : ({
|
||
att_i += 1;
|
||
att_off += 0x30;
|
||
}) {
|
||
const att_entry = attach_data + att_off;
|
||
if (ru32(this + SO.anim_frame_ctr) < ru32(att_entry + 0x20)) {
|
||
const bone_idx = @as(u32, ru16(att_entry + 4));
|
||
const bone_rt = ru32(this + SO.bone_rt_base) + bone_idx * 0x118;
|
||
// Assembly: CALL 0x71AE90 — extractAnimationByteFromKeyframes
|
||
// __fastcall(ECX=this, EDX=bone_rt, stack: anim_data, output)
|
||
const extractByte: *const fn (u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x71AE90);
|
||
extractByte(this, bone_rt, att_entry + 0x14, hierarchy + att_i * 0x20);
|
||
}
|
||
}
|
||
|
||
// Iterate child scene objects linked list
|
||
var child = ru32(this + SO.hierarchy_idx);
|
||
while (child != 0) {
|
||
// child->attach_idx at +0x1D4 (assembly-verified: MOV EAX,[ECX+0x1D4] at 0x718668)
|
||
const attach_idx = ru32(child + 0x1D4);
|
||
|
||
// Check if attachment is valid (0xFFFF = no attachment)
|
||
if (attach_idx != 0xFFFF) {
|
||
const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C);
|
||
if (visible != 0) {
|
||
const att_entry = attach_data + attach_idx * 0x30;
|
||
const bone_idx = @as(u32, ru16(att_entry + 4));
|
||
const bone_mat = bone_out_base + bone_idx * 0x40;
|
||
|
||
// Copy parent bone matrix to local
|
||
var local_1a0: [16]f32 = undefined;
|
||
for (0..16) |fi| {
|
||
local_1a0[fi] = rf32(bone_mat + @as(u32, @intCast(fi)) * 4);
|
||
}
|
||
|
||
// Apply attachment offset translation
|
||
const ox = rf32(att_entry + 8);
|
||
const oy = rf32(att_entry + 0xC);
|
||
const oz = rf32(att_entry + 0x10);
|
||
local_1a0[12] += local_1a0[0] * ox + local_1a0[4] * oy + local_1a0[8] * oz;
|
||
local_1a0[13] += local_1a0[1] * ox + local_1a0[5] * oy + local_1a0[9] * oz;
|
||
local_1a0[14] += local_1a0[2] * ox + local_1a0[6] * oy + local_1a0[10] * oz;
|
||
|
||
// Recursive call through 0x714260, matching original's CALL 0x714260.
|
||
// Goes through hook → detour → REF for child SceneObjects.
|
||
const callThrough: *const fn (u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) void = @ptrFromInt(0x714260);
|
||
callThrough(child, 0, @intFromPtr(&local_1a0), this + SO.world_pos, this + SO.render_pri, ru32(this + SO.render_scale_z));
|
||
}
|
||
}
|
||
|
||
// Next sibling in linked list
|
||
// Assembly-verified: MOV ECX,[ECX+0x1E4] at 0x718764
|
||
child = ru32(child + 0x1E4);
|
||
}
|
||
}
|