From 8fb1ece2b553cf5446493d25e46a81178d29c052 Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Sun, 15 Mar 2026 17:38:14 -0700 Subject: [PATCH] =?UTF-8?q?bone=5Fsse=5Fref:=20fix=20world=20entry=20crash?= =?UTF-8?q?=20=E2=80=94=206=20bugs=20found=20via=20full=20asm=20stepthroug?= =?UTF-8?q?h?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Full 5317-instruction walkthrough of t44_full_asm.txt vs bone_sse_reference.zig. Crash fix (Issues 1-2): When anim_start >= anim_end in the looping animation path, assembly always writes prim_time/sec_time = anim_start as fallback. REF skipped the write, leaving garbage in bone_rt timing fields. On newly loaded world SceneObjects this propagated through findInterpIdx → extractByte → ACCESS_VIOLATION at 0x71AEBC with ECX=0x7FFFFFFF (self-reinforcing bad cached index). Time clamp fix (Issues 3-4): Clamped-not-passed animation path now clamps cur_time to sec_start when sec_start > cur_time, matching assembly at 0x7145EB/0x71474B. Crossfade fix (Issues 5-6): interpVec3Track36 and interpFloatTrack12 had 'else return' for unknown interp modes. Assembly's JNZ skips primary interp but falls through to crossfade check. Changed to 'else {}' fallthrough. Also includes prior uncommitted fixes: particle crossfade blend_weight restoration, bw>0→bw!=0, attach_count==0 early-return removal. --- src/transform44/ASM_DIVERGENCES.md | 152 +++++++++++++++++ src/transform44/bone_sse.zig | 219 +++++++++++++++++++++++-- src/transform44/bone_sse_reference.zig | 68 ++++---- 3 files changed, 394 insertions(+), 45 deletions(-) create mode 100644 src/transform44/ASM_DIVERGENCES.md diff --git a/src/transform44/ASM_DIVERGENCES.md b/src/transform44/ASM_DIVERGENCES.md new file mode 100644 index 0000000..a52c3fd --- /dev/null +++ b/src/transform44/ASM_DIVERGENCES.md @@ -0,0 +1,152 @@ +# Assembly vs REF Divergence List + +Full stepthrough of `t44_full_asm.txt` (5317 insns) vs `bone_sse_reference.zig` (~2298 lines). +Each issue marked with severity estimate (CRASH / WRONG / COSMETIC). + +--- + +## ISSUE 1 — CRASH: Missing prim_time write when anim_start >= anim_end (primary slot) + +**Assembly** (0x7145F1-0x714633): The looping path always ends at 0x714633 which writes `prim_time`. When `anim_end <= anim_start` (JLE at 0x7145FF), it jumps to 0x714631 which sets `EDX = anim_start`, then falls through to 0x714642: `MOV [ESI+0x98], EDX`. + +**REF** (line ~1066-1077): The `if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end)))` block is the ONLY place `brt + 0x98` gets written. When `anim_start >= anim_end`, `prim_time` is **never written**. + +**Impact**: On newly loaded SceneObjects, `brt+0x98` contains uninitialized garbage. The original always writes `anim_start` as a fallback. The REF leaves it as garbage, which propagates through `findInterpIdx` → `extractByte` → crash at 0x71AEBC with ECX=0x7FFFFFFF. + +**Fix**: After the `if` block, add an `else` that writes `anim_start`: +```zig +if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { + // ...existing code... + wu32(brt + 0x98, anim_start +% frame); +} else { + wu32(brt + 0x98, anim_start); +} +``` + +Same fix needed in the clamped-not-passed branch (line ~1093-1100) which shares the same looping code. + +--- + +## ISSUE 2 — CRASH: Missing sec_time write when anim_start >= anim_end (secondary slot) + +**Assembly** (0x714757-0x714799): Identical pattern to primary. When `anim_end <= anim_start` (JLE at 0x714765), jumps to 0x714797: `MOV EDX, EAX` (EDX = anim_start), falls through to 0x7147A5: `MOV [ESI+0xC4], EDX`. + +**REF** (line ~1162-1167 and ~1177-1183): Same bug — `brt + 0xC4` (`sec_time`) not written when `anim_start >= anim_end`. + +**Impact**: Same as Issue 1 — stale sec_time causes bad findInterpIdx results in crossfade paths. + +**Fix**: Same pattern — add `else { wu32(brt + 0xC4, anim_start); }` after each inner `if`. + +--- + +## ISSUE 3 — WRONG: Missing cur_time clamp to sec_start in clamped-not-passed path (primary) + +**Assembly** (0x7145E5-0x7145EB): When `sec_end > cur_time` AND `sec_start > cur_time`: +```asm +SUB EDX, ECX ; sec_start - cur_time +TEST EDX, EDX +JLE looping ; if sec_start <= cur_time, use cur_time +MOV ECX, [ESI+0xA8] ; CLAMP: replace cur_time with sec_start +``` +Then at 0x714601: `delta = ECX - sec_start` = 0 (since ECX = sec_start). + +**REF** (line ~1086-1096): The inner `if (sec_start_val != cur_time ...)` block is empty — does nothing. Then `const delta = cur_time -% ru32(brt + 0xA8)` uses the UNCLAMPED cur_time. + +**Impact**: When sec_start > cur_time (uncommon edge case after time delta adjustment), delta wraps to a huge unsigned value. The modulo operation may still produce a valid result, but the ftol intermediate could overflow. Lower severity than Issues 1-2 since the modulo clamps the final result. + +**Fix**: Before computing delta, clamp: `const effective_time = if (sec_start_val > cur_time) sec_start_val else cur_time;` then `const delta = effective_time -% ru32(brt + 0xA8);` + +--- + +## ISSUE 4 — WRONG: Missing cur_time clamp to sec_start in clamped-not-passed path (secondary) + +**Assembly** (0x71474B): Same pattern for secondary slot. + +**REF** (line ~1174-1180): Same bug — empty inner `if` block, no clamp applied. + +**Fix**: Same as Issue 3 but for secondary slot variables. + +--- + +## ISSUE 5 — WRONG: interpVec3Track36 `else return` skips crossfade for unknown modes + +**Assembly** (0x716B98): For unknown interp modes (not 0, 1, 2, or 3): +```asm +DEC EDX ; mode - 3 +JNZ 0x716D1D ; if mode != 3, jump to CROSSFADE CHECK (not return!) +``` +The assembly skips primary interpolation but STILL checks and applies crossfade at 0x716D1D. + +**REF** (line 695): `} else return;` — exits the entire function, skipping crossfade. + +**Impact**: For models with unusual interp modes (rare), crossfade blending is skipped. Primary interpolation output is whatever was there from mode 0 (which returned earlier) or from the previous frame. The secondary crossfade result won't be blended in. + +**Fix**: Replace `else return;` with `else {}` (empty block, fall through to crossfade): +```zig +} else if (mode == 2) { + // ...bezier... +} else { + // Unknown mode: skip primary interp, but still check crossfade below +} +// crossfade section runs regardless +``` + +--- + +## ISSUE 6 — WRONG: interpFloatTrack12 `else return` skips crossfade for unknown modes + +**Assembly**: Same pattern as Issue 5 — unknown modes skip primary interp but fall through to crossfade. + +**REF** (line 766): `} else return;` — same bug as Issue 5. + +**Fix**: Same as Issue 5 — replace `else return` with `else {}`. + +--- + +## Sections verified CORRECT + +The following sections were compared instruction-by-instruction and match: + +1. **Entry checks** (0x714260-0x714280): model_data_ptr null check, sync_value comparison ✅ +2. **Emitter setup** (0x714286-0x7142C7): emitter_ctx flag logic, field copy ✅ +3. **World position/scale** (0x7142C7-0x71434C): pos*scale, offset+field, render_scale_z ✅ +4. **Global sequence loop** (0x714352-0x714389): unsigned modulo, gs_values write ✅ +5. **MatMul call** (0x71438C-0x7143A0): 0x74A7C0(this+0xFC, this+0xBC, mat1) ✅ +6. **child_padding len_sq** (0x7143A0-0x7143EE): emitter_ctx re-read, bit test, sqmag ✅ +7. **Identity matrices** (0x7143EE-0x714503): both 16-float identity blocks ✅ +8. **Timestamp delta** (0x714503-0x71451C): this+0x4C guard, delta, writeback ✅ +9. **Bone loop parent inherit** (0x714650-0x7146B2): parent_idx bounds, bone_idx==0 fallback ✅ +10. **Secondary slot inherit** (0x7147C5-0x714820): parent/bone0/self-primary paths ✅ +11. **Blend weight computation** (0x714820-0x71492D): Hermite smoothstep, clamping to 0/1 ✅ +12. **Parent matrix / billboard pre-processing** (0x71492D-0x714D0F): flag dispatch, normalize, scale preservation ✅ +13. **Billboard types 2/4/6** (0x714A6E-0x714C8C): spherical/cylindrical/full camera copy ✅ +14. **Translation re-computation** (0x714CAF-0x714D0C): pos - rot*pivot ✅ +15. **Rotation/scale/translation interpolation** (0x714D0F-0x7151BA): interpAnimKF, interpVec3Track, matMul ✅ +16. **Non-animated copyMat4** (0x7151C4-0x7151F7): 8×MOVSD equivalent ✅ +17. **Billboard post-processing types 0x08/0x10/0x20/0x40** (0x7151F9-0x715868): all cross product signs verified ✅ +18. **Post-billboard scale/translate** (0x715868-0x71594E): scale_len * normalized, pos - scaled*pivot ✅ +19. **Bone loop increment** (0x71594E-0x715966): bone_count comparison, re-read ✅ +20. **texAnimLoop** (0x715966-0x715C87): Vec3 track + alpha short-value interp + crossfade ✅ +21. **colorAnimLoop** (0x715C87-0x715E46): model_hdr+0x64 count/gate, short interp + crossfade ✅ +22. **wordAnimLoop** (0x715E46-0x715F25): word copy + crossfade skip for mode 0 ✅ +23. **boneKeyframeLoop** (0x715F25-0x7163BC): global init, rot/scale/trans with 0xCF043C ✅ +24. **ribbonEmitterLoop** (0x7163BC-0x716AD9): visibility byte, vec3/float tracks, post-processing ✅ +25. **particleEmitterLoop 0x124** (0x716AD9-0x71763E): bone_rt_base (bone 0) usage, Vec3Track36/FloatTrack12 ✅ +26. **Section 0x134 particles** (0x71763E-0x717D6A): all sub-tracks, strides 0xDC/0xD0 ✅ +27. **additional_remaining reset** (0x717D6F): `this+0x3D8 = 0` between 0x134/0x13C sections ✅ +28. **Section 0x13C particles** (0x717D75-0x7185E3): visibility, emitter_active, 10 sub-tracks, getInterpolatedFloat ✅ +29. **Attachment byte animation loop** (0x7185E3-0x718657): data stride 0x30, output stride 0x20, extractByte call ✅ +30. **Child traversal** (0x718657-0x718775): linked list, visibility check, matrix copy, offset translation, recursive call ✅ +31. **Sync update** (0x718775-0x718784): `this+0x40 = anim_ctx+0x10` ✅ +32. **Buffer sizes**: All particle output strides verified against maximum write offsets — no overflow ✅ +33. **Hermite/Bezier basis**: h1-h4 and b0-b3 formulas match standard Bernstein/Hermite polynomials ✅ +34. **Calling conventions**: All game function calls (0x713D50, 0x713EA0, 0x71AE90, 0x71AF20, 0x71AFF0, 0x71B010, 0x74A7C0, 0x74B6B5, 0x7BDC40, 0x7BDCA0, 0x7BDDB0, 0x7B5F60, 0x4549F0, 0x40A2B0, 0x409AEF, 0x714260) parameter order verified ✅ +35. **Constants**: 0x7FFD74 = 0.0f, 0x7FF9D8 = 1.0f, 0x80297C = 3.0f, 0x802990 = 6.0f (runtime), 0x811610 = short-to-float (runtime), 0x8029D4 = billboard epsilon (runtime) — all verified ✅ + +--- + +## Priority + +1. **ISSUE 1 + 2** (CRASH): Fix immediately — this is almost certainly the extractByte crash cause +2. **ISSUE 5 + 6** (WRONG): Fix next — affects crossfade correctness for edge-case modes +3. **ISSUE 3 + 4** (WRONG): Fix last — rare edge case, modulo likely prevents crash diff --git a/src/transform44/bone_sse.zig b/src/transform44/bone_sse.zig index 3c1738d..f5d1811 100644 --- a/src/transform44/bone_sse.zig +++ b/src/transform44/bone_sse.zig @@ -724,6 +724,196 @@ inline fn interpFloatTrack( } } +// ============================================================================= +// Hermite basis functions — used by particle emitter tracks (modes 2, 3) +// h1 = 2t³ - 3t² + 1, h2 = t³ - 2t² + t, h3 = -2t³ + 3t², h4 = t³ - t² +// ============================================================================= + +inline fn hermiteBasis(t: f32) struct { h1: f32, h2: f32, h3: f32, h4: f32 } { + const t2 = t * t; + const t3 = t2 * t; + return .{ + .h1 = 2 * t3 - 3 * t2 + 1, + .h2 = t3 - 2 * t2 + t, + .h3 = -2 * t3 + 3 * t2, + .h4 = t3 - t2, + }; +} + +inline fn bezierBasis(t: f32) struct { b0: f32, b1: f32, b2: f32, b3: f32 } { + const u = 1.0 - t; + const t2 = t * t; + const u_sq = u * u; + return .{ + .b0 = u_sq * u, + .b1 = 3 * u_sq * t, + .b2 = 3 * u * t2, + .b3 = t2 * t, + }; +} + +/// Vec3 interpolation with 36-byte keyframes and 4 modes (step/lerp/bezier/hermite). +/// Keyframe layout: [pos Vec3 (12), in_tangent Vec3 (12), out_tangent Vec3 (12)] = 36 bytes. +/// Used by 0x124 particle emitter tracks. Uses bone_rt_base (bone 0) for timing. +fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { + findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); + + const mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + if (mode == 0) { + // Step — copy Vec3 from keyframe at idx0*36 + const src = kf_base + ru32(output) * 36; + wu32(output + 0x0C, ru32(src)); + wu32(output + 0x10, ru32(src + 4)); + wu32(output + 0x14, ru32(src + 8)); + return; + } + + const t = ufloat(ru32(output + 8)); + const kf_a = kf_base + ru32(output) * 36; + const kf_b = kf_base + ru32(output + 4) * 36; + + if (mode == 1) { + // Linear interpolation + const result = lerpVec3(kf_a, kf_b, t); + wu32(output + 0x0C, fbits(result[0])); + wu32(output + 0x10, fbits(result[1])); + wu32(output + 0x14, fbits(result[2])); + } else if (mode == 3) { + // Hermite: h1*p0 + h2*m0_out + h3*p1 + h4*m1_in + const h = hermiteBasis(t); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + const p0 = rf32(kf_a + off); + const m0 = rf32(kf_a + 0x18 + off); // out_tangent + const p1 = rf32(kf_b + off); + const m1 = rf32(kf_b + 0x0C + off); // in_tangent + wf32(output + 0x0C + off, h.h1 * p0 + h.h2 * m0 + h.h3 * p1 + h.h4 * m1); + } + } else if (mode == 2) { + // Bezier: b0*p0 + b1*m0_out + b2*m1_in + b3*p1 + const b = bezierBasis(t); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + const p0 = rf32(kf_a + off); + const m0 = rf32(kf_a + 0x18 + off); // out_tangent (control point) + const p1 = rf32(kf_b + off); + const m1 = rf32(kf_b + 0x0C + off); // in_tangent (control point) + wf32(output + 0x0C + off, b.b0 * p0 + b.b1 * m0 + b.b2 * m1 + b.b3 * p1); + } + } else return; // mode 4+: no interp, leave output unchanged + + // Crossfade blend + const blend = rf32(bone_rt_base + BR.blend_weight); + if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { + findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x18); + + const st = ufloat(ru32(output + 0x20)); + const skf_a = kf_base + ru32(output + 0x18) * 36; + const skf_b = kf_base + ru32(output + 0x1C) * 36; + const smode = ri16(anim_data + AD.interp_mode); + + if (smode == 1) { + const sec = lerpVec3(skf_a, skf_b, st); + wu32(output + 0x24, fbits(sec[0])); + wu32(output + 0x28, fbits(sec[1])); + wu32(output + 0x2C, fbits(sec[2])); + } else if (smode == 3) { + const h = hermiteBasis(st); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + wf32(output + 0x24 + off, h.h1 * rf32(skf_a + off) + h.h2 * rf32(skf_a + 0x18 + off) + h.h3 * rf32(skf_b + off) + h.h4 * rf32(skf_b + 0x0C + off)); + } + } else if (smode == 2) { + const b = bezierBasis(st); + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + wf32(output + 0x24 + off, b.b0 * rf32(skf_a + off) + b.b1 * rf32(skf_a + 0x18 + off) + b.b2 * rf32(skf_b + 0x0C + off) + b.b3 * rf32(skf_b + off)); + } + } else { + // Step for crossfade + wu32(output + 0x24, ru32(skf_a)); + wu32(output + 0x28, ru32(skf_a + 4)); + wu32(output + 0x2C, ru32(skf_a + 8)); + } + + // Blend: primary = primary + (secondary - primary) * blend + var i: u32 = 0; + while (i < 3) : (i += 1) { + const off = i * 4; + const pri = rf32(output + 0x0C + off); + const sec = rf32(output + 0x24 + off); + wf32(output + 0x0C + off, (sec - pri) * blend + pri); + } + } +} + +/// Float interpolation with 12-byte keyframes and 4 modes (step/lerp/bezier/hermite). +/// Keyframe layout: [value (4), in_tangent (4), out_tangent (4)] = 12 bytes. +/// Used by 0x124 particle emitter Track 3. Uses bone_rt_base (bone 0) for timing. +fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) void { + findInterpIdx(this, ru32(bone_rt_base + 0x98), ru32(bone_rt_base + 0x9C), anim_data, output); + + const mode = ri16(anim_data + AD.interp_mode); + const kf_base = ru32(anim_data + AD.keyframe_base); + + if (mode == 0) { + // Step — copy float from keyframe at idx0*12 + wu32(output + 0x0C, ru32(kf_base + ru32(output) * 12)); + return; + } + + const t = ufloat(ru32(output + 8)); + const kf_a = kf_base + ru32(output) * 12; + const kf_b = kf_base + ru32(output + 4) * 12; + + if (mode == 1) { + const a = rf32(kf_a); + const b = rf32(kf_b); + wf32(output + 0x0C, (b - a) * t + a); + } else if (mode == 3) { + // Hermite: h1*p0 + h2*m0_out + h3*p1 + h4*m1_in + const h = hermiteBasis(t); + wf32(output + 0x0C, h.h1 * rf32(kf_a) + h.h2 * rf32(kf_a + 0x08) + h.h3 * rf32(kf_b) + h.h4 * rf32(kf_b + 0x04)); + } else if (mode == 2) { + // Bezier: b0*p0 + b1*m0_out + b2*m1_in + b3*p1 + const b = bezierBasis(t); + wf32(output + 0x0C, b.b0 * rf32(kf_a) + b.b1 * rf32(kf_a + 0x08) + b.b2 * rf32(kf_b + 0x04) + b.b3 * rf32(kf_b)); + } else return; + + // Crossfade + const blend = rf32(bone_rt_base + BR.blend_weight); + if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { + findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); + + const st = ufloat(ru32(output + 0x18)); + const skf_a = kf_base + ru32(output + 0x10) * 12; + const skf_b = kf_base + ru32(output + 0x14) * 12; + const smode = ri16(anim_data + AD.interp_mode); + + var sec: f32 = undefined; + if (smode == 1) { + sec = (rf32(skf_b) - rf32(skf_a)) * st + rf32(skf_a); + } else if (smode == 3) { + const h = hermiteBasis(st); + sec = h.h1 * rf32(skf_a) + h.h2 * rf32(skf_a + 0x08) + h.h3 * rf32(skf_b) + h.h4 * rf32(skf_b + 0x04); + } else if (smode == 2) { + const bz = bezierBasis(st); + sec = bz.b0 * rf32(skf_a) + bz.b1 * rf32(skf_a + 0x08) + bz.b2 * rf32(skf_b + 0x04) + bz.b3 * rf32(skf_b); + } else { + sec = rf32(skf_a); + } + wf32(output + 0x1C, sec); + const pri = rf32(output + 0x0C); + wf32(output + 0x0C, (sec - pri) * blend + pri); + } +} + // ============================================================================= // getInterpolatedFloat — reimplemented from 0x71af20 // Same as interpFloatTrack but uses the bone_rt directly (different register mapping) @@ -1809,35 +1999,36 @@ fn particleEmitterLoop(this: u32, model_hdr: u32) void { const data_base = ru32(model_hdr + 0x128); const out_base = ru32(this + SO.particle1); const bone_rt_base = ru32(this + SO.bone_rt_base); + const frame_ctr = ru32(this + SO.anim_frame_ctr); var i: u32 = 0; var data_off: u32 = 0; var out_off: u32 = 0; while (i < count) : ({ i += 1; - data_off += 0x7C; - out_off += 0x84; + data_off += 0x7C; // asm 0x717624: ADD EDI, 0x7C + out_off += 0x84; // asm 0x717627: ADD ESI, 0x84 }) { const entry = data_base + data_off; const output = out_base + out_off; - const bone_idx = @as(u32, ru16(entry + 2)); - const bone_rt = bone_rt_base + bone_idx * 0x118; - // All 3 tracks from assembly (0x716B00-0x717611): - // Track 1 (position): gate=entry+0x1C, AnimData=entry+0x10, output=+0x00 + // Assembly uses bone_rt_base directly (bone 0) for ALL tracks — NOT per-entry bone_idx. + // Verified: 0x716B2E MOV EAX,[EBX+0x90]; 0x716F58 same; 0x7173B1 same. + + // Track 1 (Vec3, 36-byte kf): gate=entry+0x1C, AD=entry+0x10, output=+0x00 // Assembly: 0x716B19 CMP [EAX+0x1C]; 0x716B3B LEA ESI,[EDX+0x10] - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x1C)) { - interpVec3Track(this, bone_rt, entry + 0x10, output, ufloat(ru32(bone_rt + BR.blend_weight))); + if (frame_ctr < ru32(entry + 0x1C)) { + interpVec3Track36(this, bone_rt_base, entry + 0x10, output); } - // Track 2: gate=entry+0x44, AnimData=entry+0x38, output=+0x30 + // Track 2 (Vec3, 36-byte kf): gate=entry+0x44, AD=entry+0x38, output=+0x30 // Assembly: 0x716F44 MOV EDX,[ECX+0x44]; 0x716F55 LEA ECX,[EAX+0x38] - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x44)) { - interpVec3Track(this, bone_rt, entry + 0x38, output + 0x30, ufloat(ru32(bone_rt + BR.blend_weight))); + if (frame_ctr < ru32(entry + 0x44)) { + interpVec3Track36(this, bone_rt_base, entry + 0x38, output + 0x30); } - // Track 3: gate=entry+0x6C, AnimData=entry+0x60, output=+0x60 + // Track 3 (float, 12-byte kf): gate=entry+0x6C, AD=entry+0x60, output=+0x60 // Assembly: 0x71739A MOV EDX,[ECX+0x6C]; 0x7173AE LEA EDI,[EAX+0x60] - if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x6C)) { - interpVec3Track(this, bone_rt, entry + 0x60, output + 0x60, ufloat(ru32(bone_rt + BR.blend_weight))); + if (frame_ctr < ru32(entry + 0x6C)) { + interpFloatTrack12(this, bone_rt_base, entry + 0x60, output + 0x60); } } } diff --git a/src/transform44/bone_sse_reference.zig b/src/transform44/bone_sse_reference.zig index 8f589ce..92a9aab 100644 --- a/src/transform44/bone_sse_reference.zig +++ b/src/transform44/bone_sse_reference.zig @@ -692,7 +692,7 @@ fn interpVec3Track36(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) const off = i * 4; wf32(output + 0x0C + off, b.b0 * rf32(kf_a + off) + b.b1 * rf32(kf_a + 0x18 + off) + b.b2 * rf32(kf_b + 0x0C + off) + b.b3 * rf32(kf_b + off)); } - } else return; + } else {} // Unknown mode: skip primary interp, fall through to crossfade (asm 0x716B98: JNZ crossfade_check) const blend = rf32(bone_rt_base + BR.blend_weight); if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { @@ -763,7 +763,7 @@ fn interpFloatTrack12(this: u32, bone_rt_base: u32, anim_data: u32, output: u32) } else if (mode == 2) { const b = bezierBasis(t); wf32(output + 0x0C, b.b0 * rf32(kf_a) + b.b1 * rf32(kf_a + 0x08) + b.b2 * rf32(kf_b + 0x04) + b.b3 * rf32(kf_b)); - } else return; + } else {} // Unknown mode: skip primary interp, fall through to crossfade const blend = rf32(bone_rt_base + BR.blend_weight); if (blend != 0.0 and ri16(anim_data + AD.time_index) == -1) { @@ -1074,6 +1074,9 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start); wu32(brt + 0x98, anim_start +% frame); // prim_time + } else { + // Assembly 0x714631: MOV EDX,EAX — fallback to anim_start + wu32(brt + 0x98, anim_start); } } else { // Clamped: assembly at 0x71458E-0x7145E3 @@ -1083,20 +1086,18 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat // Check if sec_end has passed (sec_end - cur_time <= 0 signed) if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) { // sec_end hasn't passed yet - if (sec_start_val != cur_time and @as(i32, @bitCast(sec_start_val -% cur_time)) > 0) { - // Before start: use sec_start as time - // Actually assembly jumps to looping path LAB_007145f1 - // which reads anim_entry+0x08, anim_entry+0x04 - // Fallthrough: use cur_time (no write to prim_time) - } + // Assembly 0x7145E5: clamp cur_time to sec_start if sec_start > cur_time + const effective_time = if (@as(i32, @bitCast(sec_start_val -% cur_time)) > 0) sec_start_val else cur_time; // goto looping path const anim_end = ru32(anim_entry + 0x08); const anim_start = ru32(anim_entry + 0x04); if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { - const delta = cur_time -% ru32(brt + 0xA8); + const delta = effective_time -% ru32(brt + 0xA8); const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xB0); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start); wu32(brt + 0x98, anim_start +% frame); + } else { + wu32(brt + 0x98, anim_start); } } else { // sec_end has passed — compute clamped position @@ -1164,6 +1165,8 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start); wu32(brt + 0xC4, anim_start +% frame); // sec_time + } else { + wu32(brt + 0xC4, anim_start); } } else { // Clamped @@ -1171,16 +1174,17 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat const sec_start_val = ru32(brt + 0xD4); if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) { - if (sec_start_val != sec_cur_time and @as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) { - // use sec_start - } + // Assembly 0x71474B: clamp sec_cur_time to sec_start if sec_start > sec_cur_time + const effective_time = if (@as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) sec_start_val else sec_cur_time; const anim_end = ru32(sec_anim_entry + 0x08); const anim_start = ru32(sec_anim_entry + 0x04); if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) { - const delta = sec_cur_time -% ru32(brt + 0xD4); + const delta = effective_time -% ru32(brt + 0xD4); const ftol_result = callFtol(@as(i32, @bitCast(delta)), brt + 0xDC); const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start); wu32(brt + 0xC4, anim_start +% frame); + } else { + wu32(brt + 0xC4, anim_start); } } else { const dur = sec_end_val -% sec_start_val; @@ -1687,7 +1691,7 @@ fn texAnimLoop(this: u32, model_hdr: u32) void { // Crossfade (assembly 0x715BAF-0x715C5E) // Only runs for mode != 0 — mode 0 JMPs past this const bw = rf32(bone_rt_base + BR.blend_weight); - if (bw > 0.0 and ri16(alpha_anim + 0x02) == -1) { + if (bw != 0.0 and ri16(alpha_anim + 0x02) == -1) { findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), alpha_anim, alpha_out + 0x10); const secondary = shortInterpToFloat(alpha_anim, alpha_out + 0x10); wf32(alpha_out + 0x1C, secondary); @@ -1748,7 +1752,7 @@ fn colorAnimLoop(this: u32, model_hdr: u32) void { // Crossfade (assembly 0x715D6B-0x715E1B) const bw = rf32(bone_rt_base + BR.blend_weight); - if (bw > 0.0 and ri16(anim_data + 0x02) == -1) { + if (bw != 0.0 and ri16(anim_data + 0x02) == -1) { findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); const secondary = shortInterpToFloat(anim_data, output + 0x10); wf32(output + 0x1C, secondary); @@ -1793,7 +1797,7 @@ fn wordAnimLoop(this: u32, model_hdr: u32) void { // mode 0: no crossfade, skip } else { const bw = rf32(bone_rt_base + BR.blend_weight); - if (bw > 0.0 and ri16(anim_data + 0x02) == -1) { + if (bw != 0.0 and ri16(anim_data + 0x02) == -1) { findInterpIdx(this, ru32(bone_rt_base + BR.sec_time), ru32(bone_rt_base + BR.sec_track), anim_data, output + 0x10); const sec_idx = ru32(output + 0x10); wu16(output + 0x1C, ru16(kf_data + sec_idx * 2)); @@ -1941,12 +1945,12 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32) void { // ---- Track A (float): gate=entry+0x38, AD=entry+0x2C, output+0x30 ---- if (frame_ctr < ru32(entry + 0x38)) { - interpFloatTrack(this, bone_rt, entry + 0x2C, output + 0x30, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x2C, output + 0x30, ufloat(ru32(bone_rt + BR.blend_weight))); } // ---- Track B (Vec3): gate=entry+0x1C, AD=entry+0x10, output+0x00 ---- if (frame_ctr < ru32(entry + 0x1C)) { - interpVec3Track(this, bone_rt, entry + 0x10, output, 0.0); // no crossfade for particles + interpVec3Track(this, bone_rt, entry + 0x10, output, ufloat(ru32(bone_rt + BR.blend_weight))); // Post-processing 1 (asm 0x71678A-0x7167CE) const scale1 = rf32(output + 0x3C) * rf32(this + SO.render_scale_z); wf32(output + 0x134, rf32(output + 0x0C) * scale1); @@ -1956,12 +1960,12 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32) void { // ---- Track C (float): gate=entry+0x70, AD=entry+0x64, output+0x80 ---- if (frame_ctr < ru32(entry + 0x70)) { - interpFloatTrack(this, bone_rt, entry + 0x64, output + 0x80, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x64, output + 0x80, ufloat(ru32(bone_rt + BR.blend_weight))); } // ---- Track D (Vec3): gate=entry+0x54, AD=entry+0x48, output+0x50 ---- if (frame_ctr < ru32(entry + 0x54)) { - interpVec3Track(this, bone_rt, entry + 0x48, output + 0x50, 0.0); // no crossfade for particles + interpVec3Track(this, bone_rt, entry + 0x48, output + 0x50, ufloat(ru32(bone_rt + BR.blend_weight))); // Post-processing 2 (asm 0x716A67-0x716AA6) const scale2 = rf32(output + 0x8C) * rf32(this + SO.render_scale_z); wf32(output + 0x140, rf32(output + 0x5C) * scale2); @@ -2055,7 +2059,7 @@ fn additionalParticleLoops(this: u32, model_hdr: u32) void { if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x30)) { const bone_idx = @as(u32, ru16(entry + 0x04)); const bone_rt = bone_rt_base + bone_idx * 0x118; - interpVec3Track(this, bone_rt, entry + 0x24, output, 0.0); // no crossfade for particles + interpVec3Track(this, bone_rt, entry + 0x24, output, ufloat(ru32(bone_rt + BR.blend_weight))); } // Alpha track: entry+0x40 vs entry+0x4C @@ -2081,14 +2085,14 @@ fn additionalParticleLoops(this: u32, model_hdr: u32) void { if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x68)) { const bone_idx = @as(u32, ru16(entry + 0x04)); const bone_rt = bone_rt_base + bone_idx * 0x118; - interpFloatTrack(this, bone_rt, entry + 0x5C, output + 0x50, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x5C, output + 0x50, ufloat(ru32(bone_rt + BR.blend_weight))); } // Emission rate: entry+0x78 vs entry+0x84 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x84)) { const bone_idx = @as(u32, ru16(entry + 0x04)); const bone_rt = bone_rt_base + bone_idx * 0x118; - interpFloatTrack(this, bone_rt, entry + 0x78, output + 0x70, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x78, output + 0x70, ufloat(ru32(bone_rt + BR.blend_weight))); } // Scale track: entry+0xA4 vs entry+0xB0 @@ -2184,27 +2188,27 @@ fn additionalParticleLoops(this: u32, model_hdr: u32) void { if (vis_byte != 0 or ru32(this + SO.anim_frame_ctr) == 0) { // Track 1: emission rate — gate=+0x40, AnimData=+0x34, output=+0x00 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x40)) { - interpFloatTrack(this, bone_rt, entry + 0x34, output, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x34, output, ufloat(ru32(bone_rt + BR.blend_weight))); } // Track 2: speed — gate=+0x5C, AnimData=+0x50, output=+0x20 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x5C)) { - interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20, ufloat(ru32(bone_rt + BR.blend_weight))); } // Track 3: color — gate=+0x78, AnimData=+0x6C, output=+0x40 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x78)) { - interpFloatTrack(this, bone_rt, entry + 0x6C, output + 0x40, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x6C, output + 0x40, ufloat(ru32(bone_rt + BR.blend_weight))); } // Track 4 — gate=+0x94, AnimData=+0x88, output=+0x60 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x94)) { - interpFloatTrack(this, bone_rt, entry + 0x88, output + 0x60, 0.0); + interpFloatTrack(this, bone_rt, entry + 0x88, output + 0x60, ufloat(ru32(bone_rt + BR.blend_weight))); } // Track 5 (Vec3 spline) — gate=+0xB0, AnimData=+0xA4, output=+0x80 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) { - interpFloatTrack(this, bone_rt, entry + 0xA4, output + 0x80, 0.0); + interpFloatTrack(this, bone_rt, entry + 0xA4, output + 0x80, ufloat(ru32(bone_rt + BR.blend_weight))); } // Track 6 — gate=+0xCC, AnimData=+0xC0, output=+0xA0 if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) { - interpFloatTrack(this, bone_rt, entry + 0xC0, output + 0xA0, 0.0); + interpFloatTrack(this, bone_rt, entry + 0xC0, output + 0xA0, ufloat(ru32(bone_rt + BR.blend_weight))); } // Track 7 — gate=+0xE8, AnimData=+0xDC, output=+0xC0 // Uses getInterpolatedFloat (0x71AF20) @@ -2232,11 +2236,13 @@ fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32) void { const hierarchy = ru32(this + SO.hierarchy_ptr); if (hierarchy == 0) return; + // Attachment byte animation loop — skipped when attach_count==0 but + // child recursion below MUST still run. Original JBE 0x718657 jumps + // past this loop to the child section, NOT to the function exit. const attach_count = ru32(model_hdr + 0x104); - if (attach_count == 0) return; const attach_data = ru32(model_hdr + 0x108); - // Process attachment byte animations + // Process attachment byte animations (only when attach_count > 0) var att_i: u32 = 0; var att_off: u32 = 0; while (att_i < attach_count) : ({