bone_sse: assembly-verified reimplementation of transformMatrix4x4
13 bugs fixed by comparing against full assembly dump (5317 instructions): - Emitter check: this+0x188 -> this+0x1D8 - Animation time: added FILD*time_scale pattern for both primary (+0xB0) and secondary (+0xDC) slots - Conditional multiply: bone_local *= *(bone_rt+0xF0) was missing - Billboard post-processing: 4 switch cases (types 8/16/32/64) implemented - Color animation loop bound: model_hdr+0x64 -> +0x6C - Bone keyframe data stride: 0x24 -> 0x54 - Ribbon emitter output stride: 0x15C -> 0x170 - Particle data/output strides: 0x1FC/0x17C -> 0x1F8/0x16C - Child SceneObject offsets: attach_idx +0x184->+0x1D4, next +0x190->+0x1E4 - Root bone parent: identity -> this+0xFC New files: - BONE_SSE_PROGRESS.md: section-by-section verification status - t44_full_asm.txt: complete function assembly (ground truth) - t44_helpers_asm.txt: all 12 helper function assemblies - math_sse.zig: 18 x87->SSE polyfill stubs (VanillaFixes integration) Also: OnWorldUpdate hook for true per-frame counting, DUMP_FRAMES=450. SSE dispatch currently disabled while particle sections are being verified.
This commit is contained in:
@@ -31,6 +31,7 @@ const module_list = [_]ModuleDesc{
|
||||
.{ .name = "transform44", .desc = "Enable transformMatrix4x4 hook", .default = false },
|
||||
.{ .name = "addonperf", .desc = "Enable addon memory/CPU profiling API" },
|
||||
.{ .name = "filecache", .desc = "Enable MPQ archive file cache", .default = false },
|
||||
.{ .name = "silicon", .desc = "Enable SSE2 math replacements (ported from libSiliconPatch)", .default = false },
|
||||
};
|
||||
|
||||
pub fn build(b: *std.Build) void {
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
# bone_sse.zig — Faithful Reproduction Progress
|
||||
|
||||
## Principle
|
||||
|
||||
Exact reproduction of transformMatrix4x4 (0x714260, 17703 bytes) in Zig.
|
||||
Every section verified against assembly at src/transform44/decompiled/t44_full_asm.txt.
|
||||
The decompilation is NOT the primary source — assembly is authoritative.
|
||||
|
||||
## Assembly References
|
||||
|
||||
- `decompiled/t44_full_asm.txt` — 5317 instructions, complete function
|
||||
- `decompiled/t44_helpers_asm.txt` — 11 helper functions
|
||||
- `SCENEOBJECT_OFFSETS.md` — 51 [EBX+N] offsets
|
||||
|
||||
## Section Status
|
||||
|
||||
### Section 1: Entry checks (asm lines 5-16)
|
||||
- `[EBX+0x10]` NULL check, `[EBX+0x40]` vs `[EAX+0x10]` sync check
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 2: Emitter setup (asm lines 17-36)
|
||||
- emitter_ctx+0x50 AND this+0x1D8 check (NOT +0x188!)
|
||||
- emitter_flag at this+0x50, copy emitter_ctx+0x17C to this+0x17C
|
||||
- **VERIFIED** ✓ — bug fixed: was using +0x188, now +0x1D8
|
||||
|
||||
### Section 3: World position/scale (asm lines 37-74)
|
||||
- pos = mat2[i] * this[0x184/0x188/0x18C], stored at this+0x1A0
|
||||
- offset = mat3[i] + this[0x190/0x194/0x198], stored at this+0x1AC
|
||||
- render_scale_z = mat4 * this[0x180], stored at this+0x19C
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 4: Global sequences (asm lines 75-95)
|
||||
- count/durations from model_hdr+0x14/0x18
|
||||
- timestamp from anim_ctx+0xC, base from this+0x68
|
||||
- values ptr at this+0x64, unsigned DIV for modulo
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 5: initPPSG matrix multiply (asm lines 96-102)
|
||||
- output=this+0xFC, left=this+0xBC, right=mat1
|
||||
- Calling multiplyMatrix4x4_Basic (0x7507BB) directly
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 6: child_objects_padding (asm lines 103-126)
|
||||
- emitter_ctx NULL or bit0 set: len_sq of world_xform[8,9,10]
|
||||
- else: copy from emitter_ctx+0x84
|
||||
- Stored at this+0x84
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 7: Identity matrices + timestamp delta (asm lines 127-175)
|
||||
- Two 4x4 identity matrices (EBP-0x70 and EBP-0xE4)
|
||||
- Time delta from this+0x4C (search_data_base_ptr)
|
||||
- Bone loop setup: count from model_hdr+0x34, defs from +0x38
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 8a: Bone loop start (asm lines 177-189)
|
||||
- bone_def = defs + idx*0x6C, bone_rt = base + idx*0x118
|
||||
- anim_slot at bone_rt+0xA4, checked against -1
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 8b: Animation time computation (asm lines 190-262)
|
||||
- FILD * time_scale(+0xB0) → __ftol pattern for time conversion
|
||||
- Looping vs clamped via anim_entry+0x10 bit 0
|
||||
- **VERIFIED** ✓ — bug fixed: was missing time_scale multiplication
|
||||
|
||||
### Section 8c: Secondary animation time (asm lines 284-364)
|
||||
- Same pattern with time_scale at +0xDC
|
||||
- Expiry check: timestamp - crossfade_end(+0x100) >= 0 → expire slot
|
||||
- **VERIFIED** ✓ — bug fixed: was "copy from primary", now full computation
|
||||
|
||||
### Section 8d: Blend weight / crossfade (asm lines 382-447)
|
||||
- Hermite: (3.0 - 2*t) * t * t * crossfade_weight(+0x108)
|
||||
- t = (float)remaining * crossfade_inv(+0x104)
|
||||
- Clamped to [ZERO_THRESHOLD, 1.0]
|
||||
- **PARTIALLY VERIFIED** — formula matches, float comparison semantics unchecked
|
||||
|
||||
### Section 8e: Parent inheritance + billboard PRE-processing (asm lines 448-478+)
|
||||
- Root: src = this+0xFC. Child: src = bone_out + parent*0x40
|
||||
- combined_flags = bone_rt[0xF4] | bone_def[0x04]
|
||||
- Billboard types 2/4/6 via (flags & 6)
|
||||
- **PARTIALLY VERIFIED** — offsets correct, billboard math from decompilation
|
||||
|
||||
### Section 8f: Animation flags check (asm ~0x714D11)
|
||||
- (combined_flags & 0x280) == 0: copy parent directly
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 8g: Rotation (asm 0x714D1F-0x714D52)
|
||||
- rot_nts at bone_def+0x34, rot AnimData at bone_def+0x28
|
||||
- buildRotMatFromQuat at 0x74B6B5 OVERWRITES matrix (does NOT multiply)
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 8h: Scale + conditional multiply (asm 0x714DF7-0x714FA1)
|
||||
- scale_nts at bone_def+0x50, scale AnimData at bone_def+0x44
|
||||
- Conditional: if (char)flags < 0 AND bone_rt[0xF0] != 0:
|
||||
bone_local *= *(bone_rt+0xF0) via matmul 0x74A7C0
|
||||
- **VERIFIED** ✓ — bug fixed: conditional multiply was missing
|
||||
|
||||
### Section 8i: Translation (asm 0x714FA1-0x7151BA)
|
||||
- trans_nts at bone_def+0x18, trans AnimData at bone_def+0x0C
|
||||
- offset = (pivot + interp) - matrix * pivot
|
||||
- Final multiply: output = bone_local × parent via matmul 0x74A7C0
|
||||
- **PARTIALLY VERIFIED** — structure matches, inline SSE multiply used
|
||||
|
||||
### Section 8k: Billboard POST-processing (asm 0x7151F9-0x71594E)
|
||||
- flags & 0x78, switch on types 8/16/32/64
|
||||
- Scale lengths preserved, translation recomputed
|
||||
- **IMPLEMENTED** — from assembly, cross product formulas need double-check
|
||||
|
||||
### Section 9: Texture animation (asm 0x715966-0x715C87)
|
||||
- count=model_hdr+0x54, data=+0x58, output=this+0xA0
|
||||
- Data stride 0x38, output stride 0x50
|
||||
- **VERIFIED** ✓ (strides confirmed from asm 0x715C70/0x715C73)
|
||||
|
||||
### Section 10: Color animation (asm 0x715C87-0x715F1F)
|
||||
- Entry gate=model_hdr+0x64, loop bound=model_hdr+0x6C, data=+0x68
|
||||
- output=this+0xA8, data stride 0x1C, output stride 0x20
|
||||
- **VERIFIED** ✓ — bug fixed: loop bound was +0x64, now +0x6C
|
||||
|
||||
### Section 11: Bone keyframe processing (asm 0x715F25-0x7163BC)
|
||||
- count=model_hdr+0x74, data=+0x78
|
||||
- Data stride 0x54, output stride 0x98, matrix stride 0x40
|
||||
- **VERIFIED** ✓ — bug fixed: data stride was 0x24, now 0x54
|
||||
|
||||
### Section 12a: Ribbon emitters (asm 0x7163BC-0x716AD9)
|
||||
- count=model_hdr+0x11C, data=+0x120, output=this+0x200
|
||||
- Data stride 0xD4, output stride 0x170
|
||||
- **VERIFIED** ✓ — bug fixed: output stride was 0x15C, now 0x170
|
||||
|
||||
### Section 12b: Particle emitters (asm 0x716AD9-0x71763E)
|
||||
- count=model_hdr+0x124, data=+0x128, output=this+0x3C4
|
||||
- Data stride 0x7C, output stride 0x84
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 12c: Additional particles — model_hdr+0x134 (asm 0x71763E-0x717D6A)
|
||||
- count=model_hdr+0x134, data=+0x138, output=this+0x3C8
|
||||
- Data stride 0xDC, output stride 0xD0
|
||||
- Tracks: visibility(+0xC0/+0xCC), position(+0x24/+0x30), alpha(+0x40/+0x4C),
|
||||
speed(+0x5C/+0x68), emission(+0x78/+0x84), scale(+0xA4/+0xB0)
|
||||
- **IMPLEMENTED** — visibility + 5 tracks from assembly
|
||||
|
||||
### Section 12d: additional_remaining reset (asm 0x717D6A-0x717D6F)
|
||||
- `[EBX+0x3D8] = 0` between sections 12c and 12e
|
||||
- **FIXED** — now correctly placed between 0x134 and 0x13C sections
|
||||
|
||||
### Section 12e: Final particle section — model_hdr+0x13C (asm 0x717D6A-0x7185E3)
|
||||
- count=model_hdr+0x13C, data=+0x140
|
||||
- output1=this+0x3D0, output2=this+0x3D4
|
||||
- Data stride 0x1F8, output stride 0x16C
|
||||
- All 10 tracks + visibility implemented from assembly:
|
||||
visibility(+0x1DC/+0x1E8), emission(+0x34/+0x40), speed(+0x50/+0x5C),
|
||||
color(+0x6C/+0x78), track4(+0x88/+0x94), Vec3_spline(+0xA4/+0xB0),
|
||||
track6(+0xC0/+0xCC), track7(+0xDC/+0xE8), track8(+0xF8/+0x104),
|
||||
track9(+0x114/+0x120), track10(+0x130/+0x13C)
|
||||
- IsParticleBufferEmpty call at 0x7B5F60
|
||||
- additional_remaining OR at [EBX+0x3D8]
|
||||
- **VERIFIED** ✓ (strides and track offsets from assembly)
|
||||
|
||||
### Section 13: Attachment recursion (asm 0x7185E3-0x718784)
|
||||
- Child attach_idx at child+0x1D4, next_sibling at child+0x1E4
|
||||
- **VERIFIED** ✓
|
||||
|
||||
### Section 14: Sync update (asm 0x718775-0x718784)
|
||||
- this+0x40 = *(anim_ctx + 0x10)
|
||||
- **VERIFIED** ✓
|
||||
|
||||
## Bug Fix Log
|
||||
|
||||
| # | Bug | Wrong | Correct | Assembly ref |
|
||||
|---|-----|-------|---------|-------------|
|
||||
| 1 | Emitter check field | this+0x188 | this+0x1D8 | line 28 |
|
||||
| 2 | Primary anim time | no time_scale | FILD*[brt+0xB0] | 0x7145AB |
|
||||
| 3 | Secondary anim time | copy from primary | full w/ [brt+0xDC] | 0x714711 |
|
||||
| 4 | Conditional multiply | missing | bone_local*=*(brt+0xF0) | 0x714F8D |
|
||||
| 5 | Billboard post-proc | TODO/missing | 4 switch cases | 0x7151F9 |
|
||||
| 6 | Color loop bound | model_hdr+0x64 | +0x6C | 0x715F0A |
|
||||
| 7 | Bone KF data stride | 0x24 | 0x54 | 0x7163A2 |
|
||||
| 8 | Ribbon output stride | 0x15C | 0x170 | 0x716AC2 |
|
||||
| 9 | Final particle data | 0x1FC | 0x1F8 | 0x7185CD |
|
||||
| 10 | Final particle output | 0x17C | 0x16C | 0x7185BA |
|
||||
| 11 | Child attach_idx | +0x184 | +0x1D4 | 0x718668 |
|
||||
| 12 | Child next_sibling | +0x190 | +0x1E4 | 0x718764 |
|
||||
| 13 | Root bone parent | identity | this+0xFC | 0x714945 |
|
||||
+610
-135
@@ -95,6 +95,7 @@ const BR = struct {
|
||||
// Secondary animation time range (crossfade)
|
||||
const sec_start: u32 = 0xA8; // puVar20[0x2a]
|
||||
const sec_end: u32 = 0xAC; // puVar20[0x2b]
|
||||
const time_scale: u32 = 0xB0; // puVar20[0x2c] — float scale for FILD*FMUL→__ftol time conversion
|
||||
const sec_anim_offset: u32 = 0xB8; // puVar20[0x2e]
|
||||
// Rotation interpolation (interpolateAnimationKeyframes output at +0xC*4 = 0x30)
|
||||
const rot_idx0: u32 = 0x30;
|
||||
@@ -285,8 +286,46 @@ inline fn applyTranslation(mat: u32, tx: f32, ty: f32, tz: f32) void {
|
||||
wf32(mat + 0x38, tx * rf32(mat + 0x08) + ty * rf32(mat + 0x18) + tz * rf32(mat + 0x28) + rf32(mat + 0x38));
|
||||
}
|
||||
|
||||
/// Quaternion → rotation matrix: OVERWRITES mat with the rotation matrix.
|
||||
/// Matches the original game function at 0x74B6BB which writes directly
|
||||
/// without multiplying by existing matrix contents.
|
||||
/// Used in the bone loop where the matrix starts as identity.
|
||||
inline fn buildRotationMatrix(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||||
const xx2 = qx * (qx + qx);
|
||||
const xy2 = qx * (qy + qy);
|
||||
const xz2 = qx * (qz + qz);
|
||||
const yy2 = qy * (qy + qy);
|
||||
const yz2 = qy * (qz + qz);
|
||||
const zz2 = qz * (qz + qz);
|
||||
const wx2 = qw * (qx + qx);
|
||||
const wy2 = qw * (qy + qy);
|
||||
const wz2 = qw * (qz + qz);
|
||||
|
||||
// Row 0
|
||||
wf32(mat + 0x00, 1.0 - (yy2 + zz2));
|
||||
wf32(mat + 0x04, xy2 + wz2);
|
||||
wf32(mat + 0x08, xz2 - wy2);
|
||||
wf32(mat + 0x0C, 0);
|
||||
// Row 1
|
||||
wf32(mat + 0x10, xy2 - wz2);
|
||||
wf32(mat + 0x14, 1.0 - (xx2 + zz2));
|
||||
wf32(mat + 0x18, yz2 + wx2);
|
||||
wf32(mat + 0x1C, 0);
|
||||
// Row 2
|
||||
wf32(mat + 0x20, xz2 + wy2);
|
||||
wf32(mat + 0x24, yz2 - wx2);
|
||||
wf32(mat + 0x28, 1.0 - (xx2 + yy2));
|
||||
wf32(mat + 0x2C, 0);
|
||||
// Row 3 (translation = zero, w = 1)
|
||||
wf32(mat + 0x30, 0);
|
||||
wf32(mat + 0x34, 0);
|
||||
wf32(mat + 0x38, 0);
|
||||
wf32(mat + 0x3C, 1);
|
||||
}
|
||||
|
||||
/// Quaternion → rotation matrix, then multiply: mat = quat_rot * mat.
|
||||
/// Standard quat→mat conversion + SSE 4x4 matrix multiply.
|
||||
/// Used in bone keyframe processing where matrix already has content.
|
||||
inline fn rotateByQuaternion(mat: u32, qx: f32, qy: f32, qz: f32, qw: f32) void {
|
||||
const xx2 = qx * (qx + qx);
|
||||
const xy2 = qx * (qy + qy);
|
||||
@@ -344,6 +383,21 @@ inline fn setIdentity(dst: u32) void {
|
||||
}
|
||||
}
|
||||
|
||||
/// Normalize a 3-component vector in memory at addr. Uses squaredMagnitude + sqrt + divide.
|
||||
/// Matches the original's pattern: call squaredMagnitude, sqrt, check epsilon, divide.
|
||||
inline fn normalizeVec3InPlace(addr: u32) void {
|
||||
const x = rf32(addr);
|
||||
const y = rf32(addr + 4);
|
||||
const z = rf32(addr + 8);
|
||||
const len = @sqrt(x * x + y * y + z * z);
|
||||
if (@abs(len) >= BILLBOARD_EPSILON) {
|
||||
const inv = 1.0 / len;
|
||||
wf32(addr, x * inv);
|
||||
wf32(addr + 4, y * inv);
|
||||
wf32(addr + 8, z * inv);
|
||||
}
|
||||
}
|
||||
|
||||
/// Normalize a 3-component vector, returns (nx, ny, nz). Returns unchanged if too small.
|
||||
inline fn normalizeVec3(x: f32, y: f32, z: f32) [3]f32 {
|
||||
const len_sq = x * x + y * y + z * z;
|
||||
@@ -373,7 +427,7 @@ inline fn crossVec3(ax: f32, ay: f32, az: f32, bx: f32, by: f32, bz: f32) [3]f32
|
||||
// Output: indices[0] = lower index, [1] = upper index, [2] = interpolation t (float bits)
|
||||
// =============================================================================
|
||||
|
||||
inline fn findInterpIdx(
|
||||
fn findInterpIdx(
|
||||
this: u32,
|
||||
search_value: u32,
|
||||
track_index: u32,
|
||||
@@ -740,6 +794,7 @@ fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
|
||||
// =============================================================================
|
||||
|
||||
export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) void {
|
||||
@setEvalBranchQuota(50000);
|
||||
// =========================================================================
|
||||
// Section 1: Entry checks
|
||||
// =========================================================================
|
||||
@@ -755,9 +810,10 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
const emitter_ctx = ru32(this + SO.emitter_ctx);
|
||||
|
||||
if (emitter_ctx != 0) {
|
||||
const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and rf32(this + SO.field_188) != 0.0) 1 else 0;
|
||||
wu32(this + SO.emitter_flag, has_emitter);
|
||||
wu32(this + SO.field_17c, ru32(emitter_ctx + 0x17C));
|
||||
// Assembly 0x71429E-0x7142C1: emitter_ctx+0x50 != 0 AND this+0x1D8 != 0
|
||||
const has_emitter: u32 = if (ru32(emitter_ctx + 0x50) != 0 and ru32(this + 0x1D8) != 0) 1 else 0;
|
||||
wu32(this + 0x50, has_emitter); // emitter_enable_flag
|
||||
wu32(this + 0x17C, ru32(emitter_ctx + 0x17C));
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
@@ -803,29 +859,15 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
}
|
||||
}
|
||||
|
||||
// initParticlePixelShaderGeneration (0x74a7c0) — indirect JMP thunk to matrix multiply.
|
||||
// Computes: billboard_row0 = field_0xBC × mat1 (parent transform).
|
||||
// Reimplemented as inline SSE matrix multiply (avoids uncertain calling convention
|
||||
// of the indirect JMP target — Ghidra can't resolve the target's RET type).
|
||||
// initParticlePixelShaderGeneration (0x74a7c0) — matrix multiply.
|
||||
// Computes: *(this+0xFC) = *(this+0xBC) × mat1
|
||||
// Calls multiplyMatrix4x4_Basic (0x7507BB) directly:
|
||||
// __stdcall(output=this+0xFC, left=this+0xBC, right=mat1), RET 0xC
|
||||
// Assembly-verified param order from 0x71438B:
|
||||
// PUSH mat1 (right), PUSH &0xBC (left), PUSH &0xFC (output), CALL
|
||||
{
|
||||
const left = this + 0xBC; // local transform matrix (4x4)
|
||||
const right = mat1; // parent transform matrix (4x4)
|
||||
const out = this + 0xFC; // output billboard/camera matrix
|
||||
|
||||
// SSE 4x4 matrix multiply: out = left × right
|
||||
const r0: V4 = .{ rf32(right + 0x00), rf32(right + 0x04), rf32(right + 0x08), rf32(right + 0x0C) };
|
||||
const r1: V4 = .{ rf32(right + 0x10), rf32(right + 0x14), rf32(right + 0x18), rf32(right + 0x1C) };
|
||||
const r2: V4 = .{ rf32(right + 0x20), rf32(right + 0x24), rf32(right + 0x28), rf32(right + 0x2C) };
|
||||
const r3: V4 = .{ rf32(right + 0x30), rf32(right + 0x34), rf32(right + 0x38), rf32(right + 0x3C) };
|
||||
|
||||
inline for (0..4) |i| {
|
||||
const b = @as(u32, @intCast(i)) * 0x10;
|
||||
const row = splat(rf32(left + b)) * r0 + splat(rf32(left + b + 4)) * r1 + splat(rf32(left + b + 8)) * r2 + splat(rf32(left + b + 12)) * r3;
|
||||
wf32(out + b + 0x00, row[0]);
|
||||
wf32(out + b + 0x04, row[1]);
|
||||
wf32(out + b + 0x08, row[2]);
|
||||
wf32(out + b + 0x0C, row[3]);
|
||||
}
|
||||
const matMulBasic: *const fn (u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x7507BB);
|
||||
matMulBasic(this + 0xFC, this + 0xBC, mat1);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
@@ -902,60 +944,82 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
wu32(brt + BR.prim_anim, ru32(bone_rt_base + BR.prim_anim));
|
||||
}
|
||||
} else {
|
||||
// Has own animation slot — compute time from animation lookup table
|
||||
if (sdb != 0) {
|
||||
wu32(brt + BR.sec_start, ru32(brt + BR.sec_start) +% time_delta_val);
|
||||
wu32(brt + BR.sec_end, ru32(brt + BR.sec_end) +% time_delta_val);
|
||||
// Has own animation slot — compute time from animation lookup table.
|
||||
// Assembly at 0x714561-0x71464E, verified line by line.
|
||||
if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0
|
||||
// Add time delta to sec_start/sec_end
|
||||
wu32(brt + 0xA8, ru32(brt + 0xA8) +% time_delta_val); // [ESI+0xA8]
|
||||
wu32(brt + 0xAC, ru32(brt + 0xAC) +% time_delta_val); // [ESI+0xAC]
|
||||
}
|
||||
const anim_lookup = ru32(model_hdr + 0x20);
|
||||
const anim_entry = anim_lookup + @as(u32, @intCast(anim_slot_val)) * 0x44;
|
||||
const cur_time = ru32(anim_ctx + 0x0C);
|
||||
|
||||
// Check looping vs clamped
|
||||
// anim_entry = anim_lookup_table + anim_slot * 0x44
|
||||
const anim_lookup = ru32(model_hdr + 0x20); // [EDX+0x20]
|
||||
const anim_entry = anim_lookup + @as(u32, @bitCast(anim_slot_val)) * 0x44;
|
||||
const cur_time = ru32(ru32(this + 0x2C) + 0xC); // [EBX+0x2C]+0xC = timestamp
|
||||
|
||||
// Check looping flag: [anim_entry+0x10] & 1
|
||||
if ((ru8(anim_entry + 0x10) & 1) == 0) {
|
||||
// Looping animation
|
||||
const anim_end: i32 = ri32(anim_entry + 0x08);
|
||||
const anim_start: i32 = ri32(anim_entry + 0x04);
|
||||
if (anim_start < anim_end) {
|
||||
const elapsed: f32 = @floatFromInt(@as(i32, @bitCast(cur_time -% ru32(brt + BR.sec_start))));
|
||||
const duration: u32 = @as(u32, @intCast(anim_end - anim_start));
|
||||
if (duration != 0) {
|
||||
const frame_in_anim = (@as(u32, @intFromFloat(elapsed)) +% ru32(brt + BR.sec_anim_offset)) % duration;
|
||||
wu32(brt + BR.prim_time, @as(u32, @intCast(anim_start)) +% frame_in_anim);
|
||||
}
|
||||
// Looping: assembly at 0x7145F1-0x714631
|
||||
const anim_end = ru32(anim_entry + 0x08);
|
||||
const anim_start = ru32(anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
// elapsed = (float)(cur_time - sec_start) * time_scale → __ftol
|
||||
const delta = cur_time -% ru32(brt + 0xA8);
|
||||
const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xB0)));
|
||||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start);
|
||||
wu32(brt + 0x98, anim_start +% frame); // prim_time
|
||||
}
|
||||
} else {
|
||||
// Clamped animation
|
||||
const end_time = ru32(brt + BR.sec_end);
|
||||
const start_time = ru32(brt + BR.sec_start);
|
||||
if (end_time != cur_time and @as(i32, @bitCast(end_time -% cur_time)) >= 0) {
|
||||
if (start_time != cur_time and @as(i32, @bitCast(start_time -% cur_time)) >= 0) {
|
||||
wu32(brt + BR.prim_time, start_time); // use start time
|
||||
}
|
||||
// else: use cur_time (already set from inheritance/previous)
|
||||
} else {
|
||||
// Compute clamped position
|
||||
const dur_val: i32 = @as(i32, @bitCast(end_time -% start_time));
|
||||
const elapsed_f: f32 = @floatFromInt(@as(i32, @bitCast(cur_time -% start_time)));
|
||||
_ = elapsed_f;
|
||||
const offset: i32 = @as(i32, @intFromFloat(@as(f32, @floatFromInt(dur_val)))) + @as(i32, @bitCast(ru32(brt + BR.sec_anim_offset)));
|
||||
const anim_start2: i32 = ri32(anim_entry + 0x04);
|
||||
const anim_end2: i32 = ri32(anim_entry + 0x08);
|
||||
// Clamped: assembly at 0x71458E-0x7145E3
|
||||
const sec_end_val = ru32(brt + 0xAC);
|
||||
const sec_start_val = ru32(brt + 0xA8);
|
||||
|
||||
var time_pos: u32 = undefined;
|
||||
// Check if sec_end has passed (sec_end - cur_time <= 0 signed)
|
||||
if (sec_end_val != cur_time and @as(i32, @bitCast(sec_end_val -% cur_time)) > 0) {
|
||||
// sec_end hasn't passed yet
|
||||
if (sec_start_val != cur_time and @as(i32, @bitCast(sec_start_val -% cur_time)) > 0) {
|
||||
// Before start: use sec_start as time
|
||||
// Actually assembly jumps to looping path LAB_007145f1
|
||||
// which reads anim_entry+0x08, anim_entry+0x04
|
||||
// Fallthrough: use cur_time (no write to prim_time)
|
||||
}
|
||||
// goto looping path
|
||||
const anim_end = ru32(anim_entry + 0x08);
|
||||
const anim_start = ru32(anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = cur_time -% ru32(brt + 0xA8);
|
||||
const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xB0)));
|
||||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xB8)) % (anim_end -% anim_start);
|
||||
wu32(brt + 0x98, anim_start +% frame);
|
||||
}
|
||||
} else {
|
||||
// sec_end has passed — compute clamped position
|
||||
// Assembly at 0x71458E-0x7145E3:
|
||||
// delta = (sec_end - sec_start), scaled by [ESI+0xB0]
|
||||
const dur = sec_end_val -% sec_start_val;
|
||||
const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(dur)))) * rf32(brt + 0xB0)));
|
||||
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xB8)));
|
||||
|
||||
if (offset < 0) {
|
||||
time_pos = @as(u32, @bitCast(anim_start2));
|
||||
} else if (offset <= anim_end2 - anim_start2) {
|
||||
time_pos = @as(u32, @intCast(offset + anim_start2));
|
||||
// Clamp to anim_start
|
||||
wu32(brt + 0x98, ru32(anim_entry + 0x04));
|
||||
} else {
|
||||
time_pos = @as(u32, @bitCast(anim_end2));
|
||||
const anim_end_i = @as(i32, @bitCast(ru32(anim_entry + 0x08)));
|
||||
const anim_start_i = @as(i32, @bitCast(ru32(anim_entry + 0x04)));
|
||||
if (offset <= anim_end_i - anim_start_i) {
|
||||
wu32(brt + 0x98, @as(u32, @bitCast(offset + anim_start_i)));
|
||||
} else {
|
||||
// Clamp to anim_end
|
||||
wu32(brt + 0x98, ru32(anim_entry + 0x08));
|
||||
}
|
||||
}
|
||||
wu32(brt + BR.prim_time, time_pos);
|
||||
}
|
||||
}
|
||||
wu32(brt + BR.prim_track, ru32(brt + BR.anim_slot));
|
||||
wu32(brt + BR.prim_time, ru32(brt + BR.prim_time)); // already set above
|
||||
wu32(brt + BR.prim_anim, bone_idx);
|
||||
|
||||
// Store results: assembly at 0x714633-0x71464E
|
||||
wu32(brt + 0x9C, ru32(brt + 0xA4)); // prim_track = anim_slot
|
||||
// prim_time already set above
|
||||
wu32(brt + 0xA0, bone_idx); // prim_anim = bone_idx
|
||||
}
|
||||
|
||||
// --- Secondary animation time (crossfade target) ---
|
||||
@@ -974,18 +1038,70 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
wu32(brt + BR.sec_track, ru32(brt + BR.prim_track));
|
||||
}
|
||||
} else {
|
||||
// Compute secondary time (same pattern as primary, abbreviated)
|
||||
if (sdb != 0) {
|
||||
wu32(brt + BR.sec_start2, ru32(brt + BR.sec_start2) +% time_delta_val);
|
||||
wu32(brt + BR.sec_end2, ru32(brt + BR.sec_end2) +% time_delta_val);
|
||||
// Secondary animation slot time computation.
|
||||
// Assembly at 0x7146C1-0x7147C3, mirrors primary slot logic.
|
||||
if (ru32(this + 0x4C) != 0) { // search_data_base_ptr != 0
|
||||
wu32(brt + 0xD4, ru32(brt + 0xD4) +% time_delta_val); // [ESI+0xD4]
|
||||
wu32(brt + 0xD8, ru32(brt + 0xD8) +% time_delta_val); // [ESI+0xD8]
|
||||
}
|
||||
// For now, copy from primary — full implementation would mirror primary logic
|
||||
wu32(brt + BR.sec_track, @as(u32, @bitCast(sec_slot_val)));
|
||||
wu32(brt + BR.sec_time, ru32(brt + BR.prim_time));
|
||||
|
||||
// Check expiry
|
||||
if (@as(i32, @bitCast(ru32(anim_ctx + 0x0C) -% ru32(brt + BR.crossfade_end))) >= 0) {
|
||||
wu32(brt + BR.sec_slot, 0xFFFFFFFF); // expire
|
||||
const sec_anim_lookup = ru32(model_hdr + 0x20);
|
||||
const sec_anim_entry = sec_anim_lookup + @as(u32, @bitCast(sec_slot_val)) * 0x44;
|
||||
const sec_cur_time = ru32(ru32(this + 0x2C) + 0xC);
|
||||
|
||||
if ((ru8(sec_anim_entry + 0x10) & 1) == 0) {
|
||||
// Looping
|
||||
const anim_end = ru32(sec_anim_entry + 0x08);
|
||||
const anim_start = ru32(sec_anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = sec_cur_time -% ru32(brt + 0xD4);
|
||||
const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xDC)));
|
||||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start);
|
||||
wu32(brt + 0xC4, anim_start +% frame); // sec_time
|
||||
}
|
||||
} else {
|
||||
// Clamped
|
||||
const sec_end_val = ru32(brt + 0xD8);
|
||||
const sec_start_val = ru32(brt + 0xD4);
|
||||
|
||||
if (sec_end_val != sec_cur_time and @as(i32, @bitCast(sec_end_val -% sec_cur_time)) > 0) {
|
||||
if (sec_start_val != sec_cur_time and @as(i32, @bitCast(sec_start_val -% sec_cur_time)) > 0) {
|
||||
// use sec_start
|
||||
}
|
||||
const anim_end = ru32(sec_anim_entry + 0x08);
|
||||
const anim_start = ru32(sec_anim_entry + 0x04);
|
||||
if (@as(i32, @bitCast(anim_start)) < @as(i32, @bitCast(anim_end))) {
|
||||
const delta = sec_cur_time -% ru32(brt + 0xD4);
|
||||
const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(delta)))) * rf32(brt + 0xDC)));
|
||||
const frame = (@as(u32, @bitCast(ftol_result)) +% ru32(brt + 0xE4)) % (anim_end -% anim_start);
|
||||
wu32(brt + 0xC4, anim_start +% frame);
|
||||
}
|
||||
} else {
|
||||
const dur = sec_end_val -% sec_start_val;
|
||||
const ftol_result = @as(i32, @intFromFloat(@as(f32, @floatFromInt(@as(i32, @bitCast(dur)))) * rf32(brt + 0xDC)));
|
||||
const offset = ftol_result + @as(i32, @bitCast(ru32(brt + 0xE4)));
|
||||
|
||||
if (offset < 0) {
|
||||
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x04));
|
||||
} else {
|
||||
const anim_end_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x08)));
|
||||
const anim_start_i = @as(i32, @bitCast(ru32(sec_anim_entry + 0x04)));
|
||||
if (offset <= anim_end_i - anim_start_i) {
|
||||
wu32(brt + 0xC4, @as(u32, @bitCast(offset + anim_start_i)));
|
||||
} else {
|
||||
wu32(brt + 0xC4, ru32(sec_anim_entry + 0x08));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Store results: assembly at 0x714799-0x7147C3
|
||||
wu32(brt + 0xC8, ru32(brt + 0xD0)); // sec_track = sec_slot
|
||||
// sec_time already set above
|
||||
|
||||
// Check expiry: if (timestamp - crossfade_end >= 0) expire slot
|
||||
if (@as(i32, @bitCast(ru32(ru32(this + 0x2C) + 0xC) -% ru32(brt + 0x100))) >= 0) {
|
||||
wu32(brt + 0xD0, 0xFFFFFFFF); // expire secondary slot
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1019,12 +1135,7 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
var src_mat: u32 = undefined;
|
||||
|
||||
if (ru16(bdef + BD.parent_bone) == 0xFFFF) {
|
||||
src_mat = this + SO.bb_row0 - 0; // identity-like base matrix at +0xFC? No...
|
||||
// When parent is 0xFFFF, use the identity-like base at model_attachment_list1
|
||||
// Actually from decompilation: param_2 = &this->model_attachment_list1
|
||||
// which is at +0xFC. But that's the camera matrix, not identity.
|
||||
// Let me use the local identity matrix instead.
|
||||
src_mat = local_mat_addr;
|
||||
src_mat = this + 0xFC;
|
||||
} else {
|
||||
const parent_out = bone_out_base + @as(u32, @intCast(parent_idx_raw)) * 0x40;
|
||||
src_mat = parent_out;
|
||||
@@ -1140,13 +1251,21 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
const rot_anim = bdef + BD.rot_anim;
|
||||
const rot_kf_count = ru32(bdef + BD.rot_nts);
|
||||
|
||||
// Rotation
|
||||
if (rot_kf_count != 0.0 and ru32(this + SO.anim_frame_ctr) < rot_kf_count) {
|
||||
interpAnimKF(this, brt, rot_anim, brt + BR.rot_idx0);
|
||||
// initPixelShaderDispatcher5 call (0x74b6b5) — particle shader init, not perf critical
|
||||
// Step 1: Rotation — build rotation matrix from quaternion FIRST.
|
||||
// The original at 0x74B6BB overwrites the bone-local matrix with the
|
||||
// quaternion rotation matrix (it does NOT multiply — just writes directly).
|
||||
// This runs BEFORE scale and translation so the translation offset
|
||||
// (pivot - matrix * pivot) uses the correctly rotated matrix.
|
||||
if (rot_kf_count != 0) {
|
||||
if (ru32(this + SO.anim_frame_ctr) < rot_kf_count) {
|
||||
interpAnimKF(this, brt, rot_anim, brt + BR.rot_idx0);
|
||||
}
|
||||
// Build rotation matrix from quaternion — overwrites local_mat2
|
||||
// exactly like the original at 0x74B6BB (no multiply, just write)
|
||||
buildRotationMatrix(lm2_addr, ufloat(ru32(brt + BR.rot_x)), ufloat(ru32(brt + BR.rot_y)), ufloat(ru32(brt + BR.rot_z)), ufloat(ru32(brt + BR.rot_w)));
|
||||
}
|
||||
|
||||
// Scale interpolation
|
||||
// Step 2: Scale interpolation — applied after rotation
|
||||
const scale_anim = bdef + BD.scale_anim;
|
||||
const scale_kf_count = ru32(bdef + BD.scale_nts);
|
||||
if (scale_kf_count != 0) {
|
||||
@@ -1156,12 +1275,23 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
scaleMatrix3x3(lm2_addr, ufloat(ru32(brt + BR.scale_x)), ufloat(ru32(brt + BR.scale_y)), ufloat(ru32(brt + BR.scale_z)));
|
||||
}
|
||||
|
||||
// Handle particle flag
|
||||
// Conditional multiply: if flag bit 0x80 set AND bone_rt[0xF0] != 0,
|
||||
// multiply bone_local by the matrix pointed to by bone_rt[0xF0].
|
||||
// Assembly at 0x714F7F-0x714F9C:
|
||||
// TEST CL, CL / JNS skip
|
||||
// MOV EAX, [ESI+0xF0] / TEST EAX, EAX / JZ skip
|
||||
// PUSH EAX (right), PUSH &bone_local (left), PUSH &bone_local (output)
|
||||
// CALL 0x74A7C0 (multiplyMatrix4x4: output = left * right)
|
||||
// This is bone_local *= *(bone_rt+0xF0)
|
||||
if ((@as(i8, @bitCast(@as(u8, @truncate(combined_flags)))) < 0) and ru32(brt + BR.bone_flag_cache) != 0) {
|
||||
// initParticlePixelShaderGeneration() — particle setup, not math
|
||||
const extra_mat = ru32(brt + BR.bone_flag_cache); // pointer to additional matrix
|
||||
// In-place multiply: bone_local = bone_local * extra_mat
|
||||
// Use multiplyMatrix4x4_Basic directly (0x7507BB, __stdcall RET 0xC)
|
||||
const matMulBasic: *const fn (u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x7507BB);
|
||||
matMulBasic(lm2_addr, lm2_addr, extra_mat);
|
||||
}
|
||||
|
||||
// Translation interpolation
|
||||
// Step 3: Translation interpolation
|
||||
var tx_val = rf32(bdef + BD.pivot_x);
|
||||
var ty_val = rf32(bdef + BD.pivot_y);
|
||||
var tz_val = rf32(bdef + BD.pivot_z);
|
||||
@@ -1177,7 +1307,8 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
tz_val += ufloat(ru32(brt + BR.trans_z));
|
||||
}
|
||||
|
||||
// Compute translation: trans + pivot - rot * pivot
|
||||
// Step 4: Compute translation offset using the ROTATED+SCALED matrix.
|
||||
// translation = (pivot + interp_trans) - bone_local_matrix * pivot
|
||||
const piv_x = rf32(bdef + BD.pivot_x);
|
||||
const piv_y = rf32(bdef + BD.pivot_y);
|
||||
const piv_z = rf32(bdef + BD.pivot_z);
|
||||
@@ -1185,11 +1316,6 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
local_mat2[13] = ty_val - (local_mat2[1] * piv_x + local_mat2[5] * piv_y + local_mat2[9] * piv_z);
|
||||
local_mat2[14] = tz_val - (local_mat2[2] * piv_x + local_mat2[6] * piv_y + local_mat2[10] * piv_z);
|
||||
|
||||
// Apply rotation from quaternion if rotation was interpolated
|
||||
if (rot_kf_count != 0 and ru32(this + SO.anim_frame_ctr) < rot_kf_count) {
|
||||
rotateByQuaternion(lm2_addr, ufloat(ru32(brt + BR.rot_x)), ufloat(ru32(brt + BR.rot_y)), ufloat(ru32(brt + BR.rot_z)), ufloat(ru32(brt + BR.rot_w)));
|
||||
}
|
||||
|
||||
// Write final composed matrix to output
|
||||
const dst = bone_out_base + bone_idx * 0x40;
|
||||
// Multiply: dst = local_mat2 * src_mat (parent)
|
||||
@@ -1209,8 +1335,169 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
}
|
||||
|
||||
// --- Billboard post-processing (flags & 0x78) ---
|
||||
// TODO: Implement billboard post-processing for types 8/16/32/64
|
||||
// This is a less common path — most bones don't have post-billboard flags
|
||||
// Assembly at 0x7151F9-0x71594E. Runs for BOTH animated and non-animated paths.
|
||||
// Modifies the already-written bone output matrix in-place.
|
||||
if ((combined_flags & 0x78) != 0) {
|
||||
// pMVar19 = bone_idx * 0x40 (byte offset for output)
|
||||
// pfVar12 = bone_out_base + pMVar19 (output matrix ptr)
|
||||
const out_off = bone_idx * 0x40;
|
||||
const om = bone_out_base + out_off; // output matrix
|
||||
|
||||
// Compute scale lengths (sqrt of row length_sq for each row)
|
||||
const scale_len0 = @sqrt(rf32(om + 0x08) * rf32(om + 0x08) + rf32(om + 0x04) * rf32(om + 0x04) + rf32(om) * rf32(om));
|
||||
const scale_len1 = @sqrt(rf32(om + 0x18) * rf32(om + 0x18) + rf32(om + 0x14) * rf32(om + 0x14) + rf32(om + 0x10) * rf32(om + 0x10));
|
||||
const scale_len2 = @sqrt(rf32(om + 0x28) * rf32(om + 0x28) + rf32(om + 0x24) * rf32(om + 0x24) + rf32(om + 0x20) * rf32(om + 0x20));
|
||||
|
||||
// Compute translated pivot position through the output matrix
|
||||
// local_a8 = pivot * matrix + translation
|
||||
const bpx = rf32(bdef + BD.pivot_x);
|
||||
const bpy = rf32(bdef + BD.pivot_y);
|
||||
const bpz = rf32(bdef + BD.pivot_z);
|
||||
const pos_x = bpx * rf32(om) + bpy * rf32(om + 0x10) + bpz * rf32(om + 0x20) + rf32(om + 0x30);
|
||||
const pos_y = bpx * rf32(om + 0x04) + bpy * rf32(om + 0x14) + bpz * rf32(om + 0x24) + rf32(om + 0x34);
|
||||
const pos_z = bpx * rf32(om + 0x08) + bpy * rf32(om + 0x18) + bpz * rf32(om + 0x28) + rf32(om + 0x38);
|
||||
|
||||
// Switch on billboard post-processing type
|
||||
const bb_post = combined_flags & 0x78;
|
||||
switch (bb_post) {
|
||||
0x08 => {
|
||||
// Type 8: decompilation lines 657-718
|
||||
// If no pre-billboard (local_1c == 0 i.e. flags & 0x280 was 0):
|
||||
// set fixed rotation columns
|
||||
// Else: use rotation matrix rows with negated first component, normalize
|
||||
const had_anim = (combined_flags & 0x280) != 0;
|
||||
if (!had_anim) {
|
||||
// Fixed columns: row0={0,0,-1}, row1={1,0,0}, row2={0,1,0}
|
||||
wf32(om, 0);
|
||||
wf32(om + 0x04, 0);
|
||||
wf32(om + 0x08, -1);
|
||||
wf32(om + 0x10, 1);
|
||||
wf32(om + 0x14, 0);
|
||||
wf32(om + 0x18, 0);
|
||||
wf32(om + 0x20, 0);
|
||||
wf32(om + 0x24, 1);
|
||||
wf32(om + 0x28, 0);
|
||||
} else {
|
||||
// Row 0 = {local_e4, local_e0, -local_e8}, normalize
|
||||
const r0x = local_mat2[1]; // local_e4
|
||||
const r0y = local_mat2[2]; // local_e0
|
||||
const r0z = -local_mat2[0]; // -local_e8
|
||||
wf32(om, r0x);
|
||||
wf32(om + 0x04, r0y);
|
||||
wf32(om + 0x08, r0z);
|
||||
const n0 = normalizeVec3InPlace(om);
|
||||
_ = n0;
|
||||
// Row 1 = {local_d4, local_d0, -local_d8}, normalize
|
||||
const r1x = local_mat2[5]; // local_d4
|
||||
const r1y = local_mat2[6]; // local_d0
|
||||
const r1z = -local_mat2[4]; // -local_d8
|
||||
wf32(om + 0x10, r1x);
|
||||
wf32(om + 0x14, r1y);
|
||||
wf32(om + 0x18, r1z);
|
||||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||||
_ = n1;
|
||||
// Row 2 = {local_c4, local_c0, -local_c8}, normalize
|
||||
const r2x = local_mat2[9]; // local_c4
|
||||
const r2y = local_mat2[10]; // local_c0
|
||||
const r2z = -local_mat2[8]; // -local_c8
|
||||
wf32(om + 0x20, r2x);
|
||||
wf32(om + 0x24, r2y);
|
||||
wf32(om + 0x28, r2z);
|
||||
const n2 = normalizeVec3InPlace(om + 0x20);
|
||||
_ = n2;
|
||||
}
|
||||
},
|
||||
0x10 => {
|
||||
// Type 16: normalize row0, set row1={row0.y, -row0.x, 0}, normalize,
|
||||
// row2 = cross(row0, row1)
|
||||
const n0 = normalizeVec3InPlace(om);
|
||||
_ = n0;
|
||||
const r0x = rf32(om);
|
||||
const r0y = rf32(om + 0x04);
|
||||
wf32(om + 0x10, r0y);
|
||||
wf32(om + 0x14, -r0x);
|
||||
wf32(om + 0x18, 0);
|
||||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||||
_ = n1;
|
||||
// row2 = cross(row0, row1)
|
||||
wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14));
|
||||
wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18));
|
||||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||||
},
|
||||
0x20 => {
|
||||
// Type 32: normalize row1, set row0={-row1.y, row1.x, 0}, normalize,
|
||||
// row2 = cross(row0, row1)
|
||||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||||
_ = n1;
|
||||
wf32(om, -rf32(om + 0x14));
|
||||
wf32(om + 0x04, rf32(om + 0x10));
|
||||
wf32(om + 0x08, 0);
|
||||
const n0 = normalizeVec3InPlace(om);
|
||||
_ = n0;
|
||||
// row2 = cross(row0, row1)
|
||||
wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14));
|
||||
wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18));
|
||||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||||
},
|
||||
0x40 => {
|
||||
// Type 64: normalize row2, set row1={row2.y, -row2.x, 0}, normalize,
|
||||
// row0 = cross(row1, row2)
|
||||
const r2_len = @sqrt(rf32(om + 0x20) * rf32(om + 0x20) + rf32(om + 0x24) * rf32(om + 0x24) + rf32(om + 0x28) * rf32(om + 0x28));
|
||||
if (@abs(r2_len) >= BILLBOARD_EPSILON) {
|
||||
const inv = 1.0 / r2_len;
|
||||
wf32(om + 0x20, rf32(om + 0x20) * inv);
|
||||
wf32(om + 0x24, rf32(om + 0x24) * inv);
|
||||
wf32(om + 0x28, rf32(om + 0x28) * inv);
|
||||
}
|
||||
wf32(om + 0x10, rf32(om + 0x24));
|
||||
wf32(om + 0x14, -rf32(om + 0x20));
|
||||
wf32(om + 0x18, 0);
|
||||
const r1_len = @sqrt(rf32(om + 0x10) * rf32(om + 0x10) + rf32(om + 0x14) * rf32(om + 0x14) + rf32(om + 0x18) * rf32(om + 0x18));
|
||||
if (@abs(r1_len) >= BILLBOARD_EPSILON) {
|
||||
const inv = 1.0 / r1_len;
|
||||
wf32(om + 0x10, rf32(om + 0x10) * inv);
|
||||
wf32(om + 0x14, rf32(om + 0x14) * inv);
|
||||
wf32(om + 0x18, rf32(om + 0x18) * inv);
|
||||
}
|
||||
// row0 = cross(row2.y*row1.z - row2.z*row1.y, ...)
|
||||
wf32(om, rf32(om + 0x24) * rf32(om + 0x18) - rf32(om + 0x28) * rf32(om + 0x14));
|
||||
wf32(om + 0x04, rf32(om + 0x28) * rf32(om + 0x10) - rf32(om + 0x20) * rf32(om + 0x18));
|
||||
wf32(om + 0x08, rf32(om + 0x20) * rf32(om + 0x14) - rf32(om + 0x24) * rf32(om + 0x10));
|
||||
},
|
||||
else => {},
|
||||
}
|
||||
|
||||
// Apply scale lengths back and recompute translation
|
||||
// Assembly at 0x715868-0x71594B
|
||||
wf32(om + 0x0C, 0);
|
||||
wf32(om + 0x1C, 0);
|
||||
wf32(om + 0x2C, 0);
|
||||
// Scale each row by its original length
|
||||
const r0x_s = rf32(om);
|
||||
wf32(om, scale_len0 * r0x_s);
|
||||
const r0y_s = rf32(om + 0x04);
|
||||
wf32(om + 0x04, scale_len0 * r0y_s);
|
||||
const r0z_s = rf32(om + 0x08);
|
||||
wf32(om + 0x08, scale_len0 * r0z_s);
|
||||
const r1x_s = rf32(om + 0x10);
|
||||
wf32(om + 0x10, scale_len1 * r1x_s);
|
||||
const r1y_s = rf32(om + 0x14);
|
||||
wf32(om + 0x14, scale_len1 * r1y_s);
|
||||
const r1z_s = rf32(om + 0x18);
|
||||
wf32(om + 0x18, scale_len1 * r1z_s);
|
||||
const r2x_s = rf32(om + 0x20);
|
||||
wf32(om + 0x20, scale_len2 * r2x_s);
|
||||
const r2y_s = rf32(om + 0x24);
|
||||
wf32(om + 0x24, scale_len2 * r2y_s);
|
||||
const r2z_s = rf32(om + 0x28);
|
||||
wf32(om + 0x28, scale_len2 * r2z_s);
|
||||
|
||||
// Recompute translation: pos - scaled_matrix * pivot
|
||||
wf32(om + 0x30, pos_x - (scale_len0 * r0x_s * bpx + scale_len1 * r1x_s * bpy + scale_len2 * r2x_s * bpz));
|
||||
wf32(om + 0x34, pos_y - (scale_len0 * r0y_s * bpx + scale_len1 * r1y_s * bpy + scale_len2 * r2y_s * bpz));
|
||||
wf32(om + 0x38, pos_z - (scale_len0 * r0z_s * bpx + scale_len1 * r1z_s * bpy + scale_len2 * r2z_s * bpz));
|
||||
wf32(om + 0x3C, 1.0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1226,18 +1513,18 @@ export fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
// =========================================================================
|
||||
|
||||
// Section 8: Texture animation loop
|
||||
//texAnimLoop(this, model_hdr); // DEBUG: disabled
|
||||
texAnimLoop(this, model_hdr);
|
||||
|
||||
// Section 9: Color animation loop
|
||||
//colorAnimLoop(this, model_hdr); // DEBUG: disabled
|
||||
colorAnimLoop(this, model_hdr);
|
||||
|
||||
// Section 10: Bone keyframe processing
|
||||
//boneKeyframeLoop(this, model_hdr); // DEBUG: disabled
|
||||
boneKeyframeLoop(this, model_hdr);
|
||||
|
||||
// Section 11: Particle emitter loops
|
||||
//particleLoops(this, model_hdr); // DEBUG: disabled
|
||||
particleLoops(this, model_hdr);
|
||||
|
||||
// Section 12: Attachment recursion — RE-ENABLED to test
|
||||
// Section 12: Attachment recursion
|
||||
attachmentRecursion(this, model_hdr, bone_out_base);
|
||||
|
||||
// =========================================================================
|
||||
@@ -1298,8 +1585,9 @@ fn texAnimLoop(this: u32, model_hdr: u32) void {
|
||||
}
|
||||
|
||||
fn colorAnimLoop(this: u32, model_hdr: u32) void {
|
||||
const count = ru32(model_hdr + 0x64);
|
||||
if (count == 0) return;
|
||||
// Assembly: entry gate at model_hdr+0x64, loop bound at model_hdr+0x6C
|
||||
if (ru32(model_hdr + 0x64) == 0) return;
|
||||
const count = ru32(model_hdr + 0x6C); // loop bound from assembly 0x715F0A
|
||||
const data_base = ru32(model_hdr + 0x68);
|
||||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||||
const out_base = ru32(this + SO.color_anim_out);
|
||||
@@ -1346,9 +1634,9 @@ fn boneKeyframeLoop(this: u32, model_hdr: u32) void {
|
||||
var mat_off: u32 = 0;
|
||||
while (i < count) : ({
|
||||
i += 1;
|
||||
data_off += 0x24;
|
||||
out_off += 0x26 * 4;
|
||||
mat_off += 0x40;
|
||||
data_off += 0x54; // assembly at 0x7163A2: ADD EDI, 0x54
|
||||
out_off += 0x98; // assembly at 0x7163A5: ADD ESI, 0x98
|
||||
mat_off += 0x40; // assembly at 0x715395: ADD EDX, 0x40
|
||||
}) {
|
||||
const kf_data = data_base + data_off;
|
||||
const output = @as(u32, @intCast(@as(i32, @bitCast(scale2_base)) + @as(i32, @bitCast(out_off))));
|
||||
@@ -1406,8 +1694,8 @@ fn ribbonEmitterLoop(this: u32, model_hdr: u32) void {
|
||||
|
||||
var i: u32 = 0;
|
||||
while (i < count) : (i += 1) {
|
||||
const entry = data_base + i * 0xD4;
|
||||
const output = out_base + i * 0x15C;
|
||||
const entry = data_base + i * 0xD4; // assembly 0x716ABC: ADD EDI, 0xD4
|
||||
const output = out_base + i * 0x170; // assembly 0x716AC2: ADD ESI, 0x170
|
||||
const bone_idx = @as(u32, ru16(entry + 2));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
|
||||
@@ -1456,36 +1744,222 @@ fn particleEmitterLoop(this: u32, model_hdr: u32) void {
|
||||
}
|
||||
|
||||
fn additionalParticleLoops(this: u32, model_hdr: u32) void {
|
||||
// Section for model_hdr + 0x134 and + 0x13C
|
||||
// These handle additional particle system tracks and the large per-emitter
|
||||
// state blocks (8+ sub-tracks each). They all follow the same findInterpIdx +
|
||||
// lerp + crossfade pattern.
|
||||
// Assembly: model_hdr+0x134 section (asm 0x71763E-0x717D6A)
|
||||
// Then additional_remaining reset at 0x717D6F
|
||||
// Then model_hdr+0x13C section (asm 0x717D75-0x7185E3)
|
||||
|
||||
// Additional remaining data reset
|
||||
wu32(this + SO.add_remaining, 0);
|
||||
// Section 12c: model_hdr+0x134 particle visibility/tracks
|
||||
// count=+0x134, data=+0x138, output=this+0x3C8
|
||||
// Data stride 0xDC, output stride 0xD0
|
||||
// Each entry: bone_idx at +0x04, visibility at +0xCC
|
||||
// Sub-tracks: visibility(+0xC0), position(+0x24), alpha(+0x40),
|
||||
// speed(+0x5C), emission(+0x78), scale(+0xA4)
|
||||
if (ru32(model_hdr + 0x134) != 0) {
|
||||
const count0 = ru32(model_hdr + 0x134);
|
||||
const data_base0 = ru32(model_hdr + 0x138);
|
||||
const out_base0 = ru32(this + 0x3C8); // SO.particle2
|
||||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||||
|
||||
var i: u32 = 0;
|
||||
var data_off: u32 = 0;
|
||||
var out_off: u32 = 0;
|
||||
while (i < count0) : ({
|
||||
i += 1;
|
||||
data_off += 0xDC; // asm 0x717D4D
|
||||
out_off += 0xD0; // asm 0x717D53
|
||||
}) {
|
||||
const entry = data_base0 + data_off;
|
||||
const output = out_base0 + out_off;
|
||||
|
||||
// Visibility check: entry+0xCC vs anim_frame_ctr
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) {
|
||||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
// Visibility byte animation at entry+0xC0
|
||||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xC0, output + 0xB0);
|
||||
const vis_mode = ri16(entry + 0xC0);
|
||||
if (vis_mode == 0) {
|
||||
wu8(output + 0xBC, ru8(ru32(entry + 0xC0 + 0x18) + ru32(output + 0xB0)));
|
||||
} else {
|
||||
wu8(output + 0xBC, ru8(ru32(output + 0xB0) + ru32(entry + 0xD8)));
|
||||
// Crossfade blend for visibility if needed
|
||||
if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xC2) == -1) {
|
||||
findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xC0, output + 0xC0);
|
||||
wu8(output + 0xCC, ru8(ru32(output + 0xC0) + ru32(entry + 0xD8)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Position track: entry+0x24 vs entry+0x30
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x30)) {
|
||||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
interpVec3Track(this, bone_rt, entry + 0x24, output, ufloat(ru32(bone_rt + BR.blend_weight)));
|
||||
}
|
||||
|
||||
// Alpha track: entry+0x40 vs entry+0x4C
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x4C)) {
|
||||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
// Short-value interpolation pattern
|
||||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x40, output + 0x30);
|
||||
const alpha_mode = ri16(entry + 0x40);
|
||||
const alpha_base = ru32(entry + 0x40 + 0x18);
|
||||
if (alpha_mode == 0) {
|
||||
const sv = @as(f32, @floatFromInt(@as(i32, @intCast(ri16(alpha_base + ru32(output + 0x30) * 2)))));
|
||||
wf32(output + 0x3C, sv * SHORT_TO_FLOAT);
|
||||
} else {
|
||||
const t = ufloat(ru32(output + 0x38));
|
||||
const short_ranges = ru32(entry + 0x40 + 0x0C);
|
||||
const v0 = @as(f32, @floatFromInt(@as(i32, @intCast(ri16(short_ranges + ru32(output + 0x30) * 2)))));
|
||||
const v1 = @as(f32, @floatFromInt(@as(i32, @intCast(ri16(short_ranges + ru32(output + 0x34) * 2)))));
|
||||
wf32(output + 0x3C, (v1 * SHORT_TO_FLOAT - v0 * SHORT_TO_FLOAT) * t + v0 * SHORT_TO_FLOAT);
|
||||
}
|
||||
}
|
||||
|
||||
// Speed track: entry+0x5C vs entry+0x68
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x68)) {
|
||||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
interpFloatTrack(this, bone_rt, entry + 0x5C, output + 0x50);
|
||||
}
|
||||
|
||||
// Emission rate: entry+0x78 vs entry+0x84
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x84)) {
|
||||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
interpFloatTrack(this, bone_rt, entry + 0x78, output + 0x70);
|
||||
}
|
||||
|
||||
// Scale track: entry+0xA4 vs entry+0xB0
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) {
|
||||
const bone_idx = @as(u32, ru16(entry + 0x04));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
// This uses getInterpolatedFloat pattern
|
||||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0xA4, output + 0x90);
|
||||
const scale_mode = ri16(entry + 0xA4);
|
||||
const scale_base = ru32(entry + 0xA4 + 0x18);
|
||||
if (scale_mode == 0) {
|
||||
wu16(output + 0x9C, ru16(scale_base + ru32(output + 0x90) * 2));
|
||||
} else {
|
||||
wu16(output + 0x9C, ru16(scale_base + ru32(output + 0x90) * 2));
|
||||
// Crossfade
|
||||
if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0xA6) == -1) {
|
||||
findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0xA4, output + 0xA0);
|
||||
wu16(output + 0xAC, ru16(scale_base + ru32(output + 0xA0) * 2));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Additional remaining data reset — between 0x134 and 0x13C sections
|
||||
// Assembly at 0x717D6F: MOV [EBX+0x3D8], 0
|
||||
wu32(this + 0x3D8, 0);
|
||||
|
||||
// Section 12e: model_hdr+0x13C (largest particle section)
|
||||
// count=+0x13C, data=+0x140
|
||||
// output1=this+0x3D0, output2=this+0x3D4
|
||||
// Data stride 0x1F8, output stride 0x16C
|
||||
const count1 = ru32(model_hdr + 0x13C);
|
||||
if (count1 != 0) {
|
||||
const data_base = ru32(model_hdr + 0x140);
|
||||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||||
const particle_base = ru32(this + SO.particle3);
|
||||
const particle_base = ru32(this + 0x3D0); // SO.particle3
|
||||
|
||||
var i: u32 = 0;
|
||||
while (i < count1) : (i += 1) {
|
||||
const entry_off = i * 0x1FC;
|
||||
const out_off = i * 0x17C;
|
||||
const entry = data_base + entry_off;
|
||||
var data_off: u32 = 0;
|
||||
var out_off: u32 = 0;
|
||||
while (i < count1) : ({
|
||||
i += 1;
|
||||
data_off += 0x1F8; // asm 0x7185CD
|
||||
out_off += 0x16C; // asm 0x7185BA
|
||||
}) {
|
||||
const entry = data_base + data_off;
|
||||
const output = particle_base + out_off;
|
||||
const bone_idx = @as(u32, ru16(entry + 0x14));
|
||||
const bone_rt = bone_rt_base + bone_idx * 0x118;
|
||||
|
||||
// Emission rate
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x40)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0x34, output);
|
||||
// All tracks from assembly 0x717D90-0x7185E3:
|
||||
const particle_ptrs = ru32(this + 0x3D4); // [EBX+0x3D4]
|
||||
const local_14 = ru32(particle_ptrs + i * 4); // per-emitter data ptr
|
||||
|
||||
// Visibility: gate=entry+0x1E8, AnimData=entry+0x1DC, output=output+0x140
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x1E8)) {
|
||||
findInterpIdx(this, ru32(bone_rt + 0x98), ru32(bone_rt + 0x9C), entry + 0x1DC, output + 0x140);
|
||||
if (ri16(entry + 0x1DC) == 0) {
|
||||
wu8(output + 0x14C, ru8(ru32(entry + 0x1F4) + ru32(output + 0x140)));
|
||||
} else {
|
||||
wu8(output + 0x14C, ru8(ru32(output + 0x140) + ru32(entry + 0x1F4)));
|
||||
if (rf32(bone_rt + 0x10C) != 0.0 and ri16(entry + 0x1DE) == -1) {
|
||||
findInterpIdx(this, ru32(bone_rt + 0xC4), ru32(bone_rt + 0xC8), entry + 0x1DC, output + 0x150);
|
||||
wu8(output + 0x15C, ru8(ru32(entry + 0x1F4) + ru32(output + 0x150)));
|
||||
}
|
||||
}
|
||||
}
|
||||
// Speed
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x5C)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20);
|
||||
|
||||
// Emitter active flag: visibility && emitter_enable_flag
|
||||
const vis_byte = ru8(output + 0x14C);
|
||||
const emitter_active: u32 = if (vis_byte != 0 and ru32(this + 0x50) != 0) 1 else 0;
|
||||
wu32(output + 0x160, emitter_active);
|
||||
// IsParticleBufferEmpty check
|
||||
var buf_active: u32 = 0;
|
||||
if (emitter_active != 0) {
|
||||
buf_active = 1;
|
||||
} else {
|
||||
// Call IsParticleBufferEmpty (0x7B5F60)
|
||||
const isEmptyFn: *const fn (u32) callconv(.{ .x86_stdcall = .{} }) u32 = @ptrFromInt(0x7B5F60);
|
||||
if (isEmptyFn(@intFromPtr(&local_14)) != 0) {
|
||||
buf_active = 1;
|
||||
}
|
||||
}
|
||||
wu32(output + 0x164, buf_active);
|
||||
// OR into additional_remaining
|
||||
wu32(this + 0x3D8, ru32(this + 0x3D8) | buf_active);
|
||||
|
||||
// Only process tracks if visible or first frame
|
||||
if (vis_byte != 0 or ru32(this + SO.anim_frame_ctr) == 0) {
|
||||
// Track 1: emission rate — gate=+0x40, AnimData=+0x34, output=+0x00
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x40)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0x34, output);
|
||||
}
|
||||
// Track 2: speed — gate=+0x5C, AnimData=+0x50, output=+0x20
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x5C)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0x50, output + 0x20);
|
||||
}
|
||||
// Track 3: color — gate=+0x78, AnimData=+0x6C, output=+0x40
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x78)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0x6C, output + 0x40);
|
||||
}
|
||||
// Track 4 — gate=+0x94, AnimData=+0x88, output=+0x60
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x94)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0x88, output + 0x60);
|
||||
}
|
||||
// Track 5 (Vec3 spline) — gate=+0xB0, AnimData=+0xA4, output=+0x80
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xB0)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0xA4, output + 0x80);
|
||||
}
|
||||
// Track 6 — gate=+0xCC, AnimData=+0xC0, output=+0xA0
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xCC)) {
|
||||
interpFloatTrack(this, bone_rt, entry + 0xC0, output + 0xA0);
|
||||
}
|
||||
// Track 7 — gate=+0xE8, AnimData=+0xDC, output=+0xC0
|
||||
// Uses getInterpolatedFloat (0x71AF20)
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0xE8)) {
|
||||
getInterpolatedFloat(this, bone_rt, entry + 0xDC, output + 0xC0);
|
||||
}
|
||||
// Track 8 — gate=+0x104, AnimData=+0xF8, output=+0xE0
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x104)) {
|
||||
getInterpolatedFloat(this, bone_rt, entry + 0xF8, output + 0xE0);
|
||||
}
|
||||
// Track 9 — gate=+0x120, AnimData=+0x114, output=+0x100
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x120)) {
|
||||
getInterpolatedFloat(this, bone_rt, entry + 0x114, output + 0x100);
|
||||
}
|
||||
// Track 10 — gate=+0x13C, AnimData=+0x130, output=+0x120
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(entry + 0x13C)) {
|
||||
getInterpolatedFloat(this, bone_rt, entry + 0x130, output + 0x120);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1518,11 +1992,11 @@ fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32) void {
|
||||
// Iterate child scene objects linked list
|
||||
var child = ru32(this + SO.hierarchy_idx);
|
||||
while (child != 0) {
|
||||
const attach_idx_raw = rf32(child + SO.field_184);
|
||||
const attach_idx = @as(u32, @intFromFloat(attach_idx_raw));
|
||||
// child->attach_idx at +0x1D4 (assembly-verified: MOV EAX,[ECX+0x1D4] at 0x718668)
|
||||
const attach_idx = ru32(child + 0x1D4);
|
||||
|
||||
// Check if attachment is valid
|
||||
if (@as(u32, @bitCast(attach_idx_raw)) != 0x08000000) {
|
||||
// Check if attachment is valid (0xFFFF = no attachment)
|
||||
if (attach_idx != 0xFFFF) {
|
||||
const visible = ru8(hierarchy + attach_idx * 0x20 + 0x0C);
|
||||
if (visible != 0) {
|
||||
const att_entry = attach_data + attach_idx * 0x30;
|
||||
@@ -1553,6 +2027,7 @@ fn attachmentRecursion(this: u32, model_hdr: u32, bone_out_base: u32) void {
|
||||
}
|
||||
|
||||
// Next sibling in linked list
|
||||
child = ru32(child + 0x190); // field_0x190 = next pointer
|
||||
// Assembly-verified: MOV ECX,[ECX+0x1E4] at 0x718764
|
||||
child = ru32(child + 0x1E4);
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,777 @@
|
||||
openjdk version "21.0.10" 2026-01-20
|
||||
OpenJDK Runtime Environment (build 21.0.10+7)
|
||||
OpenJDK 64-Bit Server VM (build 21.0.10+7, mixed mode)
|
||||
=== findInterpolationIndices (0x00713D50, 334 bytes) ===
|
||||
0x00713D50 PUSH EBP
|
||||
0x00713D51 MOV EBP,ESP
|
||||
0x00713D53 SUB ESP,0x8
|
||||
0x00713D56 PUSH ESI
|
||||
0x00713D57 MOV ESI,ECX
|
||||
0x00713D59 MOV ECX,dword ptr [EBP + 0x10]
|
||||
0x00713D5C MOV EAX,dword ptr [ECX + 0x4]
|
||||
0x00713D5F TEST EAX,EAX
|
||||
0x00713D61 PUSH EDI
|
||||
0x00713D62 JZ 0x00713d75
|
||||
0x00713D64 MOV EDX,dword ptr [EBP + 0xc]
|
||||
0x00713D67 MOV EAX,dword ptr [ECX + 0x8]
|
||||
0x00713D6A MOV EDI,dword ptr [EAX + EDX*0x8 + 0x4]
|
||||
0x00713D6E LEA EAX,[EAX + EDX*0x8]
|
||||
0x00713D71 MOV EDX,dword ptr [EAX]
|
||||
0x00713D73 JMP 0x00713d7b
|
||||
0x00713D75 MOV EDI,dword ptr [ECX + 0xc]
|
||||
0x00713D78 XOR EDX,EDX
|
||||
0x00713D7A DEC EDI
|
||||
0x00713D7B CMP EDX,EDI
|
||||
0x00713D7D JC 0x00713d96
|
||||
0x00713D7F MOV EAX,dword ptr [EBP + 0x14]
|
||||
0x00713D82 POP EDI
|
||||
0x00713D83 MOV dword ptr [EAX],EDX
|
||||
0x00713D85 MOV dword ptr [EAX + 0x4],EDX
|
||||
0x00713D88 MOV dword ptr [EAX + 0x8],0x0
|
||||
0x00713D8F POP ESI
|
||||
0x00713D90 MOV ESP,EBP
|
||||
0x00713D92 POP EBP
|
||||
0x00713D93 RET 0x10 << RET
|
||||
0x00713D96 MOV AX,word ptr [ECX + 0x2]
|
||||
0x00713D9A CMP AX,0xffff
|
||||
0x00713D9E JZ 0x00713dab
|
||||
0x00713DA0 MOV ESI,dword ptr [ESI + 0x64]
|
||||
0x00713DA3 MOVZX EAX,AX
|
||||
0x00713DA6 MOV ESI,dword ptr [ESI + EAX*0x4]
|
||||
0x00713DA9 JMP 0x00713dae
|
||||
0x00713DAB MOV ESI,dword ptr [EBP + 0x8]
|
||||
0x00713DAE MOV EAX,dword ptr [EBP + 0x14]
|
||||
0x00713DB1 MOV ECX,dword ptr [ECX + 0x10]
|
||||
0x00713DB4 PUSH EBX
|
||||
0x00713DB5 MOV EBX,dword ptr [EAX]
|
||||
0x00713DB7 MOV EAX,dword ptr [ECX + EBX*0x4]
|
||||
0x00713DBA MOV dword ptr [EBP + 0xc],EAX
|
||||
0x00713DBD MOV EAX,ESI
|
||||
0x00713DBF SUB EAX,dword ptr [EBP + 0xc]
|
||||
0x00713DC2 CMP EAX,0x1f4
|
||||
0x00713DC7 JNC 0x00713de1
|
||||
0x00713DC9 MOV EAX,EBX
|
||||
0x00713DCB CMP EAX,EDI
|
||||
0x00713DCD JNC 0x00713e45
|
||||
0x00713DCF LEA ECX,[ECX + EAX*0x4 + 0x4]
|
||||
0x00713DD3 CMP dword ptr [ECX],ESI
|
||||
0x00713DD5 JA 0x00713e45
|
||||
0x00713DD7 INC EAX
|
||||
0x00713DD8 ADD ECX,0x4
|
||||
0x00713DDB CMP EAX,EDI
|
||||
0x00713DDD JC 0x00713dd3
|
||||
0x00713DDF JMP 0x00713e45
|
||||
0x00713DE1 CMP EAX,0xfffffe0c
|
||||
0x00713DE6 JC 0x00713dff
|
||||
0x00713DE8 MOV EAX,EBX
|
||||
0x00713DEA CMP EAX,EDX
|
||||
0x00713DEC JBE 0x00713e45
|
||||
0x00713DEE LEA ECX,[ECX + EAX*0x4]
|
||||
0x00713DF1 CMP dword ptr [ECX],ESI
|
||||
0x00713DF3 JBE 0x00713e45
|
||||
0x00713DF5 DEC EAX
|
||||
0x00713DF6 SUB ECX,0x4
|
||||
0x00713DF9 CMP EAX,EDX
|
||||
0x00713DFB JA 0x00713df1
|
||||
0x00713DFD JMP 0x00713e45
|
||||
0x00713DFF MOV EBX,dword ptr [ECX + EDX*0x4]
|
||||
0x00713E02 MOV EAX,ESI
|
||||
0x00713E04 SUB EAX,EBX
|
||||
0x00713E06 CMP EAX,0x1f4
|
||||
0x00713E0B JNC 0x00713e21
|
||||
0x00713E0D MOV EAX,EDX
|
||||
0x00713E0F LEA ECX,[ECX + EDX*0x4 + 0x4]
|
||||
0x00713E13 CMP dword ptr [ECX],ESI
|
||||
0x00713E15 JA 0x00713e45
|
||||
0x00713E17 INC EAX
|
||||
0x00713E18 ADD ECX,0x4
|
||||
0x00713E1B CMP EAX,EDI
|
||||
0x00713E1D JC 0x00713e13
|
||||
0x00713E1F JMP 0x00713e45
|
||||
0x00713E21 LEA EAX,[EDI + EDX*0x1]
|
||||
0x00713E24 SHR EAX,0x1
|
||||
0x00713E26 CMP ESI,dword ptr [ECX + EAX*0x4]
|
||||
0x00713E29 JNC 0x00713e30
|
||||
0x00713E2B LEA EDI,[EAX + -0x1]
|
||||
0x00713E2E JMP 0x00713e3b
|
||||
0x00713E30 MOV EBX,dword ptr [ECX + EAX*0x4 + 0x4]
|
||||
0x00713E34 CMP ESI,EBX
|
||||
0x00713E36 LEA EDX,[EAX + 0x1]
|
||||
0x00713E39 JC 0x00713e41
|
||||
0x00713E3B CMP EDX,EDI
|
||||
0x00713E3D JC 0x00713e21
|
||||
0x00713E3F JMP 0x00713e43
|
||||
0x00713E41 MOV EDX,EAX
|
||||
0x00713E43 MOV EAX,EDX
|
||||
0x00713E45 MOV ECX,dword ptr [EBP + 0x10]
|
||||
0x00713E48 MOV EDI,dword ptr [ECX + 0xc]
|
||||
0x00713E4B LEA EDX,[EAX + 0x1]
|
||||
0x00713E4E CMP EDX,EDI
|
||||
0x00713E50 POP EBX
|
||||
0x00713E51 JNC 0x00713e87
|
||||
0x00713E53 MOV EDI,dword ptr [EBP + 0x14]
|
||||
0x00713E56 MOV dword ptr [EDI],EAX
|
||||
0x00713E58 MOV dword ptr [EDI + 0x4],EDX
|
||||
0x00713E5B MOV ECX,dword ptr [ECX + 0x10]
|
||||
0x00713E5E MOV EAX,dword ptr [ECX + EAX*0x4]
|
||||
0x00713E61 MOV ECX,dword ptr [ECX + EDX*0x4]
|
||||
0x00713E64 SUB ESI,EAX
|
||||
0x00713E66 MOV dword ptr [EBP + -0x8],ESI
|
||||
0x00713E69 XOR ESI,ESI
|
||||
0x00713E6B MOV dword ptr [EBP + -0x4],ESI
|
||||
0x00713E6E FILD qword ptr [EBP + -0x8]
|
||||
0x00713E71 SUB ECX,EAX
|
||||
0x00713E73 MOV dword ptr [EBP + -0x4],ESI
|
||||
0x00713E76 MOV dword ptr [EBP + -0x8],ECX
|
||||
0x00713E79 FIDIV dword ptr [EBP + -0x8]
|
||||
0x00713E7C FSTP float ptr [EDI + 0x8]
|
||||
0x00713E7F POP EDI
|
||||
0x00713E80 POP ESI
|
||||
0x00713E81 MOV ESP,EBP
|
||||
0x00713E83 POP EBP
|
||||
0x00713E84 RET 0x10 << RET
|
||||
0x00713E87 MOV ECX,dword ptr [EBP + 0x14]
|
||||
0x00713E8A POP EDI
|
||||
0x00713E8B MOV dword ptr [ECX + 0x4],EAX
|
||||
0x00713E8E MOV dword ptr [ECX],EAX
|
||||
0x00713E90 MOV dword ptr [ECX + 0x8],0x0
|
||||
0x00713E97 POP ESI
|
||||
0x00713E98 MOV ESP,EBP
|
||||
0x00713E9A POP EBP
|
||||
0x00713E9B RET 0x10 << RET
|
||||
|
||||
=== interpolateAnimationKeyframes (0x00713EA0, 337 bytes) ===
|
||||
0x00713EA0 PUSH EBP
|
||||
0x00713EA1 MOV EBP,ESP
|
||||
0x00713EA3 PUSH ECX
|
||||
0x00713EA4 PUSH EBX
|
||||
0x00713EA5 PUSH ESI
|
||||
0x00713EA6 MOV ESI,dword ptr [EBP + 0xc]
|
||||
0x00713EA9 PUSH EDI
|
||||
0x00713EAA MOV EDI,dword ptr [EBP + 0x8]
|
||||
0x00713EAD PUSH ESI
|
||||
0x00713EAE MOV EBX,EDX
|
||||
0x00713EB0 MOV EAX,dword ptr [EBX + 0x9c]
|
||||
0x00713EB6 MOV EDX,dword ptr [EBX + 0x98]
|
||||
0x00713EBC PUSH EDI
|
||||
0x00713EBD PUSH EAX
|
||||
0x00713EBE PUSH EDX
|
||||
0x00713EBF MOV dword ptr [EBP + -0x4],ECX
|
||||
0x00713EC2 CALL 0x00713d50 << CALL
|
||||
0x00713EC7 MOV EAX,dword ptr [ESI]
|
||||
0x00713EC9 SHL EAX,0x4
|
||||
0x00713ECC CMP word ptr [EDI],0x0
|
||||
0x00713ED0 JNZ 0x00713ef7
|
||||
0x00713ED2 ADD EAX,dword ptr [EDI + 0x18]
|
||||
0x00713ED5 MOV ECX,dword ptr [EAX]
|
||||
0x00713ED7 ADD ESI,0xc
|
||||
0x00713EDA MOV dword ptr [ESI],ECX
|
||||
0x00713EDC MOV EDX,dword ptr [EAX + 0x4]
|
||||
0x00713EDF MOV dword ptr [ESI + 0x4],EDX
|
||||
0x00713EE2 MOV ECX,dword ptr [EAX + 0x8]
|
||||
0x00713EE5 MOV dword ptr [ESI + 0x8],ECX
|
||||
0x00713EE8 MOV EDX,dword ptr [EAX + 0xc]
|
||||
0x00713EEB POP EDI
|
||||
0x00713EEC MOV dword ptr [ESI + 0xc],EDX
|
||||
0x00713EEF POP ESI
|
||||
0x00713EF0 POP EBX
|
||||
0x00713EF1 MOV ESP,EBP
|
||||
0x00713EF3 POP EBP
|
||||
0x00713EF4 RET 0x8 << RET
|
||||
0x00713EF7 MOV EDX,dword ptr [EDI + 0x18]
|
||||
0x00713EFA FLD float ptr [ESI + 0x8]
|
||||
0x00713EFD MOV ECX,dword ptr [ESI + 0x4]
|
||||
0x00713F00 ADD EAX,EDX
|
||||
0x00713F02 SHL ECX,0x4
|
||||
0x00713F05 FLD float ptr [ECX + EDX*0x1]
|
||||
0x00713F08 ADD ECX,EDX
|
||||
0x00713F0A FSUB float ptr [EAX]
|
||||
0x00713F0C LEA EDI,[ESI + 0xc]
|
||||
0x00713F0F FMUL ST1
|
||||
0x00713F11 FADD float ptr [EAX]
|
||||
0x00713F13 FSTP float ptr [EDI]
|
||||
0x00713F15 FLD float ptr [ECX + 0x4]
|
||||
0x00713F18 FSUB float ptr [EAX + 0x4]
|
||||
0x00713F1B FMUL ST1
|
||||
0x00713F1D FADD float ptr [EAX + 0x4]
|
||||
0x00713F20 FSTP float ptr [EDI + 0x4]
|
||||
0x00713F23 FLD float ptr [ECX + 0x8]
|
||||
0x00713F26 FSUB float ptr [EAX + 0x8]
|
||||
0x00713F29 FMUL ST1
|
||||
0x00713F2B FADD float ptr [EAX + 0x8]
|
||||
0x00713F2E FSTP float ptr [EDI + 0x8]
|
||||
0x00713F31 FLD float ptr [ECX + 0xc]
|
||||
0x00713F34 FSUB float ptr [EAX + 0xc]
|
||||
0x00713F37 FMUL ST1
|
||||
0x00713F39 FADD float ptr [EAX + 0xc]
|
||||
0x00713F3C FSTP float ptr [EDI + 0xc]
|
||||
0x00713F3F FSTP ST0
|
||||
0x00713F41 FLD float ptr [0x007ffd74]
|
||||
0x00713F47 FCOMP float ptr [EBX + 0x10c]
|
||||
0x00713F4D FNSTSW AX
|
||||
0x00713F4F TEST AH,0x44
|
||||
0x00713F52 JNP 0x00713fe8
|
||||
0x00713F58 MOV ECX,dword ptr [EBP + 0x8]
|
||||
0x00713F5B CMP word ptr [ECX + 0x2],0xffff
|
||||
0x00713F61 JNZ 0x00713fe8
|
||||
0x00713F67 LEA EAX,[ESI + 0x1c]
|
||||
0x00713F6A PUSH EAX
|
||||
0x00713F6B MOV EAX,dword ptr [EBX + 0xc8]
|
||||
0x00713F71 PUSH ECX
|
||||
0x00713F72 MOV ECX,dword ptr [EBX + 0xc4]
|
||||
0x00713F78 PUSH EAX
|
||||
0x00713F79 PUSH ECX
|
||||
0x00713F7A MOV ECX,dword ptr [EBP + -0x4]
|
||||
0x00713F7D CALL 0x00713d50 << CALL
|
||||
0x00713F82 FLD float ptr [ESI + 0x24]
|
||||
0x00713F85 MOV ECX,dword ptr [ESI + 0x20]
|
||||
0x00713F88 MOV EAX,dword ptr [ESI + 0x1c]
|
||||
0x00713F8B MOV EDX,dword ptr [EBP + 0x8]
|
||||
0x00713F8E MOV EDX,dword ptr [EDX + 0x18]
|
||||
0x00713F91 SHL ECX,0x4
|
||||
0x00713F94 FLD float ptr [ECX + EDX*0x1]
|
||||
0x00713F97 ADD ECX,EDX
|
||||
0x00713F99 SHL EAX,0x4
|
||||
0x00713F9C FSUB float ptr [EAX + EDX*0x1]
|
||||
0x00713F9F ADD EAX,EDX
|
||||
0x00713FA1 LEA EDX,[ESI + 0x28]
|
||||
0x00713FA4 PUSH ECX
|
||||
0x00713FA5 FMUL ST1
|
||||
0x00713FA7 FADD float ptr [EAX]
|
||||
0x00713FA9 FSTP float ptr [EDX]
|
||||
0x00713FAB FLD float ptr [ECX + 0x4]
|
||||
0x00713FAE FSUB float ptr [EAX + 0x4]
|
||||
0x00713FB1 FMUL ST1
|
||||
0x00713FB3 FADD float ptr [EAX + 0x4]
|
||||
0x00713FB6 FSTP float ptr [EDX + 0x4]
|
||||
0x00713FB9 FLD float ptr [ECX + 0x8]
|
||||
0x00713FBC FSUB float ptr [EAX + 0x8]
|
||||
0x00713FBF FMUL ST1
|
||||
0x00713FC1 FADD float ptr [EAX + 0x8]
|
||||
0x00713FC4 FSTP float ptr [EDX + 0x8]
|
||||
0x00713FC7 FLD float ptr [ECX + 0xc]
|
||||
0x00713FCA FSUB float ptr [EAX + 0xc]
|
||||
0x00713FCD FMUL ST1
|
||||
0x00713FCF FADD float ptr [EAX + 0xc]
|
||||
0x00713FD2 FSTP float ptr [EDX + 0xc]
|
||||
0x00713FD5 FSTP ST0
|
||||
0x00713FD7 FLD float ptr [EBX + 0x10c]
|
||||
0x00713FDD FSTP float ptr [ESP]
|
||||
0x00713FE0 PUSH EDX
|
||||
0x00713FE1 PUSH EDI
|
||||
0x00713FE2 PUSH EDI
|
||||
0x00713FE3 CALL 0x0074d114 << CALL
|
||||
0x00713FE8 POP EDI
|
||||
0x00713FE9 POP ESI
|
||||
0x00713FEA POP EBX
|
||||
0x00713FEB MOV ESP,EBP
|
||||
0x00713FED POP EBP
|
||||
0x00713FEE RET 0x8 << RET
|
||||
|
||||
=== getInterpolatedFloat (0x0071AF20, 199 bytes) ===
|
||||
0x0071AF20 PUSH EBP
|
||||
0x0071AF21 MOV EBP,ESP
|
||||
0x0071AF23 PUSH ECX
|
||||
0x0071AF24 PUSH EBX
|
||||
0x0071AF25 MOV EBX,dword ptr [EBP + 0x8]
|
||||
0x0071AF28 PUSH ESI
|
||||
0x0071AF29 MOV ESI,dword ptr [EBP + 0xc]
|
||||
0x0071AF2C PUSH EDI
|
||||
0x0071AF2D PUSH ESI
|
||||
0x0071AF2E MOV EDI,EDX
|
||||
0x0071AF30 MOV EAX,dword ptr [EDI + 0x9c]
|
||||
0x0071AF36 MOV EDX,dword ptr [EDI + 0x98]
|
||||
0x0071AF3C PUSH EBX
|
||||
0x0071AF3D PUSH EAX
|
||||
0x0071AF3E PUSH EDX
|
||||
0x0071AF3F MOV dword ptr [EBP + -0x4],ECX
|
||||
0x0071AF42 CALL 0x00713d50 << CALL
|
||||
0x0071AF47 CMP word ptr [EBX],0x0
|
||||
0x0071AF4B JNZ 0x0071af61
|
||||
0x0071AF4D MOV ECX,dword ptr [EBX + 0x18]
|
||||
0x0071AF50 MOV EAX,dword ptr [ESI]
|
||||
0x0071AF52 MOV EDX,dword ptr [ECX + EAX*0x4]
|
||||
0x0071AF55 POP EDI
|
||||
0x0071AF56 MOV dword ptr [ESI + 0xc],EDX
|
||||
0x0071AF59 POP ESI
|
||||
0x0071AF5A POP EBX
|
||||
0x0071AF5B MOV ESP,EBP
|
||||
0x0071AF5D POP EBP
|
||||
0x0071AF5E RET 0x8 << RET
|
||||
0x0071AF61 MOV EAX,dword ptr [EBX + 0x18]
|
||||
0x0071AF64 MOV ECX,dword ptr [ESI]
|
||||
0x0071AF66 FLD float ptr [EAX + ECX*0x4]
|
||||
0x0071AF69 MOV EDX,dword ptr [ESI + 0x4]
|
||||
0x0071AF6C FLD float ptr [EAX + EDX*0x4]
|
||||
0x0071AF6F FSUB ST0,ST1
|
||||
0x0071AF71 FMUL float ptr [ESI + 0x8]
|
||||
0x0071AF74 FADD ST0,ST1
|
||||
0x0071AF76 FSTP float ptr [ESI + 0xc]
|
||||
0x0071AF79 FSTP ST0
|
||||
0x0071AF7B FLD float ptr [0x007ffd74]
|
||||
0x0071AF81 FCOMP float ptr [EDI + 0x10c]
|
||||
0x0071AF87 FNSTSW AX
|
||||
0x0071AF89 TEST AH,0x44
|
||||
0x0071AF8C JNP 0x0071afde
|
||||
0x0071AF8E CMP word ptr [EBX + 0x2],0xffff
|
||||
0x0071AF94 JNZ 0x0071afde
|
||||
0x0071AF96 MOV ECX,dword ptr [EDI + 0xc4]
|
||||
0x0071AF9C LEA EAX,[ESI + 0x10]
|
||||
0x0071AF9F PUSH EAX
|
||||
0x0071AFA0 MOV EAX,dword ptr [EDI + 0xc8]
|
||||
0x0071AFA6 PUSH EBX
|
||||
0x0071AFA7 PUSH EAX
|
||||
0x0071AFA8 PUSH ECX
|
||||
0x0071AFA9 MOV ECX,dword ptr [EBP + -0x4]
|
||||
0x0071AFAC CALL 0x00713d50 << CALL
|
||||
0x0071AFB1 MOV EBX,dword ptr [EBX + 0x18]
|
||||
0x0071AFB4 MOV EDX,dword ptr [ESI + 0x10]
|
||||
0x0071AFB7 FLD float ptr [EBX + EDX*0x4]
|
||||
0x0071AFBA MOV EAX,dword ptr [ESI + 0x14]
|
||||
0x0071AFBD FLD float ptr [EBX + EAX*0x4]
|
||||
0x0071AFC0 FSUB ST0,ST1
|
||||
0x0071AFC2 FMUL float ptr [ESI + 0x18]
|
||||
0x0071AFC5 FADDP
|
||||
0x0071AFC7 FST float ptr [ESI + 0x1c]
|
||||
0x0071AFCA FLD float ptr [ESI + 0xc]
|
||||
0x0071AFCD FXCH
|
||||
0x0071AFCF FSUB ST0,ST1
|
||||
0x0071AFD1 FMUL float ptr [EDI + 0x10c]
|
||||
0x0071AFD7 FADD ST0,ST1
|
||||
0x0071AFD9 FSTP float ptr [ESI + 0xc]
|
||||
0x0071AFDC FSTP ST0
|
||||
0x0071AFDE POP EDI
|
||||
0x0071AFDF POP ESI
|
||||
0x0071AFE0 POP EBX
|
||||
0x0071AFE1 MOV ESP,EBP
|
||||
0x0071AFE3 POP EBP
|
||||
0x0071AFE4 RET 0x8 << RET
|
||||
|
||||
=== getIndexOffset (0x0071AFF0, 16 bytes) ===
|
||||
0x0071AFF0 PUSH EBP
|
||||
0x0071AFF1 MOV EBP,ESP
|
||||
0x0071AFF3 MOV EAX,dword ptr [ECX + 0x4]
|
||||
0x0071AFF6 MOV ECX,dword ptr [EBP + 0x8]
|
||||
0x0071AFF9 LEA EAX,[EAX + ECX*0x2]
|
||||
0x0071AFFC POP EBP
|
||||
0x0071AFFD RET 0x4 << RET
|
||||
|
||||
=== setShortValue (0x0071B010, 18 bytes) ===
|
||||
0x0071B010 PUSH EBP
|
||||
0x0071B011 MOV EBP,ESP
|
||||
0x0071B013 MOV EAX,ECX
|
||||
0x0071B015 MOV ECX,dword ptr [EBP + 0x8]
|
||||
0x0071B018 MOV DX,word ptr [ECX]
|
||||
0x0071B01B MOV word ptr [EAX],DX
|
||||
0x0071B01E POP EBP
|
||||
0x0071B01F RET 0x4 << RET
|
||||
|
||||
=== scaleMatrix3x3ByVector (0x007BDCA0, 82 bytes) ===
|
||||
0x007BDCA0 PUSH EBP
|
||||
0x007BDCA1 MOV EBP,ESP
|
||||
0x007BDCA3 MOV EAX,dword ptr [EBP + 0x8]
|
||||
0x007BDCA6 FLD float ptr [EAX]
|
||||
0x007BDCA8 FLD ST0
|
||||
0x007BDCAA FMUL float ptr [ECX]
|
||||
0x007BDCAC FSTP float ptr [ECX]
|
||||
0x007BDCAE FLD ST0
|
||||
0x007BDCB0 FMUL float ptr [ECX + 0x4]
|
||||
0x007BDCB3 FSTP float ptr [ECX + 0x4]
|
||||
0x007BDCB6 FMUL float ptr [ECX + 0x8]
|
||||
0x007BDCB9 FSTP float ptr [ECX + 0x8]
|
||||
0x007BDCBC FLD float ptr [EAX + 0x4]
|
||||
0x007BDCBF FLD ST0
|
||||
0x007BDCC1 FMUL float ptr [ECX + 0x10]
|
||||
0x007BDCC4 FSTP float ptr [ECX + 0x10]
|
||||
0x007BDCC7 FLD ST0
|
||||
0x007BDCC9 FMUL float ptr [ECX + 0x14]
|
||||
0x007BDCCC FSTP float ptr [ECX + 0x14]
|
||||
0x007BDCCF FMUL float ptr [ECX + 0x18]
|
||||
0x007BDCD2 FSTP float ptr [ECX + 0x18]
|
||||
0x007BDCD5 FLD float ptr [EAX + 0x8]
|
||||
0x007BDCD8 FLD ST0
|
||||
0x007BDCDA FMUL float ptr [ECX + 0x20]
|
||||
0x007BDCDD FSTP float ptr [ECX + 0x20]
|
||||
0x007BDCE0 FLD ST0
|
||||
0x007BDCE2 FMUL float ptr [ECX + 0x24]
|
||||
0x007BDCE5 FSTP float ptr [ECX + 0x24]
|
||||
0x007BDCE8 FMUL float ptr [ECX + 0x28]
|
||||
0x007BDCEB FSTP float ptr [ECX + 0x28]
|
||||
0x007BDCEE POP EBP
|
||||
0x007BDCEF RET 0x4 << RET
|
||||
|
||||
=== ApplyTranslationMatrix (0x007BDC40, 90 bytes) ===
|
||||
0x007BDC40 PUSH EBP
|
||||
0x007BDC41 MOV EBP,ESP
|
||||
0x007BDC43 MOV EAX,dword ptr [EBP + 0x8]
|
||||
0x007BDC46 FLD float ptr [ECX + 0x20]
|
||||
0x007BDC49 FMUL float ptr [EAX + 0x8]
|
||||
0x007BDC4C FLD float ptr [ECX + 0x10]
|
||||
0x007BDC4F FMUL float ptr [EAX + 0x4]
|
||||
0x007BDC52 FADDP
|
||||
0x007BDC54 FLD float ptr [EAX]
|
||||
0x007BDC56 FMUL float ptr [ECX]
|
||||
0x007BDC58 FADDP
|
||||
0x007BDC5A FADD float ptr [ECX + 0x30]
|
||||
0x007BDC5D FSTP float ptr [ECX + 0x30]
|
||||
0x007BDC60 FLD float ptr [ECX + 0x24]
|
||||
0x007BDC63 FMUL float ptr [EAX + 0x8]
|
||||
0x007BDC66 FLD float ptr [ECX + 0x14]
|
||||
0x007BDC69 FMUL float ptr [EAX + 0x4]
|
||||
0x007BDC6C FADDP
|
||||
0x007BDC6E FLD float ptr [ECX + 0x4]
|
||||
0x007BDC71 FMUL float ptr [EAX]
|
||||
0x007BDC73 FADDP
|
||||
0x007BDC75 FADD float ptr [ECX + 0x34]
|
||||
0x007BDC78 FSTP float ptr [ECX + 0x34]
|
||||
0x007BDC7B FLD float ptr [ECX + 0x28]
|
||||
0x007BDC7E FMUL float ptr [EAX + 0x8]
|
||||
0x007BDC81 FLD float ptr [ECX + 0x18]
|
||||
0x007BDC84 FMUL float ptr [EAX + 0x4]
|
||||
0x007BDC87 FADDP
|
||||
0x007BDC89 FLD float ptr [ECX + 0x8]
|
||||
0x007BDC8C FMUL float ptr [EAX]
|
||||
0x007BDC8E FADDP
|
||||
0x007BDC90 FADD float ptr [ECX + 0x38]
|
||||
0x007BDC93 FSTP float ptr [ECX + 0x38]
|
||||
0x007BDC96 POP EBP
|
||||
0x007BDC97 RET 0x4 << RET
|
||||
|
||||
=== rotateMatrixByQuaternion (0x007BDDB0, 333 bytes) ===
|
||||
0x007BDDB0 PUSH EBP
|
||||
0x007BDDB1 MOV EBP,ESP
|
||||
0x007BDDB3 SUB ESP,0x9c
|
||||
0x007BDDB9 MOV EAX,dword ptr [EBP + 0x8]
|
||||
0x007BDDBC FLD float ptr [EAX]
|
||||
0x007BDDBE PUSH ESI
|
||||
0x007BDDBF FADD ST0,ST0
|
||||
0x007BDDC1 MOV ESI,ECX
|
||||
0x007BDDC3 FLD float ptr [EAX + 0x4]
|
||||
0x007BDDC6 MOV dword ptr [EBP + -0x50],0x0
|
||||
0x007BDDCD FADD ST0,ST0
|
||||
0x007BDDCF MOV dword ptr [EBP + -0x40],0x0
|
||||
0x007BDDD6 FLD float ptr [EAX + 0x8]
|
||||
0x007BDDD9 FADD ST0,ST0
|
||||
0x007BDDDB FSTP float ptr [EBP + 0x8]
|
||||
0x007BDDDE FLD ST1
|
||||
0x007BDDE0 FMUL float ptr [EAX + 0xc]
|
||||
0x007BDDE3 FSTP float ptr [EBP + -0xc]
|
||||
0x007BDDE6 FLD ST0
|
||||
0x007BDDE8 FMUL float ptr [EAX + 0xc]
|
||||
0x007BDDEB FSTP float ptr [EBP + -0x1c]
|
||||
0x007BDDEE FLD float ptr [EBP + 0x8]
|
||||
0x007BDDF1 FMUL float ptr [EAX + 0xc]
|
||||
0x007BDDF4 FSTP float ptr [EBP + -0x18]
|
||||
0x007BDDF7 FXCH
|
||||
0x007BDDF9 FMUL float ptr [EAX]
|
||||
0x007BDDFB FSTP float ptr [EBP + -0x14]
|
||||
0x007BDDFE FLD ST0
|
||||
0x007BDE00 FMUL float ptr [EAX]
|
||||
0x007BDE02 FSTP float ptr [EBP + -0x8]
|
||||
0x007BDE05 FLD float ptr [EBP + 0x8]
|
||||
0x007BDE08 FMUL float ptr [EAX]
|
||||
0x007BDE0A FSTP float ptr [EBP + -0x10]
|
||||
0x007BDE0D FMUL float ptr [EAX + 0x4]
|
||||
0x007BDE10 FLD float ptr [EBP + 0x8]
|
||||
0x007BDE13 FMUL float ptr [EAX + 0x4]
|
||||
0x007BDE16 FSTP float ptr [EBP + -0x4]
|
||||
0x007BDE19 FLD float ptr [EBP + 0x8]
|
||||
0x007BDE1C FMUL float ptr [EAX + 0x8]
|
||||
0x007BDE1F FST float ptr [EBP + 0x8]
|
||||
0x007BDE22 FADD ST0,ST1
|
||||
0x007BDE24 FSUBR float ptr [0x007ff9d8]
|
||||
0x007BDE2A FLD float ptr [EBP + -0x8]
|
||||
0x007BDE2D FADD float ptr [EBP + -0x18]
|
||||
0x007BDE30 FLD float ptr [EBP + -0x10]
|
||||
0x007BDE33 FSUB float ptr [EBP + -0x1c]
|
||||
0x007BDE36 FSTP float ptr [EBP + -0x78]
|
||||
0x007BDE39 MOV EAX,dword ptr [EBP + -0x78]
|
||||
0x007BDE3C FLD float ptr [EBP + -0x8]
|
||||
0x007BDE3F MOV dword ptr [EBP + -0x54],EAX
|
||||
0x007BDE42 FSUB float ptr [EBP + -0x18]
|
||||
0x007BDE45 FSTP float ptr [EBP + -0x74]
|
||||
0x007BDE48 MOV ECX,dword ptr [EBP + -0x74]
|
||||
0x007BDE4B FLD float ptr [EBP + 0x8]
|
||||
0x007BDE4E MOV dword ptr [EBP + -0x4c],ECX
|
||||
0x007BDE51 FADD float ptr [EBP + -0x14]
|
||||
0x007BDE54 FSUBR float ptr [0x007ff9d8]
|
||||
0x007BDE5A FSTP float ptr [EBP + -0x70]
|
||||
0x007BDE5D MOV EDX,dword ptr [EBP + -0x70]
|
||||
0x007BDE60 FLD float ptr [EBP + -0x4]
|
||||
0x007BDE63 MOV dword ptr [EBP + -0x48],EDX
|
||||
0x007BDE66 FADD float ptr [EBP + -0xc]
|
||||
0x007BDE69 FSTP float ptr [EBP + -0x6c]
|
||||
0x007BDE6C MOV EAX,dword ptr [EBP + -0x6c]
|
||||
0x007BDE6F FLD float ptr [EBP + -0x10]
|
||||
0x007BDE72 MOV dword ptr [EBP + -0x44],EAX
|
||||
0x007BDE75 FADD float ptr [EBP + -0x1c]
|
||||
0x007BDE78 FSTP float ptr [EBP + -0x68]
|
||||
0x007BDE7B MOV ECX,dword ptr [EBP + -0x68]
|
||||
0x007BDE7E FLD float ptr [EBP + -0x4]
|
||||
0x007BDE81 MOV dword ptr [EBP + -0x3c],ECX
|
||||
0x007BDE84 FSUB float ptr [EBP + -0xc]
|
||||
0x007BDE87 FSTP float ptr [EBP + -0x64]
|
||||
0x007BDE8A MOV EDX,dword ptr [EBP + -0x64]
|
||||
0x007BDE8D FXCH ST2
|
||||
0x007BDE8F MOV dword ptr [EBP + -0x38],EDX
|
||||
0x007BDE92 FADD float ptr [EBP + -0x14]
|
||||
0x007BDE95 FSUBR float ptr [0x007ff9d8]
|
||||
0x007BDE9B FSTP float ptr [EBP + -0x60]
|
||||
0x007BDE9E FSTP float ptr [EBP + -0x5c]
|
||||
0x007BDEA1 FSTP float ptr [EBP + -0x58]
|
||||
0x007BDEA4 MOV EAX,dword ptr [EBP + -0x60]
|
||||
0x007BDEA7 PUSH ESI
|
||||
0x007BDEA8 LEA EDX,[EBP + -0x5c]
|
||||
0x007BDEAB LEA ECX,[EBP + 0xffffff64]
|
||||
0x007BDEB1 MOV dword ptr [EBP + -0x34],EAX
|
||||
0x007BDEB4 MOV dword ptr [EBP + -0x30],0x0
|
||||
0x007BDEBB MOV dword ptr [EBP + -0x2c],0x0
|
||||
0x007BDEC2 MOV dword ptr [EBP + -0x28],0x0
|
||||
0x007BDEC9 MOV dword ptr [EBP + -0x24],0x0
|
||||
0x007BDED0 MOV dword ptr [EBP + -0x20],0x3f800000
|
||||
0x007BDED7 CALL 0x007bc6a0 << CALL
|
||||
0x007BDEDC MOV ECX,ESI
|
||||
0x007BDEDE MOV EDX,0x8
|
||||
0x007BDEE3 MOV ESI,dword ptr [EAX]
|
||||
0x007BDEE5 MOV dword ptr [ECX],ESI
|
||||
0x007BDEE7 MOV ESI,dword ptr [EAX + 0x4]
|
||||
0x007BDEEA MOV dword ptr [ECX + 0x4],ESI
|
||||
0x007BDEED ADD ECX,0x8
|
||||
0x007BDEF0 ADD EAX,0x8
|
||||
0x007BDEF3 DEC EDX
|
||||
0x007BDEF4 JNZ 0x007bdee3
|
||||
0x007BDEF6 POP ESI
|
||||
0x007BDEF7 MOV ESP,EBP
|
||||
0x007BDEF9 POP EBP
|
||||
0x007BDEFA RET 0x4 << RET
|
||||
|
||||
=== calculateScaledInverseMatrix (0x007BD820, 347 bytes) ===
|
||||
0x007BD820 PUSH EBP
|
||||
0x007BD821 MOV EBP,ESP
|
||||
0x007BD823 SUB ESP,0x94
|
||||
0x007BD829 FLD float ptr [EBP + 0xc]
|
||||
0x007BD82C PUSH ESI
|
||||
0x007BD82D FSUB float ptr [0x007ff9d8]
|
||||
0x007BD833 PUSH EDI
|
||||
0x007BD834 MOV ESI,ECX
|
||||
0x007BD836 FABS
|
||||
0x007BD838 FCOMP float ptr [0x008026bc]
|
||||
0x007BD83E FNSTSW AX
|
||||
0x007BD840 TEST AH,0x5
|
||||
0x007BD843 JP 0x007bd858
|
||||
0x007BD845 MOV EDI,dword ptr [EBP + 0x8]
|
||||
0x007BD848 PUSH EDI
|
||||
0x007BD849 CALL 0x007bd700 << CALL
|
||||
0x007BD84E MOV EAX,EDI
|
||||
0x007BD850 POP EDI
|
||||
0x007BD851 POP ESI
|
||||
0x007BD852 MOV ESP,EBP
|
||||
0x007BD854 POP EBP
|
||||
0x007BD855 RET 0x8 << RET
|
||||
0x007BD858 MOV EAX,dword ptr [ESI + 0x28]
|
||||
0x007BD85B MOV ECX,dword ptr [ESI + 0x24]
|
||||
0x007BD85E MOV EDX,dword ptr [ESI + 0x20]
|
||||
0x007BD861 PUSH EAX
|
||||
0x007BD862 MOV EAX,dword ptr [ESI + 0x18]
|
||||
0x007BD865 PUSH ECX
|
||||
0x007BD866 MOV ECX,dword ptr [ESI + 0x14]
|
||||
0x007BD869 PUSH EDX
|
||||
0x007BD86A MOV EDX,dword ptr [ESI + 0x10]
|
||||
0x007BD86D PUSH EAX
|
||||
0x007BD86E MOV EAX,dword ptr [ESI + 0x8]
|
||||
0x007BD871 PUSH ECX
|
||||
0x007BD872 MOV ECX,dword ptr [ESI + 0x4]
|
||||
0x007BD875 PUSH EDX
|
||||
0x007BD876 MOV EDX,dword ptr [ESI]
|
||||
0x007BD878 PUSH EAX
|
||||
0x007BD879 PUSH ECX
|
||||
0x007BD87A PUSH EDX
|
||||
0x007BD87B LEA ECX,[EBP + -0x70]
|
||||
0x007BD87E CALL 0x005f8d20 << CALL
|
||||
0x007BD883 MOV EAX,dword ptr [EBP + -0x50]
|
||||
0x007BD886 MOV ECX,dword ptr [EBP + -0x5c]
|
||||
0x007BD889 MOV EDX,dword ptr [EBP + -0x68]
|
||||
0x007BD88C PUSH EAX
|
||||
0x007BD88D MOV EAX,dword ptr [EBP + -0x54]
|
||||
0x007BD890 PUSH ECX
|
||||
0x007BD891 MOV ECX,dword ptr [EBP + -0x60]
|
||||
0x007BD894 PUSH EDX
|
||||
0x007BD895 MOV EDX,dword ptr [EBP + -0x6c]
|
||||
0x007BD898 PUSH EAX
|
||||
0x007BD899 MOV EAX,dword ptr [EBP + -0x58]
|
||||
0x007BD89C PUSH ECX
|
||||
0x007BD89D MOV ECX,dword ptr [EBP + -0x64]
|
||||
0x007BD8A0 PUSH EDX
|
||||
0x007BD8A1 MOV EDX,dword ptr [EBP + -0x70]
|
||||
0x007BD8A4 PUSH EAX
|
||||
0x007BD8A5 PUSH ECX
|
||||
0x007BD8A6 PUSH EDX
|
||||
0x007BD8A7 LEA ECX,[EBP + 0xffffff6c]
|
||||
0x007BD8AD CALL 0x005f8d20 << CALL
|
||||
0x007BD8B2 FLD float ptr [EBP + 0xc]
|
||||
0x007BD8B5 FMUL float ptr [EBP + 0xc]
|
||||
0x007BD8B8 MOV ECX,dword ptr [EBP + 0xffffff70]
|
||||
0x007BD8BE MOV EAX,dword ptr [EBP + 0xffffff6c]
|
||||
0x007BD8C4 MOV EDX,dword ptr [EBP + 0xffffff74]
|
||||
0x007BD8CA FDIVR float ptr [0x007ff9d8]
|
||||
0x007BD8D0 MOV dword ptr [EBP + -0x48],ECX
|
||||
0x007BD8D3 MOV ECX,dword ptr [EBP + 0xffffff7c]
|
||||
0x007BD8D9 MOV dword ptr [EBP + -0x4c],EAX
|
||||
0x007BD8DC MOV EAX,dword ptr [EBP + 0xffffff78]
|
||||
0x007BD8E2 MOV dword ptr [EBP + -0x44],EDX
|
||||
0x007BD8E5 MOV EDX,dword ptr [EBP + -0x80]
|
||||
0x007BD8E8 MOV dword ptr [EBP + -0x38],ECX
|
||||
0x007BD8EB MOV ECX,dword ptr [EBP + -0x78]
|
||||
0x007BD8EE MOV dword ptr [EBP + -0x3c],EAX
|
||||
0x007BD8F1 MOV EAX,dword ptr [EBP + -0x7c]
|
||||
0x007BD8F4 MOV dword ptr [EBP + -0x34],EDX
|
||||
0x007BD8F7 MOV EDX,dword ptr [EBP + -0x74]
|
||||
0x007BD8FA PUSH ECX
|
||||
0x007BD8FB MOV dword ptr [EBP + -0x28],ECX
|
||||
0x007BD8FE LEA ECX,[EBP + -0x4c]
|
||||
0x007BD901 MOV dword ptr [EBP + -0x40],0x0
|
||||
0x007BD908 MOV dword ptr [EBP + -0x30],0x0
|
||||
0x007BD90F MOV dword ptr [EBP + -0x2c],EAX
|
||||
0x007BD912 MOV dword ptr [EBP + -0x24],EDX
|
||||
0x007BD915 MOV dword ptr [EBP + -0x20],0x0
|
||||
0x007BD91C MOV dword ptr [EBP + -0x1c],0x0
|
||||
0x007BD923 MOV dword ptr [EBP + -0x18],0x0
|
||||
0x007BD92A MOV dword ptr [EBP + -0x14],0x0
|
||||
0x007BD931 MOV dword ptr [EBP + -0x10],0x3f800000
|
||||
0x007BD938 FSTP float ptr [ESP]
|
||||
0x007BD93B CALL 0x007bdd00 << CALL
|
||||
0x007BD940 FLD float ptr [ESI + 0x30]
|
||||
0x007BD943 FCHS
|
||||
0x007BD945 FSTP float ptr [EBP + -0xc]
|
||||
0x007BD948 FLD float ptr [ESI + 0x34]
|
||||
0x007BD94B FCHS
|
||||
0x007BD94D FSTP float ptr [EBP + -0x8]
|
||||
0x007BD950 FLD float ptr [ESI + 0x38]
|
||||
0x007BD953 LEA EAX,[EBP + -0xc]
|
||||
0x007BD956 FCHS
|
||||
0x007BD958 PUSH EAX
|
||||
0x007BD959 FSTP float ptr [EBP + -0x4]
|
||||
0x007BD95C LEA ECX,[EBP + -0x4c]
|
||||
0x007BD95F CALL 0x007bdc40 << CALL
|
||||
0x007BD964 MOV EAX,dword ptr [EBP + 0x8]
|
||||
0x007BD967 MOV ECX,0x10
|
||||
0x007BD96C LEA ESI,[EBP + -0x4c]
|
||||
0x007BD96F MOV EDI,EAX
|
||||
0x007BD971 MOVSD.REP ES:EDI,ESI
|
||||
0x007BD973 POP EDI
|
||||
0x007BD974 POP ESI
|
||||
0x007BD975 MOV ESP,EBP
|
||||
0x007BD977 POP EBP
|
||||
0x007BD978 RET 0x8 << RET
|
||||
|
||||
=== multiplyMatrix4x4_Basic (0x007507BB, 229 bytes) ===
|
||||
0x007507BB MOV EDI,EDI
|
||||
0x007507BD PUSH EBP
|
||||
0x007507BE MOV EBP,ESP
|
||||
0x007507C0 SUB ESP,0x40
|
||||
0x007507C3 MOV EAX,dword ptr [EBP + 0x8]
|
||||
0x007507C6 PUSH EBX
|
||||
0x007507C7 PUSH ESI
|
||||
0x007507C8 MOV ESI,dword ptr [EBP + 0x10]
|
||||
0x007507CB CMP ESI,EAX
|
||||
0x007507CD PUSH EDI
|
||||
0x007507CE JNZ 0x00750841
|
||||
0x007507D0 CMP dword ptr [EBP + 0xc],EAX
|
||||
0x007507D3 JZ 0x00750833
|
||||
0x007507D5 MOV EBX,dword ptr [EBP + 0x8]
|
||||
0x007507D8 MOV ECX,dword ptr [EBP + 0xc]
|
||||
0x007507DB MOV EDX,dword ptr [EBP + 0x10]
|
||||
0x007507DE MOV EDI,0xfffffffc
|
||||
0x007507E3 MOV ESI,0xfffffff0
|
||||
0x007507E8 FLD float ptr [EDX + EDI*0x4 + 0x10]
|
||||
0x007507EC FLD float ptr [EDX + EDI*0x4 + 0x20]
|
||||
0x007507F0 FLD float ptr [EDX + EDI*0x4 + 0x30]
|
||||
0x007507F4 FLD float ptr [EDX + EDI*0x4 + 0x40]
|
||||
0x007507F8 FLD ST3
|
||||
0x007507FA FMUL float ptr [ECX + ESI*0x4 + 0x40]
|
||||
0x007507FE FLD ST3
|
||||
0x00750800 FMUL float ptr [ECX + ESI*0x4 + 0x44]
|
||||
0x00750804 FLD ST3
|
||||
0x00750806 FMUL float ptr [ECX + ESI*0x4 + 0x48]
|
||||
0x0075080A FLD ST3
|
||||
0x0075080C FMUL float ptr [ECX + ESI*0x4 + 0x4c]
|
||||
0x00750810 FXCH ST3
|
||||
0x00750812 FADDP
|
||||
0x00750814 FXCH ST2
|
||||
0x00750816 FADDP
|
||||
0x00750818 FADDP
|
||||
0x0075081A FSTP float ptr [EBX + ESI*0x4 + 0x40]
|
||||
0x0075081E ADD ESI,0x4
|
||||
0x00750821 JNZ 0x007507f8
|
||||
0x00750823 FFREE ST3
|
||||
0x00750825 FFREE ST2
|
||||
0x00750827 FFREE ST1
|
||||
0x00750829 FFREE ST0
|
||||
0x0075082B LEA EBX,[EBX + 0x4]
|
||||
0x0075082E INC EDI
|
||||
0x0075082F JNZ 0x007507e3
|
||||
0x00750831 JMP 0x00750899
|
||||
0x00750833 PUSH 0x10
|
||||
0x00750835 POP ECX
|
||||
0x00750836 LEA EDI,[EBP + -0x40]
|
||||
0x00750839 MOVSD.REP ES:EDI,ESI
|
||||
0x0075083B LEA ECX,[EBP + -0x40]
|
||||
0x0075083E MOV dword ptr [EBP + 0x10],ECX
|
||||
0x00750841 MOV EBX,dword ptr [EBP + 0x8]
|
||||
0x00750844 MOV ECX,dword ptr [EBP + 0xc]
|
||||
0x00750847 MOV EDX,dword ptr [EBP + 0x10]
|
||||
0x0075084A MOV EDI,0xfffffffc
|
||||
0x0075084F MOV ESI,0xfffffffc
|
||||
0x00750854 FLD float ptr [ECX]
|
||||
0x00750856 FLD float ptr [ECX + 0x4]
|
||||
0x00750859 FLD float ptr [ECX + 0x8]
|
||||
0x0075085C FLD float ptr [ECX + 0xc]
|
||||
0x0075085F FLD ST3
|
||||
0x00750861 FMUL float ptr [EDX + ESI*0x4 + 0x10]
|
||||
0x00750865 FLD ST3
|
||||
0x00750867 FMUL float ptr [EDX + ESI*0x4 + 0x20]
|
||||
0x0075086B FLD ST3
|
||||
0x0075086D FMUL float ptr [EDX + ESI*0x4 + 0x30]
|
||||
0x00750871 FLD ST3
|
||||
0x00750873 FMUL float ptr [EDX + ESI*0x4 + 0x40]
|
||||
0x00750877 FXCH ST3
|
||||
0x00750879 FADDP
|
||||
0x0075087B FXCH ST2
|
||||
0x0075087D FADDP
|
||||
0x0075087F FADDP
|
||||
0x00750881 FSTP float ptr [EBX + ESI*0x4 + 0x10]
|
||||
0x00750885 INC ESI
|
||||
0x00750886 JNZ 0x0075085f
|
||||
0x00750888 FFREE ST3
|
||||
0x0075088A FFREE ST2
|
||||
0x0075088C FFREE ST1
|
||||
0x0075088E FFREE ST0
|
||||
0x00750890 LEA ECX,[ECX + 0x10]
|
||||
0x00750893 LEA EBX,[EBX + 0x10]
|
||||
0x00750896 INC EDI
|
||||
0x00750897 JNZ 0x0075084f
|
||||
0x00750899 POP EDI
|
||||
0x0075089A POP ESI
|
||||
0x0075089B POP EBX
|
||||
0x0075089C LEAVE
|
||||
0x0075089D RET 0xc << RET
|
||||
|
||||
=== buildRotMatFromQuat (0x0074B6BB) NOT FOUND ===
|
||||
|
||||
=== squaredMagnitude (0x004549F0, 31 bytes) ===
|
||||
0x004549F0 FLD float ptr [ECX + 0x8]
|
||||
0x004549F3 FLD float ptr [ECX + 0x4]
|
||||
0x004549F6 FLD float ptr [ECX]
|
||||
0x004549F8 FLD ST0
|
||||
0x004549FA FMUL ST1
|
||||
0x004549FC FLD ST2
|
||||
0x004549FE FMUL ST3
|
||||
0x00454A00 FADDP
|
||||
0x00454A02 FLD ST3
|
||||
0x00454A04 FMUL ST4
|
||||
0x00454A06 FADDP
|
||||
0x00454A08 FSTP ST3
|
||||
0x00454A0A FSTP ST0
|
||||
0x00454A0C FSTP ST0
|
||||
0x00454A0E RET << RET
|
||||
|
||||
@@ -0,0 +1,424 @@
|
||||
//! SSE math polyfill — replaces x87 FPU functions with SSE equivalents.
|
||||
//!
|
||||
//! Compiled ReleaseFast even in Debug builds (separate compilation unit).
|
||||
//! Each function replaces a game x87 implementation identified from UnitXP's
|
||||
//! polyfill.cpp and libSiliconPatch's export table.
|
||||
//!
|
||||
//! Reference sources:
|
||||
//! UnitXP: reference/UnitXP_SP3/polyfill.cpp (scalar double intermediates)
|
||||
//! Silicon: /tmp/TurtleSilicon/winerosetta/libSiliconPatch.dll (closed source, symbols only)
|
||||
//!
|
||||
//! Our approach: @Vector(4, f32) SSE intrinsics where beneficial, scalar for simple ops.
|
||||
|
||||
const V4 = @Vector(4, f32);
|
||||
|
||||
// MSVC CRT — linked from the WoW process
|
||||
extern fn sinf(f32) f32;
|
||||
extern fn cosf(f32) f32;
|
||||
|
||||
// =============================================================================
|
||||
// Memory access helpers (same as clip_sse.zig / bone_sse.zig)
|
||||
// =============================================================================
|
||||
|
||||
inline fn rf32(addr: u32) f32 {
|
||||
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
inline fn wf32(addr: u32, v: f32) void {
|
||||
@as(*align(1) f32, @ptrFromInt(addr)).* = v;
|
||||
}
|
||||
inline fn splat(v: f32) V4 {
|
||||
return @splat(v);
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Vector-Matrix multiplies (0x7BCA80, 0x7BCAE0, 0x7BCB40)
|
||||
//
|
||||
// The game has three variants of vec/quat * matrix multiply, all using x87 FPU.
|
||||
// UnitXP replaces with double intermediates. We use SSE broadcast-multiply-add.
|
||||
//
|
||||
// Reference: polyfill.cpp lines 35-61 (detoured_operator_multiply_1/2/3)
|
||||
// =============================================================================
|
||||
|
||||
/// 0x7BCA80: result = vec3 * mat4 (column-major: result[i] = dot(vec, mat_col_i) + mat[3][i])
|
||||
/// __fastcall(ECX=result, EDX=vec3, stack: mat4*), RET 0x4
|
||||
export fn vecMulMat4_ColMajor(result: u32, vec: u32, mat: u32) u32 {
|
||||
const vx = rf32(vec);
|
||||
const vy = rf32(vec + 4);
|
||||
const vz = rf32(vec + 8);
|
||||
wf32(result, vx * rf32(mat) + vy * rf32(mat + 0x10) + vz * rf32(mat + 0x20) + rf32(mat + 0x30));
|
||||
wf32(result + 4, vx * rf32(mat + 0x04) + vy * rf32(mat + 0x14) + vz * rf32(mat + 0x24) + rf32(mat + 0x34));
|
||||
wf32(result + 8, vx * rf32(mat + 0x08) + vy * rf32(mat + 0x18) + vz * rf32(mat + 0x28) + rf32(mat + 0x38));
|
||||
return result;
|
||||
}
|
||||
|
||||
/// 0x7BCAE0: result = mat4 * vec3 (row-major: result[i] = dot(mat_row_i, vec) + mat[i][3])
|
||||
/// __fastcall(ECX=result, EDX=mat4, stack: vec3*), RET 0x4
|
||||
export fn matMulVec3_RowMajor(result: u32, mat: u32, vec: u32) u32 {
|
||||
const vx = rf32(vec);
|
||||
const vy = rf32(vec + 4);
|
||||
const vz = rf32(vec + 8);
|
||||
wf32(result, rf32(mat) * vx + rf32(mat + 0x04) * vy + rf32(mat + 0x08) * vz + rf32(mat + 0x0C));
|
||||
wf32(result + 4, rf32(mat + 0x10) * vx + rf32(mat + 0x14) * vy + rf32(mat + 0x18) * vz + rf32(mat + 0x1C));
|
||||
wf32(result + 8, rf32(mat + 0x20) * vx + rf32(mat + 0x24) * vy + rf32(mat + 0x28) * vz + rf32(mat + 0x2C));
|
||||
return result;
|
||||
}
|
||||
|
||||
/// 0x7BCB40: result = quat4 * mat4 (4-component: result[i] = dot(quat, mat_col_i))
|
||||
/// __fastcall(ECX=result, EDX=quat4, stack: mat4*), RET 0x4
|
||||
/// Reference: polyfill.cpp line 55-61
|
||||
export fn quatMulMat4(result: u32, quat: u32, mat: u32) u32 {
|
||||
const q: V4 = .{ rf32(quat), rf32(quat + 4), rf32(quat + 8), rf32(quat + 12) };
|
||||
inline for (0..4) |i| {
|
||||
const col_off = @as(u32, @intCast(i)) * 4;
|
||||
const c: V4 = .{ rf32(mat + col_off), rf32(mat + 0x10 + col_off), rf32(mat + 0x20 + col_off), rf32(mat + 0x30 + col_off) };
|
||||
wf32(result + col_off, @reduce(.Add, q * c));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Vector scalar operations (0x5F8CF0, 0x5132F0)
|
||||
//
|
||||
// Reference: polyfill.cpp lines 117-132
|
||||
// =============================================================================
|
||||
|
||||
/// 0x5F8CF0: result = vec3 * scalar
|
||||
/// __fastcall(ECX=result, EDX=vec3, stack: factor_float), RET 0x4
|
||||
export fn vec3MulScalar(result: u32, vec: u32, factor_bits: u32) u32 {
|
||||
const f: f32 = @bitCast(factor_bits);
|
||||
wf32(result, rf32(vec) * f);
|
||||
wf32(result + 4, rf32(vec + 4) * f);
|
||||
wf32(result + 8, rf32(vec + 8) * f);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// 0x5132F0: self *= scalar (in-place)
|
||||
/// __thiscall(ECX=self, stack: factor_float), RET 0x4
|
||||
export fn vec3MulAssign(self: u32, factor_bits: u32) u32 {
|
||||
const f: f32 = @bitCast(factor_bits);
|
||||
wf32(self, rf32(self) * f);
|
||||
wf32(self + 4, rf32(self + 4) * f);
|
||||
wf32(self + 8, rf32(self + 8) * f);
|
||||
return self;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Matrix operations (0x7BDC40, 0x7BDCA0, 0x7BDD00, 0x7BDFC0)
|
||||
//
|
||||
// ApplyTranslation and ScaleMatrix are already reimplemented inline in bone_sse.zig.
|
||||
// These standalone hooks catch calls from OUTSIDE transformMatrix4x4.
|
||||
//
|
||||
// Reference: polyfill.cpp lines 134-218
|
||||
// =============================================================================
|
||||
|
||||
/// 0x7BDC40: Apply translation through rotation matrix (in-place)
|
||||
/// mat[3][j] += dot(mat[col_j], translation) for j=0,1,2
|
||||
/// __thiscall(ECX=mat4x4, stack: vec3*), RET 0x4
|
||||
/// Reference: polyfill.cpp line 136, also bone_sse.zig applyTranslation()
|
||||
export fn applyTranslationMatrix(mat: u32, vec: u32) u32 {
|
||||
const tx = rf32(vec);
|
||||
const ty = rf32(vec + 4);
|
||||
const tz = rf32(vec + 8);
|
||||
wf32(mat + 0x30, tx * rf32(mat) + ty * rf32(mat + 0x10) + tz * rf32(mat + 0x20) + rf32(mat + 0x30));
|
||||
wf32(mat + 0x34, tx * rf32(mat + 0x04) + ty * rf32(mat + 0x14) + tz * rf32(mat + 0x24) + rf32(mat + 0x34));
|
||||
wf32(mat + 0x38, tx * rf32(mat + 0x08) + ty * rf32(mat + 0x18) + tz * rf32(mat + 0x28) + rf32(mat + 0x38));
|
||||
return vec; // original returns param_1 (vec ptr)
|
||||
}
|
||||
|
||||
/// 0x7BDCA0: Scale 3x3 rotation portion by per-axis scale vector
|
||||
/// row0 *= scale.x, row1 *= scale.y, row2 *= scale.z
|
||||
/// __thiscall(ECX=mat4x4, stack: vec3*), RET 0x4
|
||||
/// Reference: polyfill.cpp line 146, also bone_sse.zig scaleMatrix3x3()
|
||||
export fn scaleMatrix3x3ByVector(mat: u32, vec: u32) u32 {
|
||||
const sx = rf32(vec);
|
||||
const sy = rf32(vec + 4);
|
||||
const sz = rf32(vec + 8);
|
||||
// Row 0
|
||||
wf32(mat, rf32(mat) * sx);
|
||||
wf32(mat + 0x04, rf32(mat + 0x04) * sx);
|
||||
wf32(mat + 0x08, rf32(mat + 0x08) * sx);
|
||||
// Row 1
|
||||
wf32(mat + 0x10, rf32(mat + 0x10) * sy);
|
||||
wf32(mat + 0x14, rf32(mat + 0x14) * sy);
|
||||
wf32(mat + 0x18, rf32(mat + 0x18) * sy);
|
||||
// Row 2
|
||||
wf32(mat + 0x20, rf32(mat + 0x20) * sz);
|
||||
wf32(mat + 0x24, rf32(mat + 0x24) * sz);
|
||||
wf32(mat + 0x28, rf32(mat + 0x28) * sz);
|
||||
return vec;
|
||||
}
|
||||
|
||||
/// 0x7BDD00: Scale 3x3 rotation portion by uniform scalar
|
||||
/// __thiscall(ECX=mat4x4, stack: factor_float), plain RET
|
||||
/// Reference: polyfill.cpp line 162
|
||||
export fn scaleMatrix3x3ByScalar(mat: u32, factor_bits: u32) void {
|
||||
const f: f32 = @bitCast(factor_bits);
|
||||
inline for ([_]u32{ 0x00, 0x04, 0x08, 0x10, 0x14, 0x18, 0x20, 0x24, 0x28 }) |off| {
|
||||
wf32(mat + off, rf32(mat + off) * f);
|
||||
}
|
||||
}
|
||||
|
||||
/// 0x7BDFC0: 3x3 matrix multiply: result = A * B (9 elements, row-major)
|
||||
/// __fastcall(ECX=result, EDX=matA, stack: matB*), RET 0x4
|
||||
/// Reference: polyfill.cpp line 207
|
||||
export fn multiply3x3Matrix(result: u32, a: u32, b: u32) u32 {
|
||||
// result[row][col] = sum(A[row][k] * B[k][col], k=0..2)
|
||||
// 3x3 stored as 3 rows of 3 floats (stride 0x0C per row)
|
||||
inline for (0..3) |row| {
|
||||
const r = @as(u32, @intCast(row)) * 0x0C;
|
||||
inline for (0..3) |col| {
|
||||
const c = @as(u32, @intCast(col)) * 4;
|
||||
wf32(result + r + c,
|
||||
rf32(a + r) * rf32(b + c) +
|
||||
rf32(a + r + 4) * rf32(b + 0x0C + c) +
|
||||
rf32(a + r + 8) * rf32(b + 0x18 + c));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Rotation matrices (0x7BE490, 0x7BDB00)
|
||||
//
|
||||
// 0x7BDD60 (rotateMatrixByAxisAngle 4x4) is already in clip_sse.zig.
|
||||
// These are related but different entry points.
|
||||
//
|
||||
// Reference: polyfill.cpp lines 177-285
|
||||
// =============================================================================
|
||||
|
||||
/// 0x7BE490: Create 3x3 rotation matrix from axis + angle (Rodrigues formula)
|
||||
/// __fastcall(ECX=result_3x3, EDX=axis_vec3, stack: angle_float, is_unit_bool), RET 0x8
|
||||
/// Reference: polyfill.cpp line 177
|
||||
export fn createAxisAngleRotMat3x3(result: u32, axis: u32, angle_bits: u32, is_unit: u32) u32 {
|
||||
const angle: f32 = @bitCast(angle_bits);
|
||||
var ax = rf32(axis);
|
||||
var ay = rf32(axis + 4);
|
||||
var az = rf32(axis + 8);
|
||||
if (is_unit == 0) {
|
||||
const inv = 1.0 / @sqrt(ax * ax + ay * ay + az * az);
|
||||
ax *= inv;
|
||||
ay *= inv;
|
||||
az *= inv;
|
||||
}
|
||||
const c = cosf(angle);
|
||||
const s = sinf(angle);
|
||||
const t = 1.0 - c;
|
||||
wf32(result, ax * ax * t + c);
|
||||
wf32(result + 0x04, ax * ay * t + az * s);
|
||||
wf32(result + 0x08, ax * az * t - ay * s);
|
||||
wf32(result + 0x0C, ax * ay * t - az * s);
|
||||
wf32(result + 0x10, ay * ay * t + c);
|
||||
wf32(result + 0x14, ay * az * t + ax * s);
|
||||
wf32(result + 0x18, ax * az * t + ay * s);
|
||||
wf32(result + 0x1C, ay * az * t - ax * s);
|
||||
wf32(result + 0x20, az * az * t + c);
|
||||
return result;
|
||||
}
|
||||
|
||||
/// 0x7BDB00: Create 4x4 rotation matrix from axis + angle (Rodrigues + identity row/col)
|
||||
/// __fastcall(ECX=result_4x4, EDX=axis_vec3, stack: angle_float, is_unit_bool), RET 0x8
|
||||
/// Like 0x7BE490 but outputs 4x4 with identity padding.
|
||||
/// Reference: polyfill.cpp line 254
|
||||
export fn createAxisAngleRotMat4x4(result: u32, axis: u32, angle_bits: u32, is_unit: u32) u32 {
|
||||
const angle: f32 = @bitCast(angle_bits);
|
||||
var ax = rf32(axis);
|
||||
var ay = rf32(axis + 4);
|
||||
var az = rf32(axis + 8);
|
||||
if (is_unit == 0) {
|
||||
const inv = 1.0 / @sqrt(ax * ax + ay * ay + az * az);
|
||||
ax *= inv;
|
||||
ay *= inv;
|
||||
az *= inv;
|
||||
}
|
||||
const c = cosf(angle);
|
||||
const s = sinf(angle);
|
||||
const t = 1.0 - c;
|
||||
wf32(result + 0x00, ax * ax * t + c);
|
||||
wf32(result + 0x04, ax * ay * t + az * s);
|
||||
wf32(result + 0x08, ax * az * t - ay * s);
|
||||
wf32(result + 0x0C, 0);
|
||||
wf32(result + 0x10, ax * ay * t - az * s);
|
||||
wf32(result + 0x14, ay * ay * t + c);
|
||||
wf32(result + 0x18, ay * az * t + ax * s);
|
||||
wf32(result + 0x1C, 0);
|
||||
wf32(result + 0x20, ax * az * t + ay * s);
|
||||
wf32(result + 0x24, ay * az * t - ax * s);
|
||||
wf32(result + 0x28, az * az * t + c);
|
||||
wf32(result + 0x2C, 0);
|
||||
wf32(result + 0x30, 0);
|
||||
wf32(result + 0x34, 0);
|
||||
wf32(result + 0x38, 0);
|
||||
wf32(result + 0x3C, 1);
|
||||
return result;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Vector math primitives (0x672130, 0x602630, 0x4549F0, 0x699330)
|
||||
//
|
||||
// These are called thousands of times per frame from collision, terrain,
|
||||
// and rendering code. The originals use x87 FPU.
|
||||
//
|
||||
// Reference: polyfill.cpp lines 287-462
|
||||
// =============================================================================
|
||||
|
||||
/// 0x672130: Cross product: result = A x B
|
||||
/// __fastcall(ECX=result, EDX=vecA, stack: vecB*), RET 0x4
|
||||
/// Reference: polyfill.cpp line 451
|
||||
export fn crossProduct(result: u32, a: u32, b: u32) u32 {
|
||||
wf32(result, rf32(a + 4) * rf32(b + 8) - rf32(a + 8) * rf32(b + 4));
|
||||
wf32(result + 4, rf32(a + 8) * rf32(b) - rf32(a) * rf32(b + 8));
|
||||
wf32(result + 8, rf32(a) * rf32(b + 4) - rf32(a + 4) * rf32(b));
|
||||
return result;
|
||||
}
|
||||
|
||||
/// 0x602630: Dot product: return A . B (as f64)
|
||||
/// __fastcall(ECX=vecA, EDX=vecB), plain RET, returns double in ST(0)
|
||||
/// Reference: polyfill.cpp line 460
|
||||
/// TODO: verify return convention — x87 ST(0) double requires special handling
|
||||
export fn dotProduct(a: u32, b: u32) f64 {
|
||||
return @as(f64, rf32(a)) * @as(f64, rf32(b)) +
|
||||
@as(f64, rf32(a + 4)) * @as(f64, rf32(b + 4)) +
|
||||
@as(f64, rf32(a + 8)) * @as(f64, rf32(b + 8));
|
||||
}
|
||||
|
||||
/// 0x4549F0: Squared magnitude of vec3 (returns double in ST(0))
|
||||
/// __fastcall(ECX=vec3), plain RET
|
||||
/// Note: Ghidra labels this "emptyFunction" — it's NOT empty, it returns x*x+y*y+z*z
|
||||
/// Reference: polyfill.cpp line 289
|
||||
export fn squaredMagnitude(vec: u32) f64 {
|
||||
const x: f64 = @floatCast(rf32(vec));
|
||||
const y: f64 = @floatCast(rf32(vec + 4));
|
||||
const z: f64 = @floatCast(rf32(vec + 8));
|
||||
return x * x + y * y + z * z;
|
||||
}
|
||||
|
||||
/// 0x699330: Vector normalize (in-place) — from libSiliconPatch symbols
|
||||
/// __fastcall(ECX=vec3, EDX=vec3_other?), RET
|
||||
/// TODO: verify calling convention from assembly before enabling
|
||||
export fn vectorNormalize(vec: u32, _: u32) void {
|
||||
const x = rf32(vec);
|
||||
const y = rf32(vec + 4);
|
||||
const z = rf32(vec + 8);
|
||||
const len = @sqrt(x * x + y * y + z * z);
|
||||
if (len > 1.0e-7) {
|
||||
const inv = 1.0 / len;
|
||||
wf32(vec, x * inv);
|
||||
wf32(vec + 4, y * inv);
|
||||
wf32(vec + 8, z * inv);
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Polynomial evaluation (0x453620)
|
||||
//
|
||||
// Horner's method for evaluating polynomial coefficients.
|
||||
// Used by animation curves and interpolation.
|
||||
//
|
||||
// Reference: polyfill.cpp line 466
|
||||
// =============================================================================
|
||||
|
||||
/// 0x453620: Evaluate polynomial using Horner's method (returns double in ST(0))
|
||||
/// __fastcall(ECX=degree, EDX=coefficients*, stack: factor_float), RET 0x4
|
||||
export fn evaluatePolynomial(count: u32, coefficients: u32, factor_bits: u32) f64 {
|
||||
const f: f64 = @floatCast(@as(f32, @bitCast(factor_bits)));
|
||||
var result: f64 = @floatCast(rf32(coefficients));
|
||||
var i: u32 = 1;
|
||||
while (i <= count) : (i += 1) {
|
||||
result = result * f + @as(f64, @floatCast(rf32(coefficients + i * 4)));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Geometry / collision functions (0x637480, 0x6DC470, 0x632830, 0x6329E0, 0x6335D0)
|
||||
//
|
||||
// These are collision detection and terrain processing functions.
|
||||
// The originals use x87 for geometry math.
|
||||
//
|
||||
// Reference: polyfill.cpp lines 295-334 (calculatePlaneNormal, transformAABox)
|
||||
// libSiliconPatch symbols: hook_sub_632830, hook_sub_6329E0, hook_sub_6335D0
|
||||
// =============================================================================
|
||||
|
||||
/// 0x637480: Calculate plane normal from 3 points + normalize
|
||||
/// __thiscall(ECX=result_plane4, stack: p1*, p2*, p3*), RET 0xC
|
||||
/// result = {nx, ny, nz, d} where n = normalize(cross(p2-p1, p3-p1)), d = -dot(n, p1)
|
||||
/// Reference: polyfill.cpp line 297
|
||||
export fn calculatePlaneNormal(result: u32, p1: u32, p2: u32, p3: u32) void {
|
||||
// edge vectors
|
||||
const e1x: f64 = @as(f64, rf32(p2)) - @as(f64, rf32(p1));
|
||||
const e1y: f64 = @as(f64, rf32(p2 + 4)) - @as(f64, rf32(p1 + 4));
|
||||
const e1z: f64 = @as(f64, rf32(p2 + 8)) - @as(f64, rf32(p1 + 8));
|
||||
const e2x: f64 = @as(f64, rf32(p3)) - @as(f64, rf32(p1));
|
||||
const e2y: f64 = @as(f64, rf32(p3 + 4)) - @as(f64, rf32(p1 + 4));
|
||||
const e2z: f64 = @as(f64, rf32(p3 + 8)) - @as(f64, rf32(p1 + 8));
|
||||
// cross product
|
||||
const nx = e1y * e2z - e1z * e2y;
|
||||
const ny = e1z * e2x - e1x * e2z;
|
||||
const nz = e1x * e2y - e1y * e2x;
|
||||
const len = @sqrt(nx * nx + ny * ny + nz * nz);
|
||||
wf32(result, @floatCast(nx / len));
|
||||
wf32(result + 4, @floatCast(ny / len));
|
||||
wf32(result + 8, @floatCast(nz / len));
|
||||
wf32(result + 12, @floatCast(-(nx * @as(f64, rf32(p1)) + ny * @as(f64, rf32(p1 + 4)) + nz * @as(f64, rf32(p1 + 8))) / len));
|
||||
}
|
||||
|
||||
/// 0x6DC470: Transform axis-aligned bounding box by 3x3 matrix + translation
|
||||
/// __fastcall(ECX=mat3x3, EDX=vecA, stack: vecB*, boxIn*, boxOut*), RET 0xC
|
||||
/// Reference: polyfill.cpp line 312
|
||||
/// TODO: verify param order from assembly — UnitXP's C++ signature may differ
|
||||
export fn transformAABox(mat: u32, vec_a: u32, vec_b: u32, box_in: u32, box_out: u32) void {
|
||||
// The original iterates 3 outer x 3 inner, comparing min/max per component
|
||||
// Reference: polyfill.cpp lines 312-334
|
||||
var ptrs: [3]u32 = .{ mat, vec_a, vec_b };
|
||||
var out = box_out;
|
||||
var outer_i: u32 = 0;
|
||||
while (outer_i < 3) : ({
|
||||
outer_i += 1;
|
||||
out += 4;
|
||||
}) {
|
||||
var inner_i: u32 = 0;
|
||||
while (inner_i < 3) : (inner_i += 1) {
|
||||
const mat_val: f64 = @floatCast(rf32(ptrs[inner_i] + outer_i * 4));
|
||||
const test1: f64 = mat_val * @as(f64, rf32(box_in + inner_i * 4));
|
||||
const test2: f64 = @as(f64, rf32(box_in + inner_i * 4 + 12)) * mat_val;
|
||||
if (test2 <= test1) {
|
||||
wf32(out, @as(f32, @floatCast(test2 + @as(f64, rf32(out)))));
|
||||
wf32(out + 12, @as(f32, @floatCast(test1 + @as(f64, rf32(out + 12)))));
|
||||
} else {
|
||||
wf32(out, @as(f32, @floatCast(test1 + @as(f64, rf32(out)))));
|
||||
wf32(out + 12, @as(f32, @floatCast(test2 + @as(f64, rf32(out + 12)))));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// libSiliconPatch stubs moved to src/silicon/silicon.zig
|
||||
// See that module for the full catalog of ~200 silicon hooks with addresses.
|
||||
// =============================================================================
|
||||
|
||||
// =============================================================================
|
||||
// CriticalSection spin count optimization (from UnitXP)
|
||||
//
|
||||
// Not a math replacement — sets SpinCount=4000 on critical sections to reduce
|
||||
// kernel transitions. The game initializes CriticalSections with SpinCount=0,
|
||||
// causing immediate kernel waits on contention. With SpinCount=4000, the thread
|
||||
// spins in userspace first, which is faster for short-held locks.
|
||||
//
|
||||
// Reference: polyfill.cpp line 474-482
|
||||
// Implementation: hook EnterCriticalSection, set SpinCount if 0.
|
||||
// This goes in transform44.zig, not here (not SSE math).
|
||||
// =============================================================================
|
||||
|
||||
// =============================================================================
|
||||
// Blit optimization (from UnitXP)
|
||||
//
|
||||
// Replaces game's REP MOVSQ + REP MOVSB pattern with std::memcpy.
|
||||
// On modern CPUs with Enhanced REP MOVSB (ERMS), the old split approach
|
||||
// is slower than a single memcpy which the compiler optimizes.
|
||||
//
|
||||
// We already profile blit_hub (0x5A4F60) in transform44.zig.
|
||||
// To implement: add format-specific fast paths using @memcpy in the detour.
|
||||
//
|
||||
// Reference: polyfill.cpp lines 344-435
|
||||
// =============================================================================
|
||||
+331
-13
@@ -23,6 +23,26 @@ extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void;
|
||||
extern fn multiplyMatrix4x4(u32, u32, u32) u32;
|
||||
extern fn transformMatrix4x4_SSE(u32, u32, u32, u32, u32) void;
|
||||
|
||||
// math_sse.zig exports (UnitXP polyfill replacements)
|
||||
extern fn vecMulMat4_ColMajor(u32, u32, u32) u32;
|
||||
extern fn matMulVec3_RowMajor(u32, u32, u32) u32;
|
||||
extern fn quatMulMat4(u32, u32, u32) u32;
|
||||
extern fn vec3MulScalar(u32, u32, u32) u32;
|
||||
extern fn vec3MulAssign(u32, u32) u32;
|
||||
extern fn applyTranslationMatrix(u32, u32) u32;
|
||||
extern fn scaleMatrix3x3ByVector(u32, u32) u32;
|
||||
extern fn scaleMatrix3x3ByScalar(u32, u32) void;
|
||||
extern fn multiply3x3Matrix(u32, u32, u32) u32;
|
||||
extern fn createAxisAngleRotMat3x3(u32, u32, u32, u32) u32;
|
||||
extern fn createAxisAngleRotMat4x4(u32, u32, u32, u32) u32;
|
||||
extern fn crossProduct(u32, u32, u32) u32;
|
||||
extern fn dotProduct(u32, u32) f64;
|
||||
extern fn squaredMagnitude(u32) f64;
|
||||
extern fn vectorNormalize(u32, u32) void;
|
||||
extern fn evaluatePolynomial(u32, u32, u32) f64;
|
||||
extern fn calculatePlaneNormal(u32, u32, u32, u32) void;
|
||||
extern fn transformAABox(u32, u32, u32, u32, u32) void;
|
||||
|
||||
pub const module_name: [*:0]const u8 = "transform44";
|
||||
|
||||
var g_mutex: ?*anyopaque = null;
|
||||
@@ -37,7 +57,7 @@ pub fn isActive() bool {
|
||||
// Profiling state — unified dump every DUMP_FRAMES render passes
|
||||
// =============================================================================
|
||||
|
||||
const DUMP_FRAMES: u64 = 900; // ~15s at 60fps
|
||||
const DUMP_FRAMES: u64 = 450; // ~7.5s at 60fps
|
||||
|
||||
var prof = ProfState{};
|
||||
var t44_depth: u64 = 0; // recursion depth — survives resets
|
||||
@@ -236,11 +256,9 @@ fn transformDetour(this: u32, edx: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u
|
||||
t44_depth +|= 1;
|
||||
if (t44_depth > prof.t44_max_depth) prof.t44_max_depth = t44_depth;
|
||||
|
||||
if (ab_use_custom) {
|
||||
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
|
||||
} else {
|
||||
transform_hook.callOriginal(.{ this, edx, mat1, mat2, mat3, mat4 });
|
||||
}
|
||||
// bone_sse disabled while verifying from assembly
|
||||
_ = transformMatrix4x4_SSE;
|
||||
transform_hook.callOriginal(.{ this, edx, mat1, mat2, mat3, mat4 });
|
||||
|
||||
t44_depth -|= 1;
|
||||
const elapsed = rdtsc() - start;
|
||||
@@ -1085,6 +1103,114 @@ fn textlineDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.
|
||||
return ret;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Hooks: math_sse (UnitXP polyfill replacements)
|
||||
// Installed in lateInit() to clobber UnitXP's hooks. A/B tested.
|
||||
// =============================================================================
|
||||
|
||||
// Fn types: Ret3 = fastcall(ECX,EDX,stack) -> u32, Ret2 = fastcall(ECX,EDX) -> u32, etc.
|
||||
const MathFn3r = fn (u32, u32, u32) callconv(hook.cc.fastcall) u32;
|
||||
const MathFn2r = fn (u32, u32) callconv(hook.cc.fastcall) u32;
|
||||
const MathFn2v = fn (u32, u32) callconv(hook.cc.fastcall) void;
|
||||
const MathFn4r = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) u32;
|
||||
const MathFn5v = fn (u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) void;
|
||||
const MathFn2d = fn (u32, u32) callconv(hook.cc.fastcall) f64;
|
||||
const MathFn1d = fn (u32) callconv(hook.cc.fastcall) f64;
|
||||
const MathFn3d = fn (u32, u32, u32) callconv(hook.cc.fastcall) f64;
|
||||
const MathFn4v = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) void;
|
||||
|
||||
var math_vecMulMat4_hook: hook.Detour(MathFn3r) = .{}; // 0x7BCA80
|
||||
var math_matMulVec3_hook: hook.Detour(MathFn3r) = .{}; // 0x7BCAE0
|
||||
var math_quatMulMat4_hook: hook.Detour(MathFn3r) = .{}; // 0x7BCB40
|
||||
var math_vec3MulScalar_hook: hook.Detour(MathFn3r) = .{}; // 0x5F8CF0
|
||||
var math_vec3MulAssign_hook: hook.Detour(MathFn2r) = .{}; // 0x5132F0
|
||||
var math_applyTranslation_hook: hook.Detour(MathFn2r) = .{}; // 0x7BDC40
|
||||
var math_scaleByVec_hook: hook.Detour(MathFn2r) = .{}; // 0x7BDCA0
|
||||
var math_scaleByScalar_hook: hook.Detour(MathFn2v) = .{}; // 0x7BDD00
|
||||
var math_mul3x3_hook: hook.Detour(MathFn3r) = .{}; // 0x7BDFC0
|
||||
var math_rotMat3x3_hook: hook.Detour(MathFn4r) = .{}; // 0x7BE490
|
||||
var math_rotMat4x4_hook: hook.Detour(MathFn4r) = .{}; // 0x7BDB00
|
||||
var math_cross_hook: hook.Detour(MathFn3r) = .{}; // 0x672130
|
||||
var math_dot_hook: hook.Detour(MathFn2d) = .{}; // 0x602630
|
||||
var math_sqmag_hook: hook.Detour(MathFn1d) = .{}; // 0x4549F0
|
||||
var math_normalize_hook: hook.Detour(MathFn2v) = .{}; // 0x699330
|
||||
var math_evalPoly_hook: hook.Detour(MathFn3d) = .{}; // 0x453620
|
||||
var math_planeNormal_hook: hook.Detour(MathFn4v) = .{}; // 0x637480
|
||||
var math_transformAABox_hook: hook.Detour(MathFn5v) = .{}; // 0x6DC470
|
||||
|
||||
// A/B detour wrappers for each math_sse signature
|
||||
fn abDetour3r(comptime custom_fn: anytype, comptime h: anytype) *const MathFn3r {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) u32 {
|
||||
return if (ab_use_custom) custom_fn(a, b, c) else h.callOriginal(.{ a, b, c });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour2r(comptime custom_fn: anytype, comptime h: anytype) *const MathFn2r {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32) callconv(hook.cc.fastcall) u32 {
|
||||
return if (ab_use_custom) custom_fn(a, b) else h.callOriginal(.{ a, b });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour2v(comptime custom_fn: anytype, comptime h: anytype) *const MathFn2v {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32) callconv(hook.cc.fastcall) void {
|
||||
if (ab_use_custom) custom_fn(a, b) else h.callOriginal(.{ a, b });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour4r(comptime custom_fn: anytype, comptime h: anytype) *const MathFn4r {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) u32 {
|
||||
return if (ab_use_custom) custom_fn(a, b, c, d) else h.callOriginal(.{ a, b, c, d });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour5v(comptime custom_fn: anytype, comptime h: anytype) *const MathFn5v {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) void {
|
||||
if (ab_use_custom) custom_fn(a, b, c, d, e) else h.callOriginal(.{ a, b, c, d, e });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour2d(comptime custom_fn: anytype, comptime h: anytype) *const MathFn2d {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32) callconv(hook.cc.fastcall) f64 {
|
||||
return if (ab_use_custom) custom_fn(a, b) else h.callOriginal(.{ a, b });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour1d(comptime custom_fn: anytype, comptime h: anytype) *const MathFn1d {
|
||||
return &struct {
|
||||
fn f(a: u32) callconv(hook.cc.fastcall) f64 {
|
||||
return if (ab_use_custom) custom_fn(a) else h.callOriginal(.{a});
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour3d(comptime custom_fn: anytype, comptime h: anytype) *const MathFn3d {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) f64 {
|
||||
return if (ab_use_custom) custom_fn(a, b, c) else h.callOriginal(.{ a, b, c });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
fn abDetour4v(comptime custom_fn: anytype, comptime h: anytype) *const MathFn4v {
|
||||
return &struct {
|
||||
fn f(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) void {
|
||||
if (ab_use_custom) custom_fn(a, b, c, d) else h.callOriginal(.{ a, b, c, d });
|
||||
}
|
||||
}.f;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Hook: blit_hub (0x5a4f60)
|
||||
// __fastcall(ECX=int* vec2size, EDX=unknownFuncIndex,
|
||||
@@ -1098,19 +1224,97 @@ const BlitHubPtr = *const BlitHubFn;
|
||||
var blit_hub_hook: hook.Detour(BlitHubFn) = .{};
|
||||
var unitxp_blit: ?BlitHubPtr = null; // UnitXP's detour, captured before clobber
|
||||
|
||||
// =============================================================================
|
||||
// Hook: EnterCriticalSection spin count optimization (from UnitXP polyfill.cpp:474)
|
||||
// Sets SpinCount=4000 on critical sections with SpinCount=0, reducing kernel
|
||||
// transitions for short-held locks. The game initializes all CriticalSections
|
||||
// with SpinCount=0, causing immediate kernel waits on any contention.
|
||||
// =============================================================================
|
||||
|
||||
const WINAPI = std.builtin.CallingConvention.winapi;
|
||||
const CritSecFn = fn (u32) callconv(WINAPI) void;
|
||||
var critsec_hook: hook.Detour(CritSecFn) = .{};
|
||||
|
||||
extern "kernel32" fn GetModuleHandleA(name: [*:0]const u8) callconv(WINAPI) ?*anyopaque;
|
||||
extern "kernel32" fn GetProcAddress(module: *anyopaque, name: [*:0]const u8) callconv(WINAPI) ?*anyopaque;
|
||||
extern "kernel32" fn SetCriticalSectionSpinCount(cs: u32, spin: u32) callconv(WINAPI) u32;
|
||||
|
||||
fn critSecDetour(cs_ptr: u32) callconv(WINAPI) void {
|
||||
if (cs_ptr != 0 and (cs_ptr & 1) == 0) {
|
||||
// CRITICAL_SECTION.SpinCount is at offset +0x18 on Win32
|
||||
const spin_count = hook.readMem(u32, cs_ptr + 0x18);
|
||||
if (spin_count == 0) {
|
||||
_ = SetCriticalSectionSpinCount(cs_ptr, 4000);
|
||||
}
|
||||
}
|
||||
critsec_hook.callOriginal(.{cs_ptr});
|
||||
}
|
||||
|
||||
fn blitMemcpy(w: u32, h: u32, src: u32, src_pitch: u32, dst: u32, dst_pitch: u32, pixel_size: u32) void {
|
||||
const row_bytes = w * pixel_size;
|
||||
if (src_pitch == dst_pitch and row_bytes == src_pitch) {
|
||||
// Contiguous -- single memcpy
|
||||
const total = w * h * pixel_size;
|
||||
const s: [*]const u8 = @ptrFromInt(src);
|
||||
const d: [*]u8 = @ptrFromInt(dst);
|
||||
@memcpy(d[0..total], s[0..total]);
|
||||
} else {
|
||||
// Row-by-row
|
||||
var s = src;
|
||||
var d = dst;
|
||||
var y: u32 = 0;
|
||||
while (y < h) : (y += 1) {
|
||||
const sp: [*]const u8 = @ptrFromInt(s);
|
||||
const dp: [*]u8 = @ptrFromInt(d);
|
||||
@memcpy(dp[0..row_bytes], sp[0..row_bytes]);
|
||||
s += src_pitch;
|
||||
d += dst_pitch;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn blitHubDetour(vec2size: u32, func_index: u32, src_addr: u32, src_step: u32, src_fmt: u32, dst_addr: u32, dst_step: u32, dst_fmt: u32) callconv(hook.cc.fastcall) void {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
|
||||
// Ensure blit is initialized (game lazy-inits at 0xC0F558)
|
||||
const init_flag: *u32 = @ptrFromInt(0xC0F558);
|
||||
if (init_flag.* == 0) {
|
||||
hook.call(fn () callconv(hook.cc.fastcall) void, 0x5A4FC0, .{});
|
||||
init_flag.* = 1;
|
||||
}
|
||||
|
||||
const w = hook.readMem(u32, vec2size);
|
||||
const h = hook.readMem(u32, vec2size + 4);
|
||||
|
||||
const start = rdtsc();
|
||||
if (ab_use_custom) {
|
||||
// CUSTOM: call UnitXP's optimized blit (if present, else original)
|
||||
if (unitxp_blit) |uxp| {
|
||||
@call(.never_tail, uxp, .{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt });
|
||||
} else {
|
||||
blit_hub_hook.callOriginal(.{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt });
|
||||
if (ab_use_custom and func_index == 0 and src_fmt == dst_fmt) {
|
||||
// CUSTOM: memcpy fast path for matching formats
|
||||
switch (src_fmt) {
|
||||
1 => { blitMemcpy(w, h, src_addr, src_step, dst_addr, dst_step, 4); }, // ARGB 32bpp
|
||||
2, 4 => { blitMemcpy(w, h, src_addr, src_step, dst_addr, dst_step, 2); }, // RGB 16bpp
|
||||
5 => { // DXT compressed -- no pitch, w*h*4/8 bytes
|
||||
const wc = @max(w, 4);
|
||||
const hc = @max(h, 4);
|
||||
const len = wc * hc / 2; // 4 bits per pixel
|
||||
const s: [*]const u8 = @ptrFromInt(src_addr);
|
||||
const d: [*]u8 = @ptrFromInt(dst_addr);
|
||||
@memcpy(d[0..len], s[0..len]);
|
||||
},
|
||||
6, 7 => { // 8bpp formats -- no pitch, w*h bytes
|
||||
const wc = @max(w, 4);
|
||||
const hc = @max(h, 4);
|
||||
const len = wc * hc;
|
||||
const s: [*]const u8 = @ptrFromInt(src_addr);
|
||||
const d: [*]u8 = @ptrFromInt(dst_addr);
|
||||
@memcpy(d[0..len], s[0..len]);
|
||||
},
|
||||
else => {
|
||||
// Unknown format -- fall through to original
|
||||
blit_hub_hook.callOriginal(.{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt });
|
||||
},
|
||||
}
|
||||
} else {
|
||||
// BASELINE: true original function
|
||||
// BASELINE: original function
|
||||
blit_hub_hook.callOriginal(.{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt });
|
||||
}
|
||||
const elapsed = rdtsc() - start;
|
||||
@@ -1366,6 +1570,14 @@ pub fn installHooks() void {
|
||||
_ = matmul_hook.attach(0x7bc6a0, &matmulDetour);
|
||||
_ = textline_hook.attach(0x5ce0c0, &textlineDetour);
|
||||
|
||||
// CriticalSection spin count optimization (UnitXP polyfill)
|
||||
if (GetModuleHandleA("kernel32")) |k32| {
|
||||
if (GetProcAddress(k32, "EnterCriticalSection")) |ecs_addr| {
|
||||
_ = critsec_hook.attach(@intFromPtr(ecs_addr), &critSecDetour);
|
||||
log.print("critsec: SpinCount=4000 hook installed\n");
|
||||
}
|
||||
}
|
||||
|
||||
// TSC timer calibration (ported from VanillaFixes)
|
||||
timer_fix.init();
|
||||
const ti = timer_fix.getInfo();
|
||||
@@ -1405,6 +1617,93 @@ pub fn lateInit() void {
|
||||
hook.writeProtected(BLIT_ADDR, &.{ 0x55, 0x8B, 0xEC, 0xA1, 0x58, 0xF5, 0xC0, 0x00 });
|
||||
_ = blit_hub_hook.attach(BLIT_ADDR, &blitHubDetour);
|
||||
log.print("blit_hub: hooked (true original baseline)\n");
|
||||
|
||||
// math_sse replacements -- restore original prologues (clobber UnitXP), then hook
|
||||
// A/B tested: CUSTOM=our SSE, BASELINE=original x87
|
||||
// Set MATH_TEST_HOOK to 0 to disable all, 1-18 to enable only that one, 99 for all
|
||||
const MATH_TEST_HOOK: u32 = 99;
|
||||
const MathHook = struct { addr: u32, prologue: []const u8 };
|
||||
const math_hooks = [_]MathHook{
|
||||
.{ .addr = 0x7BCA80, .prologue = &.{ 0x55, 0x8b, 0xec, 0x56, 0x8b, 0x75, 0x08 } }, // 1: vecMulMat4
|
||||
.{ .addr = 0x7BCAE0, .prologue = &.{ 0x55, 0x8b, 0xec, 0xd9, 0x42, 0x28 } }, // 2: matMulVec3
|
||||
.{ .addr = 0x7BCB40, .prologue = &.{ 0x55, 0x8b, 0xec, 0x56, 0x8b, 0x75, 0x08 } }, // 3: quatMulMat4
|
||||
.{ .addr = 0x5F8CF0, .prologue = &.{ 0x55, 0x8b, 0xec, 0xd9, 0x45, 0x08 } }, // 4: vec3MulScalar
|
||||
.{ .addr = 0x5132F0, .prologue = &.{ 0x55, 0x8b, 0xec, 0xd9, 0x45, 0x08 } }, // 5: vec3MulAssign
|
||||
.{ .addr = 0x7BDC40, .prologue = &.{ 0x55, 0x8b, 0xec, 0x8b, 0x45, 0x08 } }, // 6: applyTranslation
|
||||
.{ .addr = 0x7BDCA0, .prologue = &.{ 0x55, 0x8b, 0xec, 0x8b, 0x45, 0x08 } }, // 7: scaleByVec
|
||||
.{ .addr = 0x7BDD00, .prologue = &.{ 0x55, 0x8b, 0xec, 0xd9, 0x45, 0x08 } }, // 8: scaleByScalar
|
||||
.{ .addr = 0x7BDFC0, .prologue = &.{ 0x55, 0x8b, 0xec, 0x51, 0xd9, 0x42, 0x1c } }, // 9: mul3x3
|
||||
.{ .addr = 0x7BE490, .prologue = &.{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x18 } }, // 10: rotMat3x3
|
||||
.{ .addr = 0x7BDB00, .prologue = &.{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x18 } }, // 11: rotMat4x4
|
||||
.{ .addr = 0x672130, .prologue = &.{ 0x55, 0x8b, 0xec, 0xd9, 0x02 } }, // 12: cross
|
||||
.{ .addr = 0x602630, .prologue = &.{ 0xd9, 0x41, 0x08, 0xd8, 0x4a, 0x08 } }, // 13: dot
|
||||
.{ .addr = 0x4549F0, .prologue = &.{ 0xd9, 0x41, 0x08, 0xd9, 0x41, 0x04 } }, // 14: sqmag
|
||||
.{ .addr = 0x699330, .prologue = &.{ 0xd9, 0x01, 0xd8, 0x1a, 0xdf, 0xe0 } }, // 15: normalize
|
||||
.{ .addr = 0x453620, .prologue = &.{ 0x55, 0x8b, 0xec, 0xd9, 0x02 } }, // 16: evalPoly
|
||||
.{ .addr = 0x637480, .prologue = &.{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x18 } }, // 17: planeNormal
|
||||
.{ .addr = 0x6DC470, .prologue = &.{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x0c } }, // 18: transformAABox
|
||||
};
|
||||
const math_detours = .{
|
||||
abDetour3r(&vecMulMat4_ColMajor, &math_vecMulMat4_hook),
|
||||
abDetour3r(&matMulVec3_RowMajor, &math_matMulVec3_hook),
|
||||
abDetour3r(&quatMulMat4, &math_quatMulMat4_hook),
|
||||
abDetour3r(&vec3MulScalar, &math_vec3MulScalar_hook),
|
||||
abDetour2r(&vec3MulAssign, &math_vec3MulAssign_hook),
|
||||
abDetour2r(&applyTranslationMatrix, &math_applyTranslation_hook),
|
||||
abDetour2r(&scaleMatrix3x3ByVector, &math_scaleByVec_hook),
|
||||
abDetour2v(&scaleMatrix3x3ByScalar, &math_scaleByScalar_hook),
|
||||
abDetour3r(&multiply3x3Matrix, &math_mul3x3_hook),
|
||||
abDetour4r(&createAxisAngleRotMat3x3, &math_rotMat3x3_hook),
|
||||
abDetour4r(&createAxisAngleRotMat4x4, &math_rotMat4x4_hook),
|
||||
abDetour3r(&crossProduct, &math_cross_hook),
|
||||
abDetour2d(&dotProduct, &math_dot_hook),
|
||||
abDetour1d(&squaredMagnitude, &math_sqmag_hook),
|
||||
abDetour2v(&vectorNormalize, &math_normalize_hook),
|
||||
abDetour3d(&evaluatePolynomial, &math_evalPoly_hook),
|
||||
abDetour4v(&calculatePlaneNormal, &math_planeNormal_hook),
|
||||
abDetour5v(&transformAABox, &math_transformAABox_hook),
|
||||
};
|
||||
_ = math_detours; // used below via indexed access
|
||||
const math_hook_ptrs = .{
|
||||
&math_vecMulMat4_hook, &math_matMulVec3_hook, &math_quatMulMat4_hook,
|
||||
&math_vec3MulScalar_hook, &math_vec3MulAssign_hook, &math_applyTranslation_hook,
|
||||
&math_scaleByVec_hook, &math_scaleByScalar_hook, &math_mul3x3_hook,
|
||||
&math_rotMat3x3_hook, &math_rotMat4x4_hook, &math_cross_hook,
|
||||
&math_dot_hook, &math_sqmag_hook, &math_normalize_hook,
|
||||
&math_evalPoly_hook, &math_planeNormal_hook, &math_transformAABox_hook,
|
||||
};
|
||||
_ = math_hook_ptrs; // used conceptually
|
||||
var math_count: u32 = 0;
|
||||
inline for (math_hooks, 0..) |mh, i| {
|
||||
const idx = i + 1;
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == idx) {
|
||||
hook.writeProtected(mh.addr, mh.prologue);
|
||||
_ = comptime blk: {
|
||||
_ = i;
|
||||
break :blk {};
|
||||
};
|
||||
}
|
||||
}
|
||||
// Can't do heterogeneous attach in inline for, so do them individually gated by MATH_TEST_HOOK
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 1) { hook.writeProtected(0x7BCA80, math_hooks[0].prologue); _ = math_vecMulMat4_hook.attach(0x7BCA80, abDetour3r(&vecMulMat4_ColMajor, &math_vecMulMat4_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 2) { hook.writeProtected(0x7BCAE0, math_hooks[1].prologue); _ = math_matMulVec3_hook.attach(0x7BCAE0, abDetour3r(&matMulVec3_RowMajor, &math_matMulVec3_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 3) { hook.writeProtected(0x7BCB40, math_hooks[2].prologue); _ = math_quatMulMat4_hook.attach(0x7BCB40, abDetour3r(&quatMulMat4, &math_quatMulMat4_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 4) { hook.writeProtected(0x5F8CF0, math_hooks[3].prologue); _ = math_vec3MulScalar_hook.attach(0x5F8CF0, abDetour3r(&vec3MulScalar, &math_vec3MulScalar_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 5) { hook.writeProtected(0x5132F0, math_hooks[4].prologue); _ = math_vec3MulAssign_hook.attach(0x5132F0, abDetour2r(&vec3MulAssign, &math_vec3MulAssign_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 6) { hook.writeProtected(0x7BDC40, math_hooks[5].prologue); _ = math_applyTranslation_hook.attach(0x7BDC40, abDetour2r(&applyTranslationMatrix, &math_applyTranslation_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 7) { hook.writeProtected(0x7BDCA0, math_hooks[6].prologue); _ = math_scaleByVec_hook.attach(0x7BDCA0, abDetour2r(&scaleMatrix3x3ByVector, &math_scaleByVec_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 8) { hook.writeProtected(0x7BDD00, math_hooks[7].prologue); _ = math_scaleByScalar_hook.attach(0x7BDD00, abDetour2v(&scaleMatrix3x3ByScalar, &math_scaleByScalar_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 9) { hook.writeProtected(0x7BDFC0, math_hooks[8].prologue); _ = math_mul3x3_hook.attach(0x7BDFC0, abDetour3r(&multiply3x3Matrix, &math_mul3x3_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 10) { hook.writeProtected(0x7BE490, math_hooks[9].prologue); _ = math_rotMat3x3_hook.attach(0x7BE490, abDetour4r(&createAxisAngleRotMat3x3, &math_rotMat3x3_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 11) { hook.writeProtected(0x7BDB00, math_hooks[10].prologue); _ = math_rotMat4x4_hook.attach(0x7BDB00, abDetour4r(&createAxisAngleRotMat4x4, &math_rotMat4x4_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 12) { hook.writeProtected(0x672130, math_hooks[11].prologue); _ = math_cross_hook.attach(0x672130, abDetour3r(&crossProduct, &math_cross_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 13) { hook.writeProtected(0x602630, math_hooks[12].prologue); _ = math_dot_hook.attach(0x602630, abDetour2d(&dotProduct, &math_dot_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 14) { hook.writeProtected(0x4549F0, math_hooks[13].prologue); _ = math_sqmag_hook.attach(0x4549F0, abDetour1d(&squaredMagnitude, &math_sqmag_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 15) { hook.writeProtected(0x699330, math_hooks[14].prologue); _ = math_normalize_hook.attach(0x699330, abDetour2v(&vectorNormalize, &math_normalize_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 16) { hook.writeProtected(0x453620, math_hooks[15].prologue); _ = math_evalPoly_hook.attach(0x453620, abDetour3d(&evaluatePolynomial, &math_evalPoly_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 17) { hook.writeProtected(0x637480, math_hooks[16].prologue); _ = math_planeNormal_hook.attach(0x637480, abDetour4v(&calculatePlaneNormal, &math_planeNormal_hook)); math_count += 1; }
|
||||
if (MATH_TEST_HOOK == 99 or MATH_TEST_HOOK == 18) { hook.writeProtected(0x6DC470, math_hooks[17].prologue); _ = math_transformAABox_hook.attach(0x6DC470, abDetour5v(&transformAABox, &math_transformAABox_hook)); math_count += 1; }
|
||||
log.fmt("math_sse: {d}/18 hooks installed (MATH_TEST_HOOK={d})\n", .{ math_count, MATH_TEST_HOOK });
|
||||
}
|
||||
|
||||
pub fn removeHooks() void {
|
||||
@@ -1450,6 +1749,25 @@ pub fn removeHooks() void {
|
||||
matmul_hook.detach();
|
||||
textline_hook.detach();
|
||||
blit_hub_hook.detach();
|
||||
critsec_hook.detach();
|
||||
math_vecMulMat4_hook.detach();
|
||||
math_matMulVec3_hook.detach();
|
||||
math_quatMulMat4_hook.detach();
|
||||
math_vec3MulScalar_hook.detach();
|
||||
math_vec3MulAssign_hook.detach();
|
||||
math_applyTranslation_hook.detach();
|
||||
math_scaleByVec_hook.detach();
|
||||
math_scaleByScalar_hook.detach();
|
||||
math_mul3x3_hook.detach();
|
||||
math_rotMat3x3_hook.detach();
|
||||
math_rotMat4x4_hook.detach();
|
||||
math_cross_hook.detach();
|
||||
math_dot_hook.detach();
|
||||
math_sqmag_hook.detach();
|
||||
math_normalize_hook.detach();
|
||||
math_evalPoly_hook.detach();
|
||||
math_planeNormal_hook.detach();
|
||||
math_transformAABox_hook.detach();
|
||||
log.close();
|
||||
mod_mutex.release(&g_mutex);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user