bone_sse_ref: fix M2 black screen — 5 bugs found via asm comparison
Assembly-level comparison of compiled REF against original 0x714260 revealed: 1. Billboard cross product sign error (types 0x10/0x20): computed +cross instead of -cross for components 0/1, corrupting billboard bone matrices 2. colorAnimLoop wrong count field: read model_hdr+0x6C instead of +0x64 3. colorAnimLoop wrong gate offset: checked anim_data+0x04 instead of +0x0C 4. Timestamp delta guard inverted: REF guarded on cur_ts!=0 and always wrote to this+0x4C; original guards on this+0x4C!=0 first and never seeds the field (something else initializes it) 5. Section 5 emitter_ctx cached instead of re-read after matMul call Also: build REF with x87-only target (subtract SSE/SSE2 features) to match original's FLD/FMUL/FSTP codegen, and use callVec3SqMag for all magnitude computations instead of inline SSE math. Remaining known issues (not yet fixed): - texAnimLoop alpha track missing crossfade blend - colorAnimLoop missing crossfade blend - Missing word animation section (model_hdr+0x6C/0x70) - Bisect infrastructure and diagnostic code still present (test scaffolding)
This commit is contained in:
@@ -71,11 +71,20 @@ pub fn build(b: *std.Build) void {
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
// REF uses x87-only target to match original game code structure.
|
||||
// The global target has SSE/SSE2 which generates movss/mulss;
|
||||
// the original at 0x714260 uses pure x87 (FLD/FMUL/FSTP).
|
||||
const ref_target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .windows,
|
||||
.abi = .msvc,
|
||||
.cpu_features_sub = std.Target.x86.featureSet(&.{ .sse, .sse2 }),
|
||||
});
|
||||
const bone_sse_ref_obj = b.addObject(.{
|
||||
.name = "bone_sse_ref",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/transform44/bone_sse_reference.zig"),
|
||||
.target = target,
|
||||
.target = ref_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
|
||||
@@ -15,6 +15,11 @@ const V4 = @Vector(4, f32);
|
||||
export var dbg_fpu_logged: u32 = 0;
|
||||
export var dbg_fpu_value: u16 = 0;
|
||||
|
||||
// BISECT: stop REF after this section (0 = run all, 7 = stop after bone loop, etc.)
|
||||
// Detour calls original trampoline after REF to provide any missing side effects.
|
||||
// Test sequence: 7 → renders? narrows to 8-13. Black? issue in 1-7 or structural.
|
||||
export var bisect_stop_section: u32 = 0;
|
||||
|
||||
|
||||
// =============================================================================
|
||||
// SceneObject field offsets — assembly-verified from [EBX+N] in transformMatrix4x4
|
||||
@@ -968,14 +973,16 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
|
||||
// =========================================================================
|
||||
// Section 5: child_objects_padding (len_sq of world transform translation)
|
||||
// Assembly re-reads emitter_ctx from this+0x1CC AFTER matMul (0x7143A0).
|
||||
// =========================================================================
|
||||
if (emitter_ctx == 0 or (ru8(emitter_ctx + 4) & 1) != 0) {
|
||||
const emitter_ctx_5 = ru32(this + SO.emitter_ctx);
|
||||
if (emitter_ctx_5 == 0 or (ru8(emitter_ctx_5 + 4) & 1) != 0) {
|
||||
const wx = rf32(this + SO.world_xform + 8 * 4); // [8]
|
||||
const wy = rf32(this + SO.world_xform + 9 * 4); // [9]
|
||||
const wz = rf32(this + SO.world_xform + 10 * 4); // [10]
|
||||
wu32(this + SO.child_padding, fbits(wx * wx + wy * wy + wz * wz));
|
||||
} else {
|
||||
wu32(this + SO.child_padding, ru32(emitter_ctx + 0x84));
|
||||
wu32(this + SO.child_padding, ru32(emitter_ctx_5 + 0x84));
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
@@ -998,16 +1005,17 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
};
|
||||
|
||||
// Timestamp delta tracking
|
||||
// Assembly guard: if (anim_ctx != 0 AND anim_ctx->timestamp != 0)
|
||||
// NOT guarded on the stored value at this+0x4C — must always write on first frame
|
||||
// Assembly (0x7143EE-0x71451C): outer guard is this+0x4C != 0 (NOT anim_ctx).
|
||||
// If stored value is 0, does NOTHING — never writes, never computes delta.
|
||||
// Something else must initialize this+0x4C; we must NOT seed it ourselves.
|
||||
var time_delta_val: u32 = 0;
|
||||
const cur_ts = ru32(anim_ctx + 0x0C);
|
||||
if (cur_ts != 0) {
|
||||
const sdb = ru32(this + SO.search_data_base);
|
||||
if (sdb != 0) {
|
||||
const sdb = ru32(this + SO.search_data_base);
|
||||
if (sdb != 0) {
|
||||
const cur_ts = ru32(anim_ctx + 0x0C);
|
||||
if (cur_ts != 0) {
|
||||
time_delta_val = cur_ts -% sdb;
|
||||
wu32(this + SO.search_data_base, cur_ts);
|
||||
}
|
||||
wu32(this + SO.search_data_base, cur_ts);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
@@ -1272,11 +1280,13 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
local_mat[10] = n2[2];
|
||||
} else if (bb_type == 4) {
|
||||
// Spherical billboard — inherit camera rotation with scale preservation
|
||||
// All sqmag computations MUST call game's vec3SqMag (0x4549F0)
|
||||
const cam0 = [3]f32{ rf32(this + SO.bb_row0), rf32(this + SO.bb_row0 + 4), rf32(this + SO.bb_row0 + 8) };
|
||||
const cam_len_sq0 = cam0[0] * cam0[0] + cam0[1] * cam0[1] + cam0[2] * cam0[2];
|
||||
const cam_len_sq0 = callVec3SqMag(this + SO.bb_row0);
|
||||
var s0: f32 = 1.0;
|
||||
if (cam_len_sq0 > rf32(0x0080c5c8)) {
|
||||
const mat_len_sq0 = local_mat[0] * local_mat[0] + local_mat[1] * local_mat[1] + local_mat[2] * local_mat[2];
|
||||
var tmp0 = [3]f32{ local_mat[0], local_mat[1], local_mat[2] };
|
||||
const mat_len_sq0 = callVec3SqMag(@intFromPtr(&tmp0));
|
||||
s0 = @sqrt(mat_len_sq0 / cam_len_sq0);
|
||||
}
|
||||
local_mat[0] = s0 * cam0[0];
|
||||
@@ -1286,10 +1296,11 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
const wt0 = rf32(this + SO.world_xform + 0 * 4);
|
||||
const wt1 = rf32(this + SO.world_xform + 1 * 4);
|
||||
const wt2 = rf32(this + SO.world_xform + 2 * 4);
|
||||
const wt_len_sq = wt0 * wt0 + wt1 * wt1 + wt2 * wt2;
|
||||
const wt_len_sq = callVec3SqMag(this + SO.world_xform);
|
||||
var s1: f32 = 1.0;
|
||||
if (wt_len_sq > rf32(0x0080c5c8)) {
|
||||
const mat_len_sq1 = local_mat[4] * local_mat[4] + local_mat[5] * local_mat[5] + local_mat[6] * local_mat[6];
|
||||
var tmp1 = [3]f32{ local_mat[4], local_mat[5], local_mat[6] };
|
||||
const mat_len_sq1 = callVec3SqMag(@intFromPtr(&tmp1));
|
||||
s1 = @sqrt(mat_len_sq1 / wt_len_sq);
|
||||
}
|
||||
local_mat[4] = s1 * wt0;
|
||||
@@ -1299,10 +1310,11 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
const wt4 = rf32(this + SO.world_xform + 4 * 4);
|
||||
const wt5 = rf32(this + SO.world_xform + 5 * 4);
|
||||
const wt6 = rf32(this + SO.world_xform + 6 * 4);
|
||||
const wt_len_sq2 = wt4 * wt4 + wt5 * wt5 + wt6 * wt6;
|
||||
const wt_len_sq2 = callVec3SqMag(this + SO.world_xform + 16);
|
||||
var s2: f32 = 1.0;
|
||||
if (wt_len_sq2 > rf32(0x0080c5c8)) {
|
||||
const mat_len_sq2 = local_mat[8] * local_mat[8] + local_mat[9] * local_mat[9] + local_mat[10] * local_mat[10];
|
||||
var tmp2 = [3]f32{ local_mat[8], local_mat[9], local_mat[10] };
|
||||
const mat_len_sq2 = callVec3SqMag(@intFromPtr(&tmp2));
|
||||
s2 = @sqrt(mat_len_sq2 / wt_len_sq2);
|
||||
}
|
||||
local_mat[8] = s2 * wt4;
|
||||
@@ -1437,10 +1449,11 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
const out_off = bone_idx * 0x40;
|
||||
const om = bone_out_base + out_off; // output matrix
|
||||
|
||||
// Compute scale lengths (sqrt of row length_sq for each row)
|
||||
const scale_len0 = @sqrt(rf32(om + 0x08) * rf32(om + 0x08) + rf32(om + 0x04) * rf32(om + 0x04) + rf32(om) * rf32(om));
|
||||
const scale_len1 = @sqrt(rf32(om + 0x18) * rf32(om + 0x18) + rf32(om + 0x14) * rf32(om + 0x14) + rf32(om + 0x10) * rf32(om + 0x10));
|
||||
const scale_len2 = @sqrt(rf32(om + 0x28) * rf32(om + 0x28) + rf32(om + 0x24) * rf32(om + 0x24) + rf32(om + 0x20) * rf32(om + 0x20));
|
||||
// Compute scale lengths — MUST call game's vec3SqMag (0x4549F0), not inline
|
||||
// Assembly: LEA ECX,[stack_vec3]; CALL 0x4549F0; FSQRT
|
||||
const scale_len0 = @sqrt(callVec3SqMag(om));
|
||||
const scale_len1 = @sqrt(callVec3SqMag(om + 0x10));
|
||||
const scale_len2 = @sqrt(callVec3SqMag(om + 0x20));
|
||||
|
||||
// Compute translated pivot position through the output matrix
|
||||
// local_a8 = pivot * matrix + translation
|
||||
@@ -1513,14 +1526,14 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
wf32(om + 0x18, 0);
|
||||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||||
_ = n1;
|
||||
// row2 = cross(row0, row1)
|
||||
wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14));
|
||||
wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18));
|
||||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||||
// row2 = -cross(row0, row1) — assembly uses negated cross product
|
||||
wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18));
|
||||
wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10));
|
||||
wf32(om + 0x28, rf32(om) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x10));
|
||||
},
|
||||
0x20 => {
|
||||
// Type 32: normalize row1, set row0={-row1.y, row1.x, 0}, normalize,
|
||||
// row2 = cross(row0, row1)
|
||||
// row2 = -cross(row0, row1)
|
||||
const n1 = normalizeVec3InPlace(om + 0x10);
|
||||
_ = n1;
|
||||
wf32(om, -rf32(om + 0x14));
|
||||
@@ -1528,10 +1541,10 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
wf32(om + 0x08, 0);
|
||||
const n0 = normalizeVec3InPlace(om);
|
||||
_ = n0;
|
||||
// row2 = cross(row0, row1)
|
||||
wf32(om + 0x20, rf32(om + 0x04) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x14));
|
||||
wf32(om + 0x24, rf32(om + 0x08) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x18));
|
||||
wf32(om + 0x28, rf32(om + 0x04) * rf32(om + 0x10) - rf32(om) * rf32(om + 0x14));
|
||||
// row2 = -cross(row0, row1) — assembly uses negated cross product
|
||||
wf32(om + 0x20, rf32(om + 0x08) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x18));
|
||||
wf32(om + 0x24, rf32(om) * rf32(om + 0x18) - rf32(om + 0x08) * rf32(om + 0x10));
|
||||
wf32(om + 0x28, rf32(om) * rf32(om + 0x14) - rf32(om + 0x04) * rf32(om + 0x10));
|
||||
},
|
||||
0x40 => {
|
||||
// Type 64: normalize row2, set row1={row2.y, -row2.x, 0}, normalize,
|
||||
@@ -1594,20 +1607,28 @@ export fn transformMatrix4x4_REF(this: u32, mat1: u32, mat2: u32, mat3: u32, mat
|
||||
// findInterpIdx + lerp + crossfade blend.
|
||||
// =========================================================================
|
||||
|
||||
// BISECT: stop after section 7 (bone loop)
|
||||
if (bisect_stop_section == 7) return;
|
||||
|
||||
// Section 8: Texture animation loop
|
||||
texAnimLoop(this, model_hdr);
|
||||
if (bisect_stop_section == 8) return;
|
||||
|
||||
// Section 9: Color animation loop
|
||||
colorAnimLoop(this, model_hdr);
|
||||
if (bisect_stop_section == 9) return;
|
||||
|
||||
// Section 10: Bone keyframe processing
|
||||
boneKeyframeLoop(this, model_hdr);
|
||||
if (bisect_stop_section == 10) return;
|
||||
|
||||
// Section 11: Particle emitter loops
|
||||
particleLoops(this, model_hdr);
|
||||
if (bisect_stop_section == 11) return;
|
||||
|
||||
// Section 12: Attachment recursion
|
||||
attachmentRecursion(this, model_hdr, bone_out_base);
|
||||
if (bisect_stop_section == 12) return;
|
||||
|
||||
// =========================================================================
|
||||
// Section 13: Sync update
|
||||
@@ -1665,9 +1686,9 @@ fn texAnimLoop(this: u32, model_hdr: u32) void {
|
||||
}
|
||||
|
||||
fn colorAnimLoop(this: u32, model_hdr: u32) void {
|
||||
// Assembly: entry gate at model_hdr+0x64, loop bound at model_hdr+0x6C
|
||||
if (ru32(model_hdr + 0x64) == 0) return;
|
||||
const count = ru32(model_hdr + 0x6C); // loop bound from assembly 0x715F0A
|
||||
// Assembly: model_hdr+0x64 is both entry gate AND loop count
|
||||
const count = ru32(model_hdr + 0x64);
|
||||
if (count == 0) return;
|
||||
const data_base = ru32(model_hdr + 0x68);
|
||||
const bone_rt_base = ru32(this + SO.bone_rt_base);
|
||||
const out_base = ru32(this + SO.color_anim_out);
|
||||
@@ -1682,7 +1703,7 @@ fn colorAnimLoop(this: u32, model_hdr: u32) void {
|
||||
}) {
|
||||
const anim_data = data_base + data_off;
|
||||
const output = out_base + out_off;
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(anim_data + 0x04)) {
|
||||
if (ru32(this + SO.anim_frame_ctr) < ru32(anim_data + 0x0C)) {
|
||||
findInterpIdx(this, ru32(bone_rt_base + BR.prim_time), ru32(bone_rt_base + BR.prim_track), anim_data, output);
|
||||
// Short value interpolation via game's getIndexOffset/setShortValue
|
||||
const mode = ri16(anim_data);
|
||||
|
||||
@@ -23,6 +23,7 @@ extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void;
|
||||
extern fn multiplyMatrix4x4(u32, u32, u32) u32;
|
||||
extern fn transformMatrix4x4_SSE(u32, u32, u32, u32, u32) void;
|
||||
extern fn transformMatrix4x4_REF(u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern var bisect_stop_section: u32;
|
||||
|
||||
pub const module_name: [*:0]const u8 = "transform44";
|
||||
|
||||
@@ -257,7 +258,15 @@ fn diagCompare(label: [*:0]const u8, snap: []const u8, live: u32, len: u32) void
|
||||
}
|
||||
}
|
||||
|
||||
// Test A: stripped detour — pure passthrough to REF, no profiling, no dispatch logic.
|
||||
// If black → issue is REF code structure. If renders → detour overhead was the problem.
|
||||
const DIRECT_REF_TEST = true;
|
||||
|
||||
fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(hook.cc.thiscall) void {
|
||||
if (DIRECT_REF_TEST) {
|
||||
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
|
||||
return;
|
||||
}
|
||||
const start = rdtsc();
|
||||
|
||||
const model_data = hook.readMem(u32, this + 0x10);
|
||||
@@ -286,7 +295,7 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
|
||||
|
||||
if (teardown_active) {
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
} else if (!is_early and diag_count < DIAG_MAX and t44_depth == 1) {
|
||||
} else if (false and !is_early and diag_count < DIAG_MAX and t44_depth == 1) {
|
||||
// --- DIAGNOSTIC (disabled): run original, snapshot, run REF, compare ---
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
|
||||
@@ -503,11 +512,12 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
|
||||
} else {
|
||||
// FPU state comparison: capture full x87 state after original vs REF
|
||||
if (diag_count >= DIAG_MAX and diag_count < DIAG_MAX + 3 and t44_depth == 1 and !is_early) {
|
||||
// Run original, capture FPU state
|
||||
// Run original, capture FPU + MXCSR state
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
var fpu_orig: [108]u8 align(16) = undefined;
|
||||
asm volatile ("fnsave (%[p])\n\tfrstor (%[p])"
|
||||
:: [p] "r" (@intFromPtr(&fpu_orig))
|
||||
var mxcsr_orig: u32 = 0;
|
||||
asm volatile ("fnsave (%[p])\n\tfrstor (%[p])\n\tstmxcsr (%[m])"
|
||||
:: [p] "r" (@intFromPtr(&fpu_orig)), [m] "r" (@intFromPtr(&mxcsr_orig))
|
||||
: "memory"
|
||||
);
|
||||
|
||||
@@ -515,8 +525,9 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
|
||||
@as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0;
|
||||
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
|
||||
var fpu_ref: [108]u8 align(16) = undefined;
|
||||
asm volatile ("fnsave (%[p])\n\tfrstor (%[p])"
|
||||
:: [p] "r" (@intFromPtr(&fpu_ref))
|
||||
var mxcsr_ref: u32 = 0;
|
||||
asm volatile ("fnsave (%[p])\n\tfrstor (%[p])\n\tstmxcsr (%[m])"
|
||||
:: [p] "r" (@intFromPtr(&fpu_ref)), [m] "r" (@intFromPtr(&mxcsr_ref))
|
||||
: "memory"
|
||||
);
|
||||
|
||||
@@ -530,6 +541,12 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
|
||||
const tw_r = @as(*align(1) const u32, @ptrFromInt(@intFromPtr(&fpu_ref) + 8)).*;
|
||||
log.fmt(" FPU orig: CW=0x{x:0>4} SW=0x{x:0>4} TW=0x{x:0>4}", .{ cw_o & 0xFFFF, sw_o & 0xFFFF, tw_o & 0xFFFF });
|
||||
log.fmt(" FPU ref: CW=0x{x:0>4} SW=0x{x:0>4} TW=0x{x:0>4}", .{ cw_r & 0xFFFF, sw_r & 0xFFFF, tw_r & 0xFFFF });
|
||||
// MXCSR comparison — SSE control/status, never checked before
|
||||
if (mxcsr_orig != mxcsr_ref) {
|
||||
log.fmt(" *** MXCSR DIFF: orig=0x{x:0>8} ref=0x{x:0>8} (xor=0x{x:0>8})", .{ mxcsr_orig, mxcsr_ref, mxcsr_orig ^ mxcsr_ref });
|
||||
} else {
|
||||
log.fmt(" MXCSR match: 0x{x:0>8}", .{mxcsr_orig});
|
||||
}
|
||||
// Dump ST0-ST7 (10 bytes each, starting at offset 28)
|
||||
var sti: u32 = 0;
|
||||
while (sti < 8) : (sti += 1) {
|
||||
@@ -546,7 +563,13 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
|
||||
}
|
||||
diag_count += 1;
|
||||
} else {
|
||||
// BISECT MODE: REF runs up to bisect_stop_section, then original provides rest
|
||||
transformMatrix4x4_REF(this, mat1, mat2, mat3, mat4);
|
||||
if (bisect_stop_section != 0) {
|
||||
// REF returned early — clear sync so original doesn't early-exit, then run original
|
||||
@as(*align(1) u32, @ptrFromInt(this + 0x40)).* = 0;
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1700,17 +1723,59 @@ pub fn installHooks() void {
|
||||
if (!g_is_hook_owner) return;
|
||||
|
||||
log = logging.Logger.open(module_name, .both);
|
||||
_ = transform_hook.attach(0x714260, &transformDetour);
|
||||
original_trampoline = @intCast(transform_hook.inner.trampoline);
|
||||
|
||||
// BINARY PATCH TEST: copy REF's x87 code directly over original at 0x714260.
|
||||
// No hook — game calls REF's code at the original address. Return addresses
|
||||
// from game function calls will point into game .text, not DLL .text.
|
||||
// If renders → game functions check return addresses (anti-cheat).
|
||||
// If black → issue is something else entirely.
|
||||
const BINARY_PATCH_TEST = false;
|
||||
if (BINARY_PATCH_TEST) {
|
||||
const ref_addr = @intFromPtr(&transformMatrix4x4_REF);
|
||||
const ref_size: u32 = 0x2320; // main function ends at particleLoops start (includes epilogue + FPU stubs)
|
||||
const game_addr: u32 = 0x714260;
|
||||
|
||||
// Copy REF's main function over the original (handles VirtualProtect)
|
||||
const src: [*]const u8 = @ptrFromInt(ref_addr);
|
||||
hook.writeProtected(game_addr, src[0..ref_size]);
|
||||
|
||||
// INT3-fill the remaining original bytes (crash on overrun)
|
||||
var cc_buf: [512]u8 = .{0xCC} ** 512;
|
||||
const remaining = @as(u32, 17703) - ref_size;
|
||||
var filled: u32 = 0;
|
||||
while (filled < remaining) {
|
||||
const chunk = @min(512, remaining - filled);
|
||||
hook.writeProtected(game_addr + ref_size + filled, cc_buf[0..chunk]);
|
||||
filled += chunk;
|
||||
}
|
||||
|
||||
// Fix up 2 relative CALL (E8) instructions to helper functions.
|
||||
// In the DLL, they resolve from ref_addr+site to DLL helper addresses.
|
||||
// After copy to game_addr, we recompute the rel32 to reach the same targets.
|
||||
const e8_sites = [_]u32{ 0x22CE, 0x22E6 };
|
||||
for (e8_sites) |site| {
|
||||
// Read the resolved rel32 from DLL copy to get absolute target
|
||||
const dll_call_addr = ref_addr + site;
|
||||
const target = hook.rel32Target(dll_call_addr);
|
||||
// Write new rel32 for the game-address copy
|
||||
const game_call_addr = game_addr + site;
|
||||
var rel_buf: [4]u8 = undefined;
|
||||
hook.writeRel32(&rel_buf, game_call_addr + 1, target);
|
||||
hook.writeProtected(game_call_addr + 1, &rel_buf);
|
||||
}
|
||||
|
||||
log.fmt("BINARY PATCH: {d} bytes REF@0x{x:0>8} -> 0x{x:0>8}, {d} E8 fixups\n", .{ ref_size, ref_addr, game_addr, e8_sites.len });
|
||||
} else {
|
||||
_ = transform_hook.attach(0x714260, &transformDetour);
|
||||
original_trampoline = @intCast(transform_hook.inner.trampoline);
|
||||
}
|
||||
_ = teardown_hook.attach(0x491180, &teardownDetour);
|
||||
_ = render_frame_hook.attach(0x707680, &renderFrameDetour);
|
||||
_ = exec_render_pass_hook.attach(0x708900, &execRenderPassDetour);
|
||||
_ = world_update_hook.attach(0x482EA0, &worldUpdateDetour);
|
||||
_ = teardown_hook.attach(0x491180, &teardownDetour);
|
||||
_ = render_quads_hook.attach(0x76FB00, &renderQuadsDetour);
|
||||
_ = movement_hook.attach(0x616620, &movementDetour);
|
||||
// _ = interp_kf_hook.attach(0x713ea0, &interpKfDetour); // disabled — pure passthrough, REF calls 0x713ea0 directly
|
||||
|
||||
// Perf-identified hotspot hooks
|
||||
// _ = interp_kf_hook.attach(0x713ea0, &interpKfDetour); // disabled: pure passthrough
|
||||
_ = clip_hook.attach(0x6318c0, &clipDetour);
|
||||
_ = glyph_hook.attach(0x5ca2d0, &glyphDetour);
|
||||
_ = particle_hook.attach(0x7b2a50, &particleDetour);
|
||||
@@ -1731,7 +1796,7 @@ pub fn installHooks() void {
|
||||
_ = activep_hook.attach(0x7b5a10, &activepDetour);
|
||||
_ = cbiter_hook.attach(0x404130, &cbiterDetour);
|
||||
_ = findguid_hook.attach(0x464890, &findguidDetour);
|
||||
_ = raytri2_hook.attach(0x632700, &raytri2Detour);
|
||||
// _ = raytri2_hook.attach(0x632700, &raytri2Detour);
|
||||
_ = drawbatch_hook.attach(0x70cb30, &drawbatchDetour);
|
||||
_ = findlua_hook.attach(0x702000, &findluaDetour);
|
||||
_ = spritequad_hook.attach(0x5a0f50, &spritequadDetour);
|
||||
@@ -1760,6 +1825,9 @@ pub fn installHooks() void {
|
||||
|
||||
// blit_hub installed in lateInit() to clobber UnitXP's hook
|
||||
log.print("transform44: 39 profiling hooks installed (blit_hub deferred)\n");
|
||||
if (bisect_stop_section != 0) {
|
||||
log.fmt(" BISECT MODE: REF stops after section {d}, then original trampoline\n", .{bisect_stop_section});
|
||||
}
|
||||
}
|
||||
|
||||
/// Called from engineInitDetour — after UnitXP has hooked blit_hub.
|
||||
|
||||
Reference in New Issue
Block a user