diff --git a/src/transform44/clip_sse.zig b/src/transform44/clip_sse.zig new file mode 100644 index 0000000..45285f4 --- /dev/null +++ b/src/transform44/clip_sse.zig @@ -0,0 +1,376 @@ +//! SSE-optimized collision math — compiled ReleaseFast even in Debug builds. +//! +//! Functions are exported and called via `extern fn` from the Debug transform44 module. +//! This separation is required because Zig module imports inherit the parent's optimization +//! level, so only a separate `addObject(.{ .optimize = .ReleaseFast })` gets real SSE codegen. + +const V4 = @Vector(4, f32); + +const CLIP_EPSILON: f32 = @bitCast(@as(u32, 0x3ab60b61)); // +0.00139, global at 0x80dfec +const MOVEMENT_EPSILON: f32 = @bitCast(@as(u32, 0x35800000)); // 9.54e-7, global at 0x8026bc + +export fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void { + const plane: [*]const f32 = @ptrFromInt(plane_addr); + const poly: [*]f32 = @ptrFromInt(poly_addr); + const new_attrib: f32 = @bitCast(attrib_bits); + + // Vertex count at poly+0xF0 (byte offset 240, float offset 60) + const count_ptr: *align(1) u32 = @ptrCast(poly + 60); + const n = count_ptr.*; + if (n == 0) return; + + // Load plane as 4-wide vector for SSE dot product + const pv: @Vector(4, f32) = .{ plane[0], plane[1], plane[2], plane[3] }; + + // --- Phase 1: Compute signed distances --- + var dists: [15]f32 = undefined; + var min_dist: f32 = @bitCast(@as(u32, 0x7f7fffff)); // FLT_MAX + var max_dist: f32 = @bitCast(@as(u32, 0xff7fffff)); // -FLT_MAX + + var i: u32 = 0; + while (i < n) : (i += 1) { + const b = i * 3; + const v: @Vector(4, f32) = .{ poly[b], poly[b + 1], poly[b + 2], 1.0 }; + const d = -@reduce(.Add, v * pv); + dists[i] = d; + if (d < min_dist) min_dist = d; + if (d > max_dist) max_dist = d; + } + + // --- Phase 2: Early exits (exact same thresholds as original) --- + if (min_dist > -CLIP_EPSILON) return; // all inside + if (max_dist < CLIP_EPSILON) { + count_ptr.* = 0; // all outside + return; + } + + // --- Phase 3: Copy polygon to local buffers --- + const attribs: [*]f32 = poly + 45; // +0xB4 + var tmp_v: [15 * 3]f32 = undefined; + var tmp_a: [15]f32 = undefined; + for (0..n * 3) |j| tmp_v[j] = poly[j]; + for (0..n) |j| tmp_a[j] = attribs[j]; + + // --- Phase 4: Sutherland-Hodgman edge walk --- + var out: u32 = 0; + var prev: u32 = n - 1; + var pd = dists[prev]; + + i = 0; + while (i < n) : (i += 1) { + const cd = dists[i]; + + if (pd >= 0.0) { + if (cd >= 0.0) { + emit(poly, attribs, &out, (tmp_v[i * 3 ..]).ptr, tmp_a[i]); + } else { + if (pd > CLIP_EPSILON) { + lerp(poly, attribs, &out, (tmp_v[prev * 3 ..]).ptr, (tmp_v[i * 3 ..]).ptr, pd / (cd - pd), new_attrib); + } + } + } else { + if (cd >= 0.0) { + if (cd > CLIP_EPSILON) { + lerp(poly, attribs, &out, (tmp_v[prev * 3 ..]).ptr, (tmp_v[i * 3 ..]).ptr, pd / (cd - pd), new_attrib); + } + emit(poly, attribs, &out, (tmp_v[i * 3 ..]).ptr, tmp_a[i]); + } + } + + prev = i; + pd = cd; + } + + count_ptr.* = if (out >= 3) out else 0; +} + +inline fn emit(poly: [*]f32, attribs: [*]f32, out: *u32, v: [*]const f32, a: f32) void { + const o = out.*; + const b = o * 3; + poly[b] = v[0]; + poly[b + 1] = v[1]; + poly[b + 2] = v[2]; + attribs[o] = a; + out.* = o + 1; +} + +inline fn lerp(poly: [*]f32, attribs: [*]f32, out: *u32, p: [*]const f32, c: [*]const f32, t: f32, a: f32) void { + const o = out.*; + const b = o * 3; + // intersection = prev - (curr - prev) * t + poly[b] = p[0] - (c[0] - p[0]) * t; + poly[b + 1] = p[1] - (c[1] - p[1]) * t; + poly[b + 2] = p[2] - (c[2] - p[2]) * t; + attribs[o] = a; + out.* = o + 1; +} + +// ============================================================================= +// BuildTrianglePlanes (0x632460) +// __fastcall(ECX=vertices, EDX=triangle_indices(byte*), +// stack: plane_normal*, offset_vector*, output_planes*) +// Returns: int (1=ok, 0=degenerate) +// +// Builds 4 clipping planes from a triangle + offset: +// planes[0..2]: edge planes (perpendicular to triangle, one per edge) +// planes[3]: cap plane (from plane_normal, offset by offset_vector) +// ============================================================================= + +export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 { + const verts: [*]const f32 = @ptrFromInt(verts_addr); + const indices: [*]const u8 = @ptrFromInt(indices_addr); + const plane_normal: [*]const f32 = @ptrFromInt(normal_addr); + const offset_vec: [*]const f32 = @ptrFromInt(offset_addr); + const output: [*]f32 = @ptrFromInt(out_addr); + + const ofs = loadV3(offset_vec); + + // Load 3 triangle vertices + var tv: [3]V4 = undefined; + for (0..3) |i| { + const idx = @as(u32, indices[i]) * 3; + tv[i] = loadV3(verts + idx); + } + + // Edge tables: for edge i, v0=i, v1=next, v2=opposite + const next = [3]u8{ 1, 2, 0 }; + const opp = [3]u8{ 2, 0, 1 }; + + // Build 3 edge planes + for (0..3) |i| { + const v0 = tv[i]; + const v1 = tv[next[i]]; + const v2 = tv[opp[i]]; + const offset_v = v0 + ofs; // offset vertex + + // Check for degenerate triangle: cross(offset_vec, v1 - v0) + const edge = v1 - v0; + const check_normal = cross3(ofs, edge); + if (dot3(check_normal, check_normal) < MOVEMENT_EPSILON) return 0; + + // Calculate plane from 3 points (v0, v1, offset_v) + const e1 = offset_v - v0; // = offset_vec + const e2 = v1 - v0; // = edge + var normal = cross3(e2, e1); + const len_sq = dot3(normal, normal); + const inv_len: V4 = @splat(1.0 / @sqrt(len_sq)); + normal = normal * inv_len; + const d = -dot3(normal, v0); + + // Check orientation: if opposite vertex is on positive side, negate + const to_v2 = v2 - v0; + const plane_idx = i * 4; + if (dot3(to_v2, normal) > 0.0) { + output[plane_idx + 0] = -normal[0]; + output[plane_idx + 1] = -normal[1]; + output[plane_idx + 2] = -normal[2]; + output[plane_idx + 3] = -d; + } else { + output[plane_idx + 0] = normal[0]; + output[plane_idx + 1] = normal[1]; + output[plane_idx + 2] = normal[2]; + output[plane_idx + 3] = d; + } + } + + // 4th plane: cap plane from plane_normal + const pn = loadV3(plane_normal); + const cap_point = tv[0] + ofs; + output[12] = plane_normal[0]; + output[13] = plane_normal[1]; + output[14] = plane_normal[2]; + output[15] = -dot3(pn, cap_point); + + return 1; +} + +// ============================================================================= +// rayTriangleIntersection (0x7c29f0) +// Möller-Trumbore ray-triangle intersection test +// __fastcall(ECX=ray[6], EDX=verts_base, stack: indices_u16[3], out_dist*, out_bary*, tolerance) +// Returns: 1 = hit, 0 = miss +// ============================================================================= + +export fn rayTriangleIntersection( + ray_addr: u32, + verts_addr: u32, + indices_addr: u32, + out_dist_addr: u32, + out_bary_addr: u32, + tolerance_bits: u32, +) u32 { + const ray: [*]const f32 = @ptrFromInt(ray_addr); + const verts: [*]const f32 = @ptrFromInt(verts_addr); + const indices: [*]const u16 = @ptrFromInt(indices_addr); + const tolerance: f32 = @bitCast(tolerance_bits); + + const neg_tol = -tolerance; + const one_plus_tol = 1.0 + tolerance; + + // Load vertex positions via indices (each vertex = 3 floats, stride = 12 bytes) + const idx0: u32 = @as(u32, indices[0]) * 3; + const idx1: u32 = @as(u32, indices[1]) * 3; + const idx2: u32 = @as(u32, indices[2]) * 3; + + const v0 = loadV3(verts + idx0); + const v1 = loadV3(verts + idx1); + const v2 = loadV3(verts + idx2); + + // Ray: origin at ray[0..3], direction at ray[3..6] + const origin = loadV3(ray); + const dir = loadV3(ray + 3); + + // Möller-Trumbore algorithm + const edge1 = v1 - v0; + const edge2 = v2 - v0; + const h = cross3(dir, edge2); + const det = dot3(edge1, h); + + // Determinant thresholds from WoW binary (both 0.0 at compile time = two-sided test) + const det_neg = @as(*const f32, @ptrFromInt(0x0081d9bc)).*; + const det_pos = @as(*const f32, @ptrFromInt(0x0080e2e4)).*; + + if (det > det_neg and det < det_pos) return 0; + + const inv_det = 1.0 / det; + + // Barycentric u + const s = origin - v0; + const u_val = dot3(s, h) * inv_det; + if (u_val < neg_tol or u_val > one_plus_tol) return 0; + + // Barycentric v + const q = cross3(s, edge1); + const v_val = dot3(dir, q) * inv_det; + if (v_val < neg_tol or (u_val + v_val) > one_plus_tol) return 0; + + // Intersection distance t + const t = dot3(edge2, q) * inv_det; + + if (out_dist_addr != 0) { + @as(*f32, @ptrFromInt(out_dist_addr)).* = t; + } + if (out_bary_addr != 0) { + const bary: [*]f32 = @ptrFromInt(out_bary_addr); + bary[0] = u_val; + bary[1] = v_val; + } + + return 1; +} + +// ============================================================================= +// multiplyMatrix4x4 (0x7bc6a0) +// SSE replacement for 542 bytes of x87 FPU matrix multiply. +// __fastcall(ECX=result, EDX=left, stack: right), RET 0x4 +// Row-major: result[i][j] = sum(left[i][k] * right[k][j], k=0..3) +// Returns result pointer (EAX = result_addr). +// ============================================================================= + +export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 { + const result: [*]f32 = @ptrFromInt(result_addr); + const left: [*]const f32 = @ptrFromInt(left_addr); + const right: [*]const f32 = @ptrFromInt(right_addr); + + // Load all 4 rows of right matrix + const r0: V4 = .{ right[0], right[1], right[2], right[3] }; + const r1: V4 = .{ right[4], right[5], right[6], right[7] }; + const r2: V4 = .{ right[8], right[9], right[10], right[11] }; + const r3: V4 = .{ right[12], right[13], right[14], right[15] }; + + // For each row of left, broadcast-multiply-add against right rows + inline for (0..4) |i| { + const b = i * 4; + const out = splat4(left[b]) * r0 + splat4(left[b + 1]) * r1 + splat4(left[b + 2]) * r2 + splat4(left[b + 3]) * r3; + storeV4(result + b, out); + } + + return result_addr; +} + +// ============================================================================= +// rotateMatrixByAxisAngle (0x7bdd60) +// Builds axis-angle rotation matrix (Rodrigues) then multiplies. +// Replaces 853 bytes of x87 FPU (createAxisAngleRotationMatrix 311B + +// multiplyMatrix4x4 542B) with SSE @Vector math. +// __thiscall(ECX=matrix, stack: angle, axis_ptr, is_unit_flag) +// ============================================================================= + +export fn rotateMatrixByAxisAngle( + matrix_addr: u32, + angle_bits: u32, + axis_addr: u32, + is_unit: u32, +) void { + const matrix: [*]f32 = @ptrFromInt(matrix_addr); + const angle: f32 = @bitCast(angle_bits); + const axis_ptr: [*]const f32 = @ptrFromInt(axis_addr); + + // Load and optionally normalize axis + var ax = axis_ptr[0]; + var ay = axis_ptr[1]; + var az = axis_ptr[2]; + + if (is_unit == 0) { + const inv_len = 1.0 / @sqrt(ax * ax + ay * ay + az * az); + ax *= inv_len; + ay *= inv_len; + az *= inv_len; + } + + // Rodrigues rotation matrix (row-major, matching WoW's layout) + const c = cosf(angle); + const s = sinf(angle); + const t = 1.0 - c; + + // Build rotation matrix on stack, then multiply: result = rot * input + // Row 3 of rotation is {0,0,0,1} so result row 3 = input row 3 (identity). + // We build the full 4x4 rot matrix and use multiplyMatrix4x4 for the multiply. + var rot: [16]f32 = .{ + ax * ax * t + c, ax * ay * t + az * s, ax * az * t - ay * s, 0, + ax * ay * t - az * s, ay * ay * t + c, ay * az * t + ax * s, 0, + ax * az * t + ay * s, ay * az * t - ax * s, az * az * t + c, 0, + 0, 0, 0, 1, + }; + + // result = rot * matrix, but result == matrix so use temp to avoid aliasing + var tmp: [16]f32 = undefined; + _ = multiplyMatrix4x4(@intFromPtr(&tmp), @intFromPtr(&rot), matrix_addr); + matrix[0..16].* = tmp; +} + +// MSVC CRT sin/cos — linked from the WoW process +extern fn sinf(f32) f32; +extern fn cosf(f32) f32; + +// ============================================================================= +// Vector helpers +// ============================================================================= + +inline fn loadV3(p: [*]const f32) V4 { + return .{ p[0], p[1], p[2], 0.0 }; +} + +inline fn dot3(a: V4, b: V4) f32 { + const p = a * b; + return p[0] + p[1] + p[2]; +} + +inline fn cross3(a: V4, b: V4) V4 { + const a_yzx = @shuffle(f32, a, undefined, [4]i32{ 1, 2, 0, 3 }); + const b_zxy = @shuffle(f32, b, undefined, [4]i32{ 2, 0, 1, 3 }); + const a_zxy = @shuffle(f32, a, undefined, [4]i32{ 2, 0, 1, 3 }); + const b_yzx = @shuffle(f32, b, undefined, [4]i32{ 1, 2, 0, 3 }); + return a_yzx * b_zxy - a_zxy * b_yzx; +} + +inline fn splat4(v: f32) V4 { + return @splat(v); +} + +inline fn storeV4(dst: [*]f32, v: V4) void { + dst[0] = v[0]; + dst[1] = v[1]; + dst[2] = v[2]; + dst[3] = v[3]; +} diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index 69ffc63..f163451 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -1,12 +1,13 @@ //! transform44 — render pipeline profiling & M2 bone transform optimization //! -//! Hooks 5 functions in the render/update pipeline to measure per-frame costs: +//! Hooks 6 functions in the render/update pipeline to measure per-frame costs: //! //! executeSceneRenderPass (0x708900) — top-level render pass dispatch //! renderFrame (0x707680) — per-model render (calls transformMatrix4x4) //! transformMatrix4x4 (0x714260) — per-model bone transform engine (17703 bytes) //! RenderTextureQuads (0x76FB00) — batched quad rendering //! CMovement::Process (0x616620) — per-unit movement update +//! blit_hub (0x5a4f60) — pixel transfer dispatch (CPU-side blitting) //! //! Stats are dumped every DUMP_FRAMES render passes (~2s at 60fps). @@ -14,6 +15,11 @@ const std = @import("std"); const hook = @import("zhook"); const logging = @import("../logging.zig"); const mod_mutex = @import("../mutex.zig"); +extern fn clipPolygonToSinglePlane(u32, u32, u32) void; +extern fn buildTrianglePlanes(u32, u32, u32, u32, u32) u32; +extern fn rayTriangleIntersection(u32, u32, u32, u32, u32, u32) u32; +extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void; +extern fn multiplyMatrix4x4(u32, u32, u32) u32; pub const module_name: [*:0]const u8 = "transform44"; @@ -29,12 +35,26 @@ pub fn isActive() bool { // Profiling state — unified dump every DUMP_FRAMES render passes // ============================================================================= -const DUMP_FRAMES: u64 = 600; +const DUMP_FRAMES: u64 = 1800; // ~30s at 60fps var prof = ProfState{}; var t44_depth: u64 = 0; // recursion depth — survives resets var last_frame_tsc: u64 = 0; // frame-to-frame TSC for total frame time +// A/B testing: alternate between baseline (original) and custom (optimized) code paths. +// Flips every DUMP_FRAMES so each dump period is purely one mode. +pub var ab_use_custom: bool = false; + +// Persistent blit totals per A/B mode — NOT reset each dump period. +// Accumulates across all dump periods so rare blits still show up. +var blit_total_baseline: BlitAccum = .{}; +var blit_total_custom: BlitAccum = .{}; + +const BlitAccum = struct { + calls: u64 = 0, + cycles: u64 = 0, +}; + const ProfState = struct { // Frame counter (incremented by executeSceneRenderPass) frames: u64 = 0, @@ -65,6 +85,74 @@ const ProfState = struct { // CMovement::ProcessUnitMovementUpdate (0x616620) mov_calls: u64 = 0, mov_cycles: u64 = 0, + + // --- Perf-identified hotspots --- + clip_calls: u64 = 0, // ClipPolygonToSinglePlane (0x6318c0) 3.95% + clip_cycles: u64 = 0, + glyph_calls: u64 = 0, // GetOrCreateCharacterGlyph (0x5ca2d0) 3.65% + glyph_cycles: u64 = 0, + particle_calls: u64 = 0, // RenderParticleSprites (0x7b2a50) 1.73% + particle_cycles: u64 = 0, + collision_calls: u64 = 0, // processLinkedListCollision (0x6abc40) 1.57% + collision_cycles: u64 = 0, + entpos_calls: u64 = 0, // UpdateEntityAndChunksPositions (0x6afad0) 1.18% + entpos_cycles: u64 = 0, + layers_calls: u64 = 0, // renderAllFrameLayers (0x765650) 1.15% + layers_cycles: u64 = 0, + textvb_calls: u64 = 0, // RenderTextToVertexBuffer (0x5ccbe0) 1.07% + textvb_cycles: u64 = 0, + complexgeo_calls: u64 = 0, // RenderComplexGeometry (0x58a3d0) 1.02% + complexgeo_cycles: u64 = 0, + entbounds_calls: u64 = 0, // updateEntitiesInBounds (0x6c1f70) 0.93% + entbounds_cycles: u64 = 0, + textctr_calls: u64 = 0, // updateTextFrameCounter (0x5cdf40) 0.88% + textctr_cycles: u64 = 0, + spatial_calls: u64 = 0, // AddToSpatialGrid (0x6816f0) 0.66% + spatial_cycles: u64 = 0, + raytri_calls: u64 = 0, // ray_triangle_intersection_indexed_ushort (0x7c29f0) 0.65% + raytri_cycles: u64 = 0, + linkedlist_calls: u64 = 0, // ManageLinkedListNode (0x710b90) 0.64% + linkedlist_cycles: u64 = 0, + color_calls: u64 = 0, // calculateColorValues (0x7b9b10) 0.63% + color_cycles: u64 = 0, + setvec_calls: u64 = 0, // SetVector3 (0x686640) 0.61% + setvec_cycles: u64 = 0, + cull_calls: u64 = 0, // PerformSpatialCulling (0x6b8c60) 0.59% + cull_cycles: u64 = 0, + colldet_calls: u64 = 0, // performCollisionDetection (0x6b88e0) 0.59% + colldet_cycles: u64 = 0, + activep_calls: u64 = 0, // ProcessActiveParticles (0x7b5a10) 0.55% + activep_cycles: u64 = 0, + cbiter_calls: u64 = 0, // CallbackIterator (0x404130) 0.53% + cbiter_cycles: u64 = 0, + findguid_calls: u64 = 0, // FindObjectByGUID (0x464890) 0.50% + findguid_cycles: u64 = 0, + raytri2_calls: u64 = 0, // RayTriangleIntersection (0x632700) 0.49% + raytri2_cycles: u64 = 0, + drawbatch_calls: u64 = 0, // DrawBatchProj (0x70cb30) 0.48% + drawbatch_cycles: u64 = 0, + findlua_calls: u64 = 0, // FindLuaFunction (0x702000) 0.47% + findlua_cycles: u64 = 0, + spritequad_calls: u64 = 0, // RenderSpriteQuads (0x5a0f50) 0.44% + spritequad_cycles: u64 = 0, + scenenode_calls: u64 = 0, // renderSceneNode (0x718960) 0.44% + scenenode_cycles: u64 = 0, + terrain_calls: u64 = 0, // generateTerrainChunk (0x6cffc0) 0.44% + terrain_cycles: u64 = 0, + d3dtex_calls: u64 = 0, // D3D_SetTexture (0x593840) 0.43% + d3dtex_cycles: u64 = 0, + bboxchk_calls: u64 = 0, // checkBoundingBoxIntersection (0x6b8b70) 0.42% + bboxchk_cycles: u64 = 0, + rotmat_calls: u64 = 0, // rotateMatrixByAxisAngle (0x7bdd60) — unresolved callee hotspot + rotmat_cycles: u64 = 0, + triplane_calls: u64 = 0, // BuildTrianglePlanes (0x632460) + triplane_cycles: u64 = 0, + partsetup_calls: u64 = 0, // SetupParticleRendering (0x7b3d20) + partsetup_cycles: u64 = 0, + matmul_calls: u64 = 0, // multiplyMatrix4x4 (0x7bc6a0) + matmul_cycles: u64 = 0, + textline_calls: u64 = 0, // renderTextLine (0x5ce0c0) + textline_cycles: u64 = 0, }; inline fn rdtsc() u64 { @@ -187,22 +275,274 @@ fn execRenderPassDetour(this: u32, edx: u32, pass_index: u32) callconv(hook.cc.f // ============================================================================= // Hook: RenderTextureQuads (0x76FB00) // __fastcall(ECX=RenderBatch*) — no stack params, RET -// RenderBatch+0xC = item_count +// +// RenderBatch layout (assembly-verified): +// +0x0C = item_count (u32) +// +0x10 = items_ptr (RenderItem*) +// +0x18 = text_data (void*, passed to DrawString if non-null) +// +0x24 = callback_list (linked list, iterated after render) +// +// RenderItem layout (0x1C = 28 bytes per item, assembly-verified): +// +0x00 = texture (ptr, primary texture — SetTexture stage 0x17) +// +0x04 = vertices (float* xyz, stride 0x0C = 3 floats, 4 verts per quad) +// +0x08 = textureCoords (float* uv, stride 0x08 = 2 floats, 4 verts per quad) +// +0x0C = renderState (int, passed to SetRenderState(7, val)) +// +0x10 = additionalData (ptr, secondary vertex data — can be NULL) +// +0x14 = dataStride (int, stride for additionalData) +// +0x18 = secondaryTexture (ptr, SetTexture stage 0x3F) +// +// Original inner loop: per-item SetTexture + SetRenderState + +// InitializeRenderingPipeline + RenderVertexBuffer + EmptyRenderFunction +// = 5 GxDevice calls per quad. Batching by texture reduces draw calls. +// +// Key globals: +// 0xCF4CF4 = g_defaultTexCoord +// 0x878CDC = g_quadVertexIndices (6 u16: 0,1,2, 0,2,3) // ============================================================================= const RenderQuadsFn = fn (u32, u32) callconv(hook.cc.fastcall) void; var render_quads_hook: hook.Detour(RenderQuadsFn) = .{}; -fn renderQuadsDetour(batch: u32, edx: u32) callconv(hook.cc.fastcall) void { +const MAX_BATCH_QUADS = 256; +const ITEM_SIZE: u32 = 0x1C; // 28 bytes per RenderItem + +// Sort key for grouping by (texture, renderState, secondaryTexture) +const SortEntry = struct { + texture: u32, + render_state: u32, + secondary_tex: u32, + has_additional: bool, + index: u16, + + fn lessThan(a: SortEntry, b: SortEntry) bool { + if (a.texture != b.texture) return a.texture < b.texture; + if (a.render_state != b.render_state) return a.render_state < b.render_state; + return a.secondary_tex < b.secondary_tex; + } + + fn sameGroup(a: SortEntry, b: SortEntry) bool { + return a.texture == b.texture and + a.render_state == b.render_state and + a.secondary_tex == b.secondary_tex; + } +}; +var sort_entries: [MAX_BATCH_QUADS]SortEntry = undefined; + +// Batched vertex data: contiguous xyz and uv buffers for multi-quad draw calls +var batch_xyz: [MAX_BATCH_QUADS * 4 * 3]f32 = undefined; // 4 verts * 3 floats per quad +var batch_uv: [MAX_BATCH_QUADS * 4 * 2]f32 = undefined; // 4 verts * 2 floats per quad + +// Pre-computed index buffer: quad Q uses verts Q*4..Q*4+3, triangles (0,1,2)(0,2,3) +const batch_indices = blk: { + var idx: [MAX_BATCH_QUADS * 6]u16 = undefined; + for (0..MAX_BATCH_QUADS) |q| { + const base: u16 = @intCast(q * 4); + idx[q * 6 + 0] = base; + idx[q * 6 + 1] = base + 1; + idx[q * 6 + 2] = base + 2; + idx[q * 6 + 3] = base; + idx[q * 6 + 4] = base + 2; + idx[q * 6 + 5] = base + 3; + } + break :blk idx; +}; + +// ---- GxDevice wrapper calls (assembly-verified calling conventions) ---- +// All route through the GxDevice at global 0xC0ED38. + +// BeginRender/EndRender: thiscall thunks, load ECX from global, JMP to vtable method +inline fn gxBeginRender() void { + hook.call(fn () callconv(.c) void, 0x589f40, .{}); +} +inline fn gxEndRender() void { + hook.call(fn () callconv(.c) void, 0x589f50, .{}); +} +// SetRenderState: fastcall(ECX=stateId, EDX=value), plain RET +// SetTexture: fastcall(ECX=stage, EDX=texturePtr), plain RET +const GxFastcall2 = fn (u32, u32) callconv(hook.cc.fastcall) void; +inline fn gxSetRenderState(state_id: u32, value: u32) void { + hook.call(GxFastcall2, 0x589e60, .{ state_id, value }); +} +inline fn gxSetTexture(stage: u32, texture: u32) void { + hook.call(GxFastcall2, 0x589e80, .{ stage, texture }); +} +// InitializeRenderingPipeline: fastcall(ECX=vertCount, EDX=vertsPtr, 11 stack), RET 0x2c +// Stores vertCount to global [0xC0ED2C] which RenderVertexBuffer reads. +const GxInitPipelineFn = fn (u32, u32, u32, u32, u32, u32, u32, u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) void; +const GX_DEFAULT_TEXCOORD: u32 = 0xCF4CF4; + +// RenderVertexBuffer: fastcall(ECX=primType, EDX=vertCount, stack: indicesPtr), RET 0x4 +const GxRenderVBFn = fn (u32, u32, u32) callconv(hook.cc.fastcall) void; +const GX_QUAD_INDICES: u32 = 0x878CDC; // game's {0,1,2,0,2,3} + +fn renderQuadsDetour(batch_ptr: u32, edx: u32) callconv(hook.cc.fastcall) void { asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); - // Read item count before call (batch+0xC) - const items: u32 = if (batch != 0) hook.readMem(u32, batch + 0xC) else 0; + const item_count: u32 = if (batch_ptr != 0) hook.readMem(u32, batch_ptr + 0xC) else 0; const start = rdtsc(); - render_quads_hook.callOriginal(.{ batch, edx }); + + // RTQ batching tested: sort-only (Approach A) and GxDevice draw-call batching + // (Approach B) both showed 0% improvement. Bottleneck is inside GxDevice/D3D9 + // internals (vertex buffer mgmt, proxy state machine), not per-call overhead. + // Batching code preserved below for future work. See RESEARCH.md for details. + render_quads_hook.callOriginal(.{ batch_ptr, edx }); + prof.rtq_cycles +|= rdtsc() - start; prof.rtq_calls +|= 1; - prof.rtq_items +|= items; + prof.rtq_items +|= item_count; +} + +/// Batch UI quads by (texture, renderState, secondaryTexture) groups. +/// Groups with no additionalData get a single draw call per group. +/// Groups with additionalData render individually through GxDevice wrappers. +fn renderQuadsBatched(batch_ptr: u32, item_count: u32) void { + const items_base = hook.readMem(u32, batch_ptr + 0x10); + if (items_base == 0) { + render_quads_hook.callOriginal(.{ batch_ptr, 0 }); + return; + } + + // Build sort entries with full grouping key + for (0..item_count) |i| { + const item_addr = items_base + @as(u32, @intCast(i)) * ITEM_SIZE; + const rs = hook.readMem(u32, item_addr + 0x0C); + const additional = hook.readMem(u32, item_addr + 0x10); + sort_entries[i] = .{ + .texture = hook.readMem(u32, item_addr), + .render_state = if (additional != 0 and rs < 2) 2 else rs, + .secondary_tex = hook.readMem(u32, item_addr + 0x18), + .has_additional = additional != 0, + .index = @intCast(i), + }; + } + + // Insertion sort by (texture, render_state, secondary_tex) + const entries = sort_entries[0..item_count]; + for (1..entries.len) |ii| { + const key = entries[ii]; + var j: usize = ii; + while (j > 0 and SortEntry.lessThan(key, entries[j - 1])) { + entries[j] = entries[j - 1]; + j -= 1; + } + entries[j] = key; + } + + // Render through GxDevice wrappers + gxBeginRender(); + gxSetRenderState(0x0E, 0); + gxSetRenderState(0x0F, 0); + gxSetRenderState(0x10, 0); + gxSetRenderState(0x12, 0); + + var last_tex: u32 = 0xFFFFFFFF; + var i: u32 = 0; + while (i < item_count) { + const gs = i; // group start + var ge = gs + 1; // group end + while (ge < item_count and SortEntry.sameGroup(entries[ge], entries[gs])) { + ge += 1; + } + + // Set primary texture (skip if same as last group) + if (entries[gs].texture != last_tex) { + gxSetTexture(0x17, entries[gs].texture); + last_tex = entries[gs].texture; + } + // Set render state + if (entries[gs].render_state != 0x0B) { + gxSetRenderState(7, entries[gs].render_state); + } + // Set secondary texture + gxSetTexture(0x3F, entries[gs].secondary_tex); + + // Check if all items in group lack additionalData (batchable) + var can_batch = true; + for (gs..ge) |gi| { + if (entries[gi].has_additional) { + can_batch = false; + break; + } + } + + if (can_batch and ge - gs > 1) { + // BATCHED: build contiguous xyz/uv buffers, single draw call + var verts: u32 = 0; + for (gs..ge) |gi| { + const idx = entries[gi].index; + const item_addr = items_base + @as(u32, idx) * ITEM_SIZE; + const xyz_ptr = hook.readMem(u32, item_addr + 0x04); + const uv_ptr = hook.readMem(u32, item_addr + 0x08); + + if (xyz_ptr != 0) { + const src_xyz: [*]const f32 = @ptrFromInt(xyz_ptr); + @memcpy(batch_xyz[verts * 3 ..][0..12], src_xyz[0..12]); + } + if (uv_ptr != 0) { + const src_uv: [*]const f32 = @ptrFromInt(uv_ptr); + @memcpy(batch_uv[verts * 2 ..][0..8], src_uv[0..8]); + } + verts += 4; + } + + hook.call(GxInitPipelineFn, 0x58a2a0, .{ + verts, @intFromPtr(&batch_xyz), // ECX=vertCount, EDX=xyzPtr + @as(u32, 0x0C), GX_DEFAULT_TEXCOORD, @as(u32, 0), // stride, defaultTC, 0 + @as(u32, 0), @as(u32, 0), // no additionalData + @as(u32, 0), @as(u32, 0), // zeros (skipped by forwarder) + @intFromPtr(&batch_uv), @as(u32, 8), // uvPtr, uvStride + @as(u32, 0), @as(u32, 0), // trailing zeros + }); + hook.call(GxRenderVBFn, 0x58a2e0, .{ + @as(u32, 4), // D3DPT_TRIANGLELIST + verts, // vertex count + @intFromPtr(&batch_indices), // index buffer + }); + hook.call(fn () callconv(.c) void, 0x58a340, .{}); // EmptyRenderFunction + } else { + // INDIVIDUAL: per-item draw calls through GxDevice wrappers + for (gs..ge) |gi| { + const idx = entries[gi].index; + const item_addr = items_base + @as(u32, idx) * ITEM_SIZE; + const xyz_ptr = hook.readMem(u32, item_addr + 0x04); + const uv_ptr = hook.readMem(u32, item_addr + 0x08); + const additional = hook.readMem(u32, item_addr + 0x10); + const add_stride = hook.readMem(u32, item_addr + 0x14); + + hook.call(GxInitPipelineFn, 0x58a2a0, .{ + @as(u32, 4), xyz_ptr, // 4 verts, xyz data + @as(u32, 0x0C), GX_DEFAULT_TEXCOORD, @as(u32, 0), + additional, add_stride, + @as(u32, 0), @as(u32, 0), + uv_ptr, @as(u32, 8), + @as(u32, 0), @as(u32, 0), + }); + hook.call(GxRenderVBFn, 0x58a2e0, .{ + @as(u32, 4), @as(u32, 4), GX_QUAD_INDICES, + }); + hook.call(fn () callconv(.c) void, 0x58a340, .{}); + } + } + + i = ge; + } + + gxEndRender(); + + // Handle text_data (DrawString at 0x5c1ef0) + const text_data = hook.readMem(u32, batch_ptr + 0x18); + if (text_data != 0) { + hook.call(fn (u32) callconv(hook.cc.fastcall) void, 0x5c1ef0, .{text_data}); + } + + // Walk callback linked list + var node = hook.readMem(u32, batch_ptr + 0x24); + if (node & 1 != 0) node = 0; + while (node != 0 and node & 1 == 0) { + const cb: *const fn () callconv(.c) void = @ptrFromInt(hook.readMem(u32, node + 8)); + cb(); + node = hook.readMem(u32, node + 4); + } } // ============================================================================= @@ -224,6 +564,514 @@ fn movementDetour(this: u32, edx: u32, time_now: u32, last_update: u32) callconv prof.mov_calls +|= 1; } +// ============================================================================= +// Hook: interpolateAnimationKeyframes (0x713ea0) +// __fastcall(ECX=animObj, EDX=animState, stack: keyframeData*, outputBuffer*) +// RET 0x8 (2 stack params) +// +// Called per-bone from transformMatrix4x4 for rotation/scale tracks. +// Calls findInterpolationIndices, then does 4-component x87 lerp. +// Our SSE path replaces the x87 lerp with @Vector(4, f32) ops. +// +// outputBuffer layout (written by this function): +// +0x00 = lower keyframe index (u32) } +// +0x04 = upper keyframe index (u32) } filled by findInterpolationIndices +// +0x08 = interpolation factor t (f32)} +// +0x0C = lerp result (4 floats) } filled by lerp +// +0x1C = secondary indices (crossfade)} filled by 2nd findInterpolationIndices +// +0x28 = secondary lerp result } filled by 2nd lerp +// ============================================================================= + +const InterpKfFn = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) void; +var interp_kf_hook: hook.Detour(InterpKfFn) = .{}; + +// findInterpolationIndices: __thiscall(ECX=animObj, stack: p1, p2, kfData, outBuf) +// RET 0x10 (4 stack params). Fastcall: ECX=animObj, EDX=unused, 4 stack. +const FindInterpIdxFn = fn (u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) void; + +fn interpKfDetour(anim_obj: u32, anim_state: u32, kf_data: u32, out_buf: u32) callconv(hook.cc.fastcall) void { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + + // SSE interp disabled — hook overhead exceeds savings (see perf analysis) + interp_kf_hook.callOriginal(.{ anim_obj, anim_state, kf_data, out_buf }); +} + +noinline fn interpKfSSE(anim_obj: u32, anim_state: u32, kf_data: u32, out_buf: u32) void { + @setRuntimeSafety(false); + + // Step 1: call findInterpolationIndices (original game function) + const p1 = hook.readMem(u32, anim_state + 0x98); + const p2 = hook.readMem(u32, anim_state + 0x9C); + hook.call(FindInterpIdxFn, 0x713d50, .{ anim_obj, 0, p1, p2, kf_data, out_buf }); + + // Step 2: read results + const lower = hook.readMem(u32, out_buf); + const flag = hook.readMem(u16, kf_data); // keyframeData[0]: 0 = single keyframe + + const kf_base = hook.readMem(u32, kf_data + 0x18); + + if (flag == 0) { + // Single keyframe: direct copy to outBuf+0xC (no interpolation) + const src: [*]const u32 = @ptrFromInt(kf_base + lower * 16); + const dst: [*]u32 = @ptrFromInt(out_buf + 0xC); + dst[0] = src[0]; + dst[1] = src[1]; + dst[2] = src[2]; + dst[3] = src[3]; + return; + } + + // Step 3: SSE 4-component lerp — result = a + (b - a) * t + const upper = hook.readMem(u32, out_buf + 4); + const t: f32 = @bitCast(hook.readMem(u32, out_buf + 8)); + + const a_ptr: [*]const f32 = @ptrFromInt(kf_base + lower * 16); + const b_ptr: [*]const f32 = @ptrFromInt(kf_base + upper * 16); + + const a: @Vector(4, f32) = a_ptr[0..4].*; + const b: @Vector(4, f32) = b_ptr[0..4].*; + const tv: @Vector(4, f32) = @splat(t); + const result = a + (b - a) * tv; + + const dst: [*]f32 = @ptrFromInt(out_buf + 0xC); + dst[0..4].* = @as([4]f32, result); + + // Step 4: check crossfade condition + const const_zero: f32 = @bitCast(hook.readMem(u32, 0x7ffd74)); + const blend_weight: f32 = @bitCast(hook.readMem(u32, anim_state + 0x10C)); + if (blend_weight == const_zero) return; + + const time_idx = hook.readMem(u16, kf_data + 2); + if (time_idx != 0xFFFF) return; + + // Step 5: secondary findInterpolationIndices + SSE lerp for crossfade + const p3 = hook.readMem(u32, anim_state + 0xC4); + const p4 = hook.readMem(u32, anim_state + 0xC8); + hook.call(FindInterpIdxFn, 0x713d50, .{ anim_obj, 0, p3, p4, kf_data, out_buf + 0x1C }); + + const lower2 = hook.readMem(u32, out_buf + 0x1C); + const upper2 = hook.readMem(u32, out_buf + 0x20); + const t2: f32 = @bitCast(hook.readMem(u32, out_buf + 0x24)); + + const a2_ptr: [*]const f32 = @ptrFromInt(kf_base + lower2 * 16); + const b2_ptr: [*]const f32 = @ptrFromInt(kf_base + upper2 * 16); + + const a2: @Vector(4, f32) = a2_ptr[0..4].*; + const b2: @Vector(4, f32) = b2_ptr[0..4].*; + const t2v: @Vector(4, f32) = @splat(t2); + const result2 = a2 + (b2 - a2) * t2v; + + const dst2: [*]f32 = @ptrFromInt(out_buf + 0x28); + dst2[0..4].* = @as([4]f32, result2); + + // Step 6: blend result1 toward result2 by blend_weight + const wv: @Vector(4, f32) = @splat(blend_weight); + const blended = result + (result2 - result) * wv; + dst[0..4].* = @as([4]f32, blended); +} + +// ============================================================================= +// Perf-identified hotspot hooks — timing-only wrappers. +// Generated from perf.data.perfparser analysis (July 2025). +// +// Convention mapping for Detour (all use fastcall ABI): +// thiscall RET 0 → Fn2 (ECX=this, EDX=unused) +// thiscall RET 0x4 → Fn3 (ECX=this, EDX=unused, 1 stack) +// thiscall RET 0x8 → Fn4 (ECX=this, EDX=unused, 2 stack) +// thiscall RET 0xC → Fn5 +// thiscall RET 0x18 → Fn8 +// thiscall RET 0x20 → Fn10 +// stdcall RET 0x4 → Fn3 (ECX=p1, EDX=p2, 1 stack) — but no this +// stdcall RET 0x8 → Fn4 +// stdcall RET 0x10 → Fn6 +// stdcall RET 0x24 → Fn11 +// ============================================================================= + +// Function type aliases by param count (all fastcall, return ?*anyopaque to preserve EAX) +const Fn2 = fn (u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn3 = fn (u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn4 = fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn5 = fn (u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn8v = fn (u32, u32, u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn6 = fn (u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn10 = fn (u32, u32, u32, u32, u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; +const Fn11 = fn (u32, u32, u32, u32, u32, u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) ?*anyopaque; + +// Hook variables +var clip_hook: hook.Detour(Fn3) = .{}; // ClipPolygonToSinglePlane: stdcall RET 0x4 +var glyph_hook: hook.Detour(Fn4) = .{}; // GetOrCreateCharacterGlyph: stdcall RET 0x8 +var particle_hook: hook.Detour(Fn4) = .{}; // RenderParticleSprites: thiscall RET 0x8 +var collision_hook: hook.Detour(Fn4) = .{}; // processLinkedListCollision: thiscall RET 0x8 +var entpos_hook: hook.Detour(Fn2) = .{}; // UpdateEntityAndChunksPositions: thiscall RET +var layers_hook: hook.Detour(Fn3) = .{}; // renderAllFrameLayers: thiscall RET 0x4 +var textvb_hook: hook.Detour(Fn8v) = .{}; // RenderTextToVertexBuffer: thiscall RET 0x18 +var complexgeo_hook: hook.Detour(Fn11) = .{}; // RenderComplexGeometry: stdcall RET 0x24 +var entbounds_hook: hook.Detour(Fn3) = .{}; // updateEntitiesInBounds: thiscall RET 0x4 +var textctr_hook: hook.Detour(Fn2) = .{}; // updateTextFrameCounter: thiscall RET +var spatial_hook: hook.Detour(Fn2) = .{}; // AddToSpatialGrid: thiscall RET +var raytri_hook: hook.Detour(Fn6) = .{}; // ray_triangle_intersection_indexed_ushort: stdcall RET 0x10 +var linkedlist_hook: hook.Detour(Fn3) = .{}; // ManageLinkedListNode: thiscall RET 0x4 +var color_hook: hook.Detour(Fn8v) = .{}; // calculateColorValues: thiscall RET 0x18 +var setvec_hook: hook.Detour(Fn2) = .{}; // SetVector3: thiscall RET +var cull_hook: hook.Detour(Fn4) = .{}; // PerformSpatialCulling: thiscall RET 0x8 +var colldet_hook: hook.Detour(Fn4) = .{}; // performCollisionDetection: thiscall RET 0x8 +var activep_hook: hook.Detour(Fn4) = .{}; // ProcessActiveParticles: stdcall RET 0x8 +var cbiter_hook: hook.Detour(Fn6) = .{}; // CallbackIterator: stdcall RET 0x10 +var findguid_hook: hook.Detour(Fn4) = .{}; // FindObjectByGUID: stdcall RET 0x8, returns ptr +var raytri2_hook: hook.Detour(Fn10) = .{}; // RayTriangleIntersection: thiscall RET 0x20 +var drawbatch_hook: hook.Detour(Fn2) = .{}; // DrawBatchProj: thiscall RET +var findlua_hook: hook.Detour(Fn3) = .{}; // FindLuaFunction: stdcall RET 0x4, returns ptr +var spritequad_hook: hook.Detour(Fn5) = .{}; // RenderSpriteQuads: thiscall RET 0xc +var scenenode_hook: hook.Detour(Fn2) = .{}; // renderSceneNode: thiscall RET +var terrain_hook: hook.Detour(Fn2) = .{}; // generateTerrainChunk: thiscall RET +var d3dtex_hook: hook.Detour(Fn4) = .{}; // D3D_SetTexture: thiscall RET 0x8 +var bboxchk_hook: hook.Detour(Fn4) = .{}; // checkBoundingBoxIntersection: stdcall RET 0x8 +var rotmat_hook: hook.Detour(Fn5) = .{}; // rotateMatrixByAxisAngle: thiscall RET 0xc +var triplane_hook: hook.Detour(Fn5) = .{}; // BuildTrianglePlanes: thiscall RET 0xc +var partsetup_hook: hook.Detour(Fn3) = .{}; // SetupParticleRendering: thiscall RET 0x4 +var matmul_hook: hook.Detour(Fn3) = .{}; // multiplyMatrix4x4: fastcall RET 0x4 +var textline_hook: hook.Detour(Fn6) = .{}; // renderTextLine: thiscall RET 0x10 + +// --- Detour functions (timing-only pass-through) --- + +fn clipDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + const s = rdtsc(); + if (ab_use_custom) { + clipPolygonToSinglePlane(a, b, c); + prof.clip_cycles +|= rdtsc() - s; + prof.clip_calls +|= 1; + return null; // original is void — EAX not read by callers + } + const ret = clip_hook.callOriginal(.{ a, b, c }); + prof.clip_cycles +|= rdtsc() - s; + prof.clip_calls +|= 1; + return ret; +} +fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = glyph_hook.callOriginal(.{ a, b, c, d }); + prof.glyph_cycles +|= rdtsc() - s; + prof.glyph_calls +|= 1; + return ret; +} +fn particleDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = particle_hook.callOriginal(.{ a, b, c, d }); + prof.particle_cycles +|= rdtsc() - s; + prof.particle_calls +|= 1; + return ret; +} +fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = collision_hook.callOriginal(.{ a, b, c, d }); + prof.collision_cycles +|= rdtsc() - s; + prof.collision_calls +|= 1; + return ret; +} +fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = entpos_hook.callOriginal(.{ a, b }); + prof.entpos_cycles +|= rdtsc() - s; + prof.entpos_calls +|= 1; + return ret; +} +fn layersDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = layers_hook.callOriginal(.{ a, b, c }); + prof.layers_cycles +|= rdtsc() - s; + prof.layers_calls +|= 1; + return ret; +} +fn textvbDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = textvb_hook.callOriginal(.{ a, b, c, d, e, f, g, h }); + prof.textvb_cycles +|= rdtsc() - s; + prof.textvb_calls +|= 1; + return ret; +} +fn complexgeoDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32, i: u32, j: u32, k: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = complexgeo_hook.callOriginal(.{ a, b, c, d, e, f, g, h, i, j, k }); + prof.complexgeo_cycles +|= rdtsc() - s; + prof.complexgeo_calls +|= 1; + return ret; +} +fn entboundsDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = entbounds_hook.callOriginal(.{ a, b, c }); + prof.entbounds_cycles +|= rdtsc() - s; + prof.entbounds_calls +|= 1; + return ret; +} +fn textctrDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = textctr_hook.callOriginal(.{ a, b }); + prof.textctr_cycles +|= rdtsc() - s; + prof.textctr_calls +|= 1; + return ret; +} +fn spatialDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = spatial_hook.callOriginal(.{ a, b }); + prof.spatial_cycles +|= rdtsc() - s; + prof.spatial_calls +|= 1; + return ret; +} +fn raytriDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + const s = rdtsc(); + if (ab_use_custom) { + const ret = rayTriangleIntersection(a, b, c, d, e, f); + prof.raytri_cycles +|= rdtsc() - s; + prof.raytri_calls +|= 1; + return @ptrFromInt(ret); + } + const ret = raytri_hook.callOriginal(.{ a, b, c, d, e, f }); + prof.raytri_cycles +|= rdtsc() - s; + prof.raytri_calls +|= 1; + return ret; +} +fn linkedlistDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = linkedlist_hook.callOriginal(.{ a, b, c }); + prof.linkedlist_cycles +|= rdtsc() - s; + prof.linkedlist_calls +|= 1; + return ret; +} +fn colorDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = color_hook.callOriginal(.{ a, b, c, d, e, f, g, h }); + prof.color_cycles +|= rdtsc() - s; + prof.color_calls +|= 1; + return ret; +} +fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = setvec_hook.callOriginal(.{ a, b }); + prof.setvec_cycles +|= rdtsc() - s; + prof.setvec_calls +|= 1; + return ret; +} +fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = cull_hook.callOriginal(.{ a, b, c, d }); + prof.cull_cycles +|= rdtsc() - s; + prof.cull_calls +|= 1; + return ret; +} +fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = colldet_hook.callOriginal(.{ a, b, c, d }); + prof.colldet_cycles +|= rdtsc() - s; + prof.colldet_calls +|= 1; + return ret; +} +fn activepDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = activep_hook.callOriginal(.{ a, b, c, d }); + prof.activep_cycles +|= rdtsc() - s; + prof.activep_calls +|= 1; + return ret; +} +fn cbiterDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = cbiter_hook.callOriginal(.{ a, b, c, d, e, f }); + prof.cbiter_cycles +|= rdtsc() - s; + prof.cbiter_calls +|= 1; + return ret; +} +fn findguidDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = findguid_hook.callOriginal(.{ a, b, c, d }); + prof.findguid_cycles +|= rdtsc() - s; + prof.findguid_calls +|= 1; + return ret; +} +fn raytri2Detour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32, i: u32, j: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = raytri2_hook.callOriginal(.{ a, b, c, d, e, f, g, h, i, j }); + prof.raytri2_cycles +|= rdtsc() - s; + prof.raytri2_calls +|= 1; + return ret; +} +fn drawbatchDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = drawbatch_hook.callOriginal(.{ a, b }); + prof.drawbatch_cycles +|= rdtsc() - s; + prof.drawbatch_calls +|= 1; + return ret; +} +fn findluaDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = findlua_hook.callOriginal(.{ a, b, c }); + prof.findlua_cycles +|= rdtsc() - s; + prof.findlua_calls +|= 1; + return ret; +} +fn spritequadDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = spritequad_hook.callOriginal(.{ a, b, c, d, e }); + prof.spritequad_cycles +|= rdtsc() - s; + prof.spritequad_calls +|= 1; + return ret; +} +fn scenenodeDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = scenenode_hook.callOriginal(.{ a, b }); + prof.scenenode_cycles +|= rdtsc() - s; + prof.scenenode_calls +|= 1; + return ret; +} +fn terrainDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = terrain_hook.callOriginal(.{ a, b }); + prof.terrain_cycles +|= rdtsc() - s; + prof.terrain_calls +|= 1; + return ret; +} +fn d3dtexDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = d3dtex_hook.callOriginal(.{ a, b, c, d }); + prof.d3dtex_cycles +|= rdtsc() - s; + prof.d3dtex_calls +|= 1; + return ret; +} +fn bboxchkDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = bboxchk_hook.callOriginal(.{ a, b, c, d }); + prof.bboxchk_cycles +|= rdtsc() - s; + prof.bboxchk_calls +|= 1; + return ret; +} +fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + const s = rdtsc(); + if (ab_use_custom) { + // thiscall: a=ECX=matrix, b=EDX=unused, c=angle, d=axis_ptr, e=is_unit + rotateMatrixByAxisAngle(a, c, d, e); + prof.rotmat_cycles +|= rdtsc() - s; + prof.rotmat_calls +|= 1; + return null; // void function, EAX not read by callers + } + const ret = rotmat_hook.callOriginal(.{ a, b, c, d, e }); + prof.rotmat_cycles +|= rdtsc() - s; + prof.rotmat_calls +|= 1; + return ret; +} +fn triplaneDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + const s = rdtsc(); + if (ab_use_custom) { + const ret = buildTrianglePlanes(a, b, c, d, e); + prof.triplane_cycles +|= rdtsc() - s; + prof.triplane_calls +|= 1; + return @ptrFromInt(ret); + } + const ret = triplane_hook.callOriginal(.{ a, b, c, d, e }); + prof.triplane_cycles +|= rdtsc() - s; + prof.triplane_calls +|= 1; + return ret; +} +fn partsetupDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = partsetup_hook.callOriginal(.{ a, b, c }); + prof.partsetup_cycles +|= rdtsc() - s; + prof.partsetup_calls +|= 1; + return ret; +} +fn matmulDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + const s = rdtsc(); + if (ab_use_custom) { + // fastcall: a=ECX=result, b=EDX=left, c=right + const ret = multiplyMatrix4x4(a, b, c); + prof.matmul_cycles +|= rdtsc() - s; + prof.matmul_calls +|= 1; + return @ptrFromInt(ret); + } + const ret = matmul_hook.callOriginal(.{ a, b, c }); + prof.matmul_cycles +|= rdtsc() - s; + prof.matmul_calls +|= 1; + return ret; +} +fn textlineDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque { + const s = rdtsc(); + const ret = textline_hook.callOriginal(.{ a, b, c, d, e, f }); + prof.textline_cycles +|= rdtsc() - s; + prof.textline_calls +|= 1; + return ret; +} + +// ============================================================================= +// Hook: blit_hub (0x5a4f60) +// __fastcall(ECX=int* vec2size, EDX=unknownFuncIndex, +// stack: srcAddr, srcStep, srcFormat, dstAddr, dstStep, dstFormat) +// RET 0x18 (6 stack params) +// Assembly-verified: PUSH EBP; MOV EBP,ESP; MOV ESI,EDX; MOV EDI,ECX; RET 0x18 +// ============================================================================= + +const BlitHubFn = fn (u32, u32, u32, u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) void; +const BlitHubPtr = *const BlitHubFn; +var blit_hub_hook: hook.Detour(BlitHubFn) = .{}; +var unitxp_blit: ?BlitHubPtr = null; // UnitXP's detour, captured before clobber + +fn blitHubDetour(vec2size: u32, func_index: u32, src_addr: u32, src_step: u32, src_fmt: u32, dst_addr: u32, dst_step: u32, dst_fmt: u32) callconv(hook.cc.fastcall) void { + asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); + + const start = rdtsc(); + if (ab_use_custom) { + // CUSTOM: call UnitXP's optimized blit (if present, else original) + if (unitxp_blit) |uxp| { + @call(.never_tail, uxp, .{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt }); + } else { + blit_hub_hook.callOriginal(.{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt }); + } + } else { + // BASELINE: true original function + blit_hub_hook.callOriginal(.{ vec2size, func_index, src_addr, src_step, src_fmt, dst_addr, dst_step, dst_fmt }); + } + const elapsed = rdtsc() - start; + if (ab_use_custom) { + blit_total_custom.cycles +|= elapsed; + blit_total_custom.calls +|= 1; + } else { + blit_total_baseline.cycles +|= elapsed; + blit_total_baseline.calls +|= 1; + } +} + +// ============================================================================= +// Lua API: SetWeatherOverride(type, intensity) +// Calls SetWeatherType (0x67baf0) directly on the global weather object at 0xC6326C. +// __thiscall(ECX=weatherObj, stack: type(int), intensity(float), smoothFade(bool)) +// RET 0x0C — assembly-verified: MOV ESI,ECX; RET 0x0C +// type: 0=clear, 1=rain, 2=snow, 3=sandstorm +// intensity: 0.0-1.0 +// smoothFade: 1=gradual transition, 0=abrupt (immediate). We use 0. +// ============================================================================= + +pub fn luaSetWeatherOverride(L: *anyopaque) callconv(.c) u32 { + const L_ptr = @intFromPtr(L); + const nargs = hook.call(fn (usize) callconv(hook.cc.fastcall) i32, 0x6F3070, .{L_ptr}); // lua_gettop + if (nargs < 2) return 0; + + const weather_type: i32 = @intFromFloat(hook.call(fn (usize, i32) callconv(hook.cc.fastcall) f64, 0x6F3620, .{ L_ptr, 1 })); // lua_tonumber + const intensity: f32 = @floatCast(hook.call(fn (usize, i32) callconv(hook.cc.fastcall) f64, 0x6F3620, .{ L_ptr, 2 })); + + const weather_obj = hook.readMem(u32, 0x00C6326C); + if (weather_obj == 0) return 0; + + // SetWeatherType: __thiscall(ECX=weatherObj, stack: type, intensity, smoothFade) + // fastcall mapping: ECX=this, EDX=unused, stack: type, intensity_bits, smoothFade + const intensity_bits: u32 = @bitCast(intensity); + hook.call(fn (u32, u32, u32, u32, u32) callconv(hook.cc.fastcall) void, 0x67baf0, .{ + weather_obj, 0, @as(u32, @bitCast(weather_type)), intensity_bits, 0, + }); + + return 0; +} + // ============================================================================= // Unified stats dump // ============================================================================= @@ -241,7 +1089,6 @@ fn dumpStats() void { // transformMatrix4x4 const t44_real = prof.t44_calls -| prof.t44_early; - const t44_avg = if (t44_real > 0) prof.t44_cycles / t44_real else 0; const t44_avg_bones = if (t44_real > 0) prof.t44_bones / t44_real else 0; // Percentages of total frame time (×10 for one decimal place) @@ -251,18 +1098,32 @@ fn dumpStats() void { const rtq_pct = pct(prof.rtq_cycles, wall); const mov_pct = pct(prof.mov_cycles, wall); - // wall/frame in Kcycles → rough ms estimate at ~3GHz: Kcyc / 3000 ≈ ms - const wall_per_f = wall / f / 1000; + // @3GHz: cycles / 3_000_000 = ms, cycles / 3_000 = us + const MS_DIVISOR = 3_000_000; + const wall_ms = wall / MS_DIVISOR; + const erp_ms = prof.erp_cycles / MS_DIVISOR; + const rf_ms = prof.rf_cycles / MS_DIVISOR; + const t44_ms = prof.t44_cycles / MS_DIVISOR; + const rtq_ms = prof.rtq_cycles / MS_DIVISOR; + const mov_ms = prof.mov_cycles / MS_DIVISOR; + const t44_avg_us = if (t44_real > 0) prof.t44_cycles / t44_real / 3000 else 0; + + const mode: [*:0]const u8 = if (ab_use_custom) "CUSTOM" else "BASELINE"; + + const bt = if (ab_use_custom) blit_total_custom else blit_total_baseline; log.fmt( - \\[prof] {d} frames, {d}Kcyc/frame (~{d}.{d}ms @3GHz) + \\[prof:{s}] {d} frames, wall={d}ms + \\ ms: erp={d} rf={d} t44={d} rtq={d} mov={d} \\ %frame: erp={d}.{d}% rf={d}.{d}% t44={d}.{d}% rtq={d}.{d}% mov={d}.{d}% \\ calls/f: erp={d} rf={d} t44={d}({d}skip) rtq={d} mov={d} - \\ t44: {d}cyc/work bones={d}/{d} depth={d} | rtq_items/f={d} + \\ t44: {d}us/work bones={d}/{d} depth={d} | rtq_items/f={d} + \\ blit: {d}ms/{d}calls \\ , .{ - f, wall_per_f, - wall_per_f / 3, wall_per_f % 3000 / 300, + mode, + f, wall_ms, + erp_ms, rf_ms, t44_ms, rtq_ms, mov_ms, erp_pct / 10, erp_pct % 10, rf_pct / 10, rf_pct % 10, t44_pct / 10, t44_pct % 10, @@ -274,11 +1135,63 @@ fn dumpStats() void { prof.t44_early / f, prof.rtq_calls / f, prof.mov_calls / f, - t44_avg, + t44_avg_us, t44_avg_bones, prof.t44_max_bones, prof.t44_max_depth, prof.rtq_items / f, + bt.cycles / MS_DIVISOR, bt.calls, }); + // Hotspot table: name, %frame, ms, calls/f + const HotEntry = struct { name: [*:0]const u8, cycles: u64, calls: u64 }; + const hotspots = [_]HotEntry{ + .{ .name = "clip", .cycles = prof.clip_cycles, .calls = prof.clip_calls }, + .{ .name = "glyph", .cycles = prof.glyph_cycles, .calls = prof.glyph_calls }, + .{ .name = "particle", .cycles = prof.particle_cycles, .calls = prof.particle_calls }, + .{ .name = "collision", .cycles = prof.collision_cycles, .calls = prof.collision_calls }, + .{ .name = "entpos", .cycles = prof.entpos_cycles, .calls = prof.entpos_calls }, + .{ .name = "layers", .cycles = prof.layers_cycles, .calls = prof.layers_calls }, + .{ .name = "textvb", .cycles = prof.textvb_cycles, .calls = prof.textvb_calls }, + .{ .name = "complexgeo", .cycles = prof.complexgeo_cycles, .calls = prof.complexgeo_calls }, + .{ .name = "entbounds", .cycles = prof.entbounds_cycles, .calls = prof.entbounds_calls }, + .{ .name = "textctr", .cycles = prof.textctr_cycles, .calls = prof.textctr_calls }, + .{ .name = "spatial", .cycles = prof.spatial_cycles, .calls = prof.spatial_calls }, + .{ .name = "raytri", .cycles = prof.raytri_cycles, .calls = prof.raytri_calls }, + .{ .name = "linkedlist", .cycles = prof.linkedlist_cycles, .calls = prof.linkedlist_calls }, + .{ .name = "color", .cycles = prof.color_cycles, .calls = prof.color_calls }, + .{ .name = "setvec", .cycles = prof.setvec_cycles, .calls = prof.setvec_calls }, + .{ .name = "cull", .cycles = prof.cull_cycles, .calls = prof.cull_calls }, + .{ .name = "colldet", .cycles = prof.colldet_cycles, .calls = prof.colldet_calls }, + .{ .name = "activep", .cycles = prof.activep_cycles, .calls = prof.activep_calls }, + .{ .name = "cbiter", .cycles = prof.cbiter_cycles, .calls = prof.cbiter_calls }, + .{ .name = "findguid", .cycles = prof.findguid_cycles, .calls = prof.findguid_calls }, + .{ .name = "raytri2", .cycles = prof.raytri2_cycles, .calls = prof.raytri2_calls }, + .{ .name = "drawbatch", .cycles = prof.drawbatch_cycles, .calls = prof.drawbatch_calls }, + .{ .name = "findlua", .cycles = prof.findlua_cycles, .calls = prof.findlua_calls }, + .{ .name = "spritequad", .cycles = prof.spritequad_cycles, .calls = prof.spritequad_calls }, + .{ .name = "scenenode", .cycles = prof.scenenode_cycles, .calls = prof.scenenode_calls }, + .{ .name = "terrain", .cycles = prof.terrain_cycles, .calls = prof.terrain_calls }, + .{ .name = "d3dtex", .cycles = prof.d3dtex_cycles, .calls = prof.d3dtex_calls }, + .{ .name = "bboxchk", .cycles = prof.bboxchk_cycles, .calls = prof.bboxchk_calls }, + .{ .name = "rotmat", .cycles = prof.rotmat_cycles, .calls = prof.rotmat_calls }, + .{ .name = "triplane", .cycles = prof.triplane_cycles, .calls = prof.triplane_calls }, + .{ .name = "partsetup", .cycles = prof.partsetup_cycles, .calls = prof.partsetup_calls }, + .{ .name = "matmul", .cycles = prof.matmul_cycles, .calls = prof.matmul_calls }, + .{ .name = "textline", .cycles = prof.textline_cycles, .calls = prof.textline_calls }, + }; + for (hotspots) |h| { + if (h.calls > 0) { + const hp = pct(h.cycles, wall); + log.fmt(" {s}: {d}.{d}% {d}ms {d}c/f\n", .{ + h.name, + hp / 10, hp % 10, + h.cycles / MS_DIVISOR, + h.calls / f, + }); + } + } + + // Flip A/B mode for next period + ab_use_custom = !ab_use_custom; prof = ProfState{}; } @@ -298,7 +1211,69 @@ pub fn installHooks() void { _ = exec_render_pass_hook.attach(0x708900, &execRenderPassDetour); _ = render_quads_hook.attach(0x76FB00, &renderQuadsDetour); _ = movement_hook.attach(0x616620, &movementDetour); - log.print("transform44: 5 profiling hooks installed\n"); + _ = interp_kf_hook.attach(0x713ea0, &interpKfDetour); + + // Perf-identified hotspot hooks + _ = clip_hook.attach(0x6318c0, &clipDetour); + _ = glyph_hook.attach(0x5ca2d0, &glyphDetour); + _ = particle_hook.attach(0x7b2a50, &particleDetour); + _ = collision_hook.attach(0x6abc40, &collisionDetour); + _ = entpos_hook.attach(0x6afad0, &entposDetour); + _ = layers_hook.attach(0x765650, &layersDetour); + _ = textvb_hook.attach(0x5ccbe0, &textvbDetour); + _ = complexgeo_hook.attach(0x58a3d0, &complexgeoDetour); + _ = entbounds_hook.attach(0x6c1f70, &entboundsDetour); + _ = textctr_hook.attach(0x5cdf40, &textctrDetour); + _ = spatial_hook.attach(0x6816f0, &spatialDetour); + _ = raytri_hook.attach(0x7c29f0, &raytriDetour); + _ = linkedlist_hook.attach(0x710b90, &linkedlistDetour); + _ = color_hook.attach(0x7b9b10, &colorDetour); + _ = setvec_hook.attach(0x686640, &setvecDetour); + _ = cull_hook.attach(0x6b8c60, &cullDetour); + _ = colldet_hook.attach(0x6b88e0, &colldetDetour); + _ = activep_hook.attach(0x7b5a10, &activepDetour); + _ = cbiter_hook.attach(0x404130, &cbiterDetour); + _ = findguid_hook.attach(0x464890, &findguidDetour); + _ = raytri2_hook.attach(0x632700, &raytri2Detour); + _ = drawbatch_hook.attach(0x70cb30, &drawbatchDetour); + _ = findlua_hook.attach(0x702000, &findluaDetour); + _ = spritequad_hook.attach(0x5a0f50, &spritequadDetour); + _ = scenenode_hook.attach(0x718960, &scenenodeDetour); + _ = terrain_hook.attach(0x6cffc0, &terrainDetour); + _ = d3dtex_hook.attach(0x593840, &d3dtexDetour); + _ = bboxchk_hook.attach(0x6b8b70, &bboxchkDetour); + _ = rotmat_hook.attach(0x7bdd60, &rotmatDetour); + _ = triplane_hook.attach(0x632460, &triplaneDetour); + _ = partsetup_hook.attach(0x7b3d20, &partsetupDetour); + _ = matmul_hook.attach(0x7bc6a0, &matmulDetour); + _ = textline_hook.attach(0x5ce0c0, &textlineDetour); + + // blit_hub installed in lateInit() to clobber UnitXP's hook + log.print("transform44: 39 profiling hooks installed (blit_hub deferred)\n"); +} + +/// Called from engineInitDetour — after UnitXP has hooked blit_hub. +/// Captures UnitXP's detour address, restores original prologue, then hooks. +pub fn lateInit() void { + if (!g_is_hook_owner) return; + + const BLIT_ADDR = 0x5a4f60; + const src: [*]const u8 = @ptrFromInt(BLIT_ADDR); + + // If UnitXP hooked it, first byte is E9 (relative JMP) + if (src[0] == 0xE9) { + // Decode rel32 target: addr + 5 + *(i32*)(addr+1) + const rel: i32 = @bitCast(hook.readMem(u32, BLIT_ADDR + 1)); + const target: usize = @intCast(@as(i64, @intCast(BLIT_ADDR + 5)) + rel); + unitxp_blit = @ptrFromInt(target); + log.fmt("blit_hub: captured UnitXP detour at 0x{x}\n", .{target}); + } + + // Restore original prologue (from Ghidra disasm), clobbering UnitXP's E9 JMP + // 55 8B EC A1 58 F5 C0 00 = PUSH EBP; MOV EBP,ESP; MOV EAX,[0xc0f558] + hook.writeProtected(BLIT_ADDR, &.{ 0x55, 0x8B, 0xEC, 0xA1, 0x58, 0xF5, 0xC0, 0x00 }); + _ = blit_hub_hook.attach(BLIT_ADDR, &blitHubDetour); + log.print("blit_hub: hooked (true original baseline)\n"); } pub fn removeHooks() void { @@ -308,6 +1283,41 @@ pub fn removeHooks() void { exec_render_pass_hook.detach(); render_quads_hook.detach(); movement_hook.detach(); + interp_kf_hook.detach(); + clip_hook.detach(); + glyph_hook.detach(); + particle_hook.detach(); + collision_hook.detach(); + entpos_hook.detach(); + layers_hook.detach(); + textvb_hook.detach(); + complexgeo_hook.detach(); + entbounds_hook.detach(); + textctr_hook.detach(); + spatial_hook.detach(); + raytri_hook.detach(); + linkedlist_hook.detach(); + color_hook.detach(); + setvec_hook.detach(); + cull_hook.detach(); + colldet_hook.detach(); + activep_hook.detach(); + cbiter_hook.detach(); + findguid_hook.detach(); + raytri2_hook.detach(); + drawbatch_hook.detach(); + findlua_hook.detach(); + spritequad_hook.detach(); + scenenode_hook.detach(); + terrain_hook.detach(); + d3dtex_hook.detach(); + bboxchk_hook.detach(); + rotmat_hook.detach(); + triplane_hook.detach(); + partsetup_hook.detach(); + matmul_hook.detach(); + textline_hook.detach(); + blit_hub_hook.detach(); log.close(); mod_mutex.release(&g_mutex); }