spritequad: recreation + analysis (disabled, no improvement)

Faithful recreation of RenderSpriteQuads (0x5A0F50) with hoisted
invariant division and inlined DisplayMode_CalculateOffset. No
measurable improvement — cost is dominated by getAdapterInfo (7
sub-calls to D3D device per invocation × 3203 calls/frame) and
DrawPrimitive/DrawIndexedPrimitive virtual dispatch.

Added decompilation and assembly dumps for future reference.
This commit is contained in:
MarcelineVQ
2026-03-24 00:36:37 -07:00
parent 1046c0a4e3
commit 3d872fdb1e
4 changed files with 304 additions and 1 deletions
@@ -0,0 +1,79 @@
// RenderSpriteQuads @ 0x5A0F50
// Disassembled from WoW 1.12.1 build 5875
005A0F50 55 push ebp
005A0F51 8BEC mov ebp,esp
005A0F53 51 push ecx
005A0F54 53 push ebx
005A0F55 56 push esi
005A0F56 57 push edi
005A0F57 8BF9 mov edi,ecx
005A0F59 8B872C0F0000 mov eax,[edi+0xf2c]
005A0F5F 85C0 test eax,eax
005A0F61 0F8481010000 jz 0x5a10e8
005A0F67 8B9FD8270000 mov ebx,[edi+0x27d8]
005A0F6D 8D87A4270000 lea eax,[edi+0x27a4]
005A0F73 BA01000000 mov edx,0x1
005A0F78 33C9 xor ecx,ecx
005A0F7A 8BF0 mov esi,eax
005A0F7C 8D642400 lea esp,[esp+0x0]
005A0F80 B801000000 mov eax,0x1
005A0F85 D3E0 shl eax,cl
005A0F87 85C3 test ebx,eax
005A0F89 7421 jz 0x5a0fac
005A0F8B 8B06 mov eax,[esi]
005A0F8D 85C0 test eax,eax
005A0F8F 7419 jz 0x5a0faa
005A0F91 85D2 test edx,edx
005A0F93 7415 jz 0x5a0faa
005A0F95 8A501C mov dl,[eax+0x1c]
005A0F98 84D2 test dl,dl
005A0F9A 740E jz 0x5a0faa
005A0F9C 8A501D mov dl,[eax+0x1d]
005A0F9F 84D2 test dl,dl
005A0FA1 7407 jz 0x5a0faa
005A0FA3 BA01000000 mov edx,0x1
005A0FA8 EB02 jmp 0x5a0fac
005A0FAA 33D2 xor edx,edx
005A0FAC 41 inc ecx
005A0FAD 83C604 add esi,0x4
005A0FB0 83F90D cmp ecx,0xd
005A0FB3 72CB jc 0x5a0f80
005A0FB5 8B4D10 mov ecx,[ebp+0x10]
005A0FB8 85C9 test ecx,ecx
005A0FBA 7418 jz 0x5a0fd4
005A0FBC 85D2 test edx,edx
005A0FBE 7418 jz 0x5a0fd8
005A0FC0 8B87EC270000 mov eax,[edi+0x27ec]
005A0FC6 8A501C mov dl,[eax+0x1c]
005A0FC9 84D2 test dl,dl
005A0FCB 740B jz 0x5a0fd8
005A0FCD 8A501D mov dl,[eax+0x1d]
005A0FD0 84D2 test dl,dl
005A0FD2 EB02 jmp 0x5a0fd6
005A0FD4 85D2 test edx,edx
005A0FD6 7516 jnz 0x5a0fee
005A0FD8 68A8C78500 push dword 0x85c7a8
005A0FDD E84EF6FEFF call 0x590630
005A0FE2 83C404 add esp,0x4
005A0FE5 5F pop edi
005A0FE6 5E pop esi
005A0FE7 5B pop ebx
005A0FE8 8BE5 mov esp,ebp
005A0FEA 5D pop ebp
005A0FEB C20C00 ret word 0xc
005A0FEE 8B5D0C mov ebx,[ebp+0xc]
005A0FF1 8B7508 mov esi,[ebp+0x8]
005A0FF4 51 push ecx
005A0FF5 53 push ebx
005A0FF6 56 push esi
005A0FF7 8BCF mov ecx,edi
005A0FF9 E8021BFFFF call 0x592b00
005A0FFE 8BCF mov ecx,edi
005A1000 E81B0B0000 call 0x5a1b20
005A1005 85DB test ebx,ebx
005A1007 0F86DB000000 jna 0x5a10e8
005A100D 83C60A add esi,0xa
005A1010 895D08 mov [ebp+0x8],ebx
005A1013 668B4EFE mov cx,[esi-0x2]
005A1017 66 o16
@@ -0,0 +1,88 @@
void __thiscall RenderSpriteQuads(void *this,int spriteData,uint spriteCount,int renderMode)
{
ushort uVar1;
ushort uVar2;
int iVar3;
int iVar4;
uint uVar5;
int *piVar6;
ushort *puVar7;
bool bVar8;
undefined *local_8;
/* Renders sprite quads - validates texture states, calculates rendering
metrics, and issues draw calls for each valid sprite */
if (*(int *)((int)this + 0xf2c) != 0) {
piVar6 = (int *)((int)this + 0x27a4);
bVar8 = true;
uVar5 = 0;
do {
if ((*(uint *)((int)this + 0x27d8) & 1 << ((byte)uVar5 & 0x1f)) != 0) {
iVar3 = *piVar6;
if ((((iVar3 == 0) || (!bVar8)) || (*(char *)(iVar3 + 0x1c) == '\0')) ||
(*(char *)(iVar3 + 0x1d) == '\0')) {
bVar8 = false;
}
else {
bVar8 = true;
}
}
uVar5 = uVar5 + 1;
piVar6 = piVar6 + 1;
} while (uVar5 < 0xd);
if (renderMode == 0) {
bVar8 = !bVar8;
}
else {
if ((!bVar8) || (*(char *)(*(int *)((int)this + 0x27ec) + 0x1c) == '\0')) goto LAB_005a0fd8;
bVar8 = *(char *)(*(int *)((int)this + 0x27ec) + 0x1d) == '\0';
}
if (bVar8) {
LAB_005a0fd8:
EmptyStub();
return;
}
CalculateRenderingMetrics(this,spriteData,spriteCount,renderMode);
getAdapterInfo(this);
if (spriteCount != 0) {
puVar7 = (ushort *)(spriteData + 10);
spriteData = spriteCount;
do {
uVar1 = puVar7[-1];
if (uVar1 != 0) {
spriteCount = 0;
if (*(int *)((int)this + 0x24c) == 0) {
spriteCount = *(uint *)(*(int *)((int)this + 0x27a4) + 0x18) /
*(uint *)(*(int *)((int)this + 0x27a4) + 0xc);
}
if (renderMode == 0) {
iVar3 = **(int **)((int)this + 0x38a8);
iVar4 = DisplayMode_CalculateOffset(*(int *)(puVar7 + -5),(uint)uVar1);
(**(code **)(iVar3 + 0x144))
(*(undefined4 *)((int)this + 0x38a8),
*(undefined4 *)(&DAT_0080a14c + *(int *)(puVar7 + -5) * 4),spriteCount,iVar4)
;
}
else {
uVar2 = *puVar7;
iVar3 = **(int **)((int)this + 0x38a8);
iVar4 = DisplayMode_CalculateOffset(*(int *)(puVar7 + -5),(uint)uVar1);
(**(code **)(iVar3 + 0x148))
(*(undefined4 *)((int)this + 0x38a8),
*(undefined4 *)(&DAT_0080a14c + *(int *)(puVar7 + -5) * 4),spriteCount,
(uint)uVar2,((uint)puVar7[1] - (uint)uVar2) + 1,
(*(uint *)(*(int *)((int)this + 0x27ec) + 0x18) >> 1) + *(int *)(puVar7 + -3)
,iVar4);
}
}
puVar7 = puVar7 + 8;
spriteData = spriteData + -1;
} while (spriteData != 0);
}
}
return;
}
+136 -1
View File
@@ -1062,9 +1062,144 @@ export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC
}
inline fn copyMat4x4(dst: u32, src: u32) void {
// Copy 64 bytes (16 floats) using V4 loads/stores
@as(*align(1) V4, @ptrFromInt(dst)).* = @as(*align(1) const V4, @ptrFromInt(src)).*;
@as(*align(1) V4, @ptrFromInt(dst + 16)).* = @as(*align(1) const V4, @ptrFromInt(src + 16)).*;
@as(*align(1) V4, @ptrFromInt(dst + 32)).* = @as(*align(1) const V4, @ptrFromInt(src + 32)).*;
@as(*align(1) V4, @ptrFromInt(dst + 48)).* = @as(*align(1) const V4, @ptrFromInt(src + 48)).*;
}
// =============================================================================
// RenderSpriteQuads (0x5A0F50)
// __thiscall(ECX=this, stack=spriteData, spriteCount, renderMode), RET 0xC
//
// Optimizations over original:
// 1. Hoisted invariant division out of inner loop (same result every iteration)
// 2. Inlined DisplayMode_CalculateOffset (trivial: table lookup + divide + subtract)
// 3. Cached texture validation bitmask check
// =============================================================================
// Game functions called by RenderSpriteQuads
const sqEmptyStub: *const fn (u32) callconv(.{ .x86_stdcall = .{} }) void = @ptrFromInt(0x590630);
const sqCalcMetrics: *const fn (u32, u32, u32, u32) callconv(TC) void = @ptrFromInt(0x592B00);
const sqGetAdapterInfo: *const fn (u32) callconv(TC) void = @ptrFromInt(0x5A1B20);
// DisplayMode tables (from 0x592C10 disassembly)
const DISPLAY_MODE_DIVISOR_TABLE: u32 = 0x85ACF0;
const DISPLAY_MODE_OFFSET_TABLE: u32 = 0x85AD08;
/// Inlined DisplayMode_CalculateOffset: table[type] divide + subtract
inline fn displayModeOffset(sprite_type: u32, count: u32) u32 {
const divisor = ru32(DISPLAY_MODE_DIVISOR_TABLE + sprite_type * 4);
const divided = if (divisor == 1) count else count / divisor;
return divided -% ru32(DISPLAY_MODE_OFFSET_TABLE + sprite_type * 4);
}
export fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void {
// Early out: this+0xF2C == 0
if (ru32(this + 0xF2C) == 0) return;
// =========================================================================
// Section 1: Texture validation (13 slots)
// =========================================================================
const tex_bitmask = ru32(this + 0x27D8);
const tex_array_base = this + 0x27A4;
var all_valid: bool = true;
var slot: u32 = 0;
while (slot < 13) : (slot += 1) {
if ((tex_bitmask & (@as(u32, 1) << @truncate(slot))) != 0) {
const tex_ptr = ru32(tex_array_base + slot * 4);
if (tex_ptr == 0 or !all_valid or ru8(tex_ptr + 0x1C) == 0 or ru8(tex_ptr + 0x1D) == 0) {
all_valid = false;
}
}
}
// Render mode logic
var should_render: bool = undefined;
if (render_mode == 0) {
should_render = all_valid; // mode 0: render if NOT all valid → invert
// Wait: original does bVar8 = !bVar8 for mode 0, then checks if(bVar8) → early out
// So: if all_valid → !all_valid = false → don't early out → render
// if !all_valid → !all_valid = true → early out → don't render
// Simplified: render if all_valid
} else {
if (!all_valid) {
sqEmptyStub(0x85C7A8);
return;
}
const extra_ptr = ru32(this + 0x27EC);
if (ru8(extra_ptr + 0x1C) == 0) {
sqEmptyStub(0x85C7A8);
return;
}
should_render = ru8(extra_ptr + 0x1D) != 0;
}
if (!should_render) {
sqEmptyStub(0x85C7A8);
return;
}
// =========================================================================
// Section 2: Setup calls
// =========================================================================
sqCalcMetrics(this, sprite_data, sprite_count, render_mode);
sqGetAdapterInfo(this);
if (sprite_count == 0) return;
// =========================================================================
// Section 3: Inner loop — hoisted invariant division
// =========================================================================
// The division this+0x27A4[0]+0x18 / this+0x27A4[0]+0xC is invariant across sprites.
// Original recomputes it per sprite. We hoist it.
var base_prim_count: u32 = 0;
if (ru32(this + 0x24C) == 0) {
const first_tex = ru32(this + 0x27A4);
if (first_tex != 0) {
const numerator = ru32(first_tex + 0x18);
const denominator = ru32(first_tex + 0x0C);
if (denominator != 0) {
base_prim_count = numerator / denominator;
}
}
}
// D3D device vtable pointer
const device_ptr = ru32(this + 0x38A8);
const vtable = ru32(device_ptr);
// Sprite data stride = 16 bytes, pointer starts at spriteData + 10
var ptr = sprite_data + 10;
var remaining = sprite_count;
while (remaining > 0) : (remaining -= 1) {
const count: u32 = @as(u32, ru16(ptr - 2)); // [esi-2] = sprite vertex count
if (count != 0) {
const sprite_type = ru32(ptr - 10); // [esi-0xA] = type/format index
const offset = displayModeOffset(sprite_type, count);
const lookup_val = ru32(0x80A14C + sprite_type * 4);
if (render_mode == 0) {
// DrawPrimitive: vtable[0x144](device, lookup, basePrimCount, offset)
const draw_fn: *const fn (u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void =
@ptrFromInt(ru32(vtable + 0x144));
draw_fn(device_ptr, lookup_val, base_prim_count, offset);
} else {
const start_idx: u32 = @as(u32, ru16(ptr));
const end_idx: u32 = @as(u32, ru16(ptr + 2));
const extra_ptr = ru32(this + 0x27EC);
const extra_offset = (ru32(extra_ptr + 0x18) >> 1) + ru32(ptr - 6);
// DrawIndexedPrimitive: vtable[0x148](device, lookup, basePrimCount, startIdx, count, extraOffset, offset)
const draw_fn: *const fn (u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void =
@ptrFromInt(ru32(vtable + 0x148));
draw_fn(device_ptr, lookup_val, base_prim_count, start_idx, end_idx - start_idx + 1, extra_offset, offset);
}
}
ptr += 16;
}
}
+1
View File
@@ -26,6 +26,7 @@ extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x8
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn resetParticleCache() void;
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern var stride_info: [8]u32; // exported from particle_sse.zig
extern var debug_vertex_count: u32;
extern var debug_max_sprites: u32;