From d205693dbb14cec7b59d72597e9161f37e816d1c Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Mon, 23 Mar 2026 21:56:57 -0700 Subject: [PATCH] particle: Ghidra decompilations, particle_sse.zig scaffold, bench MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Ghidra C decompilation of RenderParticleSprites (422 lines) and 5 helper functions (calculateColorValues, matVec3Transform, etc.) - particle_sse.zig with calcColorValues_SSE (10.7x bench but cache-miss bound in-game — needs inlining into full function replacement) - Bench harness for calcColorValues with correctness check - build.zig: particle_sse as separate ReleaseFast compilation unit - colorDetour reverted to pass-through (SSE has no in-game effect due to L1 cache misses on scattered ColorCtx structs) --- build.zig | 25 +- src/bench/main.zig | 97 ++++ .../decompiled/decomp_RenderParticleSprites.c | 422 ++++++++++++++++++ .../decompiled/decomp_particle_helpers.c | 241 ++++++++++ src/transform44/particle_sse.zig | 174 ++++++++ src/transform44/transform44.zig | 4 + 6 files changed, 962 insertions(+), 1 deletion(-) create mode 100644 src/transform44/decompiled/decomp_RenderParticleSprites.c create mode 100644 src/transform44/decompiled/decomp_particle_helpers.c create mode 100644 src/transform44/particle_sse.zig diff --git a/build.zig b/build.zig index 27a5e67..97b5d0a 100644 --- a/build.zig +++ b/build.zig @@ -28,7 +28,7 @@ const module_list = [_]ModuleDesc{ .{ .name = "healtextfix", .desc = "Enable SuperWoW heal text fix" }, .{ .name = "bigcursor", .desc = "Enable big cursor module" }, .{ .name = "clickthrough", .desc = "Enable GO click-through (enlarge GO model bounds)" }, - .{ .name = "dpslog", .desc = "Enable structured combat log events for addons" }, + .{ .name = "dpslog", .desc = "Enable structured combat log events for addons", .default = false }, .{ .name = "transform44", .desc = "Enable transformMatrix4x4 hook", .default = false }, .{ .name = "addonperf", .desc = "Enable addon memory/CPU profiling API", .default = false }, .{ .name = "filecache", .desc = "Enable MPQ archive file cache" }, @@ -112,6 +112,15 @@ pub fn build(b: *std.Build) void { }), }); + const particle_sse_obj = b.addObject(.{ + .name = "particle_sse", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/transform44/particle_sse.zig"), + .target = bone_sse_target, + .optimize = .ReleaseFast, + }), + }); + const lib = b.addLibrary(.{ .name = "weirdutils", .linkage = .dynamic, @@ -130,6 +139,7 @@ pub fn build(b: *std.Build) void { lib.root_module.addObject(bone_sse_ref_obj); lib.root_module.addObject(math_sse_obj); lib.root_module.addObject(silicon_sse_obj); + lib.root_module.addObject(particle_sse_obj); b.installArtifact(lib); // Benchmark harness — native x86 Linux executable for profiling SSE replacements @@ -195,10 +205,23 @@ pub fn build(b: *std.Build) void { .optimize = .ReleaseFast, }), }); + const bench_particle_sse = b.addObject(.{ + .name = "bench_particle_sse", + .root_module = b.createModule(.{ + .root_source_file = b.path("src/transform44/particle_sse.zig"), + .target = b.resolveTargetQuery(.{ + .cpu_arch = .x86, + .os_tag = .linux, + .cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }), + }), + .optimize = .ReleaseFast, + }), + }); bench.root_module.addObject(bench_math_sse); bench.root_module.addObject(bench_silicon_sse); bench.root_module.addObject(bench_bone_sse); bench.root_module.addObject(bench_bone_baseline); + bench.root_module.addObject(bench_particle_sse); bench.root_module.linkSystemLibrary("m", .{}); const install_bench = b.addInstallArtifact(bench, .{}); const bench_step = b.step("bench", "Build math_sse benchmark harness (x86 Linux)"); diff --git a/src/bench/main.zig b/src/bench/main.zig index 5c718d1..1900323 100644 --- a/src/bench/main.zig +++ b/src/bench/main.zig @@ -51,6 +51,7 @@ extern fn si_vec3Dot(u32, u32) callconv(cc_fc) f64; extern fn si_translateBoundingVol(u32, u32) callconv(cc_tc) void; extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(cc_fc) u32; extern fn si_frustumCullBBox(u32, u32, u32) callconv(cc_fc) u32; +extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(cc_tc) void; extern fn si_addVec3ToAccumulator(u32, u32) callconv(cc_tc) void; extern fn si_addToColorAccumulator(u32, u32) callconv(cc_tc) void; extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void; @@ -1769,6 +1770,9 @@ pub fn main() void { } } + // calcColorValues_SSE -- thiscall(ctx_ECX, time, scale, outColor, outAlpha1, outAlpha2, outFloat) + bench_calcColorValues(); + // si_frustumCullBBox -- fastcall(bbox_ECX, flags_EDX, radius_stack) -> u32 bench_frustumCullBBox(); @@ -1779,6 +1783,99 @@ pub fn main() void { print("\n", .{}); } +fn bench_calcColorValues() void { + // Map pages for global constants used by calculateColorValues + // 0x808AAC and 0x807A3C are in .rdata range (already mapped) + // 0x8029CC is in .rdata range (already mapped) + // 0x8015B8 is in .rdata range (already mapped) — pow exponent constant + + // Build fake ColorCtx struct + // Layout: +0x00..0x03 = base bytes [B,G,R,A], +0x04..0x10 = deltas (4×i32), + // +0x14..0x20 = alpha base/delta pairs (4×i32), +0x24 = float_base(f32), + // +0x28 = float_scale(f32), +0x2C = time_base(f32), +0x30 = time_scale(f32), + // +0x50 = alpha_power(f32) + var ctx: [0x54]u8 align(4) = std.mem.zeroes([0x54]u8); + // Base color: BGRA = {100, 150, 200, 220} + ctx[0] = 100; ctx[1] = 150; ctx[2] = 200; ctx[3] = 220; + // Deltas (i32): small values + @as(*align(1) i32, @ptrCast(ctx[0x04..0x08])).* = 10; + @as(*align(1) i32, @ptrCast(ctx[0x08..0x0C])).* = -5; + @as(*align(1) i32, @ptrCast(ctx[0x0C..0x10])).* = 8; + @as(*align(1) i32, @ptrCast(ctx[0x10..0x14])).* = -3; + // Alpha base/delta + @as(*align(1) i32, @ptrCast(ctx[0x14..0x18])).* = 200; + @as(*align(1) i32, @ptrCast(ctx[0x18..0x1C])).* = 20; + @as(*align(1) i32, @ptrCast(ctx[0x1C..0x20])).* = 180; + @as(*align(1) i32, @ptrCast(ctx[0x20..0x24])).* = 15; + // Float base/scale + @as(*align(1) f32, @ptrCast(ctx[0x24..0x28])).* = 1.0; + @as(*align(1) f32, @ptrCast(ctx[0x28..0x2C])).* = 0.5; + // Time base/scale + @as(*align(1) f32, @ptrCast(ctx[0x2C..0x30])).* = 0.0; + @as(*align(1) f32, @ptrCast(ctx[0x30..0x34])).* = 1.0; + // Alpha power = 1.0 (linear, fast path) + @as(*align(1) f32, @ptrCast(ctx[0x50..0x54])).* = 1.0; + + const time: f32 = 0.5; + const scale: f32 = 1.0; + var out_color_o: [4]u8 = .{0} ** 4; + var out_color_s: [4]u8 = .{0} ** 4; + var out_alpha1_o: u32 = 0; + var out_alpha1_s: u32 = 0; + var out_alpha2_o: u32 = 0; + var out_alpha2_s: u32 = 0; + var out_float_o: f32 = 0; + var out_float_s: f32 = 0; + + // Original: __thiscall(ECX=ctx, stack: time, scale, outColor, outAlpha1, outAlpha2, outFloat), RET 0x18 + const of = origFn(fn (u32, u32, u32, u32, u32, u32, u32) callconv(cc_tc) void, 0x7B9B10); + of(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_o), a(&out_alpha1_o), a(&out_alpha2_o), a(&out_float_o)); + calcColorValues_SSE(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_s), a(&out_alpha1_s), a(&out_alpha2_s), a(&out_float_s)); + + // The original returns float in ST(0) which we need to pop to avoid FPU stack leak + // Pop it after each call in the bench loop too + const ok = out_color_o[0] == out_color_s[0] and out_color_o[1] == out_color_s[1] and + out_color_o[2] == out_color_s[2] and out_color_o[3] == out_color_s[3] and + out_alpha1_o == out_alpha1_s and out_alpha2_o == out_alpha2_s and + compareF32(out_float_o, out_float_s); + if (!ok) { + print(" color bytes: orig=[{d},{d},{d},{d}] sse=[{d},{d},{d},{d}]\n", .{ + out_color_o[0], out_color_o[1], out_color_o[2], out_color_o[3], + out_color_s[0], out_color_s[1], out_color_s[2], out_color_s[3], + }); + print(" alpha1: orig={d} sse={d} alpha2: orig={d} sse={d}\n", .{ + out_alpha1_o, out_alpha1_s, out_alpha2_o, out_alpha2_s, + }); + print(" float: orig=0x{x} sse=0x{x}\n", .{ + @as(u32, @bitCast(out_float_o)), @as(u32, @bitCast(out_float_s)), + }); + } + + // Original returns float in ST(0) — must pop to avoid FPU stack overflow in bench loop + var t: u64 = std.math.maxInt(u64); + for (0..5) |_| { + const _t0 = rdtsc(); + for (0..ITERS) |_| { + of(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_o), a(&out_alpha1_o), a(&out_alpha2_o), a(&out_float_o)); + // Pop ST(0) to prevent FPU stack overflow + asm volatile ("fstp %%st(0)" ::: "st"); + } + const _te = rdtsc() - _t0; + if (_te < t) t = _te; + } + + var s: u64 = std.math.maxInt(u64); + for (0..5) |_| { + const _t0 = rdtsc(); + for (0..ITERS) |_| { + calcColorValues_SSE(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_s), a(&out_alpha1_s), a(&out_alpha2_s), a(&out_float_s)); + } + const _te = rdtsc() - _t0; + if (_te < s) s = _te; + } + report("calcColorValues", t, s, ok); +} + fn bench_frustumCullBBox() void { // Map runtime global pages for view-proj matrices, occlusion buffer, and flags _ = mapZeroed(0xC7B000, 0x20000); // covers 0xC7B000-0xC7D000+ (matrices, horizon buffer, globals) diff --git a/src/transform44/decompiled/decomp_RenderParticleSprites.c b/src/transform44/decompiled/decomp_RenderParticleSprites.c new file mode 100644 index 0000000..acd7857 --- /dev/null +++ b/src/transform44/decompiled/decomp_RenderParticleSprites.c @@ -0,0 +1,422 @@ +openjdk version "21.0.10" 2026-01-20 +OpenJDK Runtime Environment (build 21.0.10+7) +OpenJDK 64-Bit Server VM (build 21.0.10+7, mixed mode) +// RenderParticleSprites @ 0x7B2A50 +// Decompiled by Ghidra + + +/* WARNING: Globals starting with '_' overlap smaller symbols at the same address */ + +undefined * __thiscall +RenderParticleSprites(ParticleSystemRenderer *this,float *particleData,float **vertexBuffers) + +{ + undefined *puVar1; + int loopIndex; + uint texLoopIndex; + float *particleDataPtr; + int nextLoopIndex; + undefined **textureOffsetPtr; + undefined **textureVPtr; + uint colorDeltaR; + float10 sinRotation; + float *rotMatrix_m00; + float *rotMatrix_m01; + float *rotMatrix_m02; + float *rotMatrix_m10; + float *rotMatrix_m11; + float *rotMatrix_m12; + float *rotMatrix_m20; + float *rotMatrix_m21; + float *rotMatrix_m22; + float *transformedVelocityVector; + float *transformedVertexZ; + float *negVelocityX; + float *transformedVertexX; + float *negVelocityZ; + float *negVelocityY; + uint *colorData1; + uint *colorData2; + float *textureU; + float *textureV; + float *particleTimeIndex; + float worldPosX; + float worldPosY; + float worldPosZ; + float *vertexX; + float *vertexY; + float *vertexZ; + uint *colorValue; + float *rotationAngle; + float spriteScale; + float *particleFlags; + float textureUIndex; + float textureVIndex; + float clampedParticleTime; + undefined *clampedTimePtr; + float10 cosRotation; + float cosValue; + float scaleFactor; + float tempFloat1; + float textureCoordV; + float velocityDirection; + float *vertexBuffer; + float **vertexBufferPtr; + float zDepth; + + particleDataPtr = particleData; + colorDeltaR = 0; + /* Reading colorDeltaB+2 - likely wrong field mapping */ + /* Reading texCoordBase1+2 - another +2 offset pattern */ + if (((float)this->colorPaletteArray[2].renderFlags_source < StaticFloat1_0) || + (*(float *)&this->colorPaletteArray[2].renderFlags_prefix != + (float)COLLISION_PLANE_ZERO_THRESHOLD)) { + puVar1 = (undefined *)(this->colorPaletteArray[2].texCoordDelta2 * particleData[7]); + clampedTimePtr = COLLISION_PLANE_ZERO_THRESHOLD; + if (((float)COLLISION_PLANE_ZERO_THRESHOLD <= (float)puVar1) && + (clampedTimePtr = puVar1, (float)_DAT_007ffe58 <= (float)puVar1)) { + clampedTimePtr = _DAT_007ffe58; + } + particleTimeIndex = (float *)((float)clampedTimePtr + _DAT_008029cc); + colorDeltaR = ((uint)particleTimeIndex >> 0xe) + ((uint)particleData >> 5) & 0x7f; + } + if (((float)this->colorPaletteArray[2].renderFlags_source < StaticFloat1_0) && + ((float)this->colorPaletteArray[2].renderFlags_source < + *(float *)(&g_particleDepthBuffer + colorDeltaR * 4))) { + return (undefined *)0x0; + } + colorValue = (uint *)0x0; + calculateParticleColorAndScale + ((OrientationData *) + (this->orientationDataArray + (uint)*(byte *)(particleData + 3) * 0x60 + -0x12), + particleData[7],*(float *)((int)&this->colorPaletteArray[2].colorDeltaR + 2), + (byte *)&colorValue,(uint *)&colorData1,(uint *)&colorData2,&spriteScale); + loopIndex = UpdateLightingOffset(); + if (*(int *)(loopIndex + 0x1c) == 1) { + colorValue = (uint *)CONCAT31(CONCAT21(CONCAT11((char)((uint)colorValue >> 0x18), + (char)colorValue),(char)((uint)colorValue >> 8)) + ,(char)((uint)colorValue >> 0x10)); + rotationAngle = (float *)colorValue; + } + if (*(float *)&this->colorPaletteArray[2].renderFlags_prefix != + (float)COLLISION_PLANE_ZERO_THRESHOLD) { + spriteScale = (*(float *)(&g_particleDepthBuffer + colorDeltaR * 4) * + *(float *)&this->colorPaletteArray[2].renderFlags_prefix + + *(float *)&this->colorPaletteArray[2].field_0x1e_source) * spriteScale; + } + colorDeltaR._0_2_ = this->colorPaletteArray[2].padding_06; + colorDeltaR._2_2_ = this->colorPaletteArray[2].texCoordDelta2_prefix; + if ((colorDeltaR & 0x200) != 0) { + spriteScale = spriteScale * *(float *)(this->colorPaletteArray[3].padding_50_53 + 2); + } + transformVector3ByMatrix4x4(&worldPosX,particleDataPtr,(float *)&g_worldMatrix); + vertexBufferPtr = vertexBuffers; + texLoopIndex._0_2_ = this->colorPaletteArray[2].padding_06; + texLoopIndex._2_2_ = this->colorPaletteArray[2].texCoordDelta2_prefix; + if ((texLoopIndex & 4) != 0) { + textureUIndex = + (float)(*(int *)(this->colorPaletteArray[1].final_padding + 6) - 1U & (uint)colorData1); + textureVIndex = 0.0; + textureU = (float *)((float)(uint)textureUIndex * this->textureScaleU); + textureV = (float *)((float)((int)colorData1 >> (SUB41(this->uvCoordinateScale,0) & 0x1f)) * + this->textureScaleV); + if (this->colorPaletteArray[1].texCoordBase1 == (float)COLLISION_PLANE_ZERO_THRESHOLD) { + if ((texLoopIndex & 0x2000) == 0) { + loopIndex = 0; + do { + vertexBuffer = *vertexBuffers; + nextLoopIndex = loopIndex + 8; + vertexX = (float *)(spriteScale * *(float *)((int)&g_billboardVertexOffsetsX + loopIndex) + + worldPosX); + vertexY = (float *)(spriteScale * *(float *)((int)&g_billboardVertexOffsetsY + loopIndex) + + worldPosY); + *vertexBuffer = (float)vertexX; + vertexBuffer[1] = (float)vertexY; + vertexZ = (float *)worldPosZ; + vertexBuffer[2] = worldPosZ; + vertexBuffer = vertexBuffers[1]; + *vertexBuffer = (float)g_lightDirectionX; + vertexBuffer[1] = (float)g_lightDirectionY; + vertexBuffer[2] = (float)g_lightDirectionZ; + *vertexBuffers[2] = (float)colorValue; + vertexBuffer = vertexBuffers[3]; + tempFloat1 = *(float *)((int)&g_spriteTextureOffsetsV + loopIndex); + textureCoordV = this->textureScaleV; + *vertexBuffer = + *(float *)((int)&g_spriteTextureOffsetsU + loopIndex) * this->textureScaleU + + (float)textureU; + vertexBuffer[1] = tempFloat1 * textureCoordV + (float)textureV; + vertexBuffers[8] = (float *)((int)vertexBuffers[8] + 1); + *vertexBuffers = (float *)((int)*vertexBuffers + (int)vertexBuffers[4]); + vertexBuffers[1] = (float *)((int)vertexBuffers[1] + (int)vertexBuffers[5]); + vertexBuffers[2] = (float *)((int)vertexBuffers[2] + (int)vertexBuffers[6]); + vertexBuffers[3] = (float *)((int)vertexBuffers[3] + (int)vertexBuffers[7]); + loopIndex = nextLoopIndex; + } while (nextLoopIndex != 0x20); + } + else { + vertexBuffers = (float **)0x4; + textureVPtr = &g_transformedVertex1_Z; + textureOffsetPtr = &g_spriteTextureOffsetsV; + do { + particleDataPtr = *vertexBufferPtr; + transformedVertexZ = (float *)(spriteScale * (float)*textureVPtr); + vertexX = (float *)(spriteScale * (float)textureVPtr[-2] + worldPosX); + vertexY = (float *)(spriteScale * (float)textureVPtr[-1] + worldPosY); + vertexZ = (float *)((float)transformedVertexZ + worldPosZ); + *particleDataPtr = (float)vertexX; + particleDataPtr[1] = (float)vertexY; + particleDataPtr[2] = (float)vertexZ; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + particleDataPtr = vertexBufferPtr[3]; + tempFloat1 = this->textureScaleV; + clampedParticleTime = (float)*textureOffsetPtr; + *particleDataPtr = (float)textureOffsetPtr[-1] * this->textureScaleU + (float)textureU; + particleDataPtr[1] = tempFloat1 * clampedParticleTime + (float)textureV; + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + *vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + vertexBuffers = (float **)((int)vertexBuffers + -1); + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + textureVPtr = textureVPtr + 3; + textureOffsetPtr = textureOffsetPtr + 2; + particleDataPtr = particleData; + } while (vertexBuffers != (float **)0x0); + } + } + else { + rotationAngle = (float *)(this->colorPaletteArray[1].texCoordBase1 * particleDataPtr[7]); + if (((char)((ushort)(undefined2)texLoopIndex >> 8) < '\0') && + (((uint)particleDataPtr & 0x20) != 0)) { + rotationAngle = (float *)-(float)rotationAngle; + } + if ((texLoopIndex & 0x2000) == 0) { + particleTimeIndex = (float *)&particleData; + cosRotation = (float10)fcos((float10)(float)rotationAngle); + sinRotation = (float10)fsin((float10)(float)rotationAngle); + colorDeltaR = 0; + do { + tempFloat1 = *(float *)((int)&g_billboardVertexOffsetsX + colorDeltaR); + textureCoordV = *(float *)((int)&g_billboardVertexOffsetsY + colorDeltaR); + texLoopIndex = colorDeltaR + 8; + **vertexBuffers = + (tempFloat1 * (float)cosRotation * spriteScale + worldPosX) - + textureCoordV * (float)sinRotation * spriteScale; + (*vertexBuffers)[1] = + textureCoordV * (float)cosRotation * spriteScale + + tempFloat1 * (float)sinRotation * spriteScale + worldPosY; + (*vertexBuffers)[2] = worldPosZ; + vertexBuffer = vertexBuffers[1]; + *vertexBuffer = (float)g_lightDirectionX; + vertexBuffer[1] = (float)g_lightDirectionY; + vertexBuffer[2] = (float)g_lightDirectionZ; + *vertexBuffers[2] = (float)colorValue; + vertexBuffer = vertexBuffers[3]; + tempFloat1 = *(float *)((int)&g_spriteTextureOffsetsV + colorDeltaR); + textureCoordV = this->textureScaleV; + *vertexBuffer = + *(float *)((int)&g_spriteTextureOffsetsU + colorDeltaR) * this->textureScaleU + + (float)textureU; + vertexBuffer[1] = tempFloat1 * textureCoordV + (float)textureV; + vertexBuffers[8] = (float *)((int)vertexBuffers[8] + 1); + *vertexBuffers = (float *)((int)*vertexBuffers + (int)vertexBuffers[4]); + vertexBuffers[1] = (float *)((int)vertexBuffers[1] + (int)vertexBuffers[5]); + vertexBuffers[2] = (float *)((int)vertexBuffers[2] + (int)vertexBuffers[6]); + vertexBuffers[3] = (float *)((int)vertexBuffers[3] + (int)vertexBuffers[7]); + colorDeltaR = texLoopIndex; + } while (texLoopIndex < 0x20); + } + else { + vertexBuffers = (float **)&g_spriteTextureOffsetsV; + textureVPtr = &g_transformedVertex1_Y; + particleTimeIndex = (float *)0x4; + do { + createAxisAngleRotationMatrix3x3 + ((float *)&rotMatrix_m00, + (float *)((int)&this->colorPaletteArray[4].renderFlags + 2), + (float)rotationAngle,'\x01'); + particleDataPtr = *vertexBufferPtr; + transformedVertexZ = + (float *)((float)rotMatrix_m20 * (float)textureVPtr[-1] + + (float)rotMatrix_m22 * (float)textureVPtr[1] + + (float)rotMatrix_m21 * (float)*textureVPtr); + transformedVertexX = + (float *)(((float)rotMatrix_m02 * (float)textureVPtr[1] + + (float)rotMatrix_m01 * (float)*textureVPtr + + (float)rotMatrix_m00 * (float)textureVPtr[-1]) * spriteScale); + vertexX = (float *)((float)transformedVertexX + worldPosX); + vertexY = (float *)(((float)rotMatrix_m10 * (float)textureVPtr[-1] + + (float)rotMatrix_m12 * (float)textureVPtr[1] + + (float)rotMatrix_m11 * (float)*textureVPtr) * spriteScale + worldPosY) + ; + vertexZ = (float *)((float)transformedVertexZ * spriteScale + worldPosZ); + *particleDataPtr = (float)vertexX; + particleDataPtr[1] = (float)vertexY; + particleDataPtr[2] = (float)vertexZ; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + textureUIndex = (float)vertexBuffers[-1] * this->textureScaleU + (float)textureU; + textureVIndex = (float)*vertexBuffers * this->textureScaleV + (float)textureV; + particleDataPtr = vertexBufferPtr[3]; + *particleDataPtr = textureUIndex; + particleDataPtr[1] = textureVIndex; + vertexBuffers = vertexBuffers + 2; + textureVPtr = textureVPtr + 3; + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + *vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + particleTimeIndex = (float *)((int)particleTimeIndex + -1); + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + } while (particleTimeIndex != (float *)0x0); + particleTimeIndex = (float *)0x0; + particleDataPtr = particleData; + } + } + } + if ((this->colorPaletteArray[2].padding_06 & 8) != 0) { + textureUIndex = + (float)(*(int *)(this->colorPaletteArray[1].final_padding + 6) - 1U & (uint)colorData2); + textureVIndex = 0.0; + particleData = (float *)this->childSystemPointers; + vertexBuffers = (float **)((float)(uint)textureUIndex * this->textureScaleU); + negVelocityY = (float *)0x0; + rotationAngle = + (float *)((float)((int)colorData2 >> (SUB41(this->uvCoordinateScale,0) & 0x1f)) * + this->textureScaleV); + negVelocityX = (float *)-particleDataPtr[4]; + transformedVertexX = (float *)-particleDataPtr[5]; + negVelocityZ = (float *)-particleDataPtr[6]; + if (((this->colorPaletteArray[2].texCoordDelta2_prefix & 1) != 0) && + (particleDataPtr[7] < (float)particleData)) { + particleData = (float *)particleDataPtr[7]; + } + particleDataPtr = + (float *)transformVector4ByMatrix4x4 + ((float *)&transformedVelocityVector,(float *)&negVelocityX, + (float *)&g_worldMatrix); + tempFloat1 = (float)particleData * *particleDataPtr; + textureCoordV = (float)particleData * particleDataPtr[1]; + cosValue = tempFloat1 * tempFloat1 + textureCoordV * textureCoordV; + if (_DAT_0080c744 <= cosValue) { + vertexBuffer = *vertexBufferPtr; + velocityDirection = (float)particleData * particleDataPtr[2] + worldPosZ; + cosValue = spriteScale / SQRT(cosValue); + scaleFactor = tempFloat1 * cosValue; + cosValue = cosValue * textureCoordV; + *vertexBuffer = worldPosX - cosValue; + vertexBuffer[1] = scaleFactor + worldPosY; + vertexBuffer[2] = worldPosZ; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + particleDataPtr = vertexBufferPtr[3]; + zDepth = (float)g_spriteTextureOffsetsV * this->textureScaleV; + *particleDataPtr = (float)g_spriteTextureOffsetsU * this->textureScaleU + (float)vertexBuffers + ; + particleDataPtr[1] = zDepth + (float)rotationAngle; + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + particleDataPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + *vertexBufferPtr = particleDataPtr; + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + *particleDataPtr = worldPosX + cosValue; + particleDataPtr[1] = worldPosY - scaleFactor; + particleDataPtr[2] = worldPosZ; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + particleDataPtr = vertexBufferPtr[3]; + zDepth = (float)g_spriteTextureOffsetsV * this->textureScaleV; + *particleDataPtr = (float)g_spriteTextureOffsetsU * this->textureScaleU + (float)vertexBuffers + ; + particleDataPtr[1] = zDepth + (float)rotationAngle; + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + particleDataPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + *vertexBufferPtr = particleDataPtr; + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + *particleDataPtr = (tempFloat1 + worldPosX) - cosValue; + particleDataPtr[1] = textureCoordV + worldPosY + scaleFactor; + particleDataPtr[2] = velocityDirection; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + particleDataPtr = vertexBufferPtr[3]; + zDepth = _DAT_0087d748 * this->textureScaleV; + *particleDataPtr = _DAT_0087d744 * this->textureScaleU + (float)vertexBuffers; + particleDataPtr[1] = zDepth + (float)rotationAngle; + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + particleDataPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + *vertexBufferPtr = particleDataPtr; + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + *particleDataPtr = tempFloat1 + worldPosX + cosValue; + particleDataPtr[1] = (textureCoordV + worldPosY) - scaleFactor; + particleDataPtr[2] = velocityDirection; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + particleDataPtr = vertexBufferPtr[3]; + tempFloat1 = _DAT_0087d750 * this->textureScaleV; + *particleDataPtr = _DAT_0087d74c * this->textureScaleU + (float)vertexBuffers; + particleDataPtr[1] = tempFloat1 + (float)rotationAngle; + *vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + return (undefined *)0x1; + } + loopIndex = 0; + do { + particleDataPtr = *vertexBufferPtr; + nextLoopIndex = loopIndex + 8; + tempFloat1 = *(float *)((int)&g_billboardVertexOffsetsY + loopIndex); + *particleDataPtr = + spriteScale * *(float *)((int)&g_billboardVertexOffsetsX + loopIndex) + worldPosX; + particleDataPtr[1] = spriteScale * tempFloat1 + worldPosY; + particleDataPtr[2] = worldPosZ; + particleDataPtr = vertexBufferPtr[1]; + *particleDataPtr = (float)g_lightDirectionX; + particleDataPtr[1] = (float)g_lightDirectionY; + particleDataPtr[2] = (float)g_lightDirectionZ; + *vertexBufferPtr[2] = (float)colorValue; + particleDataPtr = vertexBufferPtr[3]; + tempFloat1 = *(float *)((int)&g_spriteTextureOffsetsV + loopIndex); + textureCoordV = this->textureScaleV; + *particleDataPtr = + *(float *)((int)&g_spriteTextureOffsetsU + loopIndex) * this->textureScaleU + + (float)vertexBuffers; + particleDataPtr[1] = tempFloat1 * textureCoordV + (float)rotationAngle; + vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1); + *vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]); + vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]); + vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]); + vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]); + loopIndex = nextLoopIndex; + } while (nextLoopIndex != 0x20); + } + return (undefined *)0x1; +} + + diff --git a/src/transform44/decompiled/decomp_particle_helpers.c b/src/transform44/decompiled/decomp_particle_helpers.c new file mode 100644 index 0000000..ebc1f1a --- /dev/null +++ b/src/transform44/decompiled/decomp_particle_helpers.c @@ -0,0 +1,241 @@ +// ============================================================ +// calculateColorValues @ 0x7B9B10 +// ============================================================ + +/* WARNING: Globals starting with '_' overlap smaller symbols at the same address */ + +void __thiscall +calculateParticleColorAndScale + (OrientationData *this,float particleTime,float colorDeltaG,byte *colorValueOut, + uint *colorData1Out,uint *colorData2Out,float *spriteScaleOut) + +{ + byte alphaChannelValue; + float10 extendedPrecisionTime; + undefined *tempVar; + float timeScaledFactor; + undefined *puVar1; + + /* Calculates particle color values, texture coordinates, and scale based on + particle time and orientation data. Uses time-scaled interpolation between + base values and deltas. Handles both standard float and extended precision + calculations. */ + /* Calculate normalized time factor: (current_time - base_time) * time_scale * + 0.99 + 0.005 */ + timeScaledFactor = + (particleTime - this->timeBase) * this->timeScale * _DAT_00808aac + _DAT_00807a3c; + /* Alpha channel: ((color_delta_alpha * time_factor + base_alpha) * + color_delta_g + 512) >> 14 */ + colorValueOut[3] = + (byte)((uint)(((float)this->colorDelta_R * timeScaledFactor + (float)this->colorBase_A) * + colorDeltaG + _DAT_008029cc) >> 0xe); + /* Blue channel: (color_delta_blue * time_factor + base_blue + 512) >> 14 */ + /* Green channel: (color_delta_green * time_factor + base_green + 512) >> 14 */ + colorValueOut[2] = + (byte)((uint)((float)this->colorDelta_G * timeScaledFactor + (float)this->colorBase_B + + _DAT_008029cc) >> 0xe); + /* Red channel: (color_delta_red * time_factor + base_red + 512) >> 14 */ + colorValueOut[1] = + (byte)((uint)((float)this->colorDelta_B * timeScaledFactor + (float)this->colorBase_G + + _DAT_008029cc) >> 0xe); + puVar1 = (undefined *) + ((uint)((float)this->colorDelta_A * timeScaledFactor + (float)this->colorBase_R + + _DAT_008029cc) >> 0xe); + *colorValueOut = (byte)puVar1; + *spriteScaleOut = timeScaledFactor * this->scaleDelta + this->scaleBase; + if (*(int *)&this->field_0x50 == 0x3f800000) { + *colorData1Out = + (uint)((float)this->texData1_Delta * timeScaledFactor + (float)this->texData1_Base + + _DAT_008029cc) >> 0xe & 0xff; + colorData1Out = + (uint *)((float)this->texData2_Delta * timeScaledFactor + (float)this->texData2_Base + + _DAT_008029cc); + } + else { + extendedPrecisionTime = (float10)callIntrinsicDispatcher(puVar1); + *colorData1Out = + (uint)(float)((float10)this->texData1_Delta * extendedPrecisionTime + + (float10)this->texData1_Base + (float10)_DAT_008029cc) >> 0xe & 0xff; + colorData1Out = + (uint *)(float)((float10)this->texData2_Delta * extendedPrecisionTime + + (float10)this->texData2_Base + (float10)_DAT_008029cc); + } + *colorData2Out = (uint)colorData1Out >> 0xe & 0xff; + return; +} + + +// ============================================================ +// matVec3Transform @ 0x7BCA80 +// ============================================================ + +void __fastcall transformVector3ByMatrix4x4(float *param_1,float *param_2,float *param_3) + +{ + float fVar1; + float fVar2; + float fVar3; + float fVar4; + float fVar5; + float fVar6; + float fVar7; + float fVar8; + float fVar9; + float fVar10; + float fVar11; + float fVar12; + float fVar13; + float fVar14; + + fVar1 = param_3[10]; + fVar2 = param_2[2]; + fVar3 = param_3[2]; + fVar4 = *param_2; + fVar5 = param_3[6]; + fVar6 = param_2[1]; + fVar7 = param_3[0xe]; + fVar8 = param_3[9]; + fVar9 = param_2[2]; + fVar10 = param_3[1]; + fVar11 = *param_2; + fVar12 = param_3[5]; + fVar13 = param_2[1]; + fVar14 = param_3[0xd]; + *param_1 = *param_2 * *param_3 + param_3[4] * param_2[1] + param_3[8] * param_2[2] + param_3[0xc]; + param_1[1] = fVar12 * fVar13 + fVar10 * fVar11 + fVar8 * fVar9 + fVar14; + param_1[2] = fVar5 * fVar6 + fVar3 * fVar4 + fVar1 * fVar2 + fVar7; + return; +} + + +// ============================================================ +// buildRotationFromAngle @ 0x7BE490 +// ============================================================ + +float * __fastcall +createAxisAngleRotationMatrix3x3(float *param_1,float *param_2,float param_3,char param_4) + +{ + float fVar1; + float fVar2; + float10 fVar3; + float fVar4; + float fVar5; + float fVar6; + float10 fVar7; + undefined *local_1c; + undefined *local_18; + undefined *local_14; + undefined *local_10; + undefined *local_c; + undefined *local_8; + + local_1c = (undefined *)*param_2; + local_18 = (undefined *)param_2[1]; + local_14 = (undefined *)param_2[2]; + if (param_4 == '\0') { + fVar1 = StaticFloat1_0 / + SQRT((float)local_1c * (float)local_1c + + (float)local_18 * (float)local_18 + (float)local_14 * (float)local_14); + local_1c = (undefined *)((float)local_1c * fVar1); + local_18 = (undefined *)((float)local_18 * fVar1); + local_14 = (undefined *)(fVar1 * (float)local_14); + } + fVar3 = (float10)fcos((float10)param_3); + fVar7 = (float10)fsin((float10)param_3); + fVar1 = (float)fVar3; + fVar2 = (float)fVar7; + fVar6 = StaticFloat1_0 - fVar1; + *param_1 = (float)local_1c * (float)local_1c * fVar6 + fVar1; + fVar4 = fVar6 * (float)local_18 * (float)local_1c; + param_1[1] = fVar4 + (float)local_14 * fVar2; + fVar5 = fVar6 * (float)local_14 * (float)local_1c; + param_1[2] = fVar5 - (float)local_18 * fVar2; + param_1[3] = fVar4 - (float)local_14 * fVar2; + param_1[4] = (float)local_18 * (float)local_18 * fVar6 + fVar1; + fVar4 = fVar6 * (float)local_14 * (float)local_18; + param_1[5] = (float)local_1c * fVar2 + fVar4; + param_1[6] = fVar5 + (float)local_18 * fVar2; + param_1[7] = fVar4 - (float)local_1c * fVar2; + param_1[8] = (float)local_14 * (float)local_14 * fVar6 + fVar1; + return param_1; +} + + +// ============================================================ +// matVec3Transform2 @ 0x7BCB40 +// ============================================================ + +void __fastcall transformVector4ByMatrix4x4(float *param_1,float *param_2,float *param_3) + +{ + float fVar1; + float fVar2; + float fVar3; + float fVar4; + float fVar5; + float fVar6; + float fVar7; + float fVar8; + float fVar9; + float fVar10; + float fVar11; + float fVar12; + float fVar13; + float fVar14; + float fVar15; + float fVar16; + float fVar17; + float fVar18; + float fVar19; + float fVar20; + float fVar21; + float fVar22; + float fVar23; + float fVar24; + + fVar1 = param_3[0xf]; + fVar2 = param_2[3]; + fVar3 = param_3[3]; + fVar4 = *param_2; + fVar5 = param_3[0xb]; + fVar6 = param_2[2]; + fVar7 = param_3[7]; + fVar8 = param_2[1]; + fVar9 = param_3[0xe]; + fVar10 = param_2[3]; + fVar11 = param_3[2]; + fVar12 = *param_2; + fVar13 = param_3[10]; + fVar14 = param_2[2]; + fVar15 = param_3[6]; + fVar16 = param_2[1]; + fVar17 = param_3[0xd]; + fVar18 = param_2[3]; + fVar19 = param_3[1]; + fVar20 = *param_2; + fVar21 = param_3[9]; + fVar22 = param_2[2]; + fVar23 = param_3[5]; + fVar24 = param_2[1]; + *param_1 = *param_2 * *param_3 + + param_3[4] * param_2[1] + param_3[8] * param_2[2] + param_3[0xc] * param_2[3]; + param_1[1] = fVar23 * fVar24 + fVar21 * fVar22 + fVar19 * fVar20 + fVar17 * fVar18; + param_1[2] = fVar15 * fVar16 + fVar13 * fVar14 + fVar11 * fVar12 + fVar9 * fVar10; + param_1[3] = fVar7 * fVar8 + fVar5 * fVar6 + fVar3 * fVar4 + fVar1 * fVar2; + return; +} + + +// ============================================================ +// setupRenderState @ 0x58A230 +// ============================================================ + +void UpdateLightingOffset(void) + +{ + GetLightingOffset((int)CGxDeviceD3d__device); + return; +} + + diff --git a/src/transform44/particle_sse.zig b/src/transform44/particle_sse.zig new file mode 100644 index 0000000..19b81e9 --- /dev/null +++ b/src/transform44/particle_sse.zig @@ -0,0 +1,174 @@ +//! particle_sse — SSE replacements for WoW 1.12.1 particle rendering pipeline. +//! +//! Compiled as a separate ReleaseFast unit (same pattern as bone_sse.zig / clip_sse.zig). +//! Functions are exported and called via `extern fn` from transform44.zig detour hooks. +//! +//! Assembly references: decompiled/asm_RenderParticleSprites.txt, +//! asm_calculateColorValues.txt, asm_SetupParticleRendering.txt + +const V4 = @Vector(4, f32); +const V4i = @Vector(4, i32); + +inline fn rf32(addr: u32) f32 { + return @as(*align(1) const f32, @ptrFromInt(addr)).*; +} + +inline fn ri32(addr: u32) i32 { + return @as(*align(1) const i32, @ptrFromInt(addr)).*; +} + +inline fn ru8(addr: u32) u8 { + return @as(*const u8, @ptrFromInt(addr)).*; +} + +inline fn ru32(addr: u32) u32 { + return @as(*align(1) const u32, @ptrFromInt(addr)).*; +} + +inline fn wf32(addr: u32, val: f32) void { + @as(*align(1) f32, @ptrFromInt(addr)).* = val; +} + +inline fn wu32(addr: u32, val: u32) void { + @as(*align(1) u32, @ptrFromInt(addr)).* = val; +} + +inline fn wu8(addr: u32, val: u8) void { + @as(*u8, @ptrFromInt(addr)).* = val; +} + +// ============================================================================= +// calculateColorValues (0x7B9B10) +// ============================================================================= +// +// __thiscall(ECX=colorCtx, stack: time, scale, outColor, outAlpha1, outAlpha2, outFloat) +// RET 0x18 (6 stack params) +// +// ColorCtx layout: +// +0x00..0x03: base color bytes [B, G, R, A] (4 bytes) +// +0x04: delta_alpha (i32) +// +0x08: delta_red (i32) +// +0x0C: delta_green (i32) +// +0x10: delta_blue (i32) +// +0x14: alpha1_base (i32) +// +0x18: alpha1_delta (i32) +// +0x1C: alpha2_base (i32) +// +0x20: alpha2_delta (i32) +// +0x24: float_base (f32) +// +0x28: float_scale (f32) +// +0x2C: time_base (f32) +// +0x30: time_scale (f32) +// +0x50: alpha_power (f32, 1.0 = linear, else calls pow) +// +// Algorithm: +// t = (time - ctx.timeBase) * ctx.timeScale * CONST1 + CONST2 +// For each color channel (A,R,G,B): +// val = (float)delta * t + (float)base_byte +// alpha channel only: val *= scale +// val += MAGIC (float-to-byte trick constant at 0x8029CC) +// outColor[ch] = (byte)(float_bits >> 14) +// outFloat = t * ctx.floatScale + ctx.floatBase +// For alpha outputs: +// if ctx.alphaPower == 1.0: linear interp +// else: pow(t * alphaPower, ...) path +// +// The "float bits >> 14" is a classic fast float-to-byte: add a large power-of-2 +// magic number so the integer value sits in the mantissa bits, then extract. +// ============================================================================= + +const CC = std.builtin.CallingConvention; +const TC: CC = .{ .x86_thiscall = .{} }; + +const std = @import("std"); + +/// SSE replacement for calculateColorValues. +/// Thiscall: ECX=ctx, stack params: time(f32), scale(f32), outColor(ptr), outAlpha1(ptr), outAlpha2(ptr), outFloat(ptr) +export fn calcColorValues_SSE( + ctx: u32, + time_bits: u32, + scale_bits: u32, + out_color: u32, + out_alpha1: u32, + out_alpha2: u32, + out_float: u32, +) callconv(TC) void { + const time: f32 = @bitCast(time_bits); + const scale: f32 = @bitCast(scale_bits); + + // Step 1: Compute interpolation parameter t + const t = (time - rf32(ctx + 0x2C)) * rf32(ctx + 0x30) * rf32(0x808AAC) + rf32(0x807A3C); + + // Step 2: Compute 4 color channels + // Load base bytes and deltas + const base_a: f32 = @floatFromInt(@as(i32, ru8(ctx + 3))); + const base_r: f32 = @floatFromInt(@as(i32, ru8(ctx + 2))); + const base_g: f32 = @floatFromInt(@as(i32, ru8(ctx + 1))); + const base_b: f32 = @floatFromInt(@as(i32, ru8(ctx + 0))); + + const delta_a: f32 = @floatFromInt(ri32(ctx + 0x04)); + const delta_r: f32 = @floatFromInt(ri32(ctx + 0x08)); + const delta_g: f32 = @floatFromInt(ri32(ctx + 0x0C)); + const delta_b: f32 = @floatFromInt(ri32(ctx + 0x10)); + + const magic: f32 = rf32(0x8029CC); + + // Alpha channel: (delta * t + base) * scale + magic + const alpha_f = @mulAdd(f32, delta_a, t, base_a) * scale + magic; + // RGB channels: delta * t + base + magic (no scale) + const red_f = @mulAdd(f32, delta_r, t, base_r) + magic; + const green_f = @mulAdd(f32, delta_g, t, base_g) + magic; + const blue_f = @mulAdd(f32, delta_b, t, base_b) + magic; + + // Extract bytes via float-bits >> 14 trick + const alpha_byte: u8 = @truncate(@as(u32, @bitCast(alpha_f)) >> 14); + const red_byte: u8 = @truncate(@as(u32, @bitCast(red_f)) >> 14); + const green_byte: u8 = @truncate(@as(u32, @bitCast(green_f)) >> 14); + const blue_byte: u8 = @truncate(@as(u32, @bitCast(blue_f)) >> 14); + + // Store color bytes: [B, G, R, A] at outColor + wu8(out_color + 0, blue_byte); + wu8(out_color + 1, green_byte); + wu8(out_color + 2, red_byte); + wu8(out_color + 3, alpha_byte); + + // Step 3: Float output = t * ctx.floatScale + ctx.floatBase + wf32(out_float, @mulAdd(f32, t, rf32(ctx + 0x28), rf32(ctx + 0x24))); + + // Step 4: Alpha outputs + const alpha_power = ru32(ctx + 0x50); + if (alpha_power == 0x3F800000) { + // Fast path: alphaPower == 1.0 (linear) + const a1_val = @mulAdd(f32, @as(f32, @floatFromInt(ri32(ctx + 0x18))), t, @as(f32, @floatFromInt(ri32(ctx + 0x14)))) + magic; + const a2_val = @mulAdd(f32, @as(f32, @floatFromInt(ri32(ctx + 0x20))), t, @as(f32, @floatFromInt(ri32(ctx + 0x1C)))) + magic; + + wu32(out_alpha1, (@as(u32, @bitCast(a1_val)) >> 14) & 0xFF); + wu32(out_alpha2, (@as(u32, @bitCast(a2_val)) >> 14) & 0xFF); + } else { + // Slow path: pow scaling. Call game's pow function. + // 0x73F90A: __cdecl pow — takes ST(0)=base, ST(1)=exponent, returns ST(0) + // t_scaled = pow(t * alphaPower, ???) + // For now, fall back to scalar computation matching the original exactly. + const ap: f32 = @bitCast(alpha_power); + const t_scaled = t * ap; + + // The original calls 0x73F90A with ST(0)=t_scaled, ST(1)=loaded from [0x8015B8] (qword) + // This is __CIpow (MSVC intrinsic pow) — ST(1)=exponent (from 0x8015B8), ST(0)=base + // We need the exponent constant. For now use @exp2/@log2 to compute pow. + // Actually: the original loads FLD qword [0x8015B8] THEN calls __CIpow. + // __CIpow expects ST(0)=x, ST(1)=y, computes x^y. So: pow(t_scaled, const_at_8015B8). + // The constant at 0x8015B8 is a f64. We read it and use std.math.pow. + const exp_val: f64 = @as(*align(1) const f64, @ptrFromInt(0x8015B8)).*; + const t_pow: f32 = @floatCast(std.math.pow(f64, @as(f64, t_scaled), exp_val)); + + const a1_delta: f32 = @floatFromInt(ri32(ctx + 0x18)); + const a1_base: f32 = @floatFromInt(ri32(ctx + 0x14)); + const a1_val = @mulAdd(f32, a1_delta, t_pow, a1_base) + magic; + + const a2_delta: f32 = @floatFromInt(ri32(ctx + 0x20)); + const a2_base: f32 = @floatFromInt(ri32(ctx + 0x1C)); + const a2_val = @mulAdd(f32, a2_delta, t_pow, a2_base) + magic; + + wu32(out_alpha1, (@as(u32, @bitCast(a1_val)) >> 14) & 0xFF); + wu32(out_alpha2, (@as(u32, @bitCast(a2_val)) >> 14) & 0xFF); + } +} diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index f5d4fa6..b29f7ad 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -22,6 +22,7 @@ extern fn rayTriangleIntersection(u32, u32, u32, u32, u32, u32) u32; extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void; extern fn multiplyMatrix4x4(u32, u32, u32) u32; extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void; +extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; /// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit) /// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame. @@ -1008,7 +1009,10 @@ fn linkedlistDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaq return ret; } fn colorDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque { + // a=ECX(ctx), b=EDX(unused), c=time, d=scale, e=outColor, f=outAlpha1, g=outAlpha2, h=outFloat const s = rdtsc(); + // SSE replacement benches at 10.7x but no in-game gain — dominated by L1 cache misses + // on scattered ColorCtx structs. Fix: inline into full RenderParticleSprites replacement. const ret = color_hook.callOriginal(.{ a, b, c, d, e, f, g, h }); prof.color_cycles +|= rdtsc() - s; prof.color_calls +|= 1;