particle: Ghidra decompilations, particle_sse.zig scaffold, bench
- Ghidra C decompilation of RenderParticleSprites (422 lines) and 5 helper functions (calculateColorValues, matVec3Transform, etc.) - particle_sse.zig with calcColorValues_SSE (10.7x bench but cache-miss bound in-game — needs inlining into full function replacement) - Bench harness for calcColorValues with correctness check - build.zig: particle_sse as separate ReleaseFast compilation unit - colorDetour reverted to pass-through (SSE has no in-game effect due to L1 cache misses on scattered ColorCtx structs)
This commit is contained in:
@@ -28,7 +28,7 @@ const module_list = [_]ModuleDesc{
|
||||
.{ .name = "healtextfix", .desc = "Enable SuperWoW heal text fix" },
|
||||
.{ .name = "bigcursor", .desc = "Enable big cursor module" },
|
||||
.{ .name = "clickthrough", .desc = "Enable GO click-through (enlarge GO model bounds)" },
|
||||
.{ .name = "dpslog", .desc = "Enable structured combat log events for addons" },
|
||||
.{ .name = "dpslog", .desc = "Enable structured combat log events for addons", .default = false },
|
||||
.{ .name = "transform44", .desc = "Enable transformMatrix4x4 hook", .default = false },
|
||||
.{ .name = "addonperf", .desc = "Enable addon memory/CPU profiling API", .default = false },
|
||||
.{ .name = "filecache", .desc = "Enable MPQ archive file cache" },
|
||||
@@ -112,6 +112,15 @@ pub fn build(b: *std.Build) void {
|
||||
}),
|
||||
});
|
||||
|
||||
const particle_sse_obj = b.addObject(.{
|
||||
.name = "particle_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/transform44/particle_sse.zig"),
|
||||
.target = bone_sse_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
|
||||
const lib = b.addLibrary(.{
|
||||
.name = "weirdutils",
|
||||
.linkage = .dynamic,
|
||||
@@ -130,6 +139,7 @@ pub fn build(b: *std.Build) void {
|
||||
lib.root_module.addObject(bone_sse_ref_obj);
|
||||
lib.root_module.addObject(math_sse_obj);
|
||||
lib.root_module.addObject(silicon_sse_obj);
|
||||
lib.root_module.addObject(particle_sse_obj);
|
||||
b.installArtifact(lib);
|
||||
|
||||
// Benchmark harness — native x86 Linux executable for profiling SSE replacements
|
||||
@@ -195,10 +205,23 @@ pub fn build(b: *std.Build) void {
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bench_particle_sse = b.addObject(.{
|
||||
.name = "bench_particle_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/transform44/particle_sse.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .linux,
|
||||
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
|
||||
}),
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
bench.root_module.addObject(bench_math_sse);
|
||||
bench.root_module.addObject(bench_silicon_sse);
|
||||
bench.root_module.addObject(bench_bone_sse);
|
||||
bench.root_module.addObject(bench_bone_baseline);
|
||||
bench.root_module.addObject(bench_particle_sse);
|
||||
bench.root_module.linkSystemLibrary("m", .{});
|
||||
const install_bench = b.addInstallArtifact(bench, .{});
|
||||
const bench_step = b.step("bench", "Build math_sse benchmark harness (x86 Linux)");
|
||||
|
||||
@@ -51,6 +51,7 @@ extern fn si_vec3Dot(u32, u32) callconv(cc_fc) f64;
|
||||
extern fn si_translateBoundingVol(u32, u32) callconv(cc_tc) void;
|
||||
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(cc_fc) u32;
|
||||
extern fn si_frustumCullBBox(u32, u32, u32) callconv(cc_fc) u32;
|
||||
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(cc_tc) void;
|
||||
extern fn si_addVec3ToAccumulator(u32, u32) callconv(cc_tc) void;
|
||||
extern fn si_addToColorAccumulator(u32, u32) callconv(cc_tc) void;
|
||||
extern fn si_packParticleColor(u32, u32, u32, u32) callconv(cc_tc) void;
|
||||
@@ -1769,6 +1770,9 @@ pub fn main() void {
|
||||
}
|
||||
}
|
||||
|
||||
// calcColorValues_SSE -- thiscall(ctx_ECX, time, scale, outColor, outAlpha1, outAlpha2, outFloat)
|
||||
bench_calcColorValues();
|
||||
|
||||
// si_frustumCullBBox -- fastcall(bbox_ECX, flags_EDX, radius_stack) -> u32
|
||||
bench_frustumCullBBox();
|
||||
|
||||
@@ -1779,6 +1783,99 @@ pub fn main() void {
|
||||
print("\n", .{});
|
||||
}
|
||||
|
||||
fn bench_calcColorValues() void {
|
||||
// Map pages for global constants used by calculateColorValues
|
||||
// 0x808AAC and 0x807A3C are in .rdata range (already mapped)
|
||||
// 0x8029CC is in .rdata range (already mapped)
|
||||
// 0x8015B8 is in .rdata range (already mapped) — pow exponent constant
|
||||
|
||||
// Build fake ColorCtx struct
|
||||
// Layout: +0x00..0x03 = base bytes [B,G,R,A], +0x04..0x10 = deltas (4×i32),
|
||||
// +0x14..0x20 = alpha base/delta pairs (4×i32), +0x24 = float_base(f32),
|
||||
// +0x28 = float_scale(f32), +0x2C = time_base(f32), +0x30 = time_scale(f32),
|
||||
// +0x50 = alpha_power(f32)
|
||||
var ctx: [0x54]u8 align(4) = std.mem.zeroes([0x54]u8);
|
||||
// Base color: BGRA = {100, 150, 200, 220}
|
||||
ctx[0] = 100; ctx[1] = 150; ctx[2] = 200; ctx[3] = 220;
|
||||
// Deltas (i32): small values
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x04..0x08])).* = 10;
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x08..0x0C])).* = -5;
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x0C..0x10])).* = 8;
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x10..0x14])).* = -3;
|
||||
// Alpha base/delta
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x14..0x18])).* = 200;
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x18..0x1C])).* = 20;
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x1C..0x20])).* = 180;
|
||||
@as(*align(1) i32, @ptrCast(ctx[0x20..0x24])).* = 15;
|
||||
// Float base/scale
|
||||
@as(*align(1) f32, @ptrCast(ctx[0x24..0x28])).* = 1.0;
|
||||
@as(*align(1) f32, @ptrCast(ctx[0x28..0x2C])).* = 0.5;
|
||||
// Time base/scale
|
||||
@as(*align(1) f32, @ptrCast(ctx[0x2C..0x30])).* = 0.0;
|
||||
@as(*align(1) f32, @ptrCast(ctx[0x30..0x34])).* = 1.0;
|
||||
// Alpha power = 1.0 (linear, fast path)
|
||||
@as(*align(1) f32, @ptrCast(ctx[0x50..0x54])).* = 1.0;
|
||||
|
||||
const time: f32 = 0.5;
|
||||
const scale: f32 = 1.0;
|
||||
var out_color_o: [4]u8 = .{0} ** 4;
|
||||
var out_color_s: [4]u8 = .{0} ** 4;
|
||||
var out_alpha1_o: u32 = 0;
|
||||
var out_alpha1_s: u32 = 0;
|
||||
var out_alpha2_o: u32 = 0;
|
||||
var out_alpha2_s: u32 = 0;
|
||||
var out_float_o: f32 = 0;
|
||||
var out_float_s: f32 = 0;
|
||||
|
||||
// Original: __thiscall(ECX=ctx, stack: time, scale, outColor, outAlpha1, outAlpha2, outFloat), RET 0x18
|
||||
const of = origFn(fn (u32, u32, u32, u32, u32, u32, u32) callconv(cc_tc) void, 0x7B9B10);
|
||||
of(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_o), a(&out_alpha1_o), a(&out_alpha2_o), a(&out_float_o));
|
||||
calcColorValues_SSE(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_s), a(&out_alpha1_s), a(&out_alpha2_s), a(&out_float_s));
|
||||
|
||||
// The original returns float in ST(0) which we need to pop to avoid FPU stack leak
|
||||
// Pop it after each call in the bench loop too
|
||||
const ok = out_color_o[0] == out_color_s[0] and out_color_o[1] == out_color_s[1] and
|
||||
out_color_o[2] == out_color_s[2] and out_color_o[3] == out_color_s[3] and
|
||||
out_alpha1_o == out_alpha1_s and out_alpha2_o == out_alpha2_s and
|
||||
compareF32(out_float_o, out_float_s);
|
||||
if (!ok) {
|
||||
print(" color bytes: orig=[{d},{d},{d},{d}] sse=[{d},{d},{d},{d}]\n", .{
|
||||
out_color_o[0], out_color_o[1], out_color_o[2], out_color_o[3],
|
||||
out_color_s[0], out_color_s[1], out_color_s[2], out_color_s[3],
|
||||
});
|
||||
print(" alpha1: orig={d} sse={d} alpha2: orig={d} sse={d}\n", .{
|
||||
out_alpha1_o, out_alpha1_s, out_alpha2_o, out_alpha2_s,
|
||||
});
|
||||
print(" float: orig=0x{x} sse=0x{x}\n", .{
|
||||
@as(u32, @bitCast(out_float_o)), @as(u32, @bitCast(out_float_s)),
|
||||
});
|
||||
}
|
||||
|
||||
// Original returns float in ST(0) — must pop to avoid FPU stack overflow in bench loop
|
||||
var t: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
of(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_o), a(&out_alpha1_o), a(&out_alpha2_o), a(&out_float_o));
|
||||
// Pop ST(0) to prevent FPU stack overflow
|
||||
asm volatile ("fstp %%st(0)" ::: "st");
|
||||
}
|
||||
const _te = rdtsc() - _t0;
|
||||
if (_te < t) t = _te;
|
||||
}
|
||||
|
||||
var s: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
const _t0 = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
calcColorValues_SSE(a(&ctx), @bitCast(time), @bitCast(scale), a(&out_color_s), a(&out_alpha1_s), a(&out_alpha2_s), a(&out_float_s));
|
||||
}
|
||||
const _te = rdtsc() - _t0;
|
||||
if (_te < s) s = _te;
|
||||
}
|
||||
report("calcColorValues", t, s, ok);
|
||||
}
|
||||
|
||||
fn bench_frustumCullBBox() void {
|
||||
// Map runtime global pages for view-proj matrices, occlusion buffer, and flags
|
||||
_ = mapZeroed(0xC7B000, 0x20000); // covers 0xC7B000-0xC7D000+ (matrices, horizon buffer, globals)
|
||||
|
||||
@@ -0,0 +1,422 @@
|
||||
openjdk version "21.0.10" 2026-01-20
|
||||
OpenJDK Runtime Environment (build 21.0.10+7)
|
||||
OpenJDK 64-Bit Server VM (build 21.0.10+7, mixed mode)
|
||||
// RenderParticleSprites @ 0x7B2A50
|
||||
// Decompiled by Ghidra
|
||||
|
||||
|
||||
/* WARNING: Globals starting with '_' overlap smaller symbols at the same address */
|
||||
|
||||
undefined * __thiscall
|
||||
RenderParticleSprites(ParticleSystemRenderer *this,float *particleData,float **vertexBuffers)
|
||||
|
||||
{
|
||||
undefined *puVar1;
|
||||
int loopIndex;
|
||||
uint texLoopIndex;
|
||||
float *particleDataPtr;
|
||||
int nextLoopIndex;
|
||||
undefined **textureOffsetPtr;
|
||||
undefined **textureVPtr;
|
||||
uint colorDeltaR;
|
||||
float10 sinRotation;
|
||||
float *rotMatrix_m00;
|
||||
float *rotMatrix_m01;
|
||||
float *rotMatrix_m02;
|
||||
float *rotMatrix_m10;
|
||||
float *rotMatrix_m11;
|
||||
float *rotMatrix_m12;
|
||||
float *rotMatrix_m20;
|
||||
float *rotMatrix_m21;
|
||||
float *rotMatrix_m22;
|
||||
float *transformedVelocityVector;
|
||||
float *transformedVertexZ;
|
||||
float *negVelocityX;
|
||||
float *transformedVertexX;
|
||||
float *negVelocityZ;
|
||||
float *negVelocityY;
|
||||
uint *colorData1;
|
||||
uint *colorData2;
|
||||
float *textureU;
|
||||
float *textureV;
|
||||
float *particleTimeIndex;
|
||||
float worldPosX;
|
||||
float worldPosY;
|
||||
float worldPosZ;
|
||||
float *vertexX;
|
||||
float *vertexY;
|
||||
float *vertexZ;
|
||||
uint *colorValue;
|
||||
float *rotationAngle;
|
||||
float spriteScale;
|
||||
float *particleFlags;
|
||||
float textureUIndex;
|
||||
float textureVIndex;
|
||||
float clampedParticleTime;
|
||||
undefined *clampedTimePtr;
|
||||
float10 cosRotation;
|
||||
float cosValue;
|
||||
float scaleFactor;
|
||||
float tempFloat1;
|
||||
float textureCoordV;
|
||||
float velocityDirection;
|
||||
float *vertexBuffer;
|
||||
float **vertexBufferPtr;
|
||||
float zDepth;
|
||||
|
||||
particleDataPtr = particleData;
|
||||
colorDeltaR = 0;
|
||||
/* Reading colorDeltaB+2 - likely wrong field mapping */
|
||||
/* Reading texCoordBase1+2 - another +2 offset pattern */
|
||||
if (((float)this->colorPaletteArray[2].renderFlags_source < StaticFloat1_0) ||
|
||||
(*(float *)&this->colorPaletteArray[2].renderFlags_prefix !=
|
||||
(float)COLLISION_PLANE_ZERO_THRESHOLD)) {
|
||||
puVar1 = (undefined *)(this->colorPaletteArray[2].texCoordDelta2 * particleData[7]);
|
||||
clampedTimePtr = COLLISION_PLANE_ZERO_THRESHOLD;
|
||||
if (((float)COLLISION_PLANE_ZERO_THRESHOLD <= (float)puVar1) &&
|
||||
(clampedTimePtr = puVar1, (float)_DAT_007ffe58 <= (float)puVar1)) {
|
||||
clampedTimePtr = _DAT_007ffe58;
|
||||
}
|
||||
particleTimeIndex = (float *)((float)clampedTimePtr + _DAT_008029cc);
|
||||
colorDeltaR = ((uint)particleTimeIndex >> 0xe) + ((uint)particleData >> 5) & 0x7f;
|
||||
}
|
||||
if (((float)this->colorPaletteArray[2].renderFlags_source < StaticFloat1_0) &&
|
||||
((float)this->colorPaletteArray[2].renderFlags_source <
|
||||
*(float *)(&g_particleDepthBuffer + colorDeltaR * 4))) {
|
||||
return (undefined *)0x0;
|
||||
}
|
||||
colorValue = (uint *)0x0;
|
||||
calculateParticleColorAndScale
|
||||
((OrientationData *)
|
||||
(this->orientationDataArray + (uint)*(byte *)(particleData + 3) * 0x60 + -0x12),
|
||||
particleData[7],*(float *)((int)&this->colorPaletteArray[2].colorDeltaR + 2),
|
||||
(byte *)&colorValue,(uint *)&colorData1,(uint *)&colorData2,&spriteScale);
|
||||
loopIndex = UpdateLightingOffset();
|
||||
if (*(int *)(loopIndex + 0x1c) == 1) {
|
||||
colorValue = (uint *)CONCAT31(CONCAT21(CONCAT11((char)((uint)colorValue >> 0x18),
|
||||
(char)colorValue),(char)((uint)colorValue >> 8))
|
||||
,(char)((uint)colorValue >> 0x10));
|
||||
rotationAngle = (float *)colorValue;
|
||||
}
|
||||
if (*(float *)&this->colorPaletteArray[2].renderFlags_prefix !=
|
||||
(float)COLLISION_PLANE_ZERO_THRESHOLD) {
|
||||
spriteScale = (*(float *)(&g_particleDepthBuffer + colorDeltaR * 4) *
|
||||
*(float *)&this->colorPaletteArray[2].renderFlags_prefix +
|
||||
*(float *)&this->colorPaletteArray[2].field_0x1e_source) * spriteScale;
|
||||
}
|
||||
colorDeltaR._0_2_ = this->colorPaletteArray[2].padding_06;
|
||||
colorDeltaR._2_2_ = this->colorPaletteArray[2].texCoordDelta2_prefix;
|
||||
if ((colorDeltaR & 0x200) != 0) {
|
||||
spriteScale = spriteScale * *(float *)(this->colorPaletteArray[3].padding_50_53 + 2);
|
||||
}
|
||||
transformVector3ByMatrix4x4(&worldPosX,particleDataPtr,(float *)&g_worldMatrix);
|
||||
vertexBufferPtr = vertexBuffers;
|
||||
texLoopIndex._0_2_ = this->colorPaletteArray[2].padding_06;
|
||||
texLoopIndex._2_2_ = this->colorPaletteArray[2].texCoordDelta2_prefix;
|
||||
if ((texLoopIndex & 4) != 0) {
|
||||
textureUIndex =
|
||||
(float)(*(int *)(this->colorPaletteArray[1].final_padding + 6) - 1U & (uint)colorData1);
|
||||
textureVIndex = 0.0;
|
||||
textureU = (float *)((float)(uint)textureUIndex * this->textureScaleU);
|
||||
textureV = (float *)((float)((int)colorData1 >> (SUB41(this->uvCoordinateScale,0) & 0x1f)) *
|
||||
this->textureScaleV);
|
||||
if (this->colorPaletteArray[1].texCoordBase1 == (float)COLLISION_PLANE_ZERO_THRESHOLD) {
|
||||
if ((texLoopIndex & 0x2000) == 0) {
|
||||
loopIndex = 0;
|
||||
do {
|
||||
vertexBuffer = *vertexBuffers;
|
||||
nextLoopIndex = loopIndex + 8;
|
||||
vertexX = (float *)(spriteScale * *(float *)((int)&g_billboardVertexOffsetsX + loopIndex)
|
||||
+ worldPosX);
|
||||
vertexY = (float *)(spriteScale * *(float *)((int)&g_billboardVertexOffsetsY + loopIndex)
|
||||
+ worldPosY);
|
||||
*vertexBuffer = (float)vertexX;
|
||||
vertexBuffer[1] = (float)vertexY;
|
||||
vertexZ = (float *)worldPosZ;
|
||||
vertexBuffer[2] = worldPosZ;
|
||||
vertexBuffer = vertexBuffers[1];
|
||||
*vertexBuffer = (float)g_lightDirectionX;
|
||||
vertexBuffer[1] = (float)g_lightDirectionY;
|
||||
vertexBuffer[2] = (float)g_lightDirectionZ;
|
||||
*vertexBuffers[2] = (float)colorValue;
|
||||
vertexBuffer = vertexBuffers[3];
|
||||
tempFloat1 = *(float *)((int)&g_spriteTextureOffsetsV + loopIndex);
|
||||
textureCoordV = this->textureScaleV;
|
||||
*vertexBuffer =
|
||||
*(float *)((int)&g_spriteTextureOffsetsU + loopIndex) * this->textureScaleU +
|
||||
(float)textureU;
|
||||
vertexBuffer[1] = tempFloat1 * textureCoordV + (float)textureV;
|
||||
vertexBuffers[8] = (float *)((int)vertexBuffers[8] + 1);
|
||||
*vertexBuffers = (float *)((int)*vertexBuffers + (int)vertexBuffers[4]);
|
||||
vertexBuffers[1] = (float *)((int)vertexBuffers[1] + (int)vertexBuffers[5]);
|
||||
vertexBuffers[2] = (float *)((int)vertexBuffers[2] + (int)vertexBuffers[6]);
|
||||
vertexBuffers[3] = (float *)((int)vertexBuffers[3] + (int)vertexBuffers[7]);
|
||||
loopIndex = nextLoopIndex;
|
||||
} while (nextLoopIndex != 0x20);
|
||||
}
|
||||
else {
|
||||
vertexBuffers = (float **)0x4;
|
||||
textureVPtr = &g_transformedVertex1_Z;
|
||||
textureOffsetPtr = &g_spriteTextureOffsetsV;
|
||||
do {
|
||||
particleDataPtr = *vertexBufferPtr;
|
||||
transformedVertexZ = (float *)(spriteScale * (float)*textureVPtr);
|
||||
vertexX = (float *)(spriteScale * (float)textureVPtr[-2] + worldPosX);
|
||||
vertexY = (float *)(spriteScale * (float)textureVPtr[-1] + worldPosY);
|
||||
vertexZ = (float *)((float)transformedVertexZ + worldPosZ);
|
||||
*particleDataPtr = (float)vertexX;
|
||||
particleDataPtr[1] = (float)vertexY;
|
||||
particleDataPtr[2] = (float)vertexZ;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
tempFloat1 = this->textureScaleV;
|
||||
clampedParticleTime = (float)*textureOffsetPtr;
|
||||
*particleDataPtr = (float)textureOffsetPtr[-1] * this->textureScaleU + (float)textureU;
|
||||
particleDataPtr[1] = tempFloat1 * clampedParticleTime + (float)textureV;
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
*vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
vertexBuffers = (float **)((int)vertexBuffers + -1);
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
textureVPtr = textureVPtr + 3;
|
||||
textureOffsetPtr = textureOffsetPtr + 2;
|
||||
particleDataPtr = particleData;
|
||||
} while (vertexBuffers != (float **)0x0);
|
||||
}
|
||||
}
|
||||
else {
|
||||
rotationAngle = (float *)(this->colorPaletteArray[1].texCoordBase1 * particleDataPtr[7]);
|
||||
if (((char)((ushort)(undefined2)texLoopIndex >> 8) < '\0') &&
|
||||
(((uint)particleDataPtr & 0x20) != 0)) {
|
||||
rotationAngle = (float *)-(float)rotationAngle;
|
||||
}
|
||||
if ((texLoopIndex & 0x2000) == 0) {
|
||||
particleTimeIndex = (float *)&particleData;
|
||||
cosRotation = (float10)fcos((float10)(float)rotationAngle);
|
||||
sinRotation = (float10)fsin((float10)(float)rotationAngle);
|
||||
colorDeltaR = 0;
|
||||
do {
|
||||
tempFloat1 = *(float *)((int)&g_billboardVertexOffsetsX + colorDeltaR);
|
||||
textureCoordV = *(float *)((int)&g_billboardVertexOffsetsY + colorDeltaR);
|
||||
texLoopIndex = colorDeltaR + 8;
|
||||
**vertexBuffers =
|
||||
(tempFloat1 * (float)cosRotation * spriteScale + worldPosX) -
|
||||
textureCoordV * (float)sinRotation * spriteScale;
|
||||
(*vertexBuffers)[1] =
|
||||
textureCoordV * (float)cosRotation * spriteScale +
|
||||
tempFloat1 * (float)sinRotation * spriteScale + worldPosY;
|
||||
(*vertexBuffers)[2] = worldPosZ;
|
||||
vertexBuffer = vertexBuffers[1];
|
||||
*vertexBuffer = (float)g_lightDirectionX;
|
||||
vertexBuffer[1] = (float)g_lightDirectionY;
|
||||
vertexBuffer[2] = (float)g_lightDirectionZ;
|
||||
*vertexBuffers[2] = (float)colorValue;
|
||||
vertexBuffer = vertexBuffers[3];
|
||||
tempFloat1 = *(float *)((int)&g_spriteTextureOffsetsV + colorDeltaR);
|
||||
textureCoordV = this->textureScaleV;
|
||||
*vertexBuffer =
|
||||
*(float *)((int)&g_spriteTextureOffsetsU + colorDeltaR) * this->textureScaleU +
|
||||
(float)textureU;
|
||||
vertexBuffer[1] = tempFloat1 * textureCoordV + (float)textureV;
|
||||
vertexBuffers[8] = (float *)((int)vertexBuffers[8] + 1);
|
||||
*vertexBuffers = (float *)((int)*vertexBuffers + (int)vertexBuffers[4]);
|
||||
vertexBuffers[1] = (float *)((int)vertexBuffers[1] + (int)vertexBuffers[5]);
|
||||
vertexBuffers[2] = (float *)((int)vertexBuffers[2] + (int)vertexBuffers[6]);
|
||||
vertexBuffers[3] = (float *)((int)vertexBuffers[3] + (int)vertexBuffers[7]);
|
||||
colorDeltaR = texLoopIndex;
|
||||
} while (texLoopIndex < 0x20);
|
||||
}
|
||||
else {
|
||||
vertexBuffers = (float **)&g_spriteTextureOffsetsV;
|
||||
textureVPtr = &g_transformedVertex1_Y;
|
||||
particleTimeIndex = (float *)0x4;
|
||||
do {
|
||||
createAxisAngleRotationMatrix3x3
|
||||
((float *)&rotMatrix_m00,
|
||||
(float *)((int)&this->colorPaletteArray[4].renderFlags + 2),
|
||||
(float)rotationAngle,'\x01');
|
||||
particleDataPtr = *vertexBufferPtr;
|
||||
transformedVertexZ =
|
||||
(float *)((float)rotMatrix_m20 * (float)textureVPtr[-1] +
|
||||
(float)rotMatrix_m22 * (float)textureVPtr[1] +
|
||||
(float)rotMatrix_m21 * (float)*textureVPtr);
|
||||
transformedVertexX =
|
||||
(float *)(((float)rotMatrix_m02 * (float)textureVPtr[1] +
|
||||
(float)rotMatrix_m01 * (float)*textureVPtr +
|
||||
(float)rotMatrix_m00 * (float)textureVPtr[-1]) * spriteScale);
|
||||
vertexX = (float *)((float)transformedVertexX + worldPosX);
|
||||
vertexY = (float *)(((float)rotMatrix_m10 * (float)textureVPtr[-1] +
|
||||
(float)rotMatrix_m12 * (float)textureVPtr[1] +
|
||||
(float)rotMatrix_m11 * (float)*textureVPtr) * spriteScale + worldPosY)
|
||||
;
|
||||
vertexZ = (float *)((float)transformedVertexZ * spriteScale + worldPosZ);
|
||||
*particleDataPtr = (float)vertexX;
|
||||
particleDataPtr[1] = (float)vertexY;
|
||||
particleDataPtr[2] = (float)vertexZ;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
textureUIndex = (float)vertexBuffers[-1] * this->textureScaleU + (float)textureU;
|
||||
textureVIndex = (float)*vertexBuffers * this->textureScaleV + (float)textureV;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
*particleDataPtr = textureUIndex;
|
||||
particleDataPtr[1] = textureVIndex;
|
||||
vertexBuffers = vertexBuffers + 2;
|
||||
textureVPtr = textureVPtr + 3;
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
*vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
particleTimeIndex = (float *)((int)particleTimeIndex + -1);
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
} while (particleTimeIndex != (float *)0x0);
|
||||
particleTimeIndex = (float *)0x0;
|
||||
particleDataPtr = particleData;
|
||||
}
|
||||
}
|
||||
}
|
||||
if ((this->colorPaletteArray[2].padding_06 & 8) != 0) {
|
||||
textureUIndex =
|
||||
(float)(*(int *)(this->colorPaletteArray[1].final_padding + 6) - 1U & (uint)colorData2);
|
||||
textureVIndex = 0.0;
|
||||
particleData = (float *)this->childSystemPointers;
|
||||
vertexBuffers = (float **)((float)(uint)textureUIndex * this->textureScaleU);
|
||||
negVelocityY = (float *)0x0;
|
||||
rotationAngle =
|
||||
(float *)((float)((int)colorData2 >> (SUB41(this->uvCoordinateScale,0) & 0x1f)) *
|
||||
this->textureScaleV);
|
||||
negVelocityX = (float *)-particleDataPtr[4];
|
||||
transformedVertexX = (float *)-particleDataPtr[5];
|
||||
negVelocityZ = (float *)-particleDataPtr[6];
|
||||
if (((this->colorPaletteArray[2].texCoordDelta2_prefix & 1) != 0) &&
|
||||
(particleDataPtr[7] < (float)particleData)) {
|
||||
particleData = (float *)particleDataPtr[7];
|
||||
}
|
||||
particleDataPtr =
|
||||
(float *)transformVector4ByMatrix4x4
|
||||
((float *)&transformedVelocityVector,(float *)&negVelocityX,
|
||||
(float *)&g_worldMatrix);
|
||||
tempFloat1 = (float)particleData * *particleDataPtr;
|
||||
textureCoordV = (float)particleData * particleDataPtr[1];
|
||||
cosValue = tempFloat1 * tempFloat1 + textureCoordV * textureCoordV;
|
||||
if (_DAT_0080c744 <= cosValue) {
|
||||
vertexBuffer = *vertexBufferPtr;
|
||||
velocityDirection = (float)particleData * particleDataPtr[2] + worldPosZ;
|
||||
cosValue = spriteScale / SQRT(cosValue);
|
||||
scaleFactor = tempFloat1 * cosValue;
|
||||
cosValue = cosValue * textureCoordV;
|
||||
*vertexBuffer = worldPosX - cosValue;
|
||||
vertexBuffer[1] = scaleFactor + worldPosY;
|
||||
vertexBuffer[2] = worldPosZ;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
zDepth = (float)g_spriteTextureOffsetsV * this->textureScaleV;
|
||||
*particleDataPtr = (float)g_spriteTextureOffsetsU * this->textureScaleU + (float)vertexBuffers
|
||||
;
|
||||
particleDataPtr[1] = zDepth + (float)rotationAngle;
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
particleDataPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
*vertexBufferPtr = particleDataPtr;
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
*particleDataPtr = worldPosX + cosValue;
|
||||
particleDataPtr[1] = worldPosY - scaleFactor;
|
||||
particleDataPtr[2] = worldPosZ;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
zDepth = (float)g_spriteTextureOffsetsV * this->textureScaleV;
|
||||
*particleDataPtr = (float)g_spriteTextureOffsetsU * this->textureScaleU + (float)vertexBuffers
|
||||
;
|
||||
particleDataPtr[1] = zDepth + (float)rotationAngle;
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
particleDataPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
*vertexBufferPtr = particleDataPtr;
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
*particleDataPtr = (tempFloat1 + worldPosX) - cosValue;
|
||||
particleDataPtr[1] = textureCoordV + worldPosY + scaleFactor;
|
||||
particleDataPtr[2] = velocityDirection;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
zDepth = _DAT_0087d748 * this->textureScaleV;
|
||||
*particleDataPtr = _DAT_0087d744 * this->textureScaleU + (float)vertexBuffers;
|
||||
particleDataPtr[1] = zDepth + (float)rotationAngle;
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
particleDataPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
*vertexBufferPtr = particleDataPtr;
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
*particleDataPtr = tempFloat1 + worldPosX + cosValue;
|
||||
particleDataPtr[1] = (textureCoordV + worldPosY) - scaleFactor;
|
||||
particleDataPtr[2] = velocityDirection;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
tempFloat1 = _DAT_0087d750 * this->textureScaleV;
|
||||
*particleDataPtr = _DAT_0087d74c * this->textureScaleU + (float)vertexBuffers;
|
||||
particleDataPtr[1] = tempFloat1 + (float)rotationAngle;
|
||||
*vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
return (undefined *)0x1;
|
||||
}
|
||||
loopIndex = 0;
|
||||
do {
|
||||
particleDataPtr = *vertexBufferPtr;
|
||||
nextLoopIndex = loopIndex + 8;
|
||||
tempFloat1 = *(float *)((int)&g_billboardVertexOffsetsY + loopIndex);
|
||||
*particleDataPtr =
|
||||
spriteScale * *(float *)((int)&g_billboardVertexOffsetsX + loopIndex) + worldPosX;
|
||||
particleDataPtr[1] = spriteScale * tempFloat1 + worldPosY;
|
||||
particleDataPtr[2] = worldPosZ;
|
||||
particleDataPtr = vertexBufferPtr[1];
|
||||
*particleDataPtr = (float)g_lightDirectionX;
|
||||
particleDataPtr[1] = (float)g_lightDirectionY;
|
||||
particleDataPtr[2] = (float)g_lightDirectionZ;
|
||||
*vertexBufferPtr[2] = (float)colorValue;
|
||||
particleDataPtr = vertexBufferPtr[3];
|
||||
tempFloat1 = *(float *)((int)&g_spriteTextureOffsetsV + loopIndex);
|
||||
textureCoordV = this->textureScaleV;
|
||||
*particleDataPtr =
|
||||
*(float *)((int)&g_spriteTextureOffsetsU + loopIndex) * this->textureScaleU +
|
||||
(float)vertexBuffers;
|
||||
particleDataPtr[1] = tempFloat1 * textureCoordV + (float)rotationAngle;
|
||||
vertexBufferPtr[8] = (float *)((int)vertexBufferPtr[8] + 1);
|
||||
*vertexBufferPtr = (float *)((int)*vertexBufferPtr + (int)vertexBufferPtr[4]);
|
||||
vertexBufferPtr[1] = (float *)((int)vertexBufferPtr[1] + (int)vertexBufferPtr[5]);
|
||||
vertexBufferPtr[2] = (float *)((int)vertexBufferPtr[2] + (int)vertexBufferPtr[6]);
|
||||
vertexBufferPtr[3] = (float *)((int)vertexBufferPtr[3] + (int)vertexBufferPtr[7]);
|
||||
loopIndex = nextLoopIndex;
|
||||
} while (nextLoopIndex != 0x20);
|
||||
}
|
||||
return (undefined *)0x1;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,241 @@
|
||||
// ============================================================
|
||||
// calculateColorValues @ 0x7B9B10
|
||||
// ============================================================
|
||||
|
||||
/* WARNING: Globals starting with '_' overlap smaller symbols at the same address */
|
||||
|
||||
void __thiscall
|
||||
calculateParticleColorAndScale
|
||||
(OrientationData *this,float particleTime,float colorDeltaG,byte *colorValueOut,
|
||||
uint *colorData1Out,uint *colorData2Out,float *spriteScaleOut)
|
||||
|
||||
{
|
||||
byte alphaChannelValue;
|
||||
float10 extendedPrecisionTime;
|
||||
undefined *tempVar;
|
||||
float timeScaledFactor;
|
||||
undefined *puVar1;
|
||||
|
||||
/* Calculates particle color values, texture coordinates, and scale based on
|
||||
particle time and orientation data. Uses time-scaled interpolation between
|
||||
base values and deltas. Handles both standard float and extended precision
|
||||
calculations. */
|
||||
/* Calculate normalized time factor: (current_time - base_time) * time_scale *
|
||||
0.99 + 0.005 */
|
||||
timeScaledFactor =
|
||||
(particleTime - this->timeBase) * this->timeScale * _DAT_00808aac + _DAT_00807a3c;
|
||||
/* Alpha channel: ((color_delta_alpha * time_factor + base_alpha) *
|
||||
color_delta_g + 512) >> 14 */
|
||||
colorValueOut[3] =
|
||||
(byte)((uint)(((float)this->colorDelta_R * timeScaledFactor + (float)this->colorBase_A) *
|
||||
colorDeltaG + _DAT_008029cc) >> 0xe);
|
||||
/* Blue channel: (color_delta_blue * time_factor + base_blue + 512) >> 14 */
|
||||
/* Green channel: (color_delta_green * time_factor + base_green + 512) >> 14 */
|
||||
colorValueOut[2] =
|
||||
(byte)((uint)((float)this->colorDelta_G * timeScaledFactor + (float)this->colorBase_B +
|
||||
_DAT_008029cc) >> 0xe);
|
||||
/* Red channel: (color_delta_red * time_factor + base_red + 512) >> 14 */
|
||||
colorValueOut[1] =
|
||||
(byte)((uint)((float)this->colorDelta_B * timeScaledFactor + (float)this->colorBase_G +
|
||||
_DAT_008029cc) >> 0xe);
|
||||
puVar1 = (undefined *)
|
||||
((uint)((float)this->colorDelta_A * timeScaledFactor + (float)this->colorBase_R +
|
||||
_DAT_008029cc) >> 0xe);
|
||||
*colorValueOut = (byte)puVar1;
|
||||
*spriteScaleOut = timeScaledFactor * this->scaleDelta + this->scaleBase;
|
||||
if (*(int *)&this->field_0x50 == 0x3f800000) {
|
||||
*colorData1Out =
|
||||
(uint)((float)this->texData1_Delta * timeScaledFactor + (float)this->texData1_Base +
|
||||
_DAT_008029cc) >> 0xe & 0xff;
|
||||
colorData1Out =
|
||||
(uint *)((float)this->texData2_Delta * timeScaledFactor + (float)this->texData2_Base +
|
||||
_DAT_008029cc);
|
||||
}
|
||||
else {
|
||||
extendedPrecisionTime = (float10)callIntrinsicDispatcher(puVar1);
|
||||
*colorData1Out =
|
||||
(uint)(float)((float10)this->texData1_Delta * extendedPrecisionTime +
|
||||
(float10)this->texData1_Base + (float10)_DAT_008029cc) >> 0xe & 0xff;
|
||||
colorData1Out =
|
||||
(uint *)(float)((float10)this->texData2_Delta * extendedPrecisionTime +
|
||||
(float10)this->texData2_Base + (float10)_DAT_008029cc);
|
||||
}
|
||||
*colorData2Out = (uint)colorData1Out >> 0xe & 0xff;
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================
|
||||
// matVec3Transform @ 0x7BCA80
|
||||
// ============================================================
|
||||
|
||||
void __fastcall transformVector3ByMatrix4x4(float *param_1,float *param_2,float *param_3)
|
||||
|
||||
{
|
||||
float fVar1;
|
||||
float fVar2;
|
||||
float fVar3;
|
||||
float fVar4;
|
||||
float fVar5;
|
||||
float fVar6;
|
||||
float fVar7;
|
||||
float fVar8;
|
||||
float fVar9;
|
||||
float fVar10;
|
||||
float fVar11;
|
||||
float fVar12;
|
||||
float fVar13;
|
||||
float fVar14;
|
||||
|
||||
fVar1 = param_3[10];
|
||||
fVar2 = param_2[2];
|
||||
fVar3 = param_3[2];
|
||||
fVar4 = *param_2;
|
||||
fVar5 = param_3[6];
|
||||
fVar6 = param_2[1];
|
||||
fVar7 = param_3[0xe];
|
||||
fVar8 = param_3[9];
|
||||
fVar9 = param_2[2];
|
||||
fVar10 = param_3[1];
|
||||
fVar11 = *param_2;
|
||||
fVar12 = param_3[5];
|
||||
fVar13 = param_2[1];
|
||||
fVar14 = param_3[0xd];
|
||||
*param_1 = *param_2 * *param_3 + param_3[4] * param_2[1] + param_3[8] * param_2[2] + param_3[0xc];
|
||||
param_1[1] = fVar12 * fVar13 + fVar10 * fVar11 + fVar8 * fVar9 + fVar14;
|
||||
param_1[2] = fVar5 * fVar6 + fVar3 * fVar4 + fVar1 * fVar2 + fVar7;
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================
|
||||
// buildRotationFromAngle @ 0x7BE490
|
||||
// ============================================================
|
||||
|
||||
float * __fastcall
|
||||
createAxisAngleRotationMatrix3x3(float *param_1,float *param_2,float param_3,char param_4)
|
||||
|
||||
{
|
||||
float fVar1;
|
||||
float fVar2;
|
||||
float10 fVar3;
|
||||
float fVar4;
|
||||
float fVar5;
|
||||
float fVar6;
|
||||
float10 fVar7;
|
||||
undefined *local_1c;
|
||||
undefined *local_18;
|
||||
undefined *local_14;
|
||||
undefined *local_10;
|
||||
undefined *local_c;
|
||||
undefined *local_8;
|
||||
|
||||
local_1c = (undefined *)*param_2;
|
||||
local_18 = (undefined *)param_2[1];
|
||||
local_14 = (undefined *)param_2[2];
|
||||
if (param_4 == '\0') {
|
||||
fVar1 = StaticFloat1_0 /
|
||||
SQRT((float)local_1c * (float)local_1c +
|
||||
(float)local_18 * (float)local_18 + (float)local_14 * (float)local_14);
|
||||
local_1c = (undefined *)((float)local_1c * fVar1);
|
||||
local_18 = (undefined *)((float)local_18 * fVar1);
|
||||
local_14 = (undefined *)(fVar1 * (float)local_14);
|
||||
}
|
||||
fVar3 = (float10)fcos((float10)param_3);
|
||||
fVar7 = (float10)fsin((float10)param_3);
|
||||
fVar1 = (float)fVar3;
|
||||
fVar2 = (float)fVar7;
|
||||
fVar6 = StaticFloat1_0 - fVar1;
|
||||
*param_1 = (float)local_1c * (float)local_1c * fVar6 + fVar1;
|
||||
fVar4 = fVar6 * (float)local_18 * (float)local_1c;
|
||||
param_1[1] = fVar4 + (float)local_14 * fVar2;
|
||||
fVar5 = fVar6 * (float)local_14 * (float)local_1c;
|
||||
param_1[2] = fVar5 - (float)local_18 * fVar2;
|
||||
param_1[3] = fVar4 - (float)local_14 * fVar2;
|
||||
param_1[4] = (float)local_18 * (float)local_18 * fVar6 + fVar1;
|
||||
fVar4 = fVar6 * (float)local_14 * (float)local_18;
|
||||
param_1[5] = (float)local_1c * fVar2 + fVar4;
|
||||
param_1[6] = fVar5 + (float)local_18 * fVar2;
|
||||
param_1[7] = fVar4 - (float)local_1c * fVar2;
|
||||
param_1[8] = (float)local_14 * (float)local_14 * fVar6 + fVar1;
|
||||
return param_1;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================
|
||||
// matVec3Transform2 @ 0x7BCB40
|
||||
// ============================================================
|
||||
|
||||
void __fastcall transformVector4ByMatrix4x4(float *param_1,float *param_2,float *param_3)
|
||||
|
||||
{
|
||||
float fVar1;
|
||||
float fVar2;
|
||||
float fVar3;
|
||||
float fVar4;
|
||||
float fVar5;
|
||||
float fVar6;
|
||||
float fVar7;
|
||||
float fVar8;
|
||||
float fVar9;
|
||||
float fVar10;
|
||||
float fVar11;
|
||||
float fVar12;
|
||||
float fVar13;
|
||||
float fVar14;
|
||||
float fVar15;
|
||||
float fVar16;
|
||||
float fVar17;
|
||||
float fVar18;
|
||||
float fVar19;
|
||||
float fVar20;
|
||||
float fVar21;
|
||||
float fVar22;
|
||||
float fVar23;
|
||||
float fVar24;
|
||||
|
||||
fVar1 = param_3[0xf];
|
||||
fVar2 = param_2[3];
|
||||
fVar3 = param_3[3];
|
||||
fVar4 = *param_2;
|
||||
fVar5 = param_3[0xb];
|
||||
fVar6 = param_2[2];
|
||||
fVar7 = param_3[7];
|
||||
fVar8 = param_2[1];
|
||||
fVar9 = param_3[0xe];
|
||||
fVar10 = param_2[3];
|
||||
fVar11 = param_3[2];
|
||||
fVar12 = *param_2;
|
||||
fVar13 = param_3[10];
|
||||
fVar14 = param_2[2];
|
||||
fVar15 = param_3[6];
|
||||
fVar16 = param_2[1];
|
||||
fVar17 = param_3[0xd];
|
||||
fVar18 = param_2[3];
|
||||
fVar19 = param_3[1];
|
||||
fVar20 = *param_2;
|
||||
fVar21 = param_3[9];
|
||||
fVar22 = param_2[2];
|
||||
fVar23 = param_3[5];
|
||||
fVar24 = param_2[1];
|
||||
*param_1 = *param_2 * *param_3 +
|
||||
param_3[4] * param_2[1] + param_3[8] * param_2[2] + param_3[0xc] * param_2[3];
|
||||
param_1[1] = fVar23 * fVar24 + fVar21 * fVar22 + fVar19 * fVar20 + fVar17 * fVar18;
|
||||
param_1[2] = fVar15 * fVar16 + fVar13 * fVar14 + fVar11 * fVar12 + fVar9 * fVar10;
|
||||
param_1[3] = fVar7 * fVar8 + fVar5 * fVar6 + fVar3 * fVar4 + fVar1 * fVar2;
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
// ============================================================
|
||||
// setupRenderState @ 0x58A230
|
||||
// ============================================================
|
||||
|
||||
void UpdateLightingOffset(void)
|
||||
|
||||
{
|
||||
GetLightingOffset((int)CGxDeviceD3d__device);
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
//! particle_sse — SSE replacements for WoW 1.12.1 particle rendering pipeline.
|
||||
//!
|
||||
//! Compiled as a separate ReleaseFast unit (same pattern as bone_sse.zig / clip_sse.zig).
|
||||
//! Functions are exported and called via `extern fn` from transform44.zig detour hooks.
|
||||
//!
|
||||
//! Assembly references: decompiled/asm_RenderParticleSprites.txt,
|
||||
//! asm_calculateColorValues.txt, asm_SetupParticleRendering.txt
|
||||
|
||||
const V4 = @Vector(4, f32);
|
||||
const V4i = @Vector(4, i32);
|
||||
|
||||
inline fn rf32(addr: u32) f32 {
|
||||
return @as(*align(1) const f32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
inline fn ri32(addr: u32) i32 {
|
||||
return @as(*align(1) const i32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
inline fn ru8(addr: u32) u8 {
|
||||
return @as(*const u8, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
inline fn ru32(addr: u32) u32 {
|
||||
return @as(*align(1) const u32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
inline fn wf32(addr: u32, val: f32) void {
|
||||
@as(*align(1) f32, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
|
||||
inline fn wu32(addr: u32, val: u32) void {
|
||||
@as(*align(1) u32, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
|
||||
inline fn wu8(addr: u32, val: u8) void {
|
||||
@as(*u8, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// calculateColorValues (0x7B9B10)
|
||||
// =============================================================================
|
||||
//
|
||||
// __thiscall(ECX=colorCtx, stack: time, scale, outColor, outAlpha1, outAlpha2, outFloat)
|
||||
// RET 0x18 (6 stack params)
|
||||
//
|
||||
// ColorCtx layout:
|
||||
// +0x00..0x03: base color bytes [B, G, R, A] (4 bytes)
|
||||
// +0x04: delta_alpha (i32)
|
||||
// +0x08: delta_red (i32)
|
||||
// +0x0C: delta_green (i32)
|
||||
// +0x10: delta_blue (i32)
|
||||
// +0x14: alpha1_base (i32)
|
||||
// +0x18: alpha1_delta (i32)
|
||||
// +0x1C: alpha2_base (i32)
|
||||
// +0x20: alpha2_delta (i32)
|
||||
// +0x24: float_base (f32)
|
||||
// +0x28: float_scale (f32)
|
||||
// +0x2C: time_base (f32)
|
||||
// +0x30: time_scale (f32)
|
||||
// +0x50: alpha_power (f32, 1.0 = linear, else calls pow)
|
||||
//
|
||||
// Algorithm:
|
||||
// t = (time - ctx.timeBase) * ctx.timeScale * CONST1 + CONST2
|
||||
// For each color channel (A,R,G,B):
|
||||
// val = (float)delta * t + (float)base_byte
|
||||
// alpha channel only: val *= scale
|
||||
// val += MAGIC (float-to-byte trick constant at 0x8029CC)
|
||||
// outColor[ch] = (byte)(float_bits >> 14)
|
||||
// outFloat = t * ctx.floatScale + ctx.floatBase
|
||||
// For alpha outputs:
|
||||
// if ctx.alphaPower == 1.0: linear interp
|
||||
// else: pow(t * alphaPower, ...) path
|
||||
//
|
||||
// The "float bits >> 14" is a classic fast float-to-byte: add a large power-of-2
|
||||
// magic number so the integer value sits in the mantissa bits, then extract.
|
||||
// =============================================================================
|
||||
|
||||
const CC = std.builtin.CallingConvention;
|
||||
const TC: CC = .{ .x86_thiscall = .{} };
|
||||
|
||||
const std = @import("std");
|
||||
|
||||
/// SSE replacement for calculateColorValues.
|
||||
/// Thiscall: ECX=ctx, stack params: time(f32), scale(f32), outColor(ptr), outAlpha1(ptr), outAlpha2(ptr), outFloat(ptr)
|
||||
export fn calcColorValues_SSE(
|
||||
ctx: u32,
|
||||
time_bits: u32,
|
||||
scale_bits: u32,
|
||||
out_color: u32,
|
||||
out_alpha1: u32,
|
||||
out_alpha2: u32,
|
||||
out_float: u32,
|
||||
) callconv(TC) void {
|
||||
const time: f32 = @bitCast(time_bits);
|
||||
const scale: f32 = @bitCast(scale_bits);
|
||||
|
||||
// Step 1: Compute interpolation parameter t
|
||||
const t = (time - rf32(ctx + 0x2C)) * rf32(ctx + 0x30) * rf32(0x808AAC) + rf32(0x807A3C);
|
||||
|
||||
// Step 2: Compute 4 color channels
|
||||
// Load base bytes and deltas
|
||||
const base_a: f32 = @floatFromInt(@as(i32, ru8(ctx + 3)));
|
||||
const base_r: f32 = @floatFromInt(@as(i32, ru8(ctx + 2)));
|
||||
const base_g: f32 = @floatFromInt(@as(i32, ru8(ctx + 1)));
|
||||
const base_b: f32 = @floatFromInt(@as(i32, ru8(ctx + 0)));
|
||||
|
||||
const delta_a: f32 = @floatFromInt(ri32(ctx + 0x04));
|
||||
const delta_r: f32 = @floatFromInt(ri32(ctx + 0x08));
|
||||
const delta_g: f32 = @floatFromInt(ri32(ctx + 0x0C));
|
||||
const delta_b: f32 = @floatFromInt(ri32(ctx + 0x10));
|
||||
|
||||
const magic: f32 = rf32(0x8029CC);
|
||||
|
||||
// Alpha channel: (delta * t + base) * scale + magic
|
||||
const alpha_f = @mulAdd(f32, delta_a, t, base_a) * scale + magic;
|
||||
// RGB channels: delta * t + base + magic (no scale)
|
||||
const red_f = @mulAdd(f32, delta_r, t, base_r) + magic;
|
||||
const green_f = @mulAdd(f32, delta_g, t, base_g) + magic;
|
||||
const blue_f = @mulAdd(f32, delta_b, t, base_b) + magic;
|
||||
|
||||
// Extract bytes via float-bits >> 14 trick
|
||||
const alpha_byte: u8 = @truncate(@as(u32, @bitCast(alpha_f)) >> 14);
|
||||
const red_byte: u8 = @truncate(@as(u32, @bitCast(red_f)) >> 14);
|
||||
const green_byte: u8 = @truncate(@as(u32, @bitCast(green_f)) >> 14);
|
||||
const blue_byte: u8 = @truncate(@as(u32, @bitCast(blue_f)) >> 14);
|
||||
|
||||
// Store color bytes: [B, G, R, A] at outColor
|
||||
wu8(out_color + 0, blue_byte);
|
||||
wu8(out_color + 1, green_byte);
|
||||
wu8(out_color + 2, red_byte);
|
||||
wu8(out_color + 3, alpha_byte);
|
||||
|
||||
// Step 3: Float output = t * ctx.floatScale + ctx.floatBase
|
||||
wf32(out_float, @mulAdd(f32, t, rf32(ctx + 0x28), rf32(ctx + 0x24)));
|
||||
|
||||
// Step 4: Alpha outputs
|
||||
const alpha_power = ru32(ctx + 0x50);
|
||||
if (alpha_power == 0x3F800000) {
|
||||
// Fast path: alphaPower == 1.0 (linear)
|
||||
const a1_val = @mulAdd(f32, @as(f32, @floatFromInt(ri32(ctx + 0x18))), t, @as(f32, @floatFromInt(ri32(ctx + 0x14)))) + magic;
|
||||
const a2_val = @mulAdd(f32, @as(f32, @floatFromInt(ri32(ctx + 0x20))), t, @as(f32, @floatFromInt(ri32(ctx + 0x1C)))) + magic;
|
||||
|
||||
wu32(out_alpha1, (@as(u32, @bitCast(a1_val)) >> 14) & 0xFF);
|
||||
wu32(out_alpha2, (@as(u32, @bitCast(a2_val)) >> 14) & 0xFF);
|
||||
} else {
|
||||
// Slow path: pow scaling. Call game's pow function.
|
||||
// 0x73F90A: __cdecl pow — takes ST(0)=base, ST(1)=exponent, returns ST(0)
|
||||
// t_scaled = pow(t * alphaPower, ???)
|
||||
// For now, fall back to scalar computation matching the original exactly.
|
||||
const ap: f32 = @bitCast(alpha_power);
|
||||
const t_scaled = t * ap;
|
||||
|
||||
// The original calls 0x73F90A with ST(0)=t_scaled, ST(1)=loaded from [0x8015B8] (qword)
|
||||
// This is __CIpow (MSVC intrinsic pow) — ST(1)=exponent (from 0x8015B8), ST(0)=base
|
||||
// We need the exponent constant. For now use @exp2/@log2 to compute pow.
|
||||
// Actually: the original loads FLD qword [0x8015B8] THEN calls __CIpow.
|
||||
// __CIpow expects ST(0)=x, ST(1)=y, computes x^y. So: pow(t_scaled, const_at_8015B8).
|
||||
// The constant at 0x8015B8 is a f64. We read it and use std.math.pow.
|
||||
const exp_val: f64 = @as(*align(1) const f64, @ptrFromInt(0x8015B8)).*;
|
||||
const t_pow: f32 = @floatCast(std.math.pow(f64, @as(f64, t_scaled), exp_val));
|
||||
|
||||
const a1_delta: f32 = @floatFromInt(ri32(ctx + 0x18));
|
||||
const a1_base: f32 = @floatFromInt(ri32(ctx + 0x14));
|
||||
const a1_val = @mulAdd(f32, a1_delta, t_pow, a1_base) + magic;
|
||||
|
||||
const a2_delta: f32 = @floatFromInt(ri32(ctx + 0x20));
|
||||
const a2_base: f32 = @floatFromInt(ri32(ctx + 0x1C));
|
||||
const a2_val = @mulAdd(f32, a2_delta, t_pow, a2_base) + magic;
|
||||
|
||||
wu32(out_alpha1, (@as(u32, @bitCast(a1_val)) >> 14) & 0xFF);
|
||||
wu32(out_alpha2, (@as(u32, @bitCast(a2_val)) >> 14) & 0xFF);
|
||||
}
|
||||
}
|
||||
@@ -22,6 +22,7 @@ extern fn rayTriangleIntersection(u32, u32, u32, u32, u32, u32) u32;
|
||||
extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void;
|
||||
extern fn multiplyMatrix4x4(u32, u32, u32) u32;
|
||||
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
|
||||
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
|
||||
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
|
||||
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
|
||||
@@ -1008,7 +1009,10 @@ fn linkedlistDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaq
|
||||
return ret;
|
||||
}
|
||||
fn colorDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
// a=ECX(ctx), b=EDX(unused), c=time, d=scale, e=outColor, f=outAlpha1, g=outAlpha2, h=outFloat
|
||||
const s = rdtsc();
|
||||
// SSE replacement benches at 10.7x but no in-game gain — dominated by L1 cache misses
|
||||
// on scattered ColorCtx structs. Fix: inline into full RenderParticleSprites replacement.
|
||||
const ret = color_hook.callOriginal(.{ a, b, c, d, e, f, g, h });
|
||||
prof.color_cycles +|= rdtsc() - s;
|
||||
prof.color_calls +|= 1;
|
||||
|
||||
Reference in New Issue
Block a user