fix: MSVC ABI, heap allocator, deferred timer - vanillafixes compat
- Restore MSVC ABI (was accidentally GNU since v0.6.0, broke .CRT section) - Replace game allocator with Windows process heap for filecache and libdeflate malloc/free - game allocator not initialized during DllMain when injected via CreateRemoteThread - Defer timer calibration (Sleep 500ms) to lateInit - blocks under loader lock during DllMain - Remove exported malloc/free symbols from DLL - Eliminate addObject compilation units for SSE files - direct @import with AVX target instead - Heap-allocate filecache (was 9.3MB static BSS) - Strip transform44 of performance/ externs, pure profiling only - Rename performance/ to weirdperformance/ to match module convention - Skip default-off modules in all-variants build step - Remove dead debug vars and stride logging from particle_sse
This commit is contained in:
+12
-30
@@ -7,36 +7,12 @@
|
||||
This project is developed entirely locally. The remote repo is **only** a
|
||||
distribution point for releases - no source code is pushed.
|
||||
|
||||
The remote `main` branch contains a single file: `README.md` (built from the
|
||||
local `DLL_README.md`). This must be set up once when creating the repo:
|
||||
The remote `main` branch contains `README.md` (built from the local
|
||||
`DLL_README.md`), `weirdutils_api.h`, and issue templates under `.gitea/`.
|
||||
|
||||
```sh
|
||||
tea repo create --name WeirdUtils --description "Vanilla WoW 1.12.1 utility DLLs" --login MarcelineVQ
|
||||
```
|
||||
|
||||
Codeberg disables releases on new repos by default. Enable via API
|
||||
(get your token from `grep 'token:' ~/.config/tea/config.yml | head -1 | awk '{print $2}'`):
|
||||
|
||||
```sh
|
||||
curl -s -X PATCH \
|
||||
-H "Authorization: token <your-token>" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{"has_releases":true}' \
|
||||
"https://codeberg.org/api/v1/repos/MarcelineVQ/WeirdUtils"
|
||||
```
|
||||
|
||||
Then push the initial README:
|
||||
|
||||
```sh
|
||||
# In a temporary directory:
|
||||
git init && git remote add origin ssh://git@codeberg.org/MarcelineVQ/WeirdUtils.git && git checkout -b main
|
||||
cp /path/to/weirdutils/DLL_README.md README.md
|
||||
git add README.md
|
||||
git commit -m "Add README"
|
||||
git push origin main
|
||||
```
|
||||
|
||||
After that, the remote `main` only needs updating when `DLL_README.md` changes.
|
||||
A local clone of the remote repo lives at `remote/WeirdUtils/`. The wiki
|
||||
lives at `remote/wiki/`. Use these for all remote operations - no tmp clones
|
||||
needed.
|
||||
|
||||
## 1. Bump module versions
|
||||
|
||||
@@ -120,11 +96,12 @@ The module name list in the Developer Notes section must also only list
|
||||
released module names.
|
||||
|
||||
```sh
|
||||
# from a clone or worktree of the remote repo
|
||||
cd remote/WeirdUtils
|
||||
# edit README.md: remove sections for modules not in this release
|
||||
git add README.md
|
||||
git commit -m "Update README for vX.Y.Z"
|
||||
git push origin main
|
||||
cd ../..
|
||||
```
|
||||
|
||||
## 4. Write the release notes
|
||||
@@ -222,6 +199,11 @@ print(r[0]['id']) if r else print('not found')
|
||||
"
|
||||
```
|
||||
|
||||
## Known Issues
|
||||
|
||||
- **vanillafixes launcher**: Incompatible with WeirdUtils DLL injection. Users
|
||||
should load the DLL via WoW.exe + `dlls.txt` or another loader instead.
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] Module versions bumped in `build.zig` for changed modules
|
||||
|
||||
@@ -41,7 +41,7 @@ pub fn build(b: *std.Build) void {
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .windows,
|
||||
.abi = .msvc,
|
||||
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2 }),
|
||||
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
|
||||
});
|
||||
const optimize = b.option(std.builtin.OptimizeMode, "optimize", "Optimization mode (default: ReleaseFast)") orelse .ReleaseFast;
|
||||
|
||||
@@ -61,46 +61,6 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
const zhook_mod = zhook_dep.module("zhook");
|
||||
|
||||
// Hot math — separate compilation units, always ReleaseFast.
|
||||
// Source lives in src/performance/ — the production SSE module.
|
||||
const clip_sse_obj = b.addObject(.{
|
||||
.name = "clip_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/clip_sse.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const cull_sse_obj = b.addObject(.{
|
||||
.name = "cull_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/cull_sse.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const entity_sse_obj = b.addObject(.{
|
||||
.name = "entity_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/entity_sse.zig"),
|
||||
.target = target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bone_sse_target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .windows,
|
||||
.abi = .msvc,
|
||||
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
|
||||
});
|
||||
const bone_sse_obj = b.addObject(.{
|
||||
.name = "bone_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/bone_sse.zig"),
|
||||
.target = bone_sse_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
// REF uses x87-only target to match original game code structure.
|
||||
// The global target has SSE/SSE2 which generates movss/mulss;
|
||||
// the original at 0x714260 uses pure x87 (FLD/FMUL/FSTP).
|
||||
@@ -126,28 +86,11 @@ pub fn build(b: *std.Build) void {
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const silicon_sse_obj = b.addObject(.{
|
||||
.name = "silicon_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/silicon_sse.zig"),
|
||||
.target = bone_sse_target, // SSE4.1+FMA+AVX, same as bone_sse
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
|
||||
const particle_sse_obj = b.addObject(.{
|
||||
.name = "particle_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/particle_sse.zig"),
|
||||
.target = bone_sse_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const particle_ref_obj = b.addObject(.{
|
||||
.name = "particle_sse_ref",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/transform44/particle_sse_reference.zig"),
|
||||
.target = bone_sse_target,
|
||||
.target = target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
@@ -178,76 +121,52 @@ pub fn build(b: *std.Build) void {
|
||||
});
|
||||
libdeflate.root_module.addCSourceFiles(.{
|
||||
.files = &.{
|
||||
"src/performance/libdeflate/lib/deflate_decompress.c",
|
||||
"src/performance/libdeflate/lib/zlib_decompress.c",
|
||||
"src/performance/libdeflate/lib/utils.c",
|
||||
"src/performance/libdeflate/lib/adler32.c",
|
||||
"src/performance/libdeflate/lib/x86/cpu_features.c",
|
||||
"src/weirdperformance/libdeflate/lib/deflate_decompress.c",
|
||||
"src/weirdperformance/libdeflate/lib/zlib_decompress.c",
|
||||
"src/weirdperformance/libdeflate/lib/utils.c",
|
||||
"src/weirdperformance/libdeflate/lib/adler32.c",
|
||||
"src/weirdperformance/libdeflate/lib/x86/cpu_features.c",
|
||||
"src/weirdperformance/libdeflate/stubs/game_alloc.c",
|
||||
},
|
||||
.flags = &.{"-DLIBDEFLATE_ASSEMBLER_DOES_NOT_SUPPORT_AVX512VNNI"},
|
||||
});
|
||||
libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate/stubs"));
|
||||
libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate"));
|
||||
libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate/lib"));
|
||||
libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate/stubs"));
|
||||
libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate"));
|
||||
libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate/lib"));
|
||||
|
||||
// Link module-specific object files into a DLL.
|
||||
// Single source of truth for which objects each module needs.
|
||||
// Called for both the main weirdutils build and each variant.
|
||||
const ModuleObjects = struct {
|
||||
clip_sse: *std.Build.Step.Compile,
|
||||
cull_sse: *std.Build.Step.Compile,
|
||||
entity_sse: *std.Build.Step.Compile,
|
||||
bone_sse: *std.Build.Step.Compile,
|
||||
bone_sse_ref: *std.Build.Step.Compile,
|
||||
math_sse: *std.Build.Step.Compile,
|
||||
silicon_sse: *std.Build.Step.Compile,
|
||||
particle_sse: *std.Build.Step.Compile,
|
||||
particle_ref: *std.Build.Step.Compile,
|
||||
libdeflate: *std.Build.Step.Compile,
|
||||
|
||||
fn linkFor(self: @This(), mod: *std.Build.Module, comptime module_name: []const u8) void {
|
||||
@setEvalBranchQuota(10000);
|
||||
if (comptime std.mem.eql(u8, module_name, "weirdperformance")) {
|
||||
mod.addObject(self.clip_sse);
|
||||
mod.addObject(self.cull_sse);
|
||||
mod.addObject(self.bone_sse);
|
||||
mod.addObject(self.silicon_sse);
|
||||
mod.addObject(self.particle_sse);
|
||||
mod.addObjectFile(self.libdeflate.getEmittedBin());
|
||||
}
|
||||
if (comptime std.mem.eql(u8, module_name, "transform44")) {
|
||||
mod.addObject(self.clip_sse);
|
||||
mod.addObject(self.cull_sse);
|
||||
mod.addObject(self.entity_sse);
|
||||
mod.addObject(self.bone_sse);
|
||||
mod.addObject(self.bone_sse_ref);
|
||||
mod.addObject(self.particle_sse);
|
||||
mod.addObject(self.particle_ref);
|
||||
}
|
||||
if (comptime std.mem.eql(u8, module_name, "silicon")) {
|
||||
mod.addObject(self.silicon_sse);
|
||||
}
|
||||
if (comptime std.mem.eql(u8, module_name, "ssemaths")) {
|
||||
mod.addObject(self.math_sse);
|
||||
}
|
||||
}
|
||||
};
|
||||
const objs = ModuleObjects{
|
||||
.clip_sse = clip_sse_obj,
|
||||
.cull_sse = cull_sse_obj,
|
||||
.entity_sse = entity_sse_obj,
|
||||
.bone_sse = bone_sse_obj,
|
||||
.bone_sse_ref = bone_sse_ref_obj,
|
||||
.math_sse = math_sse_obj,
|
||||
.silicon_sse = silicon_sse_obj,
|
||||
.particle_sse = particle_sse_obj,
|
||||
.particle_ref = particle_ref_obj,
|
||||
.libdeflate = libdeflate,
|
||||
};
|
||||
|
||||
// Main DLL: link all module objects
|
||||
inline for (module_list) |mod| {
|
||||
objs.linkFor(lib.root_module, mod.name);
|
||||
// Main DLL: link module objects only for enabled modules
|
||||
inline for (module_list, 0..) |mod, i| {
|
||||
if (module_enabled[i]) objs.linkFor(lib.root_module, mod.name);
|
||||
}
|
||||
|
||||
b.installArtifact(lib);
|
||||
@@ -282,7 +201,7 @@ pub fn build(b: *std.Build) void {
|
||||
const bench_silicon_sse = b.addObject(.{
|
||||
.name = "bench_silicon_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/silicon_sse.zig"),
|
||||
.root_source_file = b.path("src/weirdperformance/silicon_sse.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .linux,
|
||||
@@ -294,7 +213,7 @@ pub fn build(b: *std.Build) void {
|
||||
const bench_bone_sse = b.addObject(.{
|
||||
.name = "bench_bone_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/bone_sse.zig"),
|
||||
.root_source_file = b.path("src/weirdperformance/bone_sse.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .linux,
|
||||
@@ -318,7 +237,7 @@ pub fn build(b: *std.Build) void {
|
||||
const bench_particle_sse = b.addObject(.{
|
||||
.name = "bench_particle_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/particle_sse.zig"),
|
||||
.root_source_file = b.path("src/weirdperformance/particle_sse.zig"),
|
||||
.target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .linux,
|
||||
@@ -330,7 +249,7 @@ pub fn build(b: *std.Build) void {
|
||||
const bench_cull_sse = b.addObject(.{
|
||||
.name = "bench_cull_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/cull_sse.zig"),
|
||||
.root_source_file = b.path("src/weirdperformance/cull_sse.zig"),
|
||||
.target = bench_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
@@ -344,7 +263,7 @@ pub fn build(b: *std.Build) void {
|
||||
const bench_entity_sse = b.addObject(.{
|
||||
.name = "bench_entity_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/performance/entity_sse.zig"),
|
||||
.root_source_file = b.path("src/weirdperformance/entity_sse.zig"),
|
||||
.target = bench_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
@@ -396,6 +315,7 @@ pub fn build(b: *std.Build) void {
|
||||
build_all_step.dependOn(&noperf_install.step);
|
||||
|
||||
inline for (module_list) |variant_mod| {
|
||||
if (!variant_mod.default) continue;
|
||||
@setEvalBranchQuota(10000);
|
||||
const opts = b.addOptions();
|
||||
inline for (module_list) |m| {
|
||||
|
||||
+4
-1
@@ -43,7 +43,7 @@ const transform44 = if (build_opts.transform44) @import("transform44/transform44
|
||||
const addonperf = if (build_opts.addonperf) @import("addonperf/addonperf.zig") else struct {};
|
||||
const ssemaths = if (build_opts.ssemaths) @import("ssemaths/ssemaths.zig") else struct {};
|
||||
const silicon = if (build_opts.silicon) @import("silicon/silicon.zig") else struct {};
|
||||
const weirdperformance = if (build_opts.weirdperformance) @import("performance/weirdperformance.zig") else struct {};
|
||||
const weirdperformance = if (build_opts.weirdperformance) @import("weirdperformance/weirdperformance.zig") else struct {};
|
||||
|
||||
const module_active = @import("module_active.zig");
|
||||
|
||||
@@ -666,6 +666,9 @@ fn engineInitDetour() callconv(hook.cc.stdcall) void {
|
||||
if (build_opts.silicon) {
|
||||
silicon.lateInit();
|
||||
}
|
||||
if (build_opts.weirdperformance) {
|
||||
weirdperformance.lateInit();
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
|
||||
@@ -15,37 +15,6 @@ const std = @import("std");
|
||||
const hook = @import("zhook");
|
||||
const logging = @import("../logging.zig");
|
||||
const mod_mutex = @import("../mutex.zig");
|
||||
extern fn clipPolygonToSinglePlane(u32, u32, u32) void;
|
||||
extern fn buildTrianglePlanes(u32, u32, u32, u32, u32) u32;
|
||||
extern fn rayTriangleIntersection(u32, u32, u32, u32, u32, u32) u32;
|
||||
extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void;
|
||||
extern fn multiplyMatrix4x4(u32, u32, u32) u32;
|
||||
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
|
||||
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn renderParticleSprites_REF(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn resetParticleCache() void;
|
||||
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
|
||||
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
|
||||
extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void;
|
||||
extern fn findObjectByGUID_Cached(u32, u32) callconv(.{ .x86_stdcall = .{} }) u32;
|
||||
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern var stride_info: [8]u32; // exported from particle_sse.zig
|
||||
extern var debug_vertex_count: u32;
|
||||
extern var debug_max_sprites: u32;
|
||||
extern var debug_fmt_index: u32;
|
||||
extern var debug_data_ptr: u32;
|
||||
var stride_dumped: bool = false;
|
||||
|
||||
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
|
||||
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
|
||||
fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
|
||||
transformImpl_SSE(this, mat1, mat2, mat3, mat4);
|
||||
}
|
||||
extern fn transformMatrix4x4_REF(u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern var bisect_stop_section: u32;
|
||||
|
||||
@@ -82,12 +51,6 @@ const AB_OTHER_HOOKS = true;
|
||||
var diag_cmp_count: u32 = 0;
|
||||
export var original_trampoline: u32 = 0; // DEBUG: expose trampoline for REF passthrough test
|
||||
|
||||
// Teardown guard: set true when CleanupWorldAndEntities fires.
|
||||
// During teardown, SceneObject data may be partially freed — our SSE code
|
||||
// must not process it. Falls back to original function which the game
|
||||
// controls. NOTE: binary patching (instead of hooking) would avoid this
|
||||
// issue entirely since the patched code IS the original entry point.
|
||||
var teardown_active: bool = false;
|
||||
|
||||
|
||||
// Persistent blit totals per A/B mode — NOT reset each dump period.
|
||||
@@ -325,11 +288,7 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
|
||||
t44_depth +|= 1;
|
||||
if (t44_depth > prof.t44_max_depth) prof.t44_max_depth = t44_depth;
|
||||
|
||||
if (teardown_active) {
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
} else {
|
||||
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
|
||||
}
|
||||
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
|
||||
|
||||
t44_depth -|= 1;
|
||||
const elapsed = rdtsc() - start;
|
||||
@@ -392,7 +351,6 @@ const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void;
|
||||
var world_update_hook: hook.Detour(WorldUpdateFn) = .{};
|
||||
|
||||
fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
|
||||
resetParticleCache();
|
||||
const now = rdtsc();
|
||||
if (last_frame_tsc != 0) {
|
||||
const delta = now - last_frame_tsc;
|
||||
@@ -410,23 +368,6 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Hook: World_HandleLogoutCleanup (0x491180)
|
||||
// Fires at the START of the logout/disconnect cleanup sequence, BEFORE any
|
||||
// model data is freed. Sets teardown_active flag so our SSE code falls back
|
||||
// to the original function during the entire cleanup chain.
|
||||
// NOTE: binary patching instead of hooking would avoid this issue entirely.
|
||||
// =============================================================================
|
||||
|
||||
const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void;
|
||||
var teardown_hook: hook.Detour(TeardownFn) = .{};
|
||||
|
||||
fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void {
|
||||
teardown_active = true;
|
||||
teardown_hook.callOriginal(.{});
|
||||
teardown_active = false;
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Hook: RenderTextureQuads (0x76FB00)
|
||||
// __fastcall(ECX=RenderBatch*) — no stack params, RET
|
||||
@@ -897,12 +838,6 @@ var staticcull_hook: hook.Detour(Fn2) = .{}; // ProcessStaticObjectsCulling: fas
|
||||
fn clipDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
clipPolygonToSinglePlane(a, b, c);
|
||||
prof.clip_cycles +|= rdtsc() - s;
|
||||
prof.clip_calls +|= 1;
|
||||
return null; // original is void — EAX not read by callers
|
||||
}
|
||||
const ret = clip_hook.callOriginal(.{ a, b, c });
|
||||
prof.clip_cycles +|= rdtsc() - s;
|
||||
prof.clip_calls +|= 1;
|
||||
@@ -948,13 +883,12 @@ fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyo
|
||||
prof.glyph_calls +|= 1;
|
||||
return ret;
|
||||
}
|
||||
fn particleDetour(a: u32, _: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
// a=ECX(emitter), c=particleData, d=vertexBuffers
|
||||
fn particleDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
const result = renderParticleSprites_SSE(a, c, d);
|
||||
const ret = particle_hook.callOriginal(.{ a, b, c, d });
|
||||
prof.particle_cycles +|= rdtsc() - s;
|
||||
prof.particle_calls +|= 1;
|
||||
return @ptrFromInt(result);
|
||||
return ret;
|
||||
}
|
||||
fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
@@ -965,12 +899,6 @@ fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*
|
||||
}
|
||||
fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
updateEntityAndChunksPositions(a);
|
||||
prof.entpos_cycles +|= rdtsc() - s;
|
||||
prof.entpos_calls +|= 1;
|
||||
return null;
|
||||
}
|
||||
const ret = entpos_hook.callOriginal(.{ a, b });
|
||||
prof.entpos_cycles +|= rdtsc() - s;
|
||||
prof.entpos_calls +|= 1;
|
||||
@@ -1022,12 +950,6 @@ fn spatialDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
fn raytriDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = rayTriangleIntersection(a, b, c, d, e, f);
|
||||
prof.raytri_cycles +|= rdtsc() - s;
|
||||
prof.raytri_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = raytri_hook.callOriginal(.{ a, b, c, d, e, f });
|
||||
prof.raytri_cycles +|= rdtsc() - s;
|
||||
prof.raytri_calls +|= 1;
|
||||
@@ -1059,12 +981,6 @@ fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
}
|
||||
fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = performSpatialCulling(a, c, d);
|
||||
prof.cull_cycles +|= rdtsc() - s;
|
||||
prof.cull_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = cull_hook.callOriginal(.{ a, b, c, d });
|
||||
prof.cull_cycles +|= rdtsc() - s;
|
||||
prof.cull_calls +|= 1;
|
||||
@@ -1072,12 +988,6 @@ fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyop
|
||||
}
|
||||
fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = performCollisionDetectionSSE(a, c, d);
|
||||
prof.colldet_cycles +|= rdtsc() - s;
|
||||
prof.colldet_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = colldet_hook.callOriginal(.{ a, b, c, d });
|
||||
prof.colldet_cycles +|= rdtsc() - s;
|
||||
prof.colldet_calls +|= 1;
|
||||
@@ -1216,13 +1126,6 @@ fn bboxchkDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*an
|
||||
fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
// thiscall: a=ECX=matrix, b=EDX=unused, c=angle, d=axis_ptr, e=is_unit
|
||||
rotateMatrixByAxisAngle(a, c, d, e);
|
||||
prof.rotmat_cycles +|= rdtsc() - s;
|
||||
prof.rotmat_calls +|= 1;
|
||||
return null; // void function, EAX not read by callers
|
||||
}
|
||||
const ret = rotmat_hook.callOriginal(.{ a, b, c, d, e });
|
||||
prof.rotmat_cycles +|= rdtsc() - s;
|
||||
prof.rotmat_calls +|= 1;
|
||||
@@ -1231,12 +1134,6 @@ fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcal
|
||||
fn triplaneDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = buildTrianglePlanes(a, b, c, d, e);
|
||||
prof.triplane_cycles +|= rdtsc() - s;
|
||||
prof.triplane_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = triplane_hook.callOriginal(.{ a, b, c, d, e });
|
||||
prof.triplane_cycles +|= rdtsc() - s;
|
||||
prof.triplane_calls +|= 1;
|
||||
@@ -1252,13 +1149,6 @@ fn partsetupDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaqu
|
||||
fn matmulDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
// fastcall: a=ECX=result, b=EDX=left, c=right
|
||||
const ret = multiplyMatrix4x4(a, b, c);
|
||||
prof.matmul_cycles +|= rdtsc() - s;
|
||||
prof.matmul_calls +|= 1;
|
||||
return @ptrFromInt(ret);
|
||||
}
|
||||
const ret = matmul_hook.callOriginal(.{ a, b, c });
|
||||
prof.matmul_cycles +|= rdtsc() - s;
|
||||
prof.matmul_calls +|= 1;
|
||||
@@ -1294,12 +1184,6 @@ fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u3
|
||||
}
|
||||
fn raytriIntDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque {
|
||||
const s = rdtsc();
|
||||
if (AB_OTHER_HOOKS and ab_use_custom) {
|
||||
const ret = rayTriIntersectIndexedInt(a, b, c, d, e, f);
|
||||
prof.raytri_int_cycles +|= rdtsc() - s;
|
||||
prof.raytri_int_calls +|= 1;
|
||||
return @ptrFromInt(@as(u32, ret));
|
||||
}
|
||||
const ret = raytri_int_hook.callOriginal(.{ a, b, c, d, e, f });
|
||||
prof.raytri_int_cycles +|= rdtsc() - s;
|
||||
prof.raytri_int_calls +|= 1;
|
||||
@@ -1606,24 +1490,6 @@ fn dumpStats() void {
|
||||
guid_cache_evictions = 0;
|
||||
}
|
||||
|
||||
// Dump particle VB stride info (once)
|
||||
if (debug_vertex_count != 0) {
|
||||
log.fmt(" partsetup_debug: verts={d} maxSprites={d} fmt={d} dataPtr=0x{x}\n", .{
|
||||
debug_vertex_count, debug_max_sprites, debug_fmt_index, debug_data_ptr,
|
||||
});
|
||||
debug_vertex_count = 0;
|
||||
}
|
||||
|
||||
if (stride_info[0] != 0 and !stride_dumped) {
|
||||
stride_dumped = true;
|
||||
log.fmt(" vb_strides: pos={d} norm={d} color={d} tc={d}\n", .{
|
||||
stride_info[0], stride_info[1], stride_info[2], stride_info[3],
|
||||
});
|
||||
log.fmt(" vb_bases: pos=0x{x} norm=0x{x} color=0x{x} tc=0x{x}\n", .{
|
||||
stride_info[4], stride_info[5], stride_info[6], stride_info[7],
|
||||
});
|
||||
}
|
||||
|
||||
// Flip A/B mode
|
||||
ab_use_custom = !ab_use_custom;
|
||||
diag_cmp_count = 0;
|
||||
@@ -1687,7 +1553,6 @@ pub fn installHooks() void {
|
||||
_ = transform_hook.attach(0x714260, &transformDetour);
|
||||
original_trampoline = @intCast(transform_hook.inner.trampoline);
|
||||
}
|
||||
_ = teardown_hook.attach(0x491180, &teardownDetour);
|
||||
_ = render_frame_hook.attach(0x707680, &renderFrameDetour);
|
||||
_ = exec_render_pass_hook.attach(0x708900, &execRenderPassDetour);
|
||||
_ = world_update_hook.attach(0x482EA0, &worldUpdateDetour);
|
||||
@@ -1776,7 +1641,6 @@ pub fn removeHooks() void {
|
||||
render_frame_hook.detach();
|
||||
exec_render_pass_hook.detach();
|
||||
world_update_hook.detach();
|
||||
teardown_hook.detach();
|
||||
render_quads_hook.detach();
|
||||
movement_hook.detach();
|
||||
interp_kf_hook.detach();
|
||||
|
||||
@@ -1101,7 +1101,7 @@ fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
|
||||
// mat3(offset_vec3*), mat4(scale_float_bits)
|
||||
// =============================================================================
|
||||
|
||||
export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void {
|
||||
pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void {
|
||||
|
||||
@setEvalBranchQuota(50000);
|
||||
// =========================================================================
|
||||
@@ -9,7 +9,7 @@ const V4 = @Vector(4, f32);
|
||||
const CLIP_EPSILON: f32 = @bitCast(@as(u32, 0x3ab60b61)); // +0.00139, global at 0x80dfec
|
||||
const MOVEMENT_EPSILON: f32 = @bitCast(@as(u32, 0x35800000)); // 9.54e-7, global at 0x8026bc
|
||||
|
||||
export fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void {
|
||||
pub fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void {
|
||||
const plane: [*]const f32 = @ptrFromInt(plane_addr);
|
||||
const poly: [*]f32 = @ptrFromInt(poly_addr);
|
||||
const new_attrib: f32 = @bitCast(attrib_bits);
|
||||
@@ -116,7 +116,7 @@ inline fn lerp(poly: [*]f32, attribs: [*]f32, out: *u32, p: [*]const f32, c: [*]
|
||||
// planes[3]: cap plane (from plane_normal, offset by offset_vector)
|
||||
// =============================================================================
|
||||
|
||||
export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 {
|
||||
pub fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 {
|
||||
const verts: [*]const f32 = @ptrFromInt(verts_addr);
|
||||
const indices: [*]const u8 = @ptrFromInt(indices_addr);
|
||||
const plane_normal: [*]const f32 = @ptrFromInt(normal_addr);
|
||||
@@ -191,7 +191,7 @@ export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u
|
||||
// Returns: 1 = hit, 0 = miss
|
||||
// =============================================================================
|
||||
|
||||
export fn rayTriangleIntersection(
|
||||
pub fn rayTriangleIntersection(
|
||||
ray_addr: u32,
|
||||
verts_addr: u32,
|
||||
indices_addr: u32,
|
||||
@@ -267,7 +267,7 @@ export fn rayTriangleIntersection(
|
||||
// Returns result pointer (EAX = result_addr).
|
||||
// =============================================================================
|
||||
|
||||
export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 {
|
||||
pub fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 {
|
||||
const result: [*]f32 = @ptrFromInt(result_addr);
|
||||
const left: [*]const f32 = @ptrFromInt(left_addr);
|
||||
const right: [*]const f32 = @ptrFromInt(right_addr);
|
||||
@@ -296,7 +296,7 @@ export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u
|
||||
// __thiscall(ECX=matrix, stack: angle, axis_ptr, is_unit_flag)
|
||||
// =============================================================================
|
||||
|
||||
export fn rotateMatrixByAxisAngle(
|
||||
pub fn rotateMatrixByAxisAngle(
|
||||
matrix_addr: u32,
|
||||
angle_bits: u32,
|
||||
axis_addr: u32,
|
||||
@@ -15,7 +15,7 @@
|
||||
/// Compute outcodes for `count` vertices at `verts_ptr` (stride 12 bytes = 3 floats)
|
||||
/// against AABB at `bounds_ptr` (6 floats: minX, minY, minZ, maxX, maxY, maxZ).
|
||||
/// Writes results to `out_ptr` (1 byte per vertex).
|
||||
export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void {
|
||||
pub fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void {
|
||||
if (count == 0) return;
|
||||
const v_min: V4 = .{ readF32(bounds_ptr), readF32(bounds_ptr + 4), readF32(bounds_ptr + 8), 0 };
|
||||
const v_max: V4 = .{ readF32(bounds_ptr + 12), readF32(bounds_ptr + 16), readF32(bounds_ptr + 20), 0 };
|
||||
@@ -54,7 +54,7 @@ var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID
|
||||
|
||||
const origFindObjectByGUID = @as(*const fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32, @ptrFromInt(0x464890));
|
||||
|
||||
export fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
|
||||
pub fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
|
||||
const hash = (guid_lo ^ (guid_hi *% 0x9E3779B9)) & GUID_CACHE_MASK;
|
||||
const entry = &guid_cache[hash];
|
||||
|
||||
@@ -92,7 +92,7 @@ const g_grid_offset: *const f32 = @ptrFromInt(0x86861C);
|
||||
// Dot product coefficients at 0xC7CFB8..C7CFC4 (same as entpos view coeffs but different address)
|
||||
const g_spatial_coeffs: u32 = 0xC7CFB8;
|
||||
|
||||
export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void {
|
||||
pub fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void {
|
||||
// Dot product: coeff_a * obj[0x5C] + coeff_b * obj[0x60] + coeff_c * obj[0x64] + coeff_d
|
||||
const depth = readF32(g_spatial_coeffs) * readF32(obj + 0x5C) +
|
||||
readF32(g_spatial_coeffs + 4) * readF32(obj + 0x60) +
|
||||
@@ -144,7 +144,7 @@ export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void
|
||||
// RET 0x10
|
||||
// =============================================================================
|
||||
|
||||
export fn rayTriIntersectIndexedInt(
|
||||
pub fn rayTriIntersectIndexedInt(
|
||||
ray_ptr: u32,
|
||||
vert_pool: u32,
|
||||
indices_ptr: u32,
|
||||
@@ -279,7 +279,7 @@ fn computeAllOutcodes(
|
||||
/// Finds mesh data via hash, computes vertex outcodes against AABB from this+0x10,
|
||||
/// then iterates triangles: filters by visibility mask, trivial-rejects by outcode AND,
|
||||
/// adds survivors to global visible/render lists.
|
||||
export fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
pub fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
if (g_guard.* == 0) return 0;
|
||||
|
||||
const hash_table = g_guard.*;
|
||||
@@ -377,7 +377,7 @@ inline fn dot3(a: V4, b: V4) f32 {
|
||||
/// Fully inlined SSE rewrite. No external calls except FindOrCreateHashEntry.
|
||||
/// Moller-Trumbore ray-triangle intersection is inlined with SSE cross/dot,
|
||||
/// eliminating 4 SetVector3 calls and the ray_tri function call per triangle.
|
||||
export fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
pub fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
|
||||
if (g_guard.* == 0) return 0;
|
||||
|
||||
const hash_table = g_guard.*;
|
||||
@@ -52,7 +52,12 @@ const CacheSet = struct {
|
||||
lru: u8 = 0,
|
||||
};
|
||||
|
||||
var cache: [NUM_SETS]CacheSet = @splat(CacheSet{});
|
||||
const WINAPI = @import("std").builtin.CallingConvention.winapi;
|
||||
extern "kernel32" fn GetProcessHeap() callconv(WINAPI) ?*anyopaque;
|
||||
extern "kernel32" fn HeapAlloc(hHeap: ?*anyopaque, dwFlags: u32, dwBytes: usize) callconv(WINAPI) ?[*]u8;
|
||||
extern "kernel32" fn HeapFree(hHeap: ?*anyopaque, dwFlags: u32, lpMem: *anyopaque) callconv(WINAPI) i32;
|
||||
|
||||
var cache: ?[*]CacheSet = null;
|
||||
var cache_entries: u32 = 0;
|
||||
var cache_hits: u64 = 0;
|
||||
var cache_negative_hits: u64 = 0;
|
||||
@@ -66,8 +71,9 @@ var cache_miss_p2_archive: u64 = 0;
|
||||
|
||||
/// Lookup: hash picks set, check both ways for name match.
|
||||
pub fn archiveCacheLookup(h: u32, path: [*:0]const u8) ?ArchiveCacheEntry {
|
||||
const c = cache orelse return null;
|
||||
const set_idx = h & (NUM_SETS - 1);
|
||||
const set = &cache[set_idx];
|
||||
const set = &c[set_idx];
|
||||
const span = std.mem.span(path);
|
||||
|
||||
for (0..WAYS) |w| {
|
||||
@@ -93,11 +99,12 @@ pub fn computeBlockEntry(archive: u32, index: u32) u32 {
|
||||
}
|
||||
|
||||
pub fn archiveCacheInsert(h: u32, path: [*:0]const u8, outer: u32, inner: u32, block: u32, negative: bool) void {
|
||||
const c = cache orelse return;
|
||||
const span = std.mem.span(path);
|
||||
if (span.len > CACHE_NAME_LEN) return;
|
||||
|
||||
const set_idx = h & (NUM_SETS - 1);
|
||||
const set = &cache[set_idx];
|
||||
const set = &c[set_idx];
|
||||
|
||||
var target: u8 = set.lru;
|
||||
for (0..WAYS) |w| {
|
||||
@@ -135,8 +142,9 @@ pub fn recordMissP2() void { cache_miss_p2 +|= 1; }
|
||||
pub fn recordMissP2Archive() void { cache_miss_p2_archive +|= 1; }
|
||||
|
||||
pub fn getSlotOccupant(h: u32) ?[]const u8 {
|
||||
const c = cache orelse return null;
|
||||
const set_idx = h & (NUM_SETS - 1);
|
||||
const set = &cache[set_idx];
|
||||
const set = &c[set_idx];
|
||||
for (0..WAYS) |w| {
|
||||
if (set.entries[w].name_len != 0)
|
||||
return set.entries[w].name[0..set.entries[w].name_len];
|
||||
@@ -246,11 +254,19 @@ fn fileFindDetour(
|
||||
}
|
||||
|
||||
pub fn install() bool {
|
||||
const size = NUM_SETS * @sizeOf(CacheSet);
|
||||
const heap = GetProcessHeap() orelse return false;
|
||||
const ptr = HeapAlloc(heap, 0x00000008, size) orelse return false; // HEAP_ZERO_MEMORY
|
||||
cache = @alignCast(@ptrCast(ptr));
|
||||
return file_find_hook.attach(0x6549a0, &fileFindDetour) == .ok;
|
||||
}
|
||||
|
||||
pub fn remove() void {
|
||||
file_find_hook.detach();
|
||||
if (cache) |c| {
|
||||
if (GetProcessHeap()) |heap| _ = HeapFree(heap, 0, @ptrCast(c));
|
||||
cache = null;
|
||||
}
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
@@ -13,7 +13,6 @@
|
||||
|
||||
const hook_lib = @import("zhook");
|
||||
const logging = @import("../logging.zig");
|
||||
const std = @import("std");
|
||||
|
||||
// libdeflate C API
|
||||
extern fn libdeflate_alloc_decompressor() ?*anyopaque;
|
||||
@@ -24,15 +23,12 @@ var lib_available: bool = false;
|
||||
var log: logging.Logger = .{};
|
||||
|
||||
// --- Thread-local decompressor ---
|
||||
// Each thread lazily allocates its own decompressor on first use.
|
||||
// OS-managed TLS via Zig's threadlocal -- works correctly on both
|
||||
// native Windows and Wine without manual FS segment access.
|
||||
threadlocal var tls_decomp: ?*anyopaque = null;
|
||||
threadlocal var tls_decompressor: ?*anyopaque = null;
|
||||
|
||||
fn getTlsDecompressor() ?*anyopaque {
|
||||
if (tls_decomp) |d| return d;
|
||||
if (tls_decompressor) |d| return d;
|
||||
const d = libdeflate_alloc_decompressor() orelse return null;
|
||||
tls_decomp = d;
|
||||
tls_decompressor = d;
|
||||
return d;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
/* Wraps libdeflate_deflate_decompress with SEH to catch access violations.
|
||||
* libdeflate's fast path on 32-bit can crash on malformed input despite
|
||||
* SAFETY_CHECKs. This wrapper catches the crash and returns an error code. */
|
||||
|
||||
#include "libdeflate.h"
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
|
||||
__declspec(dllexport) int safe_deflate_decompress(
|
||||
struct libdeflate_decompressor *d,
|
||||
const void *in, size_t in_nbytes,
|
||||
void *out, size_t out_nbytes_avail,
|
||||
size_t *actual_out_nbytes_ret)
|
||||
{
|
||||
int result;
|
||||
__try {
|
||||
result = libdeflate_deflate_decompress(d, in, in_nbytes,
|
||||
out, out_nbytes_avail, actual_out_nbytes_ret);
|
||||
} __except(EXCEPTION_EXECUTE_HANDLER) {
|
||||
result = 1; /* LIBDEFLATE_BAD_DATA */
|
||||
}
|
||||
return result;
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,15 @@
|
||||
/* malloc/free stubs for libdeflate using the Windows process heap. */
|
||||
|
||||
#include <stddef.h>
|
||||
|
||||
__declspec(dllimport) void *__stdcall GetProcessHeap(void);
|
||||
__declspec(dllimport) void *__stdcall HeapAlloc(void *hHeap, unsigned long dwFlags, size_t dwBytes);
|
||||
__declspec(dllimport) int __stdcall HeapFree(void *hHeap, unsigned long dwFlags, void *lpMem);
|
||||
|
||||
void *malloc(size_t size) {
|
||||
return HeapAlloc(GetProcessHeap(), 0, size);
|
||||
}
|
||||
|
||||
void free(void *ptr) {
|
||||
if (ptr) HeapFree(GetProcessHeap(), 0, ptr);
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
#pragma once
|
||||
#include <stddef.h>
|
||||
extern void *malloc(size_t size);
|
||||
extern void free(void *ptr);
|
||||
@@ -0,0 +1,2 @@
|
||||
#pragma once
|
||||
/* libdeflate only uses stdio.h in debug paths — stub it out */
|
||||
@@ -0,0 +1,4 @@
|
||||
#pragma once
|
||||
#include <stddef.h>
|
||||
extern void *malloc(size_t size);
|
||||
extern void free(void *ptr);
|
||||
@@ -0,0 +1,6 @@
|
||||
#pragma once
|
||||
#include <stddef.h>
|
||||
void *memcpy(void *dest, const void *src, size_t n);
|
||||
void *memmove(void *dest, const void *src, size_t n);
|
||||
void *memset(void *s, int c, size_t n);
|
||||
int memcmp(const void *s1, const void *s2, size_t n);
|
||||
@@ -168,7 +168,6 @@ const VBState = struct {
|
||||
light: [3]u32,
|
||||
|
||||
fn load(vb: u32) VBState {
|
||||
logStrides(vb);
|
||||
return .{
|
||||
.pos = ru32(vb + VB.pos),
|
||||
.normal = ru32(vb + VB.normal),
|
||||
@@ -260,37 +259,12 @@ inline fn emitVertex(vb: u32, px: f32, py: f32, pz: f32, color: u32, tu: f32, tv
|
||||
// Reset each frame via resetParticleCache() called from the frame hook.
|
||||
var cached_render_state: u32 = 0;
|
||||
|
||||
var stride_logged: bool = false;
|
||||
var debug_logged: bool = false;
|
||||
export var debug_vertex_count: u32 = 0;
|
||||
export var debug_max_sprites: u32 = 0;
|
||||
export var debug_fmt_index: u32 = 0;
|
||||
export var debug_data_ptr: u32 = 0;
|
||||
|
||||
/// Reset per-frame caches. Call from OnWorldUpdate or executeSceneRenderPass hook.
|
||||
export fn resetParticleCache() void {
|
||||
pub fn resetParticleCache() void {
|
||||
cached_render_state = 0;
|
||||
}
|
||||
|
||||
/// Log VB strides once for analysis. Called from first VBState.load.
|
||||
fn logStrides(vb: u32) void {
|
||||
if (stride_logged) return;
|
||||
stride_logged = true;
|
||||
// Write to a known memory location that the profiler can dump, or just use
|
||||
// the debug console. For now, store in a global we can read.
|
||||
stride_info = .{
|
||||
ru32(vb + VB.pos_stride),
|
||||
ru32(vb + VB.normal_stride),
|
||||
ru32(vb + VB.color_stride),
|
||||
ru32(vb + VB.texcoord_stride),
|
||||
ru32(vb + VB.pos),
|
||||
ru32(vb + VB.normal),
|
||||
ru32(vb + VB.color),
|
||||
ru32(vb + VB.texcoord),
|
||||
};
|
||||
}
|
||||
|
||||
export var stride_info: [8]u32 = .{0} ** 8;
|
||||
|
||||
// =============================================================================
|
||||
// RenderParticleSprites (0x7B2A50)
|
||||
@@ -299,7 +273,7 @@ export var stride_info: [8]u32 = .{0} ** 8;
|
||||
//
|
||||
// Faithful recreation from assembly + Ghidra decompilation.
|
||||
// =============================================================================
|
||||
export fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 {
|
||||
pub fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 {
|
||||
const pd = particle_data; // particleData pointer (float*)
|
||||
const vb = vertex_buffers; // vertexBuffers pointer (float**)
|
||||
|
||||
@@ -762,7 +736,7 @@ const SG = struct {
|
||||
// Faithful recreation from Ghidra decompilation + assembly.
|
||||
// All game function calls preserved, matrix math inlined with V4.
|
||||
// =============================================================================
|
||||
export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void {
|
||||
pub fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void {
|
||||
// =========================================================================
|
||||
// Section 1: Identity matrices for render state
|
||||
// Optimization: use static identity instead of rebuilding on stack each call.
|
||||
@@ -999,15 +973,6 @@ export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC
|
||||
// =========================================================================
|
||||
gameRenderSorted(emitter, @intFromPtr(&vb_ptrs));
|
||||
|
||||
// DEBUG: log vertex count produced
|
||||
if (!debug_logged and vb_ptrs[8] > 0) {
|
||||
debug_logged = true;
|
||||
debug_vertex_count = vb_ptrs[8];
|
||||
debug_max_sprites = max_sprites;
|
||||
debug_fmt_index = fmt_index;
|
||||
debug_data_ptr = data_ptr;
|
||||
}
|
||||
|
||||
gameUnlockVB(vb_ptr, 0);
|
||||
gameDrawPrim(vb_ptr, fmt_index);
|
||||
|
||||
@@ -1095,7 +1060,7 @@ inline fn displayModeOffset(sprite_type: u32, count: u32) u32 {
|
||||
return divided -% ru32(DISPLAY_MODE_OFFSET_TABLE + sprite_type * 4);
|
||||
}
|
||||
|
||||
export fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void {
|
||||
pub fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void {
|
||||
// Early out: this+0xF2C == 0
|
||||
if (ru32(this + 0xF2C) == 0) return;
|
||||
|
||||
@@ -54,7 +54,7 @@ inline fn cvtss2si(x: f32) i32 {
|
||||
// --- 0x4549C0: normalizeVec3 (137K/7.5s) ---
|
||||
// Naked thiscall: ECX=vec, [ESP+4]=length_bits. RET 4. Original: 38 bytes.
|
||||
// rcpss + NR for fast reciprocal, then 3 multiplies.
|
||||
export fn si_normalizeVec3() callconv(.naked) void {
|
||||
pub fn si_normalizeVec3() callconv(.naked) void {
|
||||
asm volatile (
|
||||
// xmm0 = 1.0 / length (via rcpss + Newton-Raphson)
|
||||
\\vmovss 4(%%esp), %%xmm0
|
||||
@@ -77,7 +77,7 @@ export fn si_normalizeVec3() callconv(.naked) void {
|
||||
// out = A * B (3x4 layout: 3x3 rotation + 3 translation)
|
||||
// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz]
|
||||
// V4 per row: broadcast b[row*3+k], multiply with a's columns, accumulate.
|
||||
export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 {
|
||||
pub fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 {
|
||||
const dst: [*]f32 = @ptrFromInt(out);
|
||||
const aa: [*]const f32 = @ptrFromInt(a_ptr);
|
||||
const b: [*]const f32 = @ptrFromInt(b_ptr);
|
||||
@@ -117,7 +117,7 @@ export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 {
|
||||
// --- 0x7BDDB0: rotateMatByQuat ---
|
||||
// builds rotation matrix from quaternion, multiplies with existing 4x4 matrix
|
||||
// Uses V4 for the matrix multiply (same pattern as bone_sse)
|
||||
export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 {
|
||||
pub fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 {
|
||||
const q: [*]const f32 = @ptrFromInt(quat);
|
||||
const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3];
|
||||
const x2 = x + x; const y2 = y + y; const z2 = z + z;
|
||||
@@ -146,7 +146,7 @@ export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 {
|
||||
|
||||
// --- 0x7BB860: createRotMat3x4 ---
|
||||
// Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries.
|
||||
export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 {
|
||||
pub fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 {
|
||||
const m: [*]f32 = @ptrFromInt(out);
|
||||
const ax: [*]const f32 = @ptrFromInt(axis_ptr);
|
||||
var x = ax[0]; var y = ax[1]; var z = ax[2];
|
||||
@@ -168,7 +168,7 @@ export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normal
|
||||
|
||||
// --- 0x6329E0: distanceToPlane (525K/7.5s) ---
|
||||
// __fastcall(ECX=point, EDX=plane, stack=direction), returns f64 via ST(0), RET 0x4.
|
||||
export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 {
|
||||
pub fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 {
|
||||
const p: [*]const f32 = @ptrFromInt(point);
|
||||
const pl: [*]const f32 = @ptrFromInt(plane);
|
||||
const dir: [*]const f32 = @ptrFromInt(direction);
|
||||
@@ -183,7 +183,7 @@ export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC
|
||||
// Tests point against 6 frustum planes, produces 6-bit bitmask.
|
||||
// Scalar @mulAdd dot4 per plane — the FMA chain has best throughput for this pattern.
|
||||
// Tried: V4 batch 4 planes (gather kills it), V4 hsum (shuffle overhead kills it).
|
||||
export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 {
|
||||
pub fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 {
|
||||
const mask: *u32 = @ptrFromInt(out_mask);
|
||||
const pt = loadV3_1(point);
|
||||
var bits: u32 = 0;
|
||||
@@ -199,7 +199,7 @@ export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) ca
|
||||
|
||||
// --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) ---
|
||||
// Slab AABB test. Branchless min/max for t0/t1 swap and tmin/tmax accumulation.
|
||||
export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 {
|
||||
pub fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 {
|
||||
const bmin: [*]const f32 = @ptrFromInt(box_ptr);
|
||||
const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC);
|
||||
const start: [*]const f32 = @ptrFromInt(line_start);
|
||||
@@ -225,7 +225,7 @@ export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32)
|
||||
|
||||
// --- 0x6869C0: testOBBFrustum ---
|
||||
// Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test.
|
||||
export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 {
|
||||
pub fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 {
|
||||
const aabb: [*]const f32 = @ptrFromInt(aabb_ptr);
|
||||
const rot: [*]const f32 = @ptrFromInt(rot_ptr);
|
||||
const t: [*]const f32 = @ptrFromInt(trans_ptr);
|
||||
@@ -285,7 +285,7 @@ export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_
|
||||
|
||||
// --- 0x686B80: testSphereFrustum (375K/7.5s) ---
|
||||
// Zig thiscall: naked asm tested at 10cy (vhaddps slow), Zig dot4v at 8cy.
|
||||
export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 {
|
||||
pub fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 {
|
||||
const s: [*]const f32 = @ptrFromInt(sphere);
|
||||
const center = V4{ s[0], s[1], s[2], 1.0 };
|
||||
const r = s[3];
|
||||
@@ -299,7 +299,7 @@ export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 {
|
||||
|
||||
// --- 0x7C0570: quatSlerp ---
|
||||
// V4 for final blend, @mulAdd for dot product
|
||||
export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 {
|
||||
pub fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 {
|
||||
const dst: [*]f32 = @ptrFromInt(out);
|
||||
const av = loadV4(a_ptr);
|
||||
const bv = loadV4(b_ptr);
|
||||
@@ -326,7 +326,7 @@ export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(F
|
||||
|
||||
// --- 0x699330: isPointInsideBounds (1.7M/7.5s) ---
|
||||
// __fastcall(ECX=a, EDX=b), returns u32.
|
||||
export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 {
|
||||
pub fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 {
|
||||
const va: [*]const f32 = @ptrFromInt(a);
|
||||
const vb: [*]const f32 = @ptrFromInt(b);
|
||||
if (vb[0] <= va[0] and vb[1] <= va[1] and vb[2] <= va[2]) return 1;
|
||||
@@ -334,7 +334,7 @@ export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 {
|
||||
}
|
||||
|
||||
// --- 0x749280: calculateSinCos ---
|
||||
export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void {
|
||||
pub fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void {
|
||||
const angle: f32 = @bitCast(angle_bits);
|
||||
const sp: *f32 = @ptrFromInt(out_sin);
|
||||
const cp: *f32 = @ptrFromInt(out_cos);
|
||||
@@ -343,7 +343,7 @@ export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callco
|
||||
}
|
||||
|
||||
// --- 0x7BE5B0: createZRotMat3x3 ---
|
||||
export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 {
|
||||
pub fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 {
|
||||
const m: [*]f32 = @ptrFromInt(out);
|
||||
const angle: f32 = @bitCast(angle_bits);
|
||||
const c = @cos(angle); const s = @sin(angle);
|
||||
@@ -356,7 +356,7 @@ export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 {
|
||||
// --- 0x7BCEF0: transposeMat4x4 ---
|
||||
// Naked thiscall: ECX=src, [ESP+4]=dst. RET 4. Original: 156 bytes.
|
||||
// SSE unpacklo/unpackhi transpose: 4 loads + 4 shuffles + 4 stores.
|
||||
export fn si_transposeMat4x4() callconv(.naked) void {
|
||||
pub fn si_transposeMat4x4() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\mov 4(%%esp), %%eax
|
||||
// Load 4 rows from src (ECX)
|
||||
@@ -387,7 +387,7 @@ export fn si_transposeMat4x4() callconv(.naked) void {
|
||||
|
||||
// --- 0x7BB420: mulMat3x4InPlace ---
|
||||
// this = this * matB. V4 columns loaded upfront, write directly back (no tmp needed).
|
||||
export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 {
|
||||
pub fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 {
|
||||
const a: [*]f32 = @ptrFromInt(mat_a);
|
||||
const b: [*]const f32 = @ptrFromInt(mat_b);
|
||||
|
||||
@@ -419,7 +419,7 @@ export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 {
|
||||
// --- 0x6720F0: normalizeVec3InPlace ---
|
||||
// sqrt + reciprocal. 14cy (2.2x). rsqrt+NR tested at 15cy — no gain, compiler's
|
||||
// vsqrtss+vdivss pipeline is already optimal for scalar inverse sqrt.
|
||||
export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void {
|
||||
pub fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void {
|
||||
const v: [*]f32 = @ptrFromInt(vec);
|
||||
const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]);
|
||||
if (len > 1.0e-20) {
|
||||
@@ -432,7 +432,7 @@ export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void {
|
||||
|
||||
// --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) ---
|
||||
// thiscall(ECX=this, stack=vec). Scale is a global at 0x81207C, NOT a parameter.
|
||||
export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void {
|
||||
pub fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void {
|
||||
const obj: [*]f32 = @ptrFromInt(this);
|
||||
const v: [*]const f32 = @ptrFromInt(vec);
|
||||
const scale: f32 = @as(*const f32, @ptrFromInt(0x81207C)).*;
|
||||
@@ -447,7 +447,7 @@ export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void {
|
||||
// --- 0x71BF60: addToColorAccumulator (10K/7.5s) ---
|
||||
// Naked thiscall: ECX=this, [ESP+4]=color_ptr. RET 4. Original: 34 bytes.
|
||||
// 3 SSE adds at this+0x6C from color[0..2].
|
||||
export fn si_addToColorAccumulator() callconv(.naked) void {
|
||||
pub fn si_addToColorAccumulator() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\mov 4(%%esp), %%eax
|
||||
\\vmovss (%%eax), %%xmm0
|
||||
@@ -465,7 +465,7 @@ export fn si_addToColorAccumulator() callconv(.naked) void {
|
||||
|
||||
// --- 0x7B7A80: packParticleColor (2K/7.5s) ---
|
||||
// V4 multiply + clamp, then packed round+convert via @Vector(4, i32) for all channels at once.
|
||||
export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void {
|
||||
pub fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void {
|
||||
const base: [*]u8 = @ptrFromInt(obj);
|
||||
const out: *align(1) u32 = @ptrCast(base + 0x12C);
|
||||
const alpha = base[0x12F];
|
||||
@@ -480,7 +480,7 @@ export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32)
|
||||
// --- 0x7B7B10: setParticleAlpha (2K/7.5s) ---
|
||||
// Naked fastcall: ECX=obj, [ESP+4]=alpha_bits. RET 4.
|
||||
// Clamp alpha*255 to [0,255], write byte to obj+0x12F.
|
||||
export fn si_setParticleAlpha() callconv(.naked) void {
|
||||
pub fn si_setParticleAlpha() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\vmovss 4(%%esp), %%xmm0
|
||||
\\mov $0x437F0000, %%eax
|
||||
@@ -499,7 +499,7 @@ export fn si_setParticleAlpha() callconv(.naked) void {
|
||||
// --- 0x40A2B0: __ftol ---
|
||||
// Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64).
|
||||
// SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original)
|
||||
export fn si_ftol() callconv(.naked) void {
|
||||
pub fn si_ftol() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\sub $8, %%esp
|
||||
\\fisttpll (%%esp)
|
||||
@@ -511,7 +511,7 @@ export fn si_ftol() callconv(.naked) void {
|
||||
|
||||
// --- 0x602630: vec3Dot (31K/7.5s) ---
|
||||
// __fastcall(ECX=a, EDX=b), returns f64 via ST(0).
|
||||
export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 {
|
||||
pub fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 {
|
||||
const va: [*]const f32 = @ptrFromInt(a);
|
||||
const vb: [*]const f32 = @ptrFromInt(b);
|
||||
return @floatCast(@mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0])));
|
||||
@@ -519,7 +519,7 @@ export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 {
|
||||
|
||||
// --- 0x686820: translateBoundingVol ---
|
||||
// @mulAdd for plane distances. Scalar corner adds (stride 3 — V4 unaligned tested, slower).
|
||||
export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
|
||||
pub fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
|
||||
const obj: [*]f32 = @ptrFromInt(this);
|
||||
const off: [*]const f32 = @ptrFromInt(offset);
|
||||
const dx = off[0]; const dy = off[1]; const dz = off[2];
|
||||
@@ -543,7 +543,7 @@ export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
|
||||
// Original: 380 bytes, 2 calls to mat*vec3 (0x7BCA80), x87 perspective divide, x87 column scan.
|
||||
// SSE: inline V4 mat*vec3, SSE perspective divide, 4-wide column scan.
|
||||
// __fastcall(bbox_ECX, flags_EDX, radius_stack), RET 0x4
|
||||
export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 {
|
||||
pub fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 {
|
||||
// Early out: global occlusion flag bit 5
|
||||
if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 0x20) == 0) return 0;
|
||||
|
||||
@@ -638,7 +638,7 @@ export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(F
|
||||
// __fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack), RET 0x8
|
||||
// addGeometryToBuffer at 0x6ABD90: __fastcall(queryBox_ECX, nodeData_EDX, resultBuf_stack), RET 0x4
|
||||
// Visited sentinel: *(u32*)0xC89F20
|
||||
export fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 {
|
||||
pub fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 {
|
||||
if ((flags & 0xF0000F) == 0) return 1;
|
||||
|
||||
const addGeometryToBuffer: *const fn (u32, u32, u32) callconv(FC) void = @ptrFromInt(0x6ABD90);
|
||||
@@ -22,21 +22,8 @@ const filecache = @import("filecache.zig");
|
||||
|
||||
pub const module_name: [*:0]const u8 = "weirdperformance";
|
||||
|
||||
// Provide malloc/free for libdeflate's default allocator (linked without libc).
|
||||
// Use game's Storm memory manager:
|
||||
// ReallocMemory (0x646320): __stdcall(ptr, size, filename, line, flags) → ptr
|
||||
// When ptr=NULL, acts as malloc via AllocateBufferWithPowerOfTwo.
|
||||
// FreeMemory (0x646430): __stdcall(ptr, filename, line, flags) RET 0x10 = 4 params
|
||||
const gameRealloc: *const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) ?*anyopaque = @ptrFromInt(0x646320);
|
||||
const gameFree: *const fn (u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) u32 = @ptrFromInt(0x646430);
|
||||
|
||||
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
|
||||
return gameRealloc(0, @intCast(size), 0, 0, 0);
|
||||
}
|
||||
|
||||
export fn free(ptr: ?*anyopaque) callconv(.c) void {
|
||||
if (ptr) |p| _ = gameFree(@intFromPtr(p), 0, 0, 0);
|
||||
}
|
||||
// malloc/free for libdeflate provided by stubs/game_alloc.c (compiled into
|
||||
// the libdeflate static lib). This avoids exporting malloc/free from the DLL.
|
||||
|
||||
var g_mutex: ?*anyopaque = null;
|
||||
var g_is_hook_owner: bool = false;
|
||||
@@ -47,12 +34,18 @@ pub fn isActive() bool {
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// Extern SSE functions (from separate ReleaseFast compilation units)
|
||||
// SSE functions (imported directly to avoid addObject SizeOfImage bloat)
|
||||
// =============================================================================
|
||||
|
||||
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
|
||||
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn resetParticleCache() void;
|
||||
const bone_sse = @import("bone_sse.zig");
|
||||
const particle_sse = @import("particle_sse.zig");
|
||||
const clip_sse = @import("clip_sse.zig");
|
||||
const cull_sse = @import("cull_sse.zig");
|
||||
const silicon_sse = @import("silicon_sse.zig");
|
||||
|
||||
const transformImpl_SSE = bone_sse.transformImpl_SSE;
|
||||
const renderParticleSprites_SSE = particle_sse.renderParticleSprites_SSE;
|
||||
const resetParticleCache = particle_sse.resetParticleCache;
|
||||
|
||||
// =============================================================================
|
||||
// transformMatrix4x4 hook (0x714260)
|
||||
@@ -218,35 +211,12 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
|
||||
// =============================================================================
|
||||
|
||||
// cull_sse.zig
|
||||
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
|
||||
const performSpatialCulling = cull_sse.performSpatialCulling;
|
||||
const performCollisionDetectionSSE = cull_sse.performCollisionDetectionSSE;
|
||||
const rayTriIntersectIndexedInt = cull_sse.rayTriIntersectIndexedInt;
|
||||
|
||||
// silicon_sse.zig
|
||||
const sse = struct {
|
||||
extern fn si_normalizeVec3() callconv(.naked) void;
|
||||
extern fn si_mulMat3x4(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
extern fn si_rotateMatByQuat(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn si_createRotMat3x4(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
extern fn si_classifyPointFrustum(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn si_checkBoxLineIntersect(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
extern fn si_testOBBFrustum(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn si_testSphereFrustum(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn si_quatSlerp(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
extern fn si_calculateSinCos(u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void;
|
||||
extern fn si_createZRotMat3x3(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn si_transposeMat4x4() callconv(.naked) void;
|
||||
extern fn si_mulMat3x4InPlace(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
|
||||
extern fn si_normalizeVec3InPlace(u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn si_addVec3ToAccumulator(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn si_addToColorAccumulator() callconv(.naked) void;
|
||||
extern fn si_packParticleColor(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn si_setParticleAlpha() callconv(.naked) void;
|
||||
extern fn si_ftol() callconv(.naked) void;
|
||||
extern fn si_translateBoundingVol(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
|
||||
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
extern fn si_frustumCullBBox(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
};
|
||||
const sse = silicon_sse;
|
||||
|
||||
const PatchEntry = struct {
|
||||
target: u32,
|
||||
@@ -348,7 +318,10 @@ pub fn installHooks() void {
|
||||
// libdeflate inflate replacement
|
||||
if (inflate_hook.install()) installed += 1;
|
||||
|
||||
// TSC timer calibration + OS timer tweaks
|
||||
}
|
||||
|
||||
pub fn lateInit() void {
|
||||
if (!g_is_hook_owner) return;
|
||||
timer_fix.init();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,206 @@
|
||||
/* zconf.h -- configuration of the zlib compression library
|
||||
* Copyright (C) 1995-2024 Jean-loup Gailly, Mark Adler
|
||||
* For conditions of distribution and use, see copyright notice in zlib.h
|
||||
*/
|
||||
|
||||
#ifndef ZCONF_H
|
||||
#define ZCONF_H
|
||||
|
||||
#include "zlib_name_mangling.h"
|
||||
|
||||
#if !defined(_WIN32) && defined(__WIN32__)
|
||||
# define _WIN32
|
||||
#endif
|
||||
|
||||
/* Clang macro for detecting declspec support
|
||||
* https://clang.llvm.org/docs/LanguageExtensions.html#has-declspec-attribute
|
||||
*/
|
||||
#ifndef __has_declspec_attribute
|
||||
# define __has_declspec_attribute(x) 0
|
||||
#endif
|
||||
|
||||
#if defined(ZLIB_CONST) && !defined(z_const)
|
||||
# define z_const const
|
||||
#else
|
||||
# define z_const
|
||||
#endif
|
||||
|
||||
/* Maximum value for memLevel in deflateInit2 */
|
||||
#ifndef MAX_MEM_LEVEL
|
||||
# define MAX_MEM_LEVEL 9
|
||||
#endif
|
||||
|
||||
/* Maximum value for windowBits in deflateInit2 and inflateInit2.
|
||||
* WARNING: reducing MAX_WBITS makes minigzip unable to extract .gz files
|
||||
* created by gzip. (Files created by minigzip can still be extracted by
|
||||
* gzip.)
|
||||
*/
|
||||
#ifndef MIN_WBITS
|
||||
# define MIN_WBITS 8 /* 256 LZ77 window */
|
||||
#endif
|
||||
#ifndef MAX_WBITS
|
||||
# define MAX_WBITS 15 /* 32K LZ77 window */
|
||||
#endif
|
||||
|
||||
/* The memory requirements for deflate are (in bytes):
|
||||
(1 << (windowBits+2)) + (1 << (memLevel+9))
|
||||
that is: 128K for windowBits=15 + 128K for memLevel = 8 (default values)
|
||||
plus a few kilobytes for small objects. For example, if you want to reduce
|
||||
the default memory requirements from 256K to 128K, compile with
|
||||
make CFLAGS="-O -DMAX_WBITS=14 -DMAX_MEM_LEVEL=7"
|
||||
Of course this will generally degrade compression (there's no free lunch).
|
||||
|
||||
The memory requirements for inflate are (in bytes) 1 << windowBits
|
||||
that is, 32K for windowBits=15 (default value) plus about 7 kilobytes
|
||||
for small objects.
|
||||
*/
|
||||
|
||||
/* Type declarations */
|
||||
|
||||
|
||||
#ifndef OF /* function prototypes */
|
||||
# define OF(args) args
|
||||
#endif
|
||||
|
||||
#ifdef ZLIB_INTERNAL
|
||||
# define Z_INTERNAL ZLIB_INTERNAL
|
||||
#endif
|
||||
|
||||
/* If building or using zlib as a DLL, define ZLIB_DLL.
|
||||
* This is not mandatory, but it offers a little performance increase.
|
||||
*/
|
||||
#if defined(ZLIB_DLL) && (defined(_WIN32) || (__has_declspec_attribute(dllexport) && __has_declspec_attribute(dllimport)))
|
||||
# ifdef Z_INTERNAL
|
||||
# define Z_EXTERN extern __declspec(dllexport)
|
||||
# else
|
||||
# define Z_EXTERN extern __declspec(dllimport)
|
||||
# endif
|
||||
#endif
|
||||
|
||||
/* If building or using zlib with the WINAPI/WINAPIV calling convention,
|
||||
* define ZLIB_WINAPI.
|
||||
* Caution: the standard ZLIB1.DLL is NOT compiled using ZLIB_WINAPI.
|
||||
*/
|
||||
#if defined(ZLIB_WINAPI) && defined(_WIN32)
|
||||
# ifndef WIN32_LEAN_AND_MEAN
|
||||
# define WIN32_LEAN_AND_MEAN
|
||||
# endif
|
||||
# include <windows.h>
|
||||
/* No need for _export, use ZLIB.DEF instead. */
|
||||
/* For complete Windows compatibility, use WINAPI, not __stdcall. */
|
||||
# define Z_EXPORT WINAPI
|
||||
# define Z_EXPORTVA WINAPIV
|
||||
#endif
|
||||
|
||||
#ifndef Z_EXTERN
|
||||
# define Z_EXTERN extern
|
||||
#endif
|
||||
#ifndef Z_EXPORT
|
||||
# define Z_EXPORT
|
||||
#endif
|
||||
#ifndef Z_EXPORTVA
|
||||
# define Z_EXPORTVA
|
||||
#endif
|
||||
|
||||
/* Conditional exports */
|
||||
#define ZNG_CONDEXPORT Z_INTERNAL
|
||||
|
||||
/* For backwards compatibility */
|
||||
|
||||
#ifndef ZEXTERN
|
||||
# define ZEXTERN Z_EXTERN
|
||||
#endif
|
||||
#ifndef ZEXPORT
|
||||
# define ZEXPORT Z_EXPORT
|
||||
#endif
|
||||
#ifndef ZEXPORTVA
|
||||
# define ZEXPORTVA Z_EXPORTVA
|
||||
#endif
|
||||
#ifndef FAR
|
||||
# define FAR
|
||||
#endif
|
||||
|
||||
/* Legacy zlib typedefs for backwards compatibility. Don't assume stdint.h is defined. */
|
||||
typedef unsigned char Byte;
|
||||
typedef Byte Bytef;
|
||||
|
||||
typedef unsigned int uInt; /* 16 bits or more */
|
||||
typedef unsigned long uLong; /* 32 bits or more */
|
||||
|
||||
typedef char charf;
|
||||
typedef int intf;
|
||||
typedef uInt uIntf;
|
||||
typedef uLong uLongf;
|
||||
|
||||
typedef void const *voidpc;
|
||||
typedef void *voidpf;
|
||||
typedef void *voidp;
|
||||
|
||||
typedef unsigned int z_crc_t;
|
||||
|
||||
#if 1 /* was set to #if 1 by configure/cmake/etc */
|
||||
# define Z_HAVE_UNISTD_H
|
||||
#endif
|
||||
|
||||
#ifdef NEED_PTRDIFF_T /* may be set to #if 1 by configure/cmake/etc */
|
||||
typedef PTRDIFF_TYPE ptrdiff_t;
|
||||
#endif
|
||||
|
||||
#include <sys/types.h> /* for off_t */
|
||||
|
||||
#include <stddef.h> /* for wchar_t and NULL */
|
||||
|
||||
/* a little trick to accommodate both "#define _LARGEFILE64_SOURCE" and
|
||||
* "#define _LARGEFILE64_SOURCE 1" as requesting 64-bit operations, (even
|
||||
* though the former does not conform to the LFS document), but considering
|
||||
* both "#undef _LARGEFILE64_SOURCE" and "#define _LARGEFILE64_SOURCE 0" as
|
||||
* equivalently requesting no 64-bit operations
|
||||
*/
|
||||
#if defined(_LARGEFILE64_SOURCE) && -_LARGEFILE64_SOURCE - -1 == 1
|
||||
# undef _LARGEFILE64_SOURCE
|
||||
#endif
|
||||
|
||||
#if defined(Z_HAVE_UNISTD_H)
|
||||
# include <unistd.h> /* for SEEK_*, off_t, and _LFS64_LARGEFILE */
|
||||
# ifndef z_off_t
|
||||
# define z_off_t off_t
|
||||
# endif
|
||||
#endif
|
||||
|
||||
#if defined(_LFS64_LARGEFILE) && _LFS64_LARGEFILE-0
|
||||
# define Z_LFS64
|
||||
#endif
|
||||
|
||||
#if defined(_LARGEFILE64_SOURCE) && defined(Z_LFS64)
|
||||
# define Z_LARGE64
|
||||
#endif
|
||||
|
||||
#if defined(_FILE_OFFSET_BITS) && _FILE_OFFSET_BITS-0 == 64 && defined(Z_LFS64)
|
||||
# define Z_WANT64
|
||||
#endif
|
||||
|
||||
#if !defined(SEEK_SET)
|
||||
# define SEEK_SET 0 /* Seek from beginning of file. */
|
||||
# define SEEK_CUR 1 /* Seek from current position. */
|
||||
# define SEEK_END 2 /* Set file pointer to EOF plus "offset" */
|
||||
#endif
|
||||
|
||||
#ifndef z_off_t
|
||||
# define z_off_t long
|
||||
#endif
|
||||
|
||||
#if !defined(_WIN32) && defined(Z_LARGE64)
|
||||
# define z_off64_t off64_t
|
||||
#else
|
||||
# if defined(__MSYS__)
|
||||
# define z_off64_t _off64_t
|
||||
# elif defined(_WIN32) && !defined(__GNUC__)
|
||||
# define z_off64_t __int64
|
||||
# else
|
||||
# define z_off64_t z_off_t
|
||||
# endif
|
||||
#endif
|
||||
|
||||
typedef size_t z_size_t;
|
||||
|
||||
#endif /* ZCONF_H */
|
||||
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user