fix: MSVC ABI, heap allocator, deferred timer - vanillafixes compat

- Restore MSVC ABI (was accidentally GNU since v0.6.0, broke .CRT section)
- Replace game allocator with Windows process heap for filecache and
  libdeflate malloc/free - game allocator not initialized during DllMain
  when injected via CreateRemoteThread
- Defer timer calibration (Sleep 500ms) to lateInit - blocks under
  loader lock during DllMain
- Remove exported malloc/free symbols from DLL
- Eliminate addObject compilation units for SSE files - direct @import
  with AVX target instead
- Heap-allocate filecache (was 9.3MB static BSS)
- Strip transform44 of performance/ externs, pure profiling only
- Rename performance/ to weirdperformance/ to match module convention
- Skip default-off modules in all-variants build step
- Remove dead debug vars and stride logging from particle_sse
This commit is contained in:
MarcelineVQ
2026-03-28 05:31:31 -07:00
parent dd745e9728
commit db675c7a7e
43 changed files with 2245 additions and 405 deletions
+12 -30
View File
@@ -7,36 +7,12 @@
This project is developed entirely locally. The remote repo is **only** a
distribution point for releases - no source code is pushed.
The remote `main` branch contains a single file: `README.md` (built from the
local `DLL_README.md`). This must be set up once when creating the repo:
The remote `main` branch contains `README.md` (built from the local
`DLL_README.md`), `weirdutils_api.h`, and issue templates under `.gitea/`.
```sh
tea repo create --name WeirdUtils --description "Vanilla WoW 1.12.1 utility DLLs" --login MarcelineVQ
```
Codeberg disables releases on new repos by default. Enable via API
(get your token from `grep 'token:' ~/.config/tea/config.yml | head -1 | awk '{print $2}'`):
```sh
curl -s -X PATCH \
-H "Authorization: token <your-token>" \
-H "Content-Type: application/json" \
-d '{"has_releases":true}' \
"https://codeberg.org/api/v1/repos/MarcelineVQ/WeirdUtils"
```
Then push the initial README:
```sh
# In a temporary directory:
git init && git remote add origin ssh://git@codeberg.org/MarcelineVQ/WeirdUtils.git && git checkout -b main
cp /path/to/weirdutils/DLL_README.md README.md
git add README.md
git commit -m "Add README"
git push origin main
```
After that, the remote `main` only needs updating when `DLL_README.md` changes.
A local clone of the remote repo lives at `remote/WeirdUtils/`. The wiki
lives at `remote/wiki/`. Use these for all remote operations - no tmp clones
needed.
## 1. Bump module versions
@@ -120,11 +96,12 @@ The module name list in the Developer Notes section must also only list
released module names.
```sh
# from a clone or worktree of the remote repo
cd remote/WeirdUtils
# edit README.md: remove sections for modules not in this release
git add README.md
git commit -m "Update README for vX.Y.Z"
git push origin main
cd ../..
```
## 4. Write the release notes
@@ -222,6 +199,11 @@ print(r[0]['id']) if r else print('not found')
"
```
## Known Issues
- **vanillafixes launcher**: Incompatible with WeirdUtils DLL injection. Users
should load the DLL via WoW.exe + `dlls.txt` or another loader instead.
## Checklist
- [ ] Module versions bumped in `build.zig` for changed modules
+20 -100
View File
@@ -41,7 +41,7 @@ pub fn build(b: *std.Build) void {
.cpu_arch = .x86,
.os_tag = .windows,
.abi = .msvc,
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2 }),
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
});
const optimize = b.option(std.builtin.OptimizeMode, "optimize", "Optimization mode (default: ReleaseFast)") orelse .ReleaseFast;
@@ -61,46 +61,6 @@ pub fn build(b: *std.Build) void {
});
const zhook_mod = zhook_dep.module("zhook");
// Hot math — separate compilation units, always ReleaseFast.
// Source lives in src/performance/ — the production SSE module.
const clip_sse_obj = b.addObject(.{
.name = "clip_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/clip_sse.zig"),
.target = target,
.optimize = .ReleaseFast,
}),
});
const cull_sse_obj = b.addObject(.{
.name = "cull_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/cull_sse.zig"),
.target = target,
.optimize = .ReleaseFast,
}),
});
const entity_sse_obj = b.addObject(.{
.name = "entity_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/entity_sse.zig"),
.target = target,
.optimize = .ReleaseFast,
}),
});
const bone_sse_target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .windows,
.abi = .msvc,
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }),
});
const bone_sse_obj = b.addObject(.{
.name = "bone_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/bone_sse.zig"),
.target = bone_sse_target,
.optimize = .ReleaseFast,
}),
});
// REF uses x87-only target to match original game code structure.
// The global target has SSE/SSE2 which generates movss/mulss;
// the original at 0x714260 uses pure x87 (FLD/FMUL/FSTP).
@@ -126,28 +86,11 @@ pub fn build(b: *std.Build) void {
.optimize = .ReleaseFast,
}),
});
const silicon_sse_obj = b.addObject(.{
.name = "silicon_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/silicon_sse.zig"),
.target = bone_sse_target, // SSE4.1+FMA+AVX, same as bone_sse
.optimize = .ReleaseFast,
}),
});
const particle_sse_obj = b.addObject(.{
.name = "particle_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/particle_sse.zig"),
.target = bone_sse_target,
.optimize = .ReleaseFast,
}),
});
const particle_ref_obj = b.addObject(.{
.name = "particle_sse_ref",
.root_module = b.createModule(.{
.root_source_file = b.path("src/transform44/particle_sse_reference.zig"),
.target = bone_sse_target,
.target = target,
.optimize = .ReleaseFast,
}),
});
@@ -178,76 +121,52 @@ pub fn build(b: *std.Build) void {
});
libdeflate.root_module.addCSourceFiles(.{
.files = &.{
"src/performance/libdeflate/lib/deflate_decompress.c",
"src/performance/libdeflate/lib/zlib_decompress.c",
"src/performance/libdeflate/lib/utils.c",
"src/performance/libdeflate/lib/adler32.c",
"src/performance/libdeflate/lib/x86/cpu_features.c",
"src/weirdperformance/libdeflate/lib/deflate_decompress.c",
"src/weirdperformance/libdeflate/lib/zlib_decompress.c",
"src/weirdperformance/libdeflate/lib/utils.c",
"src/weirdperformance/libdeflate/lib/adler32.c",
"src/weirdperformance/libdeflate/lib/x86/cpu_features.c",
"src/weirdperformance/libdeflate/stubs/game_alloc.c",
},
.flags = &.{"-DLIBDEFLATE_ASSEMBLER_DOES_NOT_SUPPORT_AVX512VNNI"},
});
libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate/stubs"));
libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate"));
libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate/lib"));
libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate/stubs"));
libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate"));
libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate/lib"));
// Link module-specific object files into a DLL.
// Single source of truth for which objects each module needs.
// Called for both the main weirdutils build and each variant.
const ModuleObjects = struct {
clip_sse: *std.Build.Step.Compile,
cull_sse: *std.Build.Step.Compile,
entity_sse: *std.Build.Step.Compile,
bone_sse: *std.Build.Step.Compile,
bone_sse_ref: *std.Build.Step.Compile,
math_sse: *std.Build.Step.Compile,
silicon_sse: *std.Build.Step.Compile,
particle_sse: *std.Build.Step.Compile,
particle_ref: *std.Build.Step.Compile,
libdeflate: *std.Build.Step.Compile,
fn linkFor(self: @This(), mod: *std.Build.Module, comptime module_name: []const u8) void {
@setEvalBranchQuota(10000);
if (comptime std.mem.eql(u8, module_name, "weirdperformance")) {
mod.addObject(self.clip_sse);
mod.addObject(self.cull_sse);
mod.addObject(self.bone_sse);
mod.addObject(self.silicon_sse);
mod.addObject(self.particle_sse);
mod.addObjectFile(self.libdeflate.getEmittedBin());
}
if (comptime std.mem.eql(u8, module_name, "transform44")) {
mod.addObject(self.clip_sse);
mod.addObject(self.cull_sse);
mod.addObject(self.entity_sse);
mod.addObject(self.bone_sse);
mod.addObject(self.bone_sse_ref);
mod.addObject(self.particle_sse);
mod.addObject(self.particle_ref);
}
if (comptime std.mem.eql(u8, module_name, "silicon")) {
mod.addObject(self.silicon_sse);
}
if (comptime std.mem.eql(u8, module_name, "ssemaths")) {
mod.addObject(self.math_sse);
}
}
};
const objs = ModuleObjects{
.clip_sse = clip_sse_obj,
.cull_sse = cull_sse_obj,
.entity_sse = entity_sse_obj,
.bone_sse = bone_sse_obj,
.bone_sse_ref = bone_sse_ref_obj,
.math_sse = math_sse_obj,
.silicon_sse = silicon_sse_obj,
.particle_sse = particle_sse_obj,
.particle_ref = particle_ref_obj,
.libdeflate = libdeflate,
};
// Main DLL: link all module objects
inline for (module_list) |mod| {
objs.linkFor(lib.root_module, mod.name);
// Main DLL: link module objects only for enabled modules
inline for (module_list, 0..) |mod, i| {
if (module_enabled[i]) objs.linkFor(lib.root_module, mod.name);
}
b.installArtifact(lib);
@@ -282,7 +201,7 @@ pub fn build(b: *std.Build) void {
const bench_silicon_sse = b.addObject(.{
.name = "bench_silicon_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/silicon_sse.zig"),
.root_source_file = b.path("src/weirdperformance/silicon_sse.zig"),
.target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .linux,
@@ -294,7 +213,7 @@ pub fn build(b: *std.Build) void {
const bench_bone_sse = b.addObject(.{
.name = "bench_bone_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/bone_sse.zig"),
.root_source_file = b.path("src/weirdperformance/bone_sse.zig"),
.target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .linux,
@@ -318,7 +237,7 @@ pub fn build(b: *std.Build) void {
const bench_particle_sse = b.addObject(.{
.name = "bench_particle_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/particle_sse.zig"),
.root_source_file = b.path("src/weirdperformance/particle_sse.zig"),
.target = b.resolveTargetQuery(.{
.cpu_arch = .x86,
.os_tag = .linux,
@@ -330,7 +249,7 @@ pub fn build(b: *std.Build) void {
const bench_cull_sse = b.addObject(.{
.name = "bench_cull_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/cull_sse.zig"),
.root_source_file = b.path("src/weirdperformance/cull_sse.zig"),
.target = bench_target,
.optimize = .ReleaseFast,
}),
@@ -344,7 +263,7 @@ pub fn build(b: *std.Build) void {
const bench_entity_sse = b.addObject(.{
.name = "bench_entity_sse",
.root_module = b.createModule(.{
.root_source_file = b.path("src/performance/entity_sse.zig"),
.root_source_file = b.path("src/weirdperformance/entity_sse.zig"),
.target = bench_target,
.optimize = .ReleaseFast,
}),
@@ -396,6 +315,7 @@ pub fn build(b: *std.Build) void {
build_all_step.dependOn(&noperf_install.step);
inline for (module_list) |variant_mod| {
if (!variant_mod.default) continue;
@setEvalBranchQuota(10000);
const opts = b.addOptions();
inline for (module_list) |m| {
+4 -1
View File
@@ -43,7 +43,7 @@ const transform44 = if (build_opts.transform44) @import("transform44/transform44
const addonperf = if (build_opts.addonperf) @import("addonperf/addonperf.zig") else struct {};
const ssemaths = if (build_opts.ssemaths) @import("ssemaths/ssemaths.zig") else struct {};
const silicon = if (build_opts.silicon) @import("silicon/silicon.zig") else struct {};
const weirdperformance = if (build_opts.weirdperformance) @import("performance/weirdperformance.zig") else struct {};
const weirdperformance = if (build_opts.weirdperformance) @import("weirdperformance/weirdperformance.zig") else struct {};
const module_active = @import("module_active.zig");
@@ -666,6 +666,9 @@ fn engineInitDetour() callconv(hook.cc.stdcall) void {
if (build_opts.silicon) {
silicon.lateInit();
}
if (build_opts.weirdperformance) {
weirdperformance.lateInit();
}
}
// =============================================================================
+4 -140
View File
@@ -15,37 +15,6 @@ const std = @import("std");
const hook = @import("zhook");
const logging = @import("../logging.zig");
const mod_mutex = @import("../mutex.zig");
extern fn clipPolygonToSinglePlane(u32, u32, u32) void;
extern fn buildTrianglePlanes(u32, u32, u32, u32, u32) u32;
extern fn rayTriangleIntersection(u32, u32, u32, u32, u32, u32) u32;
extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void;
extern fn multiplyMatrix4x4(u32, u32, u32) u32;
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn renderParticleSprites_REF(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn resetParticleCache() void;
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void;
extern fn findObjectByGUID_Cached(u32, u32) callconv(.{ .x86_stdcall = .{} }) u32;
extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern var stride_info: [8]u32; // exported from particle_sse.zig
extern var debug_vertex_count: u32;
extern var debug_max_sprites: u32;
extern var debug_fmt_index: u32;
extern var debug_data_ptr: u32;
var stride_dumped: bool = false;
/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit)
/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame.
fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void {
transformImpl_SSE(this, mat1, mat2, mat3, mat4);
}
extern fn transformMatrix4x4_REF(u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern var bisect_stop_section: u32;
@@ -82,12 +51,6 @@ const AB_OTHER_HOOKS = true;
var diag_cmp_count: u32 = 0;
export var original_trampoline: u32 = 0; // DEBUG: expose trampoline for REF passthrough test
// Teardown guard: set true when CleanupWorldAndEntities fires.
// During teardown, SceneObject data may be partially freed — our SSE code
// must not process it. Falls back to original function which the game
// controls. NOTE: binary patching (instead of hooking) would avoid this
// issue entirely since the patched code IS the original entry point.
var teardown_active: bool = false;
// Persistent blit totals per A/B mode — NOT reset each dump period.
@@ -325,11 +288,7 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco
t44_depth +|= 1;
if (t44_depth > prof.t44_max_depth) prof.t44_max_depth = t44_depth;
if (teardown_active) {
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
} else {
transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4);
}
transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 });
t44_depth -|= 1;
const elapsed = rdtsc() - start;
@@ -392,7 +351,6 @@ const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void;
var world_update_hook: hook.Detour(WorldUpdateFn) = .{};
fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
resetParticleCache();
const now = rdtsc();
if (last_frame_tsc != 0) {
const delta = now - last_frame_tsc;
@@ -410,23 +368,6 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
}
}
// =============================================================================
// Hook: World_HandleLogoutCleanup (0x491180)
// Fires at the START of the logout/disconnect cleanup sequence, BEFORE any
// model data is freed. Sets teardown_active flag so our SSE code falls back
// to the original function during the entire cleanup chain.
// NOTE: binary patching instead of hooking would avoid this issue entirely.
// =============================================================================
const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void;
var teardown_hook: hook.Detour(TeardownFn) = .{};
fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void {
teardown_active = true;
teardown_hook.callOriginal(.{});
teardown_active = false;
}
// =============================================================================
// Hook: RenderTextureQuads (0x76FB00)
// __fastcall(ECX=RenderBatch*) — no stack params, RET
@@ -897,12 +838,6 @@ var staticcull_hook: hook.Detour(Fn2) = .{}; // ProcessStaticObjectsCulling: fas
fn clipDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque {
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
clipPolygonToSinglePlane(a, b, c);
prof.clip_cycles +|= rdtsc() - s;
prof.clip_calls +|= 1;
return null; // original is void — EAX not read by callers
}
const ret = clip_hook.callOriginal(.{ a, b, c });
prof.clip_cycles +|= rdtsc() - s;
prof.clip_calls +|= 1;
@@ -948,13 +883,12 @@ fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyo
prof.glyph_calls +|= 1;
return ret;
}
fn particleDetour(a: u32, _: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
// a=ECX(emitter), c=particleData, d=vertexBuffers
fn particleDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
const result = renderParticleSprites_SSE(a, c, d);
const ret = particle_hook.callOriginal(.{ a, b, c, d });
prof.particle_cycles +|= rdtsc() - s;
prof.particle_calls +|= 1;
return @ptrFromInt(result);
return ret;
}
fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
@@ -965,12 +899,6 @@ fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*
}
fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
updateEntityAndChunksPositions(a);
prof.entpos_cycles +|= rdtsc() - s;
prof.entpos_calls +|= 1;
return null;
}
const ret = entpos_hook.callOriginal(.{ a, b });
prof.entpos_cycles +|= rdtsc() - s;
prof.entpos_calls +|= 1;
@@ -1022,12 +950,6 @@ fn spatialDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
fn raytriDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque {
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = rayTriangleIntersection(a, b, c, d, e, f);
prof.raytri_cycles +|= rdtsc() - s;
prof.raytri_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = raytri_hook.callOriginal(.{ a, b, c, d, e, f });
prof.raytri_cycles +|= rdtsc() - s;
prof.raytri_calls +|= 1;
@@ -1059,12 +981,6 @@ fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque {
}
fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = performSpatialCulling(a, c, d);
prof.cull_cycles +|= rdtsc() - s;
prof.cull_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = cull_hook.callOriginal(.{ a, b, c, d });
prof.cull_cycles +|= rdtsc() - s;
prof.cull_calls +|= 1;
@@ -1072,12 +988,6 @@ fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyop
}
fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = performCollisionDetectionSSE(a, c, d);
prof.colldet_cycles +|= rdtsc() - s;
prof.colldet_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = colldet_hook.callOriginal(.{ a, b, c, d });
prof.colldet_cycles +|= rdtsc() - s;
prof.colldet_calls +|= 1;
@@ -1216,13 +1126,6 @@ fn bboxchkDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*an
fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
// thiscall: a=ECX=matrix, b=EDX=unused, c=angle, d=axis_ptr, e=is_unit
rotateMatrixByAxisAngle(a, c, d, e);
prof.rotmat_cycles +|= rdtsc() - s;
prof.rotmat_calls +|= 1;
return null; // void function, EAX not read by callers
}
const ret = rotmat_hook.callOriginal(.{ a, b, c, d, e });
prof.rotmat_cycles +|= rdtsc() - s;
prof.rotmat_calls +|= 1;
@@ -1231,12 +1134,6 @@ fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcal
fn triplaneDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque {
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = buildTrianglePlanes(a, b, c, d, e);
prof.triplane_cycles +|= rdtsc() - s;
prof.triplane_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = triplane_hook.callOriginal(.{ a, b, c, d, e });
prof.triplane_cycles +|= rdtsc() - s;
prof.triplane_calls +|= 1;
@@ -1252,13 +1149,6 @@ fn partsetupDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaqu
fn matmulDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque {
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
// fastcall: a=ECX=result, b=EDX=left, c=right
const ret = multiplyMatrix4x4(a, b, c);
prof.matmul_cycles +|= rdtsc() - s;
prof.matmul_calls +|= 1;
return @ptrFromInt(ret);
}
const ret = matmul_hook.callOriginal(.{ a, b, c });
prof.matmul_cycles +|= rdtsc() - s;
prof.matmul_calls +|= 1;
@@ -1294,12 +1184,6 @@ fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u3
}
fn raytriIntDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque {
const s = rdtsc();
if (AB_OTHER_HOOKS and ab_use_custom) {
const ret = rayTriIntersectIndexedInt(a, b, c, d, e, f);
prof.raytri_int_cycles +|= rdtsc() - s;
prof.raytri_int_calls +|= 1;
return @ptrFromInt(@as(u32, ret));
}
const ret = raytri_int_hook.callOriginal(.{ a, b, c, d, e, f });
prof.raytri_int_cycles +|= rdtsc() - s;
prof.raytri_int_calls +|= 1;
@@ -1606,24 +1490,6 @@ fn dumpStats() void {
guid_cache_evictions = 0;
}
// Dump particle VB stride info (once)
if (debug_vertex_count != 0) {
log.fmt(" partsetup_debug: verts={d} maxSprites={d} fmt={d} dataPtr=0x{x}\n", .{
debug_vertex_count, debug_max_sprites, debug_fmt_index, debug_data_ptr,
});
debug_vertex_count = 0;
}
if (stride_info[0] != 0 and !stride_dumped) {
stride_dumped = true;
log.fmt(" vb_strides: pos={d} norm={d} color={d} tc={d}\n", .{
stride_info[0], stride_info[1], stride_info[2], stride_info[3],
});
log.fmt(" vb_bases: pos=0x{x} norm=0x{x} color=0x{x} tc=0x{x}\n", .{
stride_info[4], stride_info[5], stride_info[6], stride_info[7],
});
}
// Flip A/B mode
ab_use_custom = !ab_use_custom;
diag_cmp_count = 0;
@@ -1687,7 +1553,6 @@ pub fn installHooks() void {
_ = transform_hook.attach(0x714260, &transformDetour);
original_trampoline = @intCast(transform_hook.inner.trampoline);
}
_ = teardown_hook.attach(0x491180, &teardownDetour);
_ = render_frame_hook.attach(0x707680, &renderFrameDetour);
_ = exec_render_pass_hook.attach(0x708900, &execRenderPassDetour);
_ = world_update_hook.attach(0x482EA0, &worldUpdateDetour);
@@ -1776,7 +1641,6 @@ pub fn removeHooks() void {
render_frame_hook.detach();
exec_render_pass_hook.detach();
world_update_hook.detach();
teardown_hook.detach();
render_quads_hook.detach();
movement_hook.detach();
interp_kf_hook.detach();
@@ -1101,7 +1101,7 @@ fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void {
// mat3(offset_vec3*), mat4(scale_float_bits)
// =============================================================================
export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void {
pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void {
@setEvalBranchQuota(50000);
// =========================================================================
@@ -9,7 +9,7 @@ const V4 = @Vector(4, f32);
const CLIP_EPSILON: f32 = @bitCast(@as(u32, 0x3ab60b61)); // +0.00139, global at 0x80dfec
const MOVEMENT_EPSILON: f32 = @bitCast(@as(u32, 0x35800000)); // 9.54e-7, global at 0x8026bc
export fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void {
pub fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void {
const plane: [*]const f32 = @ptrFromInt(plane_addr);
const poly: [*]f32 = @ptrFromInt(poly_addr);
const new_attrib: f32 = @bitCast(attrib_bits);
@@ -116,7 +116,7 @@ inline fn lerp(poly: [*]f32, attribs: [*]f32, out: *u32, p: [*]const f32, c: [*]
// planes[3]: cap plane (from plane_normal, offset by offset_vector)
// =============================================================================
export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 {
pub fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 {
const verts: [*]const f32 = @ptrFromInt(verts_addr);
const indices: [*]const u8 = @ptrFromInt(indices_addr);
const plane_normal: [*]const f32 = @ptrFromInt(normal_addr);
@@ -191,7 +191,7 @@ export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u
// Returns: 1 = hit, 0 = miss
// =============================================================================
export fn rayTriangleIntersection(
pub fn rayTriangleIntersection(
ray_addr: u32,
verts_addr: u32,
indices_addr: u32,
@@ -267,7 +267,7 @@ export fn rayTriangleIntersection(
// Returns result pointer (EAX = result_addr).
// =============================================================================
export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 {
pub fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 {
const result: [*]f32 = @ptrFromInt(result_addr);
const left: [*]const f32 = @ptrFromInt(left_addr);
const right: [*]const f32 = @ptrFromInt(right_addr);
@@ -296,7 +296,7 @@ export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u
// __thiscall(ECX=matrix, stack: angle, axis_ptr, is_unit_flag)
// =============================================================================
export fn rotateMatrixByAxisAngle(
pub fn rotateMatrixByAxisAngle(
matrix_addr: u32,
angle_bits: u32,
axis_addr: u32,
@@ -15,7 +15,7 @@
/// Compute outcodes for `count` vertices at `verts_ptr` (stride 12 bytes = 3 floats)
/// against AABB at `bounds_ptr` (6 floats: minX, minY, minZ, maxX, maxY, maxZ).
/// Writes results to `out_ptr` (1 byte per vertex).
export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void {
pub fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void {
if (count == 0) return;
const v_min: V4 = .{ readF32(bounds_ptr), readF32(bounds_ptr + 4), readF32(bounds_ptr + 8), 0 };
const v_max: V4 = .{ readF32(bounds_ptr + 12), readF32(bounds_ptr + 16), readF32(bounds_ptr + 20), 0 };
@@ -54,7 +54,7 @@ var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID
const origFindObjectByGUID = @as(*const fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32, @ptrFromInt(0x464890));
export fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
pub fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 {
const hash = (guid_lo ^ (guid_hi *% 0x9E3779B9)) & GUID_CACHE_MASK;
const entry = &guid_cache[hash];
@@ -92,7 +92,7 @@ const g_grid_offset: *const f32 = @ptrFromInt(0x86861C);
// Dot product coefficients at 0xC7CFB8..C7CFC4 (same as entpos view coeffs but different address)
const g_spatial_coeffs: u32 = 0xC7CFB8;
export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void {
pub fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void {
// Dot product: coeff_a * obj[0x5C] + coeff_b * obj[0x60] + coeff_c * obj[0x64] + coeff_d
const depth = readF32(g_spatial_coeffs) * readF32(obj + 0x5C) +
readF32(g_spatial_coeffs + 4) * readF32(obj + 0x60) +
@@ -144,7 +144,7 @@ export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void
// RET 0x10
// =============================================================================
export fn rayTriIntersectIndexedInt(
pub fn rayTriIntersectIndexedInt(
ray_ptr: u32,
vert_pool: u32,
indices_ptr: u32,
@@ -279,7 +279,7 @@ fn computeAllOutcodes(
/// Finds mesh data via hash, computes vertex outcodes against AABB from this+0x10,
/// then iterates triangles: filters by visibility mask, trivial-rejects by outcode AND,
/// adds survivors to global visible/render lists.
export fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
pub fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
if (g_guard.* == 0) return 0;
const hash_table = g_guard.*;
@@ -377,7 +377,7 @@ inline fn dot3(a: V4, b: V4) f32 {
/// Fully inlined SSE rewrite. No external calls except FindOrCreateHashEntry.
/// Moller-Trumbore ray-triangle intersection is inlined with SSE cross/dot,
/// eliminating 4 SetVector3 calls and the ray_tri function call per triangle.
export fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
pub fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 {
if (g_guard.* == 0) return 0;
const hash_table = g_guard.*;
@@ -52,7 +52,12 @@ const CacheSet = struct {
lru: u8 = 0,
};
var cache: [NUM_SETS]CacheSet = @splat(CacheSet{});
const WINAPI = @import("std").builtin.CallingConvention.winapi;
extern "kernel32" fn GetProcessHeap() callconv(WINAPI) ?*anyopaque;
extern "kernel32" fn HeapAlloc(hHeap: ?*anyopaque, dwFlags: u32, dwBytes: usize) callconv(WINAPI) ?[*]u8;
extern "kernel32" fn HeapFree(hHeap: ?*anyopaque, dwFlags: u32, lpMem: *anyopaque) callconv(WINAPI) i32;
var cache: ?[*]CacheSet = null;
var cache_entries: u32 = 0;
var cache_hits: u64 = 0;
var cache_negative_hits: u64 = 0;
@@ -66,8 +71,9 @@ var cache_miss_p2_archive: u64 = 0;
/// Lookup: hash picks set, check both ways for name match.
pub fn archiveCacheLookup(h: u32, path: [*:0]const u8) ?ArchiveCacheEntry {
const c = cache orelse return null;
const set_idx = h & (NUM_SETS - 1);
const set = &cache[set_idx];
const set = &c[set_idx];
const span = std.mem.span(path);
for (0..WAYS) |w| {
@@ -93,11 +99,12 @@ pub fn computeBlockEntry(archive: u32, index: u32) u32 {
}
pub fn archiveCacheInsert(h: u32, path: [*:0]const u8, outer: u32, inner: u32, block: u32, negative: bool) void {
const c = cache orelse return;
const span = std.mem.span(path);
if (span.len > CACHE_NAME_LEN) return;
const set_idx = h & (NUM_SETS - 1);
const set = &cache[set_idx];
const set = &c[set_idx];
var target: u8 = set.lru;
for (0..WAYS) |w| {
@@ -135,8 +142,9 @@ pub fn recordMissP2() void { cache_miss_p2 +|= 1; }
pub fn recordMissP2Archive() void { cache_miss_p2_archive +|= 1; }
pub fn getSlotOccupant(h: u32) ?[]const u8 {
const c = cache orelse return null;
const set_idx = h & (NUM_SETS - 1);
const set = &cache[set_idx];
const set = &c[set_idx];
for (0..WAYS) |w| {
if (set.entries[w].name_len != 0)
return set.entries[w].name[0..set.entries[w].name_len];
@@ -246,11 +254,19 @@ fn fileFindDetour(
}
pub fn install() bool {
const size = NUM_SETS * @sizeOf(CacheSet);
const heap = GetProcessHeap() orelse return false;
const ptr = HeapAlloc(heap, 0x00000008, size) orelse return false; // HEAP_ZERO_MEMORY
cache = @alignCast(@ptrCast(ptr));
return file_find_hook.attach(0x6549a0, &fileFindDetour) == .ok;
}
pub fn remove() void {
file_find_hook.detach();
if (cache) |c| {
if (GetProcessHeap()) |heap| _ = HeapFree(heap, 0, @ptrCast(c));
cache = null;
}
}
// =============================================================================
@@ -13,7 +13,6 @@
const hook_lib = @import("zhook");
const logging = @import("../logging.zig");
const std = @import("std");
// libdeflate C API
extern fn libdeflate_alloc_decompressor() ?*anyopaque;
@@ -24,15 +23,12 @@ var lib_available: bool = false;
var log: logging.Logger = .{};
// --- Thread-local decompressor ---
// Each thread lazily allocates its own decompressor on first use.
// OS-managed TLS via Zig's threadlocal -- works correctly on both
// native Windows and Wine without manual FS segment access.
threadlocal var tls_decomp: ?*anyopaque = null;
threadlocal var tls_decompressor: ?*anyopaque = null;
fn getTlsDecompressor() ?*anyopaque {
if (tls_decomp) |d| return d;
if (tls_decompressor) |d| return d;
const d = libdeflate_alloc_decompressor() orelse return null;
tls_decomp = d;
tls_decompressor = d;
return d;
}
@@ -0,0 +1,25 @@
/* Wraps libdeflate_deflate_decompress with SEH to catch access violations.
* libdeflate's fast path on 32-bit can crash on malformed input despite
* SAFETY_CHECKs. This wrapper catches the crash and returns an error code. */
#include "libdeflate.h"
#ifdef _WIN32
#include <windows.h>
__declspec(dllexport) int safe_deflate_decompress(
struct libdeflate_decompressor *d,
const void *in, size_t in_nbytes,
void *out, size_t out_nbytes_avail,
size_t *actual_out_nbytes_ret)
{
int result;
__try {
result = libdeflate_deflate_decompress(d, in, in_nbytes,
out, out_nbytes_avail, actual_out_nbytes_ret);
} __except(EXCEPTION_EXECUTE_HANDLER) {
result = 1; /* LIBDEFLATE_BAD_DATA */
}
return result;
}
#endif
@@ -0,0 +1,15 @@
/* malloc/free stubs for libdeflate using the Windows process heap. */
#include <stddef.h>
__declspec(dllimport) void *__stdcall GetProcessHeap(void);
__declspec(dllimport) void *__stdcall HeapAlloc(void *hHeap, unsigned long dwFlags, size_t dwBytes);
__declspec(dllimport) int __stdcall HeapFree(void *hHeap, unsigned long dwFlags, void *lpMem);
void *malloc(size_t size) {
return HeapAlloc(GetProcessHeap(), 0, size);
}
void free(void *ptr) {
if (ptr) HeapFree(GetProcessHeap(), 0, ptr);
}
@@ -0,0 +1,4 @@
#pragma once
#include <stddef.h>
extern void *malloc(size_t size);
extern void free(void *ptr);
@@ -0,0 +1,2 @@
#pragma once
/* libdeflate only uses stdio.h in debug paths — stub it out */
@@ -0,0 +1,4 @@
#pragma once
#include <stddef.h>
extern void *malloc(size_t size);
extern void free(void *ptr);
@@ -0,0 +1,6 @@
#pragma once
#include <stddef.h>
void *memcpy(void *dest, const void *src, size_t n);
void *memmove(void *dest, const void *src, size_t n);
void *memset(void *s, int c, size_t n);
int memcmp(const void *s1, const void *s2, size_t n);
@@ -168,7 +168,6 @@ const VBState = struct {
light: [3]u32,
fn load(vb: u32) VBState {
logStrides(vb);
return .{
.pos = ru32(vb + VB.pos),
.normal = ru32(vb + VB.normal),
@@ -260,37 +259,12 @@ inline fn emitVertex(vb: u32, px: f32, py: f32, pz: f32, color: u32, tu: f32, tv
// Reset each frame via resetParticleCache() called from the frame hook.
var cached_render_state: u32 = 0;
var stride_logged: bool = false;
var debug_logged: bool = false;
export var debug_vertex_count: u32 = 0;
export var debug_max_sprites: u32 = 0;
export var debug_fmt_index: u32 = 0;
export var debug_data_ptr: u32 = 0;
/// Reset per-frame caches. Call from OnWorldUpdate or executeSceneRenderPass hook.
export fn resetParticleCache() void {
pub fn resetParticleCache() void {
cached_render_state = 0;
}
/// Log VB strides once for analysis. Called from first VBState.load.
fn logStrides(vb: u32) void {
if (stride_logged) return;
stride_logged = true;
// Write to a known memory location that the profiler can dump, or just use
// the debug console. For now, store in a global we can read.
stride_info = .{
ru32(vb + VB.pos_stride),
ru32(vb + VB.normal_stride),
ru32(vb + VB.color_stride),
ru32(vb + VB.texcoord_stride),
ru32(vb + VB.pos),
ru32(vb + VB.normal),
ru32(vb + VB.color),
ru32(vb + VB.texcoord),
};
}
export var stride_info: [8]u32 = .{0} ** 8;
// =============================================================================
// RenderParticleSprites (0x7B2A50)
@@ -299,7 +273,7 @@ export var stride_info: [8]u32 = .{0} ** 8;
//
// Faithful recreation from assembly + Ghidra decompilation.
// =============================================================================
export fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 {
pub fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 {
const pd = particle_data; // particleData pointer (float*)
const vb = vertex_buffers; // vertexBuffers pointer (float**)
@@ -762,7 +736,7 @@ const SG = struct {
// Faithful recreation from Ghidra decompilation + assembly.
// All game function calls preserved, matrix math inlined with V4.
// =============================================================================
export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void {
pub fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void {
// =========================================================================
// Section 1: Identity matrices for render state
// Optimization: use static identity instead of rebuilding on stack each call.
@@ -999,15 +973,6 @@ export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC
// =========================================================================
gameRenderSorted(emitter, @intFromPtr(&vb_ptrs));
// DEBUG: log vertex count produced
if (!debug_logged and vb_ptrs[8] > 0) {
debug_logged = true;
debug_vertex_count = vb_ptrs[8];
debug_max_sprites = max_sprites;
debug_fmt_index = fmt_index;
debug_data_ptr = data_ptr;
}
gameUnlockVB(vb_ptr, 0);
gameDrawPrim(vb_ptr, fmt_index);
@@ -1095,7 +1060,7 @@ inline fn displayModeOffset(sprite_type: u32, count: u32) u32 {
return divided -% ru32(DISPLAY_MODE_OFFSET_TABLE + sprite_type * 4);
}
export fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void {
pub fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void {
// Early out: this+0xF2C == 0
if (ru32(this + 0xF2C) == 0) return;
@@ -54,7 +54,7 @@ inline fn cvtss2si(x: f32) i32 {
// --- 0x4549C0: normalizeVec3 (137K/7.5s) ---
// Naked thiscall: ECX=vec, [ESP+4]=length_bits. RET 4. Original: 38 bytes.
// rcpss + NR for fast reciprocal, then 3 multiplies.
export fn si_normalizeVec3() callconv(.naked) void {
pub fn si_normalizeVec3() callconv(.naked) void {
asm volatile (
// xmm0 = 1.0 / length (via rcpss + Newton-Raphson)
\\vmovss 4(%%esp), %%xmm0
@@ -77,7 +77,7 @@ export fn si_normalizeVec3() callconv(.naked) void {
// out = A * B (3x4 layout: 3x3 rotation + 3 translation)
// Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz]
// V4 per row: broadcast b[row*3+k], multiply with a's columns, accumulate.
export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 {
pub fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 {
const dst: [*]f32 = @ptrFromInt(out);
const aa: [*]const f32 = @ptrFromInt(a_ptr);
const b: [*]const f32 = @ptrFromInt(b_ptr);
@@ -117,7 +117,7 @@ export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 {
// --- 0x7BDDB0: rotateMatByQuat ---
// builds rotation matrix from quaternion, multiplies with existing 4x4 matrix
// Uses V4 for the matrix multiply (same pattern as bone_sse)
export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 {
pub fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 {
const q: [*]const f32 = @ptrFromInt(quat);
const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3];
const x2 = x + x; const y2 = y + y; const z2 = z + z;
@@ -146,7 +146,7 @@ export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 {
// --- 0x7BB860: createRotMat3x4 ---
// Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries.
export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 {
pub fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 {
const m: [*]f32 = @ptrFromInt(out);
const ax: [*]const f32 = @ptrFromInt(axis_ptr);
var x = ax[0]; var y = ax[1]; var z = ax[2];
@@ -168,7 +168,7 @@ export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normal
// --- 0x6329E0: distanceToPlane (525K/7.5s) ---
// __fastcall(ECX=point, EDX=plane, stack=direction), returns f64 via ST(0), RET 0x4.
export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 {
pub fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 {
const p: [*]const f32 = @ptrFromInt(point);
const pl: [*]const f32 = @ptrFromInt(plane);
const dir: [*]const f32 = @ptrFromInt(direction);
@@ -183,7 +183,7 @@ export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC
// Tests point against 6 frustum planes, produces 6-bit bitmask.
// Scalar @mulAdd dot4 per plane — the FMA chain has best throughput for this pattern.
// Tried: V4 batch 4 planes (gather kills it), V4 hsum (shuffle overhead kills it).
export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 {
pub fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 {
const mask: *u32 = @ptrFromInt(out_mask);
const pt = loadV3_1(point);
var bits: u32 = 0;
@@ -199,7 +199,7 @@ export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) ca
// --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) ---
// Slab AABB test. Branchless min/max for t0/t1 swap and tmin/tmax accumulation.
export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 {
pub fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 {
const bmin: [*]const f32 = @ptrFromInt(box_ptr);
const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC);
const start: [*]const f32 = @ptrFromInt(line_start);
@@ -225,7 +225,7 @@ export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32)
// --- 0x6869C0: testOBBFrustum ---
// Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test.
export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 {
pub fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 {
const aabb: [*]const f32 = @ptrFromInt(aabb_ptr);
const rot: [*]const f32 = @ptrFromInt(rot_ptr);
const t: [*]const f32 = @ptrFromInt(trans_ptr);
@@ -285,7 +285,7 @@ export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_
// --- 0x686B80: testSphereFrustum (375K/7.5s) ---
// Zig thiscall: naked asm tested at 10cy (vhaddps slow), Zig dot4v at 8cy.
export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 {
pub fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 {
const s: [*]const f32 = @ptrFromInt(sphere);
const center = V4{ s[0], s[1], s[2], 1.0 };
const r = s[3];
@@ -299,7 +299,7 @@ export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 {
// --- 0x7C0570: quatSlerp ---
// V4 for final blend, @mulAdd for dot product
export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 {
pub fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 {
const dst: [*]f32 = @ptrFromInt(out);
const av = loadV4(a_ptr);
const bv = loadV4(b_ptr);
@@ -326,7 +326,7 @@ export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(F
// --- 0x699330: isPointInsideBounds (1.7M/7.5s) ---
// __fastcall(ECX=a, EDX=b), returns u32.
export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 {
pub fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 {
const va: [*]const f32 = @ptrFromInt(a);
const vb: [*]const f32 = @ptrFromInt(b);
if (vb[0] <= va[0] and vb[1] <= va[1] and vb[2] <= va[2]) return 1;
@@ -334,7 +334,7 @@ export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 {
}
// --- 0x749280: calculateSinCos ---
export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void {
pub fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void {
const angle: f32 = @bitCast(angle_bits);
const sp: *f32 = @ptrFromInt(out_sin);
const cp: *f32 = @ptrFromInt(out_cos);
@@ -343,7 +343,7 @@ export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callco
}
// --- 0x7BE5B0: createZRotMat3x3 ---
export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 {
pub fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 {
const m: [*]f32 = @ptrFromInt(out);
const angle: f32 = @bitCast(angle_bits);
const c = @cos(angle); const s = @sin(angle);
@@ -356,7 +356,7 @@ export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 {
// --- 0x7BCEF0: transposeMat4x4 ---
// Naked thiscall: ECX=src, [ESP+4]=dst. RET 4. Original: 156 bytes.
// SSE unpacklo/unpackhi transpose: 4 loads + 4 shuffles + 4 stores.
export fn si_transposeMat4x4() callconv(.naked) void {
pub fn si_transposeMat4x4() callconv(.naked) void {
asm volatile (
\\mov 4(%%esp), %%eax
// Load 4 rows from src (ECX)
@@ -387,7 +387,7 @@ export fn si_transposeMat4x4() callconv(.naked) void {
// --- 0x7BB420: mulMat3x4InPlace ---
// this = this * matB. V4 columns loaded upfront, write directly back (no tmp needed).
export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 {
pub fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 {
const a: [*]f32 = @ptrFromInt(mat_a);
const b: [*]const f32 = @ptrFromInt(mat_b);
@@ -419,7 +419,7 @@ export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 {
// --- 0x6720F0: normalizeVec3InPlace ---
// sqrt + reciprocal. 14cy (2.2x). rsqrt+NR tested at 15cy — no gain, compiler's
// vsqrtss+vdivss pipeline is already optimal for scalar inverse sqrt.
export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void {
pub fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void {
const v: [*]f32 = @ptrFromInt(vec);
const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]);
if (len > 1.0e-20) {
@@ -432,7 +432,7 @@ export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void {
// --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) ---
// thiscall(ECX=this, stack=vec). Scale is a global at 0x81207C, NOT a parameter.
export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void {
pub fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void {
const obj: [*]f32 = @ptrFromInt(this);
const v: [*]const f32 = @ptrFromInt(vec);
const scale: f32 = @as(*const f32, @ptrFromInt(0x81207C)).*;
@@ -447,7 +447,7 @@ export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void {
// --- 0x71BF60: addToColorAccumulator (10K/7.5s) ---
// Naked thiscall: ECX=this, [ESP+4]=color_ptr. RET 4. Original: 34 bytes.
// 3 SSE adds at this+0x6C from color[0..2].
export fn si_addToColorAccumulator() callconv(.naked) void {
pub fn si_addToColorAccumulator() callconv(.naked) void {
asm volatile (
\\mov 4(%%esp), %%eax
\\vmovss (%%eax), %%xmm0
@@ -465,7 +465,7 @@ export fn si_addToColorAccumulator() callconv(.naked) void {
// --- 0x7B7A80: packParticleColor (2K/7.5s) ---
// V4 multiply + clamp, then packed round+convert via @Vector(4, i32) for all channels at once.
export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void {
pub fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void {
const base: [*]u8 = @ptrFromInt(obj);
const out: *align(1) u32 = @ptrCast(base + 0x12C);
const alpha = base[0x12F];
@@ -480,7 +480,7 @@ export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32)
// --- 0x7B7B10: setParticleAlpha (2K/7.5s) ---
// Naked fastcall: ECX=obj, [ESP+4]=alpha_bits. RET 4.
// Clamp alpha*255 to [0,255], write byte to obj+0x12F.
export fn si_setParticleAlpha() callconv(.naked) void {
pub fn si_setParticleAlpha() callconv(.naked) void {
asm volatile (
\\vmovss 4(%%esp), %%xmm0
\\mov $0x437F0000, %%eax
@@ -499,7 +499,7 @@ export fn si_setParticleAlpha() callconv(.naked) void {
// --- 0x40A2B0: __ftol ---
// Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64).
// SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original)
export fn si_ftol() callconv(.naked) void {
pub fn si_ftol() callconv(.naked) void {
asm volatile (
\\sub $8, %%esp
\\fisttpll (%%esp)
@@ -511,7 +511,7 @@ export fn si_ftol() callconv(.naked) void {
// --- 0x602630: vec3Dot (31K/7.5s) ---
// __fastcall(ECX=a, EDX=b), returns f64 via ST(0).
export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 {
pub fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 {
const va: [*]const f32 = @ptrFromInt(a);
const vb: [*]const f32 = @ptrFromInt(b);
return @floatCast(@mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0])));
@@ -519,7 +519,7 @@ export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 {
// --- 0x686820: translateBoundingVol ---
// @mulAdd for plane distances. Scalar corner adds (stride 3 — V4 unaligned tested, slower).
export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
pub fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
const obj: [*]f32 = @ptrFromInt(this);
const off: [*]const f32 = @ptrFromInt(offset);
const dx = off[0]; const dy = off[1]; const dz = off[2];
@@ -543,7 +543,7 @@ export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void {
// Original: 380 bytes, 2 calls to mat*vec3 (0x7BCA80), x87 perspective divide, x87 column scan.
// SSE: inline V4 mat*vec3, SSE perspective divide, 4-wide column scan.
// __fastcall(bbox_ECX, flags_EDX, radius_stack), RET 0x4
export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 {
pub fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 {
// Early out: global occlusion flag bit 5
if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 0x20) == 0) return 0;
@@ -638,7 +638,7 @@ export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(F
// __fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack), RET 0x8
// addGeometryToBuffer at 0x6ABD90: __fastcall(queryBox_ECX, nodeData_EDX, resultBuf_stack), RET 0x4
// Visited sentinel: *(u32*)0xC89F20
export fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 {
pub fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 {
if ((flags & 0xF0000F) == 0) return 1;
const addGeometryToBuffer: *const fn (u32, u32, u32) callconv(FC) void = @ptrFromInt(0x6ABD90);
@@ -22,21 +22,8 @@ const filecache = @import("filecache.zig");
pub const module_name: [*:0]const u8 = "weirdperformance";
// Provide malloc/free for libdeflate's default allocator (linked without libc).
// Use game's Storm memory manager:
// ReallocMemory (0x646320): __stdcall(ptr, size, filename, line, flags) → ptr
// When ptr=NULL, acts as malloc via AllocateBufferWithPowerOfTwo.
// FreeMemory (0x646430): __stdcall(ptr, filename, line, flags) RET 0x10 = 4 params
const gameRealloc: *const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) ?*anyopaque = @ptrFromInt(0x646320);
const gameFree: *const fn (u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) u32 = @ptrFromInt(0x646430);
export fn malloc(size: usize) callconv(.c) ?*anyopaque {
return gameRealloc(0, @intCast(size), 0, 0, 0);
}
export fn free(ptr: ?*anyopaque) callconv(.c) void {
if (ptr) |p| _ = gameFree(@intFromPtr(p), 0, 0, 0);
}
// malloc/free for libdeflate provided by stubs/game_alloc.c (compiled into
// the libdeflate static lib). This avoids exporting malloc/free from the DLL.
var g_mutex: ?*anyopaque = null;
var g_is_hook_owner: bool = false;
@@ -47,12 +34,18 @@ pub fn isActive() bool {
}
// =============================================================================
// Extern SSE functions (from separate ReleaseFast compilation units)
// SSE functions (imported directly to avoid addObject SizeOfImage bloat)
// =============================================================================
extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void;
extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn resetParticleCache() void;
const bone_sse = @import("bone_sse.zig");
const particle_sse = @import("particle_sse.zig");
const clip_sse = @import("clip_sse.zig");
const cull_sse = @import("cull_sse.zig");
const silicon_sse = @import("silicon_sse.zig");
const transformImpl_SSE = bone_sse.transformImpl_SSE;
const renderParticleSprites_SSE = particle_sse.renderParticleSprites_SSE;
const resetParticleCache = particle_sse.resetParticleCache;
// =============================================================================
// transformMatrix4x4 hook (0x714260)
@@ -218,35 +211,12 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void {
// =============================================================================
// cull_sse.zig
extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8;
const performSpatialCulling = cull_sse.performSpatialCulling;
const performCollisionDetectionSSE = cull_sse.performCollisionDetectionSSE;
const rayTriIntersectIndexedInt = cull_sse.rayTriIntersectIndexedInt;
// silicon_sse.zig
const sse = struct {
extern fn si_normalizeVec3() callconv(.naked) void;
extern fn si_mulMat3x4(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
extern fn si_rotateMatByQuat(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn si_createRotMat3x4(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
extern fn si_classifyPointFrustum(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn si_checkBoxLineIntersect(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
extern fn si_testOBBFrustum(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn si_testSphereFrustum(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn si_quatSlerp(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
extern fn si_calculateSinCos(u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void;
extern fn si_createZRotMat3x3(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn si_transposeMat4x4() callconv(.naked) void;
extern fn si_mulMat3x4InPlace(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32;
extern fn si_normalizeVec3InPlace(u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn si_addVec3ToAccumulator(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn si_addToColorAccumulator() callconv(.naked) void;
extern fn si_packParticleColor(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn si_setParticleAlpha() callconv(.naked) void;
extern fn si_ftol() callconv(.naked) void;
extern fn si_translateBoundingVol(u32, u32) callconv(.{ .x86_thiscall = .{} }) void;
extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
extern fn si_frustumCullBBox(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
};
const sse = silicon_sse;
const PatchEntry = struct {
target: u32,
@@ -348,7 +318,10 @@ pub fn installHooks() void {
// libdeflate inflate replacement
if (inflate_hook.install()) installed += 1;
// TSC timer calibration + OS timer tweaks
}
pub fn lateInit() void {
if (!g_is_hook_owner) return;
timer_fix.init();
}
+206
View File
@@ -0,0 +1,206 @@
/* zconf.h -- configuration of the zlib compression library
* Copyright (C) 1995-2024 Jean-loup Gailly, Mark Adler
* For conditions of distribution and use, see copyright notice in zlib.h
*/
#ifndef ZCONF_H
#define ZCONF_H
#include "zlib_name_mangling.h"
#if !defined(_WIN32) && defined(__WIN32__)
# define _WIN32
#endif
/* Clang macro for detecting declspec support
* https://clang.llvm.org/docs/LanguageExtensions.html#has-declspec-attribute
*/
#ifndef __has_declspec_attribute
# define __has_declspec_attribute(x) 0
#endif
#if defined(ZLIB_CONST) && !defined(z_const)
# define z_const const
#else
# define z_const
#endif
/* Maximum value for memLevel in deflateInit2 */
#ifndef MAX_MEM_LEVEL
# define MAX_MEM_LEVEL 9
#endif
/* Maximum value for windowBits in deflateInit2 and inflateInit2.
* WARNING: reducing MAX_WBITS makes minigzip unable to extract .gz files
* created by gzip. (Files created by minigzip can still be extracted by
* gzip.)
*/
#ifndef MIN_WBITS
# define MIN_WBITS 8 /* 256 LZ77 window */
#endif
#ifndef MAX_WBITS
# define MAX_WBITS 15 /* 32K LZ77 window */
#endif
/* The memory requirements for deflate are (in bytes):
(1 << (windowBits+2)) + (1 << (memLevel+9))
that is: 128K for windowBits=15 + 128K for memLevel = 8 (default values)
plus a few kilobytes for small objects. For example, if you want to reduce
the default memory requirements from 256K to 128K, compile with
make CFLAGS="-O -DMAX_WBITS=14 -DMAX_MEM_LEVEL=7"
Of course this will generally degrade compression (there's no free lunch).
The memory requirements for inflate are (in bytes) 1 << windowBits
that is, 32K for windowBits=15 (default value) plus about 7 kilobytes
for small objects.
*/
/* Type declarations */
#ifndef OF /* function prototypes */
# define OF(args) args
#endif
#ifdef ZLIB_INTERNAL
# define Z_INTERNAL ZLIB_INTERNAL
#endif
/* If building or using zlib as a DLL, define ZLIB_DLL.
* This is not mandatory, but it offers a little performance increase.
*/
#if defined(ZLIB_DLL) && (defined(_WIN32) || (__has_declspec_attribute(dllexport) && __has_declspec_attribute(dllimport)))
# ifdef Z_INTERNAL
# define Z_EXTERN extern __declspec(dllexport)
# else
# define Z_EXTERN extern __declspec(dllimport)
# endif
#endif
/* If building or using zlib with the WINAPI/WINAPIV calling convention,
* define ZLIB_WINAPI.
* Caution: the standard ZLIB1.DLL is NOT compiled using ZLIB_WINAPI.
*/
#if defined(ZLIB_WINAPI) && defined(_WIN32)
# ifndef WIN32_LEAN_AND_MEAN
# define WIN32_LEAN_AND_MEAN
# endif
# include <windows.h>
/* No need for _export, use ZLIB.DEF instead. */
/* For complete Windows compatibility, use WINAPI, not __stdcall. */
# define Z_EXPORT WINAPI
# define Z_EXPORTVA WINAPIV
#endif
#ifndef Z_EXTERN
# define Z_EXTERN extern
#endif
#ifndef Z_EXPORT
# define Z_EXPORT
#endif
#ifndef Z_EXPORTVA
# define Z_EXPORTVA
#endif
/* Conditional exports */
#define ZNG_CONDEXPORT Z_INTERNAL
/* For backwards compatibility */
#ifndef ZEXTERN
# define ZEXTERN Z_EXTERN
#endif
#ifndef ZEXPORT
# define ZEXPORT Z_EXPORT
#endif
#ifndef ZEXPORTVA
# define ZEXPORTVA Z_EXPORTVA
#endif
#ifndef FAR
# define FAR
#endif
/* Legacy zlib typedefs for backwards compatibility. Don't assume stdint.h is defined. */
typedef unsigned char Byte;
typedef Byte Bytef;
typedef unsigned int uInt; /* 16 bits or more */
typedef unsigned long uLong; /* 32 bits or more */
typedef char charf;
typedef int intf;
typedef uInt uIntf;
typedef uLong uLongf;
typedef void const *voidpc;
typedef void *voidpf;
typedef void *voidp;
typedef unsigned int z_crc_t;
#if 1 /* was set to #if 1 by configure/cmake/etc */
# define Z_HAVE_UNISTD_H
#endif
#ifdef NEED_PTRDIFF_T /* may be set to #if 1 by configure/cmake/etc */
typedef PTRDIFF_TYPE ptrdiff_t;
#endif
#include <sys/types.h> /* for off_t */
#include <stddef.h> /* for wchar_t and NULL */
/* a little trick to accommodate both "#define _LARGEFILE64_SOURCE" and
* "#define _LARGEFILE64_SOURCE 1" as requesting 64-bit operations, (even
* though the former does not conform to the LFS document), but considering
* both "#undef _LARGEFILE64_SOURCE" and "#define _LARGEFILE64_SOURCE 0" as
* equivalently requesting no 64-bit operations
*/
#if defined(_LARGEFILE64_SOURCE) && -_LARGEFILE64_SOURCE - -1 == 1
# undef _LARGEFILE64_SOURCE
#endif
#if defined(Z_HAVE_UNISTD_H)
# include <unistd.h> /* for SEEK_*, off_t, and _LFS64_LARGEFILE */
# ifndef z_off_t
# define z_off_t off_t
# endif
#endif
#if defined(_LFS64_LARGEFILE) && _LFS64_LARGEFILE-0
# define Z_LFS64
#endif
#if defined(_LARGEFILE64_SOURCE) && defined(Z_LFS64)
# define Z_LARGE64
#endif
#if defined(_FILE_OFFSET_BITS) && _FILE_OFFSET_BITS-0 == 64 && defined(Z_LFS64)
# define Z_WANT64
#endif
#if !defined(SEEK_SET)
# define SEEK_SET 0 /* Seek from beginning of file. */
# define SEEK_CUR 1 /* Seek from current position. */
# define SEEK_END 2 /* Set file pointer to EOF plus "offset" */
#endif
#ifndef z_off_t
# define z_off_t long
#endif
#if !defined(_WIN32) && defined(Z_LARGE64)
# define z_off64_t off64_t
#else
# if defined(__MSYS__)
# define z_off64_t _off64_t
# elif defined(_WIN32) && !defined(__GNUC__)
# define z_off64_t __int64
# else
# define z_off64_t z_off_t
# endif
#endif
typedef size_t z_size_t;
#endif /* ZCONF_H */
File diff suppressed because it is too large Load Diff