From db675c7a7e786776076789efeb572acc79c210e5 Mon Sep 17 00:00:00 2001 From: MarcelineVQ Date: Sat, 28 Mar 2026 05:31:31 -0700 Subject: [PATCH] fix: MSVC ABI, heap allocator, deferred timer - vanillafixes compat - Restore MSVC ABI (was accidentally GNU since v0.6.0, broke .CRT section) - Replace game allocator with Windows process heap for filecache and libdeflate malloc/free - game allocator not initialized during DllMain when injected via CreateRemoteThread - Defer timer calibration (Sleep 500ms) to lateInit - blocks under loader lock during DllMain - Remove exported malloc/free symbols from DLL - Eliminate addObject compilation units for SSE files - direct @import with AVX target instead - Heap-allocate filecache (was 9.3MB static BSS) - Strip transform44 of performance/ externs, pure profiling only - Rename performance/ to weirdperformance/ to match module convention - Skip default-off modules in all-variants build step - Remove dead debug vars and stride logging from particle_sse --- RELEASING.md | 42 +- build.zig | 120 +- src/main.zig | 5 +- src/transform44/transform44.zig | 144 +- .../INFLATE_RESEARCH.md | 0 .../bone_sse.zig | 2 +- .../clip_sse.zig | 10 +- .../cull_sse.zig | 12 +- .../entity_sse.zig | 0 .../filecache.zig | 24 +- .../inflate_hook.zig | 10 +- .../libdeflate/common_defs.h | 0 .../libdeflate/lib/adler32.c | 0 .../libdeflate/lib/cpu_features_common.h | 0 .../libdeflate/lib/decompress_template.h | 0 .../libdeflate/lib/deflate_constants.h | 0 .../libdeflate/lib/deflate_decompress.c | 0 .../libdeflate/lib/lib_common.h | 0 .../libdeflate/lib/utils.c | 0 .../libdeflate/lib/x86/adler32_impl.h | 0 .../libdeflate/lib/x86/adler32_template.h | 0 .../libdeflate/lib/x86/cpu_features.c | 0 .../libdeflate/lib/x86/cpu_features.h | 0 .../libdeflate/lib/x86/crc32_impl.h | 0 .../lib/x86/crc32_pclmul_template.h | 0 .../libdeflate/lib/x86/decompress_impl.h | 0 .../libdeflate/lib/x86/matchfinder_impl.h | 0 .../libdeflate/lib/zlib_constants.h | 0 .../libdeflate/lib/zlib_decompress.c | 0 .../libdeflate/libdeflate.h | 0 .../libdeflate/safe_decompress.c | 25 + .../libdeflate/stubs/game_alloc.c | 15 + .../libdeflate/stubs/malloc.h | 4 + .../libdeflate/stubs/setjmp.h | 0 src/weirdperformance/libdeflate/stubs/stdio.h | 2 + .../libdeflate/stubs/stdlib.h | 4 + .../libdeflate/stubs/string.h | 6 + .../particle_sse.zig | 43 +- .../silicon_sse.zig | 50 +- .../timer_fix.zig | 0 .../weirdperformance.zig | 67 +- src/weirdperformance/zconf.h | 206 ++ src/weirdperformance/zlib.h | 1859 +++++++++++++++++ 43 files changed, 2245 insertions(+), 405 deletions(-) rename src/{performance => weirdperformance}/INFLATE_RESEARCH.md (100%) rename src/{performance => weirdperformance}/bone_sse.zig (99%) rename src/{performance => weirdperformance}/clip_sse.zig (97%) rename src/{performance => weirdperformance}/cull_sse.zig (97%) rename src/{performance => weirdperformance}/entity_sse.zig (100%) rename src/{performance => weirdperformance}/filecache.zig (92%) rename src/{performance => weirdperformance}/inflate_hook.zig (92%) rename src/{performance => weirdperformance}/libdeflate/common_defs.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/adler32.c (100%) rename src/{performance => weirdperformance}/libdeflate/lib/cpu_features_common.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/decompress_template.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/deflate_constants.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/deflate_decompress.c (100%) rename src/{performance => weirdperformance}/libdeflate/lib/lib_common.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/utils.c (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/adler32_impl.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/adler32_template.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/cpu_features.c (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/cpu_features.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/crc32_impl.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/crc32_pclmul_template.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/decompress_impl.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/x86/matchfinder_impl.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/zlib_constants.h (100%) rename src/{performance => weirdperformance}/libdeflate/lib/zlib_decompress.c (100%) rename src/{performance => weirdperformance}/libdeflate/libdeflate.h (100%) create mode 100644 src/weirdperformance/libdeflate/safe_decompress.c create mode 100644 src/weirdperformance/libdeflate/stubs/game_alloc.c create mode 100644 src/weirdperformance/libdeflate/stubs/malloc.h rename src/{performance => weirdperformance}/libdeflate/stubs/setjmp.h (100%) create mode 100644 src/weirdperformance/libdeflate/stubs/stdio.h create mode 100644 src/weirdperformance/libdeflate/stubs/stdlib.h create mode 100644 src/weirdperformance/libdeflate/stubs/string.h rename src/{performance => weirdperformance}/particle_sse.zig (97%) rename src/{performance => weirdperformance}/silicon_sse.zig (93%) rename src/{performance => weirdperformance}/timer_fix.zig (100%) rename src/{performance => weirdperformance}/weirdperformance.zig (80%) create mode 100644 src/weirdperformance/zconf.h create mode 100644 src/weirdperformance/zlib.h diff --git a/RELEASING.md b/RELEASING.md index 9825ed9..2fba57a 100644 --- a/RELEASING.md +++ b/RELEASING.md @@ -7,36 +7,12 @@ This project is developed entirely locally. The remote repo is **only** a distribution point for releases - no source code is pushed. -The remote `main` branch contains a single file: `README.md` (built from the -local `DLL_README.md`). This must be set up once when creating the repo: +The remote `main` branch contains `README.md` (built from the local +`DLL_README.md`), `weirdutils_api.h`, and issue templates under `.gitea/`. -```sh -tea repo create --name WeirdUtils --description "Vanilla WoW 1.12.1 utility DLLs" --login MarcelineVQ -``` - -Codeberg disables releases on new repos by default. Enable via API -(get your token from `grep 'token:' ~/.config/tea/config.yml | head -1 | awk '{print $2}'`): - -```sh -curl -s -X PATCH \ - -H "Authorization: token " \ - -H "Content-Type: application/json" \ - -d '{"has_releases":true}' \ - "https://codeberg.org/api/v1/repos/MarcelineVQ/WeirdUtils" -``` - -Then push the initial README: - -```sh -# In a temporary directory: -git init && git remote add origin ssh://git@codeberg.org/MarcelineVQ/WeirdUtils.git && git checkout -b main -cp /path/to/weirdutils/DLL_README.md README.md -git add README.md -git commit -m "Add README" -git push origin main -``` - -After that, the remote `main` only needs updating when `DLL_README.md` changes. +A local clone of the remote repo lives at `remote/WeirdUtils/`. The wiki +lives at `remote/wiki/`. Use these for all remote operations - no tmp clones +needed. ## 1. Bump module versions @@ -120,11 +96,12 @@ The module name list in the Developer Notes section must also only list released module names. ```sh -# from a clone or worktree of the remote repo +cd remote/WeirdUtils # edit README.md: remove sections for modules not in this release git add README.md git commit -m "Update README for vX.Y.Z" git push origin main +cd ../.. ``` ## 4. Write the release notes @@ -222,6 +199,11 @@ print(r[0]['id']) if r else print('not found') " ``` +## Known Issues + +- **vanillafixes launcher**: Incompatible with WeirdUtils DLL injection. Users + should load the DLL via WoW.exe + `dlls.txt` or another loader instead. + ## Checklist - [ ] Module versions bumped in `build.zig` for changed modules diff --git a/build.zig b/build.zig index 5854dde..be412cc 100644 --- a/build.zig +++ b/build.zig @@ -41,7 +41,7 @@ pub fn build(b: *std.Build) void { .cpu_arch = .x86, .os_tag = .windows, .abi = .msvc, - .cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2 }), + .cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }), }); const optimize = b.option(std.builtin.OptimizeMode, "optimize", "Optimization mode (default: ReleaseFast)") orelse .ReleaseFast; @@ -61,46 +61,6 @@ pub fn build(b: *std.Build) void { }); const zhook_mod = zhook_dep.module("zhook"); - // Hot math — separate compilation units, always ReleaseFast. - // Source lives in src/performance/ — the production SSE module. - const clip_sse_obj = b.addObject(.{ - .name = "clip_sse", - .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/clip_sse.zig"), - .target = target, - .optimize = .ReleaseFast, - }), - }); - const cull_sse_obj = b.addObject(.{ - .name = "cull_sse", - .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/cull_sse.zig"), - .target = target, - .optimize = .ReleaseFast, - }), - }); - const entity_sse_obj = b.addObject(.{ - .name = "entity_sse", - .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/entity_sse.zig"), - .target = target, - .optimize = .ReleaseFast, - }), - }); - const bone_sse_target = b.resolveTargetQuery(.{ - .cpu_arch = .x86, - .os_tag = .windows, - .abi = .msvc, - .cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2, .sse3, .sse4_1, .fma, .avx }), - }); - const bone_sse_obj = b.addObject(.{ - .name = "bone_sse", - .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/bone_sse.zig"), - .target = bone_sse_target, - .optimize = .ReleaseFast, - }), - }); // REF uses x87-only target to match original game code structure. // The global target has SSE/SSE2 which generates movss/mulss; // the original at 0x714260 uses pure x87 (FLD/FMUL/FSTP). @@ -126,28 +86,11 @@ pub fn build(b: *std.Build) void { .optimize = .ReleaseFast, }), }); - const silicon_sse_obj = b.addObject(.{ - .name = "silicon_sse", - .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/silicon_sse.zig"), - .target = bone_sse_target, // SSE4.1+FMA+AVX, same as bone_sse - .optimize = .ReleaseFast, - }), - }); - - const particle_sse_obj = b.addObject(.{ - .name = "particle_sse", - .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/particle_sse.zig"), - .target = bone_sse_target, - .optimize = .ReleaseFast, - }), - }); const particle_ref_obj = b.addObject(.{ .name = "particle_sse_ref", .root_module = b.createModule(.{ .root_source_file = b.path("src/transform44/particle_sse_reference.zig"), - .target = bone_sse_target, + .target = target, .optimize = .ReleaseFast, }), }); @@ -178,76 +121,52 @@ pub fn build(b: *std.Build) void { }); libdeflate.root_module.addCSourceFiles(.{ .files = &.{ - "src/performance/libdeflate/lib/deflate_decompress.c", - "src/performance/libdeflate/lib/zlib_decompress.c", - "src/performance/libdeflate/lib/utils.c", - "src/performance/libdeflate/lib/adler32.c", - "src/performance/libdeflate/lib/x86/cpu_features.c", + "src/weirdperformance/libdeflate/lib/deflate_decompress.c", + "src/weirdperformance/libdeflate/lib/zlib_decompress.c", + "src/weirdperformance/libdeflate/lib/utils.c", + "src/weirdperformance/libdeflate/lib/adler32.c", + "src/weirdperformance/libdeflate/lib/x86/cpu_features.c", + "src/weirdperformance/libdeflate/stubs/game_alloc.c", }, .flags = &.{"-DLIBDEFLATE_ASSEMBLER_DOES_NOT_SUPPORT_AVX512VNNI"}, }); - libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate/stubs")); - libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate")); - libdeflate.root_module.addIncludePath(b.path("src/performance/libdeflate/lib")); + libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate/stubs")); + libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate")); + libdeflate.root_module.addIncludePath(b.path("src/weirdperformance/libdeflate/lib")); // Link module-specific object files into a DLL. // Single source of truth for which objects each module needs. // Called for both the main weirdutils build and each variant. const ModuleObjects = struct { - clip_sse: *std.Build.Step.Compile, - cull_sse: *std.Build.Step.Compile, - entity_sse: *std.Build.Step.Compile, - bone_sse: *std.Build.Step.Compile, bone_sse_ref: *std.Build.Step.Compile, math_sse: *std.Build.Step.Compile, - silicon_sse: *std.Build.Step.Compile, - particle_sse: *std.Build.Step.Compile, particle_ref: *std.Build.Step.Compile, libdeflate: *std.Build.Step.Compile, fn linkFor(self: @This(), mod: *std.Build.Module, comptime module_name: []const u8) void { @setEvalBranchQuota(10000); if (comptime std.mem.eql(u8, module_name, "weirdperformance")) { - mod.addObject(self.clip_sse); - mod.addObject(self.cull_sse); - mod.addObject(self.bone_sse); - mod.addObject(self.silicon_sse); - mod.addObject(self.particle_sse); mod.addObjectFile(self.libdeflate.getEmittedBin()); } if (comptime std.mem.eql(u8, module_name, "transform44")) { - mod.addObject(self.clip_sse); - mod.addObject(self.cull_sse); - mod.addObject(self.entity_sse); - mod.addObject(self.bone_sse); mod.addObject(self.bone_sse_ref); - mod.addObject(self.particle_sse); mod.addObject(self.particle_ref); } - if (comptime std.mem.eql(u8, module_name, "silicon")) { - mod.addObject(self.silicon_sse); - } if (comptime std.mem.eql(u8, module_name, "ssemaths")) { mod.addObject(self.math_sse); } } }; const objs = ModuleObjects{ - .clip_sse = clip_sse_obj, - .cull_sse = cull_sse_obj, - .entity_sse = entity_sse_obj, - .bone_sse = bone_sse_obj, .bone_sse_ref = bone_sse_ref_obj, .math_sse = math_sse_obj, - .silicon_sse = silicon_sse_obj, - .particle_sse = particle_sse_obj, .particle_ref = particle_ref_obj, .libdeflate = libdeflate, }; - // Main DLL: link all module objects - inline for (module_list) |mod| { - objs.linkFor(lib.root_module, mod.name); + // Main DLL: link module objects only for enabled modules + inline for (module_list, 0..) |mod, i| { + if (module_enabled[i]) objs.linkFor(lib.root_module, mod.name); } b.installArtifact(lib); @@ -282,7 +201,7 @@ pub fn build(b: *std.Build) void { const bench_silicon_sse = b.addObject(.{ .name = "bench_silicon_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/silicon_sse.zig"), + .root_source_file = b.path("src/weirdperformance/silicon_sse.zig"), .target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .linux, @@ -294,7 +213,7 @@ pub fn build(b: *std.Build) void { const bench_bone_sse = b.addObject(.{ .name = "bench_bone_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/bone_sse.zig"), + .root_source_file = b.path("src/weirdperformance/bone_sse.zig"), .target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .linux, @@ -318,7 +237,7 @@ pub fn build(b: *std.Build) void { const bench_particle_sse = b.addObject(.{ .name = "bench_particle_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/particle_sse.zig"), + .root_source_file = b.path("src/weirdperformance/particle_sse.zig"), .target = b.resolveTargetQuery(.{ .cpu_arch = .x86, .os_tag = .linux, @@ -330,7 +249,7 @@ pub fn build(b: *std.Build) void { const bench_cull_sse = b.addObject(.{ .name = "bench_cull_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/cull_sse.zig"), + .root_source_file = b.path("src/weirdperformance/cull_sse.zig"), .target = bench_target, .optimize = .ReleaseFast, }), @@ -344,7 +263,7 @@ pub fn build(b: *std.Build) void { const bench_entity_sse = b.addObject(.{ .name = "bench_entity_sse", .root_module = b.createModule(.{ - .root_source_file = b.path("src/performance/entity_sse.zig"), + .root_source_file = b.path("src/weirdperformance/entity_sse.zig"), .target = bench_target, .optimize = .ReleaseFast, }), @@ -396,6 +315,7 @@ pub fn build(b: *std.Build) void { build_all_step.dependOn(&noperf_install.step); inline for (module_list) |variant_mod| { + if (!variant_mod.default) continue; @setEvalBranchQuota(10000); const opts = b.addOptions(); inline for (module_list) |m| { diff --git a/src/main.zig b/src/main.zig index 67b320b..8ee8bc5 100644 --- a/src/main.zig +++ b/src/main.zig @@ -43,7 +43,7 @@ const transform44 = if (build_opts.transform44) @import("transform44/transform44 const addonperf = if (build_opts.addonperf) @import("addonperf/addonperf.zig") else struct {}; const ssemaths = if (build_opts.ssemaths) @import("ssemaths/ssemaths.zig") else struct {}; const silicon = if (build_opts.silicon) @import("silicon/silicon.zig") else struct {}; -const weirdperformance = if (build_opts.weirdperformance) @import("performance/weirdperformance.zig") else struct {}; +const weirdperformance = if (build_opts.weirdperformance) @import("weirdperformance/weirdperformance.zig") else struct {}; const module_active = @import("module_active.zig"); @@ -666,6 +666,9 @@ fn engineInitDetour() callconv(hook.cc.stdcall) void { if (build_opts.silicon) { silicon.lateInit(); } + if (build_opts.weirdperformance) { + weirdperformance.lateInit(); + } } // ============================================================================= diff --git a/src/transform44/transform44.zig b/src/transform44/transform44.zig index 327e8ad..488fd4e 100644 --- a/src/transform44/transform44.zig +++ b/src/transform44/transform44.zig @@ -15,37 +15,6 @@ const std = @import("std"); const hook = @import("zhook"); const logging = @import("../logging.zig"); const mod_mutex = @import("../mutex.zig"); -extern fn clipPolygonToSinglePlane(u32, u32, u32) void; -extern fn buildTrianglePlanes(u32, u32, u32, u32, u32) u32; -extern fn rayTriangleIntersection(u32, u32, u32, u32, u32, u32) u32; -extern fn rotateMatrixByAxisAngle(u32, u32, u32, u32) void; -extern fn multiplyMatrix4x4(u32, u32, u32) u32; -extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void; -extern fn calcColorValues_SSE(u32, u32, u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; -extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn renderParticleSprites_REF(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn resetParticleCache() void; -extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn updateEntityAndChunksPositions(u32) callconv(.{ .x86_fastcall = .{} }) void; -extern fn updateEntitiesInBoundsSSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; -extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; -extern fn addToSpatialGridSSE(u32) callconv(.{ .x86_fastcall = .{} }) void; -extern fn findObjectByGUID_Cached(u32, u32) callconv(.{ .x86_stdcall = .{} }) u32; -extern fn setupParticleRendering_SSE(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; -extern fn renderSpriteQuads_SSE(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; -extern var stride_info: [8]u32; // exported from particle_sse.zig -extern var debug_vertex_count: u32; -extern var debug_max_sprites: u32; -extern var debug_fmt_index: u32; -extern var debug_data_ptr: u32; -var stride_dumped: bool = false; - -/// Thiscall wrapper for the SSE implementation. Lives here (baseline SSE2 unit) -/// so LLVM can't inline transformImpl_SSE's alignment into the thiscall frame. -fn transformMatrix4x4_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.{ .x86_thiscall = .{} }) void { - transformImpl_SSE(this, mat1, mat2, mat3, mat4); -} extern fn transformMatrix4x4_REF(u32, u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; extern var bisect_stop_section: u32; @@ -82,12 +51,6 @@ const AB_OTHER_HOOKS = true; var diag_cmp_count: u32 = 0; export var original_trampoline: u32 = 0; // DEBUG: expose trampoline for REF passthrough test -// Teardown guard: set true when CleanupWorldAndEntities fires. -// During teardown, SceneObject data may be partially freed — our SSE code -// must not process it. Falls back to original function which the game -// controls. NOTE: binary patching (instead of hooking) would avoid this -// issue entirely since the patched code IS the original entry point. -var teardown_active: bool = false; // Persistent blit totals per A/B mode — NOT reset each dump period. @@ -325,11 +288,7 @@ fn transformDetour(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callco t44_depth +|= 1; if (t44_depth > prof.t44_max_depth) prof.t44_max_depth = t44_depth; - if (teardown_active) { - transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); - } else { - transformMatrix4x4_SSE(this, mat1, mat2, mat3, mat4); - } + transform_hook.callOriginal(.{ this, mat1, mat2, mat3, mat4 }); t44_depth -|= 1; const elapsed = rdtsc() - start; @@ -392,7 +351,6 @@ const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void; var world_update_hook: hook.Detour(WorldUpdateFn) = .{}; fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void { - resetParticleCache(); const now = rdtsc(); if (last_frame_tsc != 0) { const delta = now - last_frame_tsc; @@ -410,23 +368,6 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void { } } -// ============================================================================= -// Hook: World_HandleLogoutCleanup (0x491180) -// Fires at the START of the logout/disconnect cleanup sequence, BEFORE any -// model data is freed. Sets teardown_active flag so our SSE code falls back -// to the original function during the entire cleanup chain. -// NOTE: binary patching instead of hooking would avoid this issue entirely. -// ============================================================================= - -const TeardownFn = fn () callconv(.{ .x86_stdcall = .{} }) void; -var teardown_hook: hook.Detour(TeardownFn) = .{}; - -fn teardownDetour() callconv(.{ .x86_stdcall = .{} }) void { - teardown_active = true; - teardown_hook.callOriginal(.{}); - teardown_active = false; -} - // ============================================================================= // Hook: RenderTextureQuads (0x76FB00) // __fastcall(ECX=RenderBatch*) — no stack params, RET @@ -897,12 +838,6 @@ var staticcull_hook: hook.Detour(Fn2) = .{}; // ProcessStaticObjectsCulling: fas fn clipDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - clipPolygonToSinglePlane(a, b, c); - prof.clip_cycles +|= rdtsc() - s; - prof.clip_calls +|= 1; - return null; // original is void — EAX not read by callers - } const ret = clip_hook.callOriginal(.{ a, b, c }); prof.clip_cycles +|= rdtsc() - s; prof.clip_calls +|= 1; @@ -948,13 +883,12 @@ fn glyphDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyo prof.glyph_calls +|= 1; return ret; } -fn particleDetour(a: u32, _: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { - // a=ECX(emitter), c=particleData, d=vertexBuffers +fn particleDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); - const result = renderParticleSprites_SSE(a, c, d); + const ret = particle_hook.callOriginal(.{ a, b, c, d }); prof.particle_cycles +|= rdtsc() - s; prof.particle_calls +|= 1; - return @ptrFromInt(result); + return ret; } fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); @@ -965,12 +899,6 @@ fn collisionDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?* } fn entposDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - updateEntityAndChunksPositions(a); - prof.entpos_cycles +|= rdtsc() - s; - prof.entpos_calls +|= 1; - return null; - } const ret = entpos_hook.callOriginal(.{ a, b }); prof.entpos_cycles +|= rdtsc() - s; prof.entpos_calls +|= 1; @@ -1022,12 +950,6 @@ fn spatialDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { fn raytriDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque { asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - const ret = rayTriangleIntersection(a, b, c, d, e, f); - prof.raytri_cycles +|= rdtsc() - s; - prof.raytri_calls +|= 1; - return @ptrFromInt(ret); - } const ret = raytri_hook.callOriginal(.{ a, b, c, d, e, f }); prof.raytri_cycles +|= rdtsc() - s; prof.raytri_calls +|= 1; @@ -1059,12 +981,6 @@ fn setvecDetour(a: u32, b: u32) callconv(hook.cc.fastcall) ?*anyopaque { } fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - const ret = performSpatialCulling(a, c, d); - prof.cull_cycles +|= rdtsc() - s; - prof.cull_calls +|= 1; - return @ptrFromInt(ret); - } const ret = cull_hook.callOriginal(.{ a, b, c, d }); prof.cull_cycles +|= rdtsc() - s; prof.cull_calls +|= 1; @@ -1072,12 +988,6 @@ fn cullDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyop } fn colldetDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - const ret = performCollisionDetectionSSE(a, c, d); - prof.colldet_cycles +|= rdtsc() - s; - prof.colldet_calls +|= 1; - return @ptrFromInt(ret); - } const ret = colldet_hook.callOriginal(.{ a, b, c, d }); prof.colldet_cycles +|= rdtsc() - s; prof.colldet_calls +|= 1; @@ -1216,13 +1126,6 @@ fn bboxchkDetour(a: u32, b: u32, c: u32, d: u32) callconv(hook.cc.fastcall) ?*an fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - // thiscall: a=ECX=matrix, b=EDX=unused, c=angle, d=axis_ptr, e=is_unit - rotateMatrixByAxisAngle(a, c, d, e); - prof.rotmat_cycles +|= rdtsc() - s; - prof.rotmat_calls +|= 1; - return null; // void function, EAX not read by callers - } const ret = rotmat_hook.callOriginal(.{ a, b, c, d, e }); prof.rotmat_cycles +|= rdtsc() - s; prof.rotmat_calls +|= 1; @@ -1231,12 +1134,6 @@ fn rotmatDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcal fn triplaneDetour(a: u32, b: u32, c: u32, d: u32, e: u32) callconv(hook.cc.fastcall) ?*anyopaque { asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - const ret = buildTrianglePlanes(a, b, c, d, e); - prof.triplane_cycles +|= rdtsc() - s; - prof.triplane_calls +|= 1; - return @ptrFromInt(ret); - } const ret = triplane_hook.callOriginal(.{ a, b, c, d, e }); prof.triplane_cycles +|= rdtsc() - s; prof.triplane_calls +|= 1; @@ -1252,13 +1149,6 @@ fn partsetupDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaqu fn matmulDetour(a: u32, b: u32, c: u32) callconv(hook.cc.fastcall) ?*anyopaque { asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true }); const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - // fastcall: a=ECX=result, b=EDX=left, c=right - const ret = multiplyMatrix4x4(a, b, c); - prof.matmul_cycles +|= rdtsc() - s; - prof.matmul_calls +|= 1; - return @ptrFromInt(ret); - } const ret = matmul_hook.callOriginal(.{ a, b, c }); prof.matmul_cycles +|= rdtsc() - s; prof.matmul_calls +|= 1; @@ -1294,12 +1184,6 @@ fn rendersphDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32, g: u32, h: u3 } fn raytriIntDetour(a: u32, b: u32, c: u32, d: u32, e: u32, f: u32) callconv(hook.cc.fastcall) ?*anyopaque { const s = rdtsc(); - if (AB_OTHER_HOOKS and ab_use_custom) { - const ret = rayTriIntersectIndexedInt(a, b, c, d, e, f); - prof.raytri_int_cycles +|= rdtsc() - s; - prof.raytri_int_calls +|= 1; - return @ptrFromInt(@as(u32, ret)); - } const ret = raytri_int_hook.callOriginal(.{ a, b, c, d, e, f }); prof.raytri_int_cycles +|= rdtsc() - s; prof.raytri_int_calls +|= 1; @@ -1606,24 +1490,6 @@ fn dumpStats() void { guid_cache_evictions = 0; } - // Dump particle VB stride info (once) - if (debug_vertex_count != 0) { - log.fmt(" partsetup_debug: verts={d} maxSprites={d} fmt={d} dataPtr=0x{x}\n", .{ - debug_vertex_count, debug_max_sprites, debug_fmt_index, debug_data_ptr, - }); - debug_vertex_count = 0; - } - - if (stride_info[0] != 0 and !stride_dumped) { - stride_dumped = true; - log.fmt(" vb_strides: pos={d} norm={d} color={d} tc={d}\n", .{ - stride_info[0], stride_info[1], stride_info[2], stride_info[3], - }); - log.fmt(" vb_bases: pos=0x{x} norm=0x{x} color=0x{x} tc=0x{x}\n", .{ - stride_info[4], stride_info[5], stride_info[6], stride_info[7], - }); - } - // Flip A/B mode ab_use_custom = !ab_use_custom; diag_cmp_count = 0; @@ -1687,7 +1553,6 @@ pub fn installHooks() void { _ = transform_hook.attach(0x714260, &transformDetour); original_trampoline = @intCast(transform_hook.inner.trampoline); } - _ = teardown_hook.attach(0x491180, &teardownDetour); _ = render_frame_hook.attach(0x707680, &renderFrameDetour); _ = exec_render_pass_hook.attach(0x708900, &execRenderPassDetour); _ = world_update_hook.attach(0x482EA0, &worldUpdateDetour); @@ -1776,7 +1641,6 @@ pub fn removeHooks() void { render_frame_hook.detach(); exec_render_pass_hook.detach(); world_update_hook.detach(); - teardown_hook.detach(); render_quads_hook.detach(); movement_hook.detach(); interp_kf_hook.detach(); diff --git a/src/performance/INFLATE_RESEARCH.md b/src/weirdperformance/INFLATE_RESEARCH.md similarity index 100% rename from src/performance/INFLATE_RESEARCH.md rename to src/weirdperformance/INFLATE_RESEARCH.md diff --git a/src/performance/bone_sse.zig b/src/weirdperformance/bone_sse.zig similarity index 99% rename from src/performance/bone_sse.zig rename to src/weirdperformance/bone_sse.zig index fccf4f8..369352d 100644 --- a/src/performance/bone_sse.zig +++ b/src/weirdperformance/bone_sse.zig @@ -1101,7 +1101,7 @@ fn calcScaledInverse(this_mat: u32, out: u32, scale: f32) void { // mat3(offset_vec3*), mat4(scale_float_bits) // ============================================================================= -export fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void { +pub fn transformImpl_SSE(this: u32, mat1: u32, mat2: u32, mat3: u32, mat4: u32) callconv(.c) void { @setEvalBranchQuota(50000); // ========================================================================= diff --git a/src/performance/clip_sse.zig b/src/weirdperformance/clip_sse.zig similarity index 97% rename from src/performance/clip_sse.zig rename to src/weirdperformance/clip_sse.zig index 45285f4..e873647 100644 --- a/src/performance/clip_sse.zig +++ b/src/weirdperformance/clip_sse.zig @@ -9,7 +9,7 @@ const V4 = @Vector(4, f32); const CLIP_EPSILON: f32 = @bitCast(@as(u32, 0x3ab60b61)); // +0.00139, global at 0x80dfec const MOVEMENT_EPSILON: f32 = @bitCast(@as(u32, 0x35800000)); // 9.54e-7, global at 0x8026bc -export fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void { +pub fn clipPolygonToSinglePlane(plane_addr: u32, poly_addr: u32, attrib_bits: u32) void { const plane: [*]const f32 = @ptrFromInt(plane_addr); const poly: [*]f32 = @ptrFromInt(poly_addr); const new_attrib: f32 = @bitCast(attrib_bits); @@ -116,7 +116,7 @@ inline fn lerp(poly: [*]f32, attribs: [*]f32, out: *u32, p: [*]const f32, c: [*] // planes[3]: cap plane (from plane_normal, offset by offset_vector) // ============================================================================= -export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 { +pub fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u32, offset_addr: u32, out_addr: u32) u32 { const verts: [*]const f32 = @ptrFromInt(verts_addr); const indices: [*]const u8 = @ptrFromInt(indices_addr); const plane_normal: [*]const f32 = @ptrFromInt(normal_addr); @@ -191,7 +191,7 @@ export fn buildTrianglePlanes(verts_addr: u32, indices_addr: u32, normal_addr: u // Returns: 1 = hit, 0 = miss // ============================================================================= -export fn rayTriangleIntersection( +pub fn rayTriangleIntersection( ray_addr: u32, verts_addr: u32, indices_addr: u32, @@ -267,7 +267,7 @@ export fn rayTriangleIntersection( // Returns result pointer (EAX = result_addr). // ============================================================================= -export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 { +pub fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u32 { const result: [*]f32 = @ptrFromInt(result_addr); const left: [*]const f32 = @ptrFromInt(left_addr); const right: [*]const f32 = @ptrFromInt(right_addr); @@ -296,7 +296,7 @@ export fn multiplyMatrix4x4(result_addr: u32, left_addr: u32, right_addr: u32) u // __thiscall(ECX=matrix, stack: angle, axis_ptr, is_unit_flag) // ============================================================================= -export fn rotateMatrixByAxisAngle( +pub fn rotateMatrixByAxisAngle( matrix_addr: u32, angle_bits: u32, axis_addr: u32, diff --git a/src/performance/cull_sse.zig b/src/weirdperformance/cull_sse.zig similarity index 97% rename from src/performance/cull_sse.zig rename to src/weirdperformance/cull_sse.zig index c0f9b69..33e03cc 100644 --- a/src/performance/cull_sse.zig +++ b/src/weirdperformance/cull_sse.zig @@ -15,7 +15,7 @@ /// Compute outcodes for `count` vertices at `verts_ptr` (stride 12 bytes = 3 floats) /// against AABB at `bounds_ptr` (6 floats: minX, minY, minZ, maxX, maxY, maxZ). /// Writes results to `out_ptr` (1 byte per vertex). -export fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void { +pub fn benchComputeOutcodes(verts_ptr: u32, bounds_ptr: u32, out_ptr: u32, count: u32) void { if (count == 0) return; const v_min: V4 = .{ readF32(bounds_ptr), readF32(bounds_ptr + 4), readF32(bounds_ptr + 8), 0 }; const v_max: V4 = .{ readF32(bounds_ptr + 12), readF32(bounds_ptr + 16), readF32(bounds_ptr + 20), 0 }; @@ -54,7 +54,7 @@ var guid_cache: [GUID_CACHE_SIZE]GuidCacheEntry = [_]GuidCacheEntry{.{}} ** GUID const origFindObjectByGUID = @as(*const fn (u32, u32) callconv(.{ .x86_stdcall = .{} }) u32, @ptrFromInt(0x464890)); -export fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 { +pub fn findObjectByGUID_Cached(guid_lo: u32, guid_hi: u32) callconv(.{ .x86_stdcall = .{} }) u32 { const hash = (guid_lo ^ (guid_hi *% 0x9E3779B9)) & GUID_CACHE_MASK; const entry = &guid_cache[hash]; @@ -92,7 +92,7 @@ const g_grid_offset: *const f32 = @ptrFromInt(0x86861C); // Dot product coefficients at 0xC7CFB8..C7CFC4 (same as entpos view coeffs but different address) const g_spatial_coeffs: u32 = 0xC7CFB8; -export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void { +pub fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void { // Dot product: coeff_a * obj[0x5C] + coeff_b * obj[0x60] + coeff_c * obj[0x64] + coeff_d const depth = readF32(g_spatial_coeffs) * readF32(obj + 0x5C) + readF32(g_spatial_coeffs + 4) * readF32(obj + 0x60) + @@ -144,7 +144,7 @@ export fn addToSpatialGridSSE(obj: u32) callconv(.{ .x86_fastcall = .{} }) void // RET 0x10 // ============================================================================= -export fn rayTriIntersectIndexedInt( +pub fn rayTriIntersectIndexedInt( ray_ptr: u32, vert_pool: u32, indices_ptr: u32, @@ -279,7 +279,7 @@ fn computeAllOutcodes( /// Finds mesh data via hash, computes vertex outcodes against AABB from this+0x10, /// then iterates triangles: filters by visibility mask, trivial-rejects by outcode AND, /// adds survivors to global visible/render lists. -export fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 { +pub fn performSpatialCulling(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 { if (g_guard.* == 0) return 0; const hash_table = g_guard.*; @@ -377,7 +377,7 @@ inline fn dot3(a: V4, b: V4) f32 { /// Fully inlined SSE rewrite. No external calls except FindOrCreateHashEntry. /// Moller-Trumbore ray-triangle intersection is inlined with SSE cross/dot, /// eliminating 4 SetVector3 calls and the ray_tri function call per triangle. -export fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 { +pub fn performCollisionDetectionSSE(this: u32, key_data: u32, key_size: u32) callconv(.{ .x86_thiscall = .{} }) u32 { if (g_guard.* == 0) return 0; const hash_table = g_guard.*; diff --git a/src/performance/entity_sse.zig b/src/weirdperformance/entity_sse.zig similarity index 100% rename from src/performance/entity_sse.zig rename to src/weirdperformance/entity_sse.zig diff --git a/src/performance/filecache.zig b/src/weirdperformance/filecache.zig similarity index 92% rename from src/performance/filecache.zig rename to src/weirdperformance/filecache.zig index 1ca9950..d55d334 100644 --- a/src/performance/filecache.zig +++ b/src/weirdperformance/filecache.zig @@ -52,7 +52,12 @@ const CacheSet = struct { lru: u8 = 0, }; -var cache: [NUM_SETS]CacheSet = @splat(CacheSet{}); +const WINAPI = @import("std").builtin.CallingConvention.winapi; +extern "kernel32" fn GetProcessHeap() callconv(WINAPI) ?*anyopaque; +extern "kernel32" fn HeapAlloc(hHeap: ?*anyopaque, dwFlags: u32, dwBytes: usize) callconv(WINAPI) ?[*]u8; +extern "kernel32" fn HeapFree(hHeap: ?*anyopaque, dwFlags: u32, lpMem: *anyopaque) callconv(WINAPI) i32; + +var cache: ?[*]CacheSet = null; var cache_entries: u32 = 0; var cache_hits: u64 = 0; var cache_negative_hits: u64 = 0; @@ -66,8 +71,9 @@ var cache_miss_p2_archive: u64 = 0; /// Lookup: hash picks set, check both ways for name match. pub fn archiveCacheLookup(h: u32, path: [*:0]const u8) ?ArchiveCacheEntry { + const c = cache orelse return null; const set_idx = h & (NUM_SETS - 1); - const set = &cache[set_idx]; + const set = &c[set_idx]; const span = std.mem.span(path); for (0..WAYS) |w| { @@ -93,11 +99,12 @@ pub fn computeBlockEntry(archive: u32, index: u32) u32 { } pub fn archiveCacheInsert(h: u32, path: [*:0]const u8, outer: u32, inner: u32, block: u32, negative: bool) void { + const c = cache orelse return; const span = std.mem.span(path); if (span.len > CACHE_NAME_LEN) return; const set_idx = h & (NUM_SETS - 1); - const set = &cache[set_idx]; + const set = &c[set_idx]; var target: u8 = set.lru; for (0..WAYS) |w| { @@ -135,8 +142,9 @@ pub fn recordMissP2() void { cache_miss_p2 +|= 1; } pub fn recordMissP2Archive() void { cache_miss_p2_archive +|= 1; } pub fn getSlotOccupant(h: u32) ?[]const u8 { + const c = cache orelse return null; const set_idx = h & (NUM_SETS - 1); - const set = &cache[set_idx]; + const set = &c[set_idx]; for (0..WAYS) |w| { if (set.entries[w].name_len != 0) return set.entries[w].name[0..set.entries[w].name_len]; @@ -246,11 +254,19 @@ fn fileFindDetour( } pub fn install() bool { + const size = NUM_SETS * @sizeOf(CacheSet); + const heap = GetProcessHeap() orelse return false; + const ptr = HeapAlloc(heap, 0x00000008, size) orelse return false; // HEAP_ZERO_MEMORY + cache = @alignCast(@ptrCast(ptr)); return file_find_hook.attach(0x6549a0, &fileFindDetour) == .ok; } pub fn remove() void { file_find_hook.detach(); + if (cache) |c| { + if (GetProcessHeap()) |heap| _ = HeapFree(heap, 0, @ptrCast(c)); + cache = null; + } } // ============================================================================= diff --git a/src/performance/inflate_hook.zig b/src/weirdperformance/inflate_hook.zig similarity index 92% rename from src/performance/inflate_hook.zig rename to src/weirdperformance/inflate_hook.zig index 209543a..437437d 100644 --- a/src/performance/inflate_hook.zig +++ b/src/weirdperformance/inflate_hook.zig @@ -13,7 +13,6 @@ const hook_lib = @import("zhook"); const logging = @import("../logging.zig"); -const std = @import("std"); // libdeflate C API extern fn libdeflate_alloc_decompressor() ?*anyopaque; @@ -24,15 +23,12 @@ var lib_available: bool = false; var log: logging.Logger = .{}; // --- Thread-local decompressor --- -// Each thread lazily allocates its own decompressor on first use. -// OS-managed TLS via Zig's threadlocal -- works correctly on both -// native Windows and Wine without manual FS segment access. -threadlocal var tls_decomp: ?*anyopaque = null; +threadlocal var tls_decompressor: ?*anyopaque = null; fn getTlsDecompressor() ?*anyopaque { - if (tls_decomp) |d| return d; + if (tls_decompressor) |d| return d; const d = libdeflate_alloc_decompressor() orelse return null; - tls_decomp = d; + tls_decompressor = d; return d; } diff --git a/src/performance/libdeflate/common_defs.h b/src/weirdperformance/libdeflate/common_defs.h similarity index 100% rename from src/performance/libdeflate/common_defs.h rename to src/weirdperformance/libdeflate/common_defs.h diff --git a/src/performance/libdeflate/lib/adler32.c b/src/weirdperformance/libdeflate/lib/adler32.c similarity index 100% rename from src/performance/libdeflate/lib/adler32.c rename to src/weirdperformance/libdeflate/lib/adler32.c diff --git a/src/performance/libdeflate/lib/cpu_features_common.h b/src/weirdperformance/libdeflate/lib/cpu_features_common.h similarity index 100% rename from src/performance/libdeflate/lib/cpu_features_common.h rename to src/weirdperformance/libdeflate/lib/cpu_features_common.h diff --git a/src/performance/libdeflate/lib/decompress_template.h b/src/weirdperformance/libdeflate/lib/decompress_template.h similarity index 100% rename from src/performance/libdeflate/lib/decompress_template.h rename to src/weirdperformance/libdeflate/lib/decompress_template.h diff --git a/src/performance/libdeflate/lib/deflate_constants.h b/src/weirdperformance/libdeflate/lib/deflate_constants.h similarity index 100% rename from src/performance/libdeflate/lib/deflate_constants.h rename to src/weirdperformance/libdeflate/lib/deflate_constants.h diff --git a/src/performance/libdeflate/lib/deflate_decompress.c b/src/weirdperformance/libdeflate/lib/deflate_decompress.c similarity index 100% rename from src/performance/libdeflate/lib/deflate_decompress.c rename to src/weirdperformance/libdeflate/lib/deflate_decompress.c diff --git a/src/performance/libdeflate/lib/lib_common.h b/src/weirdperformance/libdeflate/lib/lib_common.h similarity index 100% rename from src/performance/libdeflate/lib/lib_common.h rename to src/weirdperformance/libdeflate/lib/lib_common.h diff --git a/src/performance/libdeflate/lib/utils.c b/src/weirdperformance/libdeflate/lib/utils.c similarity index 100% rename from src/performance/libdeflate/lib/utils.c rename to src/weirdperformance/libdeflate/lib/utils.c diff --git a/src/performance/libdeflate/lib/x86/adler32_impl.h b/src/weirdperformance/libdeflate/lib/x86/adler32_impl.h similarity index 100% rename from src/performance/libdeflate/lib/x86/adler32_impl.h rename to src/weirdperformance/libdeflate/lib/x86/adler32_impl.h diff --git a/src/performance/libdeflate/lib/x86/adler32_template.h b/src/weirdperformance/libdeflate/lib/x86/adler32_template.h similarity index 100% rename from src/performance/libdeflate/lib/x86/adler32_template.h rename to src/weirdperformance/libdeflate/lib/x86/adler32_template.h diff --git a/src/performance/libdeflate/lib/x86/cpu_features.c b/src/weirdperformance/libdeflate/lib/x86/cpu_features.c similarity index 100% rename from src/performance/libdeflate/lib/x86/cpu_features.c rename to src/weirdperformance/libdeflate/lib/x86/cpu_features.c diff --git a/src/performance/libdeflate/lib/x86/cpu_features.h b/src/weirdperformance/libdeflate/lib/x86/cpu_features.h similarity index 100% rename from src/performance/libdeflate/lib/x86/cpu_features.h rename to src/weirdperformance/libdeflate/lib/x86/cpu_features.h diff --git a/src/performance/libdeflate/lib/x86/crc32_impl.h b/src/weirdperformance/libdeflate/lib/x86/crc32_impl.h similarity index 100% rename from src/performance/libdeflate/lib/x86/crc32_impl.h rename to src/weirdperformance/libdeflate/lib/x86/crc32_impl.h diff --git a/src/performance/libdeflate/lib/x86/crc32_pclmul_template.h b/src/weirdperformance/libdeflate/lib/x86/crc32_pclmul_template.h similarity index 100% rename from src/performance/libdeflate/lib/x86/crc32_pclmul_template.h rename to src/weirdperformance/libdeflate/lib/x86/crc32_pclmul_template.h diff --git a/src/performance/libdeflate/lib/x86/decompress_impl.h b/src/weirdperformance/libdeflate/lib/x86/decompress_impl.h similarity index 100% rename from src/performance/libdeflate/lib/x86/decompress_impl.h rename to src/weirdperformance/libdeflate/lib/x86/decompress_impl.h diff --git a/src/performance/libdeflate/lib/x86/matchfinder_impl.h b/src/weirdperformance/libdeflate/lib/x86/matchfinder_impl.h similarity index 100% rename from src/performance/libdeflate/lib/x86/matchfinder_impl.h rename to src/weirdperformance/libdeflate/lib/x86/matchfinder_impl.h diff --git a/src/performance/libdeflate/lib/zlib_constants.h b/src/weirdperformance/libdeflate/lib/zlib_constants.h similarity index 100% rename from src/performance/libdeflate/lib/zlib_constants.h rename to src/weirdperformance/libdeflate/lib/zlib_constants.h diff --git a/src/performance/libdeflate/lib/zlib_decompress.c b/src/weirdperformance/libdeflate/lib/zlib_decompress.c similarity index 100% rename from src/performance/libdeflate/lib/zlib_decompress.c rename to src/weirdperformance/libdeflate/lib/zlib_decompress.c diff --git a/src/performance/libdeflate/libdeflate.h b/src/weirdperformance/libdeflate/libdeflate.h similarity index 100% rename from src/performance/libdeflate/libdeflate.h rename to src/weirdperformance/libdeflate/libdeflate.h diff --git a/src/weirdperformance/libdeflate/safe_decompress.c b/src/weirdperformance/libdeflate/safe_decompress.c new file mode 100644 index 0000000..7a28aab --- /dev/null +++ b/src/weirdperformance/libdeflate/safe_decompress.c @@ -0,0 +1,25 @@ +/* Wraps libdeflate_deflate_decompress with SEH to catch access violations. + * libdeflate's fast path on 32-bit can crash on malformed input despite + * SAFETY_CHECKs. This wrapper catches the crash and returns an error code. */ + +#include "libdeflate.h" + +#ifdef _WIN32 +#include + +__declspec(dllexport) int safe_deflate_decompress( + struct libdeflate_decompressor *d, + const void *in, size_t in_nbytes, + void *out, size_t out_nbytes_avail, + size_t *actual_out_nbytes_ret) +{ + int result; + __try { + result = libdeflate_deflate_decompress(d, in, in_nbytes, + out, out_nbytes_avail, actual_out_nbytes_ret); + } __except(EXCEPTION_EXECUTE_HANDLER) { + result = 1; /* LIBDEFLATE_BAD_DATA */ + } + return result; +} +#endif diff --git a/src/weirdperformance/libdeflate/stubs/game_alloc.c b/src/weirdperformance/libdeflate/stubs/game_alloc.c new file mode 100644 index 0000000..fae108b --- /dev/null +++ b/src/weirdperformance/libdeflate/stubs/game_alloc.c @@ -0,0 +1,15 @@ +/* malloc/free stubs for libdeflate using the Windows process heap. */ + +#include + +__declspec(dllimport) void *__stdcall GetProcessHeap(void); +__declspec(dllimport) void *__stdcall HeapAlloc(void *hHeap, unsigned long dwFlags, size_t dwBytes); +__declspec(dllimport) int __stdcall HeapFree(void *hHeap, unsigned long dwFlags, void *lpMem); + +void *malloc(size_t size) { + return HeapAlloc(GetProcessHeap(), 0, size); +} + +void free(void *ptr) { + if (ptr) HeapFree(GetProcessHeap(), 0, ptr); +} diff --git a/src/weirdperformance/libdeflate/stubs/malloc.h b/src/weirdperformance/libdeflate/stubs/malloc.h new file mode 100644 index 0000000..c32e4fb --- /dev/null +++ b/src/weirdperformance/libdeflate/stubs/malloc.h @@ -0,0 +1,4 @@ +#pragma once +#include +extern void *malloc(size_t size); +extern void free(void *ptr); diff --git a/src/performance/libdeflate/stubs/setjmp.h b/src/weirdperformance/libdeflate/stubs/setjmp.h similarity index 100% rename from src/performance/libdeflate/stubs/setjmp.h rename to src/weirdperformance/libdeflate/stubs/setjmp.h diff --git a/src/weirdperformance/libdeflate/stubs/stdio.h b/src/weirdperformance/libdeflate/stubs/stdio.h new file mode 100644 index 0000000..945333a --- /dev/null +++ b/src/weirdperformance/libdeflate/stubs/stdio.h @@ -0,0 +1,2 @@ +#pragma once +/* libdeflate only uses stdio.h in debug paths — stub it out */ diff --git a/src/weirdperformance/libdeflate/stubs/stdlib.h b/src/weirdperformance/libdeflate/stubs/stdlib.h new file mode 100644 index 0000000..c32e4fb --- /dev/null +++ b/src/weirdperformance/libdeflate/stubs/stdlib.h @@ -0,0 +1,4 @@ +#pragma once +#include +extern void *malloc(size_t size); +extern void free(void *ptr); diff --git a/src/weirdperformance/libdeflate/stubs/string.h b/src/weirdperformance/libdeflate/stubs/string.h new file mode 100644 index 0000000..d7e2e7a --- /dev/null +++ b/src/weirdperformance/libdeflate/stubs/string.h @@ -0,0 +1,6 @@ +#pragma once +#include +void *memcpy(void *dest, const void *src, size_t n); +void *memmove(void *dest, const void *src, size_t n); +void *memset(void *s, int c, size_t n); +int memcmp(const void *s1, const void *s2, size_t n); diff --git a/src/performance/particle_sse.zig b/src/weirdperformance/particle_sse.zig similarity index 97% rename from src/performance/particle_sse.zig rename to src/weirdperformance/particle_sse.zig index 54fc536..d4deee6 100644 --- a/src/performance/particle_sse.zig +++ b/src/weirdperformance/particle_sse.zig @@ -168,7 +168,6 @@ const VBState = struct { light: [3]u32, fn load(vb: u32) VBState { - logStrides(vb); return .{ .pos = ru32(vb + VB.pos), .normal = ru32(vb + VB.normal), @@ -260,37 +259,12 @@ inline fn emitVertex(vb: u32, px: f32, py: f32, pz: f32, color: u32, tu: f32, tv // Reset each frame via resetParticleCache() called from the frame hook. var cached_render_state: u32 = 0; -var stride_logged: bool = false; -var debug_logged: bool = false; -export var debug_vertex_count: u32 = 0; -export var debug_max_sprites: u32 = 0; -export var debug_fmt_index: u32 = 0; -export var debug_data_ptr: u32 = 0; /// Reset per-frame caches. Call from OnWorldUpdate or executeSceneRenderPass hook. -export fn resetParticleCache() void { +pub fn resetParticleCache() void { cached_render_state = 0; } -/// Log VB strides once for analysis. Called from first VBState.load. -fn logStrides(vb: u32) void { - if (stride_logged) return; - stride_logged = true; - // Write to a known memory location that the profiler can dump, or just use - // the debug console. For now, store in a global we can read. - stride_info = .{ - ru32(vb + VB.pos_stride), - ru32(vb + VB.normal_stride), - ru32(vb + VB.color_stride), - ru32(vb + VB.texcoord_stride), - ru32(vb + VB.pos), - ru32(vb + VB.normal), - ru32(vb + VB.color), - ru32(vb + VB.texcoord), - }; -} - -export var stride_info: [8]u32 = .{0} ** 8; // ============================================================================= // RenderParticleSprites (0x7B2A50) @@ -299,7 +273,7 @@ export var stride_info: [8]u32 = .{0} ** 8; // // Faithful recreation from assembly + Ghidra decompilation. // ============================================================================= -export fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 { +pub fn renderParticleSprites_SSE(emitter: u32, particle_data: u32, vertex_buffers: u32) callconv(TC) u32 { const pd = particle_data; // particleData pointer (float*) const vb = vertex_buffers; // vertexBuffers pointer (float**) @@ -762,7 +736,7 @@ const SG = struct { // Faithful recreation from Ghidra decompilation + assembly. // All game function calls preserved, matrix math inlined with V4. // ============================================================================= -export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void { +pub fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC) void { // ========================================================================= // Section 1: Identity matrices for render state // Optimization: use static identity instead of rebuilding on stack each call. @@ -999,15 +973,6 @@ export fn setupParticleRendering_SSE(emitter: u32, view_matrix: u32) callconv(TC // ========================================================================= gameRenderSorted(emitter, @intFromPtr(&vb_ptrs)); - // DEBUG: log vertex count produced - if (!debug_logged and vb_ptrs[8] > 0) { - debug_logged = true; - debug_vertex_count = vb_ptrs[8]; - debug_max_sprites = max_sprites; - debug_fmt_index = fmt_index; - debug_data_ptr = data_ptr; - } - gameUnlockVB(vb_ptr, 0); gameDrawPrim(vb_ptr, fmt_index); @@ -1095,7 +1060,7 @@ inline fn displayModeOffset(sprite_type: u32, count: u32) u32 { return divided -% ru32(DISPLAY_MODE_OFFSET_TABLE + sprite_type * 4); } -export fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void { +pub fn renderSpriteQuads_SSE(this: u32, sprite_data: u32, sprite_count: u32, render_mode: u32) callconv(TC) void { // Early out: this+0xF2C == 0 if (ru32(this + 0xF2C) == 0) return; diff --git a/src/performance/silicon_sse.zig b/src/weirdperformance/silicon_sse.zig similarity index 93% rename from src/performance/silicon_sse.zig rename to src/weirdperformance/silicon_sse.zig index 89bc7ca..478df51 100644 --- a/src/performance/silicon_sse.zig +++ b/src/weirdperformance/silicon_sse.zig @@ -54,7 +54,7 @@ inline fn cvtss2si(x: f32) i32 { // --- 0x4549C0: normalizeVec3 (137K/7.5s) --- // Naked thiscall: ECX=vec, [ESP+4]=length_bits. RET 4. Original: 38 bytes. // rcpss + NR for fast reciprocal, then 3 multiplies. -export fn si_normalizeVec3() callconv(.naked) void { +pub fn si_normalizeVec3() callconv(.naked) void { asm volatile ( // xmm0 = 1.0 / length (via rcpss + Newton-Raphson) \\vmovss 4(%%esp), %%xmm0 @@ -77,7 +77,7 @@ export fn si_normalizeVec3() callconv(.naked) void { // out = A * B (3x4 layout: 3x3 rotation + 3 translation) // Layout: [r0c0 r0c1 r0c2 | r1c0 r1c1 r1c2 | r2c0 r2c1 r2c2 | tx ty tz] // V4 per row: broadcast b[row*3+k], multiply with a's columns, accumulate. -export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 { +pub fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 { const dst: [*]f32 = @ptrFromInt(out); const aa: [*]const f32 = @ptrFromInt(a_ptr); const b: [*]const f32 = @ptrFromInt(b_ptr); @@ -117,7 +117,7 @@ export fn si_mulMat3x4(out: u32, a_ptr: u32, b_ptr: u32) callconv(FC) u32 { // --- 0x7BDDB0: rotateMatByQuat --- // builds rotation matrix from quaternion, multiplies with existing 4x4 matrix // Uses V4 for the matrix multiply (same pattern as bone_sse) -export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 { +pub fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 { const q: [*]const f32 = @ptrFromInt(quat); const x = q[0]; const y = q[1]; const z = q[2]; const w = q[3]; const x2 = x + x; const y2 = y + y; const z2 = z + z; @@ -146,7 +146,7 @@ export fn si_rotateMatByQuat(mat: u32, quat: u32) callconv(TC) u32 { // --- 0x7BB860: createRotMat3x4 --- // Rodrigues rotation matrix, 3x4 layout. Uses @mulAdd for all 9 entries. -export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 { +pub fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normalized: u32) callconv(FC) u32 { const m: [*]f32 = @ptrFromInt(out); const ax: [*]const f32 = @ptrFromInt(axis_ptr); var x = ax[0]; var y = ax[1]; var z = ax[2]; @@ -168,7 +168,7 @@ export fn si_createRotMat3x4(out: u32, axis_ptr: u32, angle_bits: u32, is_normal // --- 0x6329E0: distanceToPlane (525K/7.5s) --- // __fastcall(ECX=point, EDX=plane, stack=direction), returns f64 via ST(0), RET 0x4. -export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 { +pub fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC) f64 { const p: [*]const f32 = @ptrFromInt(point); const pl: [*]const f32 = @ptrFromInt(plane); const dir: [*]const f32 = @ptrFromInt(direction); @@ -183,7 +183,7 @@ export fn si_distanceToPlane(point: u32, plane: u32, direction: u32) callconv(FC // Tests point against 6 frustum planes, produces 6-bit bitmask. // Scalar @mulAdd dot4 per plane — the FMA chain has best throughput for this pattern. // Tried: V4 batch 4 planes (gather kills it), V4 hsum (shuffle overhead kills it). -export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 { +pub fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) callconv(TC) u32 { const mask: *u32 = @ptrFromInt(out_mask); const pt = loadV3_1(point); var bits: u32 = 0; @@ -199,7 +199,7 @@ export fn si_classifyPointFrustum(planes_ptr: u32, point: u32, out_mask: u32) ca // --- 0x6DC5A0: checkBoxLineIntersect (2.7M/7.5s) --- // Slab AABB test. Branchless min/max for t0/t1 swap and tmin/tmax accumulation. -export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 { +pub fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) callconv(FC) u32 { const bmin: [*]const f32 = @ptrFromInt(box_ptr); const bmax: [*]const f32 = @ptrFromInt(box_ptr + 0xC); const start: [*]const f32 = @ptrFromInt(line_start); @@ -225,7 +225,7 @@ export fn si_checkBoxLineIntersect(box_ptr: u32, line_start: u32, line_end: u32) // --- 0x6869C0: testOBBFrustum --- // Tests OBB against 6 frustum planes. Uses V4 for corner transform and plane test. -export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 { +pub fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ptr: u32) callconv(TC) u32 { const aabb: [*]const f32 = @ptrFromInt(aabb_ptr); const rot: [*]const f32 = @ptrFromInt(rot_ptr); const t: [*]const f32 = @ptrFromInt(trans_ptr); @@ -285,7 +285,7 @@ export fn si_testOBBFrustum(planes_ptr: u32, aabb_ptr: u32, rot_ptr: u32, trans_ // --- 0x686B80: testSphereFrustum (375K/7.5s) --- // Zig thiscall: naked asm tested at 10cy (vhaddps slow), Zig dot4v at 8cy. -export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 { +pub fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 { const s: [*]const f32 = @ptrFromInt(sphere); const center = V4{ s[0], s[1], s[2], 1.0 }; const r = s[3]; @@ -299,7 +299,7 @@ export fn si_testSphereFrustum(planes_ptr: u32, sphere: u32) callconv(TC) u32 { // --- 0x7C0570: quatSlerp --- // V4 for final blend, @mulAdd for dot product -export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 { +pub fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(FC) u32 { const dst: [*]f32 = @ptrFromInt(out); const av = loadV4(a_ptr); const bv = loadV4(b_ptr); @@ -326,7 +326,7 @@ export fn si_quatSlerp(out: u32, a_ptr: u32, t_bits: u32, b_ptr: u32) callconv(F // --- 0x699330: isPointInsideBounds (1.7M/7.5s) --- // __fastcall(ECX=a, EDX=b), returns u32. -export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 { +pub fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 { const va: [*]const f32 = @ptrFromInt(a); const vb: [*]const f32 = @ptrFromInt(b); if (vb[0] <= va[0] and vb[1] <= va[1] and vb[2] <= va[2]) return 1; @@ -334,7 +334,7 @@ export fn si_isPointInsideBounds(a: u32, b: u32) callconv(FC) u32 { } // --- 0x749280: calculateSinCos --- -export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void { +pub fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callconv(SC) void { const angle: f32 = @bitCast(angle_bits); const sp: *f32 = @ptrFromInt(out_sin); const cp: *f32 = @ptrFromInt(out_cos); @@ -343,7 +343,7 @@ export fn si_calculateSinCos(angle_bits: u32, out_sin: u32, out_cos: u32) callco } // --- 0x7BE5B0: createZRotMat3x3 --- -export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 { +pub fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 { const m: [*]f32 = @ptrFromInt(out); const angle: f32 = @bitCast(angle_bits); const c = @cos(angle); const s = @sin(angle); @@ -356,7 +356,7 @@ export fn si_createZRotMat3x3(out: u32, angle_bits: u32) callconv(TC) u32 { // --- 0x7BCEF0: transposeMat4x4 --- // Naked thiscall: ECX=src, [ESP+4]=dst. RET 4. Original: 156 bytes. // SSE unpacklo/unpackhi transpose: 4 loads + 4 shuffles + 4 stores. -export fn si_transposeMat4x4() callconv(.naked) void { +pub fn si_transposeMat4x4() callconv(.naked) void { asm volatile ( \\mov 4(%%esp), %%eax // Load 4 rows from src (ECX) @@ -387,7 +387,7 @@ export fn si_transposeMat4x4() callconv(.naked) void { // --- 0x7BB420: mulMat3x4InPlace --- // this = this * matB. V4 columns loaded upfront, write directly back (no tmp needed). -export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 { +pub fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 { const a: [*]f32 = @ptrFromInt(mat_a); const b: [*]const f32 = @ptrFromInt(mat_b); @@ -419,7 +419,7 @@ export fn si_mulMat3x4InPlace(mat_a: u32, mat_b: u32) callconv(TC) u32 { // --- 0x6720F0: normalizeVec3InPlace --- // sqrt + reciprocal. 14cy (2.2x). rsqrt+NR tested at 15cy — no gain, compiler's // vsqrtss+vdivss pipeline is already optimal for scalar inverse sqrt. -export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void { +pub fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void { const v: [*]f32 = @ptrFromInt(vec); const len = @sqrt(v[0] * v[0] + v[1] * v[1] + v[2] * v[2]); if (len > 1.0e-20) { @@ -432,7 +432,7 @@ export fn si_normalizeVec3InPlace(vec: u32) callconv(TC) void { // --- 0x71BC70: addVec3ToAccumulator (136K/7.5s) --- // thiscall(ECX=this, stack=vec). Scale is a global at 0x81207C, NOT a parameter. -export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void { +pub fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void { const obj: [*]f32 = @ptrFromInt(this); const v: [*]const f32 = @ptrFromInt(vec); const scale: f32 = @as(*const f32, @ptrFromInt(0x81207C)).*; @@ -447,7 +447,7 @@ export fn si_addVec3ToAccumulator(this: u32, vec: u32) callconv(TC) void { // --- 0x71BF60: addToColorAccumulator (10K/7.5s) --- // Naked thiscall: ECX=this, [ESP+4]=color_ptr. RET 4. Original: 34 bytes. // 3 SSE adds at this+0x6C from color[0..2]. -export fn si_addToColorAccumulator() callconv(.naked) void { +pub fn si_addToColorAccumulator() callconv(.naked) void { asm volatile ( \\mov 4(%%esp), %%eax \\vmovss (%%eax), %%xmm0 @@ -465,7 +465,7 @@ export fn si_addToColorAccumulator() callconv(.naked) void { // --- 0x7B7A80: packParticleColor (2K/7.5s) --- // V4 multiply + clamp, then packed round+convert via @Vector(4, i32) for all channels at once. -export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void { +pub fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) callconv(TC) void { const base: [*]u8 = @ptrFromInt(obj); const out: *align(1) u32 = @ptrCast(base + 0x12C); const alpha = base[0x12F]; @@ -480,7 +480,7 @@ export fn si_packParticleColor(obj: u32, r_bits: u32, g_bits: u32, b_bits: u32) // --- 0x7B7B10: setParticleAlpha (2K/7.5s) --- // Naked fastcall: ECX=obj, [ESP+4]=alpha_bits. RET 4. // Clamp alpha*255 to [0,255], write byte to obj+0x12F. -export fn si_setParticleAlpha() callconv(.naked) void { +pub fn si_setParticleAlpha() callconv(.naked) void { asm volatile ( \\vmovss 4(%%esp), %%xmm0 \\mov $0x437F0000, %%eax @@ -499,7 +499,7 @@ export fn si_setParticleAlpha() callconv(.naked) void { // --- 0x40A2B0: __ftol --- // Drop-in binary replacement. Input: ST(0). Output: EAX:EDX (i64). // SSE3 FISTTP: truncate directly from x87 (9 bytes, replaces 39-byte original) -export fn si_ftol() callconv(.naked) void { +pub fn si_ftol() callconv(.naked) void { asm volatile ( \\sub $8, %%esp \\fisttpll (%%esp) @@ -511,7 +511,7 @@ export fn si_ftol() callconv(.naked) void { // --- 0x602630: vec3Dot (31K/7.5s) --- // __fastcall(ECX=a, EDX=b), returns f64 via ST(0). -export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 { +pub fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 { const va: [*]const f32 = @ptrFromInt(a); const vb: [*]const f32 = @ptrFromInt(b); return @floatCast(@mulAdd(f32, va[2], vb[2], @mulAdd(f32, va[1], vb[1], va[0] * vb[0]))); @@ -519,7 +519,7 @@ export fn si_vec3Dot(a: u32, b: u32) callconv(FC) f64 { // --- 0x686820: translateBoundingVol --- // @mulAdd for plane distances. Scalar corner adds (stride 3 — V4 unaligned tested, slower). -export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void { +pub fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void { const obj: [*]f32 = @ptrFromInt(this); const off: [*]const f32 = @ptrFromInt(offset); const dx = off[0]; const dy = off[1]; const dz = off[2]; @@ -543,7 +543,7 @@ export fn si_translateBoundingVol(this: u32, offset: u32) callconv(TC) void { // Original: 380 bytes, 2 calls to mat*vec3 (0x7BCA80), x87 perspective divide, x87 column scan. // SSE: inline V4 mat*vec3, SSE perspective divide, 4-wide column scan. // __fastcall(bbox_ECX, flags_EDX, radius_stack), RET 0x4 -export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 { +pub fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(FC) u32 { // Early out: global occlusion flag bit 5 if ((@as(*const u8, @ptrFromInt(0xC7B2A4)).* & 0x20) == 0) return 0; @@ -638,7 +638,7 @@ export fn si_frustumCullBBox(bbox: u32, flags: u32, radius_bits: u32) callconv(F // __fastcall(listHead_ECX, queryBox_EDX, resultBuf_stack, flags_stack), RET 0x8 // addGeometryToBuffer at 0x6ABD90: __fastcall(queryBox_ECX, nodeData_EDX, resultBuf_stack), RET 0x4 // Visited sentinel: *(u32*)0xC89F20 -export fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 { +pub fn si_processLinkedListCollision(list_head: u32, query_box: u32, result_buf: u32, flags: u32) callconv(FC) u32 { if ((flags & 0xF0000F) == 0) return 1; const addGeometryToBuffer: *const fn (u32, u32, u32) callconv(FC) void = @ptrFromInt(0x6ABD90); diff --git a/src/performance/timer_fix.zig b/src/weirdperformance/timer_fix.zig similarity index 100% rename from src/performance/timer_fix.zig rename to src/weirdperformance/timer_fix.zig diff --git a/src/performance/weirdperformance.zig b/src/weirdperformance/weirdperformance.zig similarity index 80% rename from src/performance/weirdperformance.zig rename to src/weirdperformance/weirdperformance.zig index cae405d..3c04fe5 100644 --- a/src/performance/weirdperformance.zig +++ b/src/weirdperformance/weirdperformance.zig @@ -22,21 +22,8 @@ const filecache = @import("filecache.zig"); pub const module_name: [*:0]const u8 = "weirdperformance"; -// Provide malloc/free for libdeflate's default allocator (linked without libc). -// Use game's Storm memory manager: -// ReallocMemory (0x646320): __stdcall(ptr, size, filename, line, flags) → ptr -// When ptr=NULL, acts as malloc via AllocateBufferWithPowerOfTwo. -// FreeMemory (0x646430): __stdcall(ptr, filename, line, flags) RET 0x10 = 4 params -const gameRealloc: *const fn (u32, u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) ?*anyopaque = @ptrFromInt(0x646320); -const gameFree: *const fn (u32, u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) u32 = @ptrFromInt(0x646430); - -export fn malloc(size: usize) callconv(.c) ?*anyopaque { - return gameRealloc(0, @intCast(size), 0, 0, 0); -} - -export fn free(ptr: ?*anyopaque) callconv(.c) void { - if (ptr) |p| _ = gameFree(@intFromPtr(p), 0, 0, 0); -} +// malloc/free for libdeflate provided by stubs/game_alloc.c (compiled into +// the libdeflate static lib). This avoids exporting malloc/free from the DLL. var g_mutex: ?*anyopaque = null; var g_is_hook_owner: bool = false; @@ -47,12 +34,18 @@ pub fn isActive() bool { } // ============================================================================= -// Extern SSE functions (from separate ReleaseFast compilation units) +// SSE functions (imported directly to avoid addObject SizeOfImage bloat) // ============================================================================= -extern fn transformImpl_SSE(u32, u32, u32, u32, u32) callconv(.c) void; -extern fn renderParticleSprites_SSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn resetParticleCache() void; +const bone_sse = @import("bone_sse.zig"); +const particle_sse = @import("particle_sse.zig"); +const clip_sse = @import("clip_sse.zig"); +const cull_sse = @import("cull_sse.zig"); +const silicon_sse = @import("silicon_sse.zig"); + +const transformImpl_SSE = bone_sse.transformImpl_SSE; +const renderParticleSprites_SSE = particle_sse.renderParticleSprites_SSE; +const resetParticleCache = particle_sse.resetParticleCache; // ============================================================================= // transformMatrix4x4 hook (0x714260) @@ -218,35 +211,12 @@ fn worldUpdateDetour(frame_count: u32) callconv(hook.cc.fastcall) void { // ============================================================================= // cull_sse.zig -extern fn performSpatialCulling(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn performCollisionDetectionSSE(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; -extern fn rayTriIntersectIndexedInt(u32, u32, u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u8; +const performSpatialCulling = cull_sse.performSpatialCulling; +const performCollisionDetectionSSE = cull_sse.performCollisionDetectionSSE; +const rayTriIntersectIndexedInt = cull_sse.rayTriIntersectIndexedInt; // silicon_sse.zig -const sse = struct { - extern fn si_normalizeVec3() callconv(.naked) void; - extern fn si_mulMat3x4(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; - extern fn si_rotateMatByQuat(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; - extern fn si_createRotMat3x4(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; - extern fn si_classifyPointFrustum(u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; - extern fn si_checkBoxLineIntersect(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; - extern fn si_testOBBFrustum(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; - extern fn si_testSphereFrustum(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; - extern fn si_quatSlerp(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; - extern fn si_calculateSinCos(u32, u32, u32) callconv(.{ .x86_stdcall = .{} }) void; - extern fn si_createZRotMat3x3(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; - extern fn si_transposeMat4x4() callconv(.naked) void; - extern fn si_mulMat3x4InPlace(u32, u32) callconv(.{ .x86_thiscall = .{} }) u32; - extern fn si_normalizeVec3InPlace(u32) callconv(.{ .x86_thiscall = .{} }) void; - extern fn si_addVec3ToAccumulator(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; - extern fn si_addToColorAccumulator() callconv(.naked) void; - extern fn si_packParticleColor(u32, u32, u32, u32) callconv(.{ .x86_thiscall = .{} }) void; - extern fn si_setParticleAlpha() callconv(.naked) void; - extern fn si_ftol() callconv(.naked) void; - extern fn si_translateBoundingVol(u32, u32) callconv(.{ .x86_thiscall = .{} }) void; - extern fn si_processLinkedListCollision(u32, u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; - extern fn si_frustumCullBBox(u32, u32, u32) callconv(.{ .x86_fastcall = .{} }) u32; -}; +const sse = silicon_sse; const PatchEntry = struct { target: u32, @@ -348,7 +318,10 @@ pub fn installHooks() void { // libdeflate inflate replacement if (inflate_hook.install()) installed += 1; - // TSC timer calibration + OS timer tweaks +} + +pub fn lateInit() void { + if (!g_is_hook_owner) return; timer_fix.init(); } diff --git a/src/weirdperformance/zconf.h b/src/weirdperformance/zconf.h new file mode 100644 index 0000000..e4910fb --- /dev/null +++ b/src/weirdperformance/zconf.h @@ -0,0 +1,206 @@ +/* zconf.h -- configuration of the zlib compression library + * Copyright (C) 1995-2024 Jean-loup Gailly, Mark Adler + * For conditions of distribution and use, see copyright notice in zlib.h + */ + +#ifndef ZCONF_H +#define ZCONF_H + +#include "zlib_name_mangling.h" + +#if !defined(_WIN32) && defined(__WIN32__) +# define _WIN32 +#endif + +/* Clang macro for detecting declspec support + * https://clang.llvm.org/docs/LanguageExtensions.html#has-declspec-attribute + */ +#ifndef __has_declspec_attribute +# define __has_declspec_attribute(x) 0 +#endif + +#if defined(ZLIB_CONST) && !defined(z_const) +# define z_const const +#else +# define z_const +#endif + +/* Maximum value for memLevel in deflateInit2 */ +#ifndef MAX_MEM_LEVEL +# define MAX_MEM_LEVEL 9 +#endif + +/* Maximum value for windowBits in deflateInit2 and inflateInit2. + * WARNING: reducing MAX_WBITS makes minigzip unable to extract .gz files + * created by gzip. (Files created by minigzip can still be extracted by + * gzip.) + */ +#ifndef MIN_WBITS +# define MIN_WBITS 8 /* 256 LZ77 window */ +#endif +#ifndef MAX_WBITS +# define MAX_WBITS 15 /* 32K LZ77 window */ +#endif + +/* The memory requirements for deflate are (in bytes): + (1 << (windowBits+2)) + (1 << (memLevel+9)) + that is: 128K for windowBits=15 + 128K for memLevel = 8 (default values) + plus a few kilobytes for small objects. For example, if you want to reduce + the default memory requirements from 256K to 128K, compile with + make CFLAGS="-O -DMAX_WBITS=14 -DMAX_MEM_LEVEL=7" + Of course this will generally degrade compression (there's no free lunch). + + The memory requirements for inflate are (in bytes) 1 << windowBits + that is, 32K for windowBits=15 (default value) plus about 7 kilobytes + for small objects. +*/ + +/* Type declarations */ + + +#ifndef OF /* function prototypes */ +# define OF(args) args +#endif + +#ifdef ZLIB_INTERNAL +# define Z_INTERNAL ZLIB_INTERNAL +#endif + +/* If building or using zlib as a DLL, define ZLIB_DLL. + * This is not mandatory, but it offers a little performance increase. + */ +#if defined(ZLIB_DLL) && (defined(_WIN32) || (__has_declspec_attribute(dllexport) && __has_declspec_attribute(dllimport))) +# ifdef Z_INTERNAL +# define Z_EXTERN extern __declspec(dllexport) +# else +# define Z_EXTERN extern __declspec(dllimport) +# endif +#endif + +/* If building or using zlib with the WINAPI/WINAPIV calling convention, + * define ZLIB_WINAPI. + * Caution: the standard ZLIB1.DLL is NOT compiled using ZLIB_WINAPI. + */ +#if defined(ZLIB_WINAPI) && defined(_WIN32) +# ifndef WIN32_LEAN_AND_MEAN +# define WIN32_LEAN_AND_MEAN +# endif +# include + /* No need for _export, use ZLIB.DEF instead. */ + /* For complete Windows compatibility, use WINAPI, not __stdcall. */ +# define Z_EXPORT WINAPI +# define Z_EXPORTVA WINAPIV +#endif + +#ifndef Z_EXTERN +# define Z_EXTERN extern +#endif +#ifndef Z_EXPORT +# define Z_EXPORT +#endif +#ifndef Z_EXPORTVA +# define Z_EXPORTVA +#endif + +/* Conditional exports */ +#define ZNG_CONDEXPORT Z_INTERNAL + +/* For backwards compatibility */ + +#ifndef ZEXTERN +# define ZEXTERN Z_EXTERN +#endif +#ifndef ZEXPORT +# define ZEXPORT Z_EXPORT +#endif +#ifndef ZEXPORTVA +# define ZEXPORTVA Z_EXPORTVA +#endif +#ifndef FAR +# define FAR +#endif + +/* Legacy zlib typedefs for backwards compatibility. Don't assume stdint.h is defined. */ +typedef unsigned char Byte; +typedef Byte Bytef; + +typedef unsigned int uInt; /* 16 bits or more */ +typedef unsigned long uLong; /* 32 bits or more */ + +typedef char charf; +typedef int intf; +typedef uInt uIntf; +typedef uLong uLongf; + +typedef void const *voidpc; +typedef void *voidpf; +typedef void *voidp; + +typedef unsigned int z_crc_t; + +#if 1 /* was set to #if 1 by configure/cmake/etc */ +# define Z_HAVE_UNISTD_H +#endif + +#ifdef NEED_PTRDIFF_T /* may be set to #if 1 by configure/cmake/etc */ +typedef PTRDIFF_TYPE ptrdiff_t; +#endif + +#include /* for off_t */ + +#include /* for wchar_t and NULL */ + +/* a little trick to accommodate both "#define _LARGEFILE64_SOURCE" and + * "#define _LARGEFILE64_SOURCE 1" as requesting 64-bit operations, (even + * though the former does not conform to the LFS document), but considering + * both "#undef _LARGEFILE64_SOURCE" and "#define _LARGEFILE64_SOURCE 0" as + * equivalently requesting no 64-bit operations + */ +#if defined(_LARGEFILE64_SOURCE) && -_LARGEFILE64_SOURCE - -1 == 1 +# undef _LARGEFILE64_SOURCE +#endif + +#if defined(Z_HAVE_UNISTD_H) +# include /* for SEEK_*, off_t, and _LFS64_LARGEFILE */ +# ifndef z_off_t +# define z_off_t off_t +# endif +#endif + +#if defined(_LFS64_LARGEFILE) && _LFS64_LARGEFILE-0 +# define Z_LFS64 +#endif + +#if defined(_LARGEFILE64_SOURCE) && defined(Z_LFS64) +# define Z_LARGE64 +#endif + +#if defined(_FILE_OFFSET_BITS) && _FILE_OFFSET_BITS-0 == 64 && defined(Z_LFS64) +# define Z_WANT64 +#endif + +#if !defined(SEEK_SET) +# define SEEK_SET 0 /* Seek from beginning of file. */ +# define SEEK_CUR 1 /* Seek from current position. */ +# define SEEK_END 2 /* Set file pointer to EOF plus "offset" */ +#endif + +#ifndef z_off_t +# define z_off_t long +#endif + +#if !defined(_WIN32) && defined(Z_LARGE64) +# define z_off64_t off64_t +#else +# if defined(__MSYS__) +# define z_off64_t _off64_t +# elif defined(_WIN32) && !defined(__GNUC__) +# define z_off64_t __int64 +# else +# define z_off64_t z_off_t +# endif +#endif + +typedef size_t z_size_t; + +#endif /* ZCONF_H */ diff --git a/src/weirdperformance/zlib.h b/src/weirdperformance/zlib.h new file mode 100644 index 0000000..4dc5ba8 --- /dev/null +++ b/src/weirdperformance/zlib.h @@ -0,0 +1,1859 @@ +#ifndef ZLIB_H_ +#define ZLIB_H_ +/* zlib.h -- interface of the 'zlib-ng' compression library + Forked from and compatible with zlib 1.3.1 + + Copyright (C) 1995-2024 Jean-loup Gailly and Mark Adler + + This software is provided 'as-is', without any express or implied + warranty. In no event will the authors be held liable for any damages + arising from the use of this software. + + Permission is granted to anyone to use this software for any purpose, + including commercial applications, and to alter it and redistribute it + freely, subject to the following restrictions: + + 1. The origin of this software must not be misrepresented; you must not + claim that you wrote the original software. If you use this software + in a product, an acknowledgment in the product documentation would be + appreciated but is not required. + 2. Altered source versions must be plainly marked as such, and must not be + misrepresented as being the original software. + 3. This notice may not be removed or altered from any source distribution. + + Jean-loup Gailly Mark Adler + jloup@gzip.org madler@alumni.caltech.edu + + + The data format used by the zlib library is described by RFCs (Request for + Comments) 1950 to 1952 in the files https://tools.ietf.org/html/rfc1950 + (zlib format), rfc1951 (deflate format) and rfc1952 (gzip format). +*/ + +#ifdef ZNGLIB_H_ +# error Include zlib-ng.h for zlib-ng API or zlib.h for zlib-compat API but not both +#endif + +#ifndef RC_INVOKED +#include +#include + +#include "zconf.h" + +#ifndef ZCONF_H +# error Missing zconf.h add binary output directory to include directories +#endif +#endif /* RC_INVOKED */ + +#ifdef __cplusplus +extern "C" { +#endif + +#define ZLIBNG_VERSION "2.3.90" +#define ZLIBNG_VERNUM 0x02039000L /* MMNNRRSM: major minor revision status modified */ +#define ZLIBNG_VER_MAJOR 2 +#define ZLIBNG_VER_MINOR 3 +#define ZLIBNG_VER_REVISION 90 +#define ZLIBNG_VER_STATUS 0 /* 0=devel, 1-E=beta, F=Release (DEPRECATED) */ +#define ZLIBNG_VER_STATUSH 0x0 /* Hex values: 0=devel, 1-9=beta, A-E=Release Candidate, F=Release */ +#define ZLIBNG_VER_MODIFIED 0 /* non-zero if modified externally from zlib-ng */ + +#define ZLIB_VERSION "1.3.1.zlib-ng" +#define ZLIB_VERNUM 0x131f +#define ZLIB_VER_MAJOR 1 +#define ZLIB_VER_MINOR 3 +#define ZLIB_VER_REVISION 1 +#define ZLIB_VER_SUBREVISION 15 /* 15=fork (0xf) */ + +/* + The 'zlib' compression library provides in-memory compression and + decompression functions, including integrity checks of the uncompressed data. + This version of the library supports only one compression method (deflation) + but other algorithms will be added later and will have the same stream + interface. + + Compression can be done in a single step if the buffers are large enough, + or can be done by repeated calls of the compression function. In the latter + case, the application must provide more input and/or consume the output + (providing more output space) before each call. + + The compressed data format used by default by the in-memory functions is + the zlib format, which is a zlib wrapper documented in RFC 1950, wrapped + around a deflate stream, which is itself documented in RFC 1951. + + The library also supports reading and writing files in gzip (.gz) format + with an interface similar to that of stdio using the functions that start + with "gz". The gzip format is different from the zlib format. gzip is a + gzip wrapper, documented in RFC 1952, wrapped around a deflate stream. + + This library can optionally read and write gzip and raw deflate streams in + memory as well. + + The zlib format was designed to be compact and fast for use in memory + and on communications channels. The gzip format was designed for single- + file compression on file systems, has a larger header than zlib to maintain + directory information, and uses a different, slower check method than zlib. + + The library does not install any signal handler. The decoder checks + the consistency of the compressed data, so the library should never crash + even in the case of corrupted input. +*/ + +typedef void *(*alloc_func) (void *opaque, unsigned int items, unsigned int size); +typedef void (*free_func) (void *opaque, void *address); + +struct internal_state; + +typedef struct z_stream_s { + z_const unsigned char *next_in; /* next input byte */ + uint32_t avail_in; /* number of bytes available at next_in */ + unsigned long total_in; /* total number of input bytes read so far */ + + unsigned char *next_out; /* next output byte will go here */ + uint32_t avail_out; /* remaining free space at next_out */ + unsigned long total_out; /* total number of bytes output so far */ + + z_const char *msg; /* last error message, NULL if no error */ + struct internal_state *state; /* not visible by applications */ + + alloc_func zalloc; /* used to allocate the internal state */ + free_func zfree; /* used to free the internal state */ + void *opaque; /* private data object passed to zalloc and zfree */ + + int data_type; /* best guess about the data type: binary or text + for deflate, or the decoding state for inflate */ + unsigned long adler; /* Adler-32 or CRC-32 value of the uncompressed data */ + unsigned long reserved; /* reserved for future use */ +} z_stream; + +typedef z_stream *z_streamp; /* Obsolete type, retained for compatibility only */ + +/* + gzip header information passed to and from zlib routines. See RFC 1952 + for more details on the meanings of these fields. +*/ +typedef struct gz_header_s { + int text; /* true if compressed data believed to be text */ + unsigned long time; /* modification time */ + int xflags; /* extra flags (not used when writing a gzip file) */ + int os; /* operating system */ + unsigned char *extra; /* pointer to extra field or NULL if none */ + unsigned int extra_len; /* extra field length (valid if extra != NULL) */ + unsigned int extra_max; /* space at extra (only when reading header) */ + unsigned char *name; /* pointer to zero-terminated file name or NULL */ + unsigned int name_max; /* space at name (only when reading header) */ + unsigned char *comment; /* pointer to zero-terminated comment or NULL */ + unsigned int comm_max; /* space at comment (only when reading header) */ + int hcrc; /* true if there was or will be a header crc */ + int done; /* true when done reading gzip header (not used when writing a gzip file) */ +} gz_header; + +typedef gz_header *gz_headerp; + +/* + The application must update next_in and avail_in when avail_in has dropped + to zero. It must update next_out and avail_out when avail_out has dropped + to zero. The application must initialize zalloc, zfree and opaque before + calling the init function. All other fields are set by the compression + library and must not be updated by the application. + + The opaque value provided by the application will be passed as the first + parameter for calls of zalloc and zfree. This can be useful for custom + memory management. The compression library attaches no meaning to the + opaque value. + + zalloc must return NULL if there is not enough memory for the object. + If zlib is used in a multi-threaded application, zalloc and zfree must be + thread safe. In that case, zlib is thread-safe. When zalloc and zfree are + Z_NULL on entry to the initialization function, they are set to internal + routines that use the standard library functions malloc() and free(). + + The fields total_in and total_out can be used for statistics or progress + reports. After compression, total_in holds the total size of the + uncompressed data and may be saved for use by the decompressor (particularly + if the decompressor wants to decompress everything in a single step). +*/ + + /* constants */ + +#define Z_NO_FLUSH 0 +#define Z_PARTIAL_FLUSH 1 +#define Z_SYNC_FLUSH 2 +#define Z_FULL_FLUSH 3 +#define Z_FINISH 4 +#define Z_BLOCK 5 +#define Z_TREES 6 +/* Allowed flush values; see deflate() and inflate() below for details */ + +#define Z_OK 0 +#define Z_STREAM_END 1 +#define Z_NEED_DICT 2 +#define Z_ERRNO (-1) +#define Z_STREAM_ERROR (-2) +#define Z_DATA_ERROR (-3) +#define Z_MEM_ERROR (-4) +#define Z_BUF_ERROR (-5) +#define Z_VERSION_ERROR (-6) +/* Return codes for the compression/decompression functions. Negative values + * are errors, positive values are used for special but normal events. + */ + +#define Z_NO_COMPRESSION 0 +#define Z_BEST_SPEED 1 +#define Z_BEST_COMPRESSION 9 +#define Z_DEFAULT_COMPRESSION (-1) +/* compression levels */ + +#define Z_FILTERED 1 +#define Z_HUFFMAN_ONLY 2 +#define Z_RLE 3 +#define Z_FIXED 4 +#define Z_DEFAULT_STRATEGY 0 +/* compression strategy; see deflateInit2() below for details */ + +#define Z_BINARY 0 +#define Z_TEXT 1 +#define Z_ASCII Z_TEXT /* for compatibility with 1.2.2 and earlier */ +#define Z_UNKNOWN 2 +/* Possible values of the data_type field for deflate() */ + +#define Z_DEFLATED 8 +/* The deflate compression method (the only one supported in this version) */ + +#define Z_NULL 0 /* for compatibility with zlib, was for initializing zalloc, zfree, opaque */ + +#define zlib_version zlibVersion() +/* for compatibility with versions < 1.0.2 */ + + + /* basic functions */ + +Z_EXTERN const char * Z_EXPORT zlibVersion(void); +/* The application can compare zlibVersion and ZLIB_VERSION for consistency. + If the first character differs, the library code actually used is not + compatible with the zlib.h header file used by the application. This check + is automatically made by deflateInit and inflateInit. + */ + +/* +Z_EXTERN int Z_EXPORT deflateInit (z_stream *strm, int level); + + Initializes the internal stream state for compression. The fields + zalloc, zfree and opaque must be initialized before by the caller. If + zalloc and zfree are set to Z_NULL, deflateInit updates them to use default + allocation functions. total_in, total_out, adler, and msg are initialized. + + The compression level must be Z_DEFAULT_COMPRESSION, or between 0 and 9: + 1 gives best speed, 9 gives best compression, 0 gives no compression at all + (the input data is simply copied a block at a time). Z_DEFAULT_COMPRESSION + requests a default compromise between speed and compression (currently + equivalent to level 6). + + deflateInit returns Z_OK if success, Z_MEM_ERROR if there was not enough + memory, Z_STREAM_ERROR if level is not a valid compression level, or + Z_VERSION_ERROR if the zlib library version (zlib_version) is incompatible + with the version assumed by the caller (ZLIB_VERSION). msg is set to null + if there is no error message. deflateInit does not perform any compression: + this will be done by deflate(). +*/ + + +Z_EXTERN int Z_EXPORT deflate(z_stream *strm, int flush); +/* + deflate compresses as much data as possible, and stops when the input + buffer becomes empty or the output buffer becomes full. It may introduce + some output latency (reading input without producing any output) except when + forced to flush. + + The detailed semantics are as follows. deflate performs one or both of the + following actions: + + - Compress more input starting at next_in and update next_in and avail_in + accordingly. If not all input can be processed (because there is not + enough room in the output buffer), next_in and avail_in are updated and + processing will resume at this point for the next call of deflate(). + + - Generate more output starting at next_out and update next_out and avail_out + accordingly. This action is forced if the parameter flush is non zero. + Forcing flush frequently degrades the compression ratio, so this parameter + should be set only when necessary. Some output may be provided even if + flush is zero. + + Before the call of deflate(), the application should ensure that at least + one of the actions is possible, by providing more input and/or consuming more + output, and updating avail_in or avail_out accordingly; avail_out should + never be zero before the call. The application can consume the compressed + output when it wants, for example when the output buffer is full (avail_out + == 0), or after each call of deflate(). If deflate returns Z_OK and with + zero avail_out, it must be called again after making room in the output + buffer because there might be more output pending. See deflatePending(), + which can be used if desired to determine whether or not there is more output + in that case. + + Normally the parameter flush is set to Z_NO_FLUSH, which allows deflate to + decide how much data to accumulate before producing output, in order to + maximize compression. + + If the parameter flush is set to Z_SYNC_FLUSH, all pending output is + flushed to the output buffer and the output is aligned on a byte boundary, so + that the decompressor can get all input data available so far. (In + particular avail_in is zero after the call if enough output space has been + provided before the call.) Flushing may degrade compression for some + compression algorithms and so it should be used only when necessary. This + completes the current deflate block and follows it with an empty stored block + that is three bits plus filler bits to the next byte, followed by four bytes + (00 00 ff ff). + + If flush is set to Z_PARTIAL_FLUSH, all pending output is flushed to the + output buffer, but the output is not aligned to a byte boundary. All of the + input data so far will be available to the decompressor, as for Z_SYNC_FLUSH. + This completes the current deflate block and follows it with an empty fixed + codes block that is 10 bits long. This assures that enough bytes are output + in order for the decompressor to finish the block before the empty fixed + codes block. + + If flush is set to Z_BLOCK, a deflate block is completed and emitted, as + for Z_SYNC_FLUSH, but the output is not aligned on a byte boundary, and up to + seven bits of the current block are held to be written as the next byte after + the next deflate block is completed. In this case, the decompressor may not + be provided enough bits at this point in order to complete decompression of + the data provided so far to the compressor. It may need to wait for the next + block to be emitted. This is for advanced applications that need to control + the emission of deflate blocks. + + If flush is set to Z_FULL_FLUSH, all output is flushed as with + Z_SYNC_FLUSH, and the compression state is reset so that decompression can + restart from this point if previous compressed data has been damaged or if + random access is desired. Using Z_FULL_FLUSH too often can seriously degrade + compression. + + If deflate returns with avail_out == 0, this function must be called again + with the same value of the flush parameter and more output space (updated + avail_out), until the flush is complete (deflate returns with non-zero + avail_out). In the case of a Z_FULL_FLUSH or Z_SYNC_FLUSH, make sure that + avail_out is greater than six when the flush marker begins, in order to avoid + repeated flush markers upon calling deflate() again when avail_out == 0. + + If the parameter flush is set to Z_FINISH, pending input is processed, + pending output is flushed and deflate returns with Z_STREAM_END if there was + enough output space. If deflate returns with Z_OK or Z_BUF_ERROR, this + function must be called again with Z_FINISH and more output space (updated + avail_out) but no more input data, until it returns with Z_STREAM_END or an + error. After deflate has returned Z_STREAM_END, the only possible operations + on the stream are deflateReset or deflateEnd. + + Z_FINISH can be used in the first deflate call after deflateInit if all the + compression is to be done in a single step. In order to complete in one + call, avail_out must be at least the value returned by deflateBound (see + below). Then deflate is guaranteed to return Z_STREAM_END. If not enough + output space is provided, deflate will not return Z_STREAM_END, and it must + be called again as described above. + + deflate() sets strm->adler to the Adler-32 checksum of all input read + so far (that is, total_in bytes). If a gzip stream is being generated, then + strm->adler will be the CRC-32 checksum of the input read so far. (See + deflateInit2 below.) + + deflate() may update strm->data_type if it can make a good guess about + the input data type (Z_BINARY or Z_TEXT). If in doubt, the data is + considered binary. This field is only for information purposes and does not + affect the compression algorithm in any manner. + + deflate() returns Z_OK if some progress has been made (more input + processed or more output produced), Z_STREAM_END if all input has been + consumed and all output has been produced (only when flush is set to + Z_FINISH), Z_STREAM_ERROR if the stream state was inconsistent (for example + if next_in or next_out was NULL) or the state was inadvertently written over + by the application), or Z_BUF_ERROR if no progress is possible (for example + avail_in or avail_out was zero). Note that Z_BUF_ERROR is not fatal, and + deflate() can be called again with more input and more output space to + continue compressing. +*/ + + +Z_EXTERN int Z_EXPORT deflateEnd(z_stream *strm); +/* + All dynamically allocated data structures for this stream are freed. + This function discards any unprocessed input and does not flush any pending + output. + + deflateEnd returns Z_OK if success, Z_STREAM_ERROR if the + stream state was inconsistent, Z_DATA_ERROR if the stream was freed + prematurely (some input or output was discarded). In the error case, msg + may be set but then points to a static string (which must not be + deallocated). +*/ + + +/* +Z_EXTERN int Z_EXPORT inflateInit (z_stream *strm); + + Initializes the internal stream state for decompression. The fields + next_in, avail_in, zalloc, zfree and opaque must be initialized before by + the caller. In the current version of inflate, the provided input is not + read or consumed. The allocation of a sliding window will be deferred to + the first call of inflate (if the decompression does not complete on the + first call). If zalloc and zfree are set to Z_NULL, inflateInit updates + them to use default allocation functions. total_in, total_out, adler, and + msg are initialized. + + inflateInit returns Z_OK if success, Z_MEM_ERROR if there was not enough + memory, Z_VERSION_ERROR if the zlib library version is incompatible with the + version assumed by the caller, or Z_STREAM_ERROR if the parameters are + invalid, such as a null pointer to the structure. msg is set to null if + there is no error message. inflateInit does not perform any decompression. + Actual decompression will be done by inflate(). So next_in, and avail_in, + next_out, and avail_out are unused and unchanged. The current + implementation of inflateInit() does not process any header information -- + that is deferred until inflate() is called. +*/ + + +Z_EXTERN int Z_EXPORT inflate(z_stream *strm, int flush); +/* + inflate decompresses as much data as possible, and stops when the input + buffer becomes empty or the output buffer becomes full. It may introduce + some output latency (reading input without producing any output) except when + forced to flush. + + The detailed semantics are as follows. inflate performs one or both of the + following actions: + + - Decompress more input starting at next_in and update next_in and avail_in + accordingly. If not all input can be processed (because there is not + enough room in the output buffer), then next_in and avail_in are updated + accordingly, and processing will resume at this point for the next call of + inflate(). + + - Generate more output starting at next_out and update next_out and avail_out + accordingly. inflate() provides as much output as possible, until there is + no more input data or no more space in the output buffer (see below about + the flush parameter). + + Before the call of inflate(), the application should ensure that at least + one of the actions is possible, by providing more input and/or consuming more + output, and updating the next_* and avail_* values accordingly. If the + caller of inflate() does not provide both available input and available + output space, it is possible that there will be no progress made. The + application can consume the uncompressed output when it wants, for example + when the output buffer is full (avail_out == 0), or after each call of + inflate(). If inflate returns Z_OK and with zero avail_out, it must be + called again after making room in the output buffer because there might be + more output pending. + + The flush parameter of inflate() can be Z_NO_FLUSH, Z_SYNC_FLUSH, Z_FINISH, + Z_BLOCK, or Z_TREES. Z_SYNC_FLUSH requests that inflate() flush as much + output as possible to the output buffer. Z_BLOCK requests that inflate() + stop if and when it gets to the next deflate block boundary. When decoding + the zlib or gzip format, this will cause inflate() to return immediately + after the header and before the first block. When doing a raw inflate, + inflate() will go ahead and process the first block, and will return when it + gets to the end of that block, or when it runs out of data. + + The Z_BLOCK option assists in appending to or combining deflate streams. + To assist in this, on return inflate() always sets strm->data_type to the + number of unused bits in the last byte taken from strm->next_in, plus 64 if + inflate() is currently decoding the last block in the deflate stream, plus + 128 if inflate() returned immediately after decoding an end-of-block code or + decoding the complete header up to just before the first byte of the deflate + stream. The end-of-block will not be indicated until all of the uncompressed + data from that block has been written to strm->next_out. The number of + unused bits may in general be greater than seven, except when bit 7 of + data_type is set, in which case the number of unused bits will be less than + eight. data_type is set as noted here every time inflate() returns for all + flush options, and so can be used to determine the amount of currently + consumed input in bits. + + The Z_TREES option behaves as Z_BLOCK does, but it also returns when the + end of each deflate block header is reached, before any actual data in that + block is decoded. This allows the caller to determine the length of the + deflate block header for later use in random access within a deflate block. + 256 is added to the value of strm->data_type when inflate() returns + immediately after reaching the end of the deflate block header. + + inflate() should normally be called until it returns Z_STREAM_END or an + error. However if all decompression is to be performed in a single step (a + single call of inflate), the parameter flush should be set to Z_FINISH. In + this case all pending input is processed and all pending output is flushed; + avail_out must be large enough to hold all of the uncompressed data for the + operation to complete. (The size of the uncompressed data may have been + saved by the compressor for this purpose.) The use of Z_FINISH is not + required to perform an inflation in one step. However it may be used to + inform inflate that a faster approach can be used for the single inflate() + call. Z_FINISH also informs inflate to not maintain a sliding window if the + stream completes, which reduces inflate's memory footprint. If the stream + does not complete, either because not all of the stream is provided or not + enough output space is provided, then a sliding window will be allocated and + inflate() can be called again to continue the operation as if Z_NO_FLUSH had + been used. + + In this implementation, inflate() always flushes as much output as + possible to the output buffer, and always uses the faster approach on the + first call. So the effects of the flush parameter in this implementation are + on the return value of inflate() as noted below, when inflate() returns early + when Z_BLOCK or Z_TREES is used, and when inflate() avoids the allocation of + memory for a sliding window when Z_FINISH is used. + + If a preset dictionary is needed after this call (see inflateSetDictionary + below), inflate sets strm->adler to the Adler-32 checksum of the dictionary + chosen by the compressor and returns Z_NEED_DICT; otherwise it sets + strm->adler to the Adler-32 checksum of all output produced so far (that is, + total_out bytes) and returns Z_OK, Z_STREAM_END or an error code as described + below. At the end of the stream, inflate() checks that its computed Adler-32 + checksum is equal to that saved by the compressor and returns Z_STREAM_END + only if the checksum is correct. + + inflate() can decompress and check either zlib-wrapped or gzip-wrapped + deflate data. The header type is detected automatically, if requested when + initializing with inflateInit2(). Any information contained in the gzip + header is not retained unless inflateGetHeader() is used. When processing + gzip-wrapped deflate data, strm->adler32 is set to the CRC-32 of the output + produced so far. The CRC-32 is checked against the gzip trailer, as is the + uncompressed length, modulo 2^32. + + inflate() returns Z_OK if some progress has been made (more input processed + or more output produced), Z_STREAM_END if the end of the compressed data has + been reached and all uncompressed output has been produced, Z_NEED_DICT if a + preset dictionary is needed at this point, Z_DATA_ERROR if the input data was + corrupted (input stream not conforming to the zlib format or incorrect check + value, in which case strm->msg points to a string with a more specific + error), Z_STREAM_ERROR if the stream structure was inconsistent (for example + next_in or next_out was NULL, or the state was inadvertently written over + by the application), Z_MEM_ERROR if there was not enough memory, Z_BUF_ERROR + if no progress is possible or if there was not enough room in the output + buffer when Z_FINISH is used. Note that Z_BUF_ERROR is not fatal, and + inflate() can be called again with more input and more output space to + continue decompressing. If Z_DATA_ERROR is returned, the application may + then call inflateSync() to look for a good compression block if a partial + recovery of the data is to be attempted. +*/ + + +Z_EXTERN int Z_EXPORT inflateEnd(z_stream *strm); +/* + All dynamically allocated data structures for this stream are freed. + This function discards any unprocessed input and does not flush any pending + output. + + inflateEnd returns Z_OK if success, or Z_STREAM_ERROR if the stream state + was inconsistent. +*/ + + + /* Advanced functions */ + +/* + The following functions are needed only in some special applications. +*/ + +/* +Z_EXTERN int Z_EXPORT deflateInit2 (z_stream *strm, + int level, + int method, + int windowBits, + int memLevel, + int strategy); + + This is another version of deflateInit with more compression options. The + fields zalloc, zfree and opaque must be initialized before by the caller. + + The method parameter is the compression method. It must be Z_DEFLATED in + this version of the library. + + The windowBits parameter is the base two logarithm of the window size + (the size of the history buffer). It should be in the range 8..15 for this + version of the library. Larger values of this parameter result in better + compression at the expense of memory usage. The default value is 15 if + deflateInit is used instead. + + For the current implementation of deflate(), a windowBits value of 8 (a + window size of 256 bytes) is not supported. As a result, a request for 8 + will result in 9 (a 512-byte window). In that case, providing 8 to + inflateInit2() will result in an error when the zlib header with 9 is + checked against the initialization of inflate(). The remedy is to not use 8 + with deflateInit2() with this initialization, or at least in that case use 9 + with inflateInit2(). + + windowBits can also be -8..-15 for raw deflate. In this case, -windowBits + determines the window size. deflate() will then generate raw deflate data + with no zlib header or trailer, and will not compute a check value. + + windowBits can also be greater than 15 for optional gzip encoding. Add + 16 to windowBits to write a simple gzip header and trailer around the + compressed data instead of a zlib wrapper. The gzip header will have no + file name, no extra data, no comment, no modification time (set to zero), no + header crc, and the operating system will be set to the appropriate value, + if the operating system was determined at compile time. If a gzip stream is + being written, strm->adler is a CRC-32 instead of an Adler-32. + + For raw deflate or gzip encoding, a request for a 256-byte window is + rejected as invalid, since only the zlib header provides a means of + transmitting the window size to the decompressor. + + The memLevel parameter specifies how much memory should be allocated + for the internal compression state. memLevel=1 uses minimum memory but is + slow and reduces compression ratio; memLevel=9 uses maximum memory for + optimal speed. The default value is 8. See zconf.h for total memory usage + as a function of windowBits and memLevel. + + The strategy parameter is used to tune the compression algorithm. Use the + value Z_DEFAULT_STRATEGY for normal data, Z_FILTERED for data produced by a + filter (or predictor), Z_HUFFMAN_ONLY to force Huffman encoding only (no + string match), or Z_RLE to limit match distances to one (run-length + encoding). Filtered data consists mostly of small values with a somewhat + random distribution. In this case, the compression algorithm is tuned to + compress them better. The effect of Z_FILTERED is to force more Huffman + coding and less string matching; it is somewhat intermediate between + Z_DEFAULT_STRATEGY and Z_HUFFMAN_ONLY. Z_RLE is designed to be almost as + fast as Z_HUFFMAN_ONLY, but give better compression for PNG image data. The + strategy parameter only affects the compression ratio but not the + correctness of the compressed output even if it is not set appropriately. + Z_FIXED prevents the use of dynamic Huffman codes, allowing for a simpler + decoder for special applications. + + deflateInit2 returns Z_OK if success, Z_MEM_ERROR if there was not enough + memory, Z_STREAM_ERROR if any parameter is invalid (such as an invalid + method), or Z_VERSION_ERROR if the zlib library version (zlib_version) is + incompatible with the version assumed by the caller (ZLIB_VERSION). msg is + set to null if there is no error message. deflateInit2 does not perform any + compression: this will be done by deflate(). +*/ + +Z_EXTERN int Z_EXPORT deflateSetDictionary(z_stream *strm, + const unsigned char *dictionary, + unsigned int dictLength); +/* + Initializes the compression dictionary from the given byte sequence + without producing any compressed output. When using the zlib format, this + function must be called immediately after deflateInit, deflateInit2 or + deflateReset, and before any call of deflate. When doing raw deflate, this + function must be called either before any call of deflate, or immediately + after the completion of a deflate block, i.e. after all input has been + consumed and all output has been delivered when using any of the flush + options Z_BLOCK, Z_PARTIAL_FLUSH, Z_SYNC_FLUSH, or Z_FULL_FLUSH. The + compressor and decompressor must use exactly the same dictionary (see + inflateSetDictionary). + + The dictionary should consist of strings (byte sequences) that are likely + to be encountered later in the data to be compressed, with the most commonly + used strings preferably put towards the end of the dictionary. Using a + dictionary is most useful when the data to be compressed is short and can be + predicted with good accuracy; the data can then be compressed better than + with the default empty dictionary. + + Depending on the size of the compression data structures selected by + deflateInit or deflateInit2, a part of the dictionary may in effect be + discarded, for example if the dictionary is larger than the window size + provided in deflateInit or deflateInit2. Thus the strings most likely to be + useful should be put at the end of the dictionary, not at the front. In + addition, the current implementation of deflate will use at most the window + size minus 262 bytes of the provided dictionary. + + Upon return of this function, strm->adler is set to the Adler-32 value + of the dictionary; the decompressor may later use this value to determine + which dictionary has been used by the compressor. (The Adler-32 value + applies to the whole dictionary even if only a subset of the dictionary is + actually used by the compressor.) If a raw deflate was requested, then the + Adler-32 value is not computed and strm->adler is not set. + + deflateSetDictionary returns Z_OK if success, or Z_STREAM_ERROR if a + parameter is invalid (e.g. dictionary being NULL) or the stream state is + inconsistent (for example if deflate has already been called for this stream + or if not at a block boundary for raw deflate). deflateSetDictionary does + not perform any compression: this will be done by deflate(). +*/ + +Z_EXTERN int Z_EXPORT deflateGetDictionary (z_stream *strm, unsigned char *dictionary, unsigned int *dictLength); +/* + Returns the sliding dictionary being maintained by deflate. dictLength is + set to the number of bytes in the dictionary, and that many bytes are copied + to dictionary. dictionary must have enough space, where 32768 bytes is + always enough. If deflateGetDictionary() is called with dictionary equal to + Z_NULL, then only the dictionary length is returned, and nothing is copied. + Similarly, if dictLength is Z_NULL, then it is not set. + + deflateGetDictionary() may return a length less than the window size, even + when more than the window size in input has been provided. It may return up + to 258 bytes less in that case, due to how zlib's implementation of deflate + manages the sliding window and lookahead for matches, where matches can be + up to 258 bytes long. If the application needs the last window-size bytes of + input, then that would need to be saved by the application outside of zlib. + + deflateGetDictionary returns Z_OK on success, or Z_STREAM_ERROR if the + stream state is inconsistent. +*/ + +Z_EXTERN int Z_EXPORT deflateCopy(z_stream *dest, z_stream *source); +/* + Sets the destination stream as a complete copy of the source stream. + + This function can be useful when several compression strategies will be + tried, for example when there are several ways of pre-processing the input + data with a filter. The streams that will be discarded should then be freed + by calling deflateEnd. Note that deflateCopy duplicates the internal + compression state which can be quite large, so this strategy is slow and can + consume lots of memory. + + deflateCopy returns Z_OK if success, Z_MEM_ERROR if there was not + enough memory, Z_STREAM_ERROR if the source stream state was inconsistent + (such as zalloc being NULL). msg is left unchanged in both source and + destination. +*/ + +Z_EXTERN int Z_EXPORT deflateReset(z_stream *strm); +/* + This function is equivalent to deflateEnd followed by deflateInit, but + does not free and reallocate the internal compression state. The stream + will leave the compression level and any other attributes that may have been + set unchanged. total_in, total_out, adler, and msg are initialized. + + deflateReset returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent (such as zalloc or state being NULL). +*/ + +Z_EXTERN int Z_EXPORT deflateParams(z_stream *strm, int level, int strategy); +/* + Dynamically update the compression level and compression strategy. The + interpretation of level and strategy is as in deflateInit2(). This can be + used to switch between compression and straight copy of the input data, or + to switch to a different kind of input data requiring a different strategy. + If the compression approach (which is a function of the level) or the + strategy is changed, and if there have been any deflate() calls since the + state was initialized or reset, then the input available so far is + compressed with the old level and strategy using deflate(strm, Z_BLOCK). + There are three approaches for the compression levels 0, 1..3, and 4..9 + respectively. The new level and strategy will take effect at the next call + of deflate(). + + If a deflate(strm, Z_BLOCK) is performed by deflateParams(), and it does + not have enough output space to complete, then the parameter change will not + take effect. In this case, deflateParams() can be called again with the + same parameters and more output space to try again. + + In order to assure a change in the parameters on the first try, the + deflate stream should be flushed using deflate() with Z_BLOCK or other flush + request until strm.avail_out is not zero, before calling deflateParams(). + Then no more input data should be provided before the deflateParams() call. + If this is done, the old level and strategy will be applied to the data + compressed before deflateParams(), and the new level and strategy will be + applied to the data compressed after deflateParams(). + + deflateParams returns Z_OK on success, Z_STREAM_ERROR if the source stream + state was inconsistent or if a parameter was invalid, or Z_BUF_ERROR if + there was not enough output space to complete the compression of the + available input data before a change in the strategy or approach. Note that + in the case of a Z_BUF_ERROR, the parameters are not changed. A return + value of Z_BUF_ERROR is not fatal, in which case deflateParams() can be + retried with more output space. +*/ + +Z_EXTERN int Z_EXPORT deflateTune(z_stream *strm, int good_length, int max_lazy, int nice_length, int max_chain); +/* + Fine tune deflate's internal compression parameters. This should only be + used by someone who understands the algorithm used by zlib's deflate for + searching for the best matching string, and even then only by the most + fanatic optimizer trying to squeeze out the last compressed bit for their + specific input data. Read the deflate.c source code for the meaning of the + max_lazy, good_length, nice_length, and max_chain parameters. + + deflateTune() can be called after deflateInit() or deflateInit2(), and + returns Z_OK on success, or Z_STREAM_ERROR for an invalid deflate stream. + */ + +Z_EXTERN unsigned long Z_EXPORT deflateBound(z_stream *strm, unsigned long sourceLen); +/* + deflateBound() returns an upper bound on the compressed size after + deflation of sourceLen bytes. It must be called after deflateInit() or + deflateInit2(), and after deflateSetHeader(), if used. This would be used + to allocate an output buffer for deflation in a single pass, and so would be + called before deflate(). If that first deflate() call is provided the + sourceLen input bytes, an output buffer allocated to the size returned by + deflateBound(), and the flush value Z_FINISH, then deflate() is guaranteed + to return Z_STREAM_END. Note that it is possible for the compressed size to + be larger than the value returned by deflateBound() if flush options other + than Z_FINISH or Z_NO_FLUSH are used. +*/ + +Z_EXTERN int Z_EXPORT deflatePending(z_stream *strm, uint32_t *pending, int *bits); +/* + deflatePending() returns the number of bytes and bits of output that have + been generated, but not yet provided in the available output. The bytes not + provided would be due to the available output space having being consumed. + The number of bits of output not provided are between 0 and 7, where they + await more bits to join them in order to fill out a full byte. If pending + or bits are NULL, then those values are not set. + + deflatePending returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent. + */ + +Z_EXTERN int Z_EXPORT deflatePrime(z_stream *strm, int bits, int value); +/* + deflatePrime() inserts bits in the deflate output stream. The intent + is that this function is used to start off the deflate output with the bits + leftover from a previous deflate stream when appending to it. As such, this + function can only be used for raw deflate, and must be used before the first + deflate() call after a deflateInit2() or deflateReset(). bits must be less + than or equal to 16, and that many of the least significant bits of value + will be inserted in the output. + + deflatePrime returns Z_OK if success, Z_BUF_ERROR if there was not enough + room in the internal buffer to insert the bits, or Z_STREAM_ERROR if the + source stream state was inconsistent. +*/ + +Z_EXTERN int Z_EXPORT deflateSetHeader(z_stream *strm, gz_headerp head); +/* + deflateSetHeader() provides gzip header information for when a gzip + stream is requested by deflateInit2(). deflateSetHeader() may be called + after deflateInit2() or deflateReset() and before the first call of + deflate(). The text, time, os, extra field, name, and comment information + in the provided gz_header structure are written to the gzip header (xflag is + ignored -- the extra flags are set according to the compression level). The + caller must assure that, if not NULL, name and comment are terminated with + a zero byte, and that if extra is not NULL, that extra_len bytes are + available there. If hcrc is true, a gzip header crc is included. Note that + the current versions of the command-line version of gzip (up through version + 1.3.x) do not support header crc's, and will report that it is a "multi-part + gzip file" and give up. + + If deflateSetHeader is not used, the default gzip header has text false, + the time set to zero, and os set to the current operating system, with no + extra, name, or comment fields. The gzip header is returned to the default + state by deflateReset(). + + deflateSetHeader returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent. +*/ + +/* +Z_EXTERN int Z_EXPORT inflateInit2(z_stream *strm, int windowBits); + + This is another version of inflateInit with an extra parameter. The + fields next_in, avail_in, zalloc, zfree and opaque must be initialized + before by the caller. + + The windowBits parameter is the base two logarithm of the maximum window + size (the size of the history buffer). It should be in the range 8..15 for + this version of the library. The default value is 15 if inflateInit is used + instead. windowBits must be greater than or equal to the windowBits value + provided to deflateInit2() while compressing, or it must be equal to 15 if + deflateInit2() was not used. If a compressed stream with a larger window + size is given as input, inflate() will return with the error code + Z_DATA_ERROR instead of trying to allocate a larger window. + + windowBits can also be zero to request that inflate use the window size in + the zlib header of the compressed stream. + + windowBits can also be -8..-15 for raw inflate. In this case, -windowBits + determines the window size. inflate() will then process raw deflate data, + not looking for a zlib or gzip header, not generating a check value, and not + looking for any check values for comparison at the end of the stream. This + is for use with other formats that use the deflate compressed data format + such as zip. Those formats provide their own check values. If a custom + format is developed using the raw deflate format for compressed data, it is + recommended that a check value such as an Adler-32 or a CRC-32 be applied to + the uncompressed data as is done in the zlib, gzip, and zip formats. For + most applications, the zlib format should be used as is. Note that comments + above on the use in deflateInit2() applies to the magnitude of windowBits. + + windowBits can also be greater than 15 for optional gzip decoding. Add + 32 to windowBits to enable zlib and gzip decoding with automatic header + detection, or add 16 to decode only the gzip format (the zlib format will + return a Z_DATA_ERROR). If a gzip stream is being decoded, strm->adler is a + CRC-32 instead of an Adler-32. Unlike the gunzip utility and gzread() (see + below), inflate() will *not* automatically decode concatenated gzip members. + inflate() will return Z_STREAM_END at the end of the gzip member. The state + would need to be reset to continue decoding a subsequent gzip member. This + *must* be done if there is more data after a gzip member, in order for the + decompression to be compliant with the gzip standard (RFC 1952). + + inflateInit2 returns Z_OK if success, Z_MEM_ERROR if there was not enough + memory, Z_VERSION_ERROR if the zlib library version is incompatible with the + version assumed by the caller, or Z_STREAM_ERROR if the parameters are + invalid, such as a null pointer to the structure. msg is set to null if + there is no error message. inflateInit2 does not perform any decompression + apart from possibly reading the zlib header if present: actual decompression + will be done by inflate(). (So next_in and avail_in may be modified, but + next_out and avail_out are unused and unchanged.) The current implementation + of inflateInit2() does not process any header information -- that is + deferred until inflate() is called. +*/ + +Z_EXTERN int Z_EXPORT inflateSetDictionary(z_stream *strm, const unsigned char *dictionary, unsigned int dictLength); +/* + Initializes the decompression dictionary from the given uncompressed byte + sequence. This function must be called immediately after a call of inflate, + if that call returned Z_NEED_DICT. The dictionary chosen by the compressor + can be determined from the Adler-32 value returned by that call of inflate. + The compressor and decompressor must use exactly the same dictionary (see + deflateSetDictionary). For raw inflate, this function can be called at any + time to set the dictionary. If the provided dictionary is smaller than the + window and there is already data in the window, then the provided dictionary + will amend what's there. The application must ensure that the dictionary + that was used for compression is provided. + + inflateSetDictionary returns Z_OK if success, Z_STREAM_ERROR if a + parameter is invalid (e.g. dictionary being NULL) or the stream state is + inconsistent, Z_DATA_ERROR if the given dictionary doesn't match the + expected one (incorrect Adler-32 value). inflateSetDictionary does not + perform any decompression: this will be done by subsequent calls of + inflate(). +*/ + +Z_EXTERN int Z_EXPORT inflateGetDictionary(z_stream *strm, unsigned char *dictionary, unsigned int *dictLength); +/* + Returns the sliding dictionary being maintained by inflate. dictLength is + set to the number of bytes in the dictionary, and that many bytes are copied + to dictionary. dictionary must have enough space, where 32768 bytes is + always enough. If inflateGetDictionary() is called with dictionary equal to + NULL, then only the dictionary length is returned, and nothing is copied. + Similarly, if dictLength is NULL, then it is not set. + + inflateGetDictionary returns Z_OK on success, or Z_STREAM_ERROR if the + stream state is inconsistent. +*/ + +Z_EXTERN int Z_EXPORT inflateSync(z_stream *strm); +/* + Skips invalid compressed data until a possible full flush point (see above + for the description of deflate with Z_FULL_FLUSH) can be found, or until all + available input is skipped. No output is provided. + + inflateSync searches for a 00 00 FF FF pattern in the compressed data. + All full flush points have this pattern, but not all occurrences of this + pattern are full flush points. + + inflateSync returns Z_OK if a possible full flush point has been found, + Z_BUF_ERROR if no more input was provided, Z_DATA_ERROR if no flush point + has been found, or Z_STREAM_ERROR if the stream structure was inconsistent. + In the success case, the application may save the current value of + total_in which indicates where valid compressed data was found. In the + error case, the application may repeatedly call inflateSync, providing more + input each time, until success or end of the input data. +*/ + +Z_EXTERN int Z_EXPORT inflateCopy(z_stream *dest, z_stream *source); +/* + Sets the destination stream as a complete copy of the source stream. + + This function can be useful when randomly accessing a large stream. The + first pass through the stream can periodically record the inflate state, + allowing restarting inflate at those points when randomly accessing the + stream. + + inflateCopy returns Z_OK if success, Z_MEM_ERROR if there was not + enough memory, Z_STREAM_ERROR if the source stream state was inconsistent + (such as zalloc being NULL). msg is left unchanged in both source and + destination. +*/ + +Z_EXTERN int Z_EXPORT inflateReset(z_stream *strm); +/* + This function is equivalent to inflateEnd followed by inflateInit, + but does not free and reallocate the internal decompression state. The + stream will keep attributes that may have been set by inflateInit2. + total_in, total_out, adler, and msg are initialized. + + inflateReset returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent (such as zalloc or state being NULL). +*/ + +Z_EXTERN int Z_EXPORT inflateReset2(z_stream *strm, int windowBits); +/* + This function is the same as inflateReset, but it also permits changing + the wrap and window size requests. The windowBits parameter is interpreted + the same as it is for inflateInit2. If the window size is changed, then the + memory allocated for the window is freed, and the window will be reallocated + by inflate() if needed. + + inflateReset2 returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent (such as zalloc or state being NULL), or if + the windowBits parameter is invalid. +*/ + +Z_EXTERN int Z_EXPORT inflatePrime(z_stream *strm, int bits, int value); +/* + This function inserts bits in the inflate input stream. The intent is + that this function is used to start inflating at a bit position in the + middle of a byte. The provided bits will be used before any bytes are used + from next_in. This function should only be used with raw inflate, and + should be used before the first inflate() call after inflateInit2() or + inflateReset(). bits must be less than or equal to 16, and that many of the + least significant bits of value will be inserted in the input. + + If bits is negative, then the input stream bit buffer is emptied. Then + inflatePrime() can be called again to put bits in the buffer. This is used + to clear out bits leftover after feeding inflate a block description prior + to feeding inflate codes. + + inflatePrime returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent. +*/ + +Z_EXTERN long Z_EXPORT inflateMark(z_stream *strm); +/* + This function returns two values, one in the lower 16 bits of the return + value, and the other in the remaining upper bits, obtained by shifting the + return value down 16 bits. If the upper value is -1 and the lower value is + zero, then inflate() is currently decoding information outside of a block. + If the upper value is -1 and the lower value is non-zero, then inflate is in + the middle of a stored block, with the lower value equaling the number of + bytes from the input remaining to copy. If the upper value is not -1, then + it is the number of bits back from the current bit position in the input of + the code (literal or length/distance pair) currently being processed. In + that case the lower value is the number of bytes already emitted for that + code. + + A code is being processed if inflate is waiting for more input to complete + decoding of the code, or if it has completed decoding but is waiting for + more output space to write the literal or match data. + + inflateMark() is used to mark locations in the input data for random + access, which may be at bit positions, and to note those cases where the + output of a code may span boundaries of random access blocks. The current + location in the input stream can be determined from avail_in and data_type + as noted in the description for the Z_BLOCK flush parameter for inflate. + + inflateMark returns the value noted above, or -65536 if the provided + source stream state was inconsistent. +*/ + +Z_EXTERN int Z_EXPORT inflateGetHeader(z_stream *strm, gz_headerp head); +/* + inflateGetHeader() requests that gzip header information be stored in the + provided gz_header structure. inflateGetHeader() may be called after + inflateInit2() or inflateReset(), and before the first call of inflate(). + As inflate() processes the gzip stream, head->done is zero until the header + is completed, at which time head->done is set to one. If a zlib stream is + being decoded, then head->done is set to -1 to indicate that there will be + no gzip header information forthcoming. Note that Z_BLOCK or Z_TREES can be + used to force inflate() to return immediately after header processing is + complete and before any actual data is decompressed. + + The text, time, xflags, and os fields are filled in with the gzip header + contents. hcrc is set to true if there is a header CRC. (The header CRC + was valid if done is set to one.) If extra is not NULL, then extra_max + contains the maximum number of bytes to write to extra. Once done is true, + extra_len contains the actual extra field length, and extra contains the + extra field, or that field truncated if extra_max is less than extra_len. + If name is not NULL, then up to name_max characters are written there, + terminated with a zero unless the length is greater than name_max. If + comment is not NULL, then up to comm_max characters are written there, + terminated with a zero unless the length is greater than comm_max. When any + of extra, name, or comment are not NULL and the respective field is not + present in the header, then that field is set to NULL to signal its + absence. This allows the use of deflateSetHeader() with the returned + structure to duplicate the header. However if those fields are set to + allocated memory, then the application will need to save those pointers + elsewhere so that they can be eventually freed. + + If inflateGetHeader is not used, then the header information is simply + discarded. The header is always checked for validity, including the header + CRC if present. inflateReset() will reset the process to discard the header + information. The application would need to call inflateGetHeader() again to + retrieve the header from the next gzip stream. + + inflateGetHeader returns Z_OK if success, or Z_STREAM_ERROR if the source + stream state was inconsistent. +*/ + +/* +Z_EXTERN int Z_EXPORT inflateBackInit (z_stream *strm, int windowBits, unsigned char *window); + + Initialize the internal stream state for decompression using inflateBack() + calls. The fields zalloc, zfree and opaque in strm must be initialized + before the call. If zalloc and zfree are NULL, then the default library- + derived memory allocation routines are used. windowBits is the base two + logarithm of the window size, in the range 8..15. window is a caller + supplied buffer of that size. Except for special applications where it is + assured that deflate was used with small window sizes, windowBits must be 15 + and a 32K byte window must be supplied to be able to decompress general + deflate streams. + + See inflateBack() for the usage of these routines. + + inflateBackInit will return Z_OK on success, Z_STREAM_ERROR if any of + the parameters are invalid, Z_MEM_ERROR if the internal state could not be + allocated, or Z_VERSION_ERROR if the version of the library does not match + the version of the header file. +*/ + +typedef uint32_t (*in_func) (void *, z_const unsigned char * *); +typedef int (*out_func) (void *, unsigned char *, uint32_t); + +Z_EXTERN int Z_EXPORT inflateBack(z_stream *strm, in_func in, void *in_desc, out_func out, void *out_desc); +/* + inflateBack() does a raw inflate with a single call using a call-back + interface for input and output. This is potentially more efficient than + inflate() for file i/o applications, in that it avoids copying between the + output and the sliding window by simply making the window itself the output + buffer. inflate() can be faster on modern CPUs when used with large + buffers. inflateBack() trusts the application to not change the output + buffer passed by the output function, at least until inflateBack() returns. + + inflateBackInit() must be called first to allocate the internal state + and to initialize the state with the user-provided window buffer. + inflateBack() may then be used multiple times to inflate a complete, raw + deflate stream with each call. inflateBackEnd() is then called to free the + allocated state. + + A raw deflate stream is one with no zlib or gzip header or trailer. + This routine would normally be used in a utility that reads zip or gzip + files and writes out uncompressed files. The utility would decode the + header and process the trailer on its own, hence this routine expects only + the raw deflate stream to decompress. This is different from the default + behavior of inflate(), which expects a zlib header and trailer around the + deflate stream. + + inflateBack() uses two subroutines supplied by the caller that are then + called by inflateBack() for input and output. inflateBack() calls those + routines until it reads a complete deflate stream and writes out all of the + uncompressed data, or until it encounters an error. The function's + parameters and return types are defined above in the in_func and out_func + typedefs. inflateBack() will call in(in_desc, &buf) which should return the + number of bytes of provided input, and a pointer to that input in buf. If + there is no input available, in() must return zero -- buf is ignored in that + case -- and inflateBack() will return a buffer error. inflateBack() will + call out(out_desc, buf, len) to write the uncompressed data buf[0..len-1]. + out() should return zero on success, or non-zero on failure. If out() + returns non-zero, inflateBack() will return with an error. Neither in() nor + out() are permitted to change the contents of the window provided to + inflateBackInit(), which is also the buffer that out() uses to write from. + The length written by out() will be at most the window size. Any non-zero + amount of input may be provided by in(). + + For convenience, inflateBack() can be provided input on the first call by + setting strm->next_in and strm->avail_in. If that input is exhausted, then + in() will be called. Therefore strm->next_in must be initialized before + calling inflateBack(). If strm->next_in is NULL, then in() will be called + immediately for input. If strm->next_in is not NULL, then strm->avail_in + must also be initialized, and then if strm->avail_in is not zero, input will + initially be taken from strm->next_in[0 .. strm->avail_in - 1]. + + The in_desc and out_desc parameters of inflateBack() is passed as the + first parameter of in() and out() respectively when they are called. These + descriptors can be optionally used to pass any information that the caller- + supplied in() and out() functions need to do their job. + + On return, inflateBack() will set strm->next_in and strm->avail_in to + pass back any unused input that was provided by the last in() call. The + return values of inflateBack() can be Z_STREAM_END on success, Z_BUF_ERROR + if in() or out() returned an error, Z_DATA_ERROR if there was a format error + in the deflate stream (in which case strm->msg is set to indicate the nature + of the error), or Z_STREAM_ERROR if the stream was not properly initialized. + In the case of Z_BUF_ERROR, an input or output error can be distinguished + using strm->next_in which will be NULL only if in() returned an error. If + strm->next_in is not NULL, then the Z_BUF_ERROR was due to out() returning + non-zero. (in() will always be called before out(), so strm->next_in is + assured to be defined if out() returns non-zero.) Note that inflateBack() + cannot return Z_OK. +*/ + +Z_EXTERN int Z_EXPORT inflateBackEnd(z_stream *strm); +/* + All memory allocated by inflateBackInit() is freed. + + inflateBackEnd() returns Z_OK on success, or Z_STREAM_ERROR if the stream + state was inconsistent. +*/ + +Z_EXTERN unsigned long Z_EXPORT zlibCompileFlags(void); +/* Return flags indicating compile-time options. + + Type sizes, two bits each, 00 = 16 bits, 01 = 32, 10 = 64, 11 = other: + 1.0: size of unsigned int + 3.2: size of unsigned long + 5.4: size of void * (pointer) + 7.6: size of z_off_t + + Compiler, assembler, and debug options: + 8: ZLIB_DEBUG + 9: ASMV or ASMINF -- use ASM code + 10: ZLIB_WINAPI -- exported functions use the WINAPI calling convention + 11: 0 (reserved) + + One-time table building (smaller code, but not thread-safe if true): + 12: BUILDFIXED -- build static block decoding tables when needed (not supported by zlib-ng) + 13: DYNAMIC_CRC_TABLE -- build CRC calculation tables when needed + 14,15: 0 (reserved) + + Library content (indicates missing functionality): + 16: NO_GZCOMPRESS -- gz* functions cannot compress (to avoid linking + deflate code when not needed) + 17: NO_GZIP -- deflate can't write gzip streams, and inflate can't detect + and decode gzip streams (to avoid linking crc code) + 18-19: 0 (reserved) + + Operation variations (changes in library functionality): + 20: PKZIP_BUG_WORKAROUND -- slightly more permissive inflate + 21: FASTEST -- deflate algorithm with only one, lowest compression level + 22,23: 0 (reserved) + + The sprintf variant used by gzprintf (zero is best): + 24: 0 = vs*, 1 = s* -- 1 means limited to 20 arguments after the format + 25: 0 = *nprintf, 1 = *printf -- 1 means gzprintf() not secure! + 26: 0 = returns value, 1 = void -- 1 means inferred string length returned + + Remainder: + 27-31: 0 (reserved) + */ + + +#ifndef Z_SOLO + + /* utility functions */ + +/* + The following utility functions are implemented on top of the basic + stream-oriented functions. To simplify the interface, some default options + are assumed (compression level and memory usage, standard memory allocation + functions). The source code of these utility functions can be modified if + you need special options. +*/ + +Z_EXTERN int Z_EXPORT compress(unsigned char *dest, unsigned long *destLen, const unsigned char *source, unsigned long sourceLen); +/* + Compresses the source buffer into the destination buffer. sourceLen is + the byte length of the source buffer. Upon entry, destLen is the total size + of the destination buffer, which must be at least the value returned by + compressBound(sourceLen). Upon exit, destLen is the actual size of the + compressed data. compress() is equivalent to compress2() with a level + parameter of Z_DEFAULT_COMPRESSION. + + compress returns Z_OK if success, Z_MEM_ERROR if there was not + enough memory, Z_BUF_ERROR if there was not enough room in the output + buffer. +*/ + +Z_EXTERN int Z_EXPORT compress2(unsigned char *dest, unsigned long *destLen, const unsigned char *source, + unsigned long sourceLen, int level); +/* + Compresses the source buffer into the destination buffer. The level + parameter has the same meaning as in deflateInit. sourceLen is the byte + length of the source buffer. Upon entry, destLen is the total size of the + destination buffer, which must be at least the value returned by + compressBound(sourceLen). Upon exit, destLen is the actual size of the + compressed data. + + compress2 returns Z_OK if success, Z_MEM_ERROR if there was not enough + memory, Z_BUF_ERROR if there was not enough room in the output buffer, + Z_STREAM_ERROR if the level parameter is invalid. +*/ + +Z_EXTERN unsigned long Z_EXPORT compressBound(unsigned long sourceLen); +/* + compressBound() returns an upper bound on the compressed size after + compress() or compress2() on sourceLen bytes. It would be used before a + compress() or compress2() call to allocate the destination buffer. +*/ + +Z_EXTERN int Z_EXPORT uncompress(unsigned char *dest, unsigned long *destLen, const unsigned char *source, unsigned long sourceLen); +/* + Decompresses the source buffer into the destination buffer. sourceLen is + the byte length of the source buffer. Upon entry, destLen is the total size + of the destination buffer, which must be large enough to hold the entire + uncompressed data. (The size of the uncompressed data must have been saved + previously by the compressor and transmitted to the decompressor by some + mechanism outside the scope of this compression library.) Upon exit, destLen + is the actual size of the uncompressed data. + + uncompress returns Z_OK if success, Z_MEM_ERROR if there was not + enough memory, Z_BUF_ERROR if there was not enough room in the output + buffer, or Z_DATA_ERROR if the input data was corrupted or incomplete. In + the case where there is not enough room, uncompress() will fill the output + buffer with the uncompressed data up to that point. +*/ + + +Z_EXTERN int Z_EXPORT uncompress2 (unsigned char *dest, unsigned long *destLen, + const unsigned char *source, unsigned long *sourceLen); +/* + Same as uncompress, except that sourceLen is a pointer, where the + length of the source is *sourceLen. On return, *sourceLen is the number of + source bytes consumed. +*/ + + + /* gzip file access functions */ + +/* + This library supports reading and writing files in gzip (.gz) format with + an interface similar to that of stdio, using the functions that start with + "gz". The gzip format is different from the zlib format. gzip is a gzip + wrapper, documented in RFC 1952, wrapped around a deflate stream. +*/ + +typedef struct gzFile_s *gzFile; /* semi-opaque gzip file descriptor */ + +/* +Z_EXTERN gzFile Z_EXPORT gzopen(const char *path, const char *mode); + + Open the gzip (.gz) file at path for reading and decompressing, or + compressing and writing. The mode parameter is as in fopen ("rb" or "wb") + but can also include a compression level ("wb9") or a strategy: 'f' for + filtered data as in "wb6f", 'h' for Huffman-only compression as in "wb1h", + 'R' for run-length encoding as in "wb1R", or 'F' for fixed code compression + as in "wb9F". (See the description of deflateInit2 for more information + about the strategy parameter.) 'T' will request transparent writing or + appending with no compression and not using the gzip format. + + "a" can be used instead of "w" to request that the gzip stream that will + be written be appended to the file. "+" will result in an error, since + reading and writing to the same gzip file is not supported. The addition of + "x" when writing will create the file exclusively, which fails if the file + already exists. On systems that support it, the addition of "e" when + reading or writing will set the flag to close the file on an execve() call. + + These functions, as well as gzip, will read and decode a sequence of gzip + streams in a file. The append function of gzopen() can be used to create + such a file. (Also see gzflush() for another way to do this.) When + appending, gzopen does not test whether the file begins with a gzip stream, + nor does it look for the end of the gzip streams to begin appending. gzopen + will simply append a gzip stream to the existing file. + + gzopen can be used to read a file which is not in gzip format; in this + case gzread will directly read from the file without decompression. When + reading, this will be detected automatically by looking for the magic two- + byte gzip header. + + gzopen returns NULL if the file could not be opened, if there was + insufficient memory to allocate the gzFile state, or if an invalid mode was + specified (an 'r', 'w', or 'a' was not provided, or '+' was provided). + errno can be checked to determine if the reason gzopen failed was that the + file could not be opened. +*/ + +Z_EXTERN gzFile Z_EXPORT gzdopen(int fd, const char *mode); +/* + Associate a gzFile with the file descriptor fd. File descriptors are + obtained from calls like open, dup, creat, pipe or fileno (if the file has + been previously opened with fopen). The mode parameter is as in gzopen. + + The next call of gzclose on the returned gzFile will also close the file + descriptor fd, just like fclose(fdopen(fd, mode)) closes the file descriptor + fd. If you want to keep fd open, use fd = dup(fd_keep); gz = gzdopen(fd, + mode);. The duplicated descriptor should be saved to avoid a leak, since + gzdopen does not close fd if it fails. If you are using fileno() to get the + file descriptor from a FILE *, then you will have to use dup() to avoid + double-close()ing the file descriptor. Both gzclose() and fclose() will + close the associated file descriptor, so they need to have different file + descriptors. + + gzdopen returns NULL if there was insufficient memory to allocate the + gzFile state, if an invalid mode was specified (an 'r', 'w', or 'a' was not + provided, or '+' was provided), or if fd is -1. The file descriptor is not + used until the next gz* read, write, seek, or close operation, so gzdopen + will not detect if fd is invalid (unless fd is -1). +*/ + +Z_EXTERN int Z_EXPORT gzbuffer(gzFile file, unsigned size); +/* + Set the internal buffer size used by this library's functions for file to + size. The default buffer size is 8192 bytes. This function must be called + after gzopen() or gzdopen(), and before any other calls that read or write + the file. The buffer memory allocation is always deferred to the first read + or write. Three times that size in buffer space is allocated. A larger + buffer size of, for example, 64K or 128K bytes will noticeably increase the + speed of decompression (reading). + + The new buffer size also affects the maximum length for gzprintf(). + + gzbuffer() returns 0 on success, or -1 on failure, such as being called + too late. +*/ + +Z_EXTERN int Z_EXPORT gzsetparams(gzFile file, int level, int strategy); +/* + Dynamically update the compression level and strategy for file. See the + description of deflateInit2 for the meaning of these parameters. Previously + provided data is flushed before applying the parameter changes. + + gzsetparams returns Z_OK if success, Z_STREAM_ERROR if the file was not + opened for writing, Z_ERRNO if there is an error writing the flushed data, + or Z_MEM_ERROR if there is a memory allocation error. +*/ + +Z_EXTERN int Z_EXPORT gzread(gzFile file, void *buf, unsigned len); +/* + Read and decompress up to len uncompressed bytes from file into buf. If + the input file is not in gzip format, gzread copies the given number of + bytes into the buffer directly from the file. + + After reaching the end of a gzip stream in the input, gzread will continue + to read, looking for another gzip stream. Any number of gzip streams may be + concatenated in the input file, and will all be decompressed by gzread(). + If something other than a gzip stream is encountered after a gzip stream, + that remaining trailing garbage is ignored (and no error is returned). + + gzread can be used to read a gzip file that is being concurrently written. + Upon reaching the end of the input, gzread will return with the available + data. If the error code returned by gzerror is Z_OK or Z_BUF_ERROR, then + gzclearerr can be used to clear the end of file indicator in order to permit + gzread to be tried again. Z_OK indicates that a gzip stream was completed + on the last gzread. Z_BUF_ERROR indicates that the input file ended in the + middle of a gzip stream. Note that gzread does not return -1 in the event + of an incomplete gzip stream. This error is deferred until gzclose(), which + will return Z_BUF_ERROR if the last gzread ended in the middle of a gzip + stream. Alternatively, gzerror can be used before gzclose to detect this + case. + + gzread returns the number of uncompressed bytes actually read, less than + len for end of file, or -1 for error. If len is too large to fit in an int, + then nothing is read, -1 is returned, and the error state is set to + Z_STREAM_ERROR. +*/ + +Z_EXTERN size_t Z_EXPORT gzfread (void *buf, size_t size, size_t nitems, gzFile file); +/* + Read and decompress up to nitems items of size size from file into buf, + otherwise operating as gzread() does. This duplicates the interface of + stdio's fread(), with size_t request and return types. If the library + defines size_t, then z_size_t is identical to size_t. If not, then z_size_t + is an unsigned integer type that can contain a pointer. + + gzfread() returns the number of full items read of size size, or zero if + the end of the file was reached and a full item could not be read, or if + there was an error. gzerror() must be consulted if zero is returned in + order to determine if there was an error. If the multiplication of size and + nitems overflows, i.e. the product does not fit in a size_t, then nothing + is read, zero is returned, and the error state is set to Z_STREAM_ERROR. + + In the event that the end of file is reached and only a partial item is + available at the end, i.e. the remaining uncompressed data length is not a + multiple of size, then the final partial item is nevertheless read into buf + and the end-of-file flag is set. The length of the partial item read is not + provided, but could be inferred from the result of gztell(). This behavior + is the same as the behavior of fread() implementations in common libraries, + but it prevents the direct use of gzfread() to read a concurrently written + file, resetting and retrying on end-of-file, when size is not 1. +*/ + +Z_EXTERN int Z_EXPORT gzwrite(gzFile file, void const *buf, unsigned len); +/* + Compress and write the len uncompressed bytes at buf to file. gzwrite + returns the number of uncompressed bytes written or 0 in case of error. +*/ + +Z_EXTERN size_t Z_EXPORT gzfwrite(void const *buf, size_t size, size_t nitems, gzFile file); +/* + Compress and write nitems items of size size from buf to file, duplicating + the interface of stdio's fwrite(), with size_t request and return types. + + gzfwrite() returns the number of full items written of size size, or zero + if there was an error. If the multiplication of size and nitems overflows, + i.e. the product does not fit in a size_t, then nothing is written, zero + is returned, and the error state is set to Z_STREAM_ERROR. +*/ + +Z_EXTERN int Z_EXPORTVA gzprintf(gzFile file, const char *format, ...); +/* + Convert, format, compress, and write the arguments (...) to file under + control of the string format, as in fprintf. gzprintf returns the number of + uncompressed bytes actually written, or a negative zlib error code in case + of error. The number of uncompressed bytes written is limited to 8191, or + one less than the buffer size given to gzbuffer(). The caller should assure + that this limit is not exceeded. If it is exceeded, then gzprintf() will + return an error (0) with nothing written. In this case, there may also be a + buffer overflow with unpredictable consequences, which is possible only if + zlib was compiled with the insecure functions sprintf() or vsprintf(), + because the secure snprintf() or vsnprintf() functions were not available. + This can be determined using zlibCompileFlags(). +*/ + +Z_EXTERN int Z_EXPORT gzputs(gzFile file, const char *s); +/* + Compress and write the given null-terminated string s to file, excluding + the terminating null character. + + gzputs returns the number of characters written, or -1 in case of error. +*/ + +Z_EXTERN char * Z_EXPORT gzgets(gzFile file, char *buf, int len); +/* + Read and decompress bytes from file into buf, until len-1 characters are + read, or until a newline character is read and transferred to buf, or an + end-of-file condition is encountered. If any characters are read or if len + is one, the string is terminated with a null character. If no characters + are read due to an end-of-file or len is less than one, then the buffer is + left untouched. + + gzgets returns buf which is a null-terminated string, or it returns NULL + for end-of-file or in case of error. If there was an error, the contents at + buf are indeterminate. +*/ + +Z_EXTERN int Z_EXPORT gzputc(gzFile file, int c); +/* + Compress and write c, converted to an unsigned char, into file. gzputc + returns the value that was written, or -1 in case of error. +*/ + +Z_EXTERN int Z_EXPORT gzgetc(gzFile file); +/* + Read and decompress one byte from file. gzgetc returns this byte or -1 + in case of end of file or error. This is implemented as a macro for speed. + As such, it does not do all of the checking the other functions do. I.e. + it does not check to see if file is NULL, nor whether the structure file + points to has been clobbered or not. +*/ + +Z_EXTERN int Z_EXPORT gzungetc(int c, gzFile file); +/* + Push c back onto the stream for file to be read as the first character on + the next read. At least one character of push-back is always allowed. + gzungetc() returns the character pushed, or -1 on failure. gzungetc() will + fail if c is -1, and may fail if a character has been pushed but not read + yet. If gzungetc is used immediately after gzopen or gzdopen, at least the + output buffer size of pushed characters is allowed. (See gzbuffer above.) + The pushed character will be discarded if the stream is repositioned with + gzseek() or gzrewind(). +*/ + +Z_EXTERN int Z_EXPORT gzflush(gzFile file, int flush); +/* + Flush all pending output to file. The parameter flush is as in the + deflate() function. The return value is the zlib error number (see function + gzerror below). gzflush is only permitted when writing. + + If the flush parameter is Z_FINISH, the remaining data is written and the + gzip stream is completed in the output. If gzwrite() is called again, a new + gzip stream will be started in the output. gzread() is able to read such + concatenated gzip streams. + + gzflush should be called only when strictly necessary because it will + degrade compression if called too often. +*/ + +/* +Z_EXTERN z_off_t Z_EXPORT gzseek (gzFile file, z_off_t offset, int whence); + + Set the starting position to offset relative to whence for the next gzread + or gzwrite on file. The offset represents a number of bytes in the + uncompressed data stream. The whence parameter is defined as in lseek(2); + the value SEEK_END is not supported. + + If the file is opened for reading, this function is emulated but can be + extremely slow. If the file is opened for writing, only forward seeks are + supported; gzseek then compresses a sequence of zeroes up to the new + starting position. + + gzseek returns the resulting offset location as measured in bytes from + the beginning of the uncompressed stream, or -1 in case of error, in + particular if the file is opened for writing and the new starting position + would be before the current position. +*/ + +Z_EXTERN int Z_EXPORT gzrewind(gzFile file); +/* + Rewind file. This function is supported only for reading. + + gzrewind(file) is equivalent to (int)gzseek(file, 0L, SEEK_SET). +*/ + +/* +Z_EXTERN z_off_t Z_EXPORT gztell(gzFile file); + + Return the starting position for the next gzread or gzwrite on file. + This position represents a number of bytes in the uncompressed data stream, + and is zero when starting, even if appending or reading a gzip stream from + the middle of a file using gzdopen(). + + gztell(file) is equivalent to gzseek(file, 0L, SEEK_CUR) +*/ + +/* +Z_EXTERN z_off_t Z_EXPORT gzoffset(gzFile file); + + Return the current compressed (actual) read or write offset of file. This + offset includes the count of bytes that precede the gzip stream, for example + when appending or when using gzdopen() for reading. When reading, the + offset does not include as yet unused buffered input. This information can + be used for a progress indicator. On error, gzoffset() returns -1. +*/ + +Z_EXTERN int Z_EXPORT gzeof(gzFile file); +/* + Return true (1) if the end-of-file indicator for file has been set while + reading, false (0) otherwise. Note that the end-of-file indicator is set + only if the read tried to go past the end of the input, but came up short. + Therefore, just like feof(), gzeof() may return false even if there is no + more data to read, in the event that the last read request was for the exact + number of bytes remaining in the input file. This will happen if the input + file size is an exact multiple of the buffer size. + + If gzeof() returns true, then the read functions will return no more data, + unless the end-of-file indicator is reset by gzclearerr() and the input file + has grown since the previous end of file was detected. +*/ + +Z_EXTERN int Z_EXPORT gzdirect(gzFile file); +/* + Return true (1) if file is being copied directly while reading, or false + (0) if file is a gzip stream being decompressed. + + If the input file is empty, gzdirect() will return true, since the input + does not contain a gzip stream. + + If gzdirect() is used immediately after gzopen() or gzdopen() it will + cause buffers to be allocated to allow reading the file to determine if it + is a gzip file. Therefore if gzbuffer() is used, it should be called before + gzdirect(). + + When writing, gzdirect() returns true (1) if transparent writing was + requested ("wT" for the gzopen() mode), or false (0) otherwise. (Note: + gzdirect() is not needed when writing. Transparent writing must be + explicitly requested, so the application already knows the answer. When + linking statically, using gzdirect() will include all of the zlib code for + gzip file reading and decompression, which may not be desired.) +*/ + +Z_EXTERN int Z_EXPORT gzclose(gzFile file); +/* + Flush all pending output for file, if necessary, close file and + deallocate the (de)compression state. Note that once file is closed, you + cannot call gzerror with file, since its structures have been deallocated. + gzclose must not be called more than once on the same file, just as free + must not be called more than once on the same allocation. + + gzclose will return Z_STREAM_ERROR if file is not valid, Z_ERRNO on a + file operation error, Z_MEM_ERROR if out of memory, Z_BUF_ERROR if the + last read ended in the middle of a gzip stream, or Z_OK on success. +*/ + +Z_EXTERN int Z_EXPORT gzclose_r(gzFile file); +Z_EXTERN int Z_EXPORT gzclose_w(gzFile file); +/* + Same as gzclose(), but gzclose_r() is only for use when reading, and + gzclose_w() is only for use when writing or appending. The advantage to + using these instead of gzclose() is that they avoid linking in zlib + compression or decompression code that is not used when only reading or only + writing respectively. If gzclose() is used, then both compression and + decompression code will be included the application when linking to a static + zlib library. +*/ + +Z_EXTERN const char * Z_EXPORT gzerror(gzFile file, int *errnum); +/* + Return the error message for the last error which occurred on file. + errnum is set to zlib error number. If an error occurred in the file system + and not in the compression library, errnum is set to Z_ERRNO and the + application may consult errno to get the exact error code. + + The application must not modify the returned string. Future calls to + this function may invalidate the previously returned string. If file is + closed, then the string previously returned by gzerror will no longer be + available. + + gzerror() should be used to distinguish errors from end-of-file for those + functions above that do not distinguish those cases in their return values. +*/ + +Z_EXTERN void Z_EXPORT gzclearerr(gzFile file); +/* + Clear the error and end-of-file flags for file. This is analogous to the + clearerr() function in stdio. This is useful for continuing to read a gzip + file that is being written concurrently. +*/ + +#endif + + /* checksum functions */ + +/* + These functions are not related to compression but are exported + anyway because they might be useful in applications using the compression + library. +*/ + +Z_EXTERN unsigned long Z_EXPORT adler32(unsigned long adler, const unsigned char *buf, unsigned int len); +/* + Update a running Adler-32 checksum with the bytes buf[0..len-1] and + return the updated checksum. An Adler-32 value is in the range of a 32-bit + unsigned integer. If buf is Z_NULL, this function returns the required + initial value for the checksum. + + An Adler-32 checksum is almost as reliable as a CRC-32 but can be computed + much faster. + + Usage example: + + uint32_t adler = adler32(0L, NULL, 0); + + while (read_buffer(buffer, length) != EOF) { + adler = adler32(adler, buffer, length); + } + if (adler != original_adler) error(); +*/ + +Z_EXTERN unsigned long Z_EXPORT adler32_z(unsigned long adler, const unsigned char *buf, size_t len); +/* + Same as adler32(), but with a size_t length. +*/ + +/* +Z_EXTERN unsigned long Z_EXPORT adler32_combine(unsigned long adler1, unsigned long adler2, z_off_t len2); + + Combine two Adler-32 checksums into one. For two sequences of bytes, seq1 + and seq2 with lengths len1 and len2, Adler-32 checksums were calculated for + each, adler1 and adler2. adler32_combine() returns the Adler-32 checksum of + seq1 and seq2 concatenated, requiring only adler1, adler2, and len2. Note + that the z_off_t type (like off_t) is a signed integer. If len2 is + negative, the result has no meaning or utility. +*/ + +Z_EXTERN unsigned long Z_EXPORT crc32(unsigned long crc, const unsigned char *buf, unsigned int len); +/* + Update a running CRC-32 with the bytes buf[0..len-1] and return the + updated CRC-32. A CRC-32 value is in the range of a 32-bit unsigned integer. + If buf is Z_NULL, this function returns the required initial value for the + crc. Pre- and post-conditioning (one's complement) is performed within this + function so it shouldn't be done by the application. + + Usage example: + + uint32_t crc = crc32(0L, NULL, 0); + + while (read_buffer(buffer, length) != EOF) { + crc = crc32(crc, buffer, length); + } + if (crc != original_crc) error(); +*/ + +Z_EXTERN unsigned long Z_EXPORT crc32_z(unsigned long crc, const unsigned char *buf, size_t len); +/* + Same as crc32(), but with a size_t length. +*/ + +/* +Z_EXTERN unsigned long Z_EXPORT crc32_combine(unsigned long crc1, unsigned long crc2, z_off64_t len2); + + Combine two CRC-32 check values into one. For two sequences of bytes, + seq1 and seq2 with lengths len1 and len2, CRC-32 check values were + calculated for each, crc1 and crc2. crc32_combine() returns the CRC-32 + check value of seq1 and seq2 concatenated, requiring only crc1, crc2, and + len2. len2 must be non-negative. +*/ + +/* +Z_EXTERN unsigned long Z_EXPORT crc32_combine_gen(z_off_t len2); + + Return the operator corresponding to length len2, to be used with + crc32_combine_op(). len2 must be non-negative. +*/ + +Z_EXTERN unsigned long Z_EXPORT crc32_combine_op(unsigned long crc1, unsigned long crc2, + const unsigned long op); +/* + Give the same result as crc32_combine(), using op in place of len2. op is + is generated from len2 by crc32_combine_gen(). This will be faster than + crc32_combine() if the generated op is used more than once. +*/ + + + /* various hacks, don't look :) */ + +/* deflateInit and inflateInit are macros to allow checking the zlib version + * and the compiler's view of z_stream: + */ +Z_EXTERN int Z_EXPORT deflateInit_(z_stream *strm, int level, const char *version, int stream_size); +Z_EXTERN int Z_EXPORT inflateInit_(z_stream *strm, const char *version, int stream_size); +Z_EXTERN int Z_EXPORT deflateInit2_(z_stream *strm, int level, int method, int windowBits, int memLevel, + int strategy, const char *version, int stream_size); +Z_EXTERN int Z_EXPORT inflateInit2_(z_stream *strm, int windowBits, const char *version, int stream_size); +Z_EXTERN int Z_EXPORT inflateBackInit_(z_stream *strm, int windowBits, unsigned char *window, + const char *version, int stream_size); +#define deflateInit(strm, level) deflateInit_((strm), (level), ZLIB_VERSION, (int)sizeof(z_stream)) +#define inflateInit(strm) inflateInit_((strm), ZLIB_VERSION, (int)sizeof(z_stream)) +#define deflateInit2(strm, level, method, windowBits, memLevel, strategy) \ + deflateInit2_((strm), (level), (method), (windowBits), (memLevel), \ + (strategy), ZLIB_VERSION, (int)sizeof(z_stream)) +#define inflateInit2(strm, windowBits) inflateInit2_((strm), (windowBits), ZLIB_VERSION, (int)sizeof(z_stream)) +#define inflateBackInit(strm, windowBits, window) \ + inflateBackInit_((strm), (windowBits), (window), ZLIB_VERSION, (int)sizeof(z_stream)) + + +#ifndef Z_SOLO +/* gzgetc() macro and its supporting function and exposed data structure. Note + * that the real internal state is much larger than the exposed structure. + * This abbreviated structure exposes just enough for the gzgetc() macro. The + * user should not mess with these exposed elements, since their names or + * behavior could change in the future, perhaps even capriciously. They can + * only be used by the gzgetc() macro. You have been warned. + */ +struct gzFile_s { + unsigned have; + unsigned char *next; + z_off64_t pos; +}; +Z_EXTERN int Z_EXPORT gzgetc_(gzFile file); /* backward compatibility */ +# define gzgetc(g) ((g)->have ? ((g)->have--, (g)->pos++, *((g)->next)++) : (gzgetc)(g)) + +/* provide 64-bit offset functions if _LARGEFILE64_SOURCE defined, and/or + * change the regular functions to 64 bits if _FILE_OFFSET_BITS is 64 (if + * both are true, the application gets the *64 functions, and the regular + * functions are changed to 64 bits) -- in case these are set on systems + * without large file support, _LFS64_LARGEFILE must also be true + */ +#ifdef Z_LARGE64 + Z_EXTERN gzFile Z_EXPORT gzopen64(const char *, const char *); + Z_EXTERN z_off64_t Z_EXPORT gzseek64(gzFile, z_off64_t, int); + Z_EXTERN z_off64_t Z_EXPORT gztell64(gzFile); + Z_EXTERN z_off64_t Z_EXPORT gzoffset64(gzFile); + Z_EXTERN unsigned long Z_EXPORT adler32_combine64(unsigned long, unsigned long, z_off64_t); + Z_EXTERN unsigned long Z_EXPORT crc32_combine64(unsigned long, unsigned long, z_off64_t); + Z_EXTERN unsigned long Z_EXPORT crc32_combine_gen64(z_off64_t); +#endif +#endif + +#if !defined(Z_SOLO) && !defined(Z_INTERNAL) && defined(Z_WANT64) +# define gzopen gzopen64 +# define gzseek gzseek64 +# define gztell gztell64 +# define gzoffset gzoffset64 +# define adler32_combine adler32_combine64 +# define crc32_combine crc32_combine64 +# define crc32_combine_gen crc32_combine_gen64 +# ifndef Z_LARGE64 + Z_EXTERN gzFile Z_EXPORT gzopen64(const char *, const char *); + Z_EXTERN z_off_t Z_EXPORT gzseek64(gzFile, z_off_t, int); + Z_EXTERN z_off_t Z_EXPORT gztell64(gzFile); + Z_EXTERN z_off_t Z_EXPORT gzoffset64(gzFile); + Z_EXTERN unsigned long Z_EXPORT adler32_combine64(unsigned long, unsigned long, z_off_t); + Z_EXTERN unsigned long Z_EXPORT crc32_combine64(unsigned long, unsigned long, z_off_t); + Z_EXTERN unsigned long Z_EXPORT crc32_combine_gen64(z_off64_t); +# endif +#else +# ifndef Z_SOLO + Z_EXTERN gzFile Z_EXPORT gzopen(const char *, const char *); + Z_EXTERN z_off_t Z_EXPORT gzseek(gzFile, z_off_t, int); + Z_EXTERN z_off_t Z_EXPORT gztell(gzFile); + Z_EXTERN z_off_t Z_EXPORT gzoffset(gzFile); +# endif + Z_EXTERN unsigned long Z_EXPORT adler32_combine(unsigned long, unsigned long, z_off_t); + Z_EXTERN unsigned long Z_EXPORT crc32_combine(unsigned long, unsigned long, z_off_t); + Z_EXTERN unsigned long Z_EXPORT crc32_combine_gen(z_off_t); +#endif + +/* undocumented functions */ +Z_EXTERN const char * Z_EXPORT zError (int); +Z_EXTERN int Z_EXPORT inflateSyncPoint (z_stream *); +Z_EXTERN const uint32_t * Z_EXPORT get_crc_table (void); +Z_EXTERN int Z_EXPORT inflateUndermine (z_stream *, int); +Z_EXTERN int Z_EXPORT inflateValidate (z_stream *, int); +Z_EXTERN unsigned long Z_EXPORT inflateCodesUsed (z_stream *); +Z_EXTERN int Z_EXPORT inflateResetKeep (z_stream *); +Z_EXTERN int Z_EXPORT deflateResetKeep (z_stream *); + +#ifndef Z_SOLO +#if defined(_WIN32) + Z_EXTERN gzFile Z_EXPORT gzopen_w(const wchar_t *path, const char *mode); +#endif +Z_EXTERN int Z_EXPORTVA gzvprintf(gzFile file, const char *format, va_list va); +#endif + +#ifdef __cplusplus +} +#endif + +#endif /* ZLIB_H_ */