diff --git a/src/weirdperformance/AllocBench/AllocBench.lua b/src/weirdperformance/AllocBench/AllocBench.lua index 0273ddf..002522c 100644 --- a/src/weirdperformance/AllocBench/AllocBench.lua +++ b/src/weirdperformance/AllocBench/AllocBench.lua @@ -101,3 +101,395 @@ end SLASH_ALLOCBENCH1 = "/allocbench" SlashCmdList["ALLOCBENCH"] = run_bench + +-- ========================================================================= +-- GC stress test: /gcstress [size_k] [churn_pct] [duration] +-- Builds a large live heap then churns a fraction of it each frame. +-- size_k: thousands of live objects to maintain (default 200 = 200K objects) +-- churn_pct: percent of live set to replace per frame (default 5) +-- duration: seconds to run (default 15) +-- Reports frame times to detect GC stutter. +-- ========================================================================= + +local gcstress_frame = CreateFrame("Frame") +local gcstress_running = false +local gcstress_end_time = 0 +local gcstress_frame_times = {} +local gcstress_frame_count = 0 +local gcstress_last_time = 0 +local gcstress_heap = {} -- the live set +local gcstress_heap_size = 0 +local gcstress_target_size = 0 +local gcstress_churn_count = 0 +local gcstress_churn_pct = 5 +local gcstress_duration = 15 +local gcstress_phase = "idle" -- "building" or "churning" +local gcstress_build_idx = 0 + +-- Create a varied object to fill the heap +local function make_object(i) + local mode = math.mod(i, 5) + if mode == 0 then + return { name = "obj_" .. i, value = i, flag = true } + elseif mode == 1 then + return { i, i+1, i+2, i+3, tag = "array_" .. i } + elseif mode == 2 then + return { sub = { a = i, b = i * 0.5 }, id = i } + elseif mode == 3 then + return "longstring_padding_" .. i .. "_extra_data_here" + else + return { x = i, y = i+1, z = i+2, w = i+3, label = "vec_" .. i, nested = { i } } + end +end + +gcstress_frame:SetScript("OnUpdate", function() + if not gcstress_running then return end + + local now = GetTime() + if gcstress_last_time > 0 then + local dt = (now - gcstress_last_time) * 1000 + gcstress_frame_count = gcstress_frame_count + 1 + gcstress_frame_times[gcstress_frame_count] = dt + end + gcstress_last_time = now + + if gcstress_phase == "building" then + -- Build up the live set over several frames (10K per frame) + local batch = 10000 + local target = gcstress_target_size + for i = 1, batch do + gcstress_build_idx = gcstress_build_idx + 1 + if gcstress_build_idx > target then + gcstress_phase = "churning" + gcstress_heap_size = target + gcstress_frame_count = 0 -- reset frame counter, start measuring + gcstress_last_time = 0 + gcstress_end_time = GetTime() + gcstress_duration + DEFAULT_CHAT_FRAME:AddMessage(string.format( + "|cff00ff00GCStress|r: heap built (%dK objects), churning %d%%/frame for %ds...", + target / 1000, gcstress_churn_pct, gcstress_duration)) + return + end + gcstress_heap[gcstress_build_idx] = make_object(gcstress_build_idx) + end + return + end + + -- Churning phase: replace a random fraction of the live set each frame + if now >= gcstress_end_time then + gcstress_running = false + gcstress_report() + -- Clean up heap + gcstress_heap = {} + collectgarbage() + return + end + + local churn = gcstress_churn_count + for i = 1, churn do + -- Replace a random slot with a new object (old one becomes garbage) + local idx = math.random(1, gcstress_heap_size) + gcstress_heap[idx] = make_object(idx + gcstress_frame_count * 1000) + end + + -- Also create some pure garbage (temporaries that die this frame) + for i = 1, churn do + local t = { a = i, s = "tmp_" .. i } + end +end) + +function gcstress_report() + local n = gcstress_frame_count + if n == 0 then + DEFAULT_CHAT_FRAME:AddMessage("|cffff0000GCStress|r: no frames recorded") + return + end + + -- Sort frame times to get percentiles + table.sort(gcstress_frame_times) + + local sum = 0 + local max_dt = 0 + for i = 1, n do + sum = sum + gcstress_frame_times[i] + if gcstress_frame_times[i] > max_dt then + max_dt = gcstress_frame_times[i] + end + end + local avg = sum / n + local p50 = gcstress_frame_times[math.floor(n * 0.5)] + local p95 = gcstress_frame_times[math.floor(n * 0.95)] + local p99 = gcstress_frame_times[math.floor(n * 0.99)] + + -- Count stutter frames (>2x median) + local stutter_threshold = p50 * 2 + local stutters = 0 + for i = 1, n do + if gcstress_frame_times[i] > stutter_threshold then + stutters = stutters + 1 + end + end + + DEFAULT_CHAT_FRAME:AddMessage(string.format( + "|cff00ff00GCStress|r: %d frames, %dK heap, %d churn/frame", n, gcstress_heap_size / 1000, gcstress_churn_count)) + DEFAULT_CHAT_FRAME:AddMessage(string.format( + " avg=|cffffd700%.1fms|r p50=%.1f p95=%.1f p99=%.1f max=|cffff0000%.1fms|r", + avg, p50, p95, p99, max_dt)) + DEFAULT_CHAT_FRAME:AddMessage(string.format( + " stutters (>%.1fms): |cffff0000%d|r (%.1f%%)", + stutter_threshold, stutters, stutters * 100 / n)) +end + +SLASH_GCSTRESS1 = "/gcstress" +SlashCmdList["GCSTRESS"] = function(msg) + local args = {} + for w in string.gfind(msg, "%S+") do + table.insert(args, tonumber(w)) + end + + local size_k = args[1] or 200 + gcstress_churn_pct = args[2] or 5 + gcstress_duration = args[3] or 15 + + gcstress_target_size = size_k * 1000 + gcstress_churn_count = math.floor(gcstress_target_size * gcstress_churn_pct / 100) + + gcstress_heap = {} + gcstress_frame_times = {} + gcstress_frame_count = 0 + gcstress_last_time = 0 + gcstress_build_idx = 0 + gcstress_phase = "building" + gcstress_running = true + + DEFAULT_CHAT_FRAME:AddMessage(string.format( + "|cff00ff00GCStress|r: building %dK live objects...", size_k)) +end + +-- ========================================================================= +-- Lump allocation test: /lumpaloc [count_k] [obj_size] [hold_secs] +-- Allocates a burst of objects, holds them for N seconds, then releases. +-- Measures frame times during the allocation burst to detect stutter +-- from the allocator itself (not GC). +-- count_k: thousands of objects (default 100) +-- obj_size: approximate bytes per object (default 80) +-- hold_secs: seconds to hold before release (default 10) +-- ========================================================================= + +local lump_frame = CreateFrame("Frame") +local lump_running = false +local lump_phase = "idle" +local lump_heap = {} +local lump_frame_times = {} +local lump_frame_count = 0 +local lump_last_time = 0 +local lump_target = 0 +local lump_build_idx = 0 +local lump_release_time = 0 +local lump_hold_secs = 10 +local lump_obj_size = 80 +local lump_batch = 5000 + +-- Simulate realistic addon data structures: +-- Combat log entries reference spells, units, auras in cross-linked tables. +-- Damage meters keep per-player tables with per-spell breakdowns. +-- Threat meters keep sorted lists with callbacks. + +-- Shared "database" tables that many objects reference (simulates spell/unit caches) +local shared_spells = {} +local shared_units = {} +for i = 1, 200 do + shared_spells[i] = { id = i, name = "Spell_" .. i, rank = math.mod(i, 5) + 1, school = math.mod(i, 7), icon = "Interface\\Icons\\spell_" .. i } + shared_units[i] = { guid = "0x" .. i, name = "Unit_" .. i, class = math.mod(i, 9) + 1, level = 60, buffs = {}, debuffs = {} } +end + +-- Metatables for "typed" objects (addons use these heavily) +local CombatEvent_mt = { __index = { GetSource = function(self) return self.source end, GetTarget = function(self) return self.target end, GetAmount = function(self) return self.amount end } } +local PlayerData_mt = { __index = { GetDPS = function(self) return self.total / (self.duration or 1) end, AddSpell = function(self, id, amt) self.spells[id] = (self.spells[id] or 0) + amt end } } + +local function make_combat_event(i) + local e = { + timestamp = GetTime() + i * 0.001, + event = "SPELL_DAMAGE", + source = shared_units[math.mod(i, 200) + 1], + target = shared_units[math.mod(i + 50, 200) + 1], + spell = shared_spells[math.mod(i, 200) + 1], + amount = math.random(100, 5000), + overkill = 0, + school = math.mod(i, 7), + critical = math.mod(i, 4) == 0, + absorbed = math.mod(i, 10) == 0 and math.random(50, 500) or nil, + blocked = nil, + resisted = math.mod(i, 8) == 0 and math.random(20, 200) or nil, + } + setmetatable(e, CombatEvent_mt) + return e +end + +local function make_player_data(i) + local p = { + name = "Player_" .. i, + class = math.mod(i, 9) + 1, + unit = shared_units[math.mod(i, 40) + 1], + total = 0, + duration = 0, + spells = {}, + targets = {}, + timeline = {}, + auras = {}, + } + -- Fill spell breakdown (like a damage meter accumulating data) + for j = 1, 20 do + local sp = shared_spells[math.mod(i * 7 + j, 200) + 1] + p.spells[sp.name] = { hits = math.random(10, 200), total = math.random(5000, 100000), crit = math.random(5, 50), min = math.random(100, 500), max = math.random(2000, 8000) } + p.total = p.total + p.spells[sp.name].total + end + -- Fill target breakdown + for j = 1, 8 do + local tgt = shared_units[math.mod(i * 3 + j, 200) + 1] + p.targets[tgt.name] = math.random(10000, 200000) + end + -- Timeline entries (like a graph data series) + for j = 1, 30 do + p.timeline[j] = { t = j, dps = math.random(500, 3000), hps = 0 } + end + setmetatable(p, PlayerData_mt) + return p +end + +local function make_aura_tracker(i) + local a = { + unit = shared_units[math.mod(i, 200) + 1], + buffs = {}, + debuffs = {}, + callbacks = {}, + } + for j = 1, 10 do + a.buffs[j] = { spell = shared_spells[math.mod(i + j, 200) + 1], stacks = math.mod(j, 3) + 1, expires = GetTime() + math.random(5, 30), source = shared_units[math.mod(i + j + 20, 200) + 1] } + end + for j = 1, 6 do + a.debuffs[j] = { spell = shared_spells[math.mod(i * 2 + j, 200) + 1], stacks = 1, expires = GetTime() + math.random(3, 18) } + end + -- Closures referencing upvalues (common in addon callbacks) + local unit_ref = a.unit + a.callbacks.onApply = function(spell) return unit_ref.name .. " gained " .. spell.name end + a.callbacks.onFade = function(spell) return unit_ref.name .. " lost " .. spell.name end + return a +end + +local function make_sized_object(i, size) + local mode = math.mod(i, 3) + if mode == 0 then + return make_combat_event(i) + elseif mode == 1 then + return make_player_data(i) + else + return make_aura_tracker(i) + end +end + +lump_frame:SetScript("OnUpdate", function() + if not lump_running then return end + + local now = GetTime() + if lump_last_time > 0 then + local dt = (now - lump_last_time) * 1000 + lump_frame_count = lump_frame_count + 1 + lump_frame_times[lump_frame_count] = dt + end + lump_last_time = now + + if lump_phase == "allocating" then + -- Allocate a batch per frame + for j = 1, lump_batch do + lump_build_idx = lump_build_idx + 1 + if lump_build_idx > lump_target then + lump_phase = "holding" + lump_release_time = GetTime() + lump_hold_secs + DEFAULT_CHAT_FRAME:AddMessage(string.format( + "|cff00ff00LumpAlloc|r: allocated %dK objects, holding for %ds...", + lump_target / 1000, lump_hold_secs)) + return + end + lump_heap[lump_build_idx] = make_sized_object(lump_build_idx, lump_obj_size) + end + + elseif lump_phase == "holding" then + if now >= lump_release_time then + DEFAULT_CHAT_FRAME:AddMessage("|cff00ff00LumpAlloc|r: releasing all objects...") + lump_heap = {} + lump_phase = "released" + -- Let GC deal with it, keep measuring for a few more seconds + lump_release_time = GetTime() + 5 + end + + elseif lump_phase == "released" then + if now >= lump_release_time then + lump_running = false + lump_report() + collectgarbage() + end + end +end) + +function lump_report() + local n = lump_frame_count + if n == 0 then + DEFAULT_CHAT_FRAME:AddMessage("|cffff0000LumpAlloc|r: no frames recorded") + return + end + + table.sort(lump_frame_times) + + local sum = 0 + local max_dt = 0 + for i = 1, n do + sum = sum + lump_frame_times[i] + if lump_frame_times[i] > max_dt then max_dt = lump_frame_times[i] end + end + local avg = sum / n + local p50 = lump_frame_times[math.floor(n * 0.5)] + local p95 = lump_frame_times[math.floor(n * 0.95)] + local p99 = lump_frame_times[math.floor(n * 0.99)] + + local stutter_threshold = p50 * 2 + local stutters = 0 + for i = 1, n do + if lump_frame_times[i] > stutter_threshold then stutters = stutters + 1 end + end + + DEFAULT_CHAT_FRAME:AddMessage(string.format( + "|cff00ff00LumpAlloc|r: %d frames, %dK objects at ~%dB each", + n, lump_target / 1000, lump_obj_size)) + DEFAULT_CHAT_FRAME:AddMessage(string.format( + " avg=|cffffd700%.1fms|r p50=%.1f p95=%.1f p99=%.1f max=|cffff0000%.1fms|r", + avg, p50, p95, p99, max_dt)) + DEFAULT_CHAT_FRAME:AddMessage(string.format( + " stutters (>%.1fms): %d (%.1f pct)", + stutter_threshold, stutters, stutters * 100 / n)) +end + +SLASH_LUMPALOC1 = "/lumpaloc" +SlashCmdList["LUMPALOC"] = function(msg) + local args = {} + for w in string.gfind(msg, "%S+") do + table.insert(args, tonumber(w)) + end + + local count_k = args[1] or 50 + lump_obj_size = args[2] or 80 + lump_hold_secs = args[3] or 10 + + lump_target = count_k * 1000 + lump_heap = {} + lump_frame_times = {} + lump_frame_count = 0 + lump_last_time = 0 + lump_build_idx = 0 + lump_phase = "allocating" + lump_running = true + + DEFAULT_CHAT_FRAME:AddMessage(string.format( + "|cff00ff00LumpAlloc|r: allocating %dK objects (~%dB each), hold %ds...", + count_k, lump_obj_size, lump_hold_secs)) +end diff --git a/src/weirdperformance/luaalloc.zig b/src/weirdperformance/luaalloc.zig index 14c468d..c0de8ad 100644 --- a/src/weirdperformance/luaalloc.zig +++ b/src/weirdperformance/luaalloc.zig @@ -35,7 +35,6 @@ const PROFILE = false; // set true to collect size histogram, dump with dumpStat // also use VirtualAlloc with a size header. // ============================================================================ -// VirtualAlloc for 64KB-aligned slab pages. const MEM_COMMIT = 0x1000; const MEM_RESERVE = 0x2000; const MEM_RELEASE = 0x8000; @@ -46,6 +45,11 @@ extern "kernel32" fn VirtualFree(lpAddress: *anyopaque, dwSize: u32, dwFreeType: const SEGMENT_SHIFT = 16; // 64KB pages const PAGE_SIZE = 1 << SEGMENT_SHIFT; // 65536 +// Arena: reserve 256MB of address space upfront (no physical memory), +// then commit 64KB pages from it. Commit is much cheaper than +// reserve+commit and avoids the VirtualAlloc syscall overhead that +// causes allocation stutter. + // Class index values: 1-15 = slab classes, LARGE_CLASS = large alloc, 0 = unowned const LARGE_CLASS = 0xFF; diff --git a/src/weirdperformance/luagc.zig b/src/weirdperformance/luagc.zig new file mode 100644 index 0000000..4953c05 --- /dev/null +++ b/src/weirdperformance/luagc.zig @@ -0,0 +1,302 @@ +//! Incremental garbage collector for Lua 5.0. +//! +//! WoW's Lua 5.0 uses stop-the-world mark-and-sweep GC. When it triggers, +//! the entire game freezes while every Lua object is visited. With 500K+ +//! objects from addons, this causes visible stutters. +//! +//! This module hooks luaC_collectgarbage (0x6F7340) and replaces it with +//! an incremental state machine: +//! +//! IDLE -> MARK -> SWEEP -> FINALIZE -> IDLE +//! +//! Mark phase runs atomically (it's fast -- only visits reachable objects). +//! Sweep phase runs incrementally -- each time allocation pressure triggers +//! the GC, we sweep a batch of objects and return. Lua's own allocation +//! pattern drives the sweep rate: heavy allocation = faster sweep. +//! +//! No write barriers needed because mark is atomic. The mutator doesn't +//! run between mark start and mark end, so no objects can be missed. + +const hook = @import("zhook"); + +// ============================================================================ +// Lua internals +// +// lua_State layout: +// +0x10: global_State* (l_G) +// +0x60: allowhook (checked by luaC_collectgarbage before proceeding) +// +// global_State layout (verified from disassembly): +// +0x00: strt.hash (GCObject**) +// +0x04: strt.nuse (int) +// +0x08: strt.size (int) +// +0x10: rootgc (GCObject*) -- main object list +// +0x14: rootudata (GCObject*) -- userdata list (swept first for finalizers) +// +0x18: tmudata (GCObject*) -- userdata pending __gc +// +0x24: GCthreshold (lu_mem) +// +0x28: totalbytes (lu_mem) +// +// GCObject common header: +// +0x00: next (GCObject*) -- intrusive linked list +// +0x04: tt (byte) -- type tag +// +0x05: marked (byte) -- GC mark bits +// ============================================================================ + +const GS_ROOTGC = 0x10; +const GS_ROOTUDATA = 0x14; +const GS_GCTHRESHOLD = 0x24; +const GS_TOTALBYTES = 0x28; + +const OBJ_NEXT = 0x00; +const OBJ_MARKED = 0x05; + +const MARK_BIT: u8 = 0x01; + +// Batch size: number of objects to sweep per GC invocation. +// Tuned for ~0.1ms per batch at typical object sizes. +const SWEEP_BATCH = 0xFFFFFFFF; // DEBUG: sweep everything in one batch + +// Headroom: bytes of allocation allowed between sweep batches. +// Prevents GC from being re-triggered immediately after a batch. +const BATCH_HEADROOM = 64 * 1024; // 64KB + +// ============================================================================ +// Original function pointers (called directly, not hooked) +// ============================================================================ + +// lua_gc_full_collection (0x6F73E0): __fastcall(ECX=L) -- mark phase +const MarkFn = *const fn (u32) callconv(hook.cc.fastcall) void; +const lua_gc_full_collection: MarkFn = @ptrFromInt(0x6F73E0); + +// lua_gc_free_object (0x6F7260): __fastcall(ECX=L, EDX=obj) +const FreeObjFn = *const fn (u32, u32) callconv(hook.cc.fastcall) void; +const lua_gc_free_object: FreeObjFn = @ptrFromInt(0x6F7260); + +// lua_gc_sweep_all_lists (0x6F72F0): __fastcall(ECX=L, EDX=threshold) +const SweepStringsFn = *const fn (u32, u32) callconv(hook.cc.fastcall) void; +const lua_gc_sweep_all_lists: SweepStringsFn = @ptrFromInt(0x6F72F0); + +// lua_gc_shrink_memory (0x6F7370): __fastcall(ECX=L) +const ShrinkFn = *const fn (u32) callconv(hook.cc.fastcall) void; +const lua_gc_shrink_memory: ShrinkFn = @ptrFromInt(0x6F7370); + +// luaCallUserDataGC (0x6F7080): __fastcall(ECX=L) +const FinalizeFn = *const fn (u32) callconv(hook.cc.fastcall) void; +const luaCallUserDataGC: FinalizeFn = @ptrFromInt(0x6F7080); + +// ============================================================================ +// GC state machine +// ============================================================================ + +const GcPhase = enum { idle, sweeping_udata, sweeping_strings, sweeping_rootgc, finalizing }; + +var phase: GcPhase = .idle; +var sweep_ptr: u32 = 0; // pointer TO current position in linked list (so we can unlink) +var sweep_threshold: u32 = 0; // mark threshold for current cycle +var saved_L: u32 = 0; // lua_State* for calling back into Lua + +fn getGlobalState(L: u32) u32 { + return @as(*const u32, @ptrFromInt(L + 0x10)).*; +} + +fn readU32(addr: u32) u32 { + return @as(*const u32, @ptrFromInt(addr)).*; +} + +fn writeU32(addr: u32, val: u32) void { + @as(*u32, @ptrFromInt(addr)).* = val; +} + +fn readU8(addr: u32) u8 { + return @as(*const u8, @ptrFromInt(addr)).*; +} + +fn writeU8(addr: u32, val: u8) void { + @as(*u8, @ptrFromInt(addr)).* = val; +} + +/// Sweep a batch of objects from the linked list at *sweep_ptr. +/// Returns number of objects freed. +fn sweepBatch(L: u32, count: u32) u32 { + var freed: u32 = 0; + var remaining = count; + + while (remaining > 0) { + const obj = readU32(sweep_ptr); + if (obj == 0) break; // end of list + + const marked = readU8(obj + OBJ_MARKED); + if (marked > sweep_threshold) { + // Object is marked (alive) -- clear mark bit, advance + writeU8(obj + OBJ_MARKED, marked & ~MARK_BIT); + sweep_ptr = obj + OBJ_NEXT; + } else { + // Object is unmarked (dead) -- unlink and free + writeU32(sweep_ptr, readU32(obj + OBJ_NEXT)); + lua_gc_free_object(L, obj); + freed += 1; + } + remaining -= 1; + } + + return freed; +} + +/// Main hook replacing luaC_collectgarbage (0x6F7340). +/// __fastcall(ECX=lua_State*), plain RET. +var in_gc: bool = false; + +fn collectGarbageDetour(L: u32) callconv(hook.cc.fastcall) void { + if (in_gc) return; // re-entrancy guard + // Original checks L->allowhook (offset 0x60) before proceeding + if (@as(*const u32, @ptrFromInt(L + 0x60)).* == 0) return; + in_gc = true; + defer in_gc = false; + + const g = getGlobalState(L); + saved_L = L; + + switch (phase) { + .idle => { + // Start new GC cycle: run full mark phase atomically + // lua_gc_full_collection expects state set up via prior calls. + // The original luaC_collectgarbage calls it after checking allowhook + // with ECX = L still in register. We replicate this. + lua_gc_full_collection(L); + + // Mark phase done. Start sweeping userdata first (same order as original). + phase = .sweeping_udata; + sweep_ptr = g + GS_ROOTUDATA; + sweep_threshold = 0; // first sweep pass uses threshold 0x100 + // Actually the original passes param_2=0x100 for userdata sweep. + // The threshold comparison is: if marked > threshold, keep alive. + // With threshold 0x100, only objects with marked > 256 survive, + // which means nothing survives (marked is a byte, max 255). + // Wait -- that means the first udata sweep frees EVERYTHING? + // No: the original passes EDI=0 (from XOR EDX,EDX -> param_2=0), + // then luaGarbageCollect sets EDI=0x100 if param_2!=0. + // luaC_collectgarbage calls luaGarbageCollect(L, 0), so EDI=0. + // threshold=0 means: if marked > 0, keep (marked objects survive). + sweep_threshold = 0; + + // Raise GCthreshold to prevent immediate re-trigger + const totalbytes = readU32(g + GS_TOTALBYTES); + writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM); + + // Do first batch of udata sweep + _ = sweepBatch(L, SWEEP_BATCH); + + // Check if udata sweep is done + if (readU32(sweep_ptr) == 0) { + phase = .sweeping_strings; + } + }, + + .sweeping_udata => { + _ = sweepBatch(L, SWEEP_BATCH); + + if (readU32(sweep_ptr) == 0) { + phase = .sweeping_strings; + } + + // Keep threshold ahead of allocations + const totalbytes = readU32(g + GS_TOTALBYTES); + writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM); + }, + + .sweeping_strings => { + // String table sweep is not a linked list walk -- it's a hash + // table scan. Run it atomically (it's bounded by string count, + // typically fast). + lua_gc_sweep_all_lists(L, 0); + + // Now start main rootgc sweep + phase = .sweeping_rootgc; + sweep_ptr = g + GS_ROOTGC; + + _ = sweepBatch(L, SWEEP_BATCH); + + if (readU32(sweep_ptr) == 0) { + phase = .finalizing; + } + + const totalbytes = readU32(g + GS_TOTALBYTES); + writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM); + }, + + .sweeping_rootgc => { + _ = sweepBatch(L, SWEEP_BATCH); + + if (readU32(sweep_ptr) == 0) { + phase = .finalizing; + } + + const totalbytes = readU32(g + GS_TOTALBYTES); + writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM); + }, + + .finalizing => { + // Shrink string table + buffers, set final threshold + lua_gc_shrink_memory(L); + // Run __gc finalizers + luaCallUserDataGC(L); + phase = .idle; + }, + } +} + +// ============================================================================ +// luaC_link hook (0x6F7B20) -- birth-mark barrier +// +// luaC_link adds every new GC object to rootgc and sets marked=0 (white). +// During incremental sweep, white objects get freed. New objects born during +// sweep must be born BLACK (marked=1) so the sweep skips them. +// +// Original: __fastcall(ECX=L, EDX=obj, stack: type_tag), RET 0x4 +// MOV EAX, [ECX+0x10] ; global_State +// MOV EAX, [EAX+0x10] ; old rootgc head +// MOV [EDX], EAX ; obj->next = old head +// MOV ECX, [ECX+0x10] ; global_State +// MOV [ECX+0x10], EDX ; rootgc = obj +// MOV byte [EDX+0x5], 0x0 ; obj->marked = 0 (WHITE) +// MOV byte [EDX+0x4], AL ; obj->tt = type_tag +// ============================================================================ + +const LinkFn = fn (u32, u32, u32) callconv(hook.cc.fastcall) void; +var link_hook: hook.Detour(LinkFn) = .{}; + +fn linkDetour(L: u32, obj: u32, type_tag: u32) callconv(hook.cc.fastcall) void { + // Replicate original luaC_link logic + const g = getGlobalState(L); + const old_head = readU32(g + GS_ROOTGC); + writeU32(obj + OBJ_NEXT, old_head); // obj->next = old head + writeU32(g + GS_ROOTGC, obj); // rootgc = obj + writeU8(obj + 0x04, @truncate(type_tag)); // obj->tt = type_tag + + // Birth-mark: during sweep, born BLACK so sweep skips this object + if (phase != .idle) { + writeU8(obj + OBJ_MARKED, MARK_BIT); // born marked + } else { + writeU8(obj + OBJ_MARKED, 0); // born white (normal) + } +} + +// ============================================================================ +// Hook management +// ============================================================================ + +const CollectFn = fn (u32) callconv(hook.cc.fastcall) void; +var collect_hook: hook.Detour(CollectFn) = .{}; + +pub fn install() u32 { + var installed: u32 = 0; + if (collect_hook.attach(0x6F7340, &collectGarbageDetour) == .ok) installed += 1; + // if (link_hook.attach(0x6F7B20, &linkDetour) == .ok) installed += 1; // DEBUG: disabled to isolate crash + return installed; +} + +pub fn remove() void { + collect_hook.detach(); + link_hook.detach(); + phase = .idle; +} diff --git a/src/weirdperformance/weirdperformance.zig b/src/weirdperformance/weirdperformance.zig index fae0818..511e95f 100644 --- a/src/weirdperformance/weirdperformance.zig +++ b/src/weirdperformance/weirdperformance.zig @@ -43,6 +43,7 @@ const clip_sse = @import("clip_sse.zig"); const cull_sse = @import("cull_sse.zig"); const silicon_sse = @import("silicon_sse.zig"); const luaalloc = @import("luaalloc.zig"); +const luagc = @import("luagc.zig"); const renderParticleSprites_SSE = particle_sse.renderParticleSprites_SSE; const resetParticleCache = particle_sse.resetParticleCache; @@ -326,7 +327,10 @@ pub fn installHooks() void { if (inflate_hook.install()) installed += 1; // Lua slab allocator replacement - installed += luaalloc.install(); + // installed += luaalloc.install(); // disabled for testing + + // Incremental GC -- disabled, needs more research + // installed += luagc.install(); } pub fn lateInit() void { @@ -336,6 +340,7 @@ pub fn lateInit() void { pub fn removeHooks() void { if (g_is_hook_owner) { + luaalloc.dumpStats(); // dump size histogram before unhooking filecache.remove(); inflate_hook.remove(); transform_hook.detach();