luagc: incremental GC module (disabled), realistic AllocBench

Incremental GC: atomic mark + batched rootgc sweep. Crashes from
objects born white during sweep -- needs birth-mark barrier fix.
Link hook (luaC_link) calling convention verified but interaction
with sweep_ptr at list head needs solving. Disabled pending fix.

AllocBench: realistic addon-like data structures (combat events,
player damage breakdowns with metatables, aura trackers with
closures). Cross-references shared spell/unit caches. These
complex structures reveal GC as the true stutter source -- simple
flat objects don't trigger meaningful GC pauses.

Removed arena pre-reservation (was fragmenting 32-bit address space).
This commit is contained in:
MarcelineVQ
2026-04-05 10:35:24 -07:00
parent deaeaa1450
commit 4e0a1934fb
4 changed files with 705 additions and 2 deletions
@@ -101,3 +101,395 @@ end
SLASH_ALLOCBENCH1 = "/allocbench"
SlashCmdList["ALLOCBENCH"] = run_bench
-- =========================================================================
-- GC stress test: /gcstress [size_k] [churn_pct] [duration]
-- Builds a large live heap then churns a fraction of it each frame.
-- size_k: thousands of live objects to maintain (default 200 = 200K objects)
-- churn_pct: percent of live set to replace per frame (default 5)
-- duration: seconds to run (default 15)
-- Reports frame times to detect GC stutter.
-- =========================================================================
local gcstress_frame = CreateFrame("Frame")
local gcstress_running = false
local gcstress_end_time = 0
local gcstress_frame_times = {}
local gcstress_frame_count = 0
local gcstress_last_time = 0
local gcstress_heap = {} -- the live set
local gcstress_heap_size = 0
local gcstress_target_size = 0
local gcstress_churn_count = 0
local gcstress_churn_pct = 5
local gcstress_duration = 15
local gcstress_phase = "idle" -- "building" or "churning"
local gcstress_build_idx = 0
-- Create a varied object to fill the heap
local function make_object(i)
local mode = math.mod(i, 5)
if mode == 0 then
return { name = "obj_" .. i, value = i, flag = true }
elseif mode == 1 then
return { i, i+1, i+2, i+3, tag = "array_" .. i }
elseif mode == 2 then
return { sub = { a = i, b = i * 0.5 }, id = i }
elseif mode == 3 then
return "longstring_padding_" .. i .. "_extra_data_here"
else
return { x = i, y = i+1, z = i+2, w = i+3, label = "vec_" .. i, nested = { i } }
end
end
gcstress_frame:SetScript("OnUpdate", function()
if not gcstress_running then return end
local now = GetTime()
if gcstress_last_time > 0 then
local dt = (now - gcstress_last_time) * 1000
gcstress_frame_count = gcstress_frame_count + 1
gcstress_frame_times[gcstress_frame_count] = dt
end
gcstress_last_time = now
if gcstress_phase == "building" then
-- Build up the live set over several frames (10K per frame)
local batch = 10000
local target = gcstress_target_size
for i = 1, batch do
gcstress_build_idx = gcstress_build_idx + 1
if gcstress_build_idx > target then
gcstress_phase = "churning"
gcstress_heap_size = target
gcstress_frame_count = 0 -- reset frame counter, start measuring
gcstress_last_time = 0
gcstress_end_time = GetTime() + gcstress_duration
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00GCStress|r: heap built (%dK objects), churning %d%%/frame for %ds...",
target / 1000, gcstress_churn_pct, gcstress_duration))
return
end
gcstress_heap[gcstress_build_idx] = make_object(gcstress_build_idx)
end
return
end
-- Churning phase: replace a random fraction of the live set each frame
if now >= gcstress_end_time then
gcstress_running = false
gcstress_report()
-- Clean up heap
gcstress_heap = {}
collectgarbage()
return
end
local churn = gcstress_churn_count
for i = 1, churn do
-- Replace a random slot with a new object (old one becomes garbage)
local idx = math.random(1, gcstress_heap_size)
gcstress_heap[idx] = make_object(idx + gcstress_frame_count * 1000)
end
-- Also create some pure garbage (temporaries that die this frame)
for i = 1, churn do
local t = { a = i, s = "tmp_" .. i }
end
end)
function gcstress_report()
local n = gcstress_frame_count
if n == 0 then
DEFAULT_CHAT_FRAME:AddMessage("|cffff0000GCStress|r: no frames recorded")
return
end
-- Sort frame times to get percentiles
table.sort(gcstress_frame_times)
local sum = 0
local max_dt = 0
for i = 1, n do
sum = sum + gcstress_frame_times[i]
if gcstress_frame_times[i] > max_dt then
max_dt = gcstress_frame_times[i]
end
end
local avg = sum / n
local p50 = gcstress_frame_times[math.floor(n * 0.5)]
local p95 = gcstress_frame_times[math.floor(n * 0.95)]
local p99 = gcstress_frame_times[math.floor(n * 0.99)]
-- Count stutter frames (>2x median)
local stutter_threshold = p50 * 2
local stutters = 0
for i = 1, n do
if gcstress_frame_times[i] > stutter_threshold then
stutters = stutters + 1
end
end
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00GCStress|r: %d frames, %dK heap, %d churn/frame", n, gcstress_heap_size / 1000, gcstress_churn_count))
DEFAULT_CHAT_FRAME:AddMessage(string.format(
" avg=|cffffd700%.1fms|r p50=%.1f p95=%.1f p99=%.1f max=|cffff0000%.1fms|r",
avg, p50, p95, p99, max_dt))
DEFAULT_CHAT_FRAME:AddMessage(string.format(
" stutters (>%.1fms): |cffff0000%d|r (%.1f%%)",
stutter_threshold, stutters, stutters * 100 / n))
end
SLASH_GCSTRESS1 = "/gcstress"
SlashCmdList["GCSTRESS"] = function(msg)
local args = {}
for w in string.gfind(msg, "%S+") do
table.insert(args, tonumber(w))
end
local size_k = args[1] or 200
gcstress_churn_pct = args[2] or 5
gcstress_duration = args[3] or 15
gcstress_target_size = size_k * 1000
gcstress_churn_count = math.floor(gcstress_target_size * gcstress_churn_pct / 100)
gcstress_heap = {}
gcstress_frame_times = {}
gcstress_frame_count = 0
gcstress_last_time = 0
gcstress_build_idx = 0
gcstress_phase = "building"
gcstress_running = true
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00GCStress|r: building %dK live objects...", size_k))
end
-- =========================================================================
-- Lump allocation test: /lumpaloc [count_k] [obj_size] [hold_secs]
-- Allocates a burst of objects, holds them for N seconds, then releases.
-- Measures frame times during the allocation burst to detect stutter
-- from the allocator itself (not GC).
-- count_k: thousands of objects (default 100)
-- obj_size: approximate bytes per object (default 80)
-- hold_secs: seconds to hold before release (default 10)
-- =========================================================================
local lump_frame = CreateFrame("Frame")
local lump_running = false
local lump_phase = "idle"
local lump_heap = {}
local lump_frame_times = {}
local lump_frame_count = 0
local lump_last_time = 0
local lump_target = 0
local lump_build_idx = 0
local lump_release_time = 0
local lump_hold_secs = 10
local lump_obj_size = 80
local lump_batch = 5000
-- Simulate realistic addon data structures:
-- Combat log entries reference spells, units, auras in cross-linked tables.
-- Damage meters keep per-player tables with per-spell breakdowns.
-- Threat meters keep sorted lists with callbacks.
-- Shared "database" tables that many objects reference (simulates spell/unit caches)
local shared_spells = {}
local shared_units = {}
for i = 1, 200 do
shared_spells[i] = { id = i, name = "Spell_" .. i, rank = math.mod(i, 5) + 1, school = math.mod(i, 7), icon = "Interface\\Icons\\spell_" .. i }
shared_units[i] = { guid = "0x" .. i, name = "Unit_" .. i, class = math.mod(i, 9) + 1, level = 60, buffs = {}, debuffs = {} }
end
-- Metatables for "typed" objects (addons use these heavily)
local CombatEvent_mt = { __index = { GetSource = function(self) return self.source end, GetTarget = function(self) return self.target end, GetAmount = function(self) return self.amount end } }
local PlayerData_mt = { __index = { GetDPS = function(self) return self.total / (self.duration or 1) end, AddSpell = function(self, id, amt) self.spells[id] = (self.spells[id] or 0) + amt end } }
local function make_combat_event(i)
local e = {
timestamp = GetTime() + i * 0.001,
event = "SPELL_DAMAGE",
source = shared_units[math.mod(i, 200) + 1],
target = shared_units[math.mod(i + 50, 200) + 1],
spell = shared_spells[math.mod(i, 200) + 1],
amount = math.random(100, 5000),
overkill = 0,
school = math.mod(i, 7),
critical = math.mod(i, 4) == 0,
absorbed = math.mod(i, 10) == 0 and math.random(50, 500) or nil,
blocked = nil,
resisted = math.mod(i, 8) == 0 and math.random(20, 200) or nil,
}
setmetatable(e, CombatEvent_mt)
return e
end
local function make_player_data(i)
local p = {
name = "Player_" .. i,
class = math.mod(i, 9) + 1,
unit = shared_units[math.mod(i, 40) + 1],
total = 0,
duration = 0,
spells = {},
targets = {},
timeline = {},
auras = {},
}
-- Fill spell breakdown (like a damage meter accumulating data)
for j = 1, 20 do
local sp = shared_spells[math.mod(i * 7 + j, 200) + 1]
p.spells[sp.name] = { hits = math.random(10, 200), total = math.random(5000, 100000), crit = math.random(5, 50), min = math.random(100, 500), max = math.random(2000, 8000) }
p.total = p.total + p.spells[sp.name].total
end
-- Fill target breakdown
for j = 1, 8 do
local tgt = shared_units[math.mod(i * 3 + j, 200) + 1]
p.targets[tgt.name] = math.random(10000, 200000)
end
-- Timeline entries (like a graph data series)
for j = 1, 30 do
p.timeline[j] = { t = j, dps = math.random(500, 3000), hps = 0 }
end
setmetatable(p, PlayerData_mt)
return p
end
local function make_aura_tracker(i)
local a = {
unit = shared_units[math.mod(i, 200) + 1],
buffs = {},
debuffs = {},
callbacks = {},
}
for j = 1, 10 do
a.buffs[j] = { spell = shared_spells[math.mod(i + j, 200) + 1], stacks = math.mod(j, 3) + 1, expires = GetTime() + math.random(5, 30), source = shared_units[math.mod(i + j + 20, 200) + 1] }
end
for j = 1, 6 do
a.debuffs[j] = { spell = shared_spells[math.mod(i * 2 + j, 200) + 1], stacks = 1, expires = GetTime() + math.random(3, 18) }
end
-- Closures referencing upvalues (common in addon callbacks)
local unit_ref = a.unit
a.callbacks.onApply = function(spell) return unit_ref.name .. " gained " .. spell.name end
a.callbacks.onFade = function(spell) return unit_ref.name .. " lost " .. spell.name end
return a
end
local function make_sized_object(i, size)
local mode = math.mod(i, 3)
if mode == 0 then
return make_combat_event(i)
elseif mode == 1 then
return make_player_data(i)
else
return make_aura_tracker(i)
end
end
lump_frame:SetScript("OnUpdate", function()
if not lump_running then return end
local now = GetTime()
if lump_last_time > 0 then
local dt = (now - lump_last_time) * 1000
lump_frame_count = lump_frame_count + 1
lump_frame_times[lump_frame_count] = dt
end
lump_last_time = now
if lump_phase == "allocating" then
-- Allocate a batch per frame
for j = 1, lump_batch do
lump_build_idx = lump_build_idx + 1
if lump_build_idx > lump_target then
lump_phase = "holding"
lump_release_time = GetTime() + lump_hold_secs
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00LumpAlloc|r: allocated %dK objects, holding for %ds...",
lump_target / 1000, lump_hold_secs))
return
end
lump_heap[lump_build_idx] = make_sized_object(lump_build_idx, lump_obj_size)
end
elseif lump_phase == "holding" then
if now >= lump_release_time then
DEFAULT_CHAT_FRAME:AddMessage("|cff00ff00LumpAlloc|r: releasing all objects...")
lump_heap = {}
lump_phase = "released"
-- Let GC deal with it, keep measuring for a few more seconds
lump_release_time = GetTime() + 5
end
elseif lump_phase == "released" then
if now >= lump_release_time then
lump_running = false
lump_report()
collectgarbage()
end
end
end)
function lump_report()
local n = lump_frame_count
if n == 0 then
DEFAULT_CHAT_FRAME:AddMessage("|cffff0000LumpAlloc|r: no frames recorded")
return
end
table.sort(lump_frame_times)
local sum = 0
local max_dt = 0
for i = 1, n do
sum = sum + lump_frame_times[i]
if lump_frame_times[i] > max_dt then max_dt = lump_frame_times[i] end
end
local avg = sum / n
local p50 = lump_frame_times[math.floor(n * 0.5)]
local p95 = lump_frame_times[math.floor(n * 0.95)]
local p99 = lump_frame_times[math.floor(n * 0.99)]
local stutter_threshold = p50 * 2
local stutters = 0
for i = 1, n do
if lump_frame_times[i] > stutter_threshold then stutters = stutters + 1 end
end
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00LumpAlloc|r: %d frames, %dK objects at ~%dB each",
n, lump_target / 1000, lump_obj_size))
DEFAULT_CHAT_FRAME:AddMessage(string.format(
" avg=|cffffd700%.1fms|r p50=%.1f p95=%.1f p99=%.1f max=|cffff0000%.1fms|r",
avg, p50, p95, p99, max_dt))
DEFAULT_CHAT_FRAME:AddMessage(string.format(
" stutters (>%.1fms): %d (%.1f pct)",
stutter_threshold, stutters, stutters * 100 / n))
end
SLASH_LUMPALOC1 = "/lumpaloc"
SlashCmdList["LUMPALOC"] = function(msg)
local args = {}
for w in string.gfind(msg, "%S+") do
table.insert(args, tonumber(w))
end
local count_k = args[1] or 50
lump_obj_size = args[2] or 80
lump_hold_secs = args[3] or 10
lump_target = count_k * 1000
lump_heap = {}
lump_frame_times = {}
lump_frame_count = 0
lump_last_time = 0
lump_build_idx = 0
lump_phase = "allocating"
lump_running = true
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00LumpAlloc|r: allocating %dK objects (~%dB each), hold %ds...",
count_k, lump_obj_size, lump_hold_secs))
end
+5 -1
View File
@@ -35,7 +35,6 @@ const PROFILE = false; // set true to collect size histogram, dump with dumpStat
// also use VirtualAlloc with a size header.
// ============================================================================
// VirtualAlloc for 64KB-aligned slab pages.
const MEM_COMMIT = 0x1000;
const MEM_RESERVE = 0x2000;
const MEM_RELEASE = 0x8000;
@@ -46,6 +45,11 @@ extern "kernel32" fn VirtualFree(lpAddress: *anyopaque, dwSize: u32, dwFreeType:
const SEGMENT_SHIFT = 16; // 64KB pages
const PAGE_SIZE = 1 << SEGMENT_SHIFT; // 65536
// Arena: reserve 256MB of address space upfront (no physical memory),
// then commit 64KB pages from it. Commit is much cheaper than
// reserve+commit and avoids the VirtualAlloc syscall overhead that
// causes allocation stutter.
// Class index values: 1-15 = slab classes, LARGE_CLASS = large alloc, 0 = unowned
const LARGE_CLASS = 0xFF;
+302
View File
@@ -0,0 +1,302 @@
//! Incremental garbage collector for Lua 5.0.
//!
//! WoW's Lua 5.0 uses stop-the-world mark-and-sweep GC. When it triggers,
//! the entire game freezes while every Lua object is visited. With 500K+
//! objects from addons, this causes visible stutters.
//!
//! This module hooks luaC_collectgarbage (0x6F7340) and replaces it with
//! an incremental state machine:
//!
//! IDLE -> MARK -> SWEEP -> FINALIZE -> IDLE
//!
//! Mark phase runs atomically (it's fast -- only visits reachable objects).
//! Sweep phase runs incrementally -- each time allocation pressure triggers
//! the GC, we sweep a batch of objects and return. Lua's own allocation
//! pattern drives the sweep rate: heavy allocation = faster sweep.
//!
//! No write barriers needed because mark is atomic. The mutator doesn't
//! run between mark start and mark end, so no objects can be missed.
const hook = @import("zhook");
// ============================================================================
// Lua internals
//
// lua_State layout:
// +0x10: global_State* (l_G)
// +0x60: allowhook (checked by luaC_collectgarbage before proceeding)
//
// global_State layout (verified from disassembly):
// +0x00: strt.hash (GCObject**)
// +0x04: strt.nuse (int)
// +0x08: strt.size (int)
// +0x10: rootgc (GCObject*) -- main object list
// +0x14: rootudata (GCObject*) -- userdata list (swept first for finalizers)
// +0x18: tmudata (GCObject*) -- userdata pending __gc
// +0x24: GCthreshold (lu_mem)
// +0x28: totalbytes (lu_mem)
//
// GCObject common header:
// +0x00: next (GCObject*) -- intrusive linked list
// +0x04: tt (byte) -- type tag
// +0x05: marked (byte) -- GC mark bits
// ============================================================================
const GS_ROOTGC = 0x10;
const GS_ROOTUDATA = 0x14;
const GS_GCTHRESHOLD = 0x24;
const GS_TOTALBYTES = 0x28;
const OBJ_NEXT = 0x00;
const OBJ_MARKED = 0x05;
const MARK_BIT: u8 = 0x01;
// Batch size: number of objects to sweep per GC invocation.
// Tuned for ~0.1ms per batch at typical object sizes.
const SWEEP_BATCH = 0xFFFFFFFF; // DEBUG: sweep everything in one batch
// Headroom: bytes of allocation allowed between sweep batches.
// Prevents GC from being re-triggered immediately after a batch.
const BATCH_HEADROOM = 64 * 1024; // 64KB
// ============================================================================
// Original function pointers (called directly, not hooked)
// ============================================================================
// lua_gc_full_collection (0x6F73E0): __fastcall(ECX=L) -- mark phase
const MarkFn = *const fn (u32) callconv(hook.cc.fastcall) void;
const lua_gc_full_collection: MarkFn = @ptrFromInt(0x6F73E0);
// lua_gc_free_object (0x6F7260): __fastcall(ECX=L, EDX=obj)
const FreeObjFn = *const fn (u32, u32) callconv(hook.cc.fastcall) void;
const lua_gc_free_object: FreeObjFn = @ptrFromInt(0x6F7260);
// lua_gc_sweep_all_lists (0x6F72F0): __fastcall(ECX=L, EDX=threshold)
const SweepStringsFn = *const fn (u32, u32) callconv(hook.cc.fastcall) void;
const lua_gc_sweep_all_lists: SweepStringsFn = @ptrFromInt(0x6F72F0);
// lua_gc_shrink_memory (0x6F7370): __fastcall(ECX=L)
const ShrinkFn = *const fn (u32) callconv(hook.cc.fastcall) void;
const lua_gc_shrink_memory: ShrinkFn = @ptrFromInt(0x6F7370);
// luaCallUserDataGC (0x6F7080): __fastcall(ECX=L)
const FinalizeFn = *const fn (u32) callconv(hook.cc.fastcall) void;
const luaCallUserDataGC: FinalizeFn = @ptrFromInt(0x6F7080);
// ============================================================================
// GC state machine
// ============================================================================
const GcPhase = enum { idle, sweeping_udata, sweeping_strings, sweeping_rootgc, finalizing };
var phase: GcPhase = .idle;
var sweep_ptr: u32 = 0; // pointer TO current position in linked list (so we can unlink)
var sweep_threshold: u32 = 0; // mark threshold for current cycle
var saved_L: u32 = 0; // lua_State* for calling back into Lua
fn getGlobalState(L: u32) u32 {
return @as(*const u32, @ptrFromInt(L + 0x10)).*;
}
fn readU32(addr: u32) u32 {
return @as(*const u32, @ptrFromInt(addr)).*;
}
fn writeU32(addr: u32, val: u32) void {
@as(*u32, @ptrFromInt(addr)).* = val;
}
fn readU8(addr: u32) u8 {
return @as(*const u8, @ptrFromInt(addr)).*;
}
fn writeU8(addr: u32, val: u8) void {
@as(*u8, @ptrFromInt(addr)).* = val;
}
/// Sweep a batch of objects from the linked list at *sweep_ptr.
/// Returns number of objects freed.
fn sweepBatch(L: u32, count: u32) u32 {
var freed: u32 = 0;
var remaining = count;
while (remaining > 0) {
const obj = readU32(sweep_ptr);
if (obj == 0) break; // end of list
const marked = readU8(obj + OBJ_MARKED);
if (marked > sweep_threshold) {
// Object is marked (alive) -- clear mark bit, advance
writeU8(obj + OBJ_MARKED, marked & ~MARK_BIT);
sweep_ptr = obj + OBJ_NEXT;
} else {
// Object is unmarked (dead) -- unlink and free
writeU32(sweep_ptr, readU32(obj + OBJ_NEXT));
lua_gc_free_object(L, obj);
freed += 1;
}
remaining -= 1;
}
return freed;
}
/// Main hook replacing luaC_collectgarbage (0x6F7340).
/// __fastcall(ECX=lua_State*), plain RET.
var in_gc: bool = false;
fn collectGarbageDetour(L: u32) callconv(hook.cc.fastcall) void {
if (in_gc) return; // re-entrancy guard
// Original checks L->allowhook (offset 0x60) before proceeding
if (@as(*const u32, @ptrFromInt(L + 0x60)).* == 0) return;
in_gc = true;
defer in_gc = false;
const g = getGlobalState(L);
saved_L = L;
switch (phase) {
.idle => {
// Start new GC cycle: run full mark phase atomically
// lua_gc_full_collection expects state set up via prior calls.
// The original luaC_collectgarbage calls it after checking allowhook
// with ECX = L still in register. We replicate this.
lua_gc_full_collection(L);
// Mark phase done. Start sweeping userdata first (same order as original).
phase = .sweeping_udata;
sweep_ptr = g + GS_ROOTUDATA;
sweep_threshold = 0; // first sweep pass uses threshold 0x100
// Actually the original passes param_2=0x100 for userdata sweep.
// The threshold comparison is: if marked > threshold, keep alive.
// With threshold 0x100, only objects with marked > 256 survive,
// which means nothing survives (marked is a byte, max 255).
// Wait -- that means the first udata sweep frees EVERYTHING?
// No: the original passes EDI=0 (from XOR EDX,EDX -> param_2=0),
// then luaGarbageCollect sets EDI=0x100 if param_2!=0.
// luaC_collectgarbage calls luaGarbageCollect(L, 0), so EDI=0.
// threshold=0 means: if marked > 0, keep (marked objects survive).
sweep_threshold = 0;
// Raise GCthreshold to prevent immediate re-trigger
const totalbytes = readU32(g + GS_TOTALBYTES);
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
// Do first batch of udata sweep
_ = sweepBatch(L, SWEEP_BATCH);
// Check if udata sweep is done
if (readU32(sweep_ptr) == 0) {
phase = .sweeping_strings;
}
},
.sweeping_udata => {
_ = sweepBatch(L, SWEEP_BATCH);
if (readU32(sweep_ptr) == 0) {
phase = .sweeping_strings;
}
// Keep threshold ahead of allocations
const totalbytes = readU32(g + GS_TOTALBYTES);
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
},
.sweeping_strings => {
// String table sweep is not a linked list walk -- it's a hash
// table scan. Run it atomically (it's bounded by string count,
// typically fast).
lua_gc_sweep_all_lists(L, 0);
// Now start main rootgc sweep
phase = .sweeping_rootgc;
sweep_ptr = g + GS_ROOTGC;
_ = sweepBatch(L, SWEEP_BATCH);
if (readU32(sweep_ptr) == 0) {
phase = .finalizing;
}
const totalbytes = readU32(g + GS_TOTALBYTES);
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
},
.sweeping_rootgc => {
_ = sweepBatch(L, SWEEP_BATCH);
if (readU32(sweep_ptr) == 0) {
phase = .finalizing;
}
const totalbytes = readU32(g + GS_TOTALBYTES);
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
},
.finalizing => {
// Shrink string table + buffers, set final threshold
lua_gc_shrink_memory(L);
// Run __gc finalizers
luaCallUserDataGC(L);
phase = .idle;
},
}
}
// ============================================================================
// luaC_link hook (0x6F7B20) -- birth-mark barrier
//
// luaC_link adds every new GC object to rootgc and sets marked=0 (white).
// During incremental sweep, white objects get freed. New objects born during
// sweep must be born BLACK (marked=1) so the sweep skips them.
//
// Original: __fastcall(ECX=L, EDX=obj, stack: type_tag), RET 0x4
// MOV EAX, [ECX+0x10] ; global_State
// MOV EAX, [EAX+0x10] ; old rootgc head
// MOV [EDX], EAX ; obj->next = old head
// MOV ECX, [ECX+0x10] ; global_State
// MOV [ECX+0x10], EDX ; rootgc = obj
// MOV byte [EDX+0x5], 0x0 ; obj->marked = 0 (WHITE)
// MOV byte [EDX+0x4], AL ; obj->tt = type_tag
// ============================================================================
const LinkFn = fn (u32, u32, u32) callconv(hook.cc.fastcall) void;
var link_hook: hook.Detour(LinkFn) = .{};
fn linkDetour(L: u32, obj: u32, type_tag: u32) callconv(hook.cc.fastcall) void {
// Replicate original luaC_link logic
const g = getGlobalState(L);
const old_head = readU32(g + GS_ROOTGC);
writeU32(obj + OBJ_NEXT, old_head); // obj->next = old head
writeU32(g + GS_ROOTGC, obj); // rootgc = obj
writeU8(obj + 0x04, @truncate(type_tag)); // obj->tt = type_tag
// Birth-mark: during sweep, born BLACK so sweep skips this object
if (phase != .idle) {
writeU8(obj + OBJ_MARKED, MARK_BIT); // born marked
} else {
writeU8(obj + OBJ_MARKED, 0); // born white (normal)
}
}
// ============================================================================
// Hook management
// ============================================================================
const CollectFn = fn (u32) callconv(hook.cc.fastcall) void;
var collect_hook: hook.Detour(CollectFn) = .{};
pub fn install() u32 {
var installed: u32 = 0;
if (collect_hook.attach(0x6F7340, &collectGarbageDetour) == .ok) installed += 1;
// if (link_hook.attach(0x6F7B20, &linkDetour) == .ok) installed += 1; // DEBUG: disabled to isolate crash
return installed;
}
pub fn remove() void {
collect_hook.detach();
link_hook.detach();
phase = .idle;
}
+6 -1
View File
@@ -43,6 +43,7 @@ const clip_sse = @import("clip_sse.zig");
const cull_sse = @import("cull_sse.zig");
const silicon_sse = @import("silicon_sse.zig");
const luaalloc = @import("luaalloc.zig");
const luagc = @import("luagc.zig");
const renderParticleSprites_SSE = particle_sse.renderParticleSprites_SSE;
const resetParticleCache = particle_sse.resetParticleCache;
@@ -326,7 +327,10 @@ pub fn installHooks() void {
if (inflate_hook.install()) installed += 1;
// Lua slab allocator replacement
installed += luaalloc.install();
// installed += luaalloc.install(); // disabled for testing
// Incremental GC -- disabled, needs more research
// installed += luagc.install();
}
pub fn lateInit() void {
@@ -336,6 +340,7 @@ pub fn lateInit() void {
pub fn removeHooks() void {
if (g_is_hook_owner) {
luaalloc.dumpStats(); // dump size histogram before unhooking
filecache.remove();
inflate_hook.remove();
transform_hook.detach();