luagc: incremental GC module (disabled), realistic AllocBench
Incremental GC: atomic mark + batched rootgc sweep. Crashes from objects born white during sweep -- needs birth-mark barrier fix. Link hook (luaC_link) calling convention verified but interaction with sweep_ptr at list head needs solving. Disabled pending fix. AllocBench: realistic addon-like data structures (combat events, player damage breakdowns with metatables, aura trackers with closures). Cross-references shared spell/unit caches. These complex structures reveal GC as the true stutter source -- simple flat objects don't trigger meaningful GC pauses. Removed arena pre-reservation (was fragmenting 32-bit address space).
This commit is contained in:
@@ -101,3 +101,395 @@ end
|
||||
|
||||
SLASH_ALLOCBENCH1 = "/allocbench"
|
||||
SlashCmdList["ALLOCBENCH"] = run_bench
|
||||
|
||||
-- =========================================================================
|
||||
-- GC stress test: /gcstress [size_k] [churn_pct] [duration]
|
||||
-- Builds a large live heap then churns a fraction of it each frame.
|
||||
-- size_k: thousands of live objects to maintain (default 200 = 200K objects)
|
||||
-- churn_pct: percent of live set to replace per frame (default 5)
|
||||
-- duration: seconds to run (default 15)
|
||||
-- Reports frame times to detect GC stutter.
|
||||
-- =========================================================================
|
||||
|
||||
local gcstress_frame = CreateFrame("Frame")
|
||||
local gcstress_running = false
|
||||
local gcstress_end_time = 0
|
||||
local gcstress_frame_times = {}
|
||||
local gcstress_frame_count = 0
|
||||
local gcstress_last_time = 0
|
||||
local gcstress_heap = {} -- the live set
|
||||
local gcstress_heap_size = 0
|
||||
local gcstress_target_size = 0
|
||||
local gcstress_churn_count = 0
|
||||
local gcstress_churn_pct = 5
|
||||
local gcstress_duration = 15
|
||||
local gcstress_phase = "idle" -- "building" or "churning"
|
||||
local gcstress_build_idx = 0
|
||||
|
||||
-- Create a varied object to fill the heap
|
||||
local function make_object(i)
|
||||
local mode = math.mod(i, 5)
|
||||
if mode == 0 then
|
||||
return { name = "obj_" .. i, value = i, flag = true }
|
||||
elseif mode == 1 then
|
||||
return { i, i+1, i+2, i+3, tag = "array_" .. i }
|
||||
elseif mode == 2 then
|
||||
return { sub = { a = i, b = i * 0.5 }, id = i }
|
||||
elseif mode == 3 then
|
||||
return "longstring_padding_" .. i .. "_extra_data_here"
|
||||
else
|
||||
return { x = i, y = i+1, z = i+2, w = i+3, label = "vec_" .. i, nested = { i } }
|
||||
end
|
||||
end
|
||||
|
||||
gcstress_frame:SetScript("OnUpdate", function()
|
||||
if not gcstress_running then return end
|
||||
|
||||
local now = GetTime()
|
||||
if gcstress_last_time > 0 then
|
||||
local dt = (now - gcstress_last_time) * 1000
|
||||
gcstress_frame_count = gcstress_frame_count + 1
|
||||
gcstress_frame_times[gcstress_frame_count] = dt
|
||||
end
|
||||
gcstress_last_time = now
|
||||
|
||||
if gcstress_phase == "building" then
|
||||
-- Build up the live set over several frames (10K per frame)
|
||||
local batch = 10000
|
||||
local target = gcstress_target_size
|
||||
for i = 1, batch do
|
||||
gcstress_build_idx = gcstress_build_idx + 1
|
||||
if gcstress_build_idx > target then
|
||||
gcstress_phase = "churning"
|
||||
gcstress_heap_size = target
|
||||
gcstress_frame_count = 0 -- reset frame counter, start measuring
|
||||
gcstress_last_time = 0
|
||||
gcstress_end_time = GetTime() + gcstress_duration
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
"|cff00ff00GCStress|r: heap built (%dK objects), churning %d%%/frame for %ds...",
|
||||
target / 1000, gcstress_churn_pct, gcstress_duration))
|
||||
return
|
||||
end
|
||||
gcstress_heap[gcstress_build_idx] = make_object(gcstress_build_idx)
|
||||
end
|
||||
return
|
||||
end
|
||||
|
||||
-- Churning phase: replace a random fraction of the live set each frame
|
||||
if now >= gcstress_end_time then
|
||||
gcstress_running = false
|
||||
gcstress_report()
|
||||
-- Clean up heap
|
||||
gcstress_heap = {}
|
||||
collectgarbage()
|
||||
return
|
||||
end
|
||||
|
||||
local churn = gcstress_churn_count
|
||||
for i = 1, churn do
|
||||
-- Replace a random slot with a new object (old one becomes garbage)
|
||||
local idx = math.random(1, gcstress_heap_size)
|
||||
gcstress_heap[idx] = make_object(idx + gcstress_frame_count * 1000)
|
||||
end
|
||||
|
||||
-- Also create some pure garbage (temporaries that die this frame)
|
||||
for i = 1, churn do
|
||||
local t = { a = i, s = "tmp_" .. i }
|
||||
end
|
||||
end)
|
||||
|
||||
function gcstress_report()
|
||||
local n = gcstress_frame_count
|
||||
if n == 0 then
|
||||
DEFAULT_CHAT_FRAME:AddMessage("|cffff0000GCStress|r: no frames recorded")
|
||||
return
|
||||
end
|
||||
|
||||
-- Sort frame times to get percentiles
|
||||
table.sort(gcstress_frame_times)
|
||||
|
||||
local sum = 0
|
||||
local max_dt = 0
|
||||
for i = 1, n do
|
||||
sum = sum + gcstress_frame_times[i]
|
||||
if gcstress_frame_times[i] > max_dt then
|
||||
max_dt = gcstress_frame_times[i]
|
||||
end
|
||||
end
|
||||
local avg = sum / n
|
||||
local p50 = gcstress_frame_times[math.floor(n * 0.5)]
|
||||
local p95 = gcstress_frame_times[math.floor(n * 0.95)]
|
||||
local p99 = gcstress_frame_times[math.floor(n * 0.99)]
|
||||
|
||||
-- Count stutter frames (>2x median)
|
||||
local stutter_threshold = p50 * 2
|
||||
local stutters = 0
|
||||
for i = 1, n do
|
||||
if gcstress_frame_times[i] > stutter_threshold then
|
||||
stutters = stutters + 1
|
||||
end
|
||||
end
|
||||
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
"|cff00ff00GCStress|r: %d frames, %dK heap, %d churn/frame", n, gcstress_heap_size / 1000, gcstress_churn_count))
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
" avg=|cffffd700%.1fms|r p50=%.1f p95=%.1f p99=%.1f max=|cffff0000%.1fms|r",
|
||||
avg, p50, p95, p99, max_dt))
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
" stutters (>%.1fms): |cffff0000%d|r (%.1f%%)",
|
||||
stutter_threshold, stutters, stutters * 100 / n))
|
||||
end
|
||||
|
||||
SLASH_GCSTRESS1 = "/gcstress"
|
||||
SlashCmdList["GCSTRESS"] = function(msg)
|
||||
local args = {}
|
||||
for w in string.gfind(msg, "%S+") do
|
||||
table.insert(args, tonumber(w))
|
||||
end
|
||||
|
||||
local size_k = args[1] or 200
|
||||
gcstress_churn_pct = args[2] or 5
|
||||
gcstress_duration = args[3] or 15
|
||||
|
||||
gcstress_target_size = size_k * 1000
|
||||
gcstress_churn_count = math.floor(gcstress_target_size * gcstress_churn_pct / 100)
|
||||
|
||||
gcstress_heap = {}
|
||||
gcstress_frame_times = {}
|
||||
gcstress_frame_count = 0
|
||||
gcstress_last_time = 0
|
||||
gcstress_build_idx = 0
|
||||
gcstress_phase = "building"
|
||||
gcstress_running = true
|
||||
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
"|cff00ff00GCStress|r: building %dK live objects...", size_k))
|
||||
end
|
||||
|
||||
-- =========================================================================
|
||||
-- Lump allocation test: /lumpaloc [count_k] [obj_size] [hold_secs]
|
||||
-- Allocates a burst of objects, holds them for N seconds, then releases.
|
||||
-- Measures frame times during the allocation burst to detect stutter
|
||||
-- from the allocator itself (not GC).
|
||||
-- count_k: thousands of objects (default 100)
|
||||
-- obj_size: approximate bytes per object (default 80)
|
||||
-- hold_secs: seconds to hold before release (default 10)
|
||||
-- =========================================================================
|
||||
|
||||
local lump_frame = CreateFrame("Frame")
|
||||
local lump_running = false
|
||||
local lump_phase = "idle"
|
||||
local lump_heap = {}
|
||||
local lump_frame_times = {}
|
||||
local lump_frame_count = 0
|
||||
local lump_last_time = 0
|
||||
local lump_target = 0
|
||||
local lump_build_idx = 0
|
||||
local lump_release_time = 0
|
||||
local lump_hold_secs = 10
|
||||
local lump_obj_size = 80
|
||||
local lump_batch = 5000
|
||||
|
||||
-- Simulate realistic addon data structures:
|
||||
-- Combat log entries reference spells, units, auras in cross-linked tables.
|
||||
-- Damage meters keep per-player tables with per-spell breakdowns.
|
||||
-- Threat meters keep sorted lists with callbacks.
|
||||
|
||||
-- Shared "database" tables that many objects reference (simulates spell/unit caches)
|
||||
local shared_spells = {}
|
||||
local shared_units = {}
|
||||
for i = 1, 200 do
|
||||
shared_spells[i] = { id = i, name = "Spell_" .. i, rank = math.mod(i, 5) + 1, school = math.mod(i, 7), icon = "Interface\\Icons\\spell_" .. i }
|
||||
shared_units[i] = { guid = "0x" .. i, name = "Unit_" .. i, class = math.mod(i, 9) + 1, level = 60, buffs = {}, debuffs = {} }
|
||||
end
|
||||
|
||||
-- Metatables for "typed" objects (addons use these heavily)
|
||||
local CombatEvent_mt = { __index = { GetSource = function(self) return self.source end, GetTarget = function(self) return self.target end, GetAmount = function(self) return self.amount end } }
|
||||
local PlayerData_mt = { __index = { GetDPS = function(self) return self.total / (self.duration or 1) end, AddSpell = function(self, id, amt) self.spells[id] = (self.spells[id] or 0) + amt end } }
|
||||
|
||||
local function make_combat_event(i)
|
||||
local e = {
|
||||
timestamp = GetTime() + i * 0.001,
|
||||
event = "SPELL_DAMAGE",
|
||||
source = shared_units[math.mod(i, 200) + 1],
|
||||
target = shared_units[math.mod(i + 50, 200) + 1],
|
||||
spell = shared_spells[math.mod(i, 200) + 1],
|
||||
amount = math.random(100, 5000),
|
||||
overkill = 0,
|
||||
school = math.mod(i, 7),
|
||||
critical = math.mod(i, 4) == 0,
|
||||
absorbed = math.mod(i, 10) == 0 and math.random(50, 500) or nil,
|
||||
blocked = nil,
|
||||
resisted = math.mod(i, 8) == 0 and math.random(20, 200) or nil,
|
||||
}
|
||||
setmetatable(e, CombatEvent_mt)
|
||||
return e
|
||||
end
|
||||
|
||||
local function make_player_data(i)
|
||||
local p = {
|
||||
name = "Player_" .. i,
|
||||
class = math.mod(i, 9) + 1,
|
||||
unit = shared_units[math.mod(i, 40) + 1],
|
||||
total = 0,
|
||||
duration = 0,
|
||||
spells = {},
|
||||
targets = {},
|
||||
timeline = {},
|
||||
auras = {},
|
||||
}
|
||||
-- Fill spell breakdown (like a damage meter accumulating data)
|
||||
for j = 1, 20 do
|
||||
local sp = shared_spells[math.mod(i * 7 + j, 200) + 1]
|
||||
p.spells[sp.name] = { hits = math.random(10, 200), total = math.random(5000, 100000), crit = math.random(5, 50), min = math.random(100, 500), max = math.random(2000, 8000) }
|
||||
p.total = p.total + p.spells[sp.name].total
|
||||
end
|
||||
-- Fill target breakdown
|
||||
for j = 1, 8 do
|
||||
local tgt = shared_units[math.mod(i * 3 + j, 200) + 1]
|
||||
p.targets[tgt.name] = math.random(10000, 200000)
|
||||
end
|
||||
-- Timeline entries (like a graph data series)
|
||||
for j = 1, 30 do
|
||||
p.timeline[j] = { t = j, dps = math.random(500, 3000), hps = 0 }
|
||||
end
|
||||
setmetatable(p, PlayerData_mt)
|
||||
return p
|
||||
end
|
||||
|
||||
local function make_aura_tracker(i)
|
||||
local a = {
|
||||
unit = shared_units[math.mod(i, 200) + 1],
|
||||
buffs = {},
|
||||
debuffs = {},
|
||||
callbacks = {},
|
||||
}
|
||||
for j = 1, 10 do
|
||||
a.buffs[j] = { spell = shared_spells[math.mod(i + j, 200) + 1], stacks = math.mod(j, 3) + 1, expires = GetTime() + math.random(5, 30), source = shared_units[math.mod(i + j + 20, 200) + 1] }
|
||||
end
|
||||
for j = 1, 6 do
|
||||
a.debuffs[j] = { spell = shared_spells[math.mod(i * 2 + j, 200) + 1], stacks = 1, expires = GetTime() + math.random(3, 18) }
|
||||
end
|
||||
-- Closures referencing upvalues (common in addon callbacks)
|
||||
local unit_ref = a.unit
|
||||
a.callbacks.onApply = function(spell) return unit_ref.name .. " gained " .. spell.name end
|
||||
a.callbacks.onFade = function(spell) return unit_ref.name .. " lost " .. spell.name end
|
||||
return a
|
||||
end
|
||||
|
||||
local function make_sized_object(i, size)
|
||||
local mode = math.mod(i, 3)
|
||||
if mode == 0 then
|
||||
return make_combat_event(i)
|
||||
elseif mode == 1 then
|
||||
return make_player_data(i)
|
||||
else
|
||||
return make_aura_tracker(i)
|
||||
end
|
||||
end
|
||||
|
||||
lump_frame:SetScript("OnUpdate", function()
|
||||
if not lump_running then return end
|
||||
|
||||
local now = GetTime()
|
||||
if lump_last_time > 0 then
|
||||
local dt = (now - lump_last_time) * 1000
|
||||
lump_frame_count = lump_frame_count + 1
|
||||
lump_frame_times[lump_frame_count] = dt
|
||||
end
|
||||
lump_last_time = now
|
||||
|
||||
if lump_phase == "allocating" then
|
||||
-- Allocate a batch per frame
|
||||
for j = 1, lump_batch do
|
||||
lump_build_idx = lump_build_idx + 1
|
||||
if lump_build_idx > lump_target then
|
||||
lump_phase = "holding"
|
||||
lump_release_time = GetTime() + lump_hold_secs
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
"|cff00ff00LumpAlloc|r: allocated %dK objects, holding for %ds...",
|
||||
lump_target / 1000, lump_hold_secs))
|
||||
return
|
||||
end
|
||||
lump_heap[lump_build_idx] = make_sized_object(lump_build_idx, lump_obj_size)
|
||||
end
|
||||
|
||||
elseif lump_phase == "holding" then
|
||||
if now >= lump_release_time then
|
||||
DEFAULT_CHAT_FRAME:AddMessage("|cff00ff00LumpAlloc|r: releasing all objects...")
|
||||
lump_heap = {}
|
||||
lump_phase = "released"
|
||||
-- Let GC deal with it, keep measuring for a few more seconds
|
||||
lump_release_time = GetTime() + 5
|
||||
end
|
||||
|
||||
elseif lump_phase == "released" then
|
||||
if now >= lump_release_time then
|
||||
lump_running = false
|
||||
lump_report()
|
||||
collectgarbage()
|
||||
end
|
||||
end
|
||||
end)
|
||||
|
||||
function lump_report()
|
||||
local n = lump_frame_count
|
||||
if n == 0 then
|
||||
DEFAULT_CHAT_FRAME:AddMessage("|cffff0000LumpAlloc|r: no frames recorded")
|
||||
return
|
||||
end
|
||||
|
||||
table.sort(lump_frame_times)
|
||||
|
||||
local sum = 0
|
||||
local max_dt = 0
|
||||
for i = 1, n do
|
||||
sum = sum + lump_frame_times[i]
|
||||
if lump_frame_times[i] > max_dt then max_dt = lump_frame_times[i] end
|
||||
end
|
||||
local avg = sum / n
|
||||
local p50 = lump_frame_times[math.floor(n * 0.5)]
|
||||
local p95 = lump_frame_times[math.floor(n * 0.95)]
|
||||
local p99 = lump_frame_times[math.floor(n * 0.99)]
|
||||
|
||||
local stutter_threshold = p50 * 2
|
||||
local stutters = 0
|
||||
for i = 1, n do
|
||||
if lump_frame_times[i] > stutter_threshold then stutters = stutters + 1 end
|
||||
end
|
||||
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
"|cff00ff00LumpAlloc|r: %d frames, %dK objects at ~%dB each",
|
||||
n, lump_target / 1000, lump_obj_size))
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
" avg=|cffffd700%.1fms|r p50=%.1f p95=%.1f p99=%.1f max=|cffff0000%.1fms|r",
|
||||
avg, p50, p95, p99, max_dt))
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
" stutters (>%.1fms): %d (%.1f pct)",
|
||||
stutter_threshold, stutters, stutters * 100 / n))
|
||||
end
|
||||
|
||||
SLASH_LUMPALOC1 = "/lumpaloc"
|
||||
SlashCmdList["LUMPALOC"] = function(msg)
|
||||
local args = {}
|
||||
for w in string.gfind(msg, "%S+") do
|
||||
table.insert(args, tonumber(w))
|
||||
end
|
||||
|
||||
local count_k = args[1] or 50
|
||||
lump_obj_size = args[2] or 80
|
||||
lump_hold_secs = args[3] or 10
|
||||
|
||||
lump_target = count_k * 1000
|
||||
lump_heap = {}
|
||||
lump_frame_times = {}
|
||||
lump_frame_count = 0
|
||||
lump_last_time = 0
|
||||
lump_build_idx = 0
|
||||
lump_phase = "allocating"
|
||||
lump_running = true
|
||||
|
||||
DEFAULT_CHAT_FRAME:AddMessage(string.format(
|
||||
"|cff00ff00LumpAlloc|r: allocating %dK objects (~%dB each), hold %ds...",
|
||||
count_k, lump_obj_size, lump_hold_secs))
|
||||
end
|
||||
|
||||
@@ -35,7 +35,6 @@ const PROFILE = false; // set true to collect size histogram, dump with dumpStat
|
||||
// also use VirtualAlloc with a size header.
|
||||
// ============================================================================
|
||||
|
||||
// VirtualAlloc for 64KB-aligned slab pages.
|
||||
const MEM_COMMIT = 0x1000;
|
||||
const MEM_RESERVE = 0x2000;
|
||||
const MEM_RELEASE = 0x8000;
|
||||
@@ -46,6 +45,11 @@ extern "kernel32" fn VirtualFree(lpAddress: *anyopaque, dwSize: u32, dwFreeType:
|
||||
const SEGMENT_SHIFT = 16; // 64KB pages
|
||||
const PAGE_SIZE = 1 << SEGMENT_SHIFT; // 65536
|
||||
|
||||
// Arena: reserve 256MB of address space upfront (no physical memory),
|
||||
// then commit 64KB pages from it. Commit is much cheaper than
|
||||
// reserve+commit and avoids the VirtualAlloc syscall overhead that
|
||||
// causes allocation stutter.
|
||||
|
||||
// Class index values: 1-15 = slab classes, LARGE_CLASS = large alloc, 0 = unowned
|
||||
const LARGE_CLASS = 0xFF;
|
||||
|
||||
|
||||
@@ -0,0 +1,302 @@
|
||||
//! Incremental garbage collector for Lua 5.0.
|
||||
//!
|
||||
//! WoW's Lua 5.0 uses stop-the-world mark-and-sweep GC. When it triggers,
|
||||
//! the entire game freezes while every Lua object is visited. With 500K+
|
||||
//! objects from addons, this causes visible stutters.
|
||||
//!
|
||||
//! This module hooks luaC_collectgarbage (0x6F7340) and replaces it with
|
||||
//! an incremental state machine:
|
||||
//!
|
||||
//! IDLE -> MARK -> SWEEP -> FINALIZE -> IDLE
|
||||
//!
|
||||
//! Mark phase runs atomically (it's fast -- only visits reachable objects).
|
||||
//! Sweep phase runs incrementally -- each time allocation pressure triggers
|
||||
//! the GC, we sweep a batch of objects and return. Lua's own allocation
|
||||
//! pattern drives the sweep rate: heavy allocation = faster sweep.
|
||||
//!
|
||||
//! No write barriers needed because mark is atomic. The mutator doesn't
|
||||
//! run between mark start and mark end, so no objects can be missed.
|
||||
|
||||
const hook = @import("zhook");
|
||||
|
||||
// ============================================================================
|
||||
// Lua internals
|
||||
//
|
||||
// lua_State layout:
|
||||
// +0x10: global_State* (l_G)
|
||||
// +0x60: allowhook (checked by luaC_collectgarbage before proceeding)
|
||||
//
|
||||
// global_State layout (verified from disassembly):
|
||||
// +0x00: strt.hash (GCObject**)
|
||||
// +0x04: strt.nuse (int)
|
||||
// +0x08: strt.size (int)
|
||||
// +0x10: rootgc (GCObject*) -- main object list
|
||||
// +0x14: rootudata (GCObject*) -- userdata list (swept first for finalizers)
|
||||
// +0x18: tmudata (GCObject*) -- userdata pending __gc
|
||||
// +0x24: GCthreshold (lu_mem)
|
||||
// +0x28: totalbytes (lu_mem)
|
||||
//
|
||||
// GCObject common header:
|
||||
// +0x00: next (GCObject*) -- intrusive linked list
|
||||
// +0x04: tt (byte) -- type tag
|
||||
// +0x05: marked (byte) -- GC mark bits
|
||||
// ============================================================================
|
||||
|
||||
const GS_ROOTGC = 0x10;
|
||||
const GS_ROOTUDATA = 0x14;
|
||||
const GS_GCTHRESHOLD = 0x24;
|
||||
const GS_TOTALBYTES = 0x28;
|
||||
|
||||
const OBJ_NEXT = 0x00;
|
||||
const OBJ_MARKED = 0x05;
|
||||
|
||||
const MARK_BIT: u8 = 0x01;
|
||||
|
||||
// Batch size: number of objects to sweep per GC invocation.
|
||||
// Tuned for ~0.1ms per batch at typical object sizes.
|
||||
const SWEEP_BATCH = 0xFFFFFFFF; // DEBUG: sweep everything in one batch
|
||||
|
||||
// Headroom: bytes of allocation allowed between sweep batches.
|
||||
// Prevents GC from being re-triggered immediately after a batch.
|
||||
const BATCH_HEADROOM = 64 * 1024; // 64KB
|
||||
|
||||
// ============================================================================
|
||||
// Original function pointers (called directly, not hooked)
|
||||
// ============================================================================
|
||||
|
||||
// lua_gc_full_collection (0x6F73E0): __fastcall(ECX=L) -- mark phase
|
||||
const MarkFn = *const fn (u32) callconv(hook.cc.fastcall) void;
|
||||
const lua_gc_full_collection: MarkFn = @ptrFromInt(0x6F73E0);
|
||||
|
||||
// lua_gc_free_object (0x6F7260): __fastcall(ECX=L, EDX=obj)
|
||||
const FreeObjFn = *const fn (u32, u32) callconv(hook.cc.fastcall) void;
|
||||
const lua_gc_free_object: FreeObjFn = @ptrFromInt(0x6F7260);
|
||||
|
||||
// lua_gc_sweep_all_lists (0x6F72F0): __fastcall(ECX=L, EDX=threshold)
|
||||
const SweepStringsFn = *const fn (u32, u32) callconv(hook.cc.fastcall) void;
|
||||
const lua_gc_sweep_all_lists: SweepStringsFn = @ptrFromInt(0x6F72F0);
|
||||
|
||||
// lua_gc_shrink_memory (0x6F7370): __fastcall(ECX=L)
|
||||
const ShrinkFn = *const fn (u32) callconv(hook.cc.fastcall) void;
|
||||
const lua_gc_shrink_memory: ShrinkFn = @ptrFromInt(0x6F7370);
|
||||
|
||||
// luaCallUserDataGC (0x6F7080): __fastcall(ECX=L)
|
||||
const FinalizeFn = *const fn (u32) callconv(hook.cc.fastcall) void;
|
||||
const luaCallUserDataGC: FinalizeFn = @ptrFromInt(0x6F7080);
|
||||
|
||||
// ============================================================================
|
||||
// GC state machine
|
||||
// ============================================================================
|
||||
|
||||
const GcPhase = enum { idle, sweeping_udata, sweeping_strings, sweeping_rootgc, finalizing };
|
||||
|
||||
var phase: GcPhase = .idle;
|
||||
var sweep_ptr: u32 = 0; // pointer TO current position in linked list (so we can unlink)
|
||||
var sweep_threshold: u32 = 0; // mark threshold for current cycle
|
||||
var saved_L: u32 = 0; // lua_State* for calling back into Lua
|
||||
|
||||
fn getGlobalState(L: u32) u32 {
|
||||
return @as(*const u32, @ptrFromInt(L + 0x10)).*;
|
||||
}
|
||||
|
||||
fn readU32(addr: u32) u32 {
|
||||
return @as(*const u32, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
fn writeU32(addr: u32, val: u32) void {
|
||||
@as(*u32, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
|
||||
fn readU8(addr: u32) u8 {
|
||||
return @as(*const u8, @ptrFromInt(addr)).*;
|
||||
}
|
||||
|
||||
fn writeU8(addr: u32, val: u8) void {
|
||||
@as(*u8, @ptrFromInt(addr)).* = val;
|
||||
}
|
||||
|
||||
/// Sweep a batch of objects from the linked list at *sweep_ptr.
|
||||
/// Returns number of objects freed.
|
||||
fn sweepBatch(L: u32, count: u32) u32 {
|
||||
var freed: u32 = 0;
|
||||
var remaining = count;
|
||||
|
||||
while (remaining > 0) {
|
||||
const obj = readU32(sweep_ptr);
|
||||
if (obj == 0) break; // end of list
|
||||
|
||||
const marked = readU8(obj + OBJ_MARKED);
|
||||
if (marked > sweep_threshold) {
|
||||
// Object is marked (alive) -- clear mark bit, advance
|
||||
writeU8(obj + OBJ_MARKED, marked & ~MARK_BIT);
|
||||
sweep_ptr = obj + OBJ_NEXT;
|
||||
} else {
|
||||
// Object is unmarked (dead) -- unlink and free
|
||||
writeU32(sweep_ptr, readU32(obj + OBJ_NEXT));
|
||||
lua_gc_free_object(L, obj);
|
||||
freed += 1;
|
||||
}
|
||||
remaining -= 1;
|
||||
}
|
||||
|
||||
return freed;
|
||||
}
|
||||
|
||||
/// Main hook replacing luaC_collectgarbage (0x6F7340).
|
||||
/// __fastcall(ECX=lua_State*), plain RET.
|
||||
var in_gc: bool = false;
|
||||
|
||||
fn collectGarbageDetour(L: u32) callconv(hook.cc.fastcall) void {
|
||||
if (in_gc) return; // re-entrancy guard
|
||||
// Original checks L->allowhook (offset 0x60) before proceeding
|
||||
if (@as(*const u32, @ptrFromInt(L + 0x60)).* == 0) return;
|
||||
in_gc = true;
|
||||
defer in_gc = false;
|
||||
|
||||
const g = getGlobalState(L);
|
||||
saved_L = L;
|
||||
|
||||
switch (phase) {
|
||||
.idle => {
|
||||
// Start new GC cycle: run full mark phase atomically
|
||||
// lua_gc_full_collection expects state set up via prior calls.
|
||||
// The original luaC_collectgarbage calls it after checking allowhook
|
||||
// with ECX = L still in register. We replicate this.
|
||||
lua_gc_full_collection(L);
|
||||
|
||||
// Mark phase done. Start sweeping userdata first (same order as original).
|
||||
phase = .sweeping_udata;
|
||||
sweep_ptr = g + GS_ROOTUDATA;
|
||||
sweep_threshold = 0; // first sweep pass uses threshold 0x100
|
||||
// Actually the original passes param_2=0x100 for userdata sweep.
|
||||
// The threshold comparison is: if marked > threshold, keep alive.
|
||||
// With threshold 0x100, only objects with marked > 256 survive,
|
||||
// which means nothing survives (marked is a byte, max 255).
|
||||
// Wait -- that means the first udata sweep frees EVERYTHING?
|
||||
// No: the original passes EDI=0 (from XOR EDX,EDX -> param_2=0),
|
||||
// then luaGarbageCollect sets EDI=0x100 if param_2!=0.
|
||||
// luaC_collectgarbage calls luaGarbageCollect(L, 0), so EDI=0.
|
||||
// threshold=0 means: if marked > 0, keep (marked objects survive).
|
||||
sweep_threshold = 0;
|
||||
|
||||
// Raise GCthreshold to prevent immediate re-trigger
|
||||
const totalbytes = readU32(g + GS_TOTALBYTES);
|
||||
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
|
||||
|
||||
// Do first batch of udata sweep
|
||||
_ = sweepBatch(L, SWEEP_BATCH);
|
||||
|
||||
// Check if udata sweep is done
|
||||
if (readU32(sweep_ptr) == 0) {
|
||||
phase = .sweeping_strings;
|
||||
}
|
||||
},
|
||||
|
||||
.sweeping_udata => {
|
||||
_ = sweepBatch(L, SWEEP_BATCH);
|
||||
|
||||
if (readU32(sweep_ptr) == 0) {
|
||||
phase = .sweeping_strings;
|
||||
}
|
||||
|
||||
// Keep threshold ahead of allocations
|
||||
const totalbytes = readU32(g + GS_TOTALBYTES);
|
||||
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
|
||||
},
|
||||
|
||||
.sweeping_strings => {
|
||||
// String table sweep is not a linked list walk -- it's a hash
|
||||
// table scan. Run it atomically (it's bounded by string count,
|
||||
// typically fast).
|
||||
lua_gc_sweep_all_lists(L, 0);
|
||||
|
||||
// Now start main rootgc sweep
|
||||
phase = .sweeping_rootgc;
|
||||
sweep_ptr = g + GS_ROOTGC;
|
||||
|
||||
_ = sweepBatch(L, SWEEP_BATCH);
|
||||
|
||||
if (readU32(sweep_ptr) == 0) {
|
||||
phase = .finalizing;
|
||||
}
|
||||
|
||||
const totalbytes = readU32(g + GS_TOTALBYTES);
|
||||
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
|
||||
},
|
||||
|
||||
.sweeping_rootgc => {
|
||||
_ = sweepBatch(L, SWEEP_BATCH);
|
||||
|
||||
if (readU32(sweep_ptr) == 0) {
|
||||
phase = .finalizing;
|
||||
}
|
||||
|
||||
const totalbytes = readU32(g + GS_TOTALBYTES);
|
||||
writeU32(g + GS_GCTHRESHOLD, totalbytes + BATCH_HEADROOM);
|
||||
},
|
||||
|
||||
.finalizing => {
|
||||
// Shrink string table + buffers, set final threshold
|
||||
lua_gc_shrink_memory(L);
|
||||
// Run __gc finalizers
|
||||
luaCallUserDataGC(L);
|
||||
phase = .idle;
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// luaC_link hook (0x6F7B20) -- birth-mark barrier
|
||||
//
|
||||
// luaC_link adds every new GC object to rootgc and sets marked=0 (white).
|
||||
// During incremental sweep, white objects get freed. New objects born during
|
||||
// sweep must be born BLACK (marked=1) so the sweep skips them.
|
||||
//
|
||||
// Original: __fastcall(ECX=L, EDX=obj, stack: type_tag), RET 0x4
|
||||
// MOV EAX, [ECX+0x10] ; global_State
|
||||
// MOV EAX, [EAX+0x10] ; old rootgc head
|
||||
// MOV [EDX], EAX ; obj->next = old head
|
||||
// MOV ECX, [ECX+0x10] ; global_State
|
||||
// MOV [ECX+0x10], EDX ; rootgc = obj
|
||||
// MOV byte [EDX+0x5], 0x0 ; obj->marked = 0 (WHITE)
|
||||
// MOV byte [EDX+0x4], AL ; obj->tt = type_tag
|
||||
// ============================================================================
|
||||
|
||||
const LinkFn = fn (u32, u32, u32) callconv(hook.cc.fastcall) void;
|
||||
var link_hook: hook.Detour(LinkFn) = .{};
|
||||
|
||||
fn linkDetour(L: u32, obj: u32, type_tag: u32) callconv(hook.cc.fastcall) void {
|
||||
// Replicate original luaC_link logic
|
||||
const g = getGlobalState(L);
|
||||
const old_head = readU32(g + GS_ROOTGC);
|
||||
writeU32(obj + OBJ_NEXT, old_head); // obj->next = old head
|
||||
writeU32(g + GS_ROOTGC, obj); // rootgc = obj
|
||||
writeU8(obj + 0x04, @truncate(type_tag)); // obj->tt = type_tag
|
||||
|
||||
// Birth-mark: during sweep, born BLACK so sweep skips this object
|
||||
if (phase != .idle) {
|
||||
writeU8(obj + OBJ_MARKED, MARK_BIT); // born marked
|
||||
} else {
|
||||
writeU8(obj + OBJ_MARKED, 0); // born white (normal)
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Hook management
|
||||
// ============================================================================
|
||||
|
||||
const CollectFn = fn (u32) callconv(hook.cc.fastcall) void;
|
||||
var collect_hook: hook.Detour(CollectFn) = .{};
|
||||
|
||||
pub fn install() u32 {
|
||||
var installed: u32 = 0;
|
||||
if (collect_hook.attach(0x6F7340, &collectGarbageDetour) == .ok) installed += 1;
|
||||
// if (link_hook.attach(0x6F7B20, &linkDetour) == .ok) installed += 1; // DEBUG: disabled to isolate crash
|
||||
return installed;
|
||||
}
|
||||
|
||||
pub fn remove() void {
|
||||
collect_hook.detach();
|
||||
link_hook.detach();
|
||||
phase = .idle;
|
||||
}
|
||||
@@ -43,6 +43,7 @@ const clip_sse = @import("clip_sse.zig");
|
||||
const cull_sse = @import("cull_sse.zig");
|
||||
const silicon_sse = @import("silicon_sse.zig");
|
||||
const luaalloc = @import("luaalloc.zig");
|
||||
const luagc = @import("luagc.zig");
|
||||
|
||||
const renderParticleSprites_SSE = particle_sse.renderParticleSprites_SSE;
|
||||
const resetParticleCache = particle_sse.resetParticleCache;
|
||||
@@ -326,7 +327,10 @@ pub fn installHooks() void {
|
||||
if (inflate_hook.install()) installed += 1;
|
||||
|
||||
// Lua slab allocator replacement
|
||||
installed += luaalloc.install();
|
||||
// installed += luaalloc.install(); // disabled for testing
|
||||
|
||||
// Incremental GC -- disabled, needs more research
|
||||
// installed += luagc.install();
|
||||
}
|
||||
|
||||
pub fn lateInit() void {
|
||||
@@ -336,6 +340,7 @@ pub fn lateInit() void {
|
||||
|
||||
pub fn removeHooks() void {
|
||||
if (g_is_hook_owner) {
|
||||
luaalloc.dumpStats(); // dump size histogram before unhooking
|
||||
filecache.remove();
|
||||
inflate_hook.remove();
|
||||
transform_hook.detach();
|
||||
|
||||
Reference in New Issue
Block a user