luagc: per-step timing, forced GC honoring, weirdperformance compat

- Per-step rdtsc timing: ZGCStats returns max_step_us and atomic_us
- AllocBench /zgcstats frame shows step/atomic timing with color coding
- /gccompare command for timing forced GC cycles
- Honor lua_setgcthreshold forced collections (collectgarbage() with low
  threshold now triggers GC instead of being silently ignored)
- Disable luaalloc in weirdperformance when luagc is active (no conflict)
- Closures go to grayagain (covers all upvalue mutation paths)
This commit is contained in:
MarcelineVQ
2026-04-11 10:29:39 -07:00
parent ce5a2408d5
commit 8d3460cf1e
7 changed files with 1038 additions and 15 deletions
+55 -8
View File
@@ -48,6 +48,8 @@ const Stats = struct {
gray_peak: u32 = 0,
dead_freed_last: u32 = 0,
strings_freed_last: u32 = 0,
max_step_us: u32 = 0, // longest single step in microseconds (last cycle)
atomic_us: u32 = 0, // atomic phase duration in microseconds (last cycle)
};
var stats: Stats = .{};
@@ -57,6 +59,25 @@ var cur_mark_steps: u32 = 0;
var cur_sweep_steps: u32 = 0;
var cur_dead_freed: u32 = 0;
var cur_strings_freed: u32 = 0;
var cur_max_step_cycles: u64 = 0;
var cur_atomic_cycles: u64 = 0;
/// Read CPU timestamp counter for timing.
inline fn rdtsc() u64 {
var lo: u32 = undefined;
var hi: u32 = undefined;
asm volatile ("rdtsc"
: [lo] "={eax}" (lo),
[hi] "={edx}" (hi),
);
return @as(u64, hi) << 32 | lo;
}
/// Convert TSC cycles to microseconds (approximate, assumes ~3GHz).
/// Good enough for relative comparisons.
fn cyclesToUs(cycles: u64) u32 {
return @intCast(@min(cycles / 3000, 0xFFFFFFFF));
}
// =============================================================================
// GC State Machine
@@ -807,6 +828,7 @@ fn markroot(g: u32) void {
/// Process up to CHUNK_SIZE objects from the gray stack.
/// Returns true when gray stack is drained (mark complete).
fn markStep() bool {
const t0 = rdtsc();
cur_mark_steps += 1;
var processed: u32 = 0;
@@ -817,6 +839,8 @@ fn markStep() bool {
processed += 1;
}
const elapsed = rdtsc() - t0;
if (elapsed > cur_max_step_cycles) cur_max_step_cycles = elapsed;
return gray_count == 0;
}
@@ -843,6 +867,8 @@ fn propagateall() void {
/// Atomic phase. Matches Lua 5.1 atomic() (lgc.c:525-553).
/// This runs to completion in a single call -- no yielding.
fn atomicPhase(L: u32, g: u32) void {
const t0 = rdtsc();
// (1) Propagate any remaining gray objects
propagateall();
@@ -893,6 +919,8 @@ fn atomicPhase(L: u32, g: u32) void {
weak_table_count = 0;
grayagain_count = 0;
cur_atomic_cycles = rdtsc() - t0;
}
/// Walk tmudata and re-mark each entry so references survive sweep.
@@ -991,12 +1019,17 @@ fn isCleared(tv_tt: u8, tv_gc: u32) bool {
// field that leads to the current object (initially &g->rootgc).
fn sweepRootgcStep(L: u32) bool {
const t0 = rdtsc();
cur_sweep_steps += 1;
var processed: u32 = 0;
while (processed < CHUNK_SIZE) {
const obj = readU32(sweep_prev_next);
if (obj == 0) return true; // end of list, sweep done
if (obj == 0) {
const elapsed = rdtsc() - t0;
if (elapsed > cur_max_step_cycles) cur_max_step_cycles = elapsed;
return true;
}
if (!isDead(obj)) {
// Alive: set to current white for next cycle.
@@ -1013,9 +1046,10 @@ fn sweepRootgcStep(L: u32) bool {
processed += 1;
}
// Bump threshold so GC doesn't re-trigger immediately
const elapsed = rdtsc() - t0;
if (elapsed > cur_max_step_cycles) cur_max_step_cycles = elapsed;
bumpThresholdHeadroom(saved_g);
return false; // more to sweep
return false;
}
// =============================================================================
@@ -1049,6 +1083,7 @@ fn sweepRootudata(L: u32, g: u32) void {
// Chunks by processing N buckets per call.
fn sweepStringsStep(L: u32, g: u32) bool {
const t0 = rdtsc();
cur_sweep_steps += 1;
const hash_array = readU32(g + offsets.GS_strt_hash);
const bucket_count = readU32(g + offsets.GS_strt_size);
@@ -1077,6 +1112,9 @@ fn sweepStringsStep(L: u32, g: u32) bool {
buckets_done += 1;
}
const elapsed = rdtsc() - t0;
if (elapsed > cur_max_step_cycles) cur_max_step_cycles = elapsed;
if (sweep_string_bucket >= bucket_count) return true;
bumpThresholdHeadroom(g);
@@ -1232,6 +1270,8 @@ fn collectGarbageDetour(L: u32) callconv(hook.cc.fastcall) void {
cur_sweep_steps = 0;
cur_dead_freed = 0;
cur_strings_freed = 0;
cur_max_step_cycles = 0;
cur_atomic_cycles = 0;
dbg_table_count = 0;
dbg_closure_count = 0;
dbg_thread_count = 0;
@@ -1345,6 +1385,10 @@ fn finalizeCycle(L: u32, g: u32) void {
stats.sweep_steps_last = cur_sweep_steps;
stats.dead_freed_last = cur_dead_freed;
stats.strings_freed_last = cur_strings_freed;
stats.max_step_us = cyclesToUs(cur_max_step_cycles);
stats.atomic_us = cyclesToUs(cur_atomic_cycles);
cur_max_step_cycles = 0;
cur_atomic_cycles = 0;
if (stats.cycles_total <= 20 or stats.cycles_total % 100 == 0) {
const tb = readU32(g + offsets.GS_totalbytes);
@@ -1402,7 +1446,6 @@ const SetGcThresholdFn = fn (u32, u32) callconv(hook.cc.fastcall) void;
var setgcthreshold_hook: hook.Detour(SetGcThresholdFn) = .{};
fn setgcthresholdDetour(L: u32, newthreshold: u32) callconv(hook.cc.fastcall) void {
_ = newthreshold;
if (readU32(L + offsets.L_active_check) == 0) return;
// 5.1 luaC_fullgc pattern: if a forced GC arrives mid-cycle,
@@ -1450,9 +1493,11 @@ fn setgcthresholdDetour(L: u32, newthreshold: u32) callconv(hook.cc.fastcall) vo
const g = getGlobalState(L);
const tb = readU32(g + offsets.GS_totalbytes);
const thr = readU32(g + offsets.GS_gcthreshold);
if (thr > tb) return;
collectGarbageDetour(L);
const requested = @as(u64, newthreshold) << 10;
if (requested < tb) {
// Caller wants to force GC (e.g. collectgarbage() passes 0)
collectGarbageDetour(L);
}
}
// =============================================================================
@@ -1863,7 +1908,9 @@ pub fn luaZGCStats(L: lua.State) callconv(hook.cc.fastcall) i32 {
lua.pushnumber(L, @floatFromInt(stats.dead_freed_last));
lua.pushnumber(L, @floatFromInt(stats.strings_freed_last));
lua.pushnumber(L, @floatFromInt(@intFromEnum(phase)));
return 7;
lua.pushnumber(L, @floatFromInt(stats.max_step_us));
lua.pushnumber(L, @floatFromInt(stats.atomic_us));
return 9;
}
// =============================================================================
+39
View File
@@ -0,0 +1,39 @@
# weirdperformance
SSE optimization subsystem for WoW 1.12.1. Bone transforms, particle systems, frustum culling, file caching, timer fixes, and fast zlib via libdeflate. Enabled by default in build.
## WHERE TO LOOK
| File | Purpose |
|------|---------|
| `weirdperformance.zig` | Module entry, build integration |
| `bone_sse.zig` | Bone transformation via SSE SIMD |
| `particle_sse.zig` | Particle system SSE optimization |
| `clip_sse.zig` | Frustum culling with SSE |
| `cull_sse.zig` | Additional culling routines |
| `silicon_sse.zig` | LibSiliconPatch port |
| `entity_sse.zig` | Entity processing acceleration |
| `filecache.zig` | In-memory file caching |
| `inflate_hook.zig` | Compression hook, deflate integration |
| `timer_fix.zig` | High-resolution timer fixes |
| `libdeflate/` | Fast zlib replacement (libdeflate) |
## CONVENTIONS
- SSE intrinsics use `std.math.losslessCast` for float/int bit conversion
- All SSE code gated behind `has_sse41` compile check
- Filecache uses named mutex for thread-safe access
- Hook installation deferred to first render, not DLL attach
## ANTI-PATTERNS
- NEVER call SSE code without checking CPU feature support at runtime
- NEVER enable libdeflate hook before game engine init completes
- NEVER use filecache mutex during WoW UI thread - causes deadlocks
## NOTES
- libdeflate provides 2-5x decompression speedup over stock zlib
- Bone SSE assumes bone matrices are 16-byte aligned
- Timer fix resolves GetTickCount rollover on long sessions
- All hooks use per-feature named mutexes to prevent duplicate installation
@@ -102,6 +102,175 @@ end
SLASH_ALLOCBENCH1 = "/allocbench"
SlashCmdList["ALLOCBENCH"] = run_bench
-- =========================================================================
-- /gccompare: time a forced full GC cycle.
-- Builds a heap first, then times collectgarbage("collect").
-- Run with our GC and with native (DIAG_MODE=1) to compare.
-- =========================================================================
SLASH_GCCOMPARE1 = "/gccompare"
SlashCmdList["GCCOMPARE"] = function(args)
local size = tonumber(args) or 100
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00GC Compare|r: building %dk object heap...", size))
-- Build a realistic mixed-age heap
local heap = {}
local closures = {}
for i = 1, size * 1000 do
local mode = math.mod(i, 4)
if mode == 0 then
heap[i] = { name = "obj" .. i, value = i, sub = { i, i+1 } }
elseif mode == 1 then
heap[i] = "string_key_" .. i
elseif mode == 2 then
local captured = i
closures[i] = function() return captured end
heap[i] = closures[i]
else
heap[i] = { [tostring(i)] = true, flag = i > size * 500 }
end
end
-- Kill half the heap to create garbage
for i = 1, size * 500 do
heap[i] = nil
closures[i] = nil
end
DEFAULT_CHAT_FRAME:AddMessage(string.format(
" heap built: %dk live, %dk garbage. Timing GC...", size / 2, size / 2))
-- Time 5 forced collections
local times = {}
for trial = 1, 5 do
-- Recreate some garbage between trials
for i = 1, size * 100 do
local _ = { i, "tmp" .. i }
end
debugprofilestart()
collectgarbage()
local elapsed = debugprofilestop()
table.insert(times, elapsed)
DEFAULT_CHAT_FRAME:AddMessage(string.format(
" trial %d: |cffffd700%.2f ms|r", trial, elapsed))
end
-- Find min/max/avg
local min_t, max_t, sum = 999999, 0, 0
for _, t in ipairs(times) do
if t < min_t then min_t = t end
if t > max_t then max_t = t end
sum = sum + t
end
local avg = sum / table.getn(times)
local gc_type = "incremental"
if type(ZGCStats) == "function" then
local _, mark_steps = ZGCStats()
if mark_steps <= 1 then gc_type = "atomic" end
else
gc_type = "native"
end
DEFAULT_CHAT_FRAME:AddMessage(string.format(
"|cff00ff00GC Compare|r [%s]: min=%.2f max=%.2f avg=%.2f ms (%dk heap)",
gc_type, min_t, max_t, avg, size))
-- Cleanup
heap = nil
closures = nil
collectgarbage()
end
-- =========================================================================
-- zluagen stats frame — visible by default, center screen, movable
-- ZGCStats() is registered by zluagen on its first GC tick.
-- Returns: cycles_total, cycles_major, cycles_minor,
-- mark_steps_last, sweep_steps_last,
-- gray_peak, touched_peak, current_phase
-- =========================================================================
local zgc_frame = CreateFrame("Frame", "ZGCStatsFrame", UIParent)
zgc_frame:SetWidth(280)
zgc_frame:SetHeight(120)
zgc_frame:SetPoint("CENTER", UIParent, "CENTER", 0, 0)
zgc_frame:SetBackdrop({
bgFile = "Interface\\DialogFrame\\UI-DialogBox-Background",
edgeFile = "Interface\\DialogFrame\\UI-DialogBox-Border",
tile = true, tileSize = 16, edgeSize = 16,
insets = { left = 4, right = 4, top = 4, bottom = 4 },
})
zgc_frame:SetBackdropColor(0, 0, 0, 0.7)
zgc_frame:EnableMouse(true)
zgc_frame:SetMovable(true)
zgc_frame:RegisterForDrag("LeftButton")
zgc_frame:SetScript("OnDragStart", function() this:StartMoving() end)
zgc_frame:SetScript("OnDragStop", function() this:StopMovingOrSizing() end)
local zgc_title = zgc_frame:CreateFontString(nil, "OVERLAY", "GameFontNormal")
zgc_title:SetPoint("TOP", zgc_frame, "TOP", 0, -8)
zgc_title:SetText("|cff00ff00zluagen|r")
local zgc_text = zgc_frame:CreateFontString(nil, "OVERLAY", "GameFontHighlightSmall")
zgc_text:SetPoint("TOPLEFT", zgc_frame, "TOPLEFT", 12, -28)
zgc_text:SetPoint("BOTTOMRIGHT", zgc_frame, "BOTTOMRIGHT", -12, 8)
zgc_text:SetJustifyH("LEFT")
zgc_text:SetJustifyV("TOP")
zgc_text:SetText("waiting for DLL...")
local zgc_last_update = 0
zgc_frame:SetScript("OnUpdate", function()
local now = GetTime()
if now - zgc_last_update < 0.2 then return end -- ~5Hz refresh
zgc_last_update = now
if type(ZGCStats) ~= "function" then
zgc_text:SetText("|cffff8800ZGCStats() not registered|r\n\n" ..
"zluagen.dll may not be loaded.\n" ..
"Try |cffffd700/run collectgarbage()|r to trigger first GC.")
return
end
local total, mark_steps, sweep_steps, gray_peak, freed, freed_str, ph, max_step_us, atomic_us = ZGCStats()
local phase_name = "?"
if ph == 0 then phase_name = "idle"
elseif ph == 1 then phase_name = "marking"
elseif ph == 2 then phase_name = "atomic"
elseif ph == 3 then phase_name = "sweep_str"
elseif ph == 4 then phase_name = "sweeping"
elseif ph == 5 then phase_name = "finalize"
end
local step_color = "|cff00ff00"
if max_step_us > 2000 then step_color = "|cffff0000"
elseif max_step_us > 1000 then step_color = "|cffffff00"
end
zgc_text:SetText(string.format(
"cycles: |cffffd700%d|r phase: %s\n" ..
"last: |cffffd700%d|r mark / |cffffd700%d|r sweep steps\n" ..
"freed: %d obj + %d str\n" ..
"step max: %s%d us|r atomic: %s%d us|r\n" ..
"gray peak: %d",
total, phase_name,
mark_steps, sweep_steps,
freed, freed_str,
step_color, max_step_us, step_color, atomic_us,
gray_peak))
end)
-- /zgcstats toggles visibility
SLASH_ZGCSTATS1 = "/zgcstats"
SlashCmdList["ZGCSTATS"] = function()
if zgc_frame:IsVisible() then
zgc_frame:Hide()
else
zgc_frame:Show()
end
end
-- =========================================================================
-- GC stress test: /gcstress [size_k] [churn_pct] [duration]
-- Builds a large live heap then churns a fraction of it each frame.
+637
View File
@@ -0,0 +1,637 @@
# WoW Lua 5.0 GC System
Notes on the GC machinery of WoW 1.12.1's modified Lua 5.0, based on
disassembly (Ghidra) and cross-reference with the official Lua 5.0.3
source (`lgc.c`/`lgc.h`). Addresses are client-absolute (build 5875).
## High-Level Structure
WoW's Lua uses a **stop-the-world mark-and-sweep** collector. No tri-color
incremental marking, no generational logic, no write barriers in the
generational sense. Collection is atomic: the entire world freezes for one
full mark+sweep pass.
Key entry point: `luaC_collectgarbage` (0x6F7340).
Sequence inside `luaC_collectgarbage`:
1. `lua_gc_full_collection` (0x6F73E0) - the mark phase
2. `lua_gc_remove_objects(L, &g->rootudata, 0)` (0x6F7210) - sweep userdata
3. `lua_gc_sweep_all_lists(L, 0)` (0x6F72F0) - sweep string hash table
4. `lua_gc_remove_objects(L, &g->rootgc, 0)` - sweep rootgc list
5. `lua_gc_shrink_memory(L)` (0x6F7370) - string table shrink + threshold
6. `luaCallUserDataGC(L)` (0x6F7080) - run `__gc` finalizers
## GCObject Header
Every GC object starts with a common header:
```
offset 0x00 GCObject *next ; linked list next
offset 0x04 lu_byte tt ; type tag
offset 0x05 lu_byte marked ; GC mark + flags
offset 0x06 lu_byte ... ; type-specific (e.g. Table.flags)
```
## Type Tags
Standard Lua 5.0 values (verified via propagate dispatch at 0x6F7510):
| Value | Type |
|-------|----------------|
| 0 | LUA_TNIL |
| 1 | LUA_TBOOLEAN |
| 2 | LUA_TLIGHTUSERDATA |
| 3 | LUA_TNUMBER |
| 4 | LUA_TSTRING |
| 5 | LUA_TTABLE |
| 6 | LUA_TFUNCTION |
| 7 | LUA_TUSERDATA |
| 8 | LUA_TTHREAD |
| 9 | LUA_TPROTO |
`LUA_TUPVAL` exists internally (type 10?) but is NOT in the propagate
dispatch; upvalues are traversed via `mark_closure` from the containing
closure.
## `marked` Byte Layout
Bits (verified from disassembly + Lua 5.0 `lgc.h`):
```
bit 0 -- mark flag (1 = reachable/traversed)
bit 1 -- KEYWEAK (for tables with __mode key-weak)
bit 2 -- VALUEWEAK (for tables with __mode value-weak)
bit 3 -- FINALIZED (userdata: __gc already run)
bit 4 -- propagation flag (set together with bit 0; see below)
bit 5-7 -- UNUSED (but see NOTE below)
```
**Critical**: the check used by `mark_table` / `mark_closure` / etc. when
following a child reference is:
```asm
TEST byte [child+5], 0x11 ; bits 0 and 4
JNZ skip ; if either is set, skip re-marking
```
This matches Lua 5.0's `ismarked` macro: `(marked & ((1<<4)|1))`.
Either bit counts as "already processed". Bit 4 is (in standard Lua 5.0)
set for tables with `WEAKKEY|WEAKVALUE` (`(1<<4)|1 == 0x11` test). In
practice our analysis of `mark_table` didn't show explicit bit 4 writes,
so this is either dead state from the original Lua behavior or used
implicitly via macro expansion.
**Additional critical**: `mark_closure` at 0x6F7828 uses a DIFFERENT check
for upvalues:
```asm
MOV AL, [upval+5] ; whole marked byte
TEST AL, AL
JNZ skip ; any non-zero = already processed
```
And writes the entire marked byte to 0x01 at 0x6F7840:
```asm
MOV byte [upval+5], 0x01
```
**This means**: any non-zero bits in an upvalue's marked byte cause it to
be skipped. We CANNOT use any bit in an upvalue's marked byte for our own
flags. Age bits in bits 5-7 break upvalue traversal.
## Sweep Invariants
`lua_gc_remove_objects` (0x6F7210) sweeps a list. Per-object behavior:
```asm
MOV AL, [obj+5] ; load marked
XOR ECX, ECX
MOV CL, AL ; ECX = marked (zero-extended)
CMP ECX, [limit] ; threshold (typically 0)
JLE dead ; signed <=; alive otherwise
AND AL, 0xFE ; clear bit 0 on survivor
MOV [obj+5], AL ; preserves bits 1..7
... ; advance to next
dead:
; unlink and free via lua_gc_free_object (0x6F7260)
```
Properties:
- Alive if `marked > limit` (ZXed byte compared as 32-bit int)
- Alive case clears ONLY bit 0 (AND 0xFE), preserves everything else
- Dead case frees via `lua_gc_free_object` which type-dispatches to
`luaH_free` / `luaF_freeproto` / etc., eventually back to our slab
allocator's `slabFree`
## Mark Phase (`lua_gc_full_collection`)
Flow (from disassembly of 0x6F73E0):
```
// local GCState st on stack (20 bytes, see layout below)
st.tmark = st.wk = st.wv = st.wkv = NULL;
st.g = L->l_G;
markroot(&st, L); ; 0x6F7AB0
propagatemarks(&st); ; 0x6F7510
cleartablevalues(st.wkv); ; 0x6F7A20 with st.wkv
cleartablevalues(st.wv); ; 0x6F7A20 with st.wv
wkv_saved = st.wkv;
st.wkv = NULL;
st.wv = NULL;
luaC_separateudata(L); ; 0x6F6FF0 -- move dead udata to tmudata
marktmu(&st); ; 0x6F7470 -- unmark + re-mark tmudata
propagatemarks(&st); ; second propagation pass
cleartablekeys(wkv_saved); ; 0x6F7970
cleartablekeys(st.wk); ; 0x6F7970
cleartablevalues(st.wv); ; 0x6F7A20
cleartablekeys(st.wkv); ; 0x6F7970
cleartablevalues(st.wkv); ; 0x6F7A20
```
Matches the Lua 5.0.3 `mark()` function in `lgc.c` exactly.
### `GCState` Layout (20 bytes)
```
offset 0x00 GCObject *tmark ; main gray list (to-traverse)
offset 0x04 GCObject *wk ; weak-key tables pending clear
offset 0x08 GCObject *wv ; weak-value tables pending clear
offset 0x0C GCObject *wkv ; weak key+value tables
offset 0x10 global_State *g
```
Stored as a local on the caller's stack. All mark functions take
ECX = `&GCState` as their first argument (fastcall).
### `markroot` (0x6F7AB0)
Signature: `__fastcall(ECX=GCState*, EDX=lua_State*)`.
Body (from disasm):
1. If `g->defaultmeta` (at `g+0x48`) is a table (type at `g+0x40` >= 4):
`mark_object_gray(&st, defaultmeta)`
2. If `g->registry` (at `g+0x38`) is a table (type at `g+0x30` >= 4):
`mark_object_gray(&st, registry)`
3. `mark_thread(&st, g->mainthread)` (traversestack on main thread)
4. If `L != g->mainthread`: `mark_object_gray(&st, L)`
### `mark_object_gray` (reallymarkobject) (0x6F74A0)
```asm
OR byte [obj+5], 0x01 ; set mark bit
MOVZX EAX, [obj+4] ; type tag
SUB EAX, 0x5 ; 0..4 for types 5..9
CMP EAX, 0x4
JA return ; type < 5 or > 9: just mark, don't push
JMP [dispatch + EAX*4] ; type-specific: push to appropriate list
```
Dispatch (from jump table at 0x6F7574):
| Type | Push Offset | What |
|------|--------------------------------|----------------|
| 5 | `[obj+0x18] = st.tmark; ...` | Table → tmark |
| 6 | `[obj+0x08] = st.tmark; ...` | Closure → tmark |
| 7 | (just set mark, no push) | Userdata |
| 8 | `[obj+0x54] = st.tmark; ...` | Thread → tmark |
| 9 | `[obj+0x40] = st.tmark; ...` | Proto → tmark |
The offsets are the `gclist` field within each type's struct.
Note: `mark_object_gray` has **no skip check** at entry. It always sets
bit 0 and dispatches. The "skip if already marked" check happens at
CALLERS of `mark_object_gray` via `TEST 0x11` before the call.
### `propagatemarks` (0x6F7510)
Drains `st.tmark`:
```
while (st.tmark) {
o = st.tmark;
switch (o->tt) {
case 5: st.tmark = o->gclist; mark_table(&st, o); // 0x6F7590
case 6: st.tmark = o->gclist; mark_closure(&st, o); // 0x6F77B0
case 7: st.tmark = o->gclist; /* skip */ // (just next)
case 8: st.tmark = o->gclist; mark_thread(&st, o); // 0x6F7860
case 9: st.tmark = o->gclist; mark_proto(&st, o); // 0x6F7710
}
}
```
### `mark_table` (traversetable) (0x6F7590)
1. Mark metatable (`[table+8]`): if `TEST [mt+5], 0x11` == 0,
`mark_object_gray(mt)`
2. Check metatable flags byte at `[mt+6]` bit 3 for weak mode
3. Resolve `__mode` metamethod via a function call to 0x6F7BA0, parse
"k"/"v" flags
4. Update weak flags in `[table+5]` bits 1-2 (`AND 0xF9 | new_flags<<1`)
5. If weak: push onto wk/wv/wkv list instead of marking children
6. Mark hash nodes: for each `Node[i]`, check TEST 0x11 on key gc_ptr
and value gc_ptr; call `mark_object_gray` if not set
### `mark_closure` (traverseclosure) (0x6F77B0)
Two branches based on `[closure+6]` (isC flag at offset 6):
- **C closure**: iterate `[closure+0x18 + i*0x10]` TValues (upvalue array
is inline). For each: TEST 0x11, mark_object_gray if type >= 4.
- **Lua closure**:
- Mark `closure.env` at `[closure+0x18]` (TEST 0x11, mark if unset)
- Mark `closure.p` (proto) at `[closure+0x0C]`
- Iterate `[closure+0x20 + i*4]` which are UpVal* pointers.
For each UpVal:
- **`TEST AL, AL` on `[upval+5]` (whole byte)** — skip if non-zero
- Otherwise: check the upval's value at `[upval+0x10]` (tt) and
`[upval+0x18]` (gc_ptr) via TEST 0x11, mark if collectable
- Unconditionally write `[upval+5] = 0x01` to mark upval as processed
The `TEST AL, AL` is the critical constraint: upvalues use the entire
marked byte as a "processed" flag. Any non-zero value (including our age
bits) causes the upval to be skipped from traversal.
### `cleartable` Functions (0x6F7A20 and 0x6F7970)
Two separate functions:
- **0x6F7A20** (`cleartablevalues`): walks a list of weak tables linked
by `gclist`. For each table, iterates hash nodes. For each node with
non-nil value, checks `ismarked(gcvalue(value))` — if the value is
not marked (bit 0 clear), sets the value to nil (`removekey`).
- **0x6F7970** (`cleartablekeys`): similar but for keys. If the key
gc_ptr is not marked, removes the entire entry (sets key type to
LUA_TNONE, value to nil).
Both read `marked & 0x01` (or `marked & 0x11`) to determine liveness.
**Strings are auto-marked** during this check (`stringmark(s)` is called
on any collectable string, regardless of weak status, because strings
are "values" not "entries").
## Root Set
Per `markroot` analysis, the roots are:
- `g->defaultmeta` (a metatable shared by all tables without explicit one)
- `g->registry` (the registry table, holds C API references)
- Main thread's **entire stack + CallInfo ranges** (via `mark_thread`)
- Current running `L` (if different from main thread)
**NOT traditional roots** (handled as children of other roots):
- Globals table — it's `mainthread->gt`, marked by `mark_thread`
- Open upvalues — `mainthread->openupval`, marked by `mark_thread`
walking its `openupval` list (0x6F7860 traverses this)
## Write Barriers in the Stock GC
**There aren't any "write barriers" in the generational sense.** The
stop-the-world mark-and-sweep doesn't need them.
There IS a set of global variables that LOOK like barrier machinery:
- `0xCEEAC4` - a "barrier active" flag, non-zero during certain
operations
- `0xCEEAC0` - a "current gc object" slot
And the pattern at table-write sites (e.g., 0x6F7FC7 in
`lua_set_table_value`, 0x6FA945 in `lua_table_new_key`):
```asm
MOV EAX, [src + 4] ; value's gc_ptr
TEST EAX, EAX
JZ skip ; nil/non-collectable
CMP [0xCEEAC4], 0 ; barrier flag
JZ skip ; flag clear, skip
MOV [0xCEEAC0], EAX ; record the gc_ptr
skip:
```
**This is NOT a write barrier for generational GC.** It's actually the
`fix for gcvalue` / `gcvalue save` mechanism from Lua 5.0's handling of
C API functions that might trigger GC while holding an unstable
reference. The `0xCEEAC0` slot holds a GC object that must be kept alive
across a GC step.
References to `0xCEEAC0` appear in ~100 places throughout the Lua C API,
almost all following the save/restore pattern:
```
MOV EAX, [0xCEEAC0] ; save current
... ; do work that might GC
MOV [0xCEEAC0], EAX ; restore
```
Attempting to piggyback generational write-barrier logic on these sites
is misguided: the flag `0xCEEAC4` is often zero during normal execution,
so writes don't hit these code paths, and the sites only record the
VALUE, not the destination object.
## Table Write Paths (Relevant for Real Write Barriers)
All table-value writes funnel through one of two primitives:
| Function | Address | Callers |
|---------------------------|-----------|--------------------------------------------------------------------------------------------|
| `lua_table_set_value` | 0x6FA840 | `lua_rawseti` (0x6F3EA0), `luaPackVarArgs` (0x6F6200), `lua_set_table_value` (0x6F7F40), `lua_table_resize` (0x6FAB90), `StoreLuaConstant` (0x700AD0) |
| `lua_table_set_int_key` | 0x6FAD80 | `lua_rawseti` (0x6F3F60), `luaPackVarArgs` (0x6F6200), `lua_vm_execute` (0x6F8720, SETLIST opcode), `lua_table_resize` (0x6FAB90) |
Both ultimately call `lua_table_new_key` (0x6FA8A0) for new-key insertion.
`lua_table_new_key` is only called from these two primitives; barriering
both covers 100% of hash-node creation.
`lua_set_table_value` (0x6F7F40) is the high-level wrapper used by
`lua_settable`, `lua_rawset`, `lua_setfield`, `lua_setglobal`, and all
VM opcodes for table writes.
**Not covered by these primitives:**
- `lua_setmetatable` (0x6F4020) — directly writes `table->metatable`,
bypasses `lua_table_set_value`. Callers: `luaL_create_weak_table`
(0x6F4FC0), `RegisterFrameScriptReference` (0x701BD0). Rare, setup-time.
- Direct C writes by the game engine to internal tables — unquantified
risk, probably limited to engine-created structures.
- Thread stack pushes (`lua_pushvalue`, etc.) — stack is always walked
by `mark_thread` every collection, so no barrier needed as long as
threads are always traversed.
## Freeing Objects
`lua_gc_free_object` (0x6F7260) dispatches by type:
- Proto → `luaF_freeproto`
- Function → `luaF_freeclosure`
- Upval → `luaM_freelem`
- Table → `luaH_free` (frees array + node arrays + struct)
- Thread → `luaE_freethread`
- String → `luaM_free` (removes from string hash)
- Userdata → `luaM_free`
All eventually reach `luaM_realloc` (0x6FC980) which does accounting
and calls through to the allocator (our `memory_pool_allocate` hook →
slab allocator).
## Birth Mark Patches
Locations where `luaC_link` / `lua_create_string_object` write the
initial `marked` byte for new objects:
- `luaC_link` (0x6F7B20): imm8 at 0x6F7B37 (`MOV byte [EDX+5], 0x0`)
- `lua_create_string_object` (0x6F9D90): imm8 at 0x6F9DC1
The incremental GC module patches these to `0x01` during chunked sweep
so new objects are born marked-alive and survive until the next cycle.
## Generational GC Constraints (what we can and cannot do)
From all of the above, the hard constraints on any generational
retrofit:
1. **Cannot use the marked byte for age tracking.** Bits 0+4 are used
for mark state. Bits 1+2 are used for weak flags. Upvalues use the
entire byte as a "processed" flag (`TEST AL, AL`). Any stray bits
cause subtle corruption.
2. **Must use external age storage.** We use per-page bitmaps in the
slab allocator (1 bit per slot), indexed by segment + slot index.
3. **Must clear the age bitmap on free.** Freed slots return to the
slab's free list. When reallocated, the new object inherits the
freed object's bitmap bit. This caused a multi-hour debugging
session — the fix is in `slabFree` to clear the bit.
4. **Write barriers must cover all table writes that could create
old→young references.** `lua_table_set_value` and
`lua_table_set_int_key` cover ~99% of paths. `lua_setmetatable` is
the known gap.
5. **Skipping old objects during mark is achievable** by pre-setting
bit 0 on old non-touched objects before calling `lua_gc_full_collection`.
`mark_table`'s `TEST 0x11` check catches them at child references
and skips `mark_object_gray`, which prevents them from being added
to the gray list, which prevents their subtree from being traversed.
6. **Pre-marking must NOT touch non-table objects.** Closures need full
traversal because upvalue values can change via `setupvalue` which
isn't barriered. Threads need full traversal because stack contents
change without barriers. Pre-marking upvalues directly would break
the `TEST AL, AL` check in `mark_closure`.
## Heap Size and Object Count
Observed characteristics from profiling + existing luagc module:
- **Steady-state rootgc size**: on the order of **100k-300k objects** in
a typical addon-heavy gameplay session. Not precisely measured, but
the incremental sweep module uses `CHUNK_SIZE = 50000` and takes
multiple chunks on a full sweep (observed "chunk0"/"chunkN"/"final"
log lines during gameplay).
- **Pre-generational stop-the-world sweep**: up to **5 seconds** worst
case during addon loading bursts. Main cost is rootgc sweep; mark
phase alone is ~80ms.
- **String count**: not instrumented separately but visible via the
string hash sweep (`lua_gc_sweep_all_lists`) which costs ~44-92ms per
full cycle. String hash table has `g->strt.size` buckets (power of 2,
grows with `strt.nuse`).
- **Userdata count**: small (hundreds at most). Sweep is atomic and
fast; not a contributor to GC stutter.
- **Total memory footprint**: accounted via `g->totalbytes`
(`global_State + 0x28`). Typical session: tens of MB to >100 MB.
- **Allocator size classes** (from slab profiling via `luaalloc`):
dominant sizes are **40 bytes** (Lua Table nodes), **80/160/320/640**
(power-of-2 hash arrays of nodes), and **~24-48 bytes** (small
objects like TString headers, closures). Large allocations (>4KB) are
rare.
Implication: full mark of ~200k objects in ~80ms = ~400ns per object.
Full sweep at similar rates. Any generational scheme that lets us skip
90% of the heap gives ~8ms mark + sweep on the common minor path, well
under one-frame budget.
## Threading Model
**WoW 1.12 is effectively single-threaded from Lua's perspective.**
- The main game thread runs the Lua VM, executes scripts, handles
events, does rendering setup. All Lua stack manipulation, all GC
runs, all addon code executes here.
- Other OS threads exist (sound, network, worker pool, D3D device
thread) but **none of them touch the Lua state**. They communicate
via OS-level synchronization (events, queues) and never access
`global_State` or any `lua_State`.
- Our hooks run on whichever thread calls the hooked function. For
`memory_pool_allocate`, `luaC_collectgarbage`, `lua_table_set_value`,
`lua_gc_free_object`, this is always the main thread.
**Consequences for write barriers:**
- No synchronization needed on `touched_set` or the age bitmap.
- No race between a write barrier fire and a concurrent GC step.
- `in_gc` flag is sufficient to detect GC re-entrance (finalizers
allocating, etc.); no need for atomics.
- The pre-mark / post-mark phase transitions don't need memory barriers.
This is a significant simplification compared to generational GCs in
multithreaded runtimes (HotSpot, V8, etc.) where every barrier fire
needs at least a relaxed atomic.
## Using Ghidra for Discovery
The workflow we've used for every GC investigation in this project.
Ghidra is run **headless** from CLI (no GUI); see the `ghidra-cli-wow-re`
skill for setup.
### Tool Chain
- **Ghidra 11.4.2** at `/home/august/.ghidra/ghidra_11.4.2_PUBLIC/`
- **Project** `Dis` at `/media/faststore/tmp/Dis`
- **Target binary**: `WoW.exe.multi.latest` (labeled vanilla 1.12.1)
or `WoW.exe.timber` (Turtle WoW with extra labels)
- **Wrapper script**:
`/home/august/.claude/skills/ghidra-cli-wow-re/scripts/run-analysis.sh`
takes a Python/Jython script path and runs it headless against the
default binary, dumping stdout.
### Standard Steps
**Step 1: Write a Python analysis script to `/tmp`.**
Jython 2.7 with Ghidra's scripting API (`currentProgram`, `getReferencesTo`,
`getFunctionContaining`, etc.). Example patterns we use constantly:
```python
# Disassemble a range
listing = currentProgram.getListing()
def disasm(start, length, label):
print("\n=== %s (0x%08x) ===" % (label, start))
addr = toAddr(start)
end = start + length
count = 0
while count < 200:
insn = listing.getInstructionContaining(addr)
if insn is None:
b = getByte(addr)
print(" 0x%08x %02x <data>" % (addr.getOffset(), b & 0xff))
addr = addr.add(1)
else:
raw = ""
for i in range(insn.getLength()):
b = getByte(insn.getAddress().add(i))
raw += "%02x " % (b & 0xff)
print(" 0x%08x %-28s %s" % (
insn.getAddress().getOffset(), raw.strip(), insn.toString()))
addr = insn.getAddress().add(insn.getLength())
count += 1
if addr.getOffset() >= end:
break
# Find callers of a function
def get_callers(addr_int, label):
refs = getReferencesTo(toAddr(addr_int))
seen = set()
for ref in refs:
if ref.getReferenceType().isCall():
fn = getFunctionContaining(ref.getFromAddress())
if fn and fn.getEntryPoint().getOffset() not in seen:
seen.add(fn.getEntryPoint().getOffset())
print(" 0x%08x %s" % (fn.getEntryPoint().getOffset(), fn.getName()))
```
**Step 2: Run the script.**
```bash
/home/august/.claude/skills/ghidra-cli-wow-re/scripts/run-analysis.sh /tmp/my_script.py 2>/dev/null | grep "0x006f"
```
The `2>/dev/null` suppresses Ghidra's startup noise; `grep "0x006f"`
filters to output lines referencing the Lua code region (0x6F0000+).
**Step 3: Verify from raw bytes.**
The Ghidra decompiler is a best-guess engine, especially for calling
conventions. Before hooking any function:
1. Read the **prologue** (first 10-20 insns) — confirms register
conventions (`MOV ESI, EDX` ⇒ EDX is the `table` arg, etc.)
2. Check **every RET** for stack cleanup (`RET 4` ⇒ stdcall/fastcall
with 1 stack arg; plain `RET` ⇒ all regs)
3. Cross-reference struct field accesses (`[EDX+0x08]`) against
assumed struct layouts from Lua 5.0 source
### Specific Techniques Used in This Module
**Discovering the marked byte check patterns:**
```python
# Find every instruction of the form `TEST byte [reg+5], imm`
# Pattern: F6 4x 05 imm (where 4x is 40|reg)
import jarray
mem = currentProgram.getMemory()
for reg_byte in [0x40, 0x41, 0x42, 0x43, 0x45, 0x46, 0x47]:
pat = jarray.array([0xF6, reg_byte, 0x05], 'b')
cur = toAddr(0x6F0000)
while True:
found = mem.findBytes(cur, pat, None, True, monitor)
if found is None or found.getOffset() > 0x6FF000:
break
imm = getByte(found.add(3)) & 0xff
fn = getFunctionContaining(found)
print(" 0x%08x TEST [reg+5], 0x%02x in %s" % (
found.getOffset(), imm,
fn.getName() if fn else "?"))
cur = found.add(1)
```
We used this to discover that `TEST 0x11` is the dominant mark check
(bits 0+4) and that `TEST 0xFF` in closure traversal is a different
whole-byte check.
**Finding global_State offsets:**
Start from a known offset (e.g., `rootgc` at `g+0x10`, discovered from
the handoff) and trace code that accesses `[g+X]` in GC-related
functions. Compare with Lua 5.0's `global_State` struct in `lstate.h`
to map offsets to fields. Key offsets we've confirmed:
| Offset | Field | How verified |
|--------|--------------------------|----------------------------------|
| 0x0C | stringtable.hash | Sweep function walks it |
| 0x10 | rootgc | `lua_gc_remove_objects` arg |
| 0x14 | rootudata | Sweep loop iterates this |
| 0x18 | tmudata | `marktmu` walks this |
| 0x24 | GCthreshold | Threshold updates in gc_step |
| 0x28 | totalbytes | Memory accounting |
| 0x30 | registry type tag | `markroot` `[g+0x30]` check |
| 0x38 | registry ptr | `markroot` `[g+0x38]` load |
| 0x40 | defaultmeta type | `markroot` `[g+0x40]` check |
| 0x48 | defaultmeta ptr | `markroot` `[g+0x48]` load |
| 0x50 | mainthread | `markroot` `[g+0x50]` → traversestack |
**Cross-referencing with Lua 5.0 source:**
Download Lua 5.0.3 source once (`/tmp/lua5/lua-5.0.3/`) and grep for
function names and struct fields when Ghidra's decompilation is
ambiguous. The WoW build is a lightly modified Lua 5.0 — most functions
map 1:1 to their `lgc.c` / `ltable.c` / `lobject.h` counterparts.
**Verifying calling conventions:**
Zig's `fastcall` with `@callconv(.fastcall)` puts args in ECX, EDX,
then stack. We verified this matches WoW's by looking at the prologue
of every hookable function. The pattern `MOV ESI, EDX; MOV EBX, ECX`
indicates the 1st and 2nd fastcall args are being saved; stack args
are accessed via `[EBP+8]`, `[EBP+0xC]`, etc.
`RET 4` means one 4-byte stack arg is cleaned by the callee — this
must match the Zig function signature's parameter count.
## Known Unknowns
- `0xCEEAC4` flag: when exactly is it set/cleared? Understanding this
could unlock the existing Lua C API barrier mechanism as an
alternative write barrier path.
- Exact function at 0x6F6FF0 (called between the two propagate passes
in `lua_gc_full_collection`) — likely `luaC_separateudata` but not
verified.
- Whether `lua_table_resize` can move hash nodes in a way that bypasses
our barriers transiently during a write.
+9
View File
@@ -210,6 +210,15 @@ fn slabFree(ptr: u32) void {
if (seg_val == 0) return; // not ours (shouldn't happen)
const class_idx: usize = seg_val - 1;
// Clear the age bitmap bit for this slot. Without this, a reused slot
// would inherit the "old" status of its previous occupant, causing a
// freshly allocated young object to be treated as old by isOld().
const bitmap = age_bitmaps[ptr >> SEGMENT_SHIFT];
if (bitmap) |bm| {
const idx = slotIndex(ptr, class_idx);
bm[idx >> 3] &= ~(@as(u8, 1) << @intCast(idx & 7));
}
const next_ptr: *u32 = @ptrFromInt(ptr);
next_ptr.* = free_lists[class_idx];
free_lists[class_idx] = ptr;
+126 -2
View File
@@ -40,6 +40,7 @@ const lua_gc_shrink_memory: *const fn (u32) callconv(hook.cc.fastcall) void = @p
const luaCallUserDataGC: *const fn (u32) callconv(hook.cc.fastcall) void = @ptrFromInt(0x6F7080);
const lua_gc_sweep_all_lists: *const fn (u32, u32) callconv(hook.cc.fastcall) void = @ptrFromInt(0x6F72F0);
const lua_gc_remove_objects: *const fn (u32, u32, u32) callconv(hook.cc.fastcall) u32 = @ptrFromInt(0x6F7210);
const lua_gc_free_object: *const fn (u32, u32) callconv(hook.cc.fastcall) void = @ptrFromInt(0x6F7260);
var sweeping: bool = false;
var in_gc: bool = false;
@@ -124,6 +125,17 @@ fn tableSetBarrier(L: u32, table: u32, key: u32) callconv(hook.cc.fastcall) u32
return table_set_hook.callOriginal(.{ L, table, key });
}
// lua_table_set_int_key (0x6FAD80) bypasses lua_table_set_value and calls
// lua_table_new_key directly. Same fastcall signature.
var table_set_int_hook: hook.Detour(TableSetFn) = .{};
fn tableSetIntBarrier(L: u32, table: u32, int_key: u32) callconv(hook.cc.fastcall) u32 {
if (luaalloc.isOld(table)) {
addTouched(table);
}
return table_set_int_hook.callOriginal(.{ L, table, int_key });
}
// The "already swept" list: objects that survived sweep, detached from rootgc.
// swept_head -> first swept survivor, swept_tail -> last (for O(1) append).
var swept_head: u32 = 0;
@@ -185,6 +197,101 @@ fn markAllOld(g: u32) void {
}
}
/// Check if an object is in the touched set (linear scan, small set).
fn isTouched(obj: u32) bool {
var i: u32 = 0;
while (i < touched_count) : (i += 1) {
if (touched_set[i] == obj) return true;
}
return false;
}
/// Pre-mark step for minor collections.
///
/// Walks rootgc and sets bit 0 of the marked byte on old TABLES that are
/// NOT in the touched set. This causes lua_gc_mark_table's `TEST 0x11`
/// check at child references to skip these tables -- their subtrees are
/// not re-traversed by the mark phase.
///
/// Only tables (type 5) are pre-marked:
/// - closures (type 6): their upvals' values can change via setupvalue,
/// which isn't write-barriered. Must be walked each cycle.
/// - threads (type 8): stack contents change constantly. Must be walked.
/// - protos (type 9): constants change on proto load only; safe to walk.
/// - userdata (type 7): no traversal in propagate anyway.
///
/// This is safe IF every old-to-young reference in a table goes through
/// our write barriers (tableSetBarrier / tableSetIntBarrier). Any write
/// adds the table to touched_set, which we skip here.
///
/// Returns the count of pre-marked tables (for logging).
fn preMarkOldTables(g: u32) u32 {
const LUA_TTABLE: u8 = 5;
var count: u32 = 0;
var obj = readU32(g + GS_ROOTGC);
while (obj != 0) {
const tt = @as(*const u8, @ptrFromInt(obj + 4)).*;
if (tt == LUA_TTABLE and luaalloc.isOld(obj) and !isTouched(obj)) {
const m_ptr: *u8 = @ptrFromInt(obj + 5);
m_ptr.* = m_ptr.* | 0x01;
count += 1;
}
obj = readU32(obj + OBJ_NEXT);
}
return count;
}
/// Custom minor sweep of rootgc.
///
/// Walks the rootgc linked list. For each object:
/// - old (in age bitmap): skip entirely (don't touch marked byte, don't free)
/// - young, mark bit 0 cleared (unreachable): free via lua_gc_free_object
/// - young, mark bit 0 set (reachable): clear mark bit, promote to old
///
/// This mimics what lua_gc_remove_objects does for young objects but leaves
/// old objects untouched. Returns the number of freed objects.
///
/// Safe to call in one pass (non-chunked) because it's fast: old objects are
/// just a linked-list walk with a bitmap lookup, and young objects should be
/// a small fraction of the heap after the first few cycles.
fn minorSweepRootgc(L: u32, g: u32) u32 {
var freed: u32 = 0;
var prev_next_addr: u32 = g + GS_ROOTGC;
var obj = readU32(prev_next_addr);
while (obj != 0) {
const next = readU32(obj + OBJ_NEXT);
const marked_ptr: *u8 = @ptrFromInt(obj + 5);
const marked = marked_ptr.*;
if (luaalloc.isOld(obj)) {
// Old object: always keep alive (regardless of mark bit),
// but MUST clear mark bit 0 so the next mark phase re-traverses
// it. Without this, the TEST 0x11 check at the parent's reference
// check skips the old object's children, and any young object
// reachable only through the old one gets freed while alive.
marked_ptr.* = marked & 0xFE;
prev_next_addr = obj + OBJ_NEXT;
} else {
if ((marked & 0x01) == 0) {
// Unreachable young: unlink and free
writeU32(prev_next_addr, next);
lua_gc_free_object(L, obj);
freed += 1;
} else {
// Reachable young: clear mark bit 0, promote to old
marked_ptr.* = marked & 0xFE;
luaalloc.setOld(obj);
prev_next_addr = obj + OBJ_NEXT;
}
}
obj = next;
}
return freed;
}
/// Walk the list from a head pointer, find the Nth object.
/// Returns (obj_addr, count_walked). obj_addr=0 if list shorter than N.
fn findNth(head: u32, n: u32) struct { obj: u32, count: u32 } {
@@ -241,6 +348,10 @@ fn collectGarbageDetour(L: u32) callconv(hook.cc.fastcall) void {
selectCycleMode();
cycleStart();
// Pre-marking disabled: causes crashes (see GC_NOTES.md).
// Machinery kept in place (preMarkOldTables, touched set, age bitmap)
// for future reactivation once the missed-barrier path is identified.
// === Atomic: mark + udata sweep + string sweep ===
const t0 = rdtsc();
lua_gc_full_collection(L);
@@ -251,7 +362,18 @@ fn collectGarbageDetour(L: u32) callconv(hook.cc.fastcall) void {
logGc("mark", t1 - t0, 0);
logGc("udata+str", t2 - t1, 0);
// Initialize swept list
// === Minor cycle: custom single-pass rootgc sweep ===
if (!is_major) {
const tm0 = rdtsc();
const freed = minorSweepRootgc(L, g);
logGc("minor-sweep", rdtsc() - tm0, freed);
lua_gc_shrink_memory(L);
luaCallUserDataGC(L);
cycleFinish(g);
return;
}
// === Major cycle: chunked incremental sweep (existing logic) ===
swept_head = 0;
swept_tail = 0;
@@ -345,8 +467,9 @@ var collect_hook: hook.Detour(CollectFn) = .{};
pub fn install() u32 {
var installed: u32 = 0;
if (collect_hook.attach(0x6F7340, &collectGarbageDetour) == .ok) installed += 1;
// Generational write barrier
// Generational write barriers
if (table_set_hook.attach(0x6FA840, &tableSetBarrier) == .ok) installed += 1;
if (table_set_int_hook.attach(0x6FAD80, &tableSetIntBarrier) == .ok) installed += 1;
return installed;
}
@@ -375,6 +498,7 @@ pub fn remove() void {
}
collect_hook.detach();
table_set_hook.detach();
table_set_int_hook.detach();
sweeping = false;
swept_head = 0;
swept_tail = 0;
+3 -5
View File
@@ -326,11 +326,9 @@ pub fn installHooks() void {
// libdeflate inflate replacement
if (inflate_hook.install()) installed += 1;
// Lua slab allocator replacement
installed += luaalloc.install();
// GC phase profiler
installed += luagc.install();
// Lua slab allocator + GC: owned by luagc module.
// installed += luaalloc.install();
// installed += luagc.install();
}
pub fn lateInit() void {