luavm: strip to newlstr-only, add Detour overhead bench
A/B results: hash_lookup prefetch/cache both regressed (original is near-optimal at 59 bytes), VM opcode SSE2 patches were break-even. Only luaS_newlstr hash pre-check shows consistent 40-45% per-call win. Drop hash_lookup hook, VM jump table patches, and resize hook. Keep newlstr Detour with per-call A/B rdtsc and OnWorldUpdate periodic dump. Add GC_WRITE_BARRIER.md documenting the barrier globals at 0xCEEAC0/C4 found in lua_vm_execute (useful for luagc module). bench: add Detour overhead micro-benchmark. Simulates zhook's JMP+trampoline mechanism on mmap'd executable pages. Result: 5 cyc/call overhead -- negligible vs the ~100 cyc/call newlstr savings.
This commit is contained in:
@@ -252,6 +252,8 @@ pub fn main() void {
|
||||
bench_entityUpdate();
|
||||
}
|
||||
|
||||
bench_detour_overhead();
|
||||
|
||||
if (false) { // disabled: not working on these right now
|
||||
|
||||
// 1: vecMulMat4 -- fastcall(ECX=result, EDX=vec, stack=mat) -> u32
|
||||
@@ -3035,3 +3037,173 @@ fn bench_computeOutcodes() void {
|
||||
}
|
||||
report("computeOutcodes(150v)", orig_cyc, sse_cyc, ok);
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Detour overhead benchmark
|
||||
//
|
||||
// Measures the cost of zhook's Detour trampoline mechanism vs a direct call.
|
||||
//
|
||||
// Setup:
|
||||
// target_fn: a small function (luaS_newlstr-sized, ~20 instructions)
|
||||
// allocated on an executable page
|
||||
// trampoline: stolen prologue (5 bytes) + JMP back to target+5
|
||||
// hooked_target: E9 JMP to our detour_fn, which calls trampoline
|
||||
// (simulating callOriginal)
|
||||
//
|
||||
// We measure:
|
||||
// 1. Direct call to the unhooked target_fn
|
||||
// 2. Call to the hooked target_fn (goes through detour -> trampoline -> original)
|
||||
// 3. Overhead = (2) - (1) = pure Detour cost per call
|
||||
// =========================================================================
|
||||
|
||||
fn bench_detour_overhead() void {
|
||||
print("\n{s}\n", .{"=" ** 72});
|
||||
print("Detour overhead benchmark -- {d}M iterations\n", .{ITERS / 1_000_000});
|
||||
print("{s}\n", .{"=" ** 72});
|
||||
|
||||
// Allocate two executable pages: one for "target" function, one for trampoline
|
||||
const target_page = posix.mmap(
|
||||
null, 4096,
|
||||
.{ .READ = true, .WRITE = true, .EXEC = true },
|
||||
.{ .TYPE = .PRIVATE, .ANONYMOUS = true },
|
||||
-1, 0,
|
||||
) catch {
|
||||
print("FATAL: mmap target page failed\n", .{});
|
||||
return;
|
||||
};
|
||||
const tramp_page = posix.mmap(
|
||||
null, 4096,
|
||||
.{ .READ = true, .WRITE = true, .EXEC = true },
|
||||
.{ .TYPE = .PRIVATE, .ANONYMOUS = true },
|
||||
-1, 0,
|
||||
) catch {
|
||||
print("FATAL: mmap trampoline page failed\n", .{});
|
||||
return;
|
||||
};
|
||||
|
||||
// Build a small fastcall target function that does real work:
|
||||
// __fastcall(ECX=a, EDX=b) -> EAX = a ^ (b + (a << 5) + (a >> 2))
|
||||
// This approximates one iteration of the Lua hash loop.
|
||||
//
|
||||
// Machine code (x86, 15 bytes):
|
||||
// 55 push ebp
|
||||
// 89 e5 mov ebp, esp
|
||||
// 89 c8 mov eax, ecx ; eax = a
|
||||
// c1 e0 05 shl eax, 5 ; eax = a << 5
|
||||
// 01 d0 add eax, edx ; eax += b
|
||||
// 89 c8 mov eax, ecx ; -- simplified: just return ECX ^ EDX
|
||||
// 31 d0 xor eax, edx
|
||||
// 5d pop ebp
|
||||
// c3 ret
|
||||
const target_code = [_]u8{
|
||||
0x55, // push ebp
|
||||
0x89, 0xE5, // mov ebp, esp
|
||||
0x89, 0xC8, // mov eax, ecx
|
||||
0xC1, 0xE0, 0x05, // shl eax, 5
|
||||
0x01, 0xD0, // add eax, edx
|
||||
0x89, 0xC8, // mov eax, ecx (use ecx as base)
|
||||
0x31, 0xD0, // xor eax, edx
|
||||
0x5D, // pop ebp
|
||||
0xC3, // ret
|
||||
};
|
||||
@memcpy(target_page[0..target_code.len], &target_code);
|
||||
const target_addr = @intFromPtr(target_page.ptr);
|
||||
|
||||
// Save original first 5 bytes for the trampoline
|
||||
const stolen: usize = 5; // "push ebp; mov ebp, esp" = 3 bytes, but need >= 5 for JMP
|
||||
|
||||
// Actually our prologue is: 55 89 E5 89 C8 = 5 bytes exactly (push ebp, mov ebp,esp, mov eax,ecx)
|
||||
// Build trampoline: stolen bytes + JMP back to target+5
|
||||
const tramp_addr = @intFromPtr(tramp_page.ptr);
|
||||
@memcpy(tramp_page[0..stolen], target_page[0..stolen]);
|
||||
// JMP rel32 back to target + stolen
|
||||
tramp_page[stolen] = 0xE9;
|
||||
const jmp_back_rel = @as(i32, @bitCast((target_addr + stolen) -% (tramp_addr + stolen + 5)));
|
||||
@as(*align(1) i32, @ptrCast(tramp_page[stolen + 1 ..][0..4])).* = jmp_back_rel;
|
||||
|
||||
// Benchmark 1: direct call to unhooked target
|
||||
const DirectFn = *const fn (u32, u32) callconv(.{ .x86_fastcall = .{} }) u32;
|
||||
const direct_fn: DirectFn = @ptrFromInt(target_addr);
|
||||
|
||||
var direct_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
const t = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
const r = @call(.never_inline, direct_fn, .{ 0x12345678, 0xDEADBEEF });
|
||||
std.mem.doNotOptimizeAway(r);
|
||||
}
|
||||
const elapsed = rdtsc() - t;
|
||||
if (elapsed < direct_cyc) direct_cyc = elapsed;
|
||||
}
|
||||
|
||||
// Now hook the target: overwrite first 5 bytes with JMP to our detour
|
||||
// Our detour calls the trampoline (= callOriginal) and returns
|
||||
//
|
||||
// detour_fn: a small function that calls the trampoline with the same args
|
||||
// We build this as machine code too:
|
||||
// push edx ; save EDX (fastcall param2) -- trampoline expects it in EDX
|
||||
// push ecx ; save ECX
|
||||
// call trampoline ; this runs stolen bytes + JMPs back to target+5
|
||||
// ... but wait, trampoline is just the original function body.
|
||||
// The detour should call the trampoline the same way: fastcall(ECX, EDX)
|
||||
//
|
||||
// Actually simpler: the detour IS a fastcall function that just calls trampoline.
|
||||
// Machine code for passthrough detour:
|
||||
// call [trampoline] -- but we need the trampoline addr as a CALL rel32
|
||||
// ret
|
||||
const detour_offset: usize = 256; // put detour at page+256
|
||||
const detour_addr = target_addr + detour_offset; // reuse target_page space
|
||||
// CALL rel32 to trampoline
|
||||
target_page[detour_offset] = 0xE8;
|
||||
const call_rel = @as(i32, @bitCast(tramp_addr -% (detour_addr + 5)));
|
||||
@as(*align(1) i32, @ptrCast(target_page[detour_offset + 1 ..][0..4])).* = call_rel;
|
||||
// RET
|
||||
target_page[detour_offset + 5] = 0xC3;
|
||||
|
||||
// Patch target: E9 JMP rel32 to detour
|
||||
target_page[0] = 0xE9;
|
||||
const jmp_detour_rel = @as(i32, @bitCast(detour_addr -% (target_addr + 5)));
|
||||
@as(*align(1) i32, @ptrCast(target_page[1..5])).* = jmp_detour_rel;
|
||||
|
||||
// Benchmark 2: call hooked target (target -> JMP detour -> CALL trampoline -> stolen+JMP back -> rest of target -> RET -> detour RET)
|
||||
const hooked_fn: DirectFn = @ptrFromInt(target_addr);
|
||||
|
||||
var hooked_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
const t = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
const r = @call(.never_inline, hooked_fn, .{ 0x12345678, 0xDEADBEEF });
|
||||
std.mem.doNotOptimizeAway(r);
|
||||
}
|
||||
const elapsed = rdtsc() - t;
|
||||
if (elapsed < hooked_cyc) hooked_cyc = elapsed;
|
||||
}
|
||||
|
||||
const direct_avg = direct_cyc / ITERS;
|
||||
const hooked_avg = hooked_cyc / ITERS;
|
||||
const overhead = if (hooked_avg > direct_avg) hooked_avg - direct_avg else 0;
|
||||
|
||||
print("\n direct call: {d} cyc/call\n", .{direct_avg});
|
||||
print(" hooked call: {d} cyc/call\n", .{hooked_avg});
|
||||
print(" overhead: {d} cyc/call\n", .{overhead});
|
||||
|
||||
// Also measure trampoline-only (calling trampoline directly, no JMP from target)
|
||||
const tramp_fn: DirectFn = @ptrFromInt(tramp_addr);
|
||||
|
||||
// Restore target bytes so trampoline JMPs into clean code
|
||||
@memcpy(target_page[0..target_code.len], &target_code);
|
||||
|
||||
var tramp_cyc: u64 = std.math.maxInt(u64);
|
||||
for (0..5) |_| {
|
||||
const t = rdtsc();
|
||||
for (0..ITERS) |_| {
|
||||
const r = @call(.never_inline, tramp_fn, .{ 0x12345678, 0xDEADBEEF });
|
||||
std.mem.doNotOptimizeAway(r);
|
||||
}
|
||||
const elapsed = rdtsc() - t;
|
||||
if (elapsed < tramp_cyc) tramp_cyc = elapsed;
|
||||
}
|
||||
|
||||
const tramp_avg = tramp_cyc / ITERS;
|
||||
print(" trampoline: {d} cyc/call (callOriginal path, no detour JMP)\n", .{tramp_avg});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
# Lua 5.0 GC Write Barrier in lua_vm_execute
|
||||
|
||||
## Location
|
||||
|
||||
`lua_vm_execute` (0x6F8720), the main VM interpreter loop.
|
||||
The barrier appears after every TValue copy (~20 sites in the function).
|
||||
|
||||
## Globals
|
||||
|
||||
- `0xCEEAC0` -- GC barrier value (current white marker / gc object pointer)
|
||||
- `0xCEEAC4` -- GC barrier flag (non-zero = barrier is active)
|
||||
|
||||
## Pattern (from disassembly)
|
||||
|
||||
After copying a 16-byte TValue (type_tag, gc_ptr, value_lo, value_hi):
|
||||
|
||||
```asm
|
||||
mov eax, [dst + 4] ; eax = copied gc_ptr field
|
||||
test eax, eax
|
||||
jz skip ; NULL gc_ptr -> no barrier needed
|
||||
cmp dword ptr [0xCEEAC4], 0
|
||||
jz skip ; barrier disabled -> skip
|
||||
mov [0xCEEAC0], eax ; mark: write gc_ptr into barrier global
|
||||
skip:
|
||||
```
|
||||
|
||||
## What it does
|
||||
|
||||
The gc_ptr field at TValue+0x04 holds a pointer to a GC-managed object
|
||||
(string, table, closure, userdata) or NULL for non-collectable types
|
||||
(number, boolean, nil, lightuserdata).
|
||||
|
||||
When a TValue is copied (MOVE, GETGLOBAL, GETTABLE, LOADK, etc.), the
|
||||
barrier checks:
|
||||
1. Is the copied value a GC object? (gc_ptr != NULL)
|
||||
2. Is the write barrier active? (flag at 0xCEEAC4 != 0)
|
||||
3. If both true, write the gc_ptr to the barrier global at 0xCEEAC0
|
||||
|
||||
This is Lua 5.0's incremental GC write barrier. It tracks which GC objects
|
||||
have been moved/copied so the collector knows which objects are reachable
|
||||
from newly-written locations. The barrier global accumulates the "last written"
|
||||
gc object -- the actual GC uses this to avoid rescanning the full root set.
|
||||
|
||||
## Opcodes that trigger the barrier
|
||||
|
||||
Every opcode that writes a TValue to a register or table slot:
|
||||
- MOVE (op 0)
|
||||
- LOADK (op 1) -- loads constant, has gc_ptr for string constants
|
||||
- GETUPVAL (op 4)
|
||||
- GETGLOBAL (op 5)
|
||||
- GETTABLE (op 6)
|
||||
- SETGLOBAL (op 7) -- barrier on the table side
|
||||
- SETTABLE (op 9)
|
||||
- NEWTABLE (op 10)
|
||||
- SELF (op 11)
|
||||
- CONCAT (op 21)
|
||||
- CLOSURE (op 30)
|
||||
- FORLOOP (op 23) -- number only, but barrier still present
|
||||
|
||||
## Relevance to luagc module
|
||||
|
||||
Our luagc module (in weirdperformance) hooks `lua_gc_step` (0x6FAE00).
|
||||
Understanding the barrier globals is useful for:
|
||||
- Knowing when/how often the barrier fires during heavy addon activity
|
||||
- Potentially batching barrier writes if we ever replace the GC step
|
||||
- The flag at 0xCEEAC4 could be used to temporarily disable the barrier
|
||||
during bulk operations (dangerous -- must re-enable before GC runs)
|
||||
+80
-492
@@ -1,12 +1,7 @@
|
||||
//! luavm -- Lua VM hotspot optimizations.
|
||||
//!
|
||||
//! Targets the top 3 Lua CPU consumers from perf profiling (4.5% combined):
|
||||
//!
|
||||
//! 1. lua_table_get_hash_element (0x6FA760, 1.65%) -- hash chain walk with prefetch
|
||||
//! 2. lua_vm_execute (0x6F8720, 1.50%) -- jump table patches for hot opcodes
|
||||
//! 3. luaS_newlstr (0x6F9D00, 1.35%) -- hash pre-check before memcmp
|
||||
//!
|
||||
//! All three functions verified from disassembly (calling conventions, layouts).
|
||||
//! luaS_newlstr (0x6F9D00, 1.35% CPU): hash pre-check before memcmp.
|
||||
//! A/B: alternating calls, per-call rdtsc, periodic dump via OnWorldUpdate.
|
||||
|
||||
const hook = @import("zhook");
|
||||
const logging = @import("../logging.zig");
|
||||
@@ -22,117 +17,72 @@ pub fn isActive() bool {
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// 1. lua_table_get_hash_element (0x6FA760)
|
||||
// __fastcall(ECX=Table*, EDX=TString*) -> TValue*
|
||||
// 59 bytes. Pure hash chain walk through 40-byte nodes.
|
||||
//
|
||||
// Optimization: software prefetch of next node while comparing current.
|
||||
// On chains of length >= 2, hides ~100 cycle cache miss latency behind
|
||||
// the type+key comparison (~5 cycles). Average WoW addon table chains
|
||||
// are 1-3 deep; prefetch helps the 2-3 deep cases.
|
||||
//
|
||||
// Node layout (40 bytes / 0x28):
|
||||
// +0x00: key type tag (4 = LUA_TSTRING)
|
||||
// +0x08: key value (TString pointer -- interned, so pointer equality)
|
||||
// +0x10: value (TValue, 16 bytes)
|
||||
// +0x20: next chain pointer
|
||||
//
|
||||
// Table layout:
|
||||
// +0x07: lsizenode (u8, log2 of hash part size)
|
||||
// +0x10: node array base pointer
|
||||
// A/B instrumentation
|
||||
// =============================================================================
|
||||
|
||||
const LUA_TSTRING = 4;
|
||||
const NIL_OBJECT: u32 = 0x811bc0; // luaO_nilobject sentinel
|
||||
|
||||
const HashLookupFn = fn (u32, u32) callconv(hook.cc.fastcall) u32;
|
||||
var hash_lookup_hook: hook.Detour(HashLookupFn) = .{};
|
||||
|
||||
fn hashLookupDetour(table: u32, key: u32) callconv(hook.cc.fastcall) u32 {
|
||||
// Preserve callee-saved regs the compiler might not know about
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
|
||||
const lsizenode: u5 = @truncate(hook.readMem(u8, table + 7));
|
||||
const node_base: u32 = hook.readMem(u32, table + 0x10);
|
||||
const key_hash: u32 = hook.readMem(u32, key + 0x08);
|
||||
|
||||
const mask: u32 = (@as(u32, 1) << lsizenode) - 1;
|
||||
const bucket: u32 = key_hash & mask;
|
||||
var node: u32 = node_base + bucket * 40;
|
||||
|
||||
while (true) {
|
||||
// Read next pointer early -- prefetch its cache line while we compare
|
||||
const next: u32 = hook.readMem(u32, node + 0x20);
|
||||
if (next != 0) {
|
||||
asm volatile ("prefetcht0 (%[addr])"
|
||||
:
|
||||
: [addr] "r" (next),
|
||||
);
|
||||
}
|
||||
|
||||
// Check: is this node a string key matching our TString?
|
||||
const key_type: u32 = hook.readMem(u32, node);
|
||||
if (key_type == LUA_TSTRING) {
|
||||
const key_value: u32 = hook.readMem(u32, node + 0x08);
|
||||
if (key_value == key) {
|
||||
return node + 0x10; // &node->i_val
|
||||
}
|
||||
}
|
||||
|
||||
if (next == 0) return NIL_OBJECT;
|
||||
node = next;
|
||||
}
|
||||
inline fn rdtsc() u64 {
|
||||
var lo: u32 = undefined;
|
||||
var hi: u32 = undefined;
|
||||
asm volatile ("rdtsc"
|
||||
: [lo] "={eax}" (lo),
|
||||
[hi] "={edx}" (hi),
|
||||
);
|
||||
return @as(u64, hi) << 32 | lo;
|
||||
}
|
||||
|
||||
const AB_DUMP_INTERVAL: u64 = 10000;
|
||||
|
||||
const ABStats = struct {
|
||||
cycles: u64 = 0,
|
||||
calls: u64 = 0,
|
||||
};
|
||||
|
||||
var custom_ab: ABStats = .{};
|
||||
var baseline_ab: ABStats = .{};
|
||||
|
||||
// =============================================================================
|
||||
// 2. luaS_newlstr (0x6F9D00)
|
||||
// __fastcall(ECX=lua_State*, EDX=str_ptr, stack=len) -> TString*
|
||||
// RET 0x4 (cleans 1 stack param)
|
||||
//
|
||||
// Lua 5.0 string interning: hash the string, then walk the intern table
|
||||
// chain comparing length + bytes.
|
||||
//
|
||||
// Optimization: add hash pre-check before REPE CMPSB. The original code
|
||||
// checks ts->len == len, then immediately does memcmp. But many strings
|
||||
// share common lengths (4, 6, 8...) causing false-positive memcmps.
|
||||
// Adding ts->hash == our_hash eliminates these -- a 1-cycle comparison
|
||||
// that skips an O(len) memcmp.
|
||||
//
|
||||
// This is the Lua 5.0 -> 5.1 optimization that Blizzard never got.
|
||||
//
|
||||
// TString layout:
|
||||
// +0x00: next pointer (intern chain)
|
||||
// +0x04: tt (type tag)
|
||||
// +0x05: marked (GC mark)
|
||||
// +0x06: reserved
|
||||
// +0x08: hash (u32)
|
||||
// +0x0C: len (u32)
|
||||
// +0x10: inline char data
|
||||
//
|
||||
// String table (global_State+0x10 -> strt):
|
||||
// +0x04: hash bucket array (TString**)
|
||||
// +0x0C: size (number of buckets)
|
||||
// luaS_newlstr (0x6F9D00)
|
||||
// __fastcall(ECX=lua_State*, EDX=str_ptr, stack=len) -> TString*
|
||||
// RET 0x4
|
||||
// =============================================================================
|
||||
|
||||
const NewLStrFn = fn (u32, u32, u32) callconv(hook.cc.fastcall) u32;
|
||||
var newlstr_hook: hook.Detour(NewLStrFn) = .{};
|
||||
|
||||
// lua_create_string_object at 0x6F9D90:
|
||||
// __fastcall(ECX=global_State*, EDX=str_ptr, stack: len, hash)
|
||||
// Allocates and interns a new TString. We call this on cache miss.
|
||||
fn luaCreateStringObject(state: u32, str_ptr: u32, len: u32, hash: u32) u32 {
|
||||
var call_ctr: u32 = 0;
|
||||
|
||||
fn luaCreateStringObject(state: u32, str_ptr: u32, len: u32, hash_val: u32) u32 {
|
||||
return hook.call(
|
||||
fn (u32, u32, u32, u32) callconv(hook.cc.fastcall) u32,
|
||||
0x6F9D90,
|
||||
.{ state, str_ptr, len, hash },
|
||||
.{ state, str_ptr, len, hash_val },
|
||||
);
|
||||
}
|
||||
|
||||
fn newlstrDetour(state: u32, str_ptr: u32, len: u32) callconv(hook.cc.fastcall) u32 {
|
||||
asm volatile ("" ::: .{ .esi = true, .edi = true, .ebx = true });
|
||||
|
||||
// Phase 1: Hash computation (identical to original Lua 5.0 algorithm)
|
||||
// Samples every (len>>5)+1 bytes, seed = len
|
||||
const use_custom = (call_ctr & 1) == 0;
|
||||
call_ctr +%= 1;
|
||||
const t0 = rdtsc();
|
||||
|
||||
const result = if (use_custom)
|
||||
newlstrImpl(state, str_ptr, len)
|
||||
else
|
||||
newlstr_hook.callOriginal(.{ state, str_ptr, len });
|
||||
|
||||
const elapsed = rdtsc() - t0;
|
||||
if (use_custom) {
|
||||
custom_ab.cycles +|= elapsed;
|
||||
custom_ab.calls +|= 1;
|
||||
} else {
|
||||
baseline_ab.cycles +|= elapsed;
|
||||
baseline_ab.calls +|= 1;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
fn newlstrImpl(state: u32, str_ptr: u32, len: u32) u32 {
|
||||
const str: [*]const u8 = @ptrFromInt(str_ptr);
|
||||
var h: u32 = len;
|
||||
const step: u32 = (len >> 5) + 1;
|
||||
@@ -143,11 +93,9 @@ fn newlstrDetour(state: u32, str_ptr: u32, len: u32) callconv(hook.cc.fastcall)
|
||||
l1 -= step;
|
||||
}
|
||||
|
||||
// Phase 2: Intern table lookup with hash pre-check
|
||||
// global_State is at state+0x10 (lua_State -> l_G)
|
||||
const global_state: u32 = hook.readMem(u32, state + 0x10);
|
||||
const strt_hash: u32 = hook.readMem(u32, global_state + 0x04); // bucket array
|
||||
const strt_size: u32 = hook.readMem(u32, global_state + 0x0C); // bucket count
|
||||
const strt_hash: u32 = hook.readMem(u32, global_state + 0x04);
|
||||
const strt_size: u32 = hook.readMem(u32, global_state + 0x0C);
|
||||
|
||||
const bucket: u32 = h & (strt_size - 1);
|
||||
var ts: u32 = hook.readMem(u32, strt_hash + bucket * 4);
|
||||
@@ -155,35 +103,29 @@ fn newlstrDetour(state: u32, str_ptr: u32, len: u32) callconv(hook.cc.fastcall)
|
||||
while (ts != 0) {
|
||||
const ts_len: u32 = hook.readMem(u32, ts + 0x0C);
|
||||
if (ts_len == len) {
|
||||
// ** THE OPTIMIZATION: check hash before memcmp **
|
||||
const ts_hash: u32 = hook.readMem(u32, ts + 0x08);
|
||||
if (ts_hash == h) {
|
||||
// Length and hash match -- now compare actual bytes
|
||||
if (len == 0 or strEqual(str_ptr, ts + 0x10, len)) {
|
||||
return ts;
|
||||
}
|
||||
}
|
||||
}
|
||||
ts = hook.readMem(u32, ts); // ts = ts->next
|
||||
ts = hook.readMem(u32, ts);
|
||||
}
|
||||
|
||||
// Not found -- allocate and intern (ECX = lua_State*, not global_State*)
|
||||
return luaCreateStringObject(state, str_ptr, len, h);
|
||||
}
|
||||
|
||||
/// Fast string comparison. Uses dword-at-a-time for the bulk, then byte tail.
|
||||
fn strEqual(a_ptr: u32, b_ptr: u32, len: u32) bool {
|
||||
const a: [*]const u8 = @ptrFromInt(a_ptr);
|
||||
const b: [*]const u8 = @ptrFromInt(b_ptr);
|
||||
|
||||
// Compare 4 bytes at a time
|
||||
var i: u32 = 0;
|
||||
while (i + 4 <= len) : (i += 4) {
|
||||
const va = @as(*align(1) const u32, @ptrCast(a + i)).*;
|
||||
const vb = @as(*align(1) const u32, @ptrCast(b + i)).*;
|
||||
if (va != vb) return false;
|
||||
}
|
||||
// Remaining bytes
|
||||
while (i < len) : (i += 1) {
|
||||
if (a[i] != b[i]) return false;
|
||||
}
|
||||
@@ -191,381 +133,36 @@ fn strEqual(a_ptr: u32, b_ptr: u32, len: u32) bool {
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
// 3. lua_vm_execute (0x6F8720)
|
||||
// __fastcall(ECX=lua_State*) -> TValue* (or NULL)
|
||||
// 4432 bytes, 35 opcodes via jump table at 0x6F9870.
|
||||
//
|
||||
// Strategy: patch individual jump table entries to point at our optimized
|
||||
// opcode handlers. This avoids rewriting the entire 4.4KB function while
|
||||
// targeting the hot paths.
|
||||
//
|
||||
// At jump table dispatch, the register state is:
|
||||
// ESI = instruction word (full 32-bit encoded instruction)
|
||||
// EDI = R(A) pointer (base + A*16)
|
||||
// EBX = A*16 (A field scaled to TValue offset)
|
||||
// [EBP-0x04] = lua_State*
|
||||
// [EBP-0x08] = pc (already incremented past this instruction)
|
||||
// [EBP-0x0C] = base (L->base)
|
||||
// [EBP-0x10] = constants array
|
||||
// [EBP-0x14] = closure
|
||||
//
|
||||
// Dispatch loop re-entry: 0x6F8770 (fetch next instruction from [EBP-0x08])
|
||||
//
|
||||
// Optimization targets:
|
||||
//
|
||||
// a) Arithmetic (ADD=0xC, SUB=0xD, MUL=0xE, DIV=0xF):
|
||||
// Original uses x87: FLD [src+8] / FADD [src+8] / FSTP [dst+8]
|
||||
// Replace with SSE2: MOVSD xmm0,[src+8] / ADDSD xmm0,[src+8] / MOVSD [dst+8],xmm0
|
||||
// SSE2 has lower latency and better pipelining on modern CPUs.
|
||||
//
|
||||
// b) TValue copies (~20 sites):
|
||||
// Original: 4x MOV (load+store each dword, 8 instructions = 32 bytes)
|
||||
// Replace with: MOVDQU xmm0,[src] / MOVDQU [dst],xmm0 (2 instructions)
|
||||
// Saves ~12 bytes per site and reduces frontend pressure.
|
||||
//
|
||||
// c) GETGLOBAL (opcode 5):
|
||||
// Inlines the hash lookup (currently calls 0x6FA760) to avoid call
|
||||
// overhead. The hash lookup is only ~15 instructions; inlining saves
|
||||
// the CALL/RET cycle and enables the compiler to keep values in regs.
|
||||
//
|
||||
// Implementation: each patched opcode handler is a naked fn that matches
|
||||
// the register convention above, does its work, and JMPs to 0x6F8770.
|
||||
// OnWorldUpdate (0x482EA0) -- periodic stats dump
|
||||
// =============================================================================
|
||||
|
||||
const VM_DISPATCH_LOOP: u32 = 0x6F8770;
|
||||
const VM_JUMP_TABLE: u32 = 0x6F9870;
|
||||
const WorldUpdateFn = fn (u32) callconv(hook.cc.fastcall) void;
|
||||
var world_update_hook: hook.Detour(WorldUpdateFn) = .{};
|
||||
|
||||
// Opcode indices for the jump table
|
||||
const OP_MOVE: u32 = 0;
|
||||
const OP_ADD: u32 = 0xC;
|
||||
const OP_SUB: u32 = 0xD;
|
||||
const OP_MUL: u32 = 0xE;
|
||||
const OP_DIV: u32 = 0xF;
|
||||
var frame_count: u64 = 0;
|
||||
|
||||
// Store original jump table entries so we can restore them
|
||||
var original_jt_entries: [35]u32 = undefined;
|
||||
var jt_patched: bool = false;
|
||||
fn worldUpdateDetour(fc: u32) callconv(hook.cc.fastcall) void {
|
||||
frame_count +%= 1;
|
||||
|
||||
fn patchJumpTableEntry(opcode: u32, handler: u32) void {
|
||||
const entry_addr = VM_JUMP_TABLE + opcode * 4;
|
||||
// Save original
|
||||
original_jt_entries[opcode] = hook.readMem(u32, entry_addr);
|
||||
// Write our handler address
|
||||
var addr_bytes: [4]u8 = undefined;
|
||||
@as(*align(1) u32, @ptrCast(&addr_bytes)).* = handler;
|
||||
hook.writeProtected(entry_addr, &addr_bytes);
|
||||
}
|
||||
|
||||
fn restoreJumpTableEntry(opcode: u32) void {
|
||||
const entry_addr = VM_JUMP_TABLE + opcode * 4;
|
||||
var addr_bytes: [4]u8 = undefined;
|
||||
@as(*align(1) u32, @ptrCast(&addr_bytes)).* = original_jt_entries[opcode];
|
||||
hook.writeProtected(entry_addr, &addr_bytes);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// MOVE handler (opcode 0) -- SSE TValue copy
|
||||
//
|
||||
// Original (at 0x6F87CD):
|
||||
// Decodes B from ESI bits[15:23], computes R(B) = base + B*16,
|
||||
// does 4x MOV to copy 16 bytes from R(B) to R(A), then GC barrier.
|
||||
//
|
||||
// Our version: MOVDQU copy (2 instructions instead of 8), same GC barrier.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn vmMoveHandler() callconv(.naked) void {
|
||||
// AT&T syntax: src, dst. Registers prefixed with %%.
|
||||
asm volatile (
|
||||
// Decode B field: bits [15:23] of ESI
|
||||
\\ movl %%esi, %%ecx
|
||||
\\ shrl $15, %%ecx
|
||||
\\ andl $0x1ff, %%ecx
|
||||
\\ shll $4, %%ecx
|
||||
\\ addl -0x0c(%%ebp), %%ecx
|
||||
// ecx = R(B). SSE 16-byte copy: R(A) = R(B)
|
||||
\\ movdqu (%%ecx), %%xmm0
|
||||
\\ movdqu %%xmm0, (%%edi)
|
||||
// GC write barrier check
|
||||
\\ movl 4(%%edi), %%eax
|
||||
\\ testl %%eax, %%eax
|
||||
\\ jz 1f
|
||||
\\ cmpl $0, 0xCEEAC4
|
||||
\\ jz 1f
|
||||
\\ movl %%eax, 0xCEEAC0
|
||||
\\1:
|
||||
// Re-enter dispatch loop (EAX = pc, EDX = scratch for jump target)
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
:
|
||||
: [dispatch] "i" (VM_DISPATCH_LOOP),
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Arithmetic handlers -- SSE2 ADDSD/SUBSD/MULSD/DIVSD
|
||||
//
|
||||
// Each handler: decode B & C with RK check, type-check for LUA_TNUMBER,
|
||||
// fast path with SSE2, slow path calls lua_arithmetic_operation (0x6F9A80).
|
||||
//
|
||||
// Fully inlined per-handler to avoid multiline string concat issues.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn vmAddHandler() callconv(.naked) void {
|
||||
// Register layout at entry: ESI=instruction, EDI=R(A), [EBP-0x04]=L,
|
||||
// [EBP-0x08]=pc, [EBP-0x0C]=base, [EBP-0x10]=constants
|
||||
//
|
||||
// Decode: ECX=RK(C) ptr, EDX=RK(B) ptr. EDI=R(A) preserved.
|
||||
// Fast path: SSE2 addsd. Slow path: __fastcall(ECX=L, EDX=R(A),
|
||||
// stack: opB, opC, TM_ADD=5), callee cleans stack (RET 0xC).
|
||||
asm volatile (
|
||||
\\ movl %%esi, %%ecx
|
||||
\\ shrl $6, %%ecx
|
||||
\\ andl $0x1ff, %%ecx
|
||||
\\ cmpl $0xfa, %%ecx
|
||||
\\ jl 2f
|
||||
\\ subl $250, %%ecx
|
||||
\\ shll $4, %%ecx
|
||||
\\ addl -0x10(%%ebp), %%ecx
|
||||
\\ jmp 3f
|
||||
\\2: shll $4, %%ecx
|
||||
\\ addl -0x0c(%%ebp), %%ecx
|
||||
\\3: movl %%esi, %%edx
|
||||
\\ shrl $15, %%edx
|
||||
\\ andl $0x1ff, %%edx
|
||||
\\ cmpl $0xfa, %%edx
|
||||
\\ jl 4f
|
||||
\\ subl $250, %%edx
|
||||
\\ shll $4, %%edx
|
||||
\\ addl -0x10(%%ebp), %%edx
|
||||
\\ jmp 5f
|
||||
\\4: shll $4, %%edx
|
||||
\\ addl -0x0c(%%ebp), %%edx
|
||||
\\5: cmpl $3, (%%edx)
|
||||
\\ jne 6f
|
||||
\\ cmpl $3, (%%ecx)
|
||||
\\ jne 6f
|
||||
\\ movsd 8(%%edx), %%xmm0
|
||||
\\ addsd 8(%%ecx), %%xmm0
|
||||
\\ movsd %%xmm0, 8(%%edi)
|
||||
\\ movl $3, (%%edi)
|
||||
\\ movl 0xCEEAC0, %%eax
|
||||
\\ movl %%eax, 4(%%edi)
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
\\6:
|
||||
// Slow path: __fastcall(ECX=L, EDX=R(A), stack: opB, opC, tm_id)
|
||||
// Callee cleans stack (RET 0xC) -- no ESP adjustment after call.
|
||||
\\ pushl $5
|
||||
\\ pushl %%ecx
|
||||
\\ pushl %%edx
|
||||
\\ movl -0x04(%%ebp), %%ecx
|
||||
\\ movl %%edi, %%edx
|
||||
\\ movl %[arith_op], %%eax
|
||||
\\ call *%%eax
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
:
|
||||
: [dispatch] "i" (VM_DISPATCH_LOOP),
|
||||
[arith_op] "i" (@as(u32, 0x6F9A80)),
|
||||
);
|
||||
}
|
||||
|
||||
fn vmSubHandler() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\ movl %%esi, %%ecx
|
||||
\\ shrl $6, %%ecx
|
||||
\\ andl $0x1ff, %%ecx
|
||||
\\ cmpl $0xfa, %%ecx
|
||||
\\ jl 2f
|
||||
\\ subl $250, %%ecx
|
||||
\\ shll $4, %%ecx
|
||||
\\ addl -0x10(%%ebp), %%ecx
|
||||
\\ jmp 3f
|
||||
\\2: shll $4, %%ecx
|
||||
\\ addl -0x0c(%%ebp), %%ecx
|
||||
\\3: movl %%esi, %%edx
|
||||
\\ shrl $15, %%edx
|
||||
\\ andl $0x1ff, %%edx
|
||||
\\ cmpl $0xfa, %%edx
|
||||
\\ jl 4f
|
||||
\\ subl $250, %%edx
|
||||
\\ shll $4, %%edx
|
||||
\\ addl -0x10(%%ebp), %%edx
|
||||
\\ jmp 5f
|
||||
\\4: shll $4, %%edx
|
||||
\\ addl -0x0c(%%ebp), %%edx
|
||||
\\5: cmpl $3, (%%edx)
|
||||
\\ jne 6f
|
||||
\\ cmpl $3, (%%ecx)
|
||||
\\ jne 6f
|
||||
\\ movsd 8(%%edx), %%xmm0
|
||||
\\ subsd 8(%%ecx), %%xmm0
|
||||
\\ movsd %%xmm0, 8(%%edi)
|
||||
\\ movl $3, (%%edi)
|
||||
\\ movl 0xCEEAC0, %%eax
|
||||
\\ movl %%eax, 4(%%edi)
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
\\6:
|
||||
\\ pushl $6
|
||||
\\ pushl %%ecx
|
||||
\\ pushl %%edx
|
||||
\\ movl -0x04(%%ebp), %%ecx
|
||||
\\ movl %%edi, %%edx
|
||||
\\ movl %[arith_op], %%eax
|
||||
\\ call *%%eax
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
:
|
||||
: [dispatch] "i" (VM_DISPATCH_LOOP),
|
||||
[arith_op] "i" (@as(u32, 0x6F9A80)),
|
||||
);
|
||||
}
|
||||
|
||||
fn vmMulHandler() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\ movl %%esi, %%ecx
|
||||
\\ shrl $6, %%ecx
|
||||
\\ andl $0x1ff, %%ecx
|
||||
\\ cmpl $0xfa, %%ecx
|
||||
\\ jl 2f
|
||||
\\ subl $250, %%ecx
|
||||
\\ shll $4, %%ecx
|
||||
\\ addl -0x10(%%ebp), %%ecx
|
||||
\\ jmp 3f
|
||||
\\2: shll $4, %%ecx
|
||||
\\ addl -0x0c(%%ebp), %%ecx
|
||||
\\3: movl %%esi, %%edx
|
||||
\\ shrl $15, %%edx
|
||||
\\ andl $0x1ff, %%edx
|
||||
\\ cmpl $0xfa, %%edx
|
||||
\\ jl 4f
|
||||
\\ subl $250, %%edx
|
||||
\\ shll $4, %%edx
|
||||
\\ addl -0x10(%%ebp), %%edx
|
||||
\\ jmp 5f
|
||||
\\4: shll $4, %%edx
|
||||
\\ addl -0x0c(%%ebp), %%edx
|
||||
\\5: cmpl $3, (%%edx)
|
||||
\\ jne 6f
|
||||
\\ cmpl $3, (%%ecx)
|
||||
\\ jne 6f
|
||||
\\ movsd 8(%%edx), %%xmm0
|
||||
\\ mulsd 8(%%ecx), %%xmm0
|
||||
\\ movsd %%xmm0, 8(%%edi)
|
||||
\\ movl $3, (%%edi)
|
||||
\\ movl 0xCEEAC0, %%eax
|
||||
\\ movl %%eax, 4(%%edi)
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
\\6:
|
||||
\\ pushl $7
|
||||
\\ pushl %%ecx
|
||||
\\ pushl %%edx
|
||||
\\ movl -0x04(%%ebp), %%ecx
|
||||
\\ movl %%edi, %%edx
|
||||
\\ movl %[arith_op], %%eax
|
||||
\\ call *%%eax
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
:
|
||||
: [dispatch] "i" (VM_DISPATCH_LOOP),
|
||||
[arith_op] "i" (@as(u32, 0x6F9A80)),
|
||||
);
|
||||
}
|
||||
|
||||
fn vmDivHandler() callconv(.naked) void {
|
||||
asm volatile (
|
||||
\\ movl %%esi, %%ecx
|
||||
\\ shrl $6, %%ecx
|
||||
\\ andl $0x1ff, %%ecx
|
||||
\\ cmpl $0xfa, %%ecx
|
||||
\\ jl 2f
|
||||
\\ subl $250, %%ecx
|
||||
\\ shll $4, %%ecx
|
||||
\\ addl -0x10(%%ebp), %%ecx
|
||||
\\ jmp 3f
|
||||
\\2: shll $4, %%ecx
|
||||
\\ addl -0x0c(%%ebp), %%ecx
|
||||
\\3: movl %%esi, %%edx
|
||||
\\ shrl $15, %%edx
|
||||
\\ andl $0x1ff, %%edx
|
||||
\\ cmpl $0xfa, %%edx
|
||||
\\ jl 4f
|
||||
\\ subl $250, %%edx
|
||||
\\ shll $4, %%edx
|
||||
\\ addl -0x10(%%ebp), %%edx
|
||||
\\ jmp 5f
|
||||
\\4: shll $4, %%edx
|
||||
\\ addl -0x0c(%%ebp), %%edx
|
||||
\\5: cmpl $3, (%%edx)
|
||||
\\ jne 6f
|
||||
\\ cmpl $3, (%%ecx)
|
||||
\\ jne 6f
|
||||
\\ movsd 8(%%edx), %%xmm0
|
||||
\\ divsd 8(%%ecx), %%xmm0
|
||||
\\ movsd %%xmm0, 8(%%edi)
|
||||
\\ movl $3, (%%edi)
|
||||
\\ movl 0xCEEAC0, %%eax
|
||||
\\ movl %%eax, 4(%%edi)
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
\\6:
|
||||
\\ pushl $8
|
||||
\\ pushl %%ecx
|
||||
\\ pushl %%edx
|
||||
\\ movl -0x04(%%ebp), %%ecx
|
||||
\\ movl %%edi, %%edx
|
||||
\\ movl %[arith_op], %%eax
|
||||
\\ call *%%eax
|
||||
\\ movl -0x08(%%ebp), %%eax
|
||||
\\ movl %[dispatch], %%edx
|
||||
\\ jmp *%%edx
|
||||
:
|
||||
: [dispatch] "i" (VM_DISPATCH_LOOP),
|
||||
[arith_op] "i" (@as(u32, 0x6F9A80)),
|
||||
);
|
||||
}
|
||||
|
||||
fn installVmPatches() u32 {
|
||||
// Save ALL original entries first
|
||||
for (0..35) |i| {
|
||||
original_jt_entries[i] = hook.readMem(u32, VM_JUMP_TABLE + @as(u32, @intCast(i)) * 4);
|
||||
if (frame_count % AB_DUMP_INTERVAL == 0 and frame_count > 0) {
|
||||
dumpStats();
|
||||
}
|
||||
|
||||
var count: u32 = 0;
|
||||
|
||||
// MOVE -- SSE TValue copy
|
||||
patchJumpTableEntry(OP_MOVE, @intFromPtr(&vmMoveHandler));
|
||||
count += 1;
|
||||
|
||||
// Arithmetic -- SSE2 double ops
|
||||
patchJumpTableEntry(OP_ADD, @intFromPtr(&vmAddHandler));
|
||||
patchJumpTableEntry(OP_SUB, @intFromPtr(&vmSubHandler));
|
||||
patchJumpTableEntry(OP_MUL, @intFromPtr(&vmMulHandler));
|
||||
patchJumpTableEntry(OP_DIV, @intFromPtr(&vmDivHandler));
|
||||
count += 4;
|
||||
|
||||
jt_patched = true;
|
||||
return count;
|
||||
world_update_hook.callOriginal(.{fc});
|
||||
}
|
||||
|
||||
fn removeVmPatches() void {
|
||||
if (!jt_patched) return;
|
||||
restoreJumpTableEntry(OP_MOVE);
|
||||
restoreJumpTableEntry(OP_ADD);
|
||||
restoreJumpTableEntry(OP_SUB);
|
||||
restoreJumpTableEntry(OP_MUL);
|
||||
restoreJumpTableEntry(OP_DIV);
|
||||
jt_patched = false;
|
||||
fn dumpStats() void {
|
||||
const ca = custom_ab.calls;
|
||||
const ba = baseline_ab.calls;
|
||||
const c_avg: u64 = if (ca > 0) custom_ab.cycles / ca else 0;
|
||||
const b_avg: u64 = if (ba > 0) baseline_ab.cycles / ba else 0;
|
||||
|
||||
log.fmt("[luavm] {d}f newlstr: c={d} b={d} cyc/call ({d}k calls)\n", .{
|
||||
frame_count, c_avg, b_avg, ca / 1000,
|
||||
});
|
||||
|
||||
custom_ab = .{};
|
||||
baseline_ab = .{};
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
@@ -578,32 +175,23 @@ pub fn installHooks() void {
|
||||
if (!g_is_hook_owner) return;
|
||||
|
||||
log = logging.Logger.open(module_name, .both);
|
||||
var installed: u32 = 0;
|
||||
|
||||
// Hook 1: hash table lookup with prefetch
|
||||
if (hash_lookup_hook.attach(0x6FA760, &hashLookupDetour) == .ok) {
|
||||
installed += 1;
|
||||
log.print(" lua_table_get_hash_element: prefetch chain walk\n");
|
||||
}
|
||||
|
||||
// Hook 2: string interning with hash pre-check
|
||||
if (newlstr_hook.attach(0x6F9D00, &newlstrDetour) == .ok) {
|
||||
installed += 1;
|
||||
log.print(" luaS_newlstr: hash pre-check + dword compare\n");
|
||||
log.print(" newlstr: A/B hash pre-check vs original\n");
|
||||
}
|
||||
|
||||
// Hook 3: VM opcode patches (jump table)
|
||||
installed += installVmPatches();
|
||||
log.print(" lua_vm_execute: 5 opcode patches (SSE2 arith + MOVDQU copy)\n");
|
||||
if (world_update_hook.attach(0x482EA0, &worldUpdateDetour) == .ok) {
|
||||
log.print(" OnWorldUpdate: periodic stats dump\n");
|
||||
}
|
||||
|
||||
log.print("luavm: hooks installed\n");
|
||||
log.print("luavm: active\n");
|
||||
}
|
||||
|
||||
pub fn removeHooks() void {
|
||||
if (g_is_hook_owner) {
|
||||
removeVmPatches();
|
||||
dumpStats();
|
||||
world_update_hook.detach();
|
||||
newlstr_hook.detach();
|
||||
hash_lookup_hook.detach();
|
||||
log.close();
|
||||
}
|
||||
g_is_hook_owner = false;
|
||||
|
||||
Reference in New Issue
Block a user