bench: add x86 Linux micro-benchmark harness for math_sse functions
Extracts original x87 FPU bytes from WoW.exe via Ghidra, mmaps them executable, and benchmarks against our SSE replacements. Covers all 17 UnitXP polyfill functions with correctness validation and cycle counts. Maps a page at 0x7ff000 for the float 1.0 constant referenced by rotMat3x3/rotMat4x4/planeNormal via absolute address 0x7ff9d8. Build: zig build bench / zig build run-bench
This commit is contained in:
@@ -97,6 +97,41 @@ pub fn build(b: *std.Build) void {
|
||||
lib.root_module.addObject(math_sse_obj);
|
||||
b.installArtifact(lib);
|
||||
|
||||
// Benchmark harness — native x86 Linux executable for profiling SSE replacements
|
||||
{
|
||||
const bench_target = b.resolveTargetQuery(.{
|
||||
.cpu_arch = .x86,
|
||||
.os_tag = .linux,
|
||||
.cpu_features_add = std.Target.x86.featureSet(&.{ .sse, .sse2 }),
|
||||
});
|
||||
const bench_math_sse = b.addObject(.{
|
||||
.name = "bench_math_sse",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/transform44/math_sse.zig"),
|
||||
.target = bench_target,
|
||||
.optimize = .ReleaseFast,
|
||||
}),
|
||||
});
|
||||
const bench_optimize = b.option(std.builtin.OptimizeMode, "bench-opt", "Bench optimization (default: ReleaseFast)") orelse .ReleaseFast;
|
||||
const bench = b.addExecutable(.{
|
||||
.name = "bench",
|
||||
.root_module = b.createModule(.{
|
||||
.root_source_file = b.path("src/bench/main.zig"),
|
||||
.target = bench_target,
|
||||
.optimize = bench_optimize,
|
||||
}),
|
||||
});
|
||||
bench.root_module.addObject(bench_math_sse);
|
||||
bench.root_module.linkSystemLibrary("m", .{});
|
||||
const install_bench = b.addInstallArtifact(bench, .{});
|
||||
const bench_step = b.step("bench", "Build math_sse benchmark harness (x86 Linux)");
|
||||
bench_step.dependOn(&install_bench.step);
|
||||
|
||||
const run_bench = b.addRunArtifact(bench);
|
||||
const run_step = b.step("run-bench", "Build and run math_sse benchmark");
|
||||
run_step.dependOn(&run_bench.step);
|
||||
}
|
||||
|
||||
// Convenience step to build all single-module variants
|
||||
const build_all_step = b.step("all-variants", "Build all DLL variants");
|
||||
|
||||
|
||||
@@ -0,0 +1,379 @@
|
||||
//! Micro-benchmark harness for math_sse replacements.
|
||||
//!
|
||||
//! Extracts original x87 FPU function bytes from WoW.exe, maps them executable,
|
||||
//! and benchmarks against our SSE replacements. Runs on x86 Linux (32-bit).
|
||||
//!
|
||||
//! Build: zig build bench
|
||||
//! Run: zig build run-bench
|
||||
|
||||
const std = @import("std");
|
||||
const posix = std.posix;
|
||||
const linux = std.os.linux;
|
||||
const originals = @import("originals.zig");
|
||||
|
||||
// SSE implementations (C ABI — export fn from math_sse.zig)
|
||||
extern fn vecMulMat4_ColMajor(u32, u32, u32) u32;
|
||||
extern fn matMulVec3_RowMajor(u32, u32, u32) u32;
|
||||
extern fn quatMulMat4(u32, u32, u32) u32;
|
||||
extern fn vec3MulScalar(u32, u32, u32) u32;
|
||||
extern fn vec3MulAssign(u32, u32) u32;
|
||||
extern fn applyTranslationMatrix(u32, u32) u32;
|
||||
extern fn scaleMatrix3x3ByVector(u32, u32) u32;
|
||||
extern fn scaleMatrix3x3ByScalar(u32, u32) void;
|
||||
extern fn multiply3x3Matrix(u32, u32, u32) u32;
|
||||
extern fn createAxisAngleRotMat3x3(u32, u32, u32, u32) u32;
|
||||
extern fn createAxisAngleRotMat4x4(u32, u32, u32, u32) u32;
|
||||
extern fn crossProduct(u32, u32, u32) u32;
|
||||
extern fn dotProduct(u32, u32) f64;
|
||||
extern fn squaredMagnitude(u32) f64;
|
||||
extern fn evaluatePolynomial(u32, u32, u32) f64;
|
||||
extern fn calculatePlaneNormal(u32, u32, u32, u32) void;
|
||||
extern fn transformAABox(u32, u32, u32, u32, u32) void;
|
||||
|
||||
// =========================================================================
|
||||
// Infrastructure
|
||||
// =========================================================================
|
||||
|
||||
fn print(comptime fmt: []const u8, args: anytype) void {
|
||||
var buf: [1024]u8 = undefined;
|
||||
const msg = std.fmt.bufPrint(&buf, fmt, args) catch return;
|
||||
_ = linux.write(1, msg.ptr, msg.len);
|
||||
}
|
||||
|
||||
fn makeExecutable(comptime bytes: []const u8) ?[*]const u8 {
|
||||
const mem = posix.mmap(
|
||||
null, 4096,
|
||||
.{ .READ = true, .WRITE = true, .EXEC = true },
|
||||
.{ .TYPE = .PRIVATE, .ANONYMOUS = true },
|
||||
-1, 0,
|
||||
) catch return null;
|
||||
@memcpy(mem[0..bytes.len], bytes);
|
||||
return mem.ptr;
|
||||
}
|
||||
|
||||
fn mapGameConstants() bool {
|
||||
const mem = posix.mmap(
|
||||
@ptrFromInt(0x007ff000), 4096,
|
||||
.{ .READ = true, .WRITE = true },
|
||||
.{ .TYPE = .PRIVATE, .ANONYMOUS = true, .FIXED = true },
|
||||
-1, 0,
|
||||
) catch return false;
|
||||
const p: *f32 = @ptrCast(@alignCast(&mem[0x9d8]));
|
||||
p.* = 1.0;
|
||||
return true;
|
||||
}
|
||||
|
||||
inline fn rdtsc() u64 {
|
||||
var lo: u32 = undefined;
|
||||
var hi: u32 = undefined;
|
||||
asm volatile ("rdtsc"
|
||||
: [lo] "={eax}" (lo),
|
||||
[hi] "={edx}" (hi),
|
||||
);
|
||||
return (@as(u64, hi) << 32) | lo;
|
||||
}
|
||||
|
||||
fn a(ptr: anytype) u32 {
|
||||
return @intFromPtr(ptr);
|
||||
}
|
||||
|
||||
fn compareF32(x: f32, y: f32) bool {
|
||||
if (x == y) return true;
|
||||
const d = @abs(x - y);
|
||||
const m = @max(@abs(x), @abs(y));
|
||||
if (m < 1e-7) return d < 1e-7;
|
||||
return d / m < 1e-4;
|
||||
}
|
||||
|
||||
fn cmpSlice(x: []const f32, y: []const f32) bool {
|
||||
for (x, y) |a2, b| if (!compareF32(a2, b)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
fn report(name: []const u8, orig_cyc: u64, sse_cyc: u64, ok: bool) void {
|
||||
const N = ITERS;
|
||||
const op = orig_cyc / N;
|
||||
const sp = sse_cyc / N;
|
||||
const sx10 = if (sp > 0) op * 10 / sp else 0;
|
||||
print("{s:>30}: orig={d:>4} sse={d:>4} cyc/call {d}.{d}x {s}\n", .{
|
||||
name, op, sp, sx10 / 10, sx10 % 10, if (ok) "OK" else "MISMATCH",
|
||||
});
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Calling convention types for original x87 functions (game binary)
|
||||
// =========================================================================
|
||||
|
||||
const cc_fc: std.builtin.CallingConvention = .{ .x86_fastcall = .{} };
|
||||
const cc_tc: std.builtin.CallingConvention = .{ .x86_thiscall = .{} };
|
||||
|
||||
const ITERS: u64 = 2_000_000;
|
||||
|
||||
// =========================================================================
|
||||
// Test data
|
||||
// =========================================================================
|
||||
|
||||
const Vec3 = [3]f32;
|
||||
const Vec4 = [4]f32;
|
||||
const Mat3 = [9]f32;
|
||||
const Mat4 = [16]f32;
|
||||
|
||||
fn tv3() Vec3 { return .{ 1.5, -2.3, 0.7 }; }
|
||||
fn tv3b() Vec3 { return .{ 0.4, 3.1, -1.2 }; }
|
||||
fn tv3c() Vec3 { return .{ -0.8, 1.6, 2.5 }; }
|
||||
fn tq4() Vec4 { return .{ 0.5, -0.5, 0.5, 0.5 }; }
|
||||
fn tm4() Mat4 {
|
||||
return .{ 1.0, 0.2, 0.3, 0.0, 0.1, 2.0, 0.4, 0.0, 0.2, 0.1, 1.5, 0.0, 1.0, 2.0, 3.0, 1.0 };
|
||||
}
|
||||
fn tm3() Mat3 { return .{ 1.0, 0.2, 0.3, 0.1, 2.0, 0.4, 0.2, 0.1, 1.5 }; }
|
||||
fn tm3b() Mat3 { return .{ 0.5, -0.1, 0.3, 0.2, 1.0, -0.2, -0.1, 0.4, 0.8 }; }
|
||||
|
||||
// =========================================================================
|
||||
// Main
|
||||
// =========================================================================
|
||||
|
||||
pub fn main() void {
|
||||
if (!mapGameConstants()) {
|
||||
print("WARNING: could not map game constants at 0x7ff000\n", .{});
|
||||
}
|
||||
|
||||
print("\nmath_sse benchmark -- {d}M iterations per function\n", .{ITERS / 1_000_000});
|
||||
print("{s:>30} {s:>10} {s:>10} {s}\n", .{ "function", "original", "sse", "status" });
|
||||
print("{s}\n", .{"-" ** 72});
|
||||
|
||||
// 1: vecMulMat4 -- fastcall(ECX=result, EDX=vec, stack=mat) -> u32
|
||||
bench_fc3r("vecMulMat4_ColMajor", originals.vecMulMat4_ColMajor, &vecMulMat4_ColMajor, tv3(), tm4(), 3);
|
||||
|
||||
// 2: matMulVec3 -- fastcall(ECX=result, EDX=mat, stack=vec) -> u32
|
||||
bench_fc3r("matMulVec3_RowMajor", originals.matMulVec3_RowMajor, &matMulVec3_RowMajor, tm4(), tv3(), 3);
|
||||
|
||||
// 3: quatMulMat4 -- fastcall(ECX=result, EDX=quat, stack=mat) -> u32
|
||||
bench_fc3r("quatMulMat4", originals.quatMulMat4, &quatMulMat4, tq4(), tm4(), 4);
|
||||
|
||||
// 4: vec3MulScalar -- fastcall(ECX=result, EDX=vec, stack=factor_bits) -> u32
|
||||
{
|
||||
const factor: f32 = 2.5;
|
||||
const fb: u32 = @bitCast(factor);
|
||||
const v = tv3();
|
||||
var ro: Vec3 = undefined;
|
||||
var rs: Vec3 = undefined;
|
||||
const of: *const fn (u32, u32, u32) callconv(cc_fc) u32 = @ptrCast(makeExecutable(&originals.vec3MulScalar) orelse unreachable);
|
||||
_ = of(a(&ro), a(&v), fb);
|
||||
_ = vec3MulScalar(a(&rs), a(&v), fb);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t: u64 = 0;
|
||||
var s: u64 = 0;
|
||||
t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&v), fb); } t = rdtsc() - t;
|
||||
s = rdtsc(); for (0..ITERS) |_| { _ = vec3MulScalar(a(&rs), a(&v), fb); } s = rdtsc() - s;
|
||||
report("vec3MulScalar", t, s, ok);
|
||||
}
|
||||
|
||||
// 5: vec3MulAssign -- thiscall(ECX=self, stack=factor_bits) -> u32
|
||||
{
|
||||
const fb: u32 = @bitCast(@as(f32, 2.5));
|
||||
var do = tv3();
|
||||
var ds = tv3();
|
||||
const of: *const fn (u32, u32) callconv(cc_tc) u32 = @ptrCast(makeExecutable(&originals.vec3MulAssign) orelse unreachable);
|
||||
_ = of(a(&do), fb);
|
||||
_ = vec3MulAssign(a(&ds), fb);
|
||||
const ok = cmpSlice(&do, &ds);
|
||||
do = tv3();
|
||||
ds = tv3();
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&do), fb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = vec3MulAssign(a(&ds), fb); } s = rdtsc() - s;
|
||||
report("vec3MulAssign", t, s, ok);
|
||||
}
|
||||
|
||||
// 6: applyTranslation -- thiscall(ECX=mat, stack=vec) -> u32
|
||||
bench_tc2r("applyTranslation", originals.applyTranslationMatrix, &applyTranslationMatrix, tm4(), tv3(), 16);
|
||||
|
||||
// 7: scaleByVec -- thiscall(ECX=mat, stack=vec) -> u32
|
||||
bench_tc2r("scaleByVec", originals.scaleMatrix3x3ByVector, &scaleMatrix3x3ByVector, tm4(), tv3(), 16);
|
||||
|
||||
// 8: scaleByScalar -- thiscall(ECX=mat, stack=factor_bits) -> void
|
||||
{
|
||||
const fb: u32 = @bitCast(@as(f32, 0.5));
|
||||
var mo = tm4();
|
||||
var ms = tm4();
|
||||
const of: *const fn (u32, u32) callconv(cc_tc) void = @ptrCast(makeExecutable(&originals.scaleMatrix3x3ByScalar) orelse unreachable);
|
||||
of(a(&mo), fb);
|
||||
scaleMatrix3x3ByScalar(a(&ms), fb);
|
||||
const ok = cmpSlice(&mo, &ms);
|
||||
mo = tm4();
|
||||
ms = tm4();
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&mo), fb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { scaleMatrix3x3ByScalar(a(&ms), fb); } s = rdtsc() - s;
|
||||
report("scaleByScalar", t, s, ok);
|
||||
}
|
||||
|
||||
// 9: mul3x3 -- fastcall(ECX=result, EDX=matA, stack=matB) -> u32
|
||||
bench_fc3r("multiply3x3", originals.multiply3x3Matrix, &multiply3x3Matrix, tm3(), tm3b(), 9);
|
||||
|
||||
// 10: rotMat3x3 -- fastcall(ECX=result, EDX=axis, stack=angle_bits, is_unit) -> u32
|
||||
{
|
||||
const axis = Vec3{ 0.0, 1.0, 0.0 };
|
||||
const ab: u32 = @bitCast(@as(f32, 0.7854));
|
||||
var ro: Mat3 = undefined;
|
||||
var rs: Mat3 = undefined;
|
||||
const of: *const fn (u32, u32, u32, u32) callconv(cc_fc) u32 = @ptrCast(makeExecutable(&originals.createAxisAngleRotMat3x3) orelse unreachable);
|
||||
_ = of(a(&ro), a(&axis), ab, 1);
|
||||
_ = createAxisAngleRotMat3x3(a(&rs), a(&axis), ab, 1);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis), ab, 1); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = createAxisAngleRotMat3x3(a(&rs), a(&axis), ab, 1); } s = rdtsc() - s;
|
||||
report("rotMat3x3", t, s, ok);
|
||||
}
|
||||
|
||||
// 11: rotMat4x4
|
||||
{
|
||||
const axis = Vec3{ 0.0, 1.0, 0.0 };
|
||||
const ab: u32 = @bitCast(@as(f32, 0.7854));
|
||||
var ro: Mat4 = undefined;
|
||||
var rs: Mat4 = undefined;
|
||||
const of: *const fn (u32, u32, u32, u32) callconv(cc_fc) u32 = @ptrCast(makeExecutable(&originals.createAxisAngleRotMat4x4) orelse unreachable);
|
||||
_ = of(a(&ro), a(&axis), ab, 1);
|
||||
_ = createAxisAngleRotMat4x4(a(&rs), a(&axis), ab, 1);
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&ro), a(&axis), ab, 1); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = createAxisAngleRotMat4x4(a(&rs), a(&axis), ab, 1); } s = rdtsc() - s;
|
||||
report("rotMat4x4", t, s, ok);
|
||||
}
|
||||
|
||||
// 12: cross -- fastcall(ECX=result, EDX=vecA, stack=vecB) -> u32
|
||||
bench_fc3r("crossProduct", originals.crossProduct, &crossProduct, tv3(), tv3b(), 3);
|
||||
|
||||
// 13: dot -- fastcall(ECX=vecA, EDX=vecB) -> f64
|
||||
{
|
||||
const va = tv3();
|
||||
const vb = tv3b();
|
||||
const of: *const fn (u32, u32) callconv(cc_fc) f64 = @ptrCast(makeExecutable(&originals.dotProduct) orelse unreachable);
|
||||
const ov = of(a(&va), a(&vb));
|
||||
const sv = dotProduct(a(&va), a(&vb));
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&va), a(&vb)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = dotProduct(a(&va), a(&vb)); } s = rdtsc() - s;
|
||||
report("dotProduct", t, s, ok);
|
||||
}
|
||||
|
||||
// 14: sqmag -- thiscall(ECX=vec) -> f64
|
||||
{
|
||||
const v = tv3();
|
||||
const of: *const fn (u32) callconv(cc_tc) f64 = @ptrCast(makeExecutable(&originals.squaredMagnitude) orelse unreachable);
|
||||
const ov = of(a(&v));
|
||||
const sv = squaredMagnitude(a(&v));
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(a(&v)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = squaredMagnitude(a(&v)); } s = rdtsc() - s;
|
||||
report("squaredMagnitude", t, s, ok);
|
||||
}
|
||||
|
||||
// 16: evalPoly -- fastcall(ECX=count, EDX=coeffs, stack=factor_bits) -> f64
|
||||
{
|
||||
const coeffs = [4]f32{ 3.0, -2.0, 1.0, 0.5 };
|
||||
const fb: u32 = @bitCast(@as(f32, 1.5));
|
||||
const of: *const fn (u32, u32, u32) callconv(cc_fc) f64 = @ptrCast(makeExecutable(&originals.evaluatePolynomial) orelse unreachable);
|
||||
const ov = of(3, a(&coeffs), fb);
|
||||
const sv = evaluatePolynomial(3, a(&coeffs), fb);
|
||||
const ok = @abs(ov - sv) < 1e-4;
|
||||
var t = rdtsc(); for (0..ITERS) |_| { _ = of(3, a(&coeffs), fb); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { _ = evaluatePolynomial(3, a(&coeffs), fb); } s = rdtsc() - s;
|
||||
report("evaluatePolynomial", t, s, ok);
|
||||
}
|
||||
|
||||
// 17: planeNormal -- thiscall(ECX=result, stack=p1,p2,p3) -> void
|
||||
{
|
||||
const p1 = tv3();
|
||||
const p2 = tv3b();
|
||||
const p3 = tv3c();
|
||||
var ro: Vec4 = undefined;
|
||||
var rs: Vec4 = undefined;
|
||||
const of: *const fn (u32, u32, u32, u32) callconv(cc_tc) void = @ptrCast(makeExecutable(&originals.calculatePlaneNormal) orelse unreachable);
|
||||
of(a(&ro), a(&p1), a(&p2), a(&p3));
|
||||
calculatePlaneNormal(a(&rs), a(&p1), a(&p2), a(&p3));
|
||||
const ok = cmpSlice(&ro, &rs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { of(a(&ro), a(&p1), a(&p2), a(&p3)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { calculatePlaneNormal(a(&rs), a(&p1), a(&p2), a(&p3)); } s = rdtsc() - s;
|
||||
report("planeNormal", t, s, ok);
|
||||
}
|
||||
|
||||
// 18: transformAABox -- fastcall(ECX=mat, EDX=vecA, stack=vecB,boxIn,boxOut) -> void
|
||||
{
|
||||
const mat = tm3();
|
||||
const va = tv3();
|
||||
const vb = tv3b();
|
||||
const box_in = [6]f32{ -1.0, -1.0, -1.0, 1.0, 1.0, 1.0 };
|
||||
var bo: [6]f32 = .{ 0, 0, 0, 0, 0, 0 };
|
||||
var bs: [6]f32 = .{ 0, 0, 0, 0, 0, 0 };
|
||||
const of: *const fn (u32, u32, u32, u32, u32) callconv(cc_fc) void = @ptrCast(makeExecutable(&originals.transformAABox) orelse unreachable);
|
||||
of(a(&mat), a(&va), a(&vb), a(&box_in), a(&bo));
|
||||
transformAABox(a(&mat), a(&va), a(&vb), a(&box_in), a(&bs));
|
||||
const ok = cmpSlice(&bo, &bs);
|
||||
var t = rdtsc(); for (0..ITERS) |_| { bo = .{ 0, 0, 0, 0, 0, 0 }; of(a(&mat), a(&va), a(&vb), a(&box_in), a(&bo)); } t = rdtsc() - t;
|
||||
var s = rdtsc(); for (0..ITERS) |_| { bs = .{ 0, 0, 0, 0, 0, 0 }; transformAABox(a(&mat), a(&va), a(&vb), a(&box_in), a(&bs)); } s = rdtsc() - s;
|
||||
report("transformAABox", t, s, ok);
|
||||
}
|
||||
|
||||
print("\n", .{});
|
||||
}
|
||||
|
||||
// =========================================================================
|
||||
// Generic benchmarks for common signatures
|
||||
// =========================================================================
|
||||
|
||||
/// fastcall(ECX=result, EDX=paramA, stack=paramB) -> u32
|
||||
fn bench_fc3r(
|
||||
comptime name: []const u8,
|
||||
comptime orig_bytes: anytype,
|
||||
sse_fn: *const fn (u32, u32, u32) callconv(.c) u32,
|
||||
param_a: anytype,
|
||||
param_b: anytype,
|
||||
comptime result_len: usize,
|
||||
) void {
|
||||
const of: *const fn (u32, u32, u32) callconv(cc_fc) u32 = @ptrCast(makeExecutable(&orig_bytes) orelse {
|
||||
print("{s:>30}: FAILED to map\n", .{name});
|
||||
return;
|
||||
});
|
||||
var ro: [16]f32 = undefined;
|
||||
var rs: [16]f32 = undefined;
|
||||
_ = of(a(&ro), a(¶m_a), a(¶m_b));
|
||||
_ = sse_fn(a(&rs), a(¶m_a), a(¶m_b));
|
||||
const ok = cmpSlice(ro[0..result_len], rs[0..result_len]);
|
||||
|
||||
var t = rdtsc();
|
||||
for (0..ITERS) |_| { _ = of(a(&ro), a(¶m_a), a(¶m_b)); }
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
for (0..ITERS) |_| { _ = sse_fn(a(&rs), a(¶m_a), a(¶m_b)); }
|
||||
s = rdtsc() - s;
|
||||
report(name, t, s, ok);
|
||||
}
|
||||
|
||||
/// thiscall(ECX=self, stack=param) -> u32 (in-place modification)
|
||||
fn bench_tc2r(
|
||||
comptime name: []const u8,
|
||||
comptime orig_bytes: anytype,
|
||||
sse_fn: *const fn (u32, u32) callconv(.c) u32,
|
||||
self_init: anytype,
|
||||
param: anytype,
|
||||
comptime result_len: usize,
|
||||
) void {
|
||||
const of: *const fn (u32, u32) callconv(cc_tc) u32 = @ptrCast(makeExecutable(&orig_bytes) orelse {
|
||||
print("{s:>30}: FAILED to map\n", .{name});
|
||||
return;
|
||||
});
|
||||
var so: @TypeOf(self_init) = self_init;
|
||||
var ss: @TypeOf(self_init) = self_init;
|
||||
_ = of(a(&so), a(¶m));
|
||||
_ = sse_fn(a(&ss), a(¶m));
|
||||
const ok = cmpSlice(@as([*]const f32, @ptrCast(&so))[0..result_len], @as([*]const f32, @ptrCast(&ss))[0..result_len]);
|
||||
|
||||
so = self_init;
|
||||
ss = self_init;
|
||||
var t = rdtsc();
|
||||
for (0..ITERS) |_| { _ = of(a(&so), a(¶m)); }
|
||||
t = rdtsc() - t;
|
||||
var s = rdtsc();
|
||||
for (0..ITERS) |_| { _ = sse_fn(a(&ss), a(¶m)); }
|
||||
s = rdtsc() - s;
|
||||
report(name, t, s, ok);
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
//! Original x87 FPU function bytes extracted from WoW.exe (1.12.1 build 5875).
|
||||
//! Each array is the complete function body from entry point through final RET.
|
||||
//! These get mmap'd as executable and called via function pointers in the bench harness.
|
||||
|
||||
pub const vecMulMat4_ColMajor = [_]u8{ 0x55, 0x8b, 0xec, 0x56, 0x8b, 0x75, 0x08, 0xd9, 0x46, 0x28, 0x8b, 0xc1, 0xd8, 0x4a, 0x08, 0xd9, 0x46, 0x08, 0xd8, 0x0a, 0xde, 0xc1, 0xd9, 0x46, 0x18, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd8, 0x46, 0x38, 0xd9, 0x46, 0x24, 0xd8, 0x4a, 0x08, 0xd9, 0x46, 0x04, 0xd8, 0x0a, 0xde, 0xc1, 0xd9, 0x46, 0x14, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd8, 0x46, 0x34, 0xd9, 0x46, 0x20, 0xd8, 0x4a, 0x08, 0xd9, 0x46, 0x10, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd9, 0x02, 0xd8, 0x0e, 0xde, 0xc1, 0xd8, 0x46, 0x30, 0x5e, 0xd9, 0x18, 0xd9, 0x58, 0x04, 0xd9, 0x58, 0x08, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BCA80, 93 bytes
|
||||
pub const matMulVec3_RowMajor = [_]u8{ 0x55, 0x8b, 0xec, 0xd9, 0x42, 0x28, 0x56, 0x8b, 0x75, 0x08, 0xd8, 0x4e, 0x08, 0x8b, 0xc1, 0xd9, 0x42, 0x20, 0xd8, 0x0e, 0xde, 0xc1, 0xd9, 0x42, 0x24, 0xd8, 0x4e, 0x04, 0xde, 0xc1, 0xd8, 0x42, 0x2c, 0xd9, 0x42, 0x18, 0xd8, 0x4e, 0x08, 0xd9, 0x42, 0x10, 0xd8, 0x0e, 0xde, 0xc1, 0xd9, 0x42, 0x14, 0xd8, 0x4e, 0x04, 0xde, 0xc1, 0xd8, 0x42, 0x1c, 0xd9, 0x42, 0x08, 0xd8, 0x4e, 0x08, 0xd9, 0x42, 0x04, 0xd8, 0x4e, 0x04, 0xde, 0xc1, 0xd9, 0x02, 0xd8, 0x0e, 0x5e, 0xde, 0xc1, 0xd8, 0x42, 0x0c, 0xd9, 0x18, 0xd9, 0x58, 0x04, 0xd9, 0x58, 0x08, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BCAE0, 93 bytes
|
||||
pub const quatMulMat4 = [_]u8{ 0x55, 0x8b, 0xec, 0x56, 0x8b, 0x75, 0x08, 0xd9, 0x46, 0x3c, 0x8b, 0xc1, 0xd8, 0x4a, 0x0c, 0xd9, 0x46, 0x0c, 0xd8, 0x0a, 0xde, 0xc1, 0xd9, 0x46, 0x2c, 0xd8, 0x4a, 0x08, 0xde, 0xc1, 0xd9, 0x46, 0x1c, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd9, 0x46, 0x38, 0xd8, 0x4a, 0x0c, 0xd9, 0x46, 0x08, 0xd8, 0x0a, 0xde, 0xc1, 0xd9, 0x46, 0x28, 0xd8, 0x4a, 0x08, 0xde, 0xc1, 0xd9, 0x46, 0x18, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd9, 0x46, 0x34, 0xd8, 0x4a, 0x0c, 0xd9, 0x46, 0x04, 0xd8, 0x0a, 0xde, 0xc1, 0xd9, 0x46, 0x24, 0xd8, 0x4a, 0x08, 0xde, 0xc1, 0xd9, 0x46, 0x14, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd9, 0x46, 0x30, 0xd8, 0x4a, 0x0c, 0xd9, 0x46, 0x20, 0xd8, 0x4a, 0x08, 0xde, 0xc1, 0xd9, 0x46, 0x10, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd9, 0x02, 0xd8, 0x0e, 0x5e, 0xde, 0xc1, 0xd9, 0x18, 0xd9, 0x58, 0x04, 0xd9, 0x58, 0x08, 0xd9, 0x58, 0x0c, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BCB40, 140 bytes
|
||||
pub const vec3MulScalar = [_]u8{ 0x55, 0x8b, 0xec, 0xd9, 0x45, 0x08, 0x8b, 0xc1, 0xd8, 0x4a, 0x08, 0xd9, 0x45, 0x08, 0xd8, 0x4a, 0x04, 0xd9, 0x45, 0x08, 0xd8, 0x0a, 0xd9, 0x18, 0xd9, 0x58, 0x04, 0xd9, 0x58, 0x08, 0x5d, 0xc2, 0x04, 0x00 }; // 0x5F8CF0, 34 bytes
|
||||
pub const vec3MulAssign = [_]u8{ 0x55, 0x8b, 0xec, 0xd9, 0x45, 0x08, 0x8b, 0xc1, 0xd8, 0x08, 0xd9, 0x18, 0xd9, 0x45, 0x08, 0xd8, 0x48, 0x04, 0xd9, 0x58, 0x04, 0xd9, 0x45, 0x08, 0xd8, 0x48, 0x08, 0xd9, 0x58, 0x08, 0x5d, 0xc2, 0x04, 0x00 }; // 0x5132F0, 34 bytes
|
||||
pub const applyTranslationMatrix = [_]u8{ 0x55, 0x8b, 0xec, 0x8b, 0x45, 0x08, 0xd9, 0x41, 0x20, 0xd8, 0x48, 0x08, 0xd9, 0x41, 0x10, 0xd8, 0x48, 0x04, 0xde, 0xc1, 0xd9, 0x00, 0xd8, 0x09, 0xde, 0xc1, 0xd8, 0x41, 0x30, 0xd9, 0x59, 0x30, 0xd9, 0x41, 0x24, 0xd8, 0x48, 0x08, 0xd9, 0x41, 0x14, 0xd8, 0x48, 0x04, 0xde, 0xc1, 0xd9, 0x41, 0x04, 0xd8, 0x08, 0xde, 0xc1, 0xd8, 0x41, 0x34, 0xd9, 0x59, 0x34, 0xd9, 0x41, 0x28, 0xd8, 0x48, 0x08, 0xd9, 0x41, 0x18, 0xd8, 0x48, 0x04, 0xde, 0xc1, 0xd9, 0x41, 0x08, 0xd8, 0x08, 0xde, 0xc1, 0xd8, 0x41, 0x38, 0xd9, 0x59, 0x38, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BDC40, 90 bytes
|
||||
pub const scaleMatrix3x3ByVector = [_]u8{ 0x55, 0x8b, 0xec, 0x8b, 0x45, 0x08, 0xd9, 0x00, 0xd9, 0xc0, 0xd8, 0x09, 0xd9, 0x19, 0xd9, 0xc0, 0xd8, 0x49, 0x04, 0xd9, 0x59, 0x04, 0xd8, 0x49, 0x08, 0xd9, 0x59, 0x08, 0xd9, 0x40, 0x04, 0xd9, 0xc0, 0xd8, 0x49, 0x10, 0xd9, 0x59, 0x10, 0xd9, 0xc0, 0xd8, 0x49, 0x14, 0xd9, 0x59, 0x14, 0xd8, 0x49, 0x18, 0xd9, 0x59, 0x18, 0xd9, 0x40, 0x08, 0xd9, 0xc0, 0xd8, 0x49, 0x20, 0xd9, 0x59, 0x20, 0xd9, 0xc0, 0xd8, 0x49, 0x24, 0xd9, 0x59, 0x24, 0xd8, 0x49, 0x28, 0xd9, 0x59, 0x28, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BDCA0, 82 bytes
|
||||
pub const scaleMatrix3x3ByScalar = [_]u8{ 0x55, 0x8b, 0xec, 0xd9, 0x45, 0x08, 0xd8, 0x09, 0xd9, 0x19, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x04, 0xd9, 0x59, 0x04, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x08, 0xd9, 0x59, 0x08, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x10, 0xd9, 0x59, 0x10, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x14, 0xd9, 0x59, 0x14, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x18, 0xd9, 0x59, 0x18, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x20, 0xd9, 0x59, 0x20, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x24, 0xd9, 0x59, 0x24, 0xd9, 0x45, 0x08, 0xd8, 0x49, 0x28, 0xd9, 0x59, 0x28, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BDD00, 86 bytes
|
||||
pub const multiply3x3Matrix = [_]u8{ 0x55, 0x8b, 0xec, 0x51, 0xd9, 0x42, 0x1c, 0x56, 0x8b, 0x75, 0x08, 0xd8, 0x4e, 0x14, 0x8b, 0xc1, 0xd9, 0x42, 0x18, 0xd8, 0x4e, 0x08, 0xde, 0xc1, 0xd9, 0x46, 0x20, 0xd8, 0x4a, 0x20, 0xde, 0xc1, 0xd9, 0x42, 0x1c, 0xd8, 0x4e, 0x10, 0xd9, 0x42, 0x18, 0xd8, 0x4e, 0x04, 0xde, 0xc1, 0xd9, 0x42, 0x20, 0xd8, 0x4e, 0x1c, 0xde, 0xc1, 0xd9, 0x46, 0x18, 0xd8, 0x4a, 0x20, 0xd9, 0x42, 0x1c, 0xd8, 0x4e, 0x0c, 0xde, 0xc1, 0xd9, 0x42, 0x18, 0xd8, 0x0e, 0xde, 0xc1, 0xd9, 0x42, 0x14, 0xd8, 0x4e, 0x20, 0xd9, 0x42, 0x0c, 0xd8, 0x4e, 0x08, 0xde, 0xc1, 0xd9, 0x42, 0x10, 0xd8, 0x4e, 0x14, 0xde, 0xc1, 0xd9, 0x42, 0x14, 0xd8, 0x4e, 0x1c, 0xd9, 0x42, 0x10, 0xd8, 0x4e, 0x10, 0xde, 0xc1, 0xd9, 0x42, 0x0c, 0xd8, 0x4e, 0x04, 0xde, 0xc1, 0xd9, 0x42, 0x0c, 0xd8, 0x0e, 0xd9, 0x46, 0x18, 0xd8, 0x4a, 0x14, 0xde, 0xc1, 0xd9, 0x46, 0x0c, 0xd8, 0x4a, 0x10, 0xde, 0xc1, 0xd9, 0x46, 0x20, 0xd8, 0x4a, 0x08, 0xd9, 0x02, 0xd8, 0x4e, 0x08, 0xde, 0xc1, 0xd9, 0x42, 0x04, 0xd8, 0x4e, 0x14, 0xde, 0xc1, 0xd9, 0x5d, 0xfc, 0xd9, 0x46, 0x04, 0xd8, 0x0a, 0xd9, 0x46, 0x1c, 0xd8, 0x4a, 0x08, 0xde, 0xc1, 0xd9, 0x42, 0x04, 0xd8, 0x4e, 0x10, 0xde, 0xc1, 0xd9, 0x5d, 0x08, 0xd9, 0x46, 0x18, 0xd8, 0x4a, 0x08, 0xd9, 0x46, 0x0c, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0x8b, 0x4d, 0x08, 0xd9, 0x02, 0x8b, 0x55, 0xfc, 0xd8, 0x0e, 0x89, 0x48, 0x04, 0x89, 0x50, 0x08, 0x5e, 0xde, 0xc1, 0xd9, 0x18, 0xd9, 0x58, 0x0c, 0xd9, 0x58, 0x10, 0xd9, 0x58, 0x14, 0xd9, 0x58, 0x18, 0xd9, 0x58, 0x1c, 0xd9, 0x58, 0x20, 0x8b, 0xe5, 0x5d, 0xc2, 0x04, 0x00 }; // 0x7BDFC0, 247 bytes
|
||||
pub const createAxisAngleRotMat3x3 = [_]u8{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x18, 0x8b, 0x02, 0x89, 0x45, 0xe8, 0x8b, 0x42, 0x04, 0x8b, 0x52, 0x08, 0x89, 0x45, 0xec, 0x8a, 0x45, 0x0c, 0x84, 0xc0, 0x89, 0x55, 0xf0, 0x75, 0x34, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0xf0, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0xec, 0xde, 0xc1, 0xd9, 0x45, 0xe8, 0xd8, 0x4d, 0xe8, 0xde, 0xc1, 0xd9, 0xfa, 0xd8, 0x3d, 0xd8, 0xf9, 0x7f, 0x00, 0xd9, 0x45, 0xe8, 0xd8, 0xc9, 0xd9, 0x5d, 0xe8, 0xd9, 0x45, 0xec, 0xd8, 0xc9, 0xd9, 0x5d, 0xec, 0xd8, 0x4d, 0xf0, 0xd9, 0x5d, 0xf0, 0x8d, 0x45, 0xfc, 0x8d, 0x55, 0x0c, 0x89, 0x45, 0xf8, 0x89, 0x55, 0xf4, 0x8b, 0x45, 0xf8, 0x8b, 0x55, 0xf4, 0xd9, 0x45, 0x08, 0xd9, 0xfb, 0xd9, 0x18, 0xd9, 0x1a, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0xe8, 0x8b, 0xc1, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0xec, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0xe8, 0xd9, 0x45, 0xe8, 0xd8, 0x4d, 0x0c, 0xd9, 0x5d, 0xf4, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0x0c, 0xd9, 0x5d, 0xf8, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0x0c, 0xd9, 0x05, 0xd8, 0xf9, 0x7f, 0x00, 0xd8, 0x65, 0xfc, 0xd9, 0x5d, 0x0c, 0xd9, 0x45, 0xe8, 0xd8, 0x4d, 0xe8, 0xd8, 0x4d, 0x0c, 0xd8, 0x45, 0xfc, 0xd9, 0x19, 0xd9, 0x45, 0x0c, 0xd8, 0xcc, 0xd9, 0xc0, 0xd8, 0xc2, 0xd9, 0x59, 0x04, 0xd9, 0x45, 0x0c, 0xd8, 0xcb, 0xd9, 0x55, 0x08, 0xd8, 0x65, 0xf8, 0xd9, 0x59, 0x08, 0xd8, 0xe1, 0xd9, 0x59, 0x0c, 0xdd, 0xd8, 0xdd, 0xd8, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0xec, 0xd8, 0x4d, 0x0c, 0xd8, 0x45, 0xfc, 0xd9, 0x59, 0x10, 0xd9, 0x45, 0x0c, 0xd8, 0xc9, 0xdd, 0xda, 0xdd, 0xd8, 0xd9, 0x45, 0xf4, 0xd8, 0xc1, 0xd9, 0x59, 0x14, 0xd9, 0x45, 0x08, 0xd8, 0x45, 0xf8, 0xd9, 0x59, 0x18, 0xd8, 0x65, 0xf4, 0xd9, 0x59, 0x1c, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0xf0, 0xd8, 0x4d, 0x0c, 0xd8, 0x45, 0xfc, 0xd9, 0x59, 0x20, 0x8b, 0xe5, 0x5d, 0xc2, 0x08, 0x00 }; // 0x7BE490, 282 bytes
|
||||
pub const createAxisAngleRotMat4x4 = [_]u8{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x18, 0x8b, 0x02, 0x89, 0x45, 0xe8, 0x8b, 0x42, 0x04, 0x8b, 0x52, 0x08, 0x53, 0x89, 0x45, 0xec, 0x8a, 0x45, 0x0c, 0x33, 0xdb, 0x3a, 0xc3, 0x89, 0x55, 0xf0, 0x75, 0x34, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0xf0, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0xec, 0xde, 0xc1, 0xd9, 0x45, 0xe8, 0xd8, 0x4d, 0xe8, 0xde, 0xc1, 0xd9, 0xfa, 0xd8, 0x3d, 0xd8, 0xf9, 0x7f, 0x00, 0xd9, 0x45, 0xe8, 0xd8, 0xc9, 0xd9, 0x5d, 0xe8, 0xd9, 0x45, 0xec, 0xd8, 0xc9, 0xd9, 0x5d, 0xec, 0xd8, 0x4d, 0xf0, 0xd9, 0x5d, 0xf0, 0x8d, 0x45, 0xfc, 0x8d, 0x55, 0x0c, 0x89, 0x45, 0xf8, 0x89, 0x55, 0xf4, 0x8b, 0x45, 0xf8, 0x8b, 0x55, 0xf4, 0xd9, 0x45, 0x08, 0xd9, 0xfb, 0xd9, 0x18, 0xd9, 0x1a, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0xe8, 0x89, 0x59, 0x0c, 0xd9, 0x45, 0xf0, 0x89, 0x59, 0x1c, 0xd8, 0x4d, 0xec, 0x89, 0x59, 0x2c, 0xd9, 0x45, 0xf0, 0x89, 0x59, 0x30, 0xd8, 0x4d, 0xe8, 0x89, 0x59, 0x34, 0xd9, 0x45, 0xe8, 0x89, 0x59, 0x38, 0xd8, 0x4d, 0x0c, 0xc7, 0x41, 0x3c, 0x00, 0x00, 0x80, 0x3f, 0x8b, 0xc1, 0x5b, 0xd9, 0x5d, 0xf4, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0x0c, 0xd9, 0x5d, 0xf8, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0x0c, 0xd9, 0x05, 0xd8, 0xf9, 0x7f, 0x00, 0xd8, 0x65, 0xfc, 0xd9, 0x5d, 0x0c, 0xd9, 0x45, 0xe8, 0xd8, 0x4d, 0xe8, 0xd8, 0x4d, 0x0c, 0xd8, 0x45, 0xfc, 0xd9, 0x19, 0xd9, 0x45, 0x0c, 0xd8, 0xcc, 0xd9, 0xc0, 0xd8, 0xc2, 0xd9, 0x59, 0x04, 0xd9, 0x45, 0x0c, 0xd8, 0xcb, 0xd9, 0x55, 0x08, 0xd8, 0x65, 0xf8, 0xd9, 0x59, 0x08, 0xd8, 0xe1, 0xd9, 0x59, 0x10, 0xdd, 0xd8, 0xdd, 0xd8, 0xd9, 0x45, 0xec, 0xd8, 0x4d, 0xec, 0xd8, 0x4d, 0x0c, 0xd8, 0x45, 0xfc, 0xd9, 0x59, 0x14, 0xd9, 0x45, 0x0c, 0xd8, 0xc9, 0xdd, 0xda, 0xdd, 0xd8, 0xd9, 0x45, 0xf4, 0xd8, 0xc1, 0xd9, 0x59, 0x18, 0xd9, 0x45, 0x08, 0xd8, 0x45, 0xf8, 0xd9, 0x59, 0x20, 0xd8, 0x65, 0xf4, 0xd9, 0x59, 0x24, 0xd9, 0x45, 0xf0, 0xd8, 0x4d, 0xf0, 0xd8, 0x4d, 0x0c, 0xd8, 0x45, 0xfc, 0xd9, 0x59, 0x28, 0x8b, 0xe5, 0x5d, 0xc2, 0x08, 0x00 }; // 0x7BDB00, 311 bytes
|
||||
pub const crossProduct = [_]u8{ 0x55, 0x8b, 0xec, 0xd9, 0x02, 0x56, 0x8b, 0x75, 0x08, 0xd8, 0x4e, 0x04, 0x8b, 0xc1, 0xd9, 0x42, 0x04, 0xd8, 0x0e, 0xde, 0xe9, 0xd9, 0x42, 0x08, 0xd8, 0x0e, 0xd9, 0x46, 0x08, 0xd8, 0x0a, 0xde, 0xe9, 0xd9, 0x42, 0x04, 0xd8, 0x4e, 0x08, 0xd9, 0x42, 0x08, 0xd8, 0x4e, 0x04, 0x5e, 0xde, 0xe9, 0xd9, 0x18, 0xd9, 0x58, 0x04, 0xd9, 0x58, 0x08, 0x5d, 0xc2, 0x04, 0x00 }; // 0x672130, 60 bytes
|
||||
pub const dotProduct = [_]u8{ 0xd9, 0x41, 0x08, 0xd8, 0x4a, 0x08, 0xd9, 0x41, 0x04, 0xd8, 0x4a, 0x04, 0xde, 0xc1, 0xd9, 0x01, 0xd8, 0x0a, 0xde, 0xc1, 0xc3 }; // 0x602630, 21 bytes
|
||||
pub const squaredMagnitude = [_]u8{ 0xd9, 0x41, 0x08, 0xd9, 0x41, 0x04, 0xd9, 0x01, 0xd9, 0xc0, 0xd8, 0xc9, 0xd9, 0xc2, 0xd8, 0xcb, 0xde, 0xc1, 0xd9, 0xc3, 0xd8, 0xcc, 0xde, 0xc1, 0xdd, 0xdb, 0xdd, 0xd8, 0xdd, 0xd8, 0xc3 }; // 0x4549F0, 31 bytes
|
||||
pub const evaluatePolynomial = [_]u8{ 0x55, 0x8b, 0xec, 0xd9, 0x02, 0xb8, 0x01, 0x00, 0x00, 0x00, 0x3b, 0xc8, 0x72, 0x0e, 0x8b, 0xff, 0xd8, 0x4d, 0x08, 0x40, 0x3b, 0xc1, 0xd8, 0x44, 0x82, 0xfc, 0x76, 0xf4, 0x5d, 0xc2, 0x04, 0x00 }; // 0x453620, 32 bytes
|
||||
pub const calculatePlaneNormal = [_]u8{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x18, 0x8b, 0x45, 0x08, 0x8b, 0x55, 0x10, 0xd9, 0x02, 0x56, 0xd8, 0x20, 0xd9, 0x42, 0x04, 0xd8, 0x60, 0x04, 0xd9, 0x42, 0x08, 0x8b, 0x55, 0x0c, 0xd8, 0x60, 0x08, 0xd9, 0x02, 0xd8, 0x20, 0xd9, 0x5d, 0xf4, 0xd9, 0x42, 0x04, 0xd8, 0x60, 0x04, 0xd9, 0x5d, 0xf8, 0xd9, 0x42, 0x08, 0x8b, 0xd1, 0xd8, 0x60, 0x08, 0xd9, 0x45, 0xf8, 0xd8, 0xca, 0xd9, 0xc1, 0xd8, 0xcc, 0xde, 0xe9, 0xd9, 0x5d, 0xe8, 0x8b, 0x75, 0xe8, 0xd8, 0xcb, 0x89, 0x32, 0xd9, 0xc9, 0xd8, 0x4d, 0xf4, 0xde, 0xe9, 0xd9, 0x5d, 0xec, 0x8b, 0x75, 0xec, 0xd8, 0x4d, 0xf4, 0x89, 0x72, 0x04, 0xd9, 0x45, 0xf8, 0xd8, 0xca, 0xde, 0xe9, 0xd9, 0x5d, 0xf0, 0x8b, 0x75, 0xf0, 0xdd, 0xd8, 0x89, 0x72, 0x08, 0xd9, 0x41, 0x08, 0xd9, 0x41, 0x04, 0xd9, 0x01, 0xd9, 0xc0, 0xd8, 0xc9, 0xd9, 0xc2, 0xd8, 0xcb, 0xde, 0xc1, 0xd9, 0xc3, 0xd8, 0xcc, 0xde, 0xc1, 0xd9, 0xfa, 0xdd, 0xdb, 0xdd, 0xd8, 0xdd, 0xd8, 0xd8, 0x3d, 0xd8, 0xf9, 0x7f, 0x00, 0xd9, 0xc0, 0xd8, 0x09, 0xd9, 0x11, 0xd9, 0xc1, 0xd8, 0x49, 0x04, 0xd9, 0x51, 0x04, 0xd9, 0xca, 0xd8, 0x49, 0x08, 0xd9, 0x51, 0x08, 0xd8, 0x48, 0x08, 0xd9, 0xca, 0xd8, 0x48, 0x04, 0xde, 0xc2, 0x5e, 0xd8, 0x08, 0xde, 0xc1, 0xd9, 0xe0, 0xd9, 0x59, 0x0c, 0x8b, 0xe5, 0x5d, 0xc2, 0x0c, 0x00 }; // 0x637480, 200 bytes
|
||||
pub const transformAABox = [_]u8{ 0x55, 0x8b, 0xec, 0x83, 0xec, 0x0c, 0x8b, 0x45, 0x08, 0x56, 0x57, 0x8b, 0x7d, 0x0c, 0x89, 0x55, 0xf8, 0x8b, 0x55, 0x10, 0x89, 0x4d, 0xf4, 0x89, 0x45, 0xfc, 0x33, 0xf6, 0x8d, 0x64, 0x24, 0x00, 0x33, 0xc9, 0x8b, 0x44, 0x8d, 0xf4, 0xd9, 0x04, 0x30, 0x03, 0xc6, 0xd8, 0x0c, 0x8f, 0xd9, 0x44, 0x8f, 0x0c, 0xd8, 0x08, 0xd9, 0x5d, 0x08, 0xd8, 0x55, 0x08, 0xdf, 0xe0, 0xf6, 0xc4, 0x05, 0x7a, 0x09, 0xd8, 0x02, 0xd9, 0x1a, 0xd9, 0x45, 0x08, 0xeb, 0x07, 0xd9, 0x45, 0x08, 0xd8, 0x02, 0xd9, 0x1a, 0xd8, 0x42, 0x0c, 0x41, 0x83, 0xf9, 0x03, 0xd9, 0x5a, 0x0c, 0x72, 0xc5, 0x83, 0xc6, 0x04, 0x83, 0xc2, 0x04, 0x83, 0xfe, 0x0c, 0x72, 0xb8, 0x5f, 0x5e, 0x8b, 0xe5, 0x5d, 0xc2, 0x0c, 0x00 }; // 0x6DC470, 112 bytes
|
||||
Reference in New Issue
Block a user