From 420de0a3925249223aee19ce36a73dca9d1f6c1e Mon Sep 17 00:00:00 2001 From: Florian De Rop <55103921+fderop@users.noreply.github.com> Date: Thu, 1 Oct 2026 22:51:28 -0700 Subject: [PATCH 1/2] Preserve cached matrix coefficients during sprite collision queries --- Source/Entities/MOSprite.cpp | 3 +-- Source/System/Matrix.cpp | 6 +++++- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/Source/Entities/MOSprite.cpp b/Source/Entities/MOSprite.cpp index 7739e8d2e4..75c046a96e 100644 --- a/Source/Entities/MOSprite.cpp +++ b/Source/Entities/MOSprite.cpp @@ -259,8 +259,7 @@ bool MOSprite::HitTestAtPixel(int pixelX, int pixelY, bool validOnly) const { // Check the scene position in the current local space of the MO, accounting for Position, Sprite Offset, Angle and HFlipped. // TODO Account for Scale as well someday, maybe. - Matrix rotation = m_Rotation; // <- Copy to non-const variable so / operator overload works. - Vector entryPos = (distanceBetweenTestPositionAndMO / rotation).GetXFlipped(m_HFlipped) - m_SpriteOffset; + Vector entryPos = (distanceBetweenTestPositionAndMO / m_Rotation).GetXFlipped(m_HFlipped) - m_SpriteOffset; int localX = entryPos.GetFloorIntX(); int localY = entryPos.GetFloorIntY(); diff --git a/Source/System/Matrix.cpp b/Source/System/Matrix.cpp index 19418c0970..def2aaa558 100644 --- a/Source/System/Matrix.cpp +++ b/Source/System/Matrix.cpp @@ -44,7 +44,11 @@ int Matrix::Create(const Matrix& reference) { m_Rotation = reference.m_Rotation; m_Flipped[X] = reference.m_Flipped[X]; m_Flipped[Y] = reference.m_Flipped[Y]; - m_ElementsUpdated = false; + m_Elements[0][0] = reference.m_Elements[0][0]; + m_Elements[0][1] = reference.m_Elements[0][1]; + m_Elements[1][0] = reference.m_Elements[1][0]; + m_Elements[1][1] = reference.m_Elements[1][1]; + m_ElementsUpdated = reference.m_ElementsUpdated; return 0; } From b75626e1a45d42bddb70e47b37ac64fcd26cd97e Mon Sep 17 00:00:00 2001 From: Florian De Rop <55103921+fderop@users.noreply.github.com> Date: Thu, 1 Oct 2026 22:52:12 -0700 Subject: [PATCH 2/2] Add production collision transform performance benchmark --- Tools/Benchmarks/CollisionTransforms.cpp | 103 +++++++++++++++++++++++ Tools/Benchmarks/CollisionTransforms.py | 29 +++++++ Tools/Benchmarks/README.md | 38 +++++++++ 3 files changed, 170 insertions(+) create mode 100644 Tools/Benchmarks/CollisionTransforms.cpp create mode 100644 Tools/Benchmarks/CollisionTransforms.py create mode 100644 Tools/Benchmarks/README.md diff --git a/Tools/Benchmarks/CollisionTransforms.cpp b/Tools/Benchmarks/CollisionTransforms.cpp new file mode 100644 index 0000000000..ff6679421f --- /dev/null +++ b/Tools/Benchmarks/CollisionTransforms.cpp @@ -0,0 +1,103 @@ +#include "StandardIncludes.h" +#include "SLTerrain.h" +#include "SettingsMan.h" +#include "LuaMan.h" +#include "ThreadMan.h" +#include +using namespace RTE; + +struct SceneAccess : SceneMan { + static Scene* SceneMan::* currentScene() { return &SceneAccess::m_pCurrentScene; } +}; +struct TerrainFixture : SLTerrain { + TerrainFixture() { m_WrapX = false; m_WrapY = false; } +}; +struct DebrisFixture : MOSRotating { + DebrisFixture(BITMAP* bitmap, int i) { + m_aSprite = {bitmap}; + m_SpriteOffset = Vector(-31.25F, -30.75F); + m_SpriteRadius = 46.0F; + m_Pos = Vector(512.25F + (i % 16) * 0.5F, 256.75F + (i / 16) * 0.5F); + m_HFlipped = i % 2; + setRotation(i, true); + } + void setRotation(int tick, bool warm) { + m_Rotation.SetRadAngle(static_cast((tick % 257) - 128) * 0.0317F); + m_Rotation.SetXFlipped(tick % 3 == 0); + m_Rotation.SetYFlipped(tick % 5 == 0); + if (warm) { auto offset = m_Rotation * Vector(1, 2); (void)offset; } + } + const Matrix& rotation() const { return m_Rotation; } +}; +uint64_t hashValue(uint64_t hash, uint64_t v) { return (hash ^ v) * 1099511628211ULL; } +int main() { + install_allegro(SYSTEM_NONE, &errno, atexit); + ThreadMan::Construct(); + TimerMan::Construct(); + SettingsMan::Construct(); + SceneMan::Construct(); + MovableMan::Construct(); + LuaMan::Construct(); + Scene scene; + scene.Create(new TerrainFixture()); + g_SceneMan.*SceneAccess::currentScene() = &scene; + BITMAP* bitmap = create_bitmap_ex(8, 64, 64); + for (int y = 0; y < 64; ++y) for (int x = 0; x < 64; ++x) + _putpixel(bitmap, x, y, ((x * 17 + y * 13) % 7 == 0 || x < y / 3) ? ColorKeys::g_MaskColor : 21); + std::vector> objects; + for (int i = 0; i < 512; ++i) objects.push_back(std::make_unique(bitmap, i)); + uint64_t hash = 1469598103934665603ULL; + for (int phase = 0; phase < 4; ++phase) { + for (int i = 0; i < 512; ++i) { + objects[i]->setRotation(i + phase * 87, phase % 2); + Matrix before(objects[i]->rotation()); + for (int y = 203; y < 323; ++y) for (int x = 459; x < 579; ++x) + hash = hashValue(hash, objects[i]->HitTestAtPixel(x, y, false)); + const Matrix& after = objects[i]->rotation(); + if (before.m_Rotation != after.m_Rotation || before.m_Flipped[0] != after.m_Flipped[0] || before.m_Flipped[1] != after.m_Flipped[1]) return 2; + // Check source cache state without taking a Matrix copy (baseline copies invalidate it). + if (after.m_ElementsUpdated != static_cast(phase % 2)) return 3; + } + } + std::cout << "behavior_hash=" << hash << " checks=" << 4ULL * 512 * 120 * 120 << '\n'; + // Copies and vector/matrix const operators for dirty and clean matrices. + uint64_t matrixHash = 1469598103934665603ULL; + for (int i = -512; i < 512; ++i) for (int flips = 0; flips < 4; ++flips) for (int warm = 0; warm < 2; ++warm) { + Matrix source; source.SetRadAngle(i * 0.0127F); + source.SetXFlipped(flips & 1); source.SetYFlipped(flips & 2); + if (warm) { source * Vector(1, 2); } + Matrix copied(source); + for (int j = -8; j < 8; ++j) { + Vector input(j * 0.37F, j * -0.73F); + Vector forward = input * copied, backward = input / copied; + for (float f : {forward.m_X, forward.m_Y, backward.m_X, backward.m_Y}) { + uint32_t bits; std::memcpy(&bits, &f, sizeof(bits)); matrixHash = hashValue(matrixHash, bits); + } + } + if (source.m_ElementsUpdated != static_cast(warm)) return 4; + } + std::cout << "matrix_hash=" << matrixHash << '\n'; + for (bool warm : {false, true}) { + for (int i = 0; i < 512; ++i) objects[i]->setRotation(i, warm); + uint64_t hits = 0; + for (int sample = -1; sample < 7; ++sample) { + auto start = std::chrono::steady_clock::now(); + for (int repeat = 0; repeat < 128; ++repeat) for (int q = 0; q < 128; ++q) for (const auto& object : objects) + hits += object->HitTestAtPixel(496 + q % 48, 240 + q / 4, false); + double elapsed = std::chrono::duration(std::chrono::steady_clock::now() - start).count(); + if (sample >= 0) std::cout << "warm=" << warm << " sample=" << sample << " ms=" << std::fixed << std::setprecision(3) << elapsed << " queries=" << 128ULL * 128 * 512 << " hits=" << hits << '\n'; + } + } + std::vector workers; + std::array threadHits{}; + for (int t = 0; t < 4; ++t) workers.emplace_back([&, t] { + for (int repeat = 0; repeat < 32; ++repeat) for (const auto& object : objects) for (int q = 0; q < 128; ++q) + threadHits[t] += object->HitTestAtPixel(496 + q % 48, 240 + q / 4, false); + }); + for (auto& thread : workers) thread.join(); + if (!std::all_of(threadHits.begin(), threadHits.end(), [&](uint64_t v) { return v == threadHits[0]; })) return 5; + std::cout << "concurrent_hits=" << threadHits[0] << " per_thread_queries=" << 32ULL * 512 * 128 << '\n'; + objects.clear(); + destroy_bitmap(bitmap); + g_SceneMan.*SceneAccess::currentScene() = nullptr; +} diff --git a/Tools/Benchmarks/CollisionTransforms.py b/Tools/Benchmarks/CollisionTransforms.py new file mode 100644 index 0000000000..8a8d891256 --- /dev/null +++ b/Tools/Benchmarks/CollisionTransforms.py @@ -0,0 +1,29 @@ +#!/usr/bin/env python3 +import argparse +import json +import pathlib +import shlex +import subprocess + +parser = argparse.ArgumentParser(description='Build the collision benchmark with existing engine objects.') +parser.add_argument('build_directory', type=pathlib.Path) +parser.add_argument('output', type=pathlib.Path) +args = parser.parse_args() +build = args.build_directory.resolve() +output = args.output.resolve() +source = pathlib.Path(__file__).with_suffix('.cpp').resolve() +commands = json.loads((build / 'compile_commands.json').read_text()) +command = shlex.split(next(x['command'] for x in commands if x['file'].endswith('MOSprite.cpp'))) +command = command[:command.index('-MD')] +# The benchmark includes StandardIncludes.h itself. Do not use the generated PCH. +command = [arg for arg in command if arg not in {'-ICortexCommand.p', '-fpch-preprocess'}] +index = command.index('-include') +del command[index:index + 2] +object_file = str(output) + '.o' +command += ['-c', str(source), '-o', object_file] +subprocess.run(command, cwd=build, check=True) +link = subprocess.check_output(['ninja', '-C', str(build), '-t', 'commands', 'CortexCommand'], text=True).splitlines()[-1] +command = shlex.split(link) +command[command.index('-o') + 1] = str(output) +command[command.index('CortexCommand.p/Source_Main.cpp.o')] = object_file +subprocess.run(command, cwd=build, check=True) diff --git a/Tools/Benchmarks/README.md b/Tools/Benchmarks/README.md new file mode 100644 index 0000000000..ea440a3786 --- /dev/null +++ b/Tools/Benchmarks/README.md @@ -0,0 +1,38 @@ +The collision benchmark calls the production `MOSprite::HitTestAtPixel` function through 512 overlapping debris fixtures. +Each fixture uses a 64×64 bitmap with opaque and transparent pixels. +The benchmark measures rotations with valid cached coefficients and rotations with invalidated coefficients separately. +Each case includes seven samples after one warmup sample. Each sample contains 8,388,608 queries. + +The behavior checks cover 29,491,200 queries with different rotations, flips, and fractional positions. +The benchmark also prints a hash of the float bits from the production Matrix operators. +Four concurrent readers run the same queries and compare hit counts. +The benchmark makes sure that the source matrices retain their cache state. +An error returns a nonzero exit code. Baseline and changed builds must produce the same hashes and hit counts. + +Build the engine with Meson and Ninja before you compile the benchmark. +The runner uses GCC-compatible compile commands and the existing engine objects. +Run the following command from the repository root: + +```sh +python3 Tools/Benchmarks/CollisionTransforms.py build /tmp/collision-transforms +``` + +On macOS, run the benchmark with the engine library directory: + +```sh +DYLD_LIBRARY_PATH="$PWD/external/lib/macos" /tmp/collision-transforms > /tmp/collision-transforms.txt +``` + +On Linux, run the benchmark with the engine library directory: + +```sh +LD_LIBRARY_PATH="$PWD/external/lib/linux/x86_64" /tmp/collision-transforms > /tmp/collision-transforms.txt +``` + +For a comparison, build the benchmark from each revision with the same release compiler settings. +Repeat each executable three times. Alternate the order of the executables. +Compare the median sample times for each cache state. + +This benchmark measures production collision queries with controlled fixtures. +It does not measure gameplay frame times or the complete physics simulation. +The fixtures use a synthetic bitmap and a scene without terrain wrapping.