diff --git a/.github/workflows/roundtrip.yml b/.github/workflows/roundtrip.yml index 232e314..2f63f82 100644 --- a/.github/workflows/roundtrip.yml +++ b/.github/workflows/roundtrip.yml @@ -7,6 +7,10 @@ on: permissions: contents: read +concurrency: + group: embedded-roundtrip-${{ github.event_name }}-${{ github.ref }} + cancel-in-progress: true + jobs: macos-roundtrip: name: macOS embedded Racket round trip @@ -88,6 +92,10 @@ jobs: - name: Exercise embedded runtime timeout-minutes: 2 run: "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" + - name: Record embedded benchmark (informational) + timeout-minutes: 2 + run: | + "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" --benchmark windows-roundtrip: name: Windows embedded Racket round trip @@ -161,3 +169,7 @@ jobs: timeout-minutes: 2 shell: pwsh run: '& (Join-Path $env:RIVET_ROUNDTRIP_STAGE "RivetIntegration.exe")' + - name: Record embedded benchmark (informational) + timeout-minutes: 2 + shell: pwsh + run: '& (Join-Path $env:RIVET_ROUNDTRIP_STAGE "RivetIntegration.exe") --benchmark' diff --git a/docs/benchmarks.md b/docs/benchmarks.md new file mode 100644 index 0000000..af49e12 --- /dev/null +++ b/docs/benchmarks.md @@ -0,0 +1,72 @@ +# Embedded performance benchmarks + +Rivet's first performance baseline measures the same native integration hosts used by the Windows and macOS embedded round-trip tests. The benchmark therefore includes the real in-process Racket CS runtime, RVT1 transport, native client, request lifecycle, and State synchronization path instead of measuring a detached codec microbenchmark. + +## What is measured + +The integration runner accepts an optional `--benchmark` flag. Benchmark mode reports one JSON object with schema version 1 and these metrics: + +- `startup_ms`: elapsed time for `Backend::start()` / `EmbeddedRacketBackend.start()`, including Racket CS startup and the RVT1 Hello handshake; +- `rpc`: 1,000 sequential `increment(Int64) -> Int64` round trips after 50 warm-up calls; +- `state_get`: 1,000 sequential `$state/get` round trips; +- `state_set`: 500 sequential `$state/set` round trips, including the reserved `$state` Event that precedes each successful response. + +The report also records `platform`, `architecture`, and `configuration` so results from different execution contexts are not accidentally compared as one series. Integration benchmarks are built in release mode. + +Each operation group includes `iterations`, `total_ms`, and `us_per_operation`. Every timed operation still validates its result, so a fast but incorrect run fails instead of producing a number. + +Example shape: + +```json +{ + "schema_version": 1, + "platform": "macos", + "architecture": "arm64", + "configuration": "release", + "startup_ms": 42.5, + "warmup_iterations": 50, + "rpc": { + "iterations": 1000, + "total_ms": 120.0, + "us_per_operation": 120.0 + }, + "state_get": { + "iterations": 1000, + "total_ms": 130.0, + "us_per_operation": 130.0 + }, + "state_set": { + "iterations": 500, + "total_ms": 90.0, + "us_per_operation": 180.0 + } +} +``` + +The numbers above illustrate the JSON schema only; they are not Rivet performance claims or release thresholds. + +## CI behavior + +Every `Embedded Roundtrip` job first runs the existing correctness integration path, then starts a fresh integration process with `--benchmark`. The second step records one JSON report for that runner and revision. Starting a fresh process is important because the embedded Racket runtime is intentionally one-shot within a process and startup is itself one of the measured costs. + +The benchmark harness is continuously tested: invalid results, startup failures, or a broken benchmark implementation fail the job. **The measured timing values themselves do not have pass/fail thresholds.** Shared GitHub-hosted runner timing variance therefore cannot reject a pull request merely for being slower in one sample. + +The same workflow can also be launched manually from **Actions → Embedded Roundtrip → Run workflow** for an explicit revision. Read the JSON from each platform's `Record embedded benchmark (informational)` step. + +## Comparing results + +Do not treat absolute timings from shared GitHub-hosted runners as hard pass/fail limits. Host load, virtualization, CPU generation, runner image, compiler version, and Racket version can move results independently of Rivet code. + +For meaningful regression work: + +- compare the same platform, architecture, and build configuration; +- use the same Racket, compiler, and runner image where possible; +- run several samples and compare distributions or medians rather than one measurement; +- use a dedicated or self-hosted machine before establishing release thresholds; +- keep `schema_version` in captured reports so future benchmark changes remain distinguishable. + +The CI reports are intended as a convenient baseline and investigation tool. Once enough history exists on stable hardware, selected metrics can be promoted to guarded regression thresholds without redesigning the integration harness. + +## Scope + +This first baseline focuses on runtime startup and request/State latency. Event-only throughput, concurrent RPC throughput, memory/RSS, and packaged application size are intentionally separate follow-ups because they need different sampling and interpretation. Keeping them separate avoids turning one benchmark number into an ambiguous mixture of unrelated costs. diff --git a/platform/macos/Integration/Sources/main.swift b/platform/macos/Integration/Sources/main.swift index f69b4a4..6c11f97 100644 --- a/platform/macos/Integration/Sources/main.swift +++ b/platform/macos/Integration/Sources/main.swift @@ -1,10 +1,125 @@ +import Dispatch import Foundation import RivetEmbedding import RivetRuntime +#if arch(arm64) +private let benchmarkArchitecture = "arm64" +#elseif arch(x86_64) +private let benchmarkArchitecture = "x64" +#else +private let benchmarkArchitecture = "unknown" +#endif + +private struct BenchmarkMetric { + let iterations: Int + let totalMilliseconds: Double + + var microsecondsPerOperation: Double { + totalMilliseconds * 1000.0 / Double(iterations) + } + + var json: [String: Any] { + [ + "iterations": iterations, + "total_ms": totalMilliseconds, + "us_per_operation": microsecondsPerOperation + ] + } +} + +private func elapsedMilliseconds(from start: UInt64, to end: UInt64) -> Double { + Double(end - start) / 1_000_000.0 +} + +private func benchmarkOperations( + iterations: Int, + operation: (Int) async throws -> Void +) async rethrows -> BenchmarkMetric { + let start = DispatchTime.now().uptimeNanoseconds + for index in 0.. Bool { + let arguments = Array(CommandLine.arguments.dropFirst()) + if arguments.isEmpty { return false } + if arguments == ["--benchmark"] { return true } + throw IntegrationError.invalidArguments(arguments) +} + +private func runBenchmark( + backend: EmbeddedRacketBackend, + startupMilliseconds: Double +) async throws { + let warmupIterations = 50 + let rpcIterations = 1000 + let stateGetIterations = 1000 + let stateSetIterations = 500 + + for _ in 0.. URL { guard let value = environment[key], !value.isEmpty else { @@ -25,7 +140,19 @@ struct RivetIntegration { ) ) + let startupStart = DispatchTime.now().uptimeNanoseconds try backend.start() + let startupEnd = DispatchTime.now().uptimeNanoseconds + let startupMilliseconds = elapsedMilliseconds(from: startupStart, to: startupEnd) + + if benchmark { + try await runBenchmark( + backend: backend, + startupMilliseconds: startupMilliseconds + ) + backend.stop() + return + } let incremented = try await backend.client.call( "increment", @@ -71,6 +198,8 @@ enum IntegrationError: Error, CustomStringConvertible { case missingEnvironment(String) case unexpected(String, RivetValue) case restartWasAllowed + case invalidArguments([String]) + case benchmarkEncoding var description: String { switch self { @@ -80,6 +209,10 @@ enum IntegrationError: Error, CustomStringConvertible { return "unexpected \(operation) result: \(value)" case .restartWasAllowed: return "embedded Racket backend unexpectedly allowed restart after stop" + case .invalidArguments(let arguments): + return "usage: RivetIntegration [--benchmark]; received: \(arguments)" + case .benchmarkEncoding: + return "failed to encode benchmark report as UTF-8 JSON" } } } diff --git a/platform/windows/Integration/main.cpp b/platform/windows/Integration/main.cpp index c64d98e..769d1db 100644 --- a/platform/windows/Integration/main.cpp +++ b/platform/windows/Integration/main.cpp @@ -3,9 +3,11 @@ #endif #include +#include #include #include #include +#include #include #include #include @@ -15,6 +17,24 @@ namespace { +using BenchmarkClock = std::chrono::steady_clock; + +struct BenchmarkMetric { + int iterations{}; + double total_ms{}; + double us_per_operation{}; +}; + +char const* benchmark_architecture() noexcept { +#if defined(_M_ARM64) || defined(__aarch64__) + return "arm64"; +#elif defined(_M_X64) || defined(__x86_64__) + return "x64"; +#else + return "unknown"; +#endif +} + std::filesystem::path executable_path() { std::wstring buffer(32768, L'\0'); auto const length = ::GetModuleFileNameW( @@ -56,10 +76,101 @@ void progress(char const* message) { std::cerr << "[rivet-integration] " << message << "\n" << std::flush; } +double elapsed_ms(BenchmarkClock::time_point begin, + BenchmarkClock::time_point end) { + return std::chrono::duration(end - begin).count(); +} + +template +BenchmarkMetric benchmark_operations(int iterations, Operation&& operation) { + auto const begin = BenchmarkClock::now(); + for (int i = 0; i < iterations; ++i) { + operation(i); + } + auto const total_ms = elapsed_ms(begin, BenchmarkClock::now()); + return BenchmarkMetric{ + iterations, + total_ms, + (total_ms * 1000.0) / static_cast(iterations), + }; +} + +void print_metric_json(char const* name, + BenchmarkMetric const& metric, + bool trailing_comma) { + std::cout << "\"" << name << "\":{\"iterations\":" << metric.iterations + << ",\"total_ms\":" << metric.total_ms + << ",\"us_per_operation\":" << metric.us_per_operation << "}"; + if (trailing_comma) { + std::cout << ","; + } +} + +void run_benchmark(rivet::windows::Backend& backend, double startup_ms) { + constexpr int warmup_iterations = 50; + constexpr int rpc_iterations = 1000; + constexpr int state_get_iterations = 1000; + constexpr int state_set_iterations = 500; + + for (int i = 0; i < warmup_iterations; ++i) { + auto result = backend.call( + "increment", rivet::Value::List{rivet::Value(std::int64_t{41})}); + if (expect_int(result.get(), "benchmark warmup") != 42) { + throw std::runtime_error("benchmark warmup returned an unexpected value"); + } + } + + auto const rpc = benchmark_operations(rpc_iterations, [&](int) { + auto result = backend.call( + "increment", rivet::Value::List{rivet::Value(std::int64_t{41})}); + if (expect_int(result.get(), "benchmark RPC") != 42) { + throw std::runtime_error("benchmark RPC returned an unexpected value"); + } + }); + + auto const state_get = benchmark_operations(state_get_iterations, [&](int) { + auto result = backend.get_state("counter"); + auto const value = expect_int(result.get(), "benchmark state get"); + if (value != 10) { + throw std::runtime_error("benchmark state get returned an unexpected value"); + } + }); + + auto const state_set = benchmark_operations(state_set_iterations, [&](int i) { + auto const expected = std::int64_t{10 + (i & 1)}; + auto result = backend.set_state("counter", rivet::Value(expected)); + if (expect_int(result.get(), "benchmark state set") != expected) { + throw std::runtime_error("benchmark state set returned an unexpected value"); + } + }); + + std::cout << std::fixed << std::setprecision(3); + std::cout << "{\"schema_version\":1,\"platform\":\"windows\"," + << "\"architecture\":\"" << benchmark_architecture() << "\"," + << "\"configuration\":\"release\"," + << "\"startup_ms\":" << startup_ms << "," + << "\"warmup_iterations\":" << warmup_iterations << ","; + print_metric_json("rpc", rpc, true); + print_metric_json("state_get", state_get, true); + print_metric_json("state_set", state_set, false); + std::cout << "}\n"; +} + +bool benchmark_mode(int argc, char** argv) { + if (argc == 1) { + return false; + } + if (argc == 2 && std::string(argv[1]) == "--benchmark") { + return true; + } + throw std::runtime_error("usage: RivetIntegration [--benchmark]"); +} + } // namespace -int main() { +int main(int argc, char** argv) { try { + auto const benchmark = benchmark_mode(argc, argv); auto const exe = executable_path(); auto const root = exe.parent_path(); auto const runtime = root / L"runtime"; @@ -77,9 +188,17 @@ int main() { progress("starting backend"); rivet::windows::Backend backend(std::move(config)); + auto const startup_begin = BenchmarkClock::now(); backend.start(); + auto const startup_ms = elapsed_ms(startup_begin, BenchmarkClock::now()); progress("backend started"); + if (benchmark) { + run_benchmark(backend, startup_ms); + backend.stop(); + return 0; + } + progress("calling increment"); auto increment = backend.call( "increment", rivet::Value::List{rivet::Value(std::int64_t{41})});