From 7f715011c5fef8230008baf42c41a3a4ce953392 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:15:39 +0800 Subject: [PATCH 01/12] perf: add Windows embedded benchmark mode --- platform/windows/Integration/main.cpp | 109 +++++++++++++++++++++++++- 1 file changed, 108 insertions(+), 1 deletion(-) diff --git a/platform/windows/Integration/main.cpp b/platform/windows/Integration/main.cpp index 9eb527b..2c7ba73 100644 --- a/platform/windows/Integration/main.cpp +++ b/platform/windows/Integration/main.cpp @@ -3,9 +3,11 @@ #endif #include +#include #include #include #include +#include #include #include #include @@ -15,6 +17,14 @@ namespace { +using BenchmarkClock = std::chrono::steady_clock; + +struct BenchmarkMetric { + int iterations{}; + double total_ms{}; + double us_per_operation{}; +}; + std::filesystem::path executable_path() { std::wstring buffer(32768, L'\0'); auto const length = ::GetModuleFileNameW( @@ -56,10 +66,99 @@ void progress(char const* message) { std::cerr << "[rivet-integration] " << message << "\n" << std::flush; } +double elapsed_ms(BenchmarkClock::time_point begin, + BenchmarkClock::time_point end) { + return std::chrono::duration(end - begin).count(); +} + +template +BenchmarkMetric benchmark_operations(int iterations, Operation&& operation) { + auto const begin = BenchmarkClock::now(); + for (int i = 0; i < iterations; ++i) { + operation(i); + } + auto const total_ms = elapsed_ms(begin, BenchmarkClock::now()); + return BenchmarkMetric{ + iterations, + total_ms, + (total_ms * 1000.0) / static_cast(iterations), + }; +} + +void print_metric_json(char const* name, + BenchmarkMetric const& metric, + bool trailing_comma) { + std::cout << "\"" << name << "\":{"iterations":" << metric.iterations + << ",\"total_ms\":" << metric.total_ms + << ",\"us_per_operation\":" << metric.us_per_operation << "}"; + if (trailing_comma) { + std::cout << ","; + } +} + +void run_benchmark(rivet::windows::Backend& backend, double startup_ms) { + constexpr int warmup_iterations = 50; + constexpr int rpc_iterations = 1000; + constexpr int state_get_iterations = 1000; + constexpr int state_set_iterations = 500; + + for (int i = 0; i < warmup_iterations; ++i) { + auto result = backend.call( + "increment", rivet::Value::List{rivet::Value(std::int64_t{41})}); + if (expect_int(result.get(), "benchmark warmup") != 42) { + throw std::runtime_error("benchmark warmup returned an unexpected value"); + } + } + + auto const rpc = benchmark_operations(rpc_iterations, [&](int) { + auto result = backend.call( + "increment", rivet::Value::List{rivet::Value(std::int64_t{41})}); + if (expect_int(result.get(), "benchmark RPC") != 42) { + throw std::runtime_error("benchmark RPC returned an unexpected value"); + } + }); + + auto const state_get = benchmark_operations(state_get_iterations, [&](int) { + auto result = backend.get_state("counter"); + auto const value = expect_int(result.get(), "benchmark state get"); + if (value != 10) { + throw std::runtime_error("benchmark state get returned an unexpected value"); + } + }); + + auto const state_set = benchmark_operations(state_set_iterations, [&](int i) { + auto const expected = std::int64_t{10 + (i & 1)}; + auto result = backend.set_state("counter", rivet::Value(expected)); + if (expect_int(result.get(), "benchmark state set") != expected) { + throw std::runtime_error("benchmark state set returned an unexpected value"); + } + }); + + std::cout << std::fixed << std::setprecision(3); + std::cout << "{\"schema_version\":1,\"platform\":\"windows\"," + << "\"startup_ms\":" << startup_ms << "," + << "\"warmup_iterations\":" << warmup_iterations << ","; + print_metric_json("rpc", rpc, true); + print_metric_json("state_get", state_get, true); + print_metric_json("state_set", state_set, false); + std::cout << "}\n"; +} + +bool benchmark_mode(int argc, char** argv) { + if (argc == 1) { + return false; + } + if (argc == 2 && std::string(argv[1]) == "--benchmark") { + return true; + } + throw std::runtime_error("usage: RivetIntegration [--benchmark]"); +} + } // namespace -int main() { +int main(int argc, char** argv) { try { + auto const benchmark = benchmark_mode(argc, argv); auto const exe = executable_path(); auto const root = exe.parent_path(); auto const runtime = root / L"runtime"; @@ -76,9 +175,17 @@ int main() { progress("starting backend"); rivet::windows::Backend backend(std::move(config)); + auto const startup_begin = BenchmarkClock::now(); backend.start(); + auto const startup_ms = elapsed_ms(startup_begin, BenchmarkClock::now()); progress("backend started"); + if (benchmark) { + run_benchmark(backend, startup_ms); + backend.stop(); + return 0; + } + progress("calling increment"); auto increment = backend.call( "increment", rivet::Value::List{rivet::Value(std::int64_t{41})}); From 5676c1709029dabde01792b624be3bc2ab26f270 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:16:15 +0800 Subject: [PATCH 02/12] perf: add macOS embedded benchmark mode --- platform/macos/Integration/Sources/main.swift | 123 ++++++++++++++++++ 1 file changed, 123 insertions(+) diff --git a/platform/macos/Integration/Sources/main.swift b/platform/macos/Integration/Sources/main.swift index f69b4a4..e73d0af 100644 --- a/platform/macos/Integration/Sources/main.swift +++ b/platform/macos/Integration/Sources/main.swift @@ -1,10 +1,115 @@ +import Dispatch import Foundation import RivetEmbedding import RivetRuntime +private struct BenchmarkMetric { + let iterations: Int + let totalMilliseconds: Double + + var microsecondsPerOperation: Double { + totalMilliseconds * 1000.0 / Double(iterations) + } + + var json: [String: Any] { + [ + "iterations": iterations, + "total_ms": totalMilliseconds, + "us_per_operation": microsecondsPerOperation + ] + } +} + +private func elapsedMilliseconds(from start: UInt64, to end: UInt64) -> Double { + Double(end - start) / 1_000_000.0 +} + +private func benchmarkOperations( + iterations: Int, + operation: (Int) async throws -> Void +) async rethrows -> BenchmarkMetric { + let start = DispatchTime.now().uptimeNanoseconds + for index in 0.. Bool { + let arguments = Array(CommandLine.arguments.dropFirst()) + if arguments.isEmpty { return false } + if arguments == ["--benchmark"] { return true } + throw IntegrationError.invalidArguments(arguments) +} + +private func runBenchmark( + backend: EmbeddedRacketBackend, + startupMilliseconds: Double +) async throws { + let warmupIterations = 50 + let rpcIterations = 1000 + let stateGetIterations = 1000 + let stateSetIterations = 500 + + for _ in 0.. URL { guard let value = environment[key], !value.isEmpty else { @@ -25,7 +130,19 @@ struct RivetIntegration { ) ) + let startupStart = DispatchTime.now().uptimeNanoseconds try backend.start() + let startupEnd = DispatchTime.now().uptimeNanoseconds + let startupMilliseconds = elapsedMilliseconds(from: startupStart, to: startupEnd) + + if benchmark { + try await runBenchmark( + backend: backend, + startupMilliseconds: startupMilliseconds + ) + backend.stop() + return + } let incremented = try await backend.client.call( "increment", @@ -71,6 +188,8 @@ enum IntegrationError: Error, CustomStringConvertible { case missingEnvironment(String) case unexpected(String, RivetValue) case restartWasAllowed + case invalidArguments([String]) + case benchmarkEncoding var description: String { switch self { @@ -80,6 +199,10 @@ enum IntegrationError: Error, CustomStringConvertible { return "unexpected \(operation) result: \(value)" case .restartWasAllowed: return "embedded Racket backend unexpectedly allowed restart after stop" + case .invalidArguments(let arguments): + return "usage: RivetIntegration [--benchmark]; received: \(arguments)" + case .benchmarkEncoding: + return "failed to encode benchmark report as UTF-8 JSON" } } } From 6b308d310ca914419a23b030540a2b7083ac0d96 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:17:02 +0800 Subject: [PATCH 03/12] perf: expose manual embedded benchmark mode --- .github/workflows/roundtrip.yml | 28 ++++++++++++++++++++++++++-- 1 file changed, 26 insertions(+), 2 deletions(-) diff --git a/.github/workflows/roundtrip.yml b/.github/workflows/roundtrip.yml index 232e314..4354961 100644 --- a/.github/workflows/roundtrip.yml +++ b/.github/workflows/roundtrip.yml @@ -3,6 +3,12 @@ name: Embedded Roundtrip on: pull_request: workflow_dispatch: + inputs: + benchmark: + description: Run the real embedded integration hosts in benchmark mode + required: false + default: false + type: boolean permissions: contents: read @@ -87,7 +93,16 @@ jobs: echo "RIVET_TEST_RACKET_BOOT=$stage/runtime/racket.boot" >> "$GITHUB_ENV" - name: Exercise embedded runtime timeout-minutes: 2 - run: "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" + shell: bash + env: + RIVET_BENCHMARK: ${{ github.event_name == 'workflow_dispatch' && inputs.benchmark && '1' || '0' }} + run: | + set -euo pipefail + if [[ "$RIVET_BENCHMARK" == "1" ]]; then + "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" --benchmark + else + "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" + fi windows-roundtrip: name: Windows embedded Racket round trip @@ -160,4 +175,13 @@ jobs: - name: Exercise embedded runtime timeout-minutes: 2 shell: pwsh - run: '& (Join-Path $env:RIVET_ROUNDTRIP_STAGE "RivetIntegration.exe")' + env: + RIVET_BENCHMARK: ${{ github.event_name == 'workflow_dispatch' && inputs.benchmark && '1' || '0' }} + run: | + $exe = Join-Path $env:RIVET_ROUNDTRIP_STAGE 'RivetIntegration.exe' + if ($env:RIVET_BENCHMARK -eq '1') { + & $exe --benchmark + } else { + & $exe + } + if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } From e8dcf07c63ea8dcfbf98766132589f6dfe35dad6 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:17:37 +0800 Subject: [PATCH 04/12] docs: define embedded benchmark baseline --- docs/benchmarks.md | 71 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 71 insertions(+) create mode 100644 docs/benchmarks.md diff --git a/docs/benchmarks.md b/docs/benchmarks.md new file mode 100644 index 0000000..948ebe5 --- /dev/null +++ b/docs/benchmarks.md @@ -0,0 +1,71 @@ +# Embedded performance benchmarks + +Rivet's first performance baseline measures the same native integration hosts used by the Windows and macOS embedded round-trip tests. The benchmark therefore includes the real in-process Racket CS runtime, RVT1 transport, native client, request lifecycle, and State synchronization path instead of measuring a detached codec microbenchmark. + +## What is measured + +The integration runner accepts an optional `--benchmark` flag. Benchmark mode reports one JSON object with schema version 1 and these metrics: + +- `startup_ms`: elapsed time for `Backend::start()` / `EmbeddedRacketBackend.start()`, including Racket CS startup and the RVT1 Hello handshake; +- `rpc`: 1,000 sequential `increment(Int64) -> Int64` round trips after 50 warm-up calls; +- `state_get`: 1,000 sequential `$state/get` round trips; +- `state_set`: 500 sequential `$state/set` round trips, including the reserved `$state` Event that precedes each successful response. + +Each operation group includes `iterations`, `total_ms`, and `us_per_operation`. Every timed operation still validates its result, so a fast but incorrect run fails instead of producing a number. + +Example shape: + +```json +{ + "schema_version": 1, + "platform": "macos", + "startup_ms": 42.5, + "warmup_iterations": 50, + "rpc": { + "iterations": 1000, + "total_ms": 120.0, + "us_per_operation": 120.0 + }, + "state_get": { + "iterations": 1000, + "total_ms": 130.0, + "us_per_operation": 130.0 + }, + "state_set": { + "iterations": 500, + "total_ms": 90.0, + "us_per_operation": 180.0 + } +} +``` + +The numbers above illustrate the JSON schema only; they are not Rivet performance claims or release thresholds. + +## Run on GitHub Actions + +The existing `Embedded Roundtrip` workflow supports manual benchmark runs without changing pull-request behavior: + +1. Open **Actions → Embedded Roundtrip → Run workflow**. +2. Select the revision to measure. +3. Enable **Run the real embedded integration hosts in benchmark mode**. +4. Read the JSON line from each platform's `Exercise embedded runtime` step. + +Normal pull requests leave benchmark mode disabled and continue to run correctness-only round trips. + +## Comparing results + +Do not treat absolute timings from shared GitHub-hosted runners as hard pass/fail limits. Host load, virtualization, CPU generation, runner image, compiler version, and Racket version can move results independently of Rivet code. + +For meaningful regression work: + +- compare the same platform and architecture; +- use the same Racket, compiler, and runner image where possible; +- run several samples and compare distributions or medians rather than one measurement; +- use a dedicated or self-hosted machine before establishing release thresholds; +- keep `schema_version` in captured reports so future benchmark changes remain distinguishable. + +The manual Actions mode is intended as a convenient baseline and investigation tool. Once enough history exists on stable hardware, selected metrics can be promoted to guarded regression thresholds without redesigning the integration harness. + +## Scope + +This first baseline focuses on runtime startup and request/State latency. Event-only throughput, concurrent RPC throughput, memory/RSS, and packaged application size are intentionally separate follow-ups because they need different sampling and interpretation. Keeping them separate avoids turning one benchmark number into an ambiguous mixture of unrelated costs. From 23e9423cb1531358762b6aeb9197dc0602015c30 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:20:28 +0800 Subject: [PATCH 05/12] perf: exercise benchmark mode on every PR --- .github/workflows/roundtrip.yml | 35 +++++++++------------------------ 1 file changed, 9 insertions(+), 26 deletions(-) diff --git a/.github/workflows/roundtrip.yml b/.github/workflows/roundtrip.yml index 4354961..ad32360 100644 --- a/.github/workflows/roundtrip.yml +++ b/.github/workflows/roundtrip.yml @@ -3,12 +3,6 @@ name: Embedded Roundtrip on: pull_request: workflow_dispatch: - inputs: - benchmark: - description: Run the real embedded integration hosts in benchmark mode - required: false - default: false - type: boolean permissions: contents: read @@ -93,16 +87,10 @@ jobs: echo "RIVET_TEST_RACKET_BOOT=$stage/runtime/racket.boot" >> "$GITHUB_ENV" - name: Exercise embedded runtime timeout-minutes: 2 - shell: bash - env: - RIVET_BENCHMARK: ${{ github.event_name == 'workflow_dispatch' && inputs.benchmark && '1' || '0' }} - run: | - set -euo pipefail - if [[ "$RIVET_BENCHMARK" == "1" ]]; then - "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" --benchmark - else - "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" - fi + run: "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" + - name: Record embedded benchmark (informational) + timeout-minutes: 2 + run: "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" --benchmark windows-roundtrip: name: Windows embedded Racket round trip @@ -175,13 +163,8 @@ jobs: - name: Exercise embedded runtime timeout-minutes: 2 shell: pwsh - env: - RIVET_BENCHMARK: ${{ github.event_name == 'workflow_dispatch' && inputs.benchmark && '1' || '0' }} - run: | - $exe = Join-Path $env:RIVET_ROUNDTRIP_STAGE 'RivetIntegration.exe' - if ($env:RIVET_BENCHMARK -eq '1') { - & $exe --benchmark - } else { - & $exe - } - if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE } + run: '& (Join-Path $env:RIVET_ROUNDTRIP_STAGE "RivetIntegration.exe")' + - name: Record embedded benchmark (informational) + timeout-minutes: 2 + shell: pwsh + run: '& (Join-Path $env:RIVET_ROUNDTRIP_STAGE "RivetIntegration.exe") --benchmark' From dc4cb1b84401f7ad1c0d4e515f7a068cc6dfed0a Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:21:00 +0800 Subject: [PATCH 06/12] docs: explain benchmark CI behavior --- docs/benchmarks.md | 13 +++++-------- 1 file changed, 5 insertions(+), 8 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 948ebe5..c3c0d7c 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -41,16 +41,13 @@ Example shape: The numbers above illustrate the JSON schema only; they are not Rivet performance claims or release thresholds. -## Run on GitHub Actions +## CI behavior -The existing `Embedded Roundtrip` workflow supports manual benchmark runs without changing pull-request behavior: +Every `Embedded Roundtrip` job first runs the existing correctness integration path, then starts a fresh integration process with `--benchmark`. The second step records one JSON report for that runner and revision. Starting a fresh process is important because the embedded Racket runtime is intentionally one-shot within a process and startup is itself one of the measured costs. -1. Open **Actions → Embedded Roundtrip → Run workflow**. -2. Select the revision to measure. -3. Enable **Run the real embedded integration hosts in benchmark mode**. -4. Read the JSON line from each platform's `Exercise embedded runtime` step. +The benchmark harness is continuously tested: invalid results, startup failures, or a broken benchmark implementation fail the job. **The measured timing values themselves do not have pass/fail thresholds.** Shared GitHub-hosted runner timing variance therefore cannot reject a pull request merely for being slower in one sample. -Normal pull requests leave benchmark mode disabled and continue to run correctness-only round trips. +The same workflow can also be launched manually from **Actions → Embedded Roundtrip → Run workflow** for an explicit revision. Read the JSON from each platform's `Record embedded benchmark (informational)` step. ## Comparing results @@ -64,7 +61,7 @@ For meaningful regression work: - use a dedicated or self-hosted machine before establishing release thresholds; - keep `schema_version` in captured reports so future benchmark changes remain distinguishable. -The manual Actions mode is intended as a convenient baseline and investigation tool. Once enough history exists on stable hardware, selected metrics can be promoted to guarded regression thresholds without redesigning the integration harness. +The CI reports are intended as a convenient baseline and investigation tool. Once enough history exists on stable hardware, selected metrics can be promoted to guarded regression thresholds without redesigning the integration harness. ## Scope From 1f58c9996cebf80bd4c1208ba4291ad89d65fcd8 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:23:07 +0800 Subject: [PATCH 07/12] ci: fix macOS benchmark workflow syntax --- .github/workflows/roundtrip.yml | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.github/workflows/roundtrip.yml b/.github/workflows/roundtrip.yml index ad32360..45376d5 100644 --- a/.github/workflows/roundtrip.yml +++ b/.github/workflows/roundtrip.yml @@ -90,7 +90,8 @@ jobs: run: "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" - name: Record embedded benchmark (informational) timeout-minutes: 2 - run: "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" --benchmark + run: | + "$RIVET_ROUNDTRIP_STAGE/RivetIntegration" --benchmark windows-roundtrip: name: Windows embedded Racket round trip From eb6a61c9f448b012d86f53fd2108dfa72b931bf4 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:25:22 +0800 Subject: [PATCH 08/12] perf: include benchmark execution context --- platform/windows/Integration/main.cpp | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/platform/windows/Integration/main.cpp b/platform/windows/Integration/main.cpp index 2c7ba73..6f23127 100644 --- a/platform/windows/Integration/main.cpp +++ b/platform/windows/Integration/main.cpp @@ -25,6 +25,16 @@ struct BenchmarkMetric { double us_per_operation{}; }; +char const* benchmark_architecture() noexcept { +#if defined(_M_ARM64) || defined(__aarch64__) + return "arm64"; +#elif defined(_M_X64) || defined(__x86_64__) + return "x64"; +#else + return "unknown"; +#endif +} + std::filesystem::path executable_path() { std::wstring buffer(32768, L'\0'); auto const length = ::GetModuleFileNameW( @@ -136,6 +146,8 @@ void run_benchmark(rivet::windows::Backend& backend, double startup_ms) { std::cout << std::fixed << std::setprecision(3); std::cout << "{\"schema_version\":1,\"platform\":\"windows\"," + << "\"architecture\":\"" << benchmark_architecture() << "\"," + << "\"configuration\":\"release\"," << "\"startup_ms\":" << startup_ms << "," << "\"warmup_iterations\":" << warmup_iterations << ","; print_metric_json("rpc", rpc, true); From 9a0a530e3444c6f01e296cb36c5afc0f98cf6aee Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:25:58 +0800 Subject: [PATCH 09/12] perf: include benchmark execution context --- platform/macos/Integration/Sources/main.swift | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/platform/macos/Integration/Sources/main.swift b/platform/macos/Integration/Sources/main.swift index e73d0af..6c11f97 100644 --- a/platform/macos/Integration/Sources/main.swift +++ b/platform/macos/Integration/Sources/main.swift @@ -3,6 +3,14 @@ import Foundation import RivetEmbedding import RivetRuntime +#if arch(arm64) +private let benchmarkArchitecture = "arm64" +#elseif arch(x86_64) +private let benchmarkArchitecture = "x64" +#else +private let benchmarkArchitecture = "unknown" +#endif + private struct BenchmarkMetric { let iterations: Int let totalMilliseconds: Double @@ -93,6 +101,8 @@ private func runBenchmark( let report: [String: Any] = [ "schema_version": 1, "platform": "macos", + "architecture": benchmarkArchitecture, + "configuration": "release", "startup_ms": startupMilliseconds, "warmup_iterations": warmupIterations, "rpc": rpc.json, From acda91e69cd56f346223179efddac322c1a6a69d Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:26:23 +0800 Subject: [PATCH 10/12] docs: record benchmark architecture and build mode --- docs/benchmarks.md | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index c3c0d7c..af49e12 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -11,6 +11,8 @@ The integration runner accepts an optional `--benchmark` flag. Benchmark mode re - `state_get`: 1,000 sequential `$state/get` round trips; - `state_set`: 500 sequential `$state/set` round trips, including the reserved `$state` Event that precedes each successful response. +The report also records `platform`, `architecture`, and `configuration` so results from different execution contexts are not accidentally compared as one series. Integration benchmarks are built in release mode. + Each operation group includes `iterations`, `total_ms`, and `us_per_operation`. Every timed operation still validates its result, so a fast but incorrect run fails instead of producing a number. Example shape: @@ -19,6 +21,8 @@ Example shape: { "schema_version": 1, "platform": "macos", + "architecture": "arm64", + "configuration": "release", "startup_ms": 42.5, "warmup_iterations": 50, "rpc": { @@ -55,7 +59,7 @@ Do not treat absolute timings from shared GitHub-hosted runners as hard pass/fai For meaningful regression work: -- compare the same platform and architecture; +- compare the same platform, architecture, and build configuration; - use the same Racket, compiler, and runner image where possible; - run several samples and compare distributions or medians rather than one measurement; - use a dedicated or self-hosted machine before establishing release thresholds; From 23497e1ce162bce2b23ec200ac8d3f7bce525d73 Mon Sep 17 00:00:00 2001 From: TuringLambdaAI Date: Sat, 26 Sep 2026 08:28:26 +0800 Subject: [PATCH 11/12] ci: cancel superseded embedded roundtrip runs --- .github/workflows/roundtrip.yml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.github/workflows/roundtrip.yml b/.github/workflows/roundtrip.yml index 45376d5..2f63f82 100644 --- a/.github/workflows/roundtrip.yml +++ b/.github/workflows/roundtrip.yml @@ -7,6 +7,10 @@ on: permissions: contents: read +concurrency: + group: embedded-roundtrip-${{ github.event_name }}-${{ github.ref }} + cancel-in-progress: true + jobs: macos-roundtrip: name: macOS embedded Racket round trip From dd1c60ba5b71e748880c46cc8d8a5ed78d5fe4f2 Mon Sep 17 00:00:00 2001 From: "ren.ji" Date: Sun, 27 Sep 2026 07:09:38 +0800 Subject: [PATCH 12/12] fix: encode benchmark metric keys as JSON --- platform/windows/Integration/main.cpp | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/platform/windows/Integration/main.cpp b/platform/windows/Integration/main.cpp index 37e7f24..769d1db 100644 --- a/platform/windows/Integration/main.cpp +++ b/platform/windows/Integration/main.cpp @@ -98,7 +98,7 @@ BenchmarkMetric benchmark_operations(int iterations, Operation&& operation) { void print_metric_json(char const* name, BenchmarkMetric const& metric, bool trailing_comma) { - std::cout << "\"" << name << "\":{"iterations":" << metric.iterations + std::cout << "\"" << name << "\":{\"iterations\":" << metric.iterations << ",\"total_ms\":" << metric.total_ms << ",\"us_per_operation\":" << metric.us_per_operation << "}"; if (trailing_comma) {