From d0c90e173b435cf3e4312707693d612db828f682 Mon Sep 17 00:00:00 2001
From: MasterYoav
Date: Tue, 29 Sep 2026 08:04:37 +0300
Subject: [PATCH 1/5] Mac release publishes: a versioned, signed, verified
GitHub release with its appcast.
workflow_dispatch takes a version. The run stamps it and a rising build number into the bundle,
points Sparkle at releases/latest/download/appcast.xml, refuses to publish anything Gatekeeper
would not accept as notarized, and uploads xBot.dmg, the versioned archive and the appcast.
The site's button downloads the DMG directly.
---
.github/workflows/mac-release.yml | 55 ++++++++++++++++++++++++++++---
scripts/bundle-mac-app.sh | 9 +++++
scripts/generate-appcast.sh | 4 ++-
site/index.html | 2 +-
4 files changed, 64 insertions(+), 6 deletions(-)
diff --git a/.github/workflows/mac-release.yml b/.github/workflows/mac-release.yml
index f435eba..74651a3 100644
--- a/.github/workflows/mac-release.yml
+++ b/.github/workflows/mac-release.yml
@@ -2,9 +2,19 @@ name: Mac release
on:
workflow_dispatch:
-
+ inputs:
+ version:
+ description: "Marketing version to publish, e.g. 1.0.0. Leave empty to build without publishing."
+ required: false
+ default: ""
+ prerelease:
+ description: "Mark the GitHub release as a pre-release"
+ type: boolean
+ default: false
+
+# Write, for the one step that creates the GitHub release. Nothing else here pushes.
permissions:
- contents: read
+ contents: write
# Hoisted to the job so the optional steps can be gated on them.
#
@@ -16,6 +26,9 @@ env:
MACOS_SIGNING_IDENTITY: ${{ secrets.MACOS_SIGNING_IDENTITY }}
MACOS_CERTIFICATE_P12: ${{ secrets.MACOS_CERTIFICATE_P12 }}
SPARKLE_EDDSA_PRIVATE_KEY: ${{ secrets.SPARKLE_EDDSA_PRIVATE_KEY }}
+ XBOT_VERSION: ${{ inputs.version }}
+ # Sparkle compares CFBundleVersion, so it must rise with every release. The run number does.
+ XBOT_BUILD_NUMBER: ${{ github.run_number }}
jobs:
build:
@@ -48,7 +61,7 @@ jobs:
- name: Bundle .app
env:
- XBOT_APPCAST_URL: ${{ secrets.XBOT_APPCAST_URL }}
+ XBOT_APPCAST_URL: ${{ inputs.version != '' && format('https://github.com/{0}/releases/latest/download/appcast.xml', github.repository) || secrets.XBOT_APPCAST_URL }}
XBOT_SPARKLE_PUBLIC_KEY: ${{ secrets.XBOT_SPARKLE_PUBLIC_KEY }}
run: scripts/bundle-mac-app.sh
@@ -110,10 +123,13 @@ jobs:
APPLE_APP_PASSWORD: ${{ secrets.APPLE_APP_PASSWORD }}
run: scripts/sign-mac-app.sh
+ # The enclosure has to point at the asset this run is about to upload. The feed itself is
+ # served from releases/latest/download/appcast.xml, which GitHub redirects to the newest
+ # release — so each release's appcast only needs to describe itself.
- name: Generate Sparkle appcast
if: env.SPARKLE_EDDSA_PRIVATE_KEY != ''
env:
- XBOT_RELEASE_DOWNLOAD_PREFIX: ${{ secrets.XBOT_RELEASE_DOWNLOAD_PREFIX }}
+ XBOT_RELEASE_DOWNLOAD_PREFIX: ${{ inputs.version != '' && format('https://github.com/{0}/releases/download/v{1}/', github.repository, inputs.version) || secrets.XBOT_RELEASE_DOWNLOAD_PREFIX }}
run: scripts/generate-appcast.sh
- uses: actions/upload-artifact@v4
@@ -122,3 +138,34 @@ jobs:
path: |
dist/xBot.dmg
dist/releases/
+
+ # Only a signed, notarized build is published: an unsigned DMG on the download page is the
+ # "cannot be opened" dialog that stops a non-technical person for good.
+ - name: Verify what a stranger's Mac will check
+ if: inputs.version != ''
+ run: |
+ set -euo pipefail
+ mnt="$(mktemp -d)"
+ hdiutil attach -nobrowse -readonly -mountpoint "${mnt}" dist/xBot.dmg
+ trap 'hdiutil detach "${mnt}" >/dev/null' EXIT
+ spctl -a -vvv -t execute "${mnt}/XBot.app" 2>&1 | tee /dev/stderr | grep -q "Notarized Developer ID"
+ xcrun stapler validate "${mnt}/XBot.app"
+ xcrun stapler validate dist/xBot.dmg
+ test -f dist/releases/appcast.xml
+ grep -q 'sparkle:edSignature=' dist/releases/appcast.xml
+ test "$(/usr/libexec/PlistBuddy -c 'Print :CFBundleShortVersionString' "${mnt}/XBot.app/Contents/Info.plist")" = "${XBOT_VERSION}"
+
+ - name: Publish GitHub release
+ if: inputs.version != ''
+ env:
+ GH_TOKEN: ${{ github.token }}
+ run: |
+ set -euo pipefail
+ tag="v${XBOT_VERSION}"
+ flags=()
+ [[ "${{ inputs.prerelease }}" == "true" ]] && flags+=(--prerelease) || flags+=(--latest)
+ gh release create "${tag}" --repo "${GITHUB_REPOSITORY}" --target "${GITHUB_SHA}" \
+ --title "xBot ${XBOT_VERSION}" --generate-notes "${flags[@]}" \
+ "dist/xBot.dmg#xBot.dmg" \
+ "dist/releases/xBot-${XBOT_VERSION}.dmg" \
+ "dist/releases/appcast.xml"
diff --git a/scripts/bundle-mac-app.sh b/scripts/bundle-mac-app.sh
index 474ff48..4010628 100755
--- a/scripts/bundle-mac-app.sh
+++ b/scripts/bundle-mac-app.sh
@@ -31,6 +31,15 @@ cp "${RES}/xBot.icns" "${APP}/Contents/Resources/xBot.icns"
cp "${MAC}/Resources/Info.plist" "${APP}/Contents/Info.plist"
"${ROOT}/scripts/inject-sparkle-plist.sh" "${APP}/Contents/Info.plist"
+# The release workflow stamps the version it is publishing. Sparkle compares CFBundleVersion, so it
+# has to rise with every release, and the committed plist's "1" never would.
+if [[ -n "${XBOT_VERSION:-}" ]]; then
+ /usr/libexec/PlistBuddy -c "Set :CFBundleShortVersionString ${XBOT_VERSION}" "${APP}/Contents/Info.plist"
+fi
+if [[ -n "${XBOT_BUILD_NUMBER:-}" ]]; then
+ /usr/libexec/PlistBuddy -c "Set :CFBundleVersion ${XBOT_BUILD_NUMBER}" "${APP}/Contents/Info.plist"
+fi
+
chmod +x "${APP}/Contents/MacOS/XBot"
# SwiftPM's per-target resource bundles. `Bundle.module` traps when its bundle is missing, so an app
diff --git a/scripts/generate-appcast.sh b/scripts/generate-appcast.sh
index c4b428a..ec67000 100755
--- a/scripts/generate-appcast.sh
+++ b/scripts/generate-appcast.sh
@@ -37,7 +37,9 @@ if [[ -z "${SPARKLE_EDDSA_PRIVATE_KEY:-}" && -z "${SPARKLE_EDDSA_PRIVATE_KEY_FIL
fi
VERSION="$(/usr/libexec/PlistBuddy -c 'Print :CFBundleShortVersionString' "${PLIST}")"
-ARCHIVE="${RELEASES}/xBot ${VERSION}.dmg"
+# No space: GitHub renames an uploaded asset's spaces to dots, and the appcast's enclosure URL would
+# then name a file the release does not have.
+ARCHIVE="${RELEASES}/xBot-${VERSION}.dmg"
mkdir -p "${RELEASES}"
cp "${DMG}" "${ARCHIVE}"
diff --git a/site/index.html b/site/index.html
index 40e9143..3d55c01 100644
--- a/site/index.html
+++ b/site/index.html
@@ -94,7 +94,7 @@ Your own AI coworkers, on your own Mac
Create agents, give them a computer, watch them work, and take the wheel when you want to.
- Download for Mac
+ Download for Mac
Free and open source. macOS 14 Sonoma or later, Apple Silicon or Intel.
From 91275de2fdbe1f87af9fe9db8b3848815dfafc79 Mon Sep 17 00:00:00 2001
From: MasterYoav
Date: Tue, 29 Sep 2026 08:08:33 +0300
Subject: [PATCH 2/5] Docs and site: conversations are kept on this Mac
(ADR-0008), not by CopilotKit.
The website still told visitors CopilotKit stores their conversation history.
---
CLAUDE.md | 16 +++++++---------
docs/01-vision.md | 12 ++++++------
docs/README.md | 1 +
site/index.html | 13 ++++++-------
4 files changed, 20 insertions(+), 22 deletions(-)
diff --git a/CLAUDE.md b/CLAUDE.md
index 8370786..74ad8b1 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -29,8 +29,7 @@ finished. Route it through the app.
### Non-goals
- No hosted/SaaS version of xBot. The app, the engine, the agents and their browsers run on the
- user's machine. **In v1 the conversation transcript is the exception** — it lives in CopilotKit
- Intelligence, and onboarding says so before the user types a key. See ADR-0007.
+ user's machine, and so are conversations: the engine keeps them itself. See ADR-0008.
- No Mac App Store build. See `docs/decisions/0005-distribution-outside-app-store.md`.
- No Windows or Linux client in v1. The engine is portable; the client is not.
- No model of our own. xBot supplies no intelligence — the user brings keys or runs Ollama.
@@ -79,7 +78,7 @@ it — it usually does, and better than a first attempt would.
- New xBot code goes in **new files**. A moved or heavily edited upstream file is a permanent merge
conflict.
-### 1. v1 runs on CopilotKit Intelligence, and the seam to leave it is already built.
+### 1. v1 runs in local mode, on the vendor's own SSE runner made durable.
OpenBot's `runtimeCapabilities()` in `server/src/config.ts` used to **throw on startup** unless all
four of `INTELLIGENCE_API_URL`, `INTELLIGENCE_GATEWAY_WS_URL`, `INTELLIGENCE_API_KEY` and
@@ -87,13 +86,12 @@ four of `INTELLIGENCE_API_URL`, `INTELLIGENCE_GATEWAY_WS_URL`, `INTELLIGENCE_API
means local history, and a partial set still throws — for upstream's original reason, that somebody
who set two of four intended Intelligence and got it wrong.
-**v1 sets all four.** The local mode exists, boots, and serves, but `LocalIntelligence` is a spike
-that records which methods are reached and throws; it is not a provider yet.
+**v1 sets none.** Local mode runs `LocalThreadRunner` (`server/src/history/local-thread-runner.ts`),
+which extends CopilotKit's `InMemoryAgentRunner` and persists each thread to `local_threads`. The
+app has no field for a CopilotKit key. Recall/memory (pgvector) is still ADR-0001's, for v1.1.
-- ADRs: `docs/decisions/0007-...` (the decision), `docs/decisions/0001-...` (the eventual design)
-- Measured scope: `grep` says 156 references across 18 files; the **compiler** says 5 across 4.
- Trust the compiler. Widen the `RuntimeCapabilities` union and let it tell you.
-- `COPILOTKIT_LICENSE_TOKEN` is telemetry only. The runtime validates nothing.
+- ADRs: `docs/decisions/0008-...` (what shipped), `0007-...` (the seam), `0001-...` (the fuller design)
+- `LocalThreadRunner` is single-user: SSE mode has no `identifyUser`.
- **Never add a new direct call to the Intelligence client.** Go through the seam.
### 2. The model provider is a process-wide environment variable. We are making it per-agent.
diff --git a/docs/01-vision.md b/docs/01-vision.md
index aba3e04..d610d23 100644
--- a/docs/01-vision.md
+++ b/docs/01-vision.md
@@ -121,12 +121,12 @@ is genuinely local and genuinely model-agnostic.
Recorded here rather than discovered in month four.
-**⚠️ The engine has a cloud dependency we are living with in v1.** OpenBot's server refuses to
-start without CopilotKit Intelligence unless the seam described in
-[ADR-0007](decisions/0007-wrap-openbot-keep-intelligence.md) is used. It is built and the engine
-has been run without an account — but v1 ships on Intelligence, so a third party can change its
-free tier and affect the product. The seam is what keeps that from being fatal rather than
-inconvenient. [ADR-0001](decisions/0001-local-history-provider.md) remains the end state.
+**The engine's cloud dependency is gone for conversations.** OpenBot's server refused to start
+without CopilotKit Intelligence; the seam from
+[ADR-0007](decisions/0007-wrap-openbot-keep-intelligence.md) lets it run without, and
+[ADR-0008](decisions/0008-local-thread-runner.md) made local mode the one v1 ships: conversations
+are kept by the engine's own `LocalThreadRunner`. Cross-conversation memory (pgvector) is still
+[ADR-0001](decisions/0001-local-history-provider.md)'s, for v1.1.
**⚠️ Container runtimes on macOS are a licensing and UX minefield.** Docker Desktop requires a paid
licence above a company-size threshold, is a large install, and is not something we can bundle.
diff --git a/docs/README.md b/docs/README.md
index 2dac201..d7b6505 100644
--- a/docs/README.md
+++ b/docs/README.md
@@ -58,6 +58,7 @@ Architecture Decision Records. Each one exists because the decision looks wrong
| [0005](decisions/0005-distribution-outside-app-store.md) | Developer ID and a DMG, not the Mac App Store |
| [0006](decisions/0006-naming-and-trademark.md) | Open questions about the name and the visual reference |
| [0007](decisions/0007-wrap-openbot-keep-intelligence.md) | **Wrap OpenBot rather than re-engineer it, and keep Intelligence for v1.** Defers 0001 and re-orders the roadmap — read it before 01 or 03 |
+| [0008](decisions/0008-local-thread-runner.md) | **Conversations kept locally** by a durable wrapper around the vendor's SSE runner. Supersedes 0007's "keep Intelligence" for v1 |
## Conventions in these documents
diff --git a/site/index.html b/site/index.html
index 3d55c01..15dd47b 100644
--- a/site/index.html
+++ b/site/index.html
@@ -125,13 +125,12 @@ Where your things live
- One exception, and we would rather you read it here than find it later. In
- version 1, your conversation history is stored by
- CopilotKit, the service xBot's engine is built on. Your
- agents' files and browsers are not — only the transcript. Onboarding says so before you type
- a key, and
- ADR-0007
- records why and what replaces it.
+ Your conversations stay here too. The engine keeps every conversation in
+ its own database on this Mac, with no CopilotKit account and nothing to sign up for.
+ What leaves the Mac is what you send to the model provider you chose — and nothing at all
+ if that is a model running locally through Ollama.
+ ADR-0008
+ records how.
From 70ff7d4a2cc6b7adc4fcd88af2e2078b0e140993 Mon Sep 17 00:00:00 2001
From: MasterYoav
Date: Tue, 29 Sep 2026 08:32:41 +0300
Subject: [PATCH 3/5] A run whose model is quiet for ten seconds is no longer
cut off.
Bun closes a connection that writes nothing for idleTimeout seconds, 10 by default. A run over
SSE writes RUN_STARTED and then nothing until the model's first token, and a cold local model or
a busy vendor takes longer. The stream was closed under a live run: an empty reply, and the next
message refused with 'Thread already running' until the orphaned run finished.
Streaming agent runs now have no idle timeout; every other route keeps it. This was the open
intermittent fault in docs/13: with the fix, 6 of 6 full live rounds on fresh engines with a cold
model passed, where the first round without it failed.
---
engine/server/src/index.ts | 3 +
engine/server/src/xbot/stream-idle-timeout.ts | 35 ++++++
.../tests/xbot-stream-idle-timeout.test.ts | 101 ++++++++++++++++++
3 files changed, 139 insertions(+)
create mode 100644 engine/server/src/xbot/stream-idle-timeout.ts
create mode 100644 engine/server/tests/xbot-stream-idle-timeout.test.ts
diff --git a/engine/server/src/index.ts b/engine/server/src/index.ts
index eb6af4c..a6df312 100644
--- a/engine/server/src/index.ts
+++ b/engine/server/src/index.ts
@@ -88,6 +88,7 @@ import {
startWorkOfferedListener,
type WorkOfferedListener,
} from "./work/queue";
+import { keepStreamingRunsOpen } from "./xbot/stream-idle-timeout";
/**
* Who is asking, for a CopilotKit request.
@@ -1164,6 +1165,8 @@ const asChannelSocket = (ws: { data: SocketData }) =>
serve({
port,
async fetch(request, server) {
+ // xBot: before anything is awaited. See xbot/stream-idle-timeout.ts.
+ keepStreamingRunsOpen(request, server);
const url = new URL(request.url);
const streamBotId = streamPathBotId(url.pathname);
if (
diff --git a/engine/server/src/xbot/stream-idle-timeout.ts b/engine/server/src/xbot/stream-idle-timeout.ts
new file mode 100644
index 0000000..8875108
--- /dev/null
+++ b/engine/server/src/xbot/stream-idle-timeout.ts
@@ -0,0 +1,35 @@
+import type { Server } from "bun";
+
+/**
+ * Streaming agent runs are exempt from Bun's idle timeout. Nothing else is.
+ *
+ * `Bun.serve` closes a connection that has written nothing for `idleTimeout` seconds, 10 by default.
+ * A run over SSE writes RUN_STARTED and then nothing until the model's first token, and a model is
+ * routinely silent for longer than that: a local model loading into memory, a request queued behind
+ * another run on the same Ollama, a vendor under load, a long think before a tool call. The stream
+ * was then closed under a run that was still going. The person saw an empty reply, the run finished
+ * in the engine with nobody listening, and the next message on that conversation was refused with
+ * "Thread already running" until it did.
+ *
+ * Only these requests, rather than the server's timeout raised for everything: the timeout is what
+ * frees a connection a client abandoned without closing, and every other route answers in one go.
+ * A run that genuinely stops producing is the stall guard's to end (AGENT_STALL_TIMEOUT_MS), which
+ * says so to the person rather than dropping the connection.
+ */
+const AGENT_STREAM = /^\/api\/copilotkit\/agent\/[^/]+\/(run|connect)$/;
+
+export function holdsLongSilences(request: Request): boolean {
+ if (request.method !== "POST") return false;
+ if (!request.headers.get("accept")?.includes("text/event-stream")) {
+ return false;
+ }
+ return AGENT_STREAM.test(new URL(request.url).pathname);
+}
+
+/** Call first thing in `fetch`, before anything is awaited. `0` is Bun's "no idle timeout". */
+export function keepStreamingRunsOpen(
+ request: Request,
+ server: Pick, "timeout">,
+): void {
+ if (holdsLongSilences(request)) server.timeout(request, 0);
+}
diff --git a/engine/server/tests/xbot-stream-idle-timeout.test.ts b/engine/server/tests/xbot-stream-idle-timeout.test.ts
new file mode 100644
index 0000000..95514f7
--- /dev/null
+++ b/engine/server/tests/xbot-stream-idle-timeout.test.ts
@@ -0,0 +1,101 @@
+import { afterAll, describe, expect, test } from "bun:test";
+import { serve } from "bun";
+import {
+ holdsLongSilences,
+ keepStreamingRunsOpen,
+} from "../src/xbot/stream-idle-timeout";
+
+/**
+ * A run whose model says nothing for longer than Bun's idle timeout.
+ *
+ * Bun closes a connection that has written nothing for `idleTimeout` seconds, 10 by default, and a
+ * streaming run writes RUN_STARTED and then nothing until the model's first token. A cold local model,
+ * a queue behind another run, or a slow vendor is silent for longer than that. The stream was then
+ * closed under the run: the person saw an empty reply, the run carried on in the engine, and their
+ * next message was refused with "Thread already running".
+ */
+describe("streaming runs outlive the server's idle timeout", () => {
+ // Well past the timeout: Bun checks idle connections on a coarse timer, so a silence only just
+ // over one second is sometimes survived by luck.
+ const silenceMs = 6_000;
+ const server = serve({
+ port: 0,
+ // Short, so the test does not wait the default's ten seconds to show the same thing.
+ idleTimeout: 1,
+ fetch(request, server) {
+ keepStreamingRunsOpen(request, server);
+ const body = new ReadableStream({
+ async start(controller) {
+ controller.enqueue(new TextEncoder().encode("data: started\n\n"));
+ await Bun.sleep(silenceMs);
+ controller.enqueue(new TextEncoder().encode("data: finished\n\n"));
+ controller.close();
+ },
+ });
+ return new Response(body, {
+ headers: { "content-type": "text/event-stream" },
+ });
+ },
+ });
+ afterAll(() => server.stop(true));
+
+ const read = async (path: string, accept: string) => {
+ const response = await fetch(`http://127.0.0.1:${server.port}${path}`, {
+ method: "POST",
+ headers: { accept },
+ });
+ try {
+ return await response.text();
+ } catch {
+ return "(connection closed)";
+ }
+ };
+
+ test("a streaming run is not cut off while its model is silent", {
+ timeout: 15_000,
+ }, async () => {
+ const body = await read(
+ "/api/copilotkit/agent/agent_1/run",
+ "text/event-stream",
+ );
+ expect(body).toContain("data: finished");
+ });
+
+ test("anything else keeps the server's idle timeout", {
+ timeout: 15_000,
+ }, async () => {
+ const body = await read("/api/channels", "application/json");
+ expect(body).not.toContain("data: finished");
+ });
+});
+
+describe("which requests hold long silences", () => {
+ const post = (path: string, accept = "text/event-stream") =>
+ new Request(`http://engine.local${path}`, {
+ method: "POST",
+ headers: { accept },
+ });
+
+ test("an agent run that asks for a stream", () => {
+ expect(holdsLongSilences(post("/api/copilotkit/agent/a/run"))).toBe(true);
+ expect(
+ holdsLongSilences(post("/api/copilotkit/agent/agent_x%2Fy/connect")),
+ ).toBe(true);
+ });
+
+ test("not a run that asks for JSON, and not an ordinary route", () => {
+ expect(
+ holdsLongSilences(
+ post("/api/copilotkit/agent/a/run", "application/json"),
+ ),
+ ).toBe(false);
+ expect(holdsLongSilences(post("/api/channels"))).toBe(false);
+ expect(
+ holdsLongSilences(
+ new Request("http://engine.local/api/copilotkit/agent/a/run", {
+ headers: { accept: "text/event-stream" },
+ }),
+ ),
+ ).toBe(false);
+ });
+});
From 63379a3a43107e8d4c6e96b31398f7aaf02bf77f Mon Sep 17 00:00:00 2001
From: MasterYoav
Date: Tue, 29 Sep 2026 08:32:41 +0300
Subject: [PATCH 4/5] A message sent while the agent is still answering gets a
sentence, not an empty reply.
The vendor's runner throws on a busy thread, and thrown inside the SSE handler that became a 200
with no events. LocalThreadRunner now answers with RUN_ERROR, which the app already shows.
---
.../server/src/history/local-thread-runner.ts | 30 ++++++-
.../local-thread-runner.integration.test.ts | 78 ++++++++++++++++++-
2 files changed, 106 insertions(+), 2 deletions(-)
diff --git a/engine/server/src/history/local-thread-runner.ts b/engine/server/src/history/local-thread-runner.ts
index b1bc265..691c4b1 100644
--- a/engine/server/src/history/local-thread-runner.ts
+++ b/engine/server/src/history/local-thread-runner.ts
@@ -1,8 +1,10 @@
-import type { Message } from "@ag-ui/client";
+import { type BaseEvent, EventType, type Message } from "@ag-ui/client";
import {
type AgentRunnerRunRequest,
InMemoryAgentRunner,
+ ɵGLOBAL_STORE,
} from "@copilotkit/runtime/v2";
+import { of } from "rxjs";
import type { Database } from "../db/client";
import { localThreads } from "../db/schema";
import type { IntelligenceLike } from "../routines/run-turn";
@@ -50,6 +52,23 @@ export class LocalThreadRunner extends InMemoryAgentRunner {
}
override run(request: AgentRunnerRunRequest) {
+ /*
+ * A conversation that is still answering refuses the next run, and says so as an event.
+ *
+ * The vendor's runner throws "Thread already running" here. Thrown inside the SSE handler's
+ * factory, that is logged and the response — already a 200 — is closed with no events in it, so
+ * the Mac app saw a reply that was simply empty and could not tell anybody why. RUN_ERROR is the
+ * event every caller already reads as a failed turn: the app shows its sentence, and a routine or
+ * a hop, which take a lock before they get here, treat it as the refusal it is.
+ */
+ if (this.isBusy(request.threadId)) {
+ return of({
+ type: EventType.RUN_ERROR,
+ message:
+ "The agent is still answering your last message in this conversation. Wait for it to finish, then send this again.",
+ code: "thread_busy",
+ } as BaseEvent);
+ }
const carried = request.agent.messages;
const before = new Set(carried.map((message) => message.id));
const shown = request.persistedInputMessages ?? carried;
@@ -68,6 +87,15 @@ export class LocalThreadRunner extends InMemoryAgentRunner {
return [...(this.kept.get(threadId) ?? [])];
}
+ /**
+ * Whether a run would be refused. The vendor's own test, read synchronously from its store
+ * because `run` must answer synchronously, and `isRunning` is a promise.
+ */
+ private isBusy(threadId: string): boolean {
+ const store = ɵGLOBAL_STORE.peek(threadId);
+ return Boolean(store?.isRunning || store?.stopRequested);
+ }
+
private keep(threadId: string, agentId: string, messages: Message[]) {
const prior = this.kept.get(threadId) ?? [];
const known = new Set(prior.map((message) => message.id));
diff --git a/engine/server/tests/local-thread-runner.integration.test.ts b/engine/server/tests/local-thread-runner.integration.test.ts
index 1b28105..14755c6 100644
--- a/engine/server/tests/local-thread-runner.integration.test.ts
+++ b/engine/server/tests/local-thread-runner.integration.test.ts
@@ -3,7 +3,7 @@ import { randomUUID } from "node:crypto";
import type { BaseEvent, Message, RunAgentInput } from "@ag-ui/client";
import { AbstractAgent, EventType } from "@ag-ui/client";
import { eq } from "drizzle-orm";
-import { lastValueFrom, Observable } from "rxjs";
+import { lastValueFrom, Observable, Subject } from "rxjs";
import { createDatabase } from "../src/db/client";
import { localThreads } from "../src/db/schema";
import {
@@ -189,3 +189,79 @@ describe("a conversation kept locally", () => {
]);
});
});
+
+/** Says nothing until released, so a second run can arrive while it is still going. */
+class Held extends AbstractAgent {
+ private readonly gate = new Subject();
+ release() {
+ this.gate.next();
+ this.gate.complete();
+ }
+ run(input: RunAgentInput) {
+ return new Observable((subscriber) => {
+ subscriber.next({
+ type: EventType.RUN_STARTED,
+ threadId: input.threadId,
+ runId: input.runId,
+ } as BaseEvent);
+ // Whichever subscription the runner holds, and however many times it subscribes.
+ const opened = this.gate.subscribe({
+ complete: () => {
+ subscriber.next({
+ type: EventType.RUN_FINISHED,
+ threadId: input.threadId,
+ runId: input.runId,
+ } as BaseEvent);
+ subscriber.complete();
+ },
+ });
+ return () => opened.unsubscribe();
+ });
+ }
+}
+
+describe("a message sent while the conversation is still answering", () => {
+ const input = (threadId: string) => ({
+ threadId,
+ runId: randomUUID(),
+ messages: [user(randomUUID(), "hello")],
+ state: {},
+ tools: [],
+ context: [],
+ forwardedProps: {},
+ });
+
+ test("is refused with a RUN_ERROR a person can read, not a silent empty stream", async () => {
+ const threadId = newThread();
+ const runner = new LocalThreadRunner(database);
+ const first = new Held();
+ first.setMessages([]);
+ const running = lastValueFrom(
+ runner.run({ threadId, agent: first, input: input(threadId) }),
+ { defaultValue: undefined },
+ );
+
+ // Thrown, the vendor's SSE handler logs it and closes a 200 with no events in it at all.
+ const second = new Echo();
+ second.setMessages([]);
+ const events: BaseEvent[] = [];
+ await new Promise((resolve, reject) =>
+ runner
+ .run({ threadId, agent: second, input: input(threadId) })
+ .subscribe({
+ next: (e) => events.push(e),
+ error: reject,
+ complete: resolve,
+ }),
+ );
+
+ expect(events.map((event) => event.type)).toEqual([EventType.RUN_ERROR]);
+ expect((events[0] as { message?: string }).message).toContain(
+ "still answering",
+ );
+
+ first.release();
+ await running;
+ expect(await runner.isRunning({ threadId })).toBe(false);
+ });
+});
From 9f486d39a62b67f8e3bf35aa8c44f0ab0086b35e Mon Sep 17 00:00:00 2001
From: MasterYoav
Date: Tue, 29 Sep 2026 08:33:01 +0300
Subject: [PATCH 5/5] Docs: the intermittent empty reply is found and fixed.
---
README.md | 6 +++---
docs/13-launch-checklist.md | 21 ++++++++++++---------
2 files changed, 15 insertions(+), 12 deletions(-)
diff --git a/README.md b/README.md
index d154fcd..e85cd3f 100644
--- a/README.md
+++ b/README.md
@@ -67,9 +67,9 @@ process. On 27 September a throwaway engine with no CopilotKit key held a three-
with a local Ollama model, drove its browser through the client-tool loop, remembered across turns,
and kept every conversation intact through a restart.
-Still open: a second live vendor, one intermittent fault seen only under the full live suite, a
-clean-VM first run, and Developer ID signing, notarization, Sparkle keys and the first published
-release. Those last items need an account holder.
+Still open: a second live vendor, a clean-VM first run, notarization credentials that Apple
+accepts, and the first published release. Signing, Sparkle keys and the release pipeline are in
+place.
Start at [`docs/README.md`](docs/README.md). The milestone table is in
[`docs/12-roadmap.md`](docs/12-roadmap.md), and what is left, in order, is in
diff --git a/docs/13-launch-checklist.md b/docs/13-launch-checklist.md
index 94e0041..94575eb 100644
--- a/docs/13-launch-checklist.md
+++ b/docs/13-launch-checklist.md
@@ -154,13 +154,16 @@ That first run found three engine faults, each of which failed every conversatio
guidance, and Anthropic accepts a system message only first. All system texts are now merged, in
order, into one leading message (`agent-langgraph/src/history.ts`).
-**One open finding.** On a freshly started engine, roughly one full live run in six had a turn come
-back empty, or a request go unanswered, while all nine live tests ran at once. The engine logged
-"Thread already running" or, once, "No agents are registered", both of which should be impossible
-there. It was never seen with the conversation suite alone (5 of 5 fresh engines), nor with a
-logging proxy in the path (12 of 12). A leftover process, a database lock, identity, idle
-keep-alive and split request writes were each checked and ruled out. Worth one more look before
-launch, starting from the app under ordinary use rather than nine tests at once.
+**The intermittent fault, found and fixed** (29 September). On a freshly started engine, roughly one
+full live run in six had a turn come back empty, followed by "Thread already running". The cause was
+Bun's own `idleTimeout`, 10 seconds by default: a streaming run writes RUN_STARTED and then nothing
+until the model's first token, and a cold local model takes longer than that. The connection was
+closed under a live run, the reply was empty, and the orphaned run held the thread until it
+finished. A fake model that stays silent for 15 seconds reproduced it every time. Streaming agent
+runs are now exempt from the idle timeout (`server/src/xbot/stream-idle-timeout.ts`); with that, 6
+of 6 full live rounds on fresh engines with a cold model passed with no busy refusals. A message
+sent while a run genuinely is still going now gets RUN_ERROR with a sentence rather than an empty
+200.
**Two more, found starting the app on this Mac's own engine** (27 September), both leaving the
container unhealthy with nothing on its port and no way back short of deleting the data:
@@ -233,8 +236,8 @@ From a shell holding no credentials:
now needs re-running against `LocalThreadRunner` rather than a CopilotKit key — see item 5.
- On 27 September, it was: a throwaway engine in local mode with no CopilotKit key held a real
conversation with a local Ollama model, used its computer through the client-tool loop, remembered
- across turns, and kept every conversation intact through a restart. Item 5 has the details and
- the one open finding.
+ across turns, and kept every conversation intact through a restart. Item 5 has the details,
+ including the intermittent empty reply found and fixed on 29 September.
- The pinned engine is one multi-arch image (linux/amd64 and linux/arm64). On 14 September an Apple
Silicon Mac pulled it anonymously, got arm64, was healthy in about ten seconds on a port other than
3001, and passed the live suite: browser, files, shell, and a help request handed back. Before that